feat(cluster): 集群化落地 —— Manager/Worker 拆分 + 归属租约 + 跨机验证(T08)

背景:把平台从「单机单进程」改造成「1 组 Manager + N 台 Worker + 共享归属状态」,
硬约束 = 全程兼容单例模式(deployMode 默认 local;生产切换前 47 一行未动)。

主要改动
1) 数据模型 v7(SQLite 与 PG 两方言同步):新增 dsh_hosts 注册表 +
   dsh_instances.{host_id,epoch,heartbeat_at,lease_until};claimInstance 原子抢占
   (UPDATE … WHERE host_id IS NULL OR lease_until < now)+ pinInstanceHost 钉住归属。
2) 租约与 fencing:src/supervisor/lease.ts(acquire/renew/release + stillHolder 判据 +
   ttl > 2×renew 硬校验);心跳里续租,失权即向 worker 下发更高 epoch(self-fencing)。
   ⚠️ release 只清租约(lease_until),**保留 host_id** —— host_id 是「用户数据在哪台」的锚点。
3) Worker agent(src/worker/agent.ts,子命令 dshs worker):实例生命周期 + 文件面 /fs/*
   + 幂等键(operationId)+ 鉴权(timingSafeEqual);Worker 不写控制面数据
   (apiKey/uid 由 Manager 随 launch 投递,R5 收窄)。
4) 远端 Spawner + LeasedSpawner:按 host 路由(**粘性优先**:有历史归属且那台 up 就留在原地,
   否则按容量选最空的)+ 容量准入 + deployMode=cluster 装配(systemd drop-in,可回滚)。
5) bwrap 修正:**所有挂载点的中间目录统一前置 + 去重 + 由外到内**(「就近创建」会在嵌套前缀下
   遮掉已绑挂载点 ⇒ bwrap: Can't chdir);且**只能用 --tmpfs**,用 --perms 会让 47 的
   bwrap 0.4.0 直接拒启动(沙箱全挂)。
6) 跨机隧道 src/worker/tunnel.ts:SSH ControlMaster + 动态 -R 转发;**自愈由 agent 本地
   20s 定时器驱动**(不能只放 /healthz —— 心跳本身经隧道进来,断了就没人触发它)。
7) 文件面按归属路由(RemoteUserFs):实例与文件必须落在同一台机器,否则实例看不到自己的文件。
8) 观测面:dshs doctor / dshs cluster status。

验证(本次均已实跑)
- test/lease.test.mjs:SQLite 10/10 == PG 10/10
- 组件级端到端 5 个:verify-cluster-{agent,lease,fs,migrate,live}.mjs
- 真跨机(47 Manager / 106 Worker,跨云 + 反向隧道)verify-cluster-cross.mjs 九步全绿
- 域名形态访问 verify-cluster-domain.mjs(<user>.域名 → Manager → 远端实例;越权 403)
- 冒烟 scripts/smoke-*:6/8,失败项与改动前基线完全相同(无回归)
- 生产切换与回滚剧本见 dsh-server-docs/交接单/T08-集群化落地-兼容单例模式.md §16
This commit is contained in:
admin committed 2026-09-15 18:47:02 +08:00
1 parent 68c0a320ed
commit c70d5d860e
47 files changed
+5367 -14

No files matched your search

+11
View File
@@ -0,0 +1,11 @@
#!/usr/bin/env bash
export PGPASSWORD=dshs_cluster_2026
Q() { /usr/bin/psql -h 127.0.0.1 -p 15432 -U dshs -d dshs -tAc "$1"; }
echo "=== users ==="
Q "select id || ' | ' || username || ' | ' || role || ' | uid=' || coalesce(uid::text,'-') from users order by row_id"
echo "=== dsh_instances ==="
Q "select id || ' | user=' || user_id || ' | ' || role || ' | ' || status || ' | host=' || coalesce(host_id,'NULL') || ' | epoch=' || epoch from dsh_instances order by id"
echo "=== dsh_hosts ==="
Q "select id || ' | cap=' || capacity_mb || ' | used=' || used_mb || ' | ' || status from dsh_hosts order by id"
echo "=== sessions(应为 0 条 switch-verify) ==="
Q "select count(*) from sessions where user_agent='switch-verify'"
+5
View File
@@ -54,5 +54,10 @@ if (role === 'watchdog') {
})
server.listen(port, '127.0.0.1', () => {
console.log(`fake-dsh listening on ${port}`)
// 真实 dsh 启动后会打印**可直达的带 token URL**,平台就是靠这行取 launch token
// (正则:/dsh web: http://////127//.0//.0//.1://d+/////?token=([A-Za-z0-9_-]+)/)。
// 夹具必须照实吐出来,否则平台只能等满 10 s 超时 ⇒ 「登录直达会话」这条链路
// 在本机测试里**永远测不到**(2026-09-15 T08 S3 实测踩到)。
console.log(`dsh web: http://127.0.0.1:${port}/?token=FAKE_TOKEN_${process.pid}`)
})
}
+186
View File
@@ -0,0 +1,186 @@
#!/usr/bin/env node
/**
* T08 · S1.2:SQLite → Postgres 一次性数据迁移。
*
* 设计要点(都是踩过才会疼的地方):
* 1. **列清单不写死** —— 从 PG 的 information_schema 与 SQLite 的 PRAGMA 取**交集**,
* 这样 schema 演进(v4 的 folder/patch、v6 的 enabled 等)不会让脚本静默少搬字段。
* 2. **PG 表结构不由本脚本建** —— 先 import 平台自己的 `createDbAdapter`(带 dbUrl),
* 让**平台的迁移**在 PG 上建库。这样"迁移脚本"与"平台 schema"永远不会两套。
* 3. **identity 列要 `OVERRIDING SYSTEM VALUE`** —— `users.uid` 与 `audit_log.id` 是
* GENERATED ALWAYS AS IDENTITY;不覆盖就会重排 id,**uid 一变 = 所有用户文件属主失配**。
* 搬完必须 `RESTART WITH` 把序列推到 max+1,否则下一条 INSERT 撞主键。
* 4. **FK 顺序**:先 users,再 workspaces/sessions,最后引用它们的表。
* 5. `--dry-run` 只报行数,不写任何东西。
*
* 用法:
* node scripts/migrate-sqlite-to-pg.mjs --sqlite /var/lib/dshs/dshs.db \
* --pg postgres://dshs:***@127.0.0.1:15432/dshs [--dry-run]
*
* @module dshs/scripts/migrate-sqlite-to-pg
*/
import { existsSync } from 'node:fs'
import Database from 'better-sqlite3'
import pg from 'pg'
import { createDbAdapter } from '../lib/db/index.js'
import { resolveConfig } from '../lib/config.js'
/** FK 依赖顺序(父 → 子)。未列出的表会被追加到末尾并告警。 */
const ORDER = [
'users',
'workspaces',
'sessions',
'folder_plugins',
'dsh_instances',
'domains',
'credential_vault',
'business_plugins',
'audit_log',
]
/** identity 列(必须 OVERRIDING SYSTEM VALUE + 搬完 RESTART)。 */
const IDENTITY = { users: 'uid', audit_log: 'id' }
/**
* ⛔ **绝不搬**的表。
*
* `schema_migrations`:目标端的"已应用版本"标记由**平台的迁移**建立(见 `ensurePgSchema`),
* 从源库搬会把同一批版本号再插一遍 ⇒ `schema_migrations_pkey` 唯一键冲突
* (2026-09-15 实测踩到,事务已整体回滚)。语义上也应如此:**结构版本由平台在目标端决定**。
*/
const SKIP = new Set(['schema_migrations'])
function arg(name, fallback) {
const i = process.argv.indexOf(`--${name}`)
return i >= 0 && process.argv[i + 1] !== undefined ? process.argv[i + 1] : fallback
}
const sqlitePath = arg('sqlite')
const pgUrl = arg('pg')
const dryRun = process.argv.includes('--dry-run')
if (sqlitePath === undefined || pgUrl === undefined) {
console.error('用法: node scripts/migrate-sqlite-to-pg.mjs --sqlite <file> --pg <url> [--dry-run]')
process.exit(2)
}
if (!existsSync(sqlitePath)) {
console.error(`SQLite 文件不存在: ${sqlitePath}`)
process.exit(2)
}
/** 让**平台的迁移**在 PG 上建好结构(不自己写 DDL,避免两套 schema)。 */
async function ensurePgSchema() {
const config = resolveConfig({ dataRoot: '/tmp/migrate-tooling', dbUrl: pgUrl })
const db = await createDbAdapter(config)
await db.close()
}
async function main() {
const sq = new Database(sqlitePath, { readonly: true })
await ensurePgSchema()
const client = new pg.Client({ connectionString: pgUrl })
await client.connect()
const sqTables = sq
.prepare("SELECT name FROM sqlite_master WHERE type='table' AND name NOT LIKE 'sqlite_%'")
.all()
.map((r) => r.name)
const pgTables = (
await client.query("SELECT table_name FROM information_schema.tables WHERE table_schema='public'")
).rows.map((r) => r.table_name)
const common = sqTables.filter((t) => pgTables.includes(t) && !SKIP.has(t))
if (sqTables.some((t) => SKIP.has(t))) {
console.log(`按设计跳过: ${[...SKIP].join(', ')}(目标端的结构版本由平台迁移建立)`)
}
const ordered = [
...ORDER.filter((t) => common.includes(t)),
...common.filter((t) => !ORDER.includes(t)),
]
const extra = common.filter((t) => !ORDER.includes(t))
if (extra.length > 0) console.warn(`⚠️ 未在 ORDER 中声明、按末尾处理的表: ${extra.join(', ')}`)
/** 两端的列交集 —— 只搬双方都有的列。 */
async function sharedCols(table) {
const sqCols = sq.prepare(`PRAGMA table_info(${table})`).all().map((c) => c.name)
const pgCols = (
await client.query(
'SELECT column_name FROM information_schema.columns WHERE table_schema=$1 AND table_name=$2',
['public', table],
)
).rows.map((r) => r.column_name)
return sqCols.filter((c) => pgCols.includes(c))
}
const report = []
await client.query('BEGIN')
try {
for (const table of ordered) {
const cols = await sharedCols(table)
if (cols.length === 0) {
console.warn(`跳过 ${table}: 无公共列`)
continue
}
const rows = sq.prepare(`SELECT ${cols.join(',')} FROM ${table}`).all()
const idCol = IDENTITY[table]
if (!dryRun && rows.length > 0) {
const colList = idCol !== undefined ? [idCol, ...cols.filter((c) => c !== idCol)] : cols
const override = idCol !== undefined ? ' OVERRIDING SYSTEM VALUE' : ''
const chunk = 200
for (let i = 0; i < rows.length; i += chunk) {
const slice = rows.slice(i, i + chunk)
const values = []
const tuples = slice.map((row) => {
const ph = colList.map((c) => {
values.push(row[c] ?? null)
return `$${values.length}`
})
return `(${ph.join(',')})`
})
await client.query(
`INSERT INTO ${table} (${colList.join(',')})${override} VALUES ${tuples.join(',')}`,
values,
)
}
if (idCol !== undefined) {
await client.query(
`SELECT setval(pg_get_serial_sequence('${table}','${idCol}'),
(SELECT COALESCE(MAX(${idCol}),0) FROM ${table}))`,
)
}
}
report.push({ table, rows: rows.length, cols: cols.length })
}
if (dryRun) await client.query('ROLLBACK')
else await client.query('COMMIT')
} catch (err) {
await client.query('ROLLBACK')
throw err
}
console.log(`\n${dryRun ? '【DRY-RUN,已回滚】' : '【已提交】'} SQLite → PG 迁移明细`)
for (const r of report) console.log(` ${r.table.padEnd(18)} ${String(r.rows).padStart(6)} 行 / ${r.cols} 列`)
// 校验:逐表比对行数
let bad = 0
if (!dryRun) {
for (const r of report) {
const pgCount = Number((await client.query(`SELECT COUNT(*) AS c FROM ${r.table}`)).rows[0].c)
const sqCount = Number(sq.prepare(`SELECT COUNT(*) AS c FROM ${r.table}`).get().c)
if (pgCount !== sqCount) {
console.error(` ✗ 行数不符 ${r.table}: PG=${pgCount} SQLite=${sqCount}`)
bad += 1
}
}
console.log(bad === 0 ? '\n✅ 逐表行数一致' : `\n❌ ${bad} 张表行数不符`)
}
await client.end()
sq.close()
process.exit(bad === 0 ? 0 : 1)
}
main().catch((err) => {
console.error(err)
process.exit(1)
})
+37
View File
@@ -0,0 +1,37 @@
#!/usr/bin/env bash
# 存量数据并行搬运 47 → 106(4 路并发;瓶颈在源端小文件 IOPS,单流只 ~0.4MB/s)
# 搬完写 /root/push-parallel.done,供后续 cutover 判断
set -uo pipefail
KEY=/root/.ssh/dshworker_ed25519
DST=[email protected]
SSHO="-i $KEY -o BatchMode=yes -o StrictHostKeyChecking=accept-new -o Compression=no"
SRC=/var/lib/dshs
LOG=/root/push-parallel.log
: > "$LOG"
ssh -n $SSHO "$DST" 'mkdir -p /var/lib/dshs'
echo "[$(date +%T)] 目标端就绪" >> "$LOG"
# 按"大小"分组:大用户单开一路,其余合流(每路一个 tar→ssh)
one() { # $1=组名 其余=相对路径列表
local name="$1"; shift
(
cd "$SRC" || exit 1
{ for p in "$@"; do [ -e "$p" ] && echo "$p"; done; } > "/tmp/list-$name.txt"
tar --numeric-owner --files-from="/tmp/list-$name.txt" -cf - \
| ssh $SSHO "$DST" "tar -C /var/lib/dshs --numeric-owner -xf -"
echo "[$(date +%T)] $name 完成 rc=$?" >> "$LOG"
) &
}
# 组划分(按实际内容:1 个大用户 + 若干小目录)
one g1 "users/4092b965-2f68-4977-9989-68b3966f7df0"
one g2 "users/cce6d1cd-b376-4304-80f0-0e1c58c9ffde" "users/3ec95f69-6a4e-4d16-a415-56aa09396fc5"
one g3 "users/4eaeb26b-9e0c-4c68-9b61-0daf70664ae5" "users/74e8804a-e0c8-4645-84de-90dd3fae6c2b"
one g4 "users/7ba268be-6103-438c-8e6b-609b422ccbca" "users/ca3f36e0-937f-437e-b850-53b8230f20f8" \
"bundled-skills" "business-plugins" "whitelist-cache"
wait
echo "[$(date +%T)] 全部完成" >> "$LOG"
ssh -n $SSHO "$DST" 'du -sm /var/lib/dshs | cut -f1' >> "$LOG" 2>&1
touch /root/push-parallel.done
+63
View File
@@ -0,0 +1,63 @@
#!/usr/bin/env bash
# 在 47 初始化「控制面 PG」:独立数据目录 /var/lib/dshs-pg + 专用 unit dshs-pg.service
# 设计口径:控制面 DB 在 Manager 侧、仅 loopback、scram 认证。
set -uo pipefail
PGDATA=/var/lib/dshs-pg
PGPORT=15432
DBPW="${DSHS_PG_PASSWORD:-dshs_cluster_2026}"
PGVER=$(/usr/bin/postgres --version | grep -oE '[0-9]+' | head -1)
echo " PG 版本: $(/usr/bin/postgres --version)"
# 别让发行版的默认单元意外起来(我们用自己的 unit + 自己的数据目录)
systemctl disable --now postgresql 2>/dev/null >/dev/null || true
if [ ! -f "$PGDATA/PG_VERSION" ]; then
install -d -o postgres -g postgres -m 700 "$PGDATA"
su - postgres -c "/usr/bin/initdb -D $PGDATA -E UTF8 --locale=C.UTF-8 --auth-local=peer --auth-host=scram-sha-256" >/tmp/initdb.log 2>&1 \
&& echo " ✓ initdb 完成" || { echo " ✗ initdb 失败"; tail -5 /tmp/initdb.log; exit 1; }
cat >> "$PGDATA/postgresql.conf" <<CONF
# ── DSHS 控制面(2026-09-15 集群化切换)──
listen_addresses = '127.0.0.1'
port = $PGPORT
unix_socket_directories = '/var/run/postgresql'
max_connections = 100
shared_buffers = 128MB
CONF
chown postgres:postgres "$PGDATA/postgresql.conf"
echo " ✓ 已写入 listen=127.0.0.1 port=$PGPORT"
fi
cat > /etc/systemd/system/dshs-pg.service <<UNIT
[Unit]
Description=DSHS control-plane PostgreSQL (cluster mode)
After=network.target
[Service]
Type=notify
User=postgres
Group=postgres
ExecStart=/usr/bin/postgres -D $PGDATA
ExecReload=/bin/kill -HUP \$MAINPID
KillMode=mixed
TimeoutStopSec=30
Restart=on-failure
[Install]
WantedBy=multi-user.target
UNIT
systemctl daemon-reload
systemctl enable --now dshs-pg >/dev/null 2>&1
sleep 3
echo " dshs-pg: $(systemctl is-active dshs-pg) 监听: $(ss -lntp 2>/dev/null | grep -c $PGPORT)"
# 角色 + 库(幂等)
su - postgres -c "/usr/bin/psql -p $PGPORT -tAc \"select 1 from pg_roles where rolname='dshs'\"" 2>/dev/null | grep -q 1 \
|| su - postgres -c "/usr/bin/psql -p $PGPORT -c \"create role dshs login password '$DBPW'\"" >/dev/null 2>&1
su - postgres -c "/usr/bin/psql -p $PGPORT -tAc \"select 1 from pg_database where datname='dshs'\"" 2>/dev/null | grep -q 1 \
|| su - postgres -c "/usr/bin/psql -p $PGPORT -c \"create database dshs owner dshs\"" >/dev/null 2>&1
echo " --- 连接自检(dshs 角色) ---"
PGPASSWORD="$DBPW" /usr/bin/psql -h 127.0.0.1 -p $PGPORT -U dshs -d dshs -tAc "select current_user||'@'||current_database()||' pg='||version()" 2>&1 | head -1 | cut -c1-90
echo " 连接串(含密码,勿外传): postgres://dshs:$DBPW@127.0.0.1:$PGPORT/dshs"
+16
View File
@@ -0,0 +1,16 @@
#!/usr/bin/env bash
# 在 47 的控制面 PG 上建角色与库(peer 认证走 unix socket,不依赖已存在的密码)
set -uo pipefail
PGPORT=15432
DBPW="dshs_cluster_2026"
su - postgres -c "/usr/bin/psql -p $PGPORT -tAc \"select 1 from pg_roles where rolname='dshs'\"" 2>/dev/null | grep -q 1 \
|| su - postgres -c "/usr/bin/psql -p $PGPORT -c \"create role dshs login password '$DBPW'\"" >/dev/null 2>&1
su - postgres -c "/usr/bin/psql -p $PGPORT -tAc \"select 1 from pg_database where datname='dshs'\"" 2>/dev/null | grep -q 1 \
|| su - postgres -c "/usr/bin/psql -p $PGPORT -c \"create database dshs owner dshs\"" >/dev/null 2>&1
echo "=== 角色/库确认(本地 peer) ==="
su - postgres -c "/usr/bin/psql -p $PGPORT -tAc \"select rolname from pg_roles where rolname='dshs'\"" 2>/dev/null | sed 's/^/ role: /'
su - postgres -c "/usr/bin/psql -p $PGPORT -tAc \"select datname||' owner='||pg_get_userbyid(datdba) from pg_database where datname='dshs'\"" 2>/dev/null | sed 's/^/ db: /'
echo "=== 连接自检(TCP + 密码,走的应是**本机 PG 13**) ==="
PGPASSWORD="$DBPW" /usr/bin/psql -h 127.0.0.1 -p $PGPORT -U dshs -d dshs -tAc "select version()" 2>&1 | head -1 | cut -c1-80
+5 -1
View File
@@ -75,7 +75,11 @@ try {
r = await json('/api/dsh/launch', { method: 'POST', cookie, body: { folder: 'proj' } })
console.log('launch ->', r.status, r.body?.url)
assert(r.status === 200, 'launch succeeds')
assert(r.body.url === 'https://carol.test.local/', 'launch returns the subdomain URL')
// 2026-09-15(T08 S3):夹具 `fake-dsh.mjs` 现在**照实打印带 token 的 URL**(与真实 dsh 一致),
// 于是这里不能再写死成不带 token 的相等 —— 原断言是"夹具不吐 token"时的意外产物。
// 保留原意(是子域 URL、不泄露回环端口),并把 token 的存在一并纳入判据。
assert(r.body.url.startsWith('https://carol.test.local/'), 'launch returns the subdomain URL')
assert(!r.body.url.includes('127.0.0.1'), 'launch URL must not leak the loopback port')
await sleep(200)
+45
View File
@@ -0,0 +1,45 @@
#!/usr/bin/env bash
# 起一台 **cluster 模式的 Manager**(T08 跨机演练用;在 Manager 那台机器上跑)。
#
# 现场的对照(2026-09-15 演练实测):
# · Manager 在 **47**(本脚本所在机器),监听 `127.0.0.1:13080`(**不公网暴露**)
# · Worker agent 在 **106**,经 SSH 反向隧道出现在本机 `127.0.0.1:19000` / `19001`
# · 控制面 PG 也在 **106**,经同一条隧道出现在本机 `127.0.0.1:15432`
# 用法:bash scripts/start-cluster-manager.sh (env 见下方 manager.env)
在 47 上:建 manager.env、bootstrap 管理员、起 cluster Manager(127.0.0.1:13080)
set -uo pipefail
cd /opt/dshs-cluster || exit 1
cat > /opt/dshs-cluster/manager.env <<'ENVEOF'
DSHS_DEPLOY_MODE=cluster
DSHS_DB_URL=postgres://dshs:[email protected]:15432/dshs_cross
DSHS_DATA_ROOT=/opt/dshs-cluster/data
DSHS_CLUSTER_HOST_ID=m-47
DSHS_CLUSTER_AGENT_URL=http://127.0.0.1:19000
DSHS_CLUSTER_AGENT_TOKEN=cross-machine-token
DSHS_CLUSTER_INSTANCE_HOST=127.0.0.1
DSHS_CLUSTER_WORKER_DATA_ROOT=/opt/dshs-cluster/live-data
DSHS_CLUSTER_CAPACITY_MB=-1
DSHS_CLUSTER_REGISTER_SELF=0
DSHS_CLUSTER_LEASE_TTL_MS=30000
ENVEOF
set -a
# shellcheck disable=SC1091
. /opt/dshs-cluster/manager.env
set +a
echo "--- bootstrap 管理员 ---"
node lib/cli.js bootstrap-admin --username root --password crossmgr123 2>&1 | tail -1
echo "--- 起 Manager ---"
pkill -f "dshs-cluster/lib/cli.js --port 13080" 2>/dev/null
sleep 1
nohup node lib/cli.js --port 13080 --host 127.0.0.1 --log-level warn > /tmp/manager-47.log 2>&1 &
sleep 7
echo "--- 自检 ---"
echo " login.html : $(curl -s -o /dev/null -w '%{http_code}' -m 6 http://127.0.0.1:13080/login.html)"
echo " 进程 : $(pgrep -cf 'dshs-cluster/lib/cli.js --port 13080')"
echo " 日志尾部 :"
tail -4 /tmp/manager-47.log 2>/dev/null | sed 's/^/ /'
+76
View File
@@ -0,0 +1,76 @@
#!/usr/bin/env bash
# 切换 A 步(在 47 上跑):
# ① 备份 /opt/dshs/lib → /opt/dsh/backups/lib-<ts>/
# ② 覆盖 /opt/dshs/lib(T08 集群版代码)
# ③ 装 **本地 Worker**(w-47,19100,无隧道 —— Manager 同机直连)
# ④ 把既有用户(admin/guest)的归属**预置**为 w-47(否则粘性落点无处可粘、新老用户会被按容量随机调度)
# ⑤ 在 PG 里注册 w-47 / w-106 两台 worker
set -uo pipefail
TS=$(date +%Y%m%d-%H%M%S)
W47_TOKEN="dshs-worker-47-c4b7e19f"
W106_TOKEN="dshs-worker-7f3a91c05e"
PGURL="postgres://dshs:[email protected]:15432/dshs"
TARBALL=/tmp/dshs-lib-new.tgz
echo "=== ① 备份 /opt/dshs/lib ==="
mkdir -p "/opt/dsh/backups/lib-$TS"
cp -a /opt/dshs/lib "/opt/dsh/backups/lib-$TS/lib" && echo " ✓ 备份到 /opt/dsh/backups/lib-$TS/lib($(find /opt/dsh/backups/lib-$TS -type f | wc -l) 文件)"
echo "=== ② 覆盖 lib ==="
[ -f "$TARBALL" ] || { echo " ✗ 缺少 $TARBALL"; exit 1; }
rm -rf /opt/dshs/lib && tar -xzf "$TARBALL" -C /opt/dshs
echo " ✓ 已覆盖;cluster 特征检查: $(grep -l "DEPLOY_MODE" /opt/dshs/lib/config.js >/dev/null 2>&1 && echo '有 cluster 代码 ✓' || echo '✗ 未见 cluster 代码')"
echo " lease/agent/tunnel: $(ls /opt/dshs/lib/supervisor/lease.js /opt/dshs/lib/worker/agent.js /opt/dshs/lib/worker/tunnel.js 2>/dev/null | wc -l)/3"
echo "=== ③ 本地 Worker 单元(w-47,无隧道) ==="
cat > /etc/dshs-worker.env <<ENV
DSHS_DATA_ROOT=/var/lib/dshs
DSHS_ISOLATION_MODE=account
DSHS_DSH_BIN=/usr/local/bin/dsh
DSHS_BASE_UID=100000
DSH_INSTANCE_NODE_OPTIONS=--max-old-space-size=160
DSH_INSTANCE_UNIVER_SOCKET=auto
DSHS_CLUSTER_AGENT_TOKEN=$W47_TOKEN
ENV
chmod 600 /etc/dshs-worker.env
cat > /etc/systemd/system/dshs-worker.service <<UNIT
[Unit]
Description=DSHS cluster worker agent (this host = 47, local users' instances)
After=network-online.target
[Service]
Type=simple
EnvironmentFile=/etc/dshs-worker.env
ExecStart=/usr/local/bin/node /opt/dshs/lib/cli.js worker --port 19100 --host 127.0.0.1 --host-id w-47 --instance-host 127.0.0.1 --log-level info
Restart=on-failure
RestartSec=3
KillMode=mixed
[Install]
WantedBy=multi-user.target
UNIT
systemctl daemon-reload; systemctl enable dshs-worker >/dev/null 2>&1
systemctl restart dshs-worker; sleep 5
echo " dshs-worker=$(systemctl is-active dshs-worker) healthz=$(curl -s -m 6 http://127.0.0.1:19100/healthz | head -c 120)"
echo "=== ④ 既有用户归属预置为 w-47(粘性锚点) ==="
PGPASSWORD=dshs_cluster_2026 /usr/bin/psql -h 127.0.0.1 -p 15432 -U dshs -d dshs -tAc \
"insert into dsh_instances (id, user_id, role, status, host_id, epoch, heartbeat_at, lease_until)
select 'dsh-'||id, id, 'main', 'stopped', 'w-47', 0, 0, 0 from users
on conflict (id) do update set host_id='w-47', lease_until=0" 2>&1 | tail -1
PGPASSWORD=dshs_cluster_2026 /usr/bin/psql -h 127.0.0.1 -p 15432 -U dshs -d dshs -tAc \
"select u.username||' -> '||coalesce(i.host_id,'NULL') from users u left join dsh_instances i on i.user_id=u.id" 2>&1 | sed 's/^/ /'
echo "=== ⑤ 注册两台 worker ==="
PGPASSWORD=dshs_cluster_2026 /usr/bin/psql -h 127.0.0.1 -p 15432 -U dshs -d dshs -tAc \
"insert into dsh_hosts (id, endpoint, agent_token, capacity_mb, used_mb, status)
values ('w-47','http://127.0.0.1:19100','$W47_TOKEN',1024,0,'up'),
('w-106','http://127.0.0.1:19000','$W106_TOKEN',2560,0,'up')
on conflict (id) do update set endpoint=excluded.endpoint, agent_token=excluded.agent_token, capacity_mb=excluded.capacity_mb, status='up'" 2>&1 | tail -1
PGPASSWORD=dshs_cluster_2026 /usr/bin/psql -h 127.0.0.1 -p 15432 -U dshs -d dshs -tAc \
"select id||' cap='||capacity_mb||' status='||status||' ep='||endpoint from dsh_hosts order by id" 2>&1 | sed 's/^/ /'
echo "=== 回滚剧本(现在就记下) ==="
echo " rm -f /etc/systemd/system/dshs.service.d/cluster.conf && systemctl daemon-reload && \\"
echo " systemctl restart dshs # 回 SQLite 单机;lib 回滚 = cp -a /opt/dsh/backups/lib-$TS/lib /opt/dshs/lib"
echo " 备份时间戳: $TS"
+20
View File
@@ -0,0 +1,20 @@
#!/usr/bin/env bash
# A 步:把 47 的生产库 SQLite → 本机控制面 PG(停机窗口内做,避免迁移期间库变动)
set -uo pipefail
PGURL="postgres://dshs:[email protected]:15432/dshs"
DB=/var/lib/dshs/dshs.db
cd /opt/dshs-cluster || exit 1
echo "=== 1) 停 dshs(窗口开始) ==="
systemctl stop dshs
sleep 2
echo " dshs=$(systemctl is-active dshs) 实例 scope 残留: $(systemctl list-units --type=scope --all 2>/dev/null | grep -c 'dsh-' || echo 0)"
echo " 库文件: $(ls -l $DB $DB-wal 2>/dev/null | awk '{print $5}' | tr '\n' '/')"
echo "=== 2) dry-run(只报行数) ==="
node scripts/migrate-sqlite-to-pg.mjs --sqlite "$DB" --pg "$PGURL" --dry-run 2>&1 | tail -18
echo "=== 3) 真迁 ==="
node scripts/migrate-sqlite-to-pg.mjs --sqlite "$DB" --pg "$PGURL" 2>&1 | tail -18
echo " rc=$?"
echo " ⏸ 窗口保持关闭 —— 部署 worker 与 drop-in 后再一起开(见后续步骤)"
+21
View File
@@ -0,0 +1,21 @@
#!/usr/bin/env bash
# 校验:SQLite 与 PG 两侧的 users 明细 + 与磁盘目录对照(判断 2 vs 7 是孤儿目录还是迁移漏行)
set -uo pipefail
cd /opt/dshs-cluster || exit 1
echo "=== SQLite 侧(只读打开,含 WAL) ==="
node -e '
const D = require("better-sqlite3");
const db = new D("/var/lib/dshs/dshs.db", { readonly: true });
const users = db.prepare("select id, username, role, uid from users order by rowid").all();
console.log(" users 行数:", users.length);
for (const u of users) console.log(` ${u.username} role=${u.role} uid=${u.uid} id=${u.id}`);
console.log(" credential_vault:", db.prepare("select count(*) c from credential_vault").get().c);
console.log(" business_plugins:", db.prepare("select count(*) c from business_plugins").get().c);
'
echo "=== PG 侧 ==="
PGPASSWORD=dshs_cluster_2026 /usr/bin/psql -h 127.0.0.1 -p 15432 -U dshs -d dshs -tAc \
"select username||' role='||role||' uid='||coalesce(uid::text,'NULL')||' id='||id from users order by rowid" 2>&1 | sed 's/^/ /'
echo " PG users 总数: $(PGPASSWORD=dshs_cluster_2026 /usr/bin/psql -h 127.0.0.1 -p 15432 -U dshs -d dshs -tAc 'select count(*) from users')"
echo "=== 磁盘目录 vs DB ==="
echo " 目录($(ls /var/lib/dshs/users | wc -l) 个):"
ls /var/lib/dshs/users | sed 's/^/ /'
+22
View File
@@ -0,0 +1,22 @@
#!/usr/bin/env bash
# B1 步(在 47 上跑):uid 保真校验 + 建 47→106 专用密钥并打印公钥
set -uo pipefail
KEY=/root/.ssh/dshworker_ed25519
echo "=== 1) uid 保真校验(PG 与 SQLite 必须一致 —— 否则 106 上文件属主全错) ==="
echo -n " PG : "; PGPASSWORD=dshs_cluster_2026 /usr/bin/psql -h 127.0.0.1 -p 15432 -U dshs -d dshs -tAc \
"select string_agg(username||'='||coalesce(uid::text,'NULL'), ' ' order by row_id) from users" 2>&1 | head -1
echo -n " SQLite : "; node -e 'const D=require("better-sqlite3");const db=new D("/var/lib/dshs/dshs.db",{readonly:true});console.log(db.prepare("select username, uid from users order by rowid").all().map(u=>u.username+"="+u.uid).join(" "))'
echo " --- 磁盘目录属主(与 uid 对照;多出的 5 个是已删用户孤儿目录) ---"
for d in /var/lib/dshs/users/*/; do printf " %-38s uid=%s\n" "$(basename "$d")" "$(stat -c %u "$d")"; done
echo "=== 2) 建 47→106 专用密钥(仅用于 rsync) ==="
[ -f "$KEY" ] || ssh-keygen -t ed25519 -N "" -C "dshs-rsync-47to106" -f "$KEY" >/dev/null 2>&1
echo " PUBKEY=$(cat "$KEY.pub")"
echo "=== 3) 试连通 106:22 ==="
if ssh -i "$KEY" -o BatchMode=yes -o StrictHostKeyChecking=accept-new -o ConnectTimeout=8 [email protected] 'echo ok' 2>/dev/null | grep -q ok; then
echo " ✓ 已可连通(公钥已装)"
else
echo " ⏳ 尚不可连通 —— 需先把我本机把这个公钥装到 106"
fi
+38
View File
@@ -0,0 +1,38 @@
#!/usr/bin/env bash
# 切换 B 步(在 47 上跑):加 systemd drop-in → 重启 dshs(= 切换时刻,约 1-2 秒中断)
# · 用 drop-in 而非改 unit:unit 本体 hash 不变 ⇒ 回滚只需删 drop-in
# · 同时把 106 的 worker lib 也更新到同一版本(两侧代码必须一致)
set -uo pipefail
mkdir -p /etc/systemd/system/dshs.service.d
cat > /etc/systemd/system/dshs.service.d/cluster.conf <<CONF
# T08 集群化(2026-09-15):Manager 在 47、实例落在 w-47(既有用户)/ w-106(新用户)
# 回滚:删除本文件 → systemctl daemon-reload → systemctl restart dshs
[Service]
Environment="DSHS_DEPLOY_MODE=cluster"
Environment="DSHS_DB_URL=postgres://dshs:[email protected]:15432/dshs"
Environment="DSHS_CLUSTER_HOST_ID=w-47"
Environment="DSHS_CLUSTER_AGENT_URL=http://127.0.0.1:19100"
Environment="DSHS_CLUSTER_AGENT_TOKEN=dshs-worker-47-c4b7e19f"
Environment="DSHS_CLUSTER_INSTANCE_HOST=127.0.0.1"
Environment="DSHS_CLUSTER_WORKER_DATA_ROOT=/var/lib/dshs"
Environment="DSHS_CLUSTER_CAPACITY_MB=-1"
Environment="DSHS_CLUSTER_REGISTER_SELF=0"
Environment="DSHS_CLUSTER_LEASE_TTL_MS=30000"
CONF
echo " ✓ drop-in 已写($(wc -l < /etc/systemd/system/dshs.service.d/cluster.conf) 行)"
systemctl daemon-reload
systemctl restart dshs
sleep 6
echo "=== 切换后自检 ==="
echo " dshs=$(systemctl is-active dshs) | dshs-pg=$(systemctl is-active dshs-pg) | dshs-worker=$(systemctl is-active dshs-worker)"
echo " 生效 env(systemd 解析后):"
systemctl show dshs -p Environment 2>/dev/null | tr ' ' '\n' | grep -E "DEPLOY_MODE|CLUSTER_HOST_ID|CLUSTER_AGENT_URL|DB_URL" | sed 's/^/ /'
echo " 门户: login.html=$(curl -s -o /dev/null -w '%{http_code}' -m 8 http://127.0.0.1:3080/login.html)"
echo " 公网: https://alotbuy.com/login.html = $(curl -s -o /dev/null -w '%{http_code}' -m 12 https://alotbuy.com/login.html)"
echo " --- dshs cluster status ---"
cd /opt/dshs && set -a && . /etc/dshs.env && set +a && set -a && . /etc/systemd/system/dshs.service.d/cluster.conf 2>/dev/null || true
cd /opt/dshs && DSHS_DB_URL="postgres://dshs:[email protected]:15432/dshs" node lib/cli.js cluster status 2>&1 | head -8 | sed 's/^/ /'
echo " --- 回滚命令(随时可用) ---"
echo " rm -f /etc/systemd/system/dshs.service.d/cluster.conf && systemctl daemon-reload && systemctl restart dshs"
+33
View File
@@ -0,0 +1,33 @@
#!/usr/bin/env bash
# B2 步(在 47 上跑):rsync 生产 dataRoot → 106
# · --numeric-ids 保 uid/gid(否则 106 上文件属主全错 ⇒ 实例 EACCES)
# · 排除 dshs.db*(DB 权威源已是 47 的 PG)与 secret.key(凭据主密钥不外扩到 Worker —— R5 最小面)
# · ⚠️ 所有 ssh 调用带 -n:脚本本身经 stdin 传入,ssh 若不隔离 stdin 会把**脚本剩余部分**吃掉
set -uo pipefail
KEY=/root/.ssh/dshworker_ed25519
DST=[email protected]
SRC=/var/lib/dshs
SSHOPT="-n -i $KEY -o BatchMode=yes -o StrictHostKeyChecking=accept-new -o ConnectTimeout=10"
command -v rsync >/dev/null || { dnf -y install rsync >/tmp/dnf-rsync.log 2>&1 && echo " ✓ 47 装上 rsync"; }
echo "=== 连通性 + 对端 rsync ==="
ssh $SSHOPT "$DST" 'command -v rsync >/dev/null || dnf -y install rsync >/tmp/dnf-rs.log 2>&1; echo " 对端: $(rsync --version | head -1)"' 2>&1 | tail -2
echo "=== 目标端准备 ==="
ssh $SSHOPT "$DST" 'mkdir -p /var/lib/dshs && ls -ld /var/lib/dshs' 2>&1 | sed 's/^/ /'
echo "=== rsync ==="
rsync -a --numeric-ids --stats -e "ssh $SSHOPT" \
--exclude 'dshs.db' --exclude 'dshs.db-shm' --exclude 'dshs.db-wal' --exclude 'secret.key' \
"$SRC/" "$DST:/var/lib/dshs/" 2>&1 | grep -E "Number of regular files transferred|Total file size|sent [0-9]|total size is" | sed 's/^/ /'
echo "=== 目标端核对 ==="
ssh $SSHOPT "$DST" 'bash -c "
echo \" 顶层: \$(ls /var/lib/dshs | tr \"\n\" \" \")\"
echo \" users 目录数: \$(ls /var/lib/dshs/users 2>/dev/null | wc -l)\"
for d in /var/lib/dshs/users/*/; do printf \" %-38s uid=%s\n\" \"\$(basename \$d)\" \"\$(stat -c %u \$d)\"; done
echo \" bundled-skills: \$(ls /var/lib/dshs/bundled-skills 2>/dev/null | wc -l) 项\"
echo \" business-plugins: \$(ls /var/lib/dshs/business-plugins 2>/dev/null | wc -l) 项\"
echo \" ⛔ 不应存在(dshs.db/secret.key): \$(ls /var/lib/dshs/dshs.db /var/lib/dshs/secret.key 2>/dev/null | wc -l) 个(应为 0)\"
"' 2>&1
+30
View File
@@ -0,0 +1,30 @@
#!/usr/bin/env bash
# B2' 步:47 → 106 直推生产 dataRoot(tar-over-ssh;只用命令执行,绕开 rsync 协议问题)
# · tar --numeric-owner 保 uid/gid(否则 106 上属主错 ⇒ 实例 EACCES)
# · 排除 dshs.db*(权威源=47 的 PG)与 secret.key(主密钥不外扩 —— R5 最小面)
set -uo pipefail
KEY=/root/.ssh/dshworker_ed25519
DST=[email protected]
SSHO="-i $KEY -o BatchMode=yes -o StrictHostKeyChecking=accept-new"
echo "=== 1) 目标端准备(ssh -n 防抢 stdin) ==="
ssh -n $SSHO "$DST" 'mkdir -p /var/lib/dshs && echo " ready: $(ls -ld /var/lib/dshs)"'
echo "=== 2) 推送(tar → ssh → tar --numeric-owner -x) ==="
cd /var/lib/dshs
tar --numeric-owner -cf - \
--exclude=./dshs.db --exclude=./dshs.db-shm --exclude=./dshs.db-wal --exclude=./secret.key \
. | ssh $SSHO "$DST" 'tar -C /var/lib/dshs --numeric-owner -xf -'
rc=$?
echo " 管道 rc=$rc"
echo "=== 3) 目标端核对 ==="
ssh -n $SSHO "$DST" 'bash -c "
echo \" 顶层: \$(ls /var/lib/dshs | tr \"\n\" \" \")\"
echo \" 总量: \$(du -sh /var/lib/dshs | cut -f1)\"
echo \" users 目录数: \$(ls /var/lib/dshs/users | wc -l)\"
echo \" --- 属主抽样(应与 47 的 uid 一致) ---\"
for d in /var/lib/dshs/users/*/; do printf \" %-38s uid=%s\n\" \"\$(basename \$d)\" \"\$(stat -c %u \$d)\"; done
echo \" bundled-skills=\$(ls /var/lib/dshs/bundled-skills | wc -l) business-plugins=\$(ls /var/lib/dshs/business-plugins | wc -l)\"
echo \" ⛔ 不该有 dshs.db/secret.key: \$(ls /var/lib/dshs/dshs.db /var/lib/dshs/secret.key 2>/dev/null | wc -l) 个(应 0)\"
"'
+79
View File
@@ -0,0 +1,79 @@
#!/usr/bin/env bash
# 最终验证(在 47 上跑):清污染 → 重置锚点 → 重验两条路径
# ① 既有用户 guest:留 w-47 + 工作区有历史数据 + 实例页正常
# ② 新用户:落 w-106 + **文件真的写到 106 的盘** + 实例页正常
set -uo pipefail
export PGPASSWORD=dshs_cluster_2026
Q() { /usr/bin/psql -h 127.0.0.1 -p 15432 -U dshs -d dshs -tAc "$1"; }
M=http://127.0.0.1:3080
DOMAIN=alotbuy.com
T47=dshs-worker-47-c4b7e19f
T106=dshs-worker-7f3a91c05e
SSH106="ssh -n -i /root/.ssh/dshworker_ed25519 -o BatchMode=yes -o StrictHostKeyChecking=accept-new [email protected]"
mksess() {
local u="$1" T H
T=$(openssl rand -hex 32); H=$(printf '%s' "$T" | sha256sum | cut -d' ' -f1)
Q "insert into sessions (token_hash,user_id,created_at,expires_at,ip,user_agent)
select '$H', id, (extract(epoch from now())*1000)::bigint, (extract(epoch from now())*1000)::bigint+1800000, '127.0.0.1','switch-verify' from users where username='$u'" >/dev/null
printf '%s' "$T"
}
owner() { Q "select coalesce(i.host_id,'NULL')||' epoch='||coalesce(i.epoch,0) from users u left join dsh_instances i on i.user_id=u.id where u.username='$1'"; }
agent() { curl -s -m 6 -H "x-dsh-agent-token: $2" "http://127.0.0.1:$1/healthz" | grep -o '"instances":[0-9]*'; }
isapp() { grep -q '<base href=' <<<"$1" && echo "✓实例页" || echo "✗非实例页"; }
echo "########## 0) 清污染:停所有实例 + 删测试用户 ##########"
systemctl restart dshs-worker; sleep 4
$SSH106 'systemctl restart dshs-worker' >/dev/null 2>&1; sleep 4
echo " 重启后 w-47=$(agent 19100 $T47) w-106=$(agent 19000 $T106)"
AT=$(mksess admin); AC="sid=$AT"
for u in $(Q "select id from users where username like 'switchtest%' or username like 'swtest%'"); do
echo " 删测试用户 $u → $(curl -s -m 20 -X DELETE -b "$AC" "$M/api/admin/users/$u" -o /dev/null -w '%{http_code}')"
done
echo " 剩余用户: $(Q "select string_agg(username,', ') from users")"
echo "########## 1) 重置 guest 锚点(host_id=w-47) ##########"
Q "update dsh_instances set host_id='w-47', epoch=0, lease_until=0, status='stopped'
where user_id=(select id from users where username='guest')" >/dev/null
echo " $(owner guest)"
echo
echo "########## 2) 路径①:既有用户 guest ##########"
GT=$(mksess guest); GC="sid=$GT"
E=$(curl -s -m 60 -X POST -b "$GC" -H 'content-type: application/json' -d '{}' "$M/api/dsh/enter")
sleep 3
echo " 归属: $(owner guest) ← 期望 w-47"
echo " w-47=$(agent 19100 $T47) w-106=$(agent 19000 $T106)"
echo " 工作区条目: $(curl -s -m 10 -b "$GC" "$M/api/desktop/tree" | grep -o '"name":"[^"]*"' | head -4 | tr '\n' ' ')"
PAGE=$(curl -s -m 25 -L -b "$GC" -H "Host: guest.$DOMAIN" "$M/" | head -c 300)
echo " 实例页: $(isapp "$PAGE")"
echo
echo "########## 3) 路径②:新用户 ##########"
NU="swtest2$(date +%H%M%S)"
echo " register=$(curl -s -m 15 -X POST -H 'content-type: application/json' -d "{\"username\":\"$NU\",\"password\":\"SwitchTest123\"}" "$M/api/auth/register" -o /dev/null -w '%{http_code}')"
NID=$(Q "select id from users where username='$NU'")
echo " approve=$(curl -s -m 15 -X POST -b "$AC" "$M/api/admin/users/$NID/approve" -o /dev/null -w '%{http_code}')"
NC=$(curl -s -m 15 -X POST -H 'content-type: application/json' -d "{\"username\":\"$NU\",\"password\":\"SwitchTest123\"}" -D - "$M/api/auth/login" -o /dev/null | grep -i '^set-cookie' | head -1 | grep -oP 'sid=[^;]+')
echo " mkdir=$(curl -s -m 15 -X POST -b "$NC" -H 'content-type: application/json' -d '{"path":"proj"}' "$M/api/fs/mkdir" -o /dev/null -w '%{http_code}') ← 首次触达应把归属钉住"
echo " 钉住后归属: $(owner "$NU") ← 期望 w-106(与下面的 launch 必须同台)"
echo " upload=$(curl -s -m 20 -X POST -b "$NC" -H 'content-type: application/json' -d "{\"path\":\"proj\",\"name\":\"hello.txt\",\"data\":\"$(printf 'hi-from-switch' | base64 -w0)\"}" "$M/api/fs/upload" -o /dev/null -w '%{http_code}')"
echo " launch=$(curl -s -m 60 -X POST -b "$NC" -H 'content-type: application/json' -d '{"folder":"proj"}' "$M/api/dsh/launch" -o /dev/null -w '%{http_code}')"
sleep 4
echo " 归属: $(owner "$NU") ← 期望 w-106(粘性保持)"
echo " w-106=$(agent 19000 $T106) w-47=$(agent 19100 $T47)"
echo " --- 落盘取证 ---"
L47=$($SSH106 "ls /var/lib/dshs/users/$NID/ws/proj/hello.txt 2>/dev/null" 2>/dev/null || true)
echo " 106 盘: ${L47:-不存在}"
echo " 47 盘: $(ls /var/lib/dshs/users/$NID/ws/proj/hello.txt 2>/dev/null || echo 不存在(应不存在 ✓))"
NP=$(curl -s -m 25 -L -b "$NC" -H "Host: $NU.$DOMAIN" "$M/" | head -c 300)
echo " 实例页(Host: $NU.$DOMAIN): $(isapp "$NP")"
echo
echo "########## 收尾 ##########"
curl -s -m 40 -X POST -b "$GC" "$M/api/dsh/stop" -o /dev/null -w " guest stop=%{http_code}\n"
curl -s -m 40 -X POST -b "$NC" "$M/api/dsh/stop" -o /dev/null -w " newuser stop=%{http_code}\n"
echo " stop 后 guest 归属(**不应再被清空**): $(owner guest)"
Q "delete from sessions where user_agent='switch-verify'" >/dev/null
echo " 残留临时 session: $(Q "select count(*) from sessions where user_agent='switch-verify'")(应 0)"
echo " 新用户待清: $NU / $NID"
+63
View File
@@ -0,0 +1,63 @@
#!/usr/bin/env bash
# 切换后功能验证 v2(在 47 上跑)
# ① 既有用户 guest:应留 w-47,且**工作区有历史数据**(文件在 47 的盘上)
# ② 新用户:应落 w-106,且**建的文件真的出现在 106 的盘上**(文件面路由已修)
# 判据:实例页必须带 <base href="/"(门户页不算);归属看 PG;文件落盘看两台机器磁盘
set -uo pipefail
PG() { PGPASSWORD=dshs_cluster_2026 /usr/bin/psql -h 127.0.0.1 -p 15432 -U dshs -d dshs -tAc "$1"; }
M=http://127.0.0.1:3080
DOMAIN=alotbuy.com
T47=dshs-worker-47-c4b7e19f
T106=dshs-worker-7f3a91c05e
mksess() {
local u="$1" T H
T=$(openssl rand -hex 32); H=$(printf '%s' "$T" | sha256sum | cut -d' ' -f1)
PG "insert into sessions (token_hash,user_id,created_at,expires_at,ip,user_agent)
select '$H', id, (extract(epoch from now())*1000)::bigint, (extract(epoch from now())*1000)::bigint+1800000, '127.0.0.1','switch-verify' from users where username='$u'" >/dev/null
printf '%s' "$T"
}
owner() { PG "select u.username||' -> '||coalesce(i.host_id,'NULL')||' epoch='||i.epoch from users u left join dsh_instances i on i.user_id=u.id where u.username='$1'"; }
agent() { curl -s -m 6 -H "x-dsh-agent-token: $2" "http://127.0.0.1:$1/healthz" | grep -o '"instances":[0-9]*'; }
isapp() { grep -q '<base href=' <<<"$1" && echo "✓实例页" || echo "✗非实例页"; }
echo "############ ① 既有用户 guest(应留 w-47 + 工作区有数据) ############"
GT=$(mksess guest); GC="sid=$GT"
E=$(curl -s -m 60 -X POST -b "$GC" -H 'content-type: application/json' -d '{}' "$M/api/dsh/enter")
echo " enter → $(head -c 130 <<<"$E")"
sleep 3
echo " 归属: $(owner guest) ← 期望 w-47"
echo " w-47=$(agent 19100 $T47) w-106=$(agent 19000 $T106)"
echo " 工作区首项: $(curl -s -m 10 -b "$GC" "$M/api/desktop/tree" | head -c 150)"
URL=$(grep -o '"url":"[^"]*"' <<<"$E" | head -1 | cut -d'"' -f4)
PAGE=$(curl -s -m 25 -L -b "$GC" -H "Host: guest.$DOMAIN" "$M/" | head -c 300)
echo " 实例页: $(isapp "$PAGE")"
echo " 47 盘上 guest 工作区条目: $(ls /var/lib/dshs/users/4092b965-2f68-4977-9989-68b3966f7df0/ws 2>/dev/null | wc -l) 项(>0 = 数据在 47 ✓)"
echo
echo "############ ② 新用户(应落 w-106 + 文件真的写到 106 盘) ############"
NU="swtest$(date +%H%M%S)"
AT=$(mksess admin); AC="sid=$AT"
echo " register=$(curl -s -m 15 -X POST -H 'content-type: application/json' -d "{\"username\":\"$NU\",\"password\":\"SwitchTest123\"}" "$M/api/auth/register" -o /dev/null -w '%{http_code}')"
NU_ID=$(PG "select id from users where username='$NU'")
echo " approve=$(curl -s -m 15 -X POST -b "$AC" "$M/api/admin/users/$NU_ID/approve" -o /dev/null -w '%{http_code}')"
NC=$(curl -s -m 15 -X POST -H 'content-type: application/json' -d "{\"username\":\"$NU\",\"password\":\"SwitchTest123\"}" -D - "$M/api/auth/login" -o /dev/null | grep -i '^set-cookie' | head -1 | grep -oP 'sid=[^;]+')
echo " mkdir=$(curl -s -m 15 -X POST -b "$NC" -H 'content-type: application/json' -d '{"path":"proj"}' "$M/api/fs/mkdir" -o /dev/null -w '%{http_code}')"
echo " upload=$(curl -s -m 20 -X POST -b "$NC" -H 'content-type: application/json' -d "{\"path\":\"proj\",\"name\":\"hello.txt\",\"data\":\"$(printf 'hi-from-switch' | base64 -w0)\"}" "$M/api/fs/upload" -o /dev/null -w '%{http_code}')"
echo " launch=$(curl -s -m 60 -X POST -b "$NC" -H 'content-type: application/json' -d '{"folder":"proj"}' "$M/api/dsh/launch" -o /dev/null -w '%{http_code}')"
sleep 4
echo " 归属: $(owner "$NU") ← 期望 w-106"
echo " w-106=$(agent 19000 $T106) w-47=$(agent 19100 $T47)"
echo " --- 文件落盘取证(这才是文件面路由修好的证据) ---"
echo " 47 盘: $(ls /var/lib/dshs/users/$NU_ID/ws/proj/hello.txt 2>/dev/null && echo 存在 || echo '不存在 ✓(不应在 47)')"
echo " 106 盘: $(ssh -n -i /root/.ssh/dshworker_ed25519 -o BatchMode=yes -o StrictHostKeyChecking=accept-new [email protected] "ls /var/lib/dshs/users/$NU_ID/ws/proj/hello.txt 2>/dev/null" 2>/dev/null && echo 存在✓ || echo '不存在 ✗')"
NU_PAGE=$(curl -s -m 25 -L -b "$NC" -H "Host: $NU.$DOMAIN" "$M/" | head -c 300)
echo " 实例页(Host: $NU.$DOMAIN): $(isapp "$NU_PAGE")"
echo
echo "############ 收尾 ############"
curl -s -m 40 -X POST -b "$GC" "$M/api/dsh/stop" -o /dev/null -w " guest stop=%{http_code}\n"
curl -s -m 40 -X POST -b "$NC" "$M/api/dsh/stop" -o /dev/null -w " newuser stop=%{http_code}\n"
PG "delete from sessions where user_agent='switch-verify'" >/dev/null
echo " 残留临时 session: $(PG "select count(*) from sessions where user_agent='switch-verify'")(应 0)"
echo " 新用户记录: $NU / $NU_ID"
+53
View File
@@ -0,0 +1,53 @@
#!/usr/bin/env bash
# C 步(在 106 上跑):把生产 Worker agent 装成 systemd 单元
# · env 与 47 的生产实例侧对齐(DSH_INSTANCE_* 必须一致 —— 实例是在 Worker 上 spawn 的)
# · token 走 env 文件(600)而不是命令行,避免 ps 泄露
# · 反向隧道复用演练时那把 key(其公钥已在 47 的 authorized_keys 里,restrict,port-forwarding)
set -uo pipefail
TOKEN="${WORKER_TOKEN:-dshs-worker-7f3a91c05e}"
cat > /etc/dshs-worker.env <<ENV
# DSHS 集群 Worker(2026-09-15 切换)—— 与 47 /etc/dshs.env 的**实例侧**条目保持一致
DSHS_DATA_ROOT=/var/lib/dshs
DSHS_ISOLATION_MODE=account
DSHS_DSH_BIN=/usr/bin/dsh
DSHS_BASE_UID=100000
DSH_INSTANCE_NODE_OPTIONS=--max-old-space-size=160
DSH_INSTANCE_UNIVER_SOCKET=auto
# 控制通道:Worker 主动拨 47 的反向隧道(公网入方向被云安全组挡住 ⇒ 只能这个方向)
DSHS_CLUSTER_AGENT_TOKEN=$TOKEN
[email protected]:32022
DSHS_TUNNEL_IDENTITY=/root/.ssh/tunnel_ed25519
ENV
chmod 600 /etc/dshs-worker.env
cat > /etc/systemd/system/dshs-worker.service <<UNIT
[Unit]
Description=DSHS cluster worker agent (hosts per-user dsh instances on 106)
After=network-online.target
Wants=network-online.target
[Service]
Type=simple
EnvironmentFile=/etc/dshs-worker.env
ExecStart=/usr/bin/node /opt/dshs-cluster/lib/cli.js worker --port 19000 --host 127.0.0.1 --host-id w-106 --instance-host 127.0.0.1 --log-level info
Restart=on-failure
RestartSec=3
KillMode=mixed
[Install]
WantedBy=multi-user.target
UNIT
systemctl daemon-reload
systemctl enable dshs-worker >/dev/null 2>&1
echo " 单元已装并 enable;token 长度=${#TOKEN}"
# 旧的手工 agent 若在跑先停(按端口定位)
pid=$(ss -lntpH 'sport = :19000' 2>/dev/null | grep -oP 'pid=\K[0-9]+' | head -1)
[ -n "$pid" ] && kill "$pid" && sleep 2 && echo " 已停旧的手工 agent pid=$pid"
systemctl restart dshs-worker
sleep 6
echo " dshs-worker=$(systemctl is-active dshs-worker)"
echo " healthz: $(curl -s -m 6 http://127.0.0.1:19000/healthz | head -c 200)"
+36
View File
@@ -0,0 +1,36 @@
#!/usr/bin/env bash
# 切换收尾:清理验证残留 + 健康检查(在 47 上跑)
set -uo pipefail
export PGPASSWORD=dshs_cluster_2026
Q() { /usr/bin/psql -h 127.0.0.1 -p 15432 -U dshs -d dshs -tAc "$1"; }
M=http://127.0.0.1:3080
echo "=== 1) 删掉验证留下的测试用户 ==="
T=$(openssl rand -hex 32); H=$(printf '%s' "$T" | sha256sum | cut -d' ' -f1)
Q "insert into sessions (token_hash,user_id,created_at,expires_at,ip,user_agent)
select '$H', id, (extract(epoch from now())*1000)::bigint, (extract(epoch from now())*1000)::bigint+600000, '127.0.0.1','switch-cleanup' from users where username='admin'" >/dev/null
for u in $(Q "select id from users where username like 'swtest%' or username like 'switchtest%'"); do
echo " delete $u → $(curl -s -m 20 -X DELETE -b "sid=$T" "$M/api/admin/users/$u" -o /dev/null -w '%{http_code}')"
done
Q "delete from sessions where user_agent='switch-cleanup'" >/dev/null
echo " 剩余用户: $(Q "select string_agg(username||'('||role||')', ', ') from users")"
echo "=== 2) 停掉验证期间起的实例 ==="
systemctl restart dshs-worker; sleep 4
ssh -n -i /root/.ssh/dshworker_ed25519 -o BatchMode=yes -o StrictHostKeyChecking=accept-new [email protected] 'systemctl restart dshs-worker' >/dev/null 2>&1
sleep 4
echo " w-47=$(curl -s -m 6 -H 'x-dsh-agent-token: dshs-worker-47-c4b7e19f' http://127.0.0.1:19100/healthz | grep -o '\"instances\":[0-9]*')"
echo " w-106=$(ssh -n -i /root/.ssh/dshworker_ed25519 -o BatchMode=yes -o StrictHostKeyChecking=accept-new [email protected] 'curl -s -m 6 -H "x-dsh-agent-token: dshs-worker-7f3a91c05e" http://127.0.0.1:19000/healthz | grep -o .instances.:[0-9]*' 2>/dev/null)"
echo "=== 3) 健康检查 ==="
echo " dshs=$(systemctl is-active dshs) dshs-pg=$(systemctl is-active dshs-pg) dshs-worker=$(systemctl is-active dshs-worker)"
echo " 门户公网: $(curl -s -o /dev/null -w '%{http_code}' -m 12 https://alotbuy.com/login.html)"
echo " --- dshs cluster status ---"
cd /opt/dshs && set -a && . /etc/dshs.env && set +a
DSHS_DEPLOY_MODE=cluster DSHS_DB_URL="postgres://dshs:[email protected]:15432/dshs" \
node lib/cli.js cluster status 2>&1 | head -9 | sed 's/^/ /'
echo " --- dshs doctor ---"
DSHS_DEPLOY_MODE=cluster DSHS_DB_URL="postgres://dshs:[email protected]:15432/dshs" \
node lib/cli.js doctor 2>&1 | grep -cE "^✓" | sed 's/^/ ✓ 项数: /'
echo " --- 实例归属总览 ---"
Q "select u.username || ' → ' || coalesce(i.host_id,'(未指派)') || ' status=' || coalesce(i.status,'-') from users u left join dsh_instances i on i.user_id=u.id order by u.row_id" | sed 's/^/ /'
+31
View File
@@ -0,0 +1,31 @@
#!/usr/bin/env bash
# 拆除演练环境(为生产切换让路):
# · 47:演练 Manager(127.0.0.1:13080)
# · 106:两个演练 agent(19000/19001)⇒ 会 teardown 实例并关闭隧道
# 目的:① 释放 47 的 15432(隧道转发占用 → 生产 PG 要用)
# ② 避免"演练 Manager + 生产 Manager 抢同一个 agent"
set -uo pipefail
echo "=== 47 侧:停演练 Manager(按端口定位,绝不碰生产 3080) ==="
PID=$(ss -lntpH 'sport = :13080' 2>/dev/null | grep -oP 'pid=\K[0-9]+' | head -1)
if [ -n "$PID" ]; then kill "$PID" && echo " 已停演练 Manager pid=$PID"; else echo " 13080 无监听"; fi
sleep 2
echo "=== 106 侧:停两个演练 agent ==="
for port in 19000 19001; do
pid=$(ss -lntpH "sport = :$port" 2>/dev/null | grep -oP 'pid=\K[0-9]+' | head -1)
[ -n "$pid" ] && kill "$pid" && echo " 已停 agent($port) pid=$pid" || echo " $port 无监听"
done
sleep 4
echo " 残留 dsh 实例: $(pgrep -cf 'dsh --profile' || echo 0)"
echo " 隧道进程: $(pgrep -cf 'tunnel_ed25519' || echo 0)"
echo "=== 47 侧:15432 是否已释放(隧道转发应已消失) ==="
ss -lntp 2>/dev/null | grep 15432 || echo " ✓ 15432 已空闲"
echo "=== 加固:若隧道 sshd 残留,按端口收掉 ==="
for port in 19000 19001 15432; do
pid=$(ss -lntpH "sport = :$port" 2>/dev/null | grep -oP 'pid=\K[0-9]+' | head -1)
[ -n "$pid" ] && kill "$pid" 2>/dev/null && echo " 收掉残留监听 $port pid=$pid"
done
sleep 2
ss -lntp 2>/dev/null | grep -E "15432|1900[01]" || echo " ✓ 三个端口都已空闲"
+195
View File
@@ -0,0 +1,195 @@
/**
* T08 S3 · 端到端验证:**Manager 经 RemoteSpawner 把实例起在 worker agent 上**。
*
* 与 `smoke-dsh.mjs` 的区别:那条走的是"本机直接 spawn",这条**多了一跳 HTTP**
* (Manager → agent → LocalSpawner),因此它验证的是 S3 真正的交付物:
* ① 路由/代理层**一行没改**就能工作(`Spawner` 抽象 + `endpointFor` 的 host:port);
* ② **launch token 回传**(P0-6)—— 否则"登录直达会话"与 401 自愈会失效;
* ③ **幂等键**:同一 operationId 重发不会起第二个实例(Manager 超时重试是常态);
* ④ **self-fencing**:`/fence` 下发的 epoch 更高时,agent 主动停掉自己那个实例。
*
* 刻意用 **soft 隔离 + stand-in fake-dsh**:本测试要验的是**跨机协议**,
* 不是沙箱(沙箱另有 S1.6 的双机证据)。用 account 模式反而会被"夹具路径必须在
* 沙箱绑定集内"这条夹具限制干扰(见 `交接单/T08-§10.4`)。
*
* 运行:node scripts/verify-cluster-agent.mjs
*/
import { mkdirSync, mkdtempSync, rmSync } from 'node:fs'
import { tmpdir } from 'node:os'
import { dirname, join } from 'node:path'
import { fileURLToPath } from 'node:url'
import { buildServer } from '../lib/web/server.js'
import { buildWorkerAgent, AGENT_TOKEN_HEADER } from '../lib/worker/agent.js'
import { resolveConfig } from '../lib/config.js'
import { hashPassword } from '../lib/web/auth.js'
function assert(condition, message) {
if (!condition) throw new Error('ASSERT: ' + message)
}
const here = dirname(fileURLToPath(import.meta.url))
const fakeDsh = join(here, 'fake-dsh.mjs')
const TOKEN = 'verify-cluster-agent-token'
const dataRoot = mkdtempSync(join(tmpdir(), 'dsh-cluster-'))
let agentApp
let agentHandle
let app
try {
// ── 1) 起 worker agent(进程内,端口随机)──────────────────────────────
const agentConfig = resolveConfig({
port: 0,
dbPath: ':memory:',
dataRoot,
dshCommand: [process.execPath, fakeDsh],
clusterHostId: 'w-1',
})
const agent = buildWorkerAgent(agentConfig, {
hostId: 'w-1',
token: TOKEN,
port: 0,
host: '127.0.0.1',
instanceHost: '127.0.0.1',
logLevel: 'warn',
})
agentApp = agent.app
agentHandle = agent // 收尾要用 agent.stop()(会 teardown 本机实例),否则子进程孤儿化
await agentApp.listen({ host: '127.0.0.1', port: 0 })
const agentUrl = `http://127.0.0.1:${agentApp.server.address().port}`
console.log('agent ->', agentUrl)
const agentJson = async (path, { method = 'GET', body } = {}) => {
const res = await fetch(agentUrl + path, {
method,
headers: {
[AGENT_TOKEN_HEADER]: TOKEN,
...(body ? { 'content-type': 'application/json' } : {}),
},
body: body ? JSON.stringify(body) : undefined,
})
const text = await res.text()
return { status: res.status, body: text ? JSON.parse(text) : null }
}
// agent 存活 + 鉴权(不带 token 必须 401)
const hz = await agentJson('/healthz')
assert(hz.status === 200 && hz.body.hostId === 'w-1', 'agent healthz')
const noAuth = await fetch(agentUrl + '/instances')
assert(noAuth.status === 401, 'agent 拒绝无凭据请求')
// ── 2) 起 Manager(deployMode=cluster → RemoteSpawner)─────────────────
const managerConfig = resolveConfig({
port: 0,
dbPath: ':memory:',
dataRoot,
deployMode: 'cluster',
clusterAgentUrl: agentUrl,
clusterAgentToken: TOKEN,
clusterInstanceHost: '127.0.0.1',
})
app = await buildServer(managerConfig)
await app.listen({ port: 0 })
const base = `http://127.0.0.1:${app.server.address().port}`
console.log('manager ->', base, '(deployMode=cluster)')
await app.db.createUser({
id: 'u1',
username: 'carol',
passHash: await hashPassword('carolpass123'),
role: 'active',
homeDir: '/tmp/u1-home',
})
mkdirSync(join(dataRoot, 'users', 'u1', 'ws', 'proj'), { recursive: true })
const json = async (path, { method = 'GET', body, cookie } = {}) => {
const res = await fetch(base + path, {
method,
headers: {
...(body ? { 'content-type': 'application/json' } : {}),
...(cookie ? { cookie } : {}),
},
body: body ? JSON.stringify(body) : undefined,
})
const text = await res.text()
return { status: res.status, body: text ? JSON.parse(text) : null, setCookie: res.headers.get('set-cookie') }
}
// ── 3) 登录 → 拉起(实例实际落在 agent 上)────────────────────────────
let r = await json('/api/auth/login', { method: 'POST', body: { username: 'carol', password: 'carolpass123' } })
assert(r.status === 200, 'login succeeds')
const cookie = r.setCookie.split(';')[0]
r = await json('/api/dsh/status', { cookie })
assert(r.body.running === false, 'not running initially')
r = await json('/api/dsh/launch', { method: 'POST', cookie, body: { folder: 'proj' } })
console.log('launch ->', r.status, r.body?.url ? 'url 已返回' : r.body)
assert(r.status === 200, 'launch succeeds')
// ② launch token 回传(P0-6):URL 里必须带 token,否则"登录直达"失效
assert(typeof r.body.url === 'string' && r.body.url.includes('token='), 'launch token 必须回传到 URL')
// 实例真的在 **worker** 上(而不是 Manager 本机)
const onAgent = await agentJson('/instances')
assert(onAgent.body.instances.length === 1, 'worker 上有 1 个实例')
assert(onAgent.body.instances[0].userId === 'u1', 'worker 上的实例属于 u1')
console.log('agent 视角 -> 实例数', onAgent.body.instances.length)
r = await json('/api/dsh/status', { cookie })
assert(r.body.running === true, 'running after launch')
// ── 4) 代理链路(endpointFor → agent 给的 host:port)──────────────────
let proxyText
for (let i = 0; i < 20; i += 1) {
try {
const res = await fetch(`${base}/u/u1/dsh/hello`, { headers: { cookie } })
if (res.status === 200) {
proxyText = await res.text()
break
}
} catch {
/* 子进程还没监听,重试 */
}
await new Promise((resolve) => setTimeout(resolve, 100))
}
assert(proxyText !== undefined && proxyText.includes('fake-dsh'), 'proxy reaches the child DSH(经远端协议)')
console.log('proxy -> 200 且命中 fake-dsh')
// ── 5) 幂等键:同 operationId 重发不得起第二个实例 ─────────────────────
const opId = 'verify-idempotent-1'
const l1 = await agentJson('/launch', { method: 'POST', body: { userId: 'u1', folder: join(dataRoot, 'users', 'u1', 'ws', 'proj'), patch: undefined, operationId: opId } })
const l2 = await agentJson('/launch', { method: 'POST', body: { userId: 'u1', folder: join(dataRoot, 'users', 'u1', 'ws', 'proj'), patch: undefined, operationId: opId } })
assert(l1.status === 200 && l2.status === 200, '重复 launch 不报错')
const afterIdem = await agentJson('/instances')
assert(afterIdem.body.instances.length === 1, '幂等:仍然只有 1 个实例')
console.log('幂等 -> 同 operationId 重发后实例数仍为', afterIdem.body.instances.length)
// ── 6) self-fencing:更高 epoch 下发 ⇒ agent 主动停掉自己那个实例 ───────
await agentJson('/launch', { method: 'POST', body: { userId: 'u1', epoch: 1, operationId: 'verify-epoch-1' } })
const f1 = await agentJson('/fence', { method: 'POST', body: { userId: 'u1', epoch: 1 } })
assert(f1.body.fenced === false, 'epoch 相同 ⇒ 不被 fence')
const f2 = await agentJson('/fence', { method: 'POST', body: { userId: 'u1', epoch: 2 } })
assert(f2.body.fenced === true, 'epoch 更高 ⇒ self-fence')
const afterFence = await agentJson('/instances')
assert(afterFence.body.instances.length === 0, 'fence 后实例已停')
console.log('self-fence -> epoch 1→2 触发,实例已停止')
// ── 7) 停止 ───────────────────────────────────────────────────────────
r = await json('/api/dsh/stop', { method: 'POST', cookie })
assert(r.status === 200, 'stop succeeds')
r = await json('/api/dsh/status', { cookie })
assert(r.body.running === false, 'stopped after stop')
console.log('\nOK: cluster 模式(Manager → worker agent → 实例)端到端通过')
console.log(' ✓ 路由/代理层零改动 ✓ launch token 回传 ✓ 幂等键 ✓ self-fencing')
} finally {
await app?.close()
// ⚠️ 必须走 agent.stop():它先 teardown 本机实例再关 HTTP —— 否则 fake-dsh 孤儿会继承
// stdout,管道不关 ⇒ ssh / CI 挂死(2026-09-15 实测)。
await agentHandle?.stop()
await new Promise((resolve) => setTimeout(resolve, 500))
try {
rmSync(dataRoot, { recursive: true, force: true })
} catch {
// best-effort:Windows 上子进程的 cwd 还在里面时会 EBUSY(temp 目录会被系统回收)
}
}
+184
View File
@@ -0,0 +1,184 @@
/**
* T08 · **真跨机演练**驱动脚本(在 Manager 那台机器上运行)。
*
* 与 `verify-cluster-live.mjs`(同机、脚本自己起进程)的区别:这里**假设两侧都已部署好**:
* · Manager 运行在**本机**(47)`http://127.0.0.1:13080`
* · Worker agent 运行在**另一台机器**(106),经 **SSH 反向隧道**出现在本机 `127.0.0.1:19000`
* · 控制面 PG 也在**另一台机器**(106)上,经隧道出现在本机 `127.0.0.1:15432`
* 它回答的是本次演练的核心问题:**跨机到底能不能用**(含跨机代理取页面、跨 worker 迁移)。
*
* 运行(在 47 上):MANAGER=http://127.0.0.1:13080 AGENT_TOKEN=cross-machine-token \
* AGENT2=http://127.0.0.1:19001 node scripts/verify-cluster-cross.mjs
*/
const MANAGER = process.env.MANAGER ?? 'http://127.0.0.1:13080'
const TOKEN = process.env.AGENT_TOKEN ?? 'cross-machine-token'
const AGENT1 = process.env.AGENT1 ?? 'http://127.0.0.1:19000'
const AGENT2 = process.env.AGENT2 ?? ''
const ADMIN_PW = process.env.ADMIN_PW ?? 'crossmgr123'
const USER_PW = 'crossuser123'
function assert(condition, message) {
if (!condition) throw new Error('ASSERT: ' + message)
}
const sleep = (ms) => new Promise((r) => setTimeout(r, ms))
const json = async (path, { method = 'GET', body, cookie } = {}) => {
const res = await fetch(MANAGER + path, {
method,
headers: { ...(body ? { 'content-type': 'application/json' } : {}), ...(cookie ? { cookie } : {}) },
body: body ? JSON.stringify(body) : undefined,
signal: AbortSignal.timeout(30_000),
})
const text = await res.text()
let parsed = null
try {
parsed = text === '' ? null : JSON.parse(text)
} catch {
parsed = { raw: text.slice(0, 200) }
}
return { status: res.status, body: parsed, setCookie: res.headers.get('set-cookie') }
}
/** 直接问 worker(绕过 Manager)—— 证明实例真的落在**那台机器**上。 */
const agent = async (base, path) => {
const res = await fetch(base + path, { headers: { 'x-dsh-agent-token': TOKEN }, signal: AbortSignal.timeout(10_000) })
const text = await res.text()
return { status: res.status, body: text === '' ? null : JSON.parse(text) }
}
/** 取页面:跟随重定向(dsh 首页 303),并对启动窗口的断连做重试。 */
async function fetchPage(url, cookie, tries = 40) {
let status = 0
let snippet = ''
for (let i = 0; i < tries; i += 1) {
try {
const res = await fetch(MANAGER + url, { headers: { cookie }, redirect: 'follow', signal: AbortSignal.timeout(20_000) })
status = res.status
if (res.status === 200) {
snippet = (await res.text()).slice(0, 200)
break
}
} catch {
status = 0
}
await sleep(1000)
}
return { status, snippet }
}
async function waitRunning(cookie, tries = 60) {
for (let i = 0; i < tries; i += 1) {
const st = await json('/api/dsh/status', { cookie })
if (st.body?.running === true) return st.body
if (st.body?.instance?.status === 'crashed') return st.body
await sleep(1000)
}
return await json('/api/dsh/status', { cookie }).then((r) => r.body)
}
try {
console.log('=== 跨机演练:Manager=%s Worker=%s ===', MANAGER, AGENT1)
// ── 0) 两侧可达性(跨机链路的第一层证据)─────────────────────────────
const h1 = await agent(AGENT1, '/healthz')
assert(h1.status === 200 && h1.body.hostId === 'w-106', `Worker w-106 应可达(实际 ${JSON.stringify(h1.body)})`)
assert(h1.body.tunnel?.ready === true, `Worker 侧隧道应就绪(实际 ${JSON.stringify(h1.body.tunnel)})`)
console.log('⓪ worker 可达 -> %s(隧道 ready,已转发 %s)', h1.body.hostId, JSON.stringify(h1.body.tunnel.ports))
// ── 1) 管理面:注册 worker(join 脚本干的事)──────────────────────────
const adm = await json('/api/auth/login', { method: 'POST', body: { username: 'root', password: ADMIN_PW } })
assert(adm.status === 200, `管理员登录失败 ${adm.status}`)
const adminCookie = adm.setCookie.split(';')[0]
let r = await json('/api/admin/hosts', {
method: 'POST',
cookie: adminCookie,
body: { id: 'w-106', endpoint: AGENT1, token: TOKEN, capacityMb: 4096 },
})
assert(r.status === 200, `注册 w-106 失败 ${r.status}`)
const hosts = await json('/api/admin/hosts', { cookie: adminCookie })
assert(hosts.body.hosts.some((h) => h.id === 'w-106'), 'w-106 出现在 worker 目录')
assert(!('agentToken' in (hosts.body.hosts[0] ?? {})), '**绝不下发 agentToken**')
console.log('① 注册 -> w-106(列表不含 agentToken)')
// ── 2) 用户流程 ───────────────────────────────────────────────────────
const uname = `crossuser${Date.now() % 100000}`
r = await json('/api/auth/register', { method: 'POST', body: { username: uname, password: USER_PW } })
assert(r.status === 201, `注册应 201(实际 ${r.status})`)
const users = await json('/api/admin/users', { cookie: adminCookie })
const target = users.body.users.find((u) => u.username === uname)
assert(target !== undefined, '管理员能看到待审用户')
r = await json(`/api/admin/users/${target.id}/approve`, { method: 'POST', cookie: adminCookie })
assert(r.status === 200, `审批应 200(实际 ${r.status})`)
const login = await json('/api/auth/login', { method: 'POST', body: { username: uname, password: USER_PW } })
assert(login.status === 200, `用户登录失败 ${login.status}`)
const cookie = login.setCookie.split(';')[0]
console.log('② 用户流程 -> 注册→审批→登录(uid=%s)', target.id)
// ── 3) 文件面跨机(Manager 在 47、目录落在 106)───────────────────────
r = await json('/api/fs/mkdir', { method: 'POST', cookie, body: { path: 'proj' } })
assert(r.status === 200, `mkdir 应 200(实际 ${r.status} ${JSON.stringify(r.body)})`)
console.log('③ 文件面 -> mkdir 经隧道落到 106 的 worker')
// ── 4) 拉起实例(真 dsh 在 **106** 上)────────────────────────────────
r = await json('/api/dsh/launch', { method: 'POST', cookie, body: { folder: 'proj' } })
assert(r.status === 200, `launch 应 200(实际 ${r.status} ${JSON.stringify(r.body)})`)
const st = await waitRunning(cookie)
assert(st?.running === true, `实例应 running(实际 ${JSON.stringify(st)?.slice(0, 300)})`)
const onAgent = await agent(AGENT1, '/instances')
assert(onAgent.body.instances.length === 1, 'worker(106) 上有 1 个实例')
console.log('④ 拉起 -> running=true,**实例在 106 上**(worker /instances=%d)', onAgent.body.instances.length)
// ── 5) 登录直达 + **跨机取页面**(本演练的核心证据)───────────────────
const enter = await json('/api/dsh/enter', { method: 'POST', cookie })
assert(enter.status === 200, `enter 应 200(实际 ${enter.status})`)
const url = enter.body.url
assert(typeof url === 'string' && url.includes('token='), `enter 应带 token(实际 ${url})`)
const page = await fetchPage(url, cookie)
assert(page.status === 200, `**跨机页面**应 200(实际 ${page.status})`)
console.log('⑤ 跨机页面 -> 200(47 的 Manager 代理到 106 的实例;片段 %s)', page.snippet.replace(/\s+/g, ' ').slice(0, 60))
// ── 6) 第二台 worker(106 上模拟的第二台服务器)+ 迁移 ────────────────
if (AGENT2 !== '') {
const h2 = await agent(AGENT2, '/healthz')
assert(h2.status === 200, `第二台 worker 应可达(实际 ${h2.status})`)
r = await json('/api/admin/hosts', {
method: 'POST',
cookie: adminCookie,
body: { id: 'w-106b', endpoint: AGENT2, token: TOKEN, capacityMb: 4096 },
})
assert(r.status === 200, `注册 w-106b 失败 ${r.status}`)
console.log('⑥ 第二台 -> %s(模拟的第二台服务器)已注册', h2.body.hostId)
r = await json(`/api/admin/users/${target.id}/dsh/migrate`, {
method: 'POST',
cookie: adminCookie,
body: { targetHost: 'w-106b' },
})
assert(r.status === 200, `迁移应 200(实际 ${r.status} ${JSON.stringify(r.body)})`)
assert(r.body.to === 'w-106b', `迁移目标应为 w-106b(实际 ${r.body.to})`)
await waitRunning(cookie)
const a1 = await agent(AGENT1, '/instances')
const a2 = await agent(AGENT2, '/instances')
assert(a1.body.instances.length === 0 && a2.body.instances.length === 1, '实例应从 w-106 移到 w-106b')
console.log('⑦ 跨机迁移 -> %s → %s(epoch=%d),源机已空、目标机有 1 个实例', r.body.from, r.body.to, r.body.epoch)
const enter2 = await json('/api/dsh/enter', { method: 'POST', cookie })
assert(enter2.status === 200 && enter2.body.url !== url, '迁移后 enter 应给**新** URL')
const page2 = await fetchPage(enter2.body.url, cookie)
assert(page2.status === 200, `迁移后页面应 200(实际 ${page2.status})`)
console.log('⑧ 迁移后 -> 新 token URL 页面 200')
} else {
console.log('⑥⑦⑧ 跳过(未提供 AGENT2)')
}
// ── 9) 收尾 ───────────────────────────────────────────────────────────
r = await json('/api/dsh/stop', { method: 'POST', cookie })
assert(r.status === 200, `stop 应 200(实际 ${r.status})`)
console.log('⑨ 停止 -> ok')
console.log('\nOK: **真跨机**(47 当 Manager / 106 当 Worker,隧道跨界)演练通过')
console.log(' ✓ worker 可达 ✓ 注册 ✓ 用户流程 ✓ 文件面跨机 ✓ 实例在 106 ✓ 跨机取页面 ✓ 跨 worker 迁移')
} finally {
/* 不主动清理:实例由调用方决定留或停(演练后要观察现场) */
}
+167
View File
@@ -0,0 +1,167 @@
/**
* T08 · **域名形态访问**验证(在演练环境做:不动生产、不动 DNS、不动证书)。
*
* 要回答的问题:生产切到 cluster(Manager 在 47、实例在 106)后,
* **按域名形态访问**(`<用户名>.alotbuy.com`)还能不能正常落到 106 上的实例?
*
* 做法:给演练 Manager 设一个**测试 baseDomain**,用**显式 `Host` 头**打进去。
*
* ⚠️ 关键坑(2026-09-15 实际踩到,两次假阳性都源于它):**`fetch` 会静默丢弃 `Host` 头**
* (Fetch 规范把它列为禁止头,undici 直接忽略)⇒ 请求落到"无租户"的门户路由、回 200 门户页,
* 看起来"验证通过"其实是假的。⇒ **必须用 curl(`-H Host:`)**,且判据不能只看状态码。
*
* 运行(在 47 上):MANAGER=http://127.0.0.1:13080 BASE_DOMAIN=test.alotbuy.com \
* AGENT=http://127.0.0.1:19000 AGENT_TOKEN=cross-machine-token \
* node scripts/verify-cluster-domain.mjs
*/
import { execFileSync } from 'node:child_process'
import { readFileSync } from 'node:fs'
const MANAGER = process.env.MANAGER ?? 'http://127.0.0.1:13080'
const BASE_DOMAIN = process.env.BASE_DOMAIN ?? 'test.alotbuy.com'
const AGENT = process.env.AGENT ?? 'http://127.0.0.1:19000'
const TOKEN = process.env.AGENT_TOKEN ?? 'cross-machine-token'
const ADMIN_PW = process.env.ADMIN_PW ?? 'crossmgr123'
const USER_PW = 'domainuser123'
function assert(condition, message) {
if (!condition) throw new Error('ASSERT: ' + message)
}
const sleep = (ms) => new Promise((r) => setTimeout(r, ms))
const json = async (path, { method = 'GET', body, cookie } = {}) => {
const res = await fetch(MANAGER + path, {
method,
headers: { ...(body ? { 'content-type': 'application/json' } : {}), ...(cookie ? { cookie } : {}) },
body: body ? JSON.stringify(body) : undefined,
signal: AbortSignal.timeout(30_000),
})
const text = await res.text()
let parsed = null
try {
parsed = text === '' ? null : JSON.parse(text)
} catch {
parsed = { raw: text.slice(0, 200) }
}
return { status: res.status, body: parsed, setCookie: res.headers.get('set-cookie') }
}
/**
* 用 **curl** 带 `Host` 头取页面(`-L` 跟随重定向 ⇒ 等价真实浏览器)。
* 返回 `{ status, body }`,body 从临时文件读(避免编码/二进制问题)。
*/
function getByHost(sub, path, cookie) {
const out = '/tmp/dompage.html'
const args = [
'-s',
'-L',
'--max-time',
'25',
'-o',
out,
'-w',
'%{http_code}',
'-H',
`Host: ${sub}.${BASE_DOMAIN}`,
...(cookie ? ['-b', cookie] : []),
`${MANAGER}${path}`,
]
let status = '0'
try {
status = execFileSync('curl', args, { encoding: 'utf8' }).trim()
} catch {
status = '0'
}
let body = ''
try {
body = readFileSync(out, 'utf8')
} catch {
body = ''
}
return { status: Number(status), body }
}
/**
* 判据:**dsh 实例页**带 `<base href="/">`(子路径与子域两种形态都带);平台门户页不带。
* ⚠️ 只靠"含 dsh 字样"会把门户页误判成实例页(实测踩过这个假阳性)。
*/
const isDshApp = (html) => typeof html === 'string' && html.includes('<base href=')
const describe = (html) => {
const hit = []
if (isDshApp(html)) hit.push('base-href')
if (html.includes('/api/auth/login')) hit.push('platform-login')
const t = /<title>([^<]*)<\/title>/.exec(html)
return `${hit.join(',') || '(无特征)'} | title=${t === null ? '?' : t[1].trim()} | 首100字: ${html.replace(/\s+/g, ' ').slice(0, 100)}`
}
async function waitRunning(cookie, tries = 60) {
for (let i = 0; i < tries; i += 1) {
const st = await json('/api/dsh/status', { cookie })
if (st.body?.running === true) return true
await sleep(1000)
}
return false
}
try {
console.log('=== 域名形态验证:baseDomain=%s(Manager=%s)===', BASE_DOMAIN, MANAGER)
const adm = await json('/api/auth/login', { method: 'POST', body: { username: 'root', password: ADMIN_PW } })
assert(adm.status === 200, `管理员登录失败 ${adm.status}`)
const adminCookie = adm.setCookie.split(';')[0]
const uname = `domuser${Date.now() % 100000}`
let r = await json('/api/auth/register', { method: 'POST', body: { username: uname, password: USER_PW } })
assert(r.status === 201, `注册应 201(实际 ${r.status})`)
const users = await json('/api/admin/users', { cookie: adminCookie })
const target = users.body.users.find((u) => u.username === uname)
assert(target !== undefined, '能看到待审用户')
await json(`/api/admin/users/${target.id}/approve`, { method: 'POST', cookie: adminCookie })
const login = await json('/api/auth/login', { method: 'POST', body: { username: uname, password: USER_PW } })
assert(login.status === 200, `用户登录失败 ${login.status}`)
const cookie = login.setCookie.split(';')[0]
console.log('① 用户 -> %s(uid=%s)', uname, target.id)
r = await json('/api/fs/mkdir', { method: 'POST', cookie, body: { path: 'proj' } })
assert(r.status === 200, `mkdir 失败 ${r.status}`)
r = await json('/api/dsh/launch', { method: 'POST', cookie, body: { folder: 'proj' } })
assert(r.status === 200, `launch 失败 ${r.status} ${JSON.stringify(r.body)}`)
assert(await waitRunning(cookie), '实例应 running')
console.log('② 拉起 -> running=true')
const enter = await json('/api/dsh/enter', { method: 'POST', cookie })
assert(enter.status === 200, `enter 失败 ${enter.status}`)
const url = enter.body.url
console.log('③ 直达 URL -> %s', url)
assert(url.startsWith('https://'), `baseDomain 生效时应为 https://<子域>/(实际 ${url})`)
assert(url.includes(`${uname}.${BASE_DOMAIN}`), `URL 应含用户名子域(实际 ${url})`)
// ④ 子域形态访问:**必须带 token**(真 dsh 没 token 只给自己的登录页)
const token = new URL(url).searchParams.get('token') ?? ''
assert(token !== '', `enter URL 应带 token(实际 ${url})`)
let page = { status: 0, body: '' }
for (let i = 0; i < 40; i += 1) {
page = getByHost(uname, `/?token=${encodeURIComponent(token)}`, cookie)
if (page.status === 200 && isDshApp(page.body)) break
await sleep(1000)
}
assert(page.status === 200, `子域访问应 200(实际 ${page.status})`)
assert(isDshApp(page.body), `子域访问必须是**真的 dsh 实例页**(实际 ${describe(page.body)})`)
console.log('④ 子域访问 -> 200 且是**真 dsh 实例页**(Host: %s.%s → 106 上的实例)', uname, BASE_DOMAIN)
// ⑤ 越权对照:拿 A 的 cookie 访问**另一个真实用户**(root)的子域 ⇒ 必须 401/403
const other = getByHost('root', `/?token=${encodeURIComponent(token)}`, cookie)
assert(!isDshApp(other.body), `越权响应绝不能是实例页(实际 ${describe(other.body)})`)
assert([401, 403].includes(other.status), `用 A 的 cookie 访问 root 子域应 401/403(实际 ${other.status})`)
console.log('⑤ 越权对照 -> 用 A 的 cookie 访问 root 子域 = %d(正确拒绝)', other.status)
// ⑥ 独立取证:实例确实在 106
const onAgent = await fetch(`${AGENT}/instances`, { headers: { 'x-dsh-agent-token': TOKEN } }).then((x) => x.json())
assert(onAgent.instances.length >= 1, 'worker(106) 上应有实例')
console.log('⑥ 取证 -> 实例确实在 106(worker /instances=%d)', onAgent.instances.length)
await json('/api/dsh/stop', { method: 'POST', cookie })
console.log('\nOK: **域名形态访问**在 cluster 下可用(子域 → Manager(47) → 实例(106)),且越权被拒')
} finally {
/* 现场保留 */
}
+155
View File
@@ -0,0 +1,155 @@
/**
* T08 S5 · 跨机文件面验证。
*
* 关键设计:**Manager 的 `dataRoot` 故意与 worker 的 `dataRoot` 不同** ——
* 只有这样"文件面真的走了远端"才被证明;若两个 root 相同,本地实现也能碰巧通过。
*
* 验的是:
* ① 门户的路由(`/api/desktop/tree`、`/api/fs/*`)在 cluster 模式下照常工作(**路由零改动**);
* ② 文件**落在 worker 的 dataRoot 下**、且**不在** Manager 的 dataRoot 下;
* ③ 路径安全与本地**同源**(`bad_path` 走同一条 `resolveWithinRoot`);
* ④ `resolvePath` 返回的是**实例眼里的路径**(按 worker 的 dataRoot 算)。
*
* 运行:node scripts/verify-cluster-fs.mjs
*/
import { existsSync, mkdtempSync, rmSync } from 'node:fs'
import { tmpdir } from 'node:os'
import { dirname, join } from 'node:path'
import { fileURLToPath } from 'node:url'
import { buildServer } from '../lib/web/server.js'
import { buildWorkerAgent, AGENT_TOKEN_HEADER } from '../lib/worker/agent.js'
import { resolveConfig } from '../lib/config.js'
import { hashPassword } from '../lib/web/auth.js'
function assert(condition, message) {
if (!condition) throw new Error('ASSERT: ' + message)
}
const here = dirname(fileURLToPath(import.meta.url))
const fakeDsh = join(here, 'fake-dsh.mjs')
const TOKEN = 'verify-cluster-fs-token'
const workerRoot = mkdtempSync(join(tmpdir(), 'dsh-cfs-worker-'))
const managerRoot = mkdtempSync(join(tmpdir(), 'dsh-cfs-manager-'))
let agentApp
let agentHandle
let app
try {
// ── worker agent(dataRoot = workerRoot)────────────────────────────────
const agent = buildWorkerAgent(
resolveConfig({ port: 0, dbPath: ':memory:', dataRoot: workerRoot, dshCommand: [process.execPath, fakeDsh], clusterHostId: 'w-1' }),
{ hostId: 'w-1', token: TOKEN, port: 0, host: '127.0.0.1', instanceHost: '127.0.0.1', logLevel: 'warn' },
)
agentApp = agent.app
agentHandle = agent
await agentApp.listen({ host: '127.0.0.1', port: 0 })
const agentUrl = `http://127.0.0.1:${agentApp.server.address().port}`
console.log('worker -> dataRoot %s', workerRoot)
// ── Manager(dataRoot = managerRoot ≠ workerRoot;显式告知 worker 的 root)──
app = await buildServer(
resolveConfig({
port: 0,
dbPath: ':memory:',
dataRoot: managerRoot,
deployMode: 'cluster',
clusterAgentUrl: agentUrl,
clusterAgentToken: TOKEN,
clusterInstanceHost: '127.0.0.1',
clusterHostId: 'm-1',
clusterWorkerDataRoot: workerRoot,
}),
)
await app.listen({ port: 0 })
const base = `http://127.0.0.1:${app.server.address().port}`
console.log('manager -> dataRoot %s(与 worker 不同 ⇒ 能证明走远端)', managerRoot)
const json = async (path, { method = 'GET', body, cookie } = {}) => {
const res = await fetch(base + path, {
method,
headers: { ...(body ? { 'content-type': 'application/json' } : {}), ...(cookie ? { cookie } : {}) },
body: body ? JSON.stringify(body) : undefined,
})
const text = await res.text()
return { status: res.status, body: text ? JSON.parse(text) : null, setCookie: res.headers.get('set-cookie') }
}
await app.db.createUser({
id: 'u1',
username: 'bob',
passHash: await hashPassword('bobpass123'),
role: 'active',
homeDir: '/tmp/u1-home',
})
// 用户根必须建在 **worker 上**(这一步本身就走远端)
await app.userFs.initUserRoot('u1')
// ④ resolvePath = 实例眼里的路径(按 worker 的 dataRoot)
const resolved = app.userFs.resolvePath('u1', 'proj')
assert(resolved === join(workerRoot, 'users', 'u1', 'ws', 'proj'), `resolvePath 应按 worker 的 root 计算(实际 ${resolved})`)
console.log('④ resolvePath -> %s', resolved)
// ── ① 门户路由(零改动)──────────────────────────────────────────────
let r = await json('/api/auth/login', { method: 'POST', body: { username: 'bob', password: 'bobpass123' } })
assert(r.status === 200, 'login succeeds')
const cookie = r.setCookie.split(';')[0]
r = await json('/api/desktop/tree', { cookie })
assert(r.status === 200 && r.body.entries.length === 0, '空工作区列出 0 项')
r = await json('/api/fs/mkdir', { method: 'POST', cookie, body: { path: 'proj' } })
assert(r.status === 200, `mkdir 经远端成功(实际 ${r.status} ${JSON.stringify(r.body)})`)
r = await json('/api/fs/upload', {
method: 'POST',
cookie,
body: { path: 'proj', name: 'hello.txt', data: Buffer.from('hi there').toString('base64') },
})
assert(r.status === 200, `upload 经远端成功(实际 ${r.status})`)
r = await json('/api/desktop/tree', { cookie })
assert(r.status === 200 && r.body.entries.length === 1, '工作区里出现了 proj')
console.log('① 门户路由 -> tree/mkdir/upload 全部经远端通过')
// ── ② 文件真的落在 worker 上 ──────────────────────────────────────────
const onWorker = join(workerRoot, 'users', 'u1', 'ws', 'proj', 'hello.txt')
const onManager = join(managerRoot, 'users', 'u1', 'ws', 'proj', 'hello.txt')
assert(existsSync(onWorker), `文件应落在 worker:${onWorker}`)
assert(!existsSync(onManager), `文件不该出现在 Manager 本地:${onManager}`)
console.log('② 落点 -> worker 有、manager 无(确认走远端)')
// ── ③ 路径安全与本地同源 ──────────────────────────────────────────────
r = await json('/api/fs/mkdir', { method: 'POST', cookie, body: { path: '../evil' } })
assert(r.status === 400 && r.body.error === 'bad_path', `越界路径应 400 bad_path(实际 ${r.status} ${JSON.stringify(r.body)})`)
for (const bad of ['..', '../../etc']) {
let threw = false
try {
app.userFs.resolvePath('u1', bad)
} catch (err) {
threw = err.code === 'bad_path'
}
assert(threw, `resolvePath(${bad}) 应抛 bad_path`)
}
console.log('③ 路径安全 -> bad_path 与本地同源(走同一个 resolveWithinRoot)')
// 下载回读(readFile 经远端)—— 注意该路由回的是**原始字节**,不是 JSON
const dl = await fetch(`${base}/api/fs/download?path=proj/hello.txt`, { headers: { cookie } })
assert(dl.status === 200, `download 经远端成功(实际 ${dl.status})`)
const downloaded = await dl.text()
assert(downloaded === 'hi there', `下载内容应为上传的原文(实际 ${JSON.stringify(downloaded)})`)
console.log(' 下载回读 -> readFile 经远端成功(内容逐字一致)')
console.log('\nOK: 跨机文件面(RemoteUserFs → agent /fs/*)通过')
console.log(' ✓ 门户路由零改动 ✓ 落在 worker ✓ 路径安全同源 ✓ resolvePath 按 worker 计算')
} finally {
await app?.close()
await agentHandle?.stop()
await new Promise((r) => setTimeout(r, 300))
for (const dir of [workerRoot, managerRoot]) {
try {
rmSync(dir, { recursive: true, force: true })
} catch {
/* best-effort */
}
}
}
+208
View File
@@ -0,0 +1,208 @@
/**
* T08 S4 · 归属租约端到端验证(1a 形态:多 Manager + 一个 worker agent + 共享 PG)。
*
* 验的是 S4 的四条承重行为:
* ① **归属真的落库**:launch 后 `dsh_instances` 有 `host_id` / `epoch` / `lease_until`;
* ② **单写者**:另一个 Manager(另一个 worker 身份)在租约存活期内**拉不起来**同一用户
* ⇒ 抛 `LeaseBusyError`(退让,不是接管);
* ③ **stop 释放归属** ⇒ 别人立刻能接(不用等 TTL);
* ④ **失权即 self-fence**:租约被抢走后,原持有者下一次心跳会把"更高 epoch"下发给 worker,
* 由 worker **停掉自己那个实例**(防双写的最后一道防线)。
* ⑤ 顺带验**注册 + 心跳**:`dsh_hosts` 里有记录且 `last_heartbeat` 持续更新。
*
* 需要 PG(两个 Manager 必须共享 DB,否则谈不上"归属"):
* CLUSTER_TEST_PG_URL=postgres://dshs:[email protected]:15432/dshs_smoke node scripts/verify-cluster-lease.mjs
*/
import { mkdirSync, mkdtempSync, rmSync } from 'node:fs'
import { tmpdir } from 'node:os'
import { dirname, join } from 'node:path'
import { fileURLToPath } from 'node:url'
import pg from 'pg'
import { buildServer } from '../lib/web/server.js'
import { buildWorkerAgent, AGENT_TOKEN_HEADER } from '../lib/worker/agent.js'
import { resolveConfig } from '../lib/config.js'
import { hashPassword } from '../lib/web/auth.js'
function assert(condition, message) {
if (!condition) throw new Error('ASSERT: ' + message)
}
const sleep = (ms) => new Promise((r) => setTimeout(r, ms))
const PG_URL = process.env.CLUSTER_TEST_PG_URL
if (PG_URL === undefined || PG_URL === '') {
console.error('需要 CLUSTER_TEST_PG_URL(两个 Manager 必须共享同一个库)')
process.exit(2)
}
const TOKEN = 'verify-cluster-lease-token'
const here = dirname(fileURLToPath(import.meta.url))
const fakeDsh = join(here, 'fake-dsh.mjs')
const dataRoot = mkdtempSync(join(tmpdir(), 'dsh-lease-'))
// 短 TTL:TTL=1200ms > 2×renew=500ms(满足不变量),便于在秒级制造"过期/被抢"
process.env.DSHS_CLUSTER_LEASE_TTL_MS = '1200'
process.env.DSHS_CLUSTER_LEASE_RENEW_MS = '500'
let agentApp
let agentHandle
const managers = []
/** 清空测试库里的三张表(该库专供本测试)。 */
async function resetPg() {
const client = new pg.Client({ connectionString: PG_URL })
await client.connect()
await client.query('DELETE FROM dsh_instances')
await client.query('DELETE FROM dsh_hosts')
await client.query('DELETE FROM users')
await client.end()
}
/** 起一个 Manager(cluster 模式)。hostId 即"它绑定的 worker 身份"。 */
async function startManager(hostId) {
const app = await buildServer(
resolveConfig({
port: 0,
dbUrl: PG_URL,
dataRoot,
deployMode: 'cluster',
clusterAgentUrl: agentUrl,
clusterAgentToken: TOKEN,
clusterInstanceHost: '127.0.0.1',
clusterHostId: hostId,
}),
)
await app.listen({ port: 0 })
managers.push(app)
return app
}
let agentUrl = ''
try {
await resetPg()
// ── 0) worker agent ───────────────────────────────────────────────────
const agentConfig = resolveConfig({
port: 0,
dbPath: ':memory:',
dataRoot,
dshCommand: [process.execPath, fakeDsh],
clusterHostId: 'w-1',
})
const agent = buildWorkerAgent(agentConfig, {
hostId: 'w-1',
token: TOKEN,
port: 0,
host: '127.0.0.1',
instanceHost: '127.0.0.1',
logLevel: 'warn',
})
agentApp = agent.app
agentHandle = agent
await agentApp.listen({ host: '127.0.0.1', port: 0 })
agentUrl = `http://127.0.0.1:${agentApp.server.address().port}`
const agentJson = async (path, { method = 'GET', body } = {}) => {
const res = await fetch(agentUrl + path, {
method,
headers: { [AGENT_TOKEN_HEADER]: TOKEN, ...(body ? { 'content-type': 'application/json' } : {}) },
body: body ? JSON.stringify(body) : undefined,
})
const text = await res.text()
return { status: res.status, body: text ? JSON.parse(text) : null }
}
// ── 1) 两个 Manager(不同 worker 身份)────────────────────────────────
const m1 = await startManager('m-1')
const m2 = await startManager('m-2')
console.log('manager -> m-1 %s / m-2 %s(共享 PG + 同一 agent)', m1.supervisor.hostId, m2.supervisor.hostId)
await m1.db.createUser({
id: 'u1',
username: 'carol',
passHash: await hashPassword('carolpass123'),
role: 'active',
homeDir: '/tmp/u1-home',
})
mkdirSync(join(dataRoot, 'users', 'u1', 'ws', 'proj'), { recursive: true })
const folder = join(dataRoot, 'users', 'u1', 'ws', 'proj')
// ⑤ 注册 + 心跳:dsh_hosts 里应有记录(启动即注册 + 立即一次心跳)
const hosts = await m1.db.listDshHosts()
assert(hosts.length === 2, `dsh_hosts 应有 2 条(实际 ${hosts.length})`)
assert(hosts.every((h) => h.status === 'up' && h.lastHeartbeat !== null), 'worker 状态 up 且有心跳时间')
console.log('注册/心跳 -> dsh_hosts =', hosts.map((h) => `${h.id}:${h.status}`).join(', '))
// ── 2) m-1 拉起:归属必须落库 ─────────────────────────────────────────
const inst1 = await m1.supervisor.launch('u1', folder)
assert(inst1.userId === 'u1', 'launch 返回实例')
let row = await m1.db.findUserInstance('u1', 'main')
assert(row.hostId === 'm-1', `归属应落库为 m-1(实际 ${row.hostId})`)
assert(row.epoch === 1, `首次抢占 epoch 应为 1(实际 ${row.epoch})`)
assert(row.leaseUntil > Date.now(), 'lease_until 应在未来')
console.log('① 归属落库 -> host_id=%s epoch=%d lease_until=+%dms', row.hostId, row.epoch, row.leaseUntil - Date.now())
// worker 侧真的收到了 epoch=1(`/fence` 同值 ⇒ 不该被 fence)
const f0 = await agentJson('/fence', { method: 'POST', body: { userId: 'u1', epoch: 1 } })
assert(f0.body.fenced === false, 'worker 已记录 epoch=1(同值不 fence)')
console.log(' worker 已记录 epoch=1')
// ── 3) 单写者:m-2 在租约存活期内拉不起来 ────────────────────────────
let busy
try {
await m2.supervisor.launch('u1', folder)
} catch (err) {
busy = err
}
assert(busy !== undefined, 'm-2 必须拉起失败')
assert(busy.name === 'LeaseBusyError', `应是 LeaseBusyError(实际 ${busy.name})`)
assert(busy.holder === 'm-1', `错误里应带持有者 m-1(实际 ${busy.holder})`)
const stillMine = await m1.db.findUserInstance('u1', 'main')
assert(stillMine.hostId === 'm-1' && stillMine.epoch === 1, '失败方不得改动归属')
console.log('② 单写者 -> m-2 抛 LeaseBusyError(holder=%s),归属未被改动', busy.holder)
// ── 4) stop 释放 ⇒ 别人立刻能接(不用等 TTL)──────────────────────────
await m1.supervisor.stop('u1')
row = await m1.db.findUserInstance('u1', 'main')
assert(row.hostId === null, 'stop 后归属应清空')
// 多机语义(S6 起):不显式指定目标机时由 `selectHost` 按容量挑"最优的那台",
// 不一定是 m-2 ⇒ 这一步要验的是"释放后可被接管",所以**显式指定 m-2**(确定性)。
const inst2 = await m2.supervisor.launch('u1', folder, undefined, { hostId: 'm-2' })
assert(inst2.userId === 'u1', 'm-2 拉起成功')
row = await m2.db.findUserInstance('u1', 'main')
assert(row.hostId === 'm-2', `归属应转给 m-2(实际 ${row.hostId})`)
assert(row.epoch === 2, `epoch 必须递增到 2(实际 ${row.epoch})`)
console.log('③ 释放即接手 -> host_id=m-2 epoch=%d(epoch 单调递增)', row.epoch)
// ── 5) 失权即 self-fence ─────────────────────────────────────────────
// 让 m-2 也"死掉"(停心跳)→ 等待 TTL 过期 → m-1 抢占(epoch=3)
m2.supervisor.stopHeartbeat()
await sleep(1400)
const stolen = await m1.db.claimInstance('u1', 'm-1', 60_000)
assert(stolen.ok === true && stolen.epoch === 3, `m-1 过期后应能抢到 epoch=3(实际 ${JSON.stringify(stolen)})`)
console.log(' m-1 在租约过期后抢回(epoch=3)')
// m-2 的下一跳心跳发现自己失权 ⇒ 给 worker 下发更高 epoch ⇒ worker 停掉自己那个实例
const before = await agentJson('/instances')
await m2.supervisor.tick()
await sleep(200)
const after = await agentJson('/instances')
assert(after.body.instances.length === 0, `失权方心跳后实例应被停(前 ${before.body.instances.length} → 后 ${after.body.instances.length})`)
console.log('④ 失权即 fence -> worker 实例数 %d → %d(self-fencing 生效)', before.body.instances.length, after.body.instances.length)
// 归属仍在 m-1 名下(fence 不会误清他人的归属)
row = await m1.db.findUserInstance('u1', 'main')
assert(row.hostId === 'm-1', 'fence 不该清掉持有者的归属')
console.log(' 归属仍在 m-1 名下(未被误清)')
console.log('\nOK: 归属租约(1a 形态)端到端通过')
console.log(' ✓ 归属落库 ✓ 单写者(LeaseBusyError) ✓ 释放即接手 ✓ 失权即 self-fence ✓ 注册+心跳')
} finally {
for (const app of managers) await app?.close()
await agentHandle?.stop()
await sleep(500)
try {
rmSync(dataRoot, { recursive: true, force: true })
} catch {
/* best-effort */
}
}
+365
View File
@@ -0,0 +1,365 @@
/**
* T08 · **真实部署**端到端功能确认(不是单进程内测试)。
*
* 与 `verify-cluster-*.mjs` 的区别(那些是**组件级**验证,两个 Fastify 跑在同一进程里):
* 这里**真的起进程** —— 2 个 `dshs worker` agent + 1 个 Manager 都是独立进程,
* 经**真 HTTP**(127.0.0.1 端口)与**真 PG** 通信,实例是**真 `dsh` 子进程**(account 隔离)。
* 它回答的是最后一个问题:**这套东西按部署形态装起来,到底能不能用。**
*
* 检查链路(一条真实的用户路径):
* ① bootstrap-admin → ③ register → approve(平台现有流程)
* ④ 登录 → ⑤ 建文件夹(经 RemoteUserFs 落到 worker)→ ⑥ launch(经 agent 起真 dsh)
* ⑦ 轮询 status → ⑧ **取实例页面(经代理)**← 功能确认的关键一步
* ⑨ stop → ⑩ 起第二台 worker → 注册 → **迁移** → 再取一次页面
* ⑪ `dshs doctor` / `dshs cluster status`(观测面)
*
* 需要:106 上 PG 已在 127.0.0.1:15432;以 root 运行(account 隔离要 setpriv/systemd-run)。
* 运行:CLUSTER_LIVE_PG_URL=postgres://dshs:[email protected]:15432/dshs_live node scripts/verify-cluster-live.mjs
*/
import { spawn, spawnSync } from 'node:child_process'
import { createWriteStream, mkdirSync, readFileSync, rmSync } from 'node:fs'
import { dirname, join } from 'node:path'
import { fileURLToPath } from 'node:url'
function assert(condition, message) {
if (!condition) throw new Error('ASSERT: ' + message)
}
const sleep = (ms) => new Promise((r) => setTimeout(r, ms))
const PG_URL = process.env.CLUSTER_LIVE_PG_URL
if (PG_URL === undefined || PG_URL === '') {
console.error('需要 CLUSTER_LIVE_PG_URL')
process.exit(2)
}
const here = dirname(fileURLToPath(import.meta.url))
const repoRoot = join(here, '..')
const CLI = join(repoRoot, 'lib', 'cli.js')
const TOKEN = 'live-cluster-agent-token'
const DATA_ROOT = process.env.CLUSTER_LIVE_DATA_ROOT ?? '/opt/dshs-cluster/live-data'
const ISO = process.env.CLUSTER_LIVE_ISOLATION ?? 'account'
const M_PORT = Number(process.env.CLUSTER_LIVE_MANAGER_PORT ?? 13080)
const A1_PORT = Number(process.env.CLUSTER_LIVE_AGENT1_PORT ?? 19000)
const A2_PORT = Number(process.env.CLUSTER_LIVE_AGENT2_PORT ?? 19001)
const ADMIN_PW = 'liveadmin123'
const USER_PW = 'liveuser123'
const procs = []
/** 起一个子进程并记下来(收尾统一 SIGTERM ⇒ agent 会先 teardown 实例再退出)。 */
function run(label, args, env) {
const child = spawn(process.execPath, args, {
cwd: repoRoot,
env: { ...process.env, ...env },
stdio: ['ignore', 'pipe', 'pipe'],
})
// ⚠️ **必须留日志**:不留就只能看到"实例没了"而看不到为什么(2026-09-15 实测踩到)
const logPath = `/tmp/live-${label}.log`
const stream = createWriteStream(logPath, { flags: 'w' })
child.stdout.pipe(stream)
child.stderr.pipe(stream)
child.on('exit', (code) => {
if (code !== null && code !== 0 && !stopping) console.error(`[${label}] 提前退出 code=${code}(日志 ${logPath})`)
})
procs.push({ label, child, logPath })
return child
}
/** 打印某个子进程日志的尾部(诊断用)。 */
function tailLog(label, lines = 12) {
const found = procs.find((p) => p.label === label)
if (found === undefined) return
try {
const text = readFileSync(found.logPath, 'utf8').trimEnd().split('\n')
console.error(` ── ${label} 日志尾部 ──`)
for (const line of text.slice(-lines)) console.error(' ', line.slice(0, 200))
} catch {
/* 没日志就算了 */
}
}
let stopping = false
function shutdownAll() {
stopping = true
for (const { child } of procs) {
try {
child.kill('SIGTERM')
} catch {
/* 已退出 */
}
}
}
/** 轮询等一个 URL 可用。 */
async function waitHttp(url, timeoutMs, what) {
const deadline = Date.now() + timeoutMs
let last = ''
while (Date.now() < deadline) {
try {
const res = await fetch(url, { signal: AbortSignal.timeout(2000) })
if (res.status < 500) return res.status
last = `HTTP ${res.status}`
} catch (err) {
last = err instanceof Error ? err.message : String(err)
}
await sleep(300)
}
throw new Error(`等待 ${what} 超时(${url}):${last}`)
}
const base = `http://127.0.0.1:${M_PORT}`
const json = async (path, { method = 'GET', body, cookie } = {}) => {
const res = await fetch(base + path, {
method,
headers: { ...(body ? { 'content-type': 'application/json' } : {}), ...(cookie ? { cookie } : {}) },
body: body ? JSON.stringify(body) : undefined,
signal: AbortSignal.timeout(30_000),
})
const text = await res.text()
let parsed = null
try {
parsed = text === '' ? null : JSON.parse(text)
} catch {
parsed = { raw: text.slice(0, 200) }
}
return { status: res.status, body: parsed, setCookie: res.headers.get('set-cookie') }
}
try {
console.log('=== 真实部署检查:dataRoot=%s 隔离=%s ===', DATA_ROOT, ISO)
rmSync(DATA_ROOT, { recursive: true, force: true })
mkdirSync(DATA_ROOT, { recursive: true })
// 干净的 PG 库(本检查专用)
const psql = (sql) =>
spawnSync('su', ['-', 'postgres', '-c', `/usr/bin/psql -p 15432 -q -c "${sql}"`], { encoding: 'utf8' })
psql('DROP DATABASE IF EXISTS dshs_live')
psql('CREATE DATABASE dshs_live OWNER dshs')
console.log('PG -> dshs_live 已重建')
// ── ① worker agent(独立进程)─────────────────────────────────────────
const agentEnv = { DSHS_DATA_ROOT: DATA_ROOT, DSHS_ISOLATION_MODE: ISO }
run('agent-w-1', [CLI, 'worker', '--token', TOKEN, '--port', String(A1_PORT), '--host', '127.0.0.1', '--host-id', 'w-1', '--instance-host', '127.0.0.1', '--log-level', 'warn'], agentEnv)
await waitHttp(`http://127.0.0.1:${A1_PORT}/healthz`, 15_000, 'agent w-1')
console.log('worker w-1 -> http://127.0.0.1:%d 就绪', A1_PORT)
// ── ② Manager(独立进程,cluster 模式)────────────────────────────────
const managerEnv = {
DSHS_DEPLOY_MODE: 'cluster',
DSHS_DB_URL: PG_URL,
DSHS_DATA_ROOT: DATA_ROOT,
DSHS_CLUSTER_HOST_ID: 'm-1',
DSHS_CLUSTER_AGENT_URL: `http://127.0.0.1:${A1_PORT}`,
DSHS_CLUSTER_AGENT_TOKEN: TOKEN,
DSHS_CLUSTER_INSTANCE_HOST: '127.0.0.1',
DSHS_CLUSTER_WORKER_DATA_ROOT: DATA_ROOT,
DSHS_CLUSTER_CAPACITY_MB: '-1', // Manager 自己**不承载实例**
DSHS_CLUSTER_REGISTER_SELF: '0', // 专用 Manager ⇒ **不自注册**(一个 agent 只应有一条 host 记录)
DSHS_CLUSTER_LEASE_TTL_MS: '30000',
}
run('manager', [CLI, '--port', String(M_PORT), '--host', '127.0.0.1', '--log-level', 'warn'], managerEnv)
await waitHttp(`${base}/login.html`, 20_000, 'Manager')
console.log('manager -> %s 就绪(deployMode=cluster)', base)
// ── ③ bootstrap-admin(首次建管理员)──────────────────────────────────
const boot = spawnSync(process.execPath, [CLI, 'bootstrap-admin', '--username', 'root', '--password', ADMIN_PW], {
cwd: repoRoot,
// 用 local 模式初始化管理员 root:它就是这台机上的目录,与 worker 用同一个 DATA_ROOT
env: { ...process.env, DSHS_DATA_ROOT: DATA_ROOT, DSHS_DB_URL: PG_URL },
encoding: 'utf8',
})
assert(boot.status === 0, `bootstrap-admin 失败:${boot.stderr?.slice(0, 300)}`)
console.log('管理员 -> root 已创建(bootstrap-admin)')
// ── ④ 真实用户流程:注册 → 审批 → 登录 ────────────────────────────────
let adm = await json('/api/auth/login', { method: 'POST', body: { username: 'root', password: ADMIN_PW } })
assert(adm.status === 200, `管理员登录失败:${adm.status}`)
const adminCookie = adm.setCookie.split(';')[0]
let r = await json('/api/auth/register', { method: 'POST', body: { username: 'liveuser', password: USER_PW } })
assert(r.status === 201, `注册应 201(实际 ${r.status} ${JSON.stringify(r.body)})`)
const users = await json('/api/admin/users', { cookie: adminCookie })
const target = users.body.users.find((u) => u.username === 'liveuser')
assert(target !== undefined, '管理员能列出待审用户')
r = await json(`/api/admin/users/${target.id}/approve`, { method: 'POST', cookie: adminCookie })
assert(r.status === 200, `审批应 200(实际 ${r.status})`)
const login = await json('/api/auth/login', { method: 'POST', body: { username: 'liveuser', password: USER_PW } })
assert(login.status === 200, `用户登录失败:${login.status}`)
const cookie = login.setCookie.split(';')[0]
console.log('① 用户流程 -> 注册 → 审批 → 登录 全部通过(uid=%s)', target.id)
// 显式注册 w-1(= join 脚本那一步:agent 已在跑,调管理面登记)
r = await json('/api/admin/hosts', {
method: 'POST',
cookie: adminCookie,
body: { id: 'w-1', endpoint: `http://127.0.0.1:${A1_PORT}`, token: TOKEN, capacityMb: 4096 },
})
assert(r.status === 200, `注册 w-1 应 200(实际 ${r.status})`)
// ── ⑤ 建文件夹(经 RemoteUserFs 落到 worker)─────────────────────────
r = await json('/api/fs/mkdir', { method: 'POST', cookie, body: { path: 'proj' } })
assert(r.status === 200, `mkdir 应 200(实际 ${r.status} ${JSON.stringify(r.body)})`)
assert(
(await json('/api/desktop/tree', { cookie })).body.entries.some((e) => e.name === 'proj'),
'工作区里出现 proj',
)
console.log('② 文件面 -> mkdir 落库到 worker(经 agent /fs/mkdir)')
// ── ⑥ 拉起实例(经 agent 起**真 dsh**)───────────────────────────────
r = await json('/api/dsh/launch', { method: 'POST', cookie, body: { folder: 'proj' } })
assert(r.status === 200, `launch 应 200(实际 ${r.status} ${JSON.stringify(r.body)})`)
assert(typeof r.body.url === 'string' && r.body.url.startsWith('/u/'), `launch 返回子路径形态 URL(实际 ${r.body.url})`)
console.log('③ 拉起 -> %s(真 dsh 启动中,token 稍后才吐)', r.body.url)
// 直接问 worker(绕过 Manager):实例到底在不在 agent 手上
const onAgent = await fetch(`http://127.0.0.1:${A1_PORT}/instances`, { headers: { 'x-dsh-agent-token': TOKEN } })
console.log(' worker 视角 -> /instances = %s', (await onAgent.text()).slice(0, 200))
// ── ⑦ 轮询到「running **且** launch token 到位」 ─────────────────────
// 真 dsh 与 fake-dsh 不同:进程起来 ≠ 已打印 token。**launch token 回传(P0-6)**
// 只有在 token 真的从 worker 传回 Manager 之后才算成立,所以这里要等到它。
const t0 = Date.now()
let running = false
let tokenSeen = ''
let lastStatus = null
for (let i = 0; i < 90; i += 1) {
const st = await json('/api/dsh/status', { cookie })
lastStatus = st.body
const main = st.body.main
if (main?.status === 'crashed') {
console.error(' 实例崩溃:exitCode=%s lastError=%s', main.exitCode, String(main.lastError).slice(0, 400))
break
}
// 注意:`/api/dsh/status` 的实例视图是**精简视图**(id/port/status/restarts),
// **不含 launchToken** ⇒ token 的存在性用下面的 `/api/dsh/enter` 判定(它回带 token 的 URL)。
if (st.body.running === true) {
running = true
tokenSeen = st.body.url ?? ''
break
}
await sleep(1000)
}
if (!running) {
tailLog('agent-w-1', 20)
tailLog('manager', 10)
}
assert(running, `实例应在 90s 内 running(实际 ${JSON.stringify(lastStatus)?.slice(0, 500)})`)
console.log('④ 状态 -> running=true(真 dsh,隔离=%s,耗时 %ds)', ISO, Math.round((Date.now() - t0) / 1000))
// ── ⑧ **登录直达**(P0-6 的真实端到端):enter 走"复用已运行实例"分支 ⇒ 带 token 的 URL
const enter = await json('/api/dsh/enter', { method: 'POST', cookie })
assert(enter.status === 200, `enter 应 200(实际 ${enter.status} ${JSON.stringify(enter.body)})`)
const launchUrl = enter.body.url
assert(typeof launchUrl === 'string' && launchUrl.includes('token='), `enter 应返回**带 token** 的直达 URL(实际 ${launchUrl})`)
console.log('⑤ 登录直达 -> %s', launchUrl)
let pageStatus = 0
let pageSnippet = ''
for (let i = 0; i < 40; i += 1) {
// 真实浏览器会**跟随重定向**(dsh 首页 303 → 应用页)⇒ 这里也跟随,否则会误判为失败。
// ⚠️ 必须 try/catch:实例刚 spawn 时正在初始化,代理可能中途断连(`other side closed`),
// 这是**启动窗口的正常现象**,重试即可 —— 不捕获会让检查在第一次尝试就失败。
try {
const res = await fetch(base + launchUrl, { headers: { cookie }, redirect: 'follow', signal: AbortSignal.timeout(15_000) })
pageStatus = res.status
if (res.status === 200) {
pageSnippet = (await res.text()).slice(0, 400)
break
}
} catch {
pageStatus = 0
}
await sleep(1000)
}
assert(pageStatus === 200, `实例页面应最终 200(实际 ${pageStatus})`)
console.log('⑥ 实例页面 -> 200(经 Manager 代理到 worker 上 account 沙箱内的真 dsh;已跟随 303 重定向)')
// ── ⑨ 停止 ────────────────────────────────────────────────────────────
r = await json('/api/dsh/stop', { method: 'POST', cookie })
assert(r.status === 200, `stop 应 200(实际 ${r.status})`)
await sleep(500)
assert((await json('/api/dsh/status', { cookie })).body.running === false, 'stop 后 running=false')
console.log('⑦ 停止 -> ok')
// ── ⑩ 第二台 worker + 迁移 ────────────────────────────────────────────
run('agent-w-2', [CLI, 'worker', '--token', TOKEN, '--port', String(A2_PORT), '--host', '127.0.0.1', '--host-id', 'w-2', '--instance-host', '127.0.0.1', '--log-level', 'warn'], agentEnv)
await waitHttp(`http://127.0.0.1:${A2_PORT}/healthz`, 15_000, 'agent w-2')
r = await json('/api/admin/hosts', {
method: 'POST',
cookie: adminCookie,
body: { id: 'w-2', endpoint: `http://127.0.0.1:${A2_PORT}`, token: TOKEN, capacityMb: 4096 },
})
assert(r.status === 200, `注册 w-2 应 200(实际 ${r.status})`)
const hosts = await json('/api/admin/hosts', { cookie: adminCookie })
assert(hosts.body.hosts.some((h) => h.id === 'w-2'), 'w-2 出现在 worker 目录')
assert(!('agentToken' in (hosts.body.hosts[0] ?? {})), '**绝不下发 agentToken**')
console.log('⑧ 第二台 -> w-2 已注册(且列表不含 agentToken)')
// 重新拉起(此刻只有 w-1 是候选 ⇒ 确定性落 w-1),再注册 w-2、再迁移
r = await json('/api/dsh/launch', { method: 'POST', cookie, body: { folder: 'proj' } })
assert(r.status === 200, `再次 launch 应 200(实际 ${r.status} ${JSON.stringify(r.body)})`)
for (let i = 0; i < 40 && !(await json('/api/dsh/status', { cookie })).body.running; i += 1) await sleep(1000)
const owner = (await json('/api/dsh/status', { cookie })).body
assert(owner.running === true, '重新拉起后 running=true')
r = await json(`/api/admin/users/${target.id}/dsh/migrate`, {
method: 'POST',
cookie: adminCookie,
body: { targetHost: 'w-2' },
})
assert(r.status === 200, `迁移应 200(实际 ${r.status} ${JSON.stringify(r.body)})`)
assert(r.body.to === 'w-2', `迁移目标应为 w-2(实际 ${r.body.to})`)
for (let i = 0; i < 30 && !(await json('/api/dsh/status', { cookie })).body.running; i += 1) await sleep(1000)
console.log('⑨ 迁移 -> %s → %s(epoch=%d)', r.body.from, r.body.to, r.body.epoch)
// ⚠️ 迁移后实例是**新进程 ⇒ 新 launch token**:旧 URL 里的 token 已失效(404 是**预期**行为)。
// 真实用户会重新走 `/api/dsh/enter`(门户的"进入工作区"就是这个接口)拿**新** URL ⇒ 这里照做。
let newUrl = ''
pageStatus = 0
for (let i = 0; i < 40; i += 1) {
try {
const en = await json('/api/dsh/enter', { method: 'POST', cookie })
if (en.status === 200 && typeof en.body.url === 'string') {
newUrl = en.body.url
const res = await fetch(base + newUrl, { headers: { cookie }, redirect: 'follow', signal: AbortSignal.timeout(15_000) })
pageStatus = res.status
if (res.status === 200) {
pageSnippet = (await res.text()).slice(0, 200)
break
}
}
} catch {
pageStatus = 0
}
await sleep(1000)
}
if (newUrl === '' || newUrl === launchUrl) {
// 诊断:两台 agent 各自认为有什么 + DB 归属如何
for (const [label, port] of [['w-1', A1_PORT], ['w-2', A2_PORT]]) {
const res = await fetch(`http://127.0.0.1:${port}/instances`, { headers: { 'x-dsh-agent-token': TOKEN } })
const body = await res.text()
console.error(` ${label} /instances = ${body.slice(0, 220)}`)
const st = await fetch(`http://127.0.0.1:${port}/status/${target.id}`, { headers: { 'x-dsh-agent-token': TOKEN } })
console.error(` ${label} /status = ${(await st.text()).slice(0, 220)}`)
}
tailLog('agent-w-2', 16)
tailLog('manager', 8)
}
assert(newUrl !== '' && newUrl !== launchUrl, `迁移后 enter 应给**新** URL(旧 ${launchUrl} / 新 ${newUrl})`)
assert(pageStatus === 200, `迁移后经新 URL 的实例页面应 200(实际 ${pageStatus})`)
console.log('⑩ 迁移后 -> enter 返回新 token URL,页面 200(经 w-2)')
// ── ⑪ 观测面 ──────────────────────────────────────────────────────────
const doctor = spawnSync(process.execPath, [CLI, 'doctor'], { cwd: repoRoot, env: { ...process.env, ...managerEnv }, encoding: 'utf8' })
const status = spawnSync(process.execPath, [CLI, 'cluster', 'status'], { cwd: repoRoot, env: { ...process.env, ...managerEnv }, encoding: 'utf8' })
console.log('⑪ dshs doctor -> rc=%d(0 = 无硬失败)', doctor.status ?? -1)
console.log(String(status.stdout).split('\n').slice(0, 8).map((l) => ' ' + l).join('\n'))
// 收尾:停实例(避免留下 dsh 子进程)
await json('/api/dsh/stop', { method: 'POST', cookie })
console.log('\nOK: 真实部署(2 个 worker agent 进程 + 1 个 Manager 进程 + 真 PG)端到端功能确认通过')
console.log(' ✓ 用户流程 ✓ 文件面跨机 ✓ 真 dsh 拉起并可从公网侧取页面 ✓ 停止 ✓ 注册+迁移+迁移后复验 ✓ 观测面')
console.log(' 页面片段:%s', pageSnippet.replace(/\s+/g, ' ').slice(0, 80))
} finally {
shutdownAll()
await sleep(1500)
}
+246
View File
@@ -0,0 +1,246 @@
/**
* T08 S6 · 多 worker + 容量准入 + **计划内迁移**验证。
*
* 这一条是整套设计的落点:**实例可迁移**。它同时验证:
* ① **容量准入**:`selectHost` 把"已用 + 预留 > 容量"的机排除掉 ⇒ 实例落到还有余量的那台;
* ② **归属与实例一致**:`dsh_instances.host_id` 指向实例真正所在的那台;
* ③ **迁移三步**(drain → 目标机拉起 → 归属原子更新):`host_id` 换台、`epoch` 单调 +1;
* ④ **迁移后代理照常**:`endpointFor` 按新归属路由,页面仍 200;
* ⑤ **数据不搬家也能用**:两台 worker **共享同一 dataRoot**(模拟共享存储 / 同路径基线)。
*
* 需要 PG(归属在 DB 里,两个 Manager/worker 共享):
* CLUSTER_TEST_PG_URL=postgres://dshs:[email protected]:15432/dshs_smoke node scripts/verify-cluster-migrate.mjs
*/
import { existsSync, mkdirSync, mkdtempSync, rmSync } from 'node:fs'
import { tmpdir } from 'node:os'
import { dirname, join } from 'node:path'
import { fileURLToPath } from 'node:url'
import pg from 'pg'
import { buildServer } from '../lib/web/server.js'
import { buildWorkerAgent, AGENT_TOKEN_HEADER } from '../lib/worker/agent.js'
import { resolveConfig } from '../lib/config.js'
import { hashPassword } from '../lib/web/auth.js'
function assert(condition, message) {
if (!condition) throw new Error('ASSERT: ' + message)
}
const sleep = (ms) => new Promise((r) => setTimeout(r, ms))
const PG_URL = process.env.CLUSTER_TEST_PG_URL
if (PG_URL === undefined || PG_URL === '') {
console.error('需要 CLUSTER_TEST_PG_URL')
process.exit(2)
}
const TOKEN = 'verify-cluster-migrate-token'
const here = dirname(fileURLToPath(import.meta.url))
const fakeDsh = join(here, 'fake-dsh.mjs')
/** 两台 worker **共享同一 dataRoot** = 模拟共享存储 / "所有 worker 同路径"的基线约定。 */
const sharedRoot = mkdtempSync(join(tmpdir(), 'dsh-migrate-'))
// Manager 自己不承载实例(capacity=-1);两台 worker 声明 4096MB
process.env.DSHS_CLUSTER_CAPACITY_MB = '-1'
let app
const agents = []
async function resetPg() {
const client = new pg.Client({ connectionString: PG_URL })
await client.connect()
await client.query('DELETE FROM dsh_instances')
await client.query('DELETE FROM dsh_hosts')
await client.query('DELETE FROM users')
await client.end()
}
async function startAgent(hostId) {
const config = resolveConfig({
port: 0,
dbPath: ':memory:',
dataRoot: sharedRoot,
dshCommand: [process.execPath, fakeDsh],
clusterHostId: hostId,
})
const agent = buildWorkerAgent(config, {
hostId,
token: TOKEN,
port: 0,
host: '127.0.0.1',
instanceHost: '127.0.0.1',
logLevel: 'warn',
})
await agent.app.listen({ host: '127.0.0.1', port: 0 })
agents.push(agent) // 整个 handle:收尾要用 stop() 收实例
const url = `http://127.0.0.1:${agent.app.server.address().port}`
const call = async (path, { method = 'GET', body } = {}) => {
const res = await fetch(url + path, {
method,
headers: { [AGENT_TOKEN_HEADER]: TOKEN, ...(body ? { 'content-type': 'application/json' } : {}) },
body: body ? JSON.stringify(body) : undefined,
})
const text = await res.text()
return { status: res.status, body: text ? JSON.parse(text) : null }
}
/** 该机上的实例数(对账口径)。 */
const instanceCount = async () => (await call('/instances')).body.instances.length
return { hostId, url, call, instanceCount }
}
try {
await resetPg()
// ── 0) 两台 worker ────────────────────────────────────────────────────
const a = await startAgent('w-a')
const b = await startAgent('w-b')
console.log('worker -> w-a %s / w-b %s(共享 dataRoot)', a.url, b.url)
// ── 1) Manager(默认 agent 指 w-a;自己 capacity=-1 不承载)─────────────
app = await buildServer(
resolveConfig({
port: 0,
dbUrl: PG_URL,
dataRoot: sharedRoot,
deployMode: 'cluster',
clusterAgentUrl: a.url,
clusterAgentToken: TOKEN,
clusterInstanceHost: '127.0.0.1',
clusterHostId: 'm-1',
clusterWorkerDataRoot: sharedRoot,
}),
)
await app.listen({ port: 0 })
const base = `http://127.0.0.1:${app.server.address().port}`
// 注册两台 worker(join 脚本走的就是这个 API)
await app.db.upsertDshHost({ id: 'w-a', endpoint: a.url, agentToken: TOKEN, capacityMb: 4096 })
await app.db.upsertDshHost({ id: 'w-b', endpoint: b.url, agentToken: TOKEN, capacityMb: 4096 })
// 把 w-b 的已用水位抬高到"再来一个实例就超" ⇒ 用来验证**准入拒绝**
await app.db.setDshHostStatus('w-b', 'up', 3800, Date.now())
// admin 账号(迁移 API 需要)
await app.db.createUser({
id: 'admin-1',
username: 'root',
passHash: await hashPassword('rootpass123'),
role: 'admin',
homeDir: '/tmp/admin-home',
})
await app.db.createUser({
id: 'u1',
username: 'carol',
passHash: await hashPassword('carolpass123'),
role: 'active',
homeDir: '/tmp/u1-home',
})
await app.userFs.initUserRoot('u1')
// 门户流程里 folder 是用户从「我的文件」里挑的**已存在**目录 ⇒ 这里先建出来
mkdirSync(join(sharedRoot, 'users', 'u1', 'ws', 'proj'), { recursive: true })
const json = async (path, { method = 'GET', body, cookie } = {}) => {
const res = await fetch(base + path, {
method,
headers: { ...(body ? { 'content-type': 'application/json' } : {}), ...(cookie ? { cookie } : {}) },
body: body ? JSON.stringify(body) : undefined,
})
const text = await res.text()
return { status: res.status, body: text ? JSON.parse(text) : null, setCookie: res.headers.get('set-cookie') }
}
// ── 2) 容量准入:w-b 水位高 ⇒ 必须落到 w-a ────────────────────────────
const c = await json('/api/auth/login', { method: 'POST', body: { username: 'carol', password: 'carolpass123' } })
assert(c.status === 200, 'user login')
const userCookie = c.setCookie.split(';')[0]
let r = await json('/api/dsh/launch', { method: 'POST', cookie: userCookie, body: { folder: 'proj' } })
assert(r.status === 200, `launch 经远端成功(实际 ${r.status} ${JSON.stringify(r.body)})`)
let row = await app.db.findUserInstance('u1', 'main')
assert(row.hostId === 'w-a', `容量准入应选 w-a(w-b 已 3800+512>4096);实际 ${row.hostId}`)
assert(row.epoch === 1, `首次抢占 epoch=1(实际 ${row.epoch})`)
assert((await a.instanceCount()) === 1, 'w-a 上有 1 个实例')
assert((await b.instanceCount()) === 0, 'w-b 上 0 个实例')
console.log('① 容量准入 -> 落到 w-a(w-b 因水位被排除),host_id=w-a epoch=1')
// 代理照常
let proxyText
for (let i = 0; i < 20; i += 1) {
const res = await fetch(`${base}/u/u1/dsh/hello`, { headers: { cookie: userCookie } })
if (res.status === 200) {
proxyText = await res.text()
break
}
await sleep(100)
}
assert(proxyText !== undefined && proxyText.includes('fake-dsh'), '迁移前代理 200(经 w-a)')
console.log(' 迁移前代理 -> 200(经 w-a)')
// 顺便在用户工作区放个文件(迁移后要还在 —— 共享存储场景)
await json('/api/fs/upload', {
method: 'POST',
cookie: userCookie,
body: { path: 'proj', name: 'keep.txt', data: Buffer.from('survives migration').toString('base64') },
})
// ── 3) 迁移到 w-b ─────────────────────────────────────────────────────
const adm = await json('/api/auth/login', { method: 'POST', body: { username: 'root', password: 'rootpass123' } })
assert(adm.status === 200, 'admin login')
const adminCookie = adm.setCookie.split(';')[0]
r = await json('/api/admin/users/u1/dsh/migrate', {
method: 'POST',
cookie: adminCookie,
body: { targetHost: 'w-b' },
})
assert(r.status === 200, `迁移成功(实际 ${r.status} ${JSON.stringify(r.body)})`)
assert(r.body.from === 'w-a' && r.body.to === 'w-b', `迁移方向 w-a→w-b(实际 ${JSON.stringify(r.body)})`)
assert(r.body.epoch === 2, `epoch 应 +1 到 2(实际 ${r.body.epoch})`)
row = await app.db.findUserInstance('u1', 'main')
assert(row.hostId === 'w-b' && row.epoch === 2, '归属已原子更新到 w-b / epoch=2')
assert((await a.instanceCount()) === 0, 'w-a 上实例已停(drain 生效)')
assert((await b.instanceCount()) === 1, 'w-b 上有 1 个实例')
console.log('② 迁移 -> w-a → w-b,host_id=w-b epoch=%d,源机实例已停', r.body.epoch)
// 迁移后代理照常(按新归属路由到 w-b)
proxyText = undefined
for (let i = 0; i < 20; i += 1) {
const res = await fetch(`${base}/u/u1/dsh/hello`, { headers: { cookie: userCookie } })
if (res.status === 200) {
proxyText = await res.text()
break
}
await sleep(100)
}
assert(proxyText !== undefined && proxyText.includes('fake-dsh'), '迁移后代理 200(经 w-b)')
// ── 4) 数据还在(共享存储)────────────────────────────────────────────
const keptPath = join(sharedRoot, 'users', 'u1', 'ws', 'proj', 'keep.txt')
assert(existsSync(keptPath), `迁移后文件仍在:${keptPath}`)
const dl = await fetch(`${base}/api/fs/download?path=proj/keep.txt`, { headers: { cookie: userCookie } })
assert(dl.status === 200 && (await dl.text()) === 'survives migration', '迁移后仍能下载到原文')
console.log('③ 迁移后 -> 代理 200(经 w-b)、文件可读(数据不搬家)')
// ── 5) 已在该机 + 目标机不存在 ⇒ 明确报错(不静默)────────────────────
r = await json('/api/admin/users/u1/dsh/migrate', { method: 'POST', cookie: adminCookie, body: { targetHost: 'w-b' } })
assert(r.status === 409 && r.body.error === 'already_there', `重复迁移应 409 already_there(实际 ${r.status})`)
r = await json('/api/admin/users/u1/dsh/migrate', { method: 'POST', cookie: adminCookie, body: { targetHost: 'nope' } })
assert(r.status === 404 && r.body.error === 'unknown_host', `未知目标机应 404(实际 ${r.status})`)
console.log('④ 边界 -> already_there / unknown_host 都明确报错')
console.log('\nOK: 多 worker + 容量准入 + 迁移通过')
console.log(' ✓ 容量准入 ✓ 归属与实例一致 ✓ 迁移(drain→拉起→epoch+1) ✓ 迁移后代理/文件正常')
} finally {
// ⚠️ 收尾必须**停掉还活着的实例**:否则 fake-dsh 子进程会继承 stdout,
// 管道永不关闭 ⇒ ssh / CI 会一直挂在这里(2026-09-15 实测踩到)。
try {
await app?.supervisor?.stop('u1')
} catch {
/* best-effort */
}
await app?.close()
for (const h of agents) await h?.stop()
await sleep(500)
try {
rmSync(sharedRoot, { recursive: true, force: true })
} catch {
/* best-effort */
}
}