Files
dsh_shenxian/scripts/migrate-sqlite-to-pg.mjs
T
admin c70d5d860e feat(cluster): 集群化落地 —— Manager/Worker 拆分 + 归属租约 + 跨机验证(T08)
背景:把平台从「单机单进程」改造成「1 组 Manager + N 台 Worker + 共享归属状态」,
硬约束 = 全程兼容单例模式(deployMode 默认 local;生产切换前 47 一行未动)。

主要改动
1) 数据模型 v7(SQLite 与 PG 两方言同步):新增 dsh_hosts 注册表 +
   dsh_instances.{host_id,epoch,heartbeat_at,lease_until};claimInstance 原子抢占
   (UPDATE … WHERE host_id IS NULL OR lease_until < now)+ pinInstanceHost 钉住归属。
2) 租约与 fencing:src/supervisor/lease.ts(acquire/renew/release + stillHolder 判据 +
   ttl > 2×renew 硬校验);心跳里续租,失权即向 worker 下发更高 epoch(self-fencing)。
   ⚠️ release 只清租约(lease_until),**保留 host_id** —— host_id 是「用户数据在哪台」的锚点。
3) Worker agent(src/worker/agent.ts,子命令 dshs worker):实例生命周期 + 文件面 /fs/*
   + 幂等键(operationId)+ 鉴权(timingSafeEqual);Worker 不写控制面数据
   (apiKey/uid 由 Manager 随 launch 投递,R5 收窄)。
4) 远端 Spawner + LeasedSpawner:按 host 路由(**粘性优先**:有历史归属且那台 up 就留在原地,
   否则按容量选最空的)+ 容量准入 + deployMode=cluster 装配(systemd drop-in,可回滚)。
5) bwrap 修正:**所有挂载点的中间目录统一前置 + 去重 + 由外到内**(「就近创建」会在嵌套前缀下
   遮掉已绑挂载点 ⇒ bwrap: Can't chdir);且**只能用 --tmpfs**,用 --perms 会让 47 的
   bwrap 0.4.0 直接拒启动(沙箱全挂)。
6) 跨机隧道 src/worker/tunnel.ts:SSH ControlMaster + 动态 -R 转发;**自愈由 agent 本地
   20s 定时器驱动**(不能只放 /healthz —— 心跳本身经隧道进来,断了就没人触发它)。
7) 文件面按归属路由(RemoteUserFs):实例与文件必须落在同一台机器,否则实例看不到自己的文件。
8) 观测面:dshs doctor / dshs cluster status。

验证(本次均已实跑)
- test/lease.test.mjs:SQLite 10/10 == PG 10/10
- 组件级端到端 5 个:verify-cluster-{agent,lease,fs,migrate,live}.mjs
- 真跨机(47 Manager / 106 Worker,跨云 + 反向隧道)verify-cluster-cross.mjs 九步全绿
- 域名形态访问 verify-cluster-domain.mjs(<user>.域名 → Manager → 远端实例;越权 403)
- 冒烟 scripts/smoke-*:6/8,失败项与改动前基线完全相同(无回归)
- 生产切换与回滚剧本见 dsh-server-docs/交接单/T08-集群化落地-兼容单例模式.md §16
2026-09-15 18:47:02 +08:00

187 lines
6.9 KiB
JavaScript
Raw Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
#!/usr/bin/env node
/**
* T08 · S1.2:SQLite → Postgres 一次性数据迁移。
*
* 设计要点(都是踩过才会疼的地方):
* 1. **列清单不写死** —— 从 PG 的 information_schema 与 SQLite 的 PRAGMA 取**交集**,
* 这样 schema 演进(v4 的 folder/patch、v6 的 enabled 等)不会让脚本静默少搬字段。
* 2. **PG 表结构不由本脚本建** —— 先 import 平台自己的 `createDbAdapter`(带 dbUrl),
* 让**平台的迁移**在 PG 上建库。这样"迁移脚本"与"平台 schema"永远不会两套。
* 3. **identity 列要 `OVERRIDING SYSTEM VALUE`** —— `users.uid` 与 `audit_log.id` 是
* GENERATED ALWAYS AS IDENTITY;不覆盖就会重排 id,**uid 一变 = 所有用户文件属主失配**。
* 搬完必须 `RESTART WITH` 把序列推到 max+1,否则下一条 INSERT 撞主键。
* 4. **FK 顺序**:先 users,再 workspaces/sessions,最后引用它们的表。
* 5. `--dry-run` 只报行数,不写任何东西。
*
* 用法:
* node scripts/migrate-sqlite-to-pg.mjs --sqlite /var/lib/dshs/dshs.db \
* --pg postgres://dshs:***@127.0.0.1:15432/dshs [--dry-run]
*
* @module dshs/scripts/migrate-sqlite-to-pg
*/
import { existsSync } from 'node:fs'
import Database from 'better-sqlite3'
import pg from 'pg'
import { createDbAdapter } from '../lib/db/index.js'
import { resolveConfig } from '../lib/config.js'
/** FK 依赖顺序(父 → 子)。未列出的表会被追加到末尾并告警。 */
const ORDER = [
'users',
'workspaces',
'sessions',
'folder_plugins',
'dsh_instances',
'domains',
'credential_vault',
'business_plugins',
'audit_log',
]
/** identity 列(必须 OVERRIDING SYSTEM VALUE + 搬完 RESTART)。 */
const IDENTITY = { users: 'uid', audit_log: 'id' }
/**
* ⛔ **绝不搬**的表。
*
* `schema_migrations`:目标端的"已应用版本"标记由**平台的迁移**建立(见 `ensurePgSchema`),
* 从源库搬会把同一批版本号再插一遍 ⇒ `schema_migrations_pkey` 唯一键冲突
* (2026-09-15 实测踩到,事务已整体回滚)。语义上也应如此:**结构版本由平台在目标端决定**。
*/
const SKIP = new Set(['schema_migrations'])
function arg(name, fallback) {
const i = process.argv.indexOf(`--${name}`)
return i >= 0 && process.argv[i + 1] !== undefined ? process.argv[i + 1] : fallback
}
const sqlitePath = arg('sqlite')
const pgUrl = arg('pg')
const dryRun = process.argv.includes('--dry-run')
if (sqlitePath === undefined || pgUrl === undefined) {
console.error('用法: node scripts/migrate-sqlite-to-pg.mjs --sqlite <file> --pg <url> [--dry-run]')
process.exit(2)
}
if (!existsSync(sqlitePath)) {
console.error(`SQLite 文件不存在: ${sqlitePath}`)
process.exit(2)
}
/** 让**平台的迁移**在 PG 上建好结构(不自己写 DDL,避免两套 schema)。 */
async function ensurePgSchema() {
const config = resolveConfig({ dataRoot: '/tmp/migrate-tooling', dbUrl: pgUrl })
const db = await createDbAdapter(config)
await db.close()
}
async function main() {
const sq = new Database(sqlitePath, { readonly: true })
await ensurePgSchema()
const client = new pg.Client({ connectionString: pgUrl })
await client.connect()
const sqTables = sq
.prepare("SELECT name FROM sqlite_master WHERE type='table' AND name NOT LIKE 'sqlite_%'")
.all()
.map((r) => r.name)
const pgTables = (
await client.query("SELECT table_name FROM information_schema.tables WHERE table_schema='public'")
).rows.map((r) => r.table_name)
const common = sqTables.filter((t) => pgTables.includes(t) && !SKIP.has(t))
if (sqTables.some((t) => SKIP.has(t))) {
console.log(`按设计跳过: ${[...SKIP].join(', ')}(目标端的结构版本由平台迁移建立)`)
}
const ordered = [
...ORDER.filter((t) => common.includes(t)),
...common.filter((t) => !ORDER.includes(t)),
]
const extra = common.filter((t) => !ORDER.includes(t))
if (extra.length > 0) console.warn(`⚠️ 未在 ORDER 中声明、按末尾处理的表: ${extra.join(', ')}`)
/** 两端的列交集 —— 只搬双方都有的列。 */
async function sharedCols(table) {
const sqCols = sq.prepare(`PRAGMA table_info(${table})`).all().map((c) => c.name)
const pgCols = (
await client.query(
'SELECT column_name FROM information_schema.columns WHERE table_schema=$1 AND table_name=$2',
['public', table],
)
).rows.map((r) => r.column_name)
return sqCols.filter((c) => pgCols.includes(c))
}
const report = []
await client.query('BEGIN')
try {
for (const table of ordered) {
const cols = await sharedCols(table)
if (cols.length === 0) {
console.warn(`跳过 ${table}: 无公共列`)
continue
}
const rows = sq.prepare(`SELECT ${cols.join(',')} FROM ${table}`).all()
const idCol = IDENTITY[table]
if (!dryRun && rows.length > 0) {
const colList = idCol !== undefined ? [idCol, ...cols.filter((c) => c !== idCol)] : cols
const override = idCol !== undefined ? ' OVERRIDING SYSTEM VALUE' : ''
const chunk = 200
for (let i = 0; i < rows.length; i += chunk) {
const slice = rows.slice(i, i + chunk)
const values = []
const tuples = slice.map((row) => {
const ph = colList.map((c) => {
values.push(row[c] ?? null)
return `$${values.length}`
})
return `(${ph.join(',')})`
})
await client.query(
`INSERT INTO ${table} (${colList.join(',')})${override} VALUES ${tuples.join(',')}`,
values,
)
}
if (idCol !== undefined) {
await client.query(
`SELECT setval(pg_get_serial_sequence('${table}','${idCol}'),
(SELECT COALESCE(MAX(${idCol}),0) FROM ${table}))`,
)
}
}
report.push({ table, rows: rows.length, cols: cols.length })
}
if (dryRun) await client.query('ROLLBACK')
else await client.query('COMMIT')
} catch (err) {
await client.query('ROLLBACK')
throw err
}
console.log(`\n${dryRun ? '【DRY-RUN,已回滚】' : '【已提交】'} SQLite → PG 迁移明细`)
for (const r of report) console.log(` ${r.table.padEnd(18)} ${String(r.rows).padStart(6)} 行 / ${r.cols} 列`)
// 校验:逐表比对行数
let bad = 0
if (!dryRun) {
for (const r of report) {
const pgCount = Number((await client.query(`SELECT COUNT(*) AS c FROM ${r.table}`)).rows[0].c)
const sqCount = Number(sq.prepare(`SELECT COUNT(*) AS c FROM ${r.table}`).get().c)
if (pgCount !== sqCount) {
console.error(` ✗ 行数不符 ${r.table}: PG=${pgCount} SQLite=${sqCount}`)
bad += 1
}
}
console.log(bad === 0 ? '\n✅ 逐表行数一致' : `\n❌ ${bad} 张表行数不符`)
}
await client.end()
sq.close()
process.exit(bad === 0 ? 0 : 1)
}
main().catch((err) => {
console.error(err)
process.exit(1)
})