feat(cluster): 集群化落地 —— Manager/Worker 拆分 + 归属租约 + 跨机验证(T08)

背景:把平台从「单机单进程」改造成「1 组 Manager + N 台 Worker + 共享归属状态」,
硬约束 = 全程兼容单例模式(deployMode 默认 local;生产切换前 47 一行未动)。

主要改动
1) 数据模型 v7(SQLite 与 PG 两方言同步):新增 dsh_hosts 注册表 +
   dsh_instances.{host_id,epoch,heartbeat_at,lease_until};claimInstance 原子抢占
   (UPDATE … WHERE host_id IS NULL OR lease_until < now)+ pinInstanceHost 钉住归属。
2) 租约与 fencing:src/supervisor/lease.ts(acquire/renew/release + stillHolder 判据 +
   ttl > 2×renew 硬校验);心跳里续租,失权即向 worker 下发更高 epoch(self-fencing)。
   ⚠️ release 只清租约(lease_until),**保留 host_id** —— host_id 是「用户数据在哪台」的锚点。
3) Worker agent(src/worker/agent.ts,子命令 dshs worker):实例生命周期 + 文件面 /fs/*
   + 幂等键(operationId)+ 鉴权(timingSafeEqual);Worker 不写控制面数据
   (apiKey/uid 由 Manager 随 launch 投递,R5 收窄)。
4) 远端 Spawner + LeasedSpawner:按 host 路由(**粘性优先**:有历史归属且那台 up 就留在原地,
   否则按容量选最空的)+ 容量准入 + deployMode=cluster 装配(systemd drop-in,可回滚)。
5) bwrap 修正:**所有挂载点的中间目录统一前置 + 去重 + 由外到内**(「就近创建」会在嵌套前缀下
   遮掉已绑挂载点 ⇒ bwrap: Can't chdir);且**只能用 --tmpfs**,用 --perms 会让 47 的
   bwrap 0.4.0 直接拒启动(沙箱全挂)。
6) 跨机隧道 src/worker/tunnel.ts:SSH ControlMaster + 动态 -R 转发;**自愈由 agent 本地
   20s 定时器驱动**(不能只放 /healthz —— 心跳本身经隧道进来,断了就没人触发它)。
7) 文件面按归属路由(RemoteUserFs):实例与文件必须落在同一台机器,否则实例看不到自己的文件。
8) 观测面:dshs doctor / dshs cluster status。

验证(本次均已实跑)
- test/lease.test.mjs:SQLite 10/10 == PG 10/10
- 组件级端到端 5 个:verify-cluster-{agent,lease,fs,migrate,live}.mjs
- 真跨机(47 Manager / 106 Worker,跨云 + 反向隧道)verify-cluster-cross.mjs 九步全绿
- 域名形态访问 verify-cluster-domain.mjs(<user>.域名 → Manager → 远端实例;越权 403)
- 冒烟 scripts/smoke-*:6/8,失败项与改动前基线完全相同(无回归)
- 生产切换与回滚剧本见 dsh-server-docs/交接单/T08-集群化落地-兼容单例模式.md §16
This commit is contained in:
admin committed 2026-09-15 18:47:02 +08:00
1 parent 68c0a320ed
commit c70d5d860e
47 files changed
+5367 -14

No files matched your search

+39
View File
@@ -7,12 +7,15 @@
import type {
BusinessPlugin,
ClaimResult,
CredentialKey,
CredentialKeyMeta,
CredentialLandingRow,
CreateSessionInput,
CreateUserInput,
Domain,
DshHost,
DshHostStatus,
DshInstance,
DshInstanceRole,
DshInstanceStatus,
@@ -20,6 +23,7 @@ import type {
SessionRow,
SessionUser,
UpsertBusinessPluginInput,
UpsertDshHostInput,
UpsertDshInstanceInput,
User,
UserRole,
@@ -104,6 +108,41 @@ export interface DbAdapter {
): Promise<boolean>
deleteInstance(id: string): Promise<boolean>
deleteUserInstances(userId: string): Promise<void>
// ── 集群化:worker 注册表 + 实例归属/租约(v7;T08 S2 / 设计 §3.1–§3.2)──
// ⚠️ local 模式**不写**这些表(`LocalSpawner` 靠进程内 Map + 单机互斥),
// 所以这些方法在单机路径上恒为"空/未认领",不影响现有行为。
/** 注册/更新一台 worker(join 幂等:同 id 重复执行 = 更新)。 */
upsertDshHost(input: UpsertDshHostInput): Promise<DshHost>
findDshHost(id: string): Promise<DshHost | undefined>
listDshHosts(): Promise<DshHost[]>
/** 心跳/状态上报:可只改状态,或同时带上容量水位与心跳时间。 */
setDshHostStatus(
id: string,
status: DshHostStatus,
usedMb?: number,
heartbeatAt?: number,
): Promise<boolean>
/**
* **原子抢占**某用户的 main 实例归属(承重墙,见设计 §3.2)。
* 仅当"无人持有 **或** 租约已过期"才成功;成功时 `epoch` +1(fencing)。
* 返回 `ok:false` = 有人在管 ⇒ 调用方**退让**(不是接管)。
*/
claimInstance(
userId: string,
hostId: string,
ttlMs: number,
meta?: { folder?: string; patch?: string },
): Promise<ClaimResult>
/** 续租。**必须带 epoch**:不匹配说明已被他人抢占 ⇒ 本次续租失败(fencing)。 */
renewInstanceLease(userId: string, hostId: string, epoch: number, ttlMs: number): Promise<boolean>
/** 主动释放(停实例时)。同样带 epoch 校验,避免误清他人的归属。 */
releaseInstanceLease(userId: string, hostId: string, epoch: number): Promise<boolean>
/** 钉住归属(首次触达工作区时用):只写 `host_id`,不动 epoch/租约。 */
pinInstanceHost(userId: string, hostId: string): Promise<void>
/** 租约已过期、但仍标着归属的实例 —— 供巡检/自愈(**不代表可以立即接管**,见 R9)。 */
listExpiredInstanceLeases(now: number): Promise<DshInstance[]>
/** 某 worker 上的全部实例 —— 对账用**一次拿回整机**(替代逐用户查询)。 */
listInstancesByHost(hostId: string): Promise<DshInstance[]>
// lifecycle
close(): Promise<void>
}
+156 -1
View File
@@ -11,20 +11,25 @@ import type { DbAdapter } from './adapter.js'
import { mapPgError } from './errors.js'
import { runPgMigrations } from './schema.js'
import {
clusterInstanceId,
toBusinessPlugin,
toDomain,
toDshHost,
toDshInstance,
toPublicUser,
toSession,
toUser,
toWorkspace,
type BusinessPlugin,
type ClaimResult,
type CredentialKey,
type CredentialKeyMeta,
type CredentialLandingRow,
type CreateSessionInput,
type CreateUserInput,
type Domain,
type DshHost,
type DshHostStatus,
type DshInstance,
type DshInstanceRole,
type DshInstanceStatus,
@@ -32,6 +37,7 @@ import {
type SessionRow,
type SessionUser,
type UpsertBusinessPluginInput,
type UpsertDshHostInput,
type UpsertDshInstanceInput,
type User,
type UserRole,
@@ -47,8 +53,10 @@ types.setTypeParser(20, (value: string) => Number(value))
const USER_COLS = 'id, username, pass_hash, role, home_dir, api_key_ref, created_at, approved_by, uid'
const DOMAIN_COLS = 'id, user_id, domain, verified, nginx_config, updated_at'
const BUSINESS_PLUGIN_COLS = 'id, name, description, version, tgz_path, file_size, uploaded_by, created_at, updated_at'
const HOST_COLS = 'id, endpoint, agent_token, capacity_mb, used_mb, status, last_heartbeat'
const INSTANCE_COLS =
'id, user_id, workspace_id, role, pid, port, status, started_at, last_exit, exit_code, last_error, folder, patch'
'id, user_id, workspace_id, role, pid, port, status, started_at, last_exit, exit_code, last_error, folder, patch, '
+ 'host_id, epoch, heartbeat_at, lease_until' // v7 集群化归属/租约(T08 S2)—— 漏了它们会让 hostId 恒为 null
/** Run `fn` on a dedicated client inside a BEGIN/COMMIT/ROLLBACK transaction. */
export async function withTx<T>(pool: Pool, fn: (client: PoolClient) => Promise<T>): Promise<T> {
@@ -587,6 +595,153 @@ export class PgAdapter implements DbAdapter {
await this.pool.query('DELETE FROM dsh_instances WHERE user_id = $1', [userId])
}
// ── 集群化:worker 注册表 + 归属/租约(v7;T08 S2 / 设计 §3.1–§3.2)────────
// 与 `repo.ts` 的同名 SQLite 实现**逐条对齐**(两套实现并存是本库既有事实,
// 见档案 19 §C8):任何 schema/语义变更都要**两侧同改**,否则切库时才炸。
async upsertDshHost(input: UpsertDshHostInput): Promise<DshHost> {
try {
const { rows } = await this.pool.query(
`INSERT INTO dsh_hosts (id, endpoint, agent_token, capacity_mb, used_mb, status, last_heartbeat)
VALUES ($1, $2, $3, $4, 0, $5, NULL)
ON CONFLICT(id) DO UPDATE SET
endpoint = excluded.endpoint,
agent_token = excluded.agent_token,
capacity_mb = excluded.capacity_mb,
status = excluded.status
RETURNING id, endpoint, agent_token, capacity_mb, used_mb, status, last_heartbeat`,
[input.id, input.endpoint, input.agentToken, input.capacityMb, input.status ?? 'up'],
)
return toDshHost(rows[0] as Record<string, unknown>)
} catch (e) {
mapPgError(e)
}
}
async findDshHost(id: string): Promise<DshHost | undefined> {
const { rows } = await this.pool.query(
`SELECT ${HOST_COLS} FROM dsh_hosts WHERE id = $1`,
[id],
)
return rows.length > 0 ? toDshHost(rows[0] as Record<string, unknown>) : undefined
}
async listDshHosts(): Promise<DshHost[]> {
const { rows } = await this.pool.query(`SELECT ${HOST_COLS} FROM dsh_hosts ORDER BY id ASC`)
return rows.map((row) => toDshHost(row as Record<string, unknown>))
}
async setDshHostStatus(
id: string,
status: DshHostStatus,
usedMb?: number,
heartbeatAt?: number,
): Promise<boolean> {
const result = await this.pool.query(
`UPDATE dsh_hosts
SET status = $1,
used_mb = COALESCE($2, used_mb),
last_heartbeat = COALESCE($3, last_heartbeat)
WHERE id = $4`,
[status, usedMb ?? null, heartbeatAt ?? null, id],
)
return (result.rowCount ?? 0) > 0
}
/**
* **原子抢占**(承重墙):PG 侧用 `UPDATE … RETURNING` —— 只有真正更新到行才返回行,
* 比"先读后写"少一次竞态窗口(SQLite 侧用 `changes` 判定,语义等价)。
*/
async claimInstance(
userId: string,
hostId: string,
ttlMs: number,
meta?: { folder?: string; patch?: string },
): Promise<ClaimResult> {
const now = Date.now()
const id = clusterInstanceId(userId)
await this.pool.query(
`INSERT INTO dsh_instances (id, user_id, role, status) VALUES ($1, $2, 'main', 'starting')
ON CONFLICT(id) DO NOTHING`,
[id, userId],
)
// folder/patch 一起落库:迁移要能复现启动参数(见 repo.ts 同名处注释)
const res = await this.pool.query(
`UPDATE dsh_instances
SET host_id = $1, epoch = epoch + 1, heartbeat_at = $2, lease_until = $3,
folder = COALESCE($4, folder), patch = COALESCE($5, patch)
WHERE id = $6 AND (host_id IS NULL OR lease_until < $2)
RETURNING epoch, lease_until`,
[hostId, now, now + ttlMs, meta?.folder ?? null, meta?.patch ?? null, id],
)
if (res.rows.length > 0) {
const row = res.rows[0] as { epoch: number; lease_until: number }
return { ok: true, epoch: row.epoch, leaseUntil: row.lease_until }
}
const cur = await this.pool.query('SELECT host_id, lease_until FROM dsh_instances WHERE id = $1', [id])
const row = cur.rows[0] as { host_id: string | null; lease_until: number } | undefined
return { ok: false, holder: row?.host_id ?? null, leaseUntil: row?.lease_until ?? 0 }
}
async renewInstanceLease(
userId: string,
hostId: string,
epoch: number,
ttlMs: number,
): Promise<boolean> {
const now = Date.now()
const result = await this.pool.query(
`UPDATE dsh_instances SET heartbeat_at = $1, lease_until = $2
WHERE id = $3 AND host_id = $4 AND epoch = $5`,
[now, now + ttlMs, clusterInstanceId(userId), hostId, epoch],
)
return (result.rowCount ?? 0) > 0
}
/** 只清租约、**保留 host_id**(见 repo.ts 同名函数的长注释)。 */
async releaseInstanceLease(userId: string, hostId: string, epoch: number): Promise<boolean> {
const result = await this.pool.query(
`UPDATE dsh_instances SET lease_until = 0
WHERE id = $1 AND host_id = $2 AND epoch = $3`,
[clusterInstanceId(userId), hostId, epoch],
)
return (result.rowCount ?? 0) > 0
}
/** 钉住归属(首次触达工作区时用):只写 host_id。 */
async pinInstanceHost(userId: string, hostId: string): Promise<void> {
const now = Date.now()
const id = clusterInstanceId(userId)
await this.pool.query(
`INSERT INTO dsh_instances (id, user_id, role, status, host_id, epoch, heartbeat_at, lease_until)
VALUES ($1, $2, 'main', 'stopped', $3, 0, 0, 0)
ON CONFLICT(id) DO NOTHING`,
[id, userId, hostId],
)
await this.pool.query(
`UPDATE dsh_instances SET host_id = $1 WHERE id = $2 AND (host_id IS NULL OR lease_until < $3)`,
[hostId, id, now],
)
}
async listExpiredInstanceLeases(now: number): Promise<DshInstance[]> {
const { rows } = await this.pool.query(
`SELECT ${INSTANCE_COLS} FROM dsh_instances
WHERE role = 'main' AND host_id IS NOT NULL AND lease_until < $1 AND status <> 'stopped'
ORDER BY lease_until ASC`,
[now],
)
return rows.map((row) => toDshInstance(row as Record<string, unknown>))
}
async listInstancesByHost(hostId: string): Promise<DshInstance[]> {
const { rows } = await this.pool.query(
`SELECT ${INSTANCE_COLS} FROM dsh_instances WHERE host_id = $1 ORDER BY user_id ASC`,
[hostId],
)
return rows.map((row) => toDshInstance(row as Record<string, unknown>))
}
async close(): Promise<void> {
await this.pool.end()
}
+168 -1
View File
@@ -12,20 +12,25 @@ import { randomUUID } from 'node:crypto'
import type { Database } from './connection.js'
import { prepare } from './prepared.js'
import {
clusterInstanceId,
toBusinessPlugin,
toDomain,
toDshHost,
toDshInstance,
toPublicUser,
toSession,
toUser,
toWorkspace,
type BusinessPlugin,
type ClaimResult,
type CredentialKey,
type CredentialKeyMeta,
type CredentialLandingRow,
type CreateSessionInput,
type CreateUserInput,
type Domain,
type DshHost,
type DshHostStatus,
type DshInstance,
type DshInstanceRole,
type DshInstanceStatus,
@@ -33,6 +38,7 @@ import {
type SessionRow,
type SessionUser,
type UpsertBusinessPluginInput,
type UpsertDshHostInput,
type UpsertDshInstanceInput,
type User,
type UserRole,
@@ -43,7 +49,8 @@ const USER_COLS = 'id, username, pass_hash, role, home_dir, api_key_ref, created
const DOMAIN_COLS = 'id, user_id, domain, verified, nginx_config, updated_at'
const BUSINESS_PLUGIN_COLS = 'id, name, description, version, tgz_path, file_size, uploaded_by, created_at, updated_at'
const INSTANCE_COLS =
'id, user_id, workspace_id, role, pid, port, status, started_at, last_exit, exit_code, last_error, folder, patch'
'id, user_id, workspace_id, role, pid, port, status, started_at, last_exit, exit_code, last_error, folder, patch, '
+ 'host_id, epoch, heartbeat_at, lease_until' // v7 集群化归属/租约(T08 S2)—— 漏了它们会让 hostId 恒为 null
export function createUser(db: Database, input: CreateUserInput, baseUid: number): User {
const createdAt = Date.now()
@@ -582,3 +589,163 @@ export function deleteBusinessPlugin(db: Database, id: string): boolean {
const info = prepare(db, 'DELETE FROM business_plugins WHERE id = ?').run(id)
return info.changes > 0
}
// ── 集群化:worker 注册表 + 实例归属/租约(v7;T08 S2 / 设计 §3.1–§3.2)────────
//
// ⚠️ local 模式**不调用**这些函数(`LocalSpawner` 靠进程内 Map + 单机互斥),
// 所以它们的存在不会改变现有单机行为。
const HOST_COLS = 'id, endpoint, agent_token, capacity_mb, used_mb, status, last_heartbeat'
/** 注册/更新一台 worker。join 幂等:同 id 重复执行 = 更新(并把它标回 `up`)。 */
export function upsertDshHost(db: Database, input: UpsertDshHostInput): DshHost {
prepare(db, `
INSERT INTO dsh_hosts (id, endpoint, agent_token, capacity_mb, used_mb, status, last_heartbeat)
VALUES (?, ?, ?, ?, 0, ?, NULL)
ON CONFLICT(id) DO UPDATE SET
endpoint = excluded.endpoint,
agent_token = excluded.agent_token,
capacity_mb = excluded.capacity_mb,
status = excluded.status
`).run(input.id, input.endpoint, input.agentToken, input.capacityMb, input.status ?? 'up')
const row = prepare(db, `SELECT ${HOST_COLS} FROM dsh_hosts WHERE id = ?`).get(input.id)
return toDshHost(row as Record<string, unknown>)
}
export function findDshHost(db: Database, id: string): DshHost | undefined {
const row = prepare(db, `SELECT ${HOST_COLS} FROM dsh_hosts WHERE id = ?`).get(id)
return row ? toDshHost(row as Record<string, unknown>) : undefined
}
export function listDshHosts(db: Database): DshHost[] {
const rows = prepare(db, `SELECT ${HOST_COLS} FROM dsh_hosts ORDER BY id ASC`).all() as Array<
Record<string, unknown>
>
return rows.map((row) => toDshHost(row))
}
/** 心跳/状态上报(只更新显式给出的字段,避免 heartbeat 覆盖 status)。 */
export function setDshHostStatus(
db: Database,
id: string,
status: DshHostStatus,
usedMb?: number,
heartbeatAt?: number,
): boolean {
const info = prepare(db, `
UPDATE dsh_hosts
SET status = ?,
used_mb = COALESCE(?, used_mb),
last_heartbeat = COALESCE(?, last_heartbeat)
WHERE id = ?
`).run(status, usedMb ?? null, heartbeatAt ?? null, id)
return info.changes > 0
}
/**
* **原子抢占**某用户 main 实例的归属(承重墙)。仅当"无人持有 **或** 租约已过期"才成功,
* 成功时 `epoch` +1(fencing token)。失败时返回当前持有者与租约到期时刻。
*/
export function claimInstance(
db: Database,
userId: string,
hostId: string,
ttlMs: number,
meta?: { folder?: string; patch?: string },
): ClaimResult {
const now = Date.now()
const id = clusterInstanceId(userId)
// 新用户没有 dsh_instances 行 ⇒ 先保证行存在(否则 UPDATE 影响 0 行被误判为"有人在管")。
prepare(db, `
INSERT INTO dsh_instances (id, user_id, role, status)
VALUES (?, ?, 'main', 'starting')
ON CONFLICT(id) DO NOTHING
`).run(id, userId)
// ⚠️ **必须把 folder/patch 一起落库**(2026-09-15 实测踩到):集群模式下实例行是这里建的,
// 而 local 模式不写库 ⇒ 若这里不记,`folder` 永远是 NULL,**迁移时复现不了启动参数**
// (表现为 `bwrap: Can't chdir to :` 空路径 ⇒ 崩溃循环)。用 COALESCE 保证不覆盖已有值。
const info = prepare(db, `
UPDATE dsh_instances
SET host_id = ?, epoch = epoch + 1, heartbeat_at = ?, lease_until = ?,
folder = COALESCE(?, folder), patch = COALESCE(?, patch)
WHERE id = ? AND (host_id IS NULL OR lease_until < ?)
`).run(hostId, now, now + ttlMs, meta?.folder ?? null, meta?.patch ?? null, id, now)
const row = prepare(db, 'SELECT host_id, epoch, lease_until FROM dsh_instances WHERE id = ?').get(id) as
| { host_id: string | null; epoch: number; lease_until: number }
| undefined
if (row === undefined) return { ok: false, holder: null, leaseUntil: 0 }
return info.changes > 0
? { ok: true, epoch: row.epoch, leaseUntil: row.lease_until }
: { ok: false, holder: row.host_id, leaseUntil: row.lease_until }
}
/** 续租。**必须带 epoch**:不匹配说明已被他人抢占 ⇒ 返回 false(fencing 生效)。 */
export function renewInstanceLease(
db: Database,
userId: string,
hostId: string,
epoch: number,
ttlMs: number,
): boolean {
const now = Date.now()
const info = prepare(db, `
UPDATE dsh_instances SET heartbeat_at = ?, lease_until = ?
WHERE id = ? AND host_id = ? AND epoch = ?
`).run(now, now + ttlMs, clusterInstanceId(userId), hostId, epoch)
return info.changes > 0
}
/**
* 主动释放**租约**(停实例时)。带 epoch 校验,避免误清他人的归属。
*
* ⚠️ **只清 `lease_until`,保留 `host_id`**(2026-09-15 生产切换暴露):
* `host_id` 的语义是「**这个用户的数据在哪台机器**」—— 用户的工作区在**本地盘**上,
* 把归属一起清掉就等于**丢掉粘性锚点**,下次启动可能被调度到没有他数据的机器上(工作区看起来是空的)。
* 「谁现在在托管」是**租约**(`lease_until`)的语义,所以释放只该清租约。
*/
export function releaseInstanceLease(db: Database, userId: string, hostId: string, epoch: number): boolean {
const info = prepare(db, `
UPDATE dsh_instances SET lease_until = 0
WHERE id = ? AND host_id = ? AND epoch = ?
`).run(clusterInstanceId(userId), hostId, epoch)
return info.changes > 0
}
/**
* **钉住**某用户的归属(首次触达其工作区时用):只写 `host_id`,不动 epoch/租约。
*
* 为什么需要:新用户还没有归属,`selectHost` 会在**写文件那一步**与**launch 那一步**各自选一次,
* 两次可能选到不同机器 ⇒ 「文件写到 A、实例起在 B」⇒ 实例看不到自己的文件(2026-09-15 实测)。
* 首次触达就把归属钉住,后续(含 launch)都走粘性,两面必然一致。
*/
export function pinInstanceHost(db: Database, userId: string, hostId: string): void {
const now = Date.now()
const id = clusterInstanceId(userId)
prepare(db, `
INSERT INTO dsh_instances (id, user_id, role, status, host_id, epoch, heartbeat_at, lease_until)
VALUES (?, ?, 'main', 'stopped', ?, 0, 0, 0)
ON CONFLICT(id) DO NOTHING
`).run(id, userId, hostId)
prepare(db, `
UPDATE dsh_instances SET host_id = ?
WHERE id = ? AND (host_id IS NULL OR lease_until < ?)
`).run(hostId, id, now)
}
/** 租约过期但仍标着归属的 main 实例(供巡检/自愈;**不等于可以立即接管**,见 R9)。 */
export function listExpiredInstanceLeases(db: Database, now: number): DshInstance[] {
const rows = prepare(db, `
SELECT ${INSTANCE_COLS} FROM dsh_instances
WHERE role = 'main' AND host_id IS NOT NULL AND lease_until < ? AND status <> 'stopped'
ORDER BY lease_until ASC
`).all(now) as Array<Record<string, unknown>>
return rows.map((row) => toDshInstance(row))
}
/** 某 worker 上的全部实例 —— 对账**一次拿回整机**(替代逐用户查询)。 */
export function listInstancesByHost(db: Database, hostId: string): DshInstance[] {
const rows = prepare(db, `SELECT ${INSTANCE_COLS} FROM dsh_instances WHERE host_id = ? ORDER BY user_id ASC`).all(
hostId,
) as Array<Record<string, unknown>>
return rows.map((row) => toDshInstance(row))
}
+54
View File
@@ -292,6 +292,59 @@ ALTER TABLE credential_vault ADD COLUMN models TEXT;
ALTER TABLE users ADD COLUMN shared_model_enabled INTEGER NOT NULL DEFAULT 1;
`
// v7: 集群化 —— worker 注册表 + 实例归属/租约(T08 S2;设计 §3.1/§3.2)。
//
// 为什么需要它:local 模式靠"进程内 Map + 单机"天然保证「一个用户只有一个活实例」;
// 多机后这个保证必须落到 DB 的**原子 CAS** 上,否则两个 worker 会同时写同一个
// `$DSH_HOME`(会话日志 append 冲突 ⇒ 数据损坏)。
//
// · `dsh_hosts` = worker 注册表:agent 地址、容量、水位、心跳时间。
// · `dsh_instances.{host_id, epoch, heartbeat_at, lease_until}` = 归属与租约。
// `epoch` 是 **fencing token**:抢占时 +1,旧持有者的写入据此被拒(防脑裂双写)。
//
// 抢占语义(两方言同款,见 `repo.ts` 的 claimInstance / `pg.ts` 同名方法):
// `INSERT … ON CONFLICT(id) DO UPDATE SET … WHERE host_id IS NULL OR lease_until < now`
// —— 冲突时仅在"无人持有或租约过期"才更新;否则**不动行也不报错**,
// 调用方以「受影响行数 0」判定"有人在管"。
// ⚠️ 时间戳一律 **epoch 毫秒 BIGINT**(与全库一致,勿用 timestamptz)。
const SQLITE_V7 = `
CREATE TABLE IF NOT EXISTS dsh_hosts (
id TEXT PRIMARY KEY,
endpoint TEXT NOT NULL,
agent_token TEXT NOT NULL,
capacity_mb INTEGER NOT NULL DEFAULT 0,
used_mb INTEGER NOT NULL DEFAULT 0,
status TEXT NOT NULL DEFAULT 'up'
CHECK (status IN ('up','draining','down')),
last_heartbeat INTEGER
);
ALTER TABLE dsh_instances ADD COLUMN host_id TEXT;
ALTER TABLE dsh_instances ADD COLUMN epoch INTEGER NOT NULL DEFAULT 0;
ALTER TABLE dsh_instances ADD COLUMN heartbeat_at INTEGER NOT NULL DEFAULT 0;
ALTER TABLE dsh_instances ADD COLUMN lease_until INTEGER NOT NULL DEFAULT 0;
CREATE INDEX IF NOT EXISTS idx_dsh_instances_host ON dsh_instances (host_id);
CREATE INDEX IF NOT EXISTS idx_dsh_instances_lease ON dsh_instances (lease_until);
`
const PG_V7 = `
CREATE TABLE IF NOT EXISTS dsh_hosts (
id TEXT PRIMARY KEY,
endpoint TEXT NOT NULL,
agent_token TEXT NOT NULL,
capacity_mb BIGINT NOT NULL DEFAULT 0,
used_mb BIGINT NOT NULL DEFAULT 0,
status TEXT NOT NULL DEFAULT 'up'
CHECK (status IN ('up','draining','down')),
last_heartbeat BIGINT
);
ALTER TABLE dsh_instances ADD COLUMN host_id TEXT;
ALTER TABLE dsh_instances ADD COLUMN epoch BIGINT NOT NULL DEFAULT 0;
ALTER TABLE dsh_instances ADD COLUMN heartbeat_at BIGINT NOT NULL DEFAULT 0;
ALTER TABLE dsh_instances ADD COLUMN lease_until BIGINT NOT NULL DEFAULT 0;
CREATE INDEX IF NOT EXISTS idx_dsh_instances_host ON dsh_instances (host_id);
CREATE INDEX IF NOT EXISTS idx_dsh_instances_lease ON dsh_instances (lease_until);
`
interface Migration {
version: number
name: string
@@ -306,6 +359,7 @@ const MIGRATIONS: readonly Migration[] = [
{ version: 4, name: 'instance desired state', sqlite: SQLITE_V4, pg: PG_V4 },
{ version: 5, name: 'business plugin candidate pool', sqlite: SQLITE_V5, pg: PG_V5 },
{ version: 6, name: 'user model providers', sqlite: SQLITE_V6, pg: PG_V6 },
{ version: 7, name: 'cluster host registry + instance lease', sqlite: SQLITE_V7, pg: PG_V7 },
]
/** Apply unapplied SQLite migrations inside a single transaction. */
+73
View File
@@ -57,18 +57,32 @@ import {
setUserRole as setUserRoleSync,
setUserUid as setUserUidSync,
toggleCredentialKey as toggleCredentialKeySync,
// 集群化(v7;T08 S2)
claimInstance as claimInstanceSync,
findDshHost as findDshHostSync,
listDshHosts as listDshHostsSync,
listExpiredInstanceLeases as listExpiredInstanceLeasesSync,
listInstancesByHost as listInstancesByHostSync,
pinInstanceHost as pinInstanceHostSync,
releaseInstanceLease as releaseInstanceLeaseSync,
renewInstanceLease as renewInstanceLeaseSync,
setDshHostStatus as setDshHostStatusSync,
upsertDshHost as upsertDshHostSync,
upsertBusinessPlugin as upsertBusinessPluginSync,
upsertDomain as upsertDomainSync,
upsertInstance as upsertInstanceSync,
} from './repo.js'
import type {
BusinessPlugin,
ClaimResult,
CredentialKey,
CredentialKeyMeta,
CredentialLandingRow,
CreateSessionInput,
CreateUserInput,
Domain,
DshHost,
DshHostStatus,
DshInstance,
DshInstanceRole,
DshInstanceStatus,
@@ -76,6 +90,7 @@ import type {
SessionRow,
SessionUser,
UpsertBusinessPluginInput,
UpsertDshHostInput,
UpsertDshInstanceInput,
User,
UserRole,
@@ -321,6 +336,64 @@ export class SqliteAdapter implements DbAdapter {
deleteUserInstancesSync(this.db, userId)
}
// ── 集群化:worker 注册表 + 归属/租约(v7;T08 S2)────────────────────────
// local 模式不会走到这些方法(`LocalSpawner` 不写库),它们只是让
// **SQLite 侧与 PG 侧行为一致** —— 测试与单机试跑都需要。
async upsertDshHost(input: UpsertDshHostInput): Promise<DshHost> {
try {
return upsertDshHostSync(this.db, input)
} catch (e) {
mapSqliteError(e)
}
}
async findDshHost(id: string): Promise<DshHost | undefined> {
return findDshHostSync(this.db, id)
}
async listDshHosts(): Promise<DshHost[]> {
return listDshHostsSync(this.db)
}
async setDshHostStatus(
id: string,
status: DshHostStatus,
usedMb?: number,
heartbeatAt?: number,
): Promise<boolean> {
return setDshHostStatusSync(this.db, id, status, usedMb, heartbeatAt)
}
async claimInstance(
userId: string,
hostId: string,
ttlMs: number,
meta?: { folder?: string; patch?: string },
): Promise<ClaimResult> {
return claimInstanceSync(this.db, userId, hostId, ttlMs, meta)
}
async renewInstanceLease(userId: string, hostId: string, epoch: number, ttlMs: number): Promise<boolean> {
return renewInstanceLeaseSync(this.db, userId, hostId, epoch, ttlMs)
}
async releaseInstanceLease(userId: string, hostId: string, epoch: number): Promise<boolean> {
return releaseInstanceLeaseSync(this.db, userId, hostId, epoch)
}
async pinInstanceHost(userId: string, hostId: string): Promise<void> {
pinInstanceHostSync(this.db, userId, hostId)
}
async listExpiredInstanceLeases(now: number): Promise<DshInstance[]> {
return listExpiredInstanceLeasesSync(this.db, now)
}
async listInstancesByHost(hostId: string): Promise<DshInstance[]> {
return listInstancesByHostSync(this.db, hostId)
}
async close(): Promise<void> {
this.db.close()
}
+74
View File
@@ -90,6 +90,15 @@ export interface DshInstance {
folder: string | null
/** Rendered Cordis patch content (not a path — the control plane holds no user volume). */
patch: string | null
// ── 集群化归属与租约(v7;T08 S2 / 设计 §3.1)—— local 模式下恒为 null/0 ──
/** 托管该实例的 worker(`dsh_hosts.id`);null = 未被任何 worker 认领。 */
hostId: string | null
/** **fencing token**:每次抢占 +1;旧持有者的写入据此被拒(防脑裂双写)。 */
epoch: number
/** 最近一次心跳(epoch 毫秒)。 */
heartbeatAt: number
/** 租约到期时刻(epoch 毫秒);早于 now 即可被他人抢占。 */
leaseUntil: number
}
/** A named per-user credential key (secret never exposed). */
@@ -262,6 +271,10 @@ export function toDshInstance(row: Record<string, unknown>): DshInstance {
lastError: (row.last_error as string | null) ?? null,
folder: (row.folder as string | null) ?? null,
patch: (row.patch as string | null) ?? null,
hostId: (row.host_id as string | null) ?? null,
epoch: (row.epoch as number | null) ?? 0,
heartbeatAt: (row.heartbeat_at as number | null) ?? 0,
leaseUntil: (row.lease_until as number | null) ?? 0,
}
}
@@ -289,3 +302,64 @@ export function toBusinessPlugin(row: Record<string, unknown>): BusinessPlugin {
updatedAt: row.updated_at as number,
}
}
// ── 集群化:worker 注册表与租约结果(v7;T08 S2 / 设计 §3.1–§3.2)──────────────
/** Worker 健康状态(`dsh_hosts.status` 的 CHECK 镜像)。 */
export type DshHostStatus = 'up' | 'draining' | 'down'
/** 一台承载用户实例的 worker(= 设计里的 Worker 节点)。 */
export interface DshHost {
id: string
/** agent 的内网地址,如 `10.0.1.11:9000`。 */
endpoint: string
/** 内部 HMAC 密钥(**只应存在于 DB 与 Manager 内存**,绝不经 API 返回)。 */
agentToken: string
/** 该机可用内存预算(MB);0 = 不承载实例(只做门户/控制)。 */
capacityMb: number
/** 由心跳上报的已用内存(MB)。 */
usedMb: number
status: DshHostStatus
/** 最近心跳(epoch 毫秒);null = 从未上报。 */
lastHeartbeat: number | null
}
/** Upsert payload for `dsh_hosts`(join 脚本/管理面用)。 */
export interface UpsertDshHostInput {
id: string
endpoint: string
agentToken: string
capacityMb: number
status?: DshHostStatus
}
/**
* 抢占结果。`ok:false` 时带回**当前持有者**与租约到期时刻,便于调用方决定
* "退让"还是"报告异常"(**不要据此接管** —— 见项目红线 R9)。
*/
export type ClaimResult =
| { ok: true; epoch: number; leaseUntil: number }
| { ok: false; holder: string | null; leaseUntil: number }
/**
* 集群模式下 main 实例行的**确定性 id**。
*
* 为什么需要确定性:租约是以 **(user, role='main')** 为单位的,`dsh_instances.id` 只是载体;
* 若每次 spawn 用随机 id,抢占时会插出多行 ⇒ 归属判断失效。local 模式仍用随机 id
* (它不写库),集群路径一律走这里。
*/
export function clusterInstanceId(userId: string): string {
return `dsh-${userId}`
}
export function toDshHost(row: Record<string, unknown>): DshHost {
return {
id: row.id as string,
endpoint: row.endpoint as string,
agentToken: row.agent_token as string,
capacityMb: (row.capacity_mb as number | null) ?? 0,
usedMb: (row.used_mb as number | null) ?? 0,
status: row.status as DshHostStatus,
lastHeartbeat: (row.last_heartbeat as number | null) ?? null,
}
}