diff --git a/CLAUDE.md b/CLAUDE.md index ca3879b..d0a2bbd 100644 --- a/CLAUDE.md +++ b/CLAUDE.md @@ -40,6 +40,7 @@ A client-only single-page app (React + Vite + TypeScript) to plan a [metal-stack - A `Partition` is a metal-stack failure domain (one site / room) — the term is always "partition", never "zone". A plan has no topology variant: the partitions it holds say what it is. - Fabric types per partition: `leaf-spine` | `leaf-spine-superspine` (`FabricConfig.fabricType`); superspines and storage leaves only enter the BOM/validation when configured. - **Central rack**: spines, exits, superspines, mgmt spines and mgmt servers live together in a central rack per partition; compute racks hold ToR leaves, worker/storage server groups, and a mgmt leaf. There is no separate OOB switch — the mgmt leaf carries BMC/out-of-band access. +- **Control plane** (`Plan.controlPlane`, plan-level: one control plane serves every partition): the Kubernetes cluster running metal-api, masterdata-api, the IPAM and their databases. `hosting: 'kaas'` is a managed cluster that orders no hardware and shows as a capsule on each partition's routers; `hosting: 'on-prem'` are nodes this plan buys, either in the host partition's central rack (attached to its exit switches, `placement: 'central-rack'`) or in a control-plane rack of their own with a leaf pair uplinked to the spines (`'own-rack'`). `src/derive/controlPlane.ts` is the single source of those placement rules and the BOM, the elevations, the topology and validation all ask it; its leaves count in `spinePortsPerSpine()` and its node ports in the exit switch budget. These nodes are never metal-stack-managed machines, so they stay out of `planNodes()` and the compatibility list does not apply. Validation is deliberately thin: an error without nodes, a warning below three (etcd quorum). Where the cluster runs is free per the deployment guide, so nothing checks reachability. - **Rack kinds** (`Rack.kind`): `single` (one physical rack) or `rack-group` (three physical racks sharing the middle rack's leaf pair and mgmt leaf; chassis are distributed evenly per server group — each goes to the physical rack holding the fewest chassis of its group, the least-used by height units among those, ties mid → left → right). `physicalRacks()` in `derive/rackLayout.ts` is the single source of that spread; the topology graph and the elevations both consume it, so a rack group always shows as three physical racks — in the racks view inside one dashed group box (`RackElevation.group`). A group's `name` names the group ("Rack group 1"); its physical racks are named by `memberNames` (left, middle, right). Names are editable, but defaults are unique per partition: `newRack()` / `withRackKind()` in `model/defaults.ts` number every physical rack above the highest `Rack ` in use (a group takes three numbers, switching kinds keeps names unique), and validation warns on duplicate physical rack names. BOM sections, IP-plan subnets and rack-level validation use the group name; elevations and the topology use the physical names. In both views the mgmt leaf sits at the top of a compute rack, above the leaves. Leaf port capacity counts only spine uplinks against front-panel ports — leaf↔mgmt connectivity uses the switches' dedicated OOB mgmt port. - **Management network** (`FabricConfig.mgmt`): has its own topology — `layer: 'l2' | 'l3'` and `redundant: boolean`; redundancy drives the count of mgmt spines and mgmt servers (2 vs 1, `mgmtDeviceCount()`). Production tiers use Edgecore AS7726 (the AS7712 is vendor end-of-life), management tiers AS4630/AS4625. Every switch has exactly one management interface: leaves connect it to the rack's mgmt leaf, central-rack switches and routers to the mgmt spines — derived, never configured. `FabricConfig.leafSpineLinks` (default 1) is the number of 100G links from each leaf to each spine. External networks (internet, company networks, storage) attach in the central rack: internet and company networks at the internet routers (at the exits when a partition has no routers), a storage network at the partition's storage leaves, or at the exits when it has none (`attachesAtStorageLeaves()` in `derive/topology.ts` is the single source of that rule, used by the derivation and by `Diagram` to draw storage capsules over the storage box). `filterTopology()` drops an external network whose attachment points all fell away. - **GPUs.** `ServerGroup.gpu` (optional `{ modelId, perNode }`) fits every node of a group alike: a BOM line at `nodes × perNode`, watts folded into the chassis slot so the rack power estimate picks them up, and a validation error when `perNode` exceeds the server model's catalog `gpuCapable` (absent = takes no GPUs). `gpusForServer()` populates the dropdown, so the field only appears for GPU-capable models. diff --git a/README.md b/README.md index 8e40d1f..db416b3 100644 --- a/README.md +++ b/README.md @@ -159,14 +159,15 @@ default plan. Use **Export JSON** to save a plan file and **Import JSON** to loa The whole app operates on a single `Plan` document, described by Zod schemas in `src/model/plan.ts`. Everything else is derived from it and never stored: -| Module | Derives | -| -------------------------- | ------------------------------------------------------- | -| `src/derive/bom.ts` | BOM lines and quantities | -| `src/derive/topology.ts` | the topology graph (nodes and links) | -| `src/derive/rackLayout.ts` | rack elevations and the rack-group spread | -| `src/derive/validate.ts` | validation issues | -| `src/derive/nodes.ts` | node tallies per rack, partition and plan | -| `src/derive/ip/` | CIDR arithmetic, the IP address plan and its validation | +| Module | Derives | +| ---------------------------- | ------------------------------------------------------- | +| `src/derive/bom.ts` | BOM lines and quantities | +| `src/derive/topology.ts` | the topology graph (nodes and links) | +| `src/derive/controlPlane.ts` | where the control plane's hardware lands | +| `src/derive/rackLayout.ts` | rack elevations and the rack-group spread | +| `src/derive/validate.ts` | validation issues | +| `src/derive/nodes.ts` | node tallies per rack, partition and plan | +| `src/derive/ip/` | CIDR arithmetic, the IP address plan and its validation | Hardware facts — part numbers, port counts, height units, nodes per chassis, and metal-stack compatibility — live in `src/model/catalog.ts`. The compatibility data mirrors the official diff --git a/src/derive/bom.test.ts b/src/derive/bom.test.ts index a60fd1b..b25d475 100644 --- a/src/derive/bom.test.ts +++ b/src/derive/bom.test.ts @@ -441,3 +441,71 @@ describe('deriveBom scope', () => { expect(spares).not.toContain('cable-mtp-trunk#spare') }) }) + +describe('control plane', () => { + /** A plan whose control plane runs on-prem, `patch` applied on top. */ + function onPrem(patch: Partial = {}): Plan { + const plan = createEmptyPlan() + plan.controlPlane = { ...plan.controlPlane, hosting: 'on-prem', ...patch } + return plan + } + + function line(plan: Plan, catalogId: string) { + return deriveBom(plan).find((l) => l.catalogId === catalogId) + } + + it('orders nothing for a managed control plane', () => { + const kaas = createEmptyPlan() + expect(kaas.controlPlane.hosting).toBe('kaas') + // The default plan has mgmt servers of the same model; on-prem adds to it. + const managed = line(kaas, 'server-mgmt-121h')?.quantity ?? 0 + expect(line(onPrem(), 'server-mgmt-121h')?.quantity).toBe(managed + 3) + }) + + it('adds nodes, NICs, optics and mgmt copper in the central rack', () => { + const plan = onPrem() + const nodes = line(plan, 'server-mgmt-121h')! + const nodeReason = nodes.reasons.find((r) => r.detail.includes('control plane nodes'))! + expect(nodeReason).toMatchObject({ quantity: 3, where: 'Central rack' }) + + // 3 nodes × 1 dual-port 25G NIC, 6 server ports, 2 breakout groups. + const nics = line(plan, 'nic-e810-xxvda2')! + expect(nics.reasons.find((r) => r.detail.includes('control plane'))?.quantity).toBe(3) + const optics = line(plan, 'sfp-25g-sr')! + expect(optics.reasons.find((r) => r.detail.includes('control plane'))?.quantity).toBe(6) + const breakouts = line(plan, 'cable-mtp-breakout')! + expect(breakouts.reasons.find((r) => r.detail.includes('control plane'))?.quantity).toBe(2) + + // One mgmt interface per node to the mgmt spines. + const copper = line(plan, 'cable-rj45')! + expect( + copper.reasons.find((r) => r.detail.includes('control plane nodes × 1 mgmt interface')), + ).toMatchObject({ quantity: 3, where: 'Central rack' }) + }) + + it('uses 100G point to point for 2x100G nodes', () => { + const plan = onPrem({ uplink: '2x100G' }) + const nics = line(plan, 'nic-e810-cqda2')! + expect(nics.reasons.find((r) => r.detail.includes('control plane'))?.quantity).toBe(3) + const trunks = line(plan, 'cable-mtp-trunk')! + expect(trunks.reasons.find((r) => r.detail.includes('control plane'))?.quantity).toBe(6) + // The nodes contribute no breakout; the rack's workers still do. + const breakouts = line(plan, 'cable-mtp-breakout')! + expect(breakouts.reasons.some((r) => r.detail.includes('control plane'))).toBe(false) + }) + + it('gives an own rack its leaves, licenses and mgmt leaf', () => { + const plan = onPrem({ placement: 'own-rack' }) + const where = plan.controlPlane.rack.name + const leaves = line(plan, 'switch-as7726')! + expect(leaves.reasons.find((r) => r.where === where)).toMatchObject({ quantity: 2 }) + const licenses = line(plan, 'lic-sonic-eb-100g')! + expect(licenses.reasons.some((r) => r.where === where)).toBe(true) + const mgmtLeaf = line(plan, 'switch-as4630')! + expect(mgmtLeaf.reasons.some((r) => r.where === where)).toBe(true) + + // Those leaves uplink to the spines like any other rack's. + const central = line(createEmptyPlan(), 'cable-mtp-trunk')?.quantity ?? 0 + expect(line(plan, 'cable-mtp-trunk')!.quantity).toBeGreaterThan(central) + }) +}) diff --git a/src/derive/bom.ts b/src/derive/bom.ts index 79c0a73..4f0e924 100644 --- a/src/derive/bom.ts +++ b/src/derive/bom.ts @@ -6,7 +6,9 @@ import { type Plan, type Rack, type ServerGroup, + type UplinkSpeed, } from '../model/plan' +import { controlPlaneLeafCount, hasOwnRack, inCentralRack } from './controlPlane' // The BOM is always derived from the Plan, never stored. Every quantity rule // lives here and gets a unit test in bom.test.ts. Every rule also records a @@ -235,35 +237,48 @@ function addServerGroup(bom: BomBuilder, group: ServerGroup): void { bom.add(group.gpu.modelId, gpus, `${group.count} ${role} nodes × ${group.gpu.perNode} GPU`) } + addNodeUplinks(bom, group.count, group.uplink, role) +} + +/** NIC, transceivers and cables for dual-attached nodes, whatever they are: + * server groups on their rack's leaves and the control-plane nodes on the + * switch they hang off. `what` names them in the reasons ("worker", + * "control plane"). */ +function addNodeUplinks(bom: BomBuilder, nodes: number, uplink: UplinkSpeed, what: string): void { // Every node carries one dual-port NIC matching its uplink speed. - const uplinkPorts = 2 * group.count - if (group.uplink === '2x25G') { - bom.add('nic-e810-xxvda2', group.count, `${group.count} ${role} nodes × 1 NIC`) - // 25G server ports terminate on 100G leaf ports via 4x25G breakout: - // server side gets a 25G-SR transceiver per port, the leaf side one + const uplinkPorts = 2 * nodes + if (uplink === '2x25G') { + bom.add('nic-e810-xxvda2', nodes, `${nodes} ${what} nodes × 1 NIC`) + // 25G server ports terminate on 100G switch ports via 4x25G breakout: + // server side gets a 25G-SR transceiver per port, the switch side one // 100G-SR4 per started group of four, joined by an MTP breakout cable. - bom.add('sfp-25g-sr', uplinkPorts, `${group.count} ${role} nodes × 2 server ports`) - const leafPorts = Math.ceil(uplinkPorts / 4) + bom.add('sfp-25g-sr', uplinkPorts, `${nodes} ${what} nodes × 2 server ports`) + const switchPorts = Math.ceil(uplinkPorts / 4) bom.add( 'sfp-100g-sr4', - leafPorts, - `${uplinkPorts} × 25G ${role} ports / 4 per breakout, leaf side`, + switchPorts, + `${uplinkPorts} × 25G ${what} ports / 4 per breakout, switch side`, + ) + bom.add( + 'cable-mtp-breakout', + switchPorts, + `${uplinkPorts} × 25G ${what} ports / 4 per breakout`, ) - bom.add('cable-mtp-breakout', leafPorts, `${uplinkPorts} × 25G ${role} ports / 4 per breakout`) } else { - bom.add('nic-e810-cqda2', group.count, `${group.count} ${role} nodes × 1 NIC`) + bom.add('nic-e810-cqda2', nodes, `${nodes} ${what} nodes × 1 NIC`) // 100G point-to-point: a 100G-SR4 transceiver on each end plus an MTP // trunk cable per link. - bom.add('sfp-100g-sr4', 2 * uplinkPorts, `${uplinkPorts} × 100G ${role} links × 2 ends`) - bom.add('cable-mtp-trunk', uplinkPorts, `${uplinkPorts} × 100G ${role} links`) + bom.add('sfp-100g-sr4', 2 * uplinkPorts, `${uplinkPorts} × 100G ${what} links × 2 ends`) + bom.add('cable-mtp-trunk', uplinkPorts, `${uplinkPorts} × 100G ${what} links`) } } /** Links each spine terminates: `leafSpineLinks` per leaf, one per exit, - * superspine and storage leaf. */ -export function spinePortsPerSpine(partition: Partition): number { + * superspine and storage leaf. `controlPlaneLeaves` are the leaves of a + * separate control-plane rack, which uplink like any other leaves. */ +export function spinePortsPerSpine(partition: Partition, controlPlaneLeaves = 0): number { const { fabric } = partition - const leaves = partition.racks.reduce((n, r) => n + r.leafCount, 0) + const leaves = partition.racks.reduce((n, r) => n + r.leafCount, 0) + controlPlaneLeaves const superspines = fabric.fabricType === 'leaf-spine-superspine' ? fabric.superspineCount : 0 return ( leaves * fabric.leafSpineLinks + fabric.exitSwitchCount + superspines + fabric.storageLeafCount @@ -275,8 +290,8 @@ export function routerLinks(partition: Partition): number { return 2 * partition.fabric.routerCount * partition.fabric.exitSwitchCount } -function addFabricLinks(bom: BomBuilder, partition: Partition): void { - const perSpine = spinePortsPerSpine(partition) +function addFabricLinks(bom: BomBuilder, plan: Plan, partition: Partition): void { + const perSpine = spinePortsPerSpine(partition, controlPlaneLeafCount(plan, partition)) const links = perSpine * partition.fabric.spineCount const why = `${perSpine} links per spine × ${partition.fabric.spineCount} spines` bom.add('sfp-100g-sr4', 2 * links, `${links} fabric links × 2 ends (${why})`) @@ -340,7 +355,76 @@ function addMgmtLinks(bom: BomBuilder, partition: Partition, central: BomBuilder /** Section the central-rack rules are reported under. */ const CENTRAL_RACK = 'Central rack' -function addPartition(bom: BomBuilder, partition: Partition): void { +/** The on-prem control-plane cluster: its nodes, their uplinks and, for a + * control-plane rack of its own, the leaf pair and mgmt leaf that serve + * them. A KaaS control plane orders nothing. Emitted inside the host + * partition's central-rack block so the section order stays physical. */ +function addControlPlane( + bom: BomBuilder, + plan: Plan, + partition: Partition, + prodCentral: BomBuilder, + mgmtCentral: BomBuilder, +): void { + const cp = plan.controlPlane + const central = inCentralRack(plan, partition) + const ownRack = hasOwnRack(plan, partition) + if (!central && !ownRack) return + + const { fabric } = partition + const nodes = cp.nodeCount + const where = central ? CENTRAL_RACK : cp.rack.name + const prod = central ? prodCentral : bom.on('production').at(where) + const mgmt = central ? mgmtCentral : bom.on('management').at(where) + + prod.add(cp.nodeModelId, nodes, `${nodes} control plane nodes`) + addNodeUplinks(prod, nodes, cp.uplink, 'control plane') + + if (ownRack) { + // A rack of its own: leaves uplinked to the spines like a compute + // rack's, plus a mgmt leaf for the nodes' BMC ports. + addSwitch( + prod, + fabric.nos, + cp.rack.leafModelId, + cp.rack.leafCount, + `${cp.rack.leafCount} leaves`, + ) + addSwitch( + mgmt, + fabric.nos, + fabric.mgmt.leafModelId, + fabric.mgmt.leafPerRack, + `${fabric.mgmt.leafPerRack} mgmt leaves`, + ) + mgmt.add('cable-rj45', nodes, `${nodes} control plane nodes × 1 BMC port to the mgmt leaf`) + mgmt.add( + 'cable-rj45', + cp.rack.leafCount, + `${cp.rack.leafCount} leaves × 1 mgmt interface to the mgmt leaf`, + ) + const mgmtCount = mgmtDeviceCount(fabric.mgmt) + const speed = mgmtUplinkSpeed(fabric.mgmt.leafModelId, fabric.mgmt.spineModelId) + const links = fabric.mgmt.leafPerRack * mgmtCount + const why = `${fabric.mgmt.leafPerRack} mgmt leaves × ${mgmtCount} mgmt spines` + mgmt.add( + speed === '25G' ? 'sfp-25g-sr' : 'sfp-10g-sr', + 2 * links, + `${links} mgmt uplinks × 2 ends (${why})`, + ) + mgmt.add('cable-lc-duplex', links, `${links} mgmt uplinks (${why})`) + } else { + // In the central rack the nodes reach the management network the same + // way every device there does: one interface to the mgmt spines. + mgmt.add( + 'cable-rj45', + nodes, + `${nodes} control plane nodes × 1 mgmt interface to the mgmt spines`, + ) + } +} + +function addPartition(bom: BomBuilder, plan: Plan, partition: Partition): void { const { fabric } = partition // Network and location are tagged once here, so every rule below inherits // the network it belongs to and the section it is reported under. @@ -395,7 +479,8 @@ function addPartition(bom: BomBuilder, partition: Partition): void { // compute rack does, and the racks follow in plan order. That makes the // reason sections come out in physical order for free: for any line, the // central rack is the first section, then Rack 1, Rack 2 and so on. - addFabricLinks(prodCentral, partition) + addControlPlane(bom, plan, partition, prodCentral, mgmtCentral) + addFabricLinks(prodCentral, plan, partition) addMgmtLinks(bom.on('management'), partition, mgmtCentral) for (const rack of partition.racks) { @@ -446,7 +531,7 @@ export function deriveBom(plan: Plan, scope: BomScope = 'all'): BomLine[] { const multi = plan.partitions.length > 1 const bom = new BomBuilder(scope) for (const partition of plan.partitions) { - addPartition(multi ? bom.inPartition(partition.name) : bom, partition) + addPartition(multi ? bom.inPartition(partition.name) : bom, plan, partition) } const lines = bom.build() return [...lines, ...spareLines(lines, plan.sparesPerLine)] @@ -459,7 +544,7 @@ export function deriveBomByPartition( ): { partition: Partition; lines: BomLine[] }[] { return plan.partitions.map((partition) => { const bom = new BomBuilder(scope) - addPartition(bom, partition) + addPartition(bom, plan, partition) return { partition, lines: bom.build() } }) } diff --git a/src/derive/controlPlane.test.ts b/src/derive/controlPlane.test.ts new file mode 100644 index 0000000..9fd0137 --- /dev/null +++ b/src/derive/controlPlane.test.ts @@ -0,0 +1,73 @@ +import { describe, expect, it } from 'vitest' +import { createEmptyPlan, defaultPartition } from '../model/defaults' +import type { Plan } from '../model/plan' +import { + controlPlaneFootprint, + controlPlaneHost, + controlPlaneLeafCount, + controlPlaneSwitchPorts, + hasOnPremControlPlane, + hasOwnRack, + inCentralRack, +} from './controlPlane' + +/** A plan with an on-prem control plane, `patch` applied on top. */ +function onPrem(patch: Partial = {}): Plan { + const plan = createEmptyPlan() + plan.controlPlane = { ...plan.controlPlane, hosting: 'on-prem', ...patch } + return plan +} + +describe('controlPlane', () => { + it('has no hardware for a managed cluster', () => { + const plan = createEmptyPlan() + expect(plan.controlPlane.hosting).toBe('kaas') + expect(controlPlaneHost(plan)).toBeUndefined() + expect(hasOnPremControlPlane(plan)).toBe(false) + expect(inCentralRack(plan, plan.partitions[0])).toBe(false) + expect(controlPlaneLeafCount(plan, plan.partitions[0])).toBe(0) + }) + + it('hosts on-prem nodes in the first partition unless one is chosen', () => { + const plan = onPrem() + plan.partitions.push(defaultPartition('Partition 2')) + expect(controlPlaneHost(plan)?.id).toBe(plan.partitions[0].id) + + plan.controlPlane.partitionId = plan.partitions[1].id + expect(controlPlaneHost(plan)?.id).toBe(plan.partitions[1].id) + expect(inCentralRack(plan, plan.partitions[1])).toBe(true) + expect(inCentralRack(plan, plan.partitions[0])).toBe(false) + }) + + it('falls back to the first partition when the chosen one is gone', () => { + const plan = onPrem({ partitionId: 'removed' }) + expect(controlPlaneHost(plan)?.id).toBe(plan.partitions[0].id) + }) + + it('puts the leaves in the host partition only for an own rack', () => { + const central = onPrem() + expect(hasOwnRack(central, central.partitions[0])).toBe(false) + expect(controlPlaneLeafCount(central, central.partitions[0])).toBe(0) + + const own = onPrem({ placement: 'own-rack' }) + expect(hasOwnRack(own, own.partitions[0])).toBe(true) + expect(inCentralRack(own, own.partitions[0])).toBe(false) + expect(controlPlaneLeafCount(own, own.partitions[0])).toBe(2) + }) + + it('counts switch ports as breakout groups at 25G and one to one at 100G', () => { + // 3 nodes × 2 ports = 6 ports of 25G → 2 breakout groups of four. + expect(controlPlaneSwitchPorts(onPrem().controlPlane)).toBe(2) + expect(controlPlaneSwitchPorts(onPrem({ uplink: '2x100G' }).controlPlane)).toBe(6) + expect(controlPlaneSwitchPorts(onPrem({ nodeCount: 2 }).controlPlane)).toBe(1) + }) + + it('sums height and power of the nodes', () => { + // SYS-121H-TNR: 1U, 500 W each. + expect(controlPlaneFootprint(onPrem().controlPlane)).toEqual({ units: 3, watts: 1500 }) + expect(controlPlaneFootprint(onPrem({ nodeCount: 0 }).controlPlane)).toEqual({ + units: 0, + watts: 0, + }) + }) +}) diff --git a/src/derive/controlPlane.ts b/src/derive/controlPlane.ts new file mode 100644 index 0000000..55cb3cb --- /dev/null +++ b/src/derive/controlPlane.ts @@ -0,0 +1,71 @@ +import { catalog } from '../model/catalog' +import type { ControlPlane, Partition, Plan } from '../model/plan' + +// Where the control plane's hardware lands, derived from Plan.controlPlane. +// This is the single source of the placement rules: the BOM, the rack +// elevations, the topology graph and validation all ask here instead of +// re-reading `hosting` and `placement` themselves. +// +// A KaaS control plane has no hardware at all, so every function below +// reports "nothing" for it. On-prem nodes either join the host partition's +// central rack, attaching to its exit switches, or sit in a control-plane +// rack of their own behind a leaf pair that uplinks to the spines like any +// compute rack's leaves. + +/** The partition whose site hosts the on-prem nodes: the configured one, + * or the first in the plan when unset or gone. Undefined for KaaS and for + * a plan without partitions. */ +export function controlPlaneHost(plan: Plan): Partition | undefined { + if (plan.controlPlane.hosting !== 'on-prem') return undefined + const byId = plan.partitions.find((p) => p.id === plan.controlPlane.partitionId) + return byId ?? plan.partitions[0] +} + +/** Is the control plane hardware this plan has to order? */ +export function hasOnPremControlPlane(plan: Plan): boolean { + return plan.controlPlane.hosting === 'on-prem' && plan.controlPlane.nodeCount > 0 +} + +/** Does the on-prem control plane live in this partition's central rack? */ +export function inCentralRack(plan: Plan, partition: Partition): boolean { + const cp = plan.controlPlane + return ( + hasOnPremControlPlane(plan) && + cp.placement === 'central-rack' && + controlPlaneHost(plan)?.id === partition.id + ) +} + +/** Does this partition hold the separate control-plane rack? */ +export function hasOwnRack(plan: Plan, partition: Partition): boolean { + const cp = plan.controlPlane + return ( + hasOnPremControlPlane(plan) && + cp.placement === 'own-rack' && + controlPlaneHost(plan)?.id === partition.id + ) +} + +/** Leaves of the control-plane rack in this partition, 0 when it has none. + * They terminate spine uplinks exactly like a compute rack's leaves, so + * every spine-side count has to include them. */ +export function controlPlaneLeafCount(plan: Plan, partition: Partition): number { + return hasOwnRack(plan, partition) ? plan.controlPlane.rack.leafCount : 0 +} + +/** Ports the control-plane nodes take on the switch they attach to: two + * 100G ports per node at 2x100G, one 100G port per started group of four + * 25G ports at 2x25G (4x25G breakout, as for server uplinks). */ +export function controlPlaneSwitchPorts(cp: ControlPlane): number { + const ports = 2 * cp.nodeCount + return cp.uplink === '2x100G' ? ports : Math.ceil(ports / 4) +} + +/** Height units and power the nodes add to the rack they sit in. */ +export function controlPlaneFootprint(cp: ControlPlane): { units: number; watts: number } { + const item = catalog[cp.nodeModelId] + return { + units: cp.nodeCount * (item?.heightUnits ?? 1), + watts: cp.nodeCount * (item?.powerWatts ?? 0), + } +} diff --git a/src/derive/rackLayout.ts b/src/derive/rackLayout.ts index 1dd84c3..82e9613 100644 --- a/src/derive/rackLayout.ts +++ b/src/derive/rackLayout.ts @@ -7,13 +7,14 @@ import { type Rack, } from '../model/plan' import { chassisCount } from './bom' +import { hasOwnRack, inCentralRack } from './controlPlane' // Derives physical rack elevations (which device sits in which height // units) from the Plan. Like the BOM, the layout is always computed, never // stored. Devices fill each rack from the top: network gear first, then // server chassis — the classic ToR arrangement. -export type SlotKind = 'network' | 'mgmt' | 'server' | 'storage' +export type SlotKind = 'network' | 'mgmt' | 'server' | 'storage' | 'control-plane' export interface RackSlot { label: string @@ -199,6 +200,15 @@ export function deriveRackLayout(plan: Plan): PartitionRackLayout[] { ...device('Storage leaf', fabric.storageLeafModelId, fabric.storageLeafCount, 'storage'), ...device('Mgmt spine', fabric.mgmt.spineModelId, mgmtCount, 'mgmt'), ...device('Mgmt server', fabric.mgmt.serverModelId, mgmtCount, 'mgmt'), + // On-prem control-plane nodes, when they share the central rack. + ...(inCentralRack(plan, partition) + ? device( + 'Control plane node', + plan.controlPlane.nodeModelId, + plan.controlPlane.nodeCount, + 'control-plane', + ) + : []), ] const central: RackElevation = { id: `${partition.id}/central`, @@ -222,10 +232,32 @@ export function deriveRackLayout(plan: Plan): PartitionRackLayout[] { })), ) + // The control-plane rack, when the nodes get one of their own: its + // leaf pair and mgmt leaf on top, the nodes below, like a compute rack. + const cp = plan.controlPlane + const controlPlaneRacks: RackElevation[] = hasOwnRack(plan, partition) + ? [ + { + id: `${partition.id}/control-plane`, + name: cp.rack.name, + heightUnits: cp.rack.heightUnits, + maxPowerWatts: cp.rack.maxPowerWatts, + ...place( + [ + ...device('Mgmt leaf', fabric.mgmt.leafModelId, fabric.mgmt.leafPerRack, 'mgmt'), + ...device('Leaf', cp.rack.leafModelId, cp.rack.leafCount, 'network'), + ...device('Control plane node', cp.nodeModelId, cp.nodeCount, 'control-plane'), + ], + cp.rack.heightUnits, + ), + }, + ] + : [] + return { partitionId: partition.id, partitionName: partition.name, - racks: [central, ...racks], + racks: [central, ...racks, ...controlPlaneRacks], } }) } diff --git a/src/derive/topology.test.ts b/src/derive/topology.test.ts index dd31f96..b60edf0 100644 --- a/src/derive/topology.test.ts +++ b/src/derive/topology.test.ts @@ -1,7 +1,7 @@ import { describe, expect, it } from 'vitest' -import { createEmptyPlan, withRackKind } from '../model/defaults' +import { createEmptyPlan, defaultPartition, withRackKind } from '../model/defaults' import type { Plan } from '../model/plan' -import { deriveTopology, filterTopology } from './topology' +import { CONTROL_PLANE_RACK_ID, deriveTopology, filterTopology } from './topology' describe('deriveTopology', () => { it('derives the central rack and compute racks for the default plan', () => { @@ -235,3 +235,63 @@ describe('filterTopology', () => { expect(g.links.every((l) => !l.to.includes('leaf') && !l.from.includes('leaf'))).toBe(true) }) }) + +describe('control plane in the topology', () => { + function onPrem(patch: Partial = {}) { + const plan = createEmptyPlan() + plan.controlPlane = { ...plan.controlPlane, hosting: 'on-prem', ...patch } + return plan + } + + it('hangs a managed control plane off the routers of every partition', () => { + const plan = createEmptyPlan() + plan.partitions.push(defaultPartition('Partition 2')) + const graph = deriveTopology(plan) + for (const partition of graph.partitions) { + const cp = partition.controlPlane! + expect(cp.managed).toBe(true) + expect(cp.node.kind).toBe('control-plane') + // Titled by what it is; the subtitle says how it is hosted. + expect(cp.node.label).toBe('Control plane') + expect(cp.node.sublabel).toBe('managed Kubernetes') + const targets = graph.links.filter((l) => l.from === cp.node.id).map((l) => l.to) + expect(targets.sort()).toEqual(partition.central.routers.map((n) => n.id).sort()) + } + // Two partitions, two separate capsules. + expect(graph.partitions[0].controlPlane!.node.id).not.toBe( + graph.partitions[1].controlPlane!.node.id, + ) + }) + + it('links on-prem nodes in the central rack to the exits', () => { + const graph = deriveTopology(onPrem()) + const partition = graph.partitions[0] + const cp = partition.controlPlane! + expect(cp.managed).toBe(false) + expect(cp.node.sublabel).toBe('3 × SYS-121H-TNR') + const links = graph.links.filter((l) => l.from === cp.node.id) + expect(links.map((l) => l.to).sort()).toEqual(partition.central.exits.map((n) => n.id).sort()) + expect(links.every((l) => l.network === 'production' && l.speed === '25G')).toBe(true) + }) + + it('draws an own rack with leaves uplinked to the spines', () => { + const graph = deriveTopology(onPrem({ placement: 'own-rack' })) + const partition = graph.partitions[0] + expect(partition.controlPlane).toBeUndefined() + const rack = partition.racks.find((r) => r.id === CONTROL_PLANE_RACK_ID)! + expect(rack.name).toBe('Control plane rack') + expect(rack.leaves).toHaveLength(2) + expect(rack.serverGroups[0].kind).toBe('control-plane') + const uplinks = graph.links.filter((l) => rack.leaves.some((n) => n.id === l.from)) + expect(uplinks).toHaveLength(4) // 2 leaves × 2 spines + const mgmt = graph.links.filter((l) => rack.mgmtLeaves.some((n) => n.id === l.from)) + expect(mgmt).toHaveLength(2) // 1 mgmt leaf × 2 mgmt spines + }) + + it('drops the control plane in management mode, keeps it in central mode', () => { + const graph = deriveTopology(onPrem()) + expect(filterTopology(graph, 'management').partitions[0].controlPlane).toBeUndefined() + expect(filterTopology(graph, 'central').partitions[0].controlPlane).toBeDefined() + expect(filterTopology(graph, 'production').partitions[0].controlPlane).toBeDefined() + }) +}) diff --git a/src/derive/topology.ts b/src/derive/topology.ts index 0430511..73e91e1 100644 --- a/src/derive/topology.ts +++ b/src/derive/topology.ts @@ -1,4 +1,5 @@ import { catalog } from '../model/catalog' +import { controlPlaneHost } from './controlPlane' import { physicalRacks, type RackPosition } from './rackLayout' import { mgmtDeviceCount, @@ -25,6 +26,7 @@ export type TopoNodeKind = | 'mgmt-leaf' | 'mgmt-server' | 'server-group' + | 'control-plane' | 'external-network' export interface TopoNode { @@ -78,6 +80,11 @@ export interface TopoPartition { central: TopoCentralRack storageLeaves: TopoNode[] racks: TopoRack[] + /** The control plane, when this partition carries it: a KaaS cluster + * (`managed`, drawn as a capsule like an external network) or on-prem + * nodes in the central rack. On-prem nodes in a rack of their own are a + * TopoRack in `racks` instead, so they lay out like any other rack. */ + controlPlane?: TopoControlPlane /** External networks attached to this partition, at the routers, the * exits or the storage leaves (see deriveTopology). A network that * attaches to every partition appears once per partition, so each is @@ -85,6 +92,13 @@ export interface TopoPartition { externalNetworks: TopoNode[] } +/** The control plane as the diagram needs it: the node plus whether it is + * a managed cluster somewhere else (capsule) or hardware in this rack. */ +export interface TopoControlPlane { + node: TopoNode + managed: boolean +} + export interface TopologyGraph { partitions: TopoPartition[] links: TopoLink[] @@ -303,6 +317,106 @@ function derivePartition(partition: Partition, links: TopoLink[]): TopoPartition } } +/** The control plane's place in the graph. KaaS hangs off the routers of + * every partition (or their exits), the same way an external network + * does, because that is the connection the partitions need to it. On-prem + * nodes are hardware: in the central rack they attach to the exits, in a + * rack of their own they sit behind that rack's leaves. */ +const MANAGED_SUBLABEL = 'managed Kubernetes' + +function addControlPlane(plan: Plan, partitions: TopoPartition[], links: TopoLink[]): void { + const cp = plan.controlPlane + const label = 'Control plane' + + if (cp.hosting === 'kaas') { + for (const partition of partitions) { + // Titled by what it is, like every other node; the subtitle says how + // it is hosted, where the on-prem node box carries its hardware. + const node: TopoNode = { + id: `cp/${partition.id}`, + kind: 'control-plane', + label, + sublabel: MANAGED_SUBLABEL, + } + partition.controlPlane = { node, managed: true } + const attach = + partition.central.routers.length > 0 ? partition.central.routers : partition.central.exits + for (const device of attach) { + links.push({ from: node.id, to: device.id, count: 1, network: 'external' }) + } + } + return + } + + const hostId = controlPlaneHost(plan)?.id + const host = partitions.find((p) => p.id === hostId) + if (!host || cp.nodeCount === 0) return + // Kept short: the node box is as narrow as a switch box. + const sublabel = `${cp.nodeCount} × ${part(cp.nodeModelId)}` + + if (cp.placement === 'central-rack') { + const node: TopoNode = { id: `cp/${host.id}`, kind: 'control-plane', label, sublabel } + host.controlPlane = { node, managed: false } + for (const exit of host.central.exits) { + links.push({ + from: node.id, + to: exit.id, + count: 2 * cp.nodeCount, + speed: cp.uplink === '2x100G' ? '100G' : '25G', + network: 'production', + }) + } + return + } + + // A rack of its own: leaves to every spine, mgmt leaf to every mgmt + // spine, and the nodes drawn as one box inside the rack. + const partition = controlPlaneHost(plan)! + const { mgmt } = partition.fabric + const prefix = `cp/${host.id}/` + const leaves = tier(`${prefix}leaf`, 'leaf', 'Leaf', cp.rack.leafModelId, cp.rack.leafCount) + const mgmtLeaves = tier( + `${prefix}mgmtleaf`, + 'mgmt-leaf', + 'Mgmt leaf', + mgmt.leafModelId, + mgmt.leafPerRack, + ) + for (const leaf of leaves) { + for (const spine of host.central.spines) { + links.push({ + from: leaf.id, + to: spine.id, + count: partition.fabric.leafSpineLinks, + speed: '100G', + network: 'production', + }) + } + } + for (const mgmtLeaf of mgmtLeaves) { + for (const mgmtSpine of host.central.mgmtSpines) { + links.push({ + from: mgmtLeaf.id, + to: mgmtSpine.id, + count: 1, + speed: '1G', + network: 'management', + }) + } + } + host.racks.push({ + id: CONTROL_PLANE_RACK_ID, + name: cp.rack.name, + leaves, + mgmtLeaves, + serverGroups: [{ id: `${prefix}nodes`, kind: 'control-plane', label, sublabel }], + }) +} + +/** Rack id of the separate control-plane rack in the graph; navigation + * maps it to the control plane section instead of a plan rack. */ +export const CONTROL_PLANE_RACK_ID = 'control-plane' + /** Whether an external network node hangs off the partition's storage * leaves rather than off its routers or exits. The single source of that * rule: the derivation links it there, the diagram draws it there. */ @@ -313,6 +427,7 @@ export function attachesAtStorageLeaves(partition: TopoPartition, node: TopoNode export function deriveTopology(plan: Plan): TopologyGraph { const links: TopoLink[] = [] const partitions = plan.partitions.map((partition) => derivePartition(partition, links)) + addControlPlane(plan, partitions, links) // External networks attach in their partition (or in every partition when // unset), as one node per partition they attach to. @@ -365,10 +480,11 @@ function nodeVisible(node: TopoNode, mode: TopologyMode): boolean { /** The subgraph for a view mode: nodes of the other network are dropped, * compute racks and storage are dropped in central mode, external - * networks in management mode, and links are kept only when both ends - * remain and belong to the shown network. An external network whose - * attachment points all fell away (a storage network on storage leaves, - * in central mode) goes with them instead of floating unconnected. */ + * networks and the control plane in management mode, and links are kept + * only when both ends remain and belong to the shown network. An external + * network whose attachment points all fell away (a storage network on + * storage leaves, in central mode) goes with them instead of floating + * unconnected. */ export function filterTopology(graph: TopologyGraph, mode: TopologyMode): TopologyGraph { const keep = (nodes: TopoNode[]) => nodes.filter((n) => nodeVisible(n, mode)) const partitions = graph.partitions.map((p): TopoPartition => ({ @@ -392,6 +508,10 @@ export function filterTopology(graph: TopologyGraph, mode: TopologyMode): Topolo serverGroups: keep(r.serverGroups), })), externalNetworks: mode === 'management' ? [] : p.externalNetworks, + // The control plane is production or external, never management; in + // central mode the capsule and the central-rack node both stay, + // because what they attach to stays too. + controlPlane: mode === 'management' ? undefined : p.controlPlane, })) const ids = new Set() for (const p of partitions) { @@ -403,6 +523,7 @@ export function filterTopology(graph: TopologyGraph, mode: TopologyMode): Topolo ...p.central.mgmtSpines, ...p.central.mgmtServers, ...p.storageLeaves, + ...(p.controlPlane ? [p.controlPlane.node] : []), ...p.racks.flatMap((r) => [...r.leaves, ...r.mgmtLeaves, ...r.serverGroups]), ]) { ids.add(n.id) diff --git a/src/derive/validate.test.ts b/src/derive/validate.test.ts index 39889d7..e02de5a 100644 --- a/src/derive/validate.test.ts +++ b/src/derive/validate.test.ts @@ -378,3 +378,64 @@ describe('issues fixed in the Advanced section', () => { expect(capacity!.target.field).toBeUndefined() }) }) + +describe('control plane validation', () => { + function onPrem(patch: Partial = {}): Plan { + const plan = basePlan() + plan.controlPlane = { ...plan.controlPlane, hosting: 'on-prem', ...patch } + return plan + } + + it('says nothing about a managed control plane, even without routers', () => { + const plan = basePlan() + plan.partitions[0].fabric.routerCount = 0 + plan.externalNetworks = [] + const own = validatePlan(plan).filter((i) => i.target.section === 'control-plane') + expect(own).toEqual([]) + }) + + it('reports an on-prem control plane without nodes as an error', () => { + const issues = validatePlan(onPrem({ nodeCount: 0 })).filter( + (i) => i.target.section === 'control-plane', + ) + expect(issues).toHaveLength(1) + expect(issues[0].severity).toBe('error') + expect(issues[0].message).toContain('no nodes') + }) + + it('warns below three nodes, and is happy with three', () => { + const two = validatePlan(onPrem({ nodeCount: 2 })).filter( + (i) => i.target.section === 'control-plane', + ) + expect(two).toHaveLength(1) + expect(two[0].severity).toBe('warning') + expect(two[0].message).toContain('quorum') + + expect(validatePlan(onPrem()).filter((i) => i.target.section === 'control-plane')).toEqual([]) + }) + + it('counts the nodes against the exit switch ports', () => { + // AS7726-32X: 32 ports of 100G. 2 spines + 2 × 2 routers = 6, so the + // control plane has to eat the rest for the budget to blow. + const plan = onPrem({ uplink: '2x100G', nodeCount: 14 }) + const exits = validatePlan(plan).find((i) => i.message.includes('Exit switch capacity')) + expect(exits).toBeDefined() + expect(exits!.message).toContain('for the control plane nodes') + expect(exits!.target.section).toBe('control-plane') + + // Same node count at 25G needs a quarter of the ports and fits. + const breakout = onPrem({ nodeCount: 14 }) + expect(validatePlan(breakout).some((i) => i.message.includes('Exit switch capacity'))).toBe( + false, + ) + }) + + it("counts an own rack's leaves against the spine ports", () => { + const plan = onPrem({ placement: 'own-rack' }) + plan.partitions[0].fabric.leafSpineLinks = 10 + plan.controlPlane.rack.leafCount = 3 + // The rack's own two leaves plus the control plane rack's three. + const spine = validatePlan(plan).find((i) => i.message.includes('Spine capacity')) + expect(spine?.message).toContain('5 leaves × 10') + }) +}) diff --git a/src/derive/validate.ts b/src/derive/validate.ts index fc9a288..d6ac61c 100644 --- a/src/derive/validate.ts +++ b/src/derive/validate.ts @@ -15,6 +15,7 @@ import { spineBandwidth, } from './bandwidth' import { deriveBom, mgmtUplinkSpeed, rackBmcPorts, spinePortsPerSpine } from './bom' +import { controlPlaneLeafCount, controlPlaneSwitchPorts, inCentralRack } from './controlPlane' import { validateIpPlan } from './ip/validateIp' import { deriveRackLayout, formatPower } from './rackLayout' @@ -35,8 +36,9 @@ import { deriveRackLayout, formatPower } from './rackLayout' export interface IssueTarget { partitionId?: string rackId?: string - /** Issues of another tab: 'ips' for the IP plan. */ - section?: 'ips' + /** Issues outside a partition: 'ips' for the IP plan tab, 'control-plane' + * for the plan's control plane section. */ + section?: 'ips' | 'control-plane' /** Field id within the section (e.g. "ipv4.shootPodCidr"). For a rack or * central rack section, 'advanced' means the fix is in its folded * Advanced section, which navigation then opens. */ @@ -251,7 +253,7 @@ function checkGpu(issues: Issue[], scope: Scope, group: ServerGroup): void { } } -function validatePartition(issues: Issue[], partition: Partition): void { +function validatePartition(issues: Issue[], plan: Plan, partition: Partition): void { const scope: Scope = { where: partition.name, target: { partitionId: partition.id }, @@ -315,9 +317,10 @@ function validatePartition(issues: Issue[], partition: Partition): void { // leaf, exit switch and (if present) superspine. const spine = catalog[fabric.spineModelId] if (spine && fabric.spineCount > 0) { - const leaves = partition.racks.reduce((n, r) => n + r.leafCount, 0) + const cpLeaves = controlPlaneLeafCount(plan, partition) + const leaves = partition.racks.reduce((n, r) => n + r.leafCount, 0) + cpLeaves const superspines = fabric.fabricType === 'leaf-spine-superspine' ? fabric.superspineCount : 0 - const needed = spinePortsPerSpine(partition) + const needed = spinePortsPerSpine(partition, cpLeaves) const available = portCount(spine, '100G') if (needed > available) { report( @@ -346,18 +349,22 @@ function validatePartition(issues: Issue[], partition: Partition): void { ) } - // Exit switch port budget: one 100G port per spine plus two per router. + // Exit switch port budget: one 100G port per spine, two per router and, + // when the control-plane nodes hang off the exits, their ports too (25G + // uplinks go through 4x25G breakout, as everywhere else). const exit = catalog[fabric.exitModelId] if (exit && fabric.exitSwitchCount > 0) { - const needed = fabric.spineCount + 2 * fabric.routerCount + const cpPorts = inCentralRack(plan, partition) ? controlPlaneSwitchPorts(plan.controlPlane) : 0 + const needed = fabric.spineCount + 2 * fabric.routerCount + cpPorts const available = portCount(exit, '100G') if (needed > available) { + const cpWhy = cpPorts > 0 ? `, ${cpPorts} for the control plane nodes` : '' report( issues, - scope, + cpPorts > 0 ? { where: 'Control plane', target: { section: 'control-plane' } } : scope, 'error', `Exit switch capacity exceeded: each exit needs ${needed} 100G ports ` + - `(${fabric.spineCount} spines, 2 × ${fabric.routerCount} routers), ` + + `(${fabric.spineCount} spines, 2 × ${fabric.routerCount} routers${cpWhy}), ` + `but ${itemLabel(fabric.exitModelId)} has ${available}.`, ) } @@ -430,9 +437,38 @@ function checkAvailability(issues: Issue[], plan: Plan): void { } } +/** The control plane's own checks. Where it runs is free (the deployment + * guide is explicit that a managed cluster is fine), so only the on-prem + * node count is checked here; the ports those nodes take are part of the + * exit switch budget in validatePartition. */ +function validateControlPlane(issues: Issue[], plan: Plan): void { + const cp = plan.controlPlane + if (cp.hosting !== 'on-prem') return + const scope = { where: 'Control plane', target: { section: 'control-plane' as const } } + if (cp.nodeCount === 0) { + report( + issues, + scope, + 'error', + 'The control plane runs on-prem but has no nodes: it needs a Kubernetes cluster to run on.', + ) + return + } + if (cp.nodeCount < 3) { + report( + issues, + scope, + 'warning', + `Only ${cp.nodeCount} control plane node(s): etcd needs three for quorum, ` + + 'so the cluster cannot survive the loss of one.', + ) + } +} + export function validatePlan(plan: Plan): Issue[] { const issues: Issue[] = [] - for (const partition of plan.partitions) validatePartition(issues, partition) + for (const partition of plan.partitions) validatePartition(issues, plan, partition) + validateControlPlane(issues, plan) // Physical height: no rack may hold more units than it has. for (const layout of deriveRackLayout(plan)) { diff --git a/src/model/defaults.ts b/src/model/defaults.ts index 24ddbcc..f20d63f 100644 --- a/src/model/defaults.ts +++ b/src/model/defaults.ts @@ -1,5 +1,6 @@ import { defaultIpPlan } from './ipPlan' import { SCHEMA_VERSION } from './migrate' +import { defaultControlPlane } from './plan' import type { FabricConfig, Partition, Plan, Rack, RackDefaults } from './plan' function id(): string { @@ -144,6 +145,7 @@ export function createEmptyPlan(): Plan { partitions: [defaultPartition('Partition 1')], sparesPerLine: 2, ipPlan: defaultIpPlan(), + controlPlane: defaultControlPlane(), externalNetworks: [ { id: id(), diff --git a/src/model/plan.ts b/src/model/plan.ts index 97c1217..f65856d 100644 --- a/src/model/plan.ts +++ b/src/model/plan.ts @@ -149,6 +149,61 @@ export const ExternalNetworkSchema = z.object({ }) export type ExternalNetwork = z.infer +// Where the metal-stack control plane runs. It is a Kubernetes cluster +// carrying metal-api, masterdata-api, go-ipam, NSQ, RethinkDB, their +// backup-restore sidecars and an ingress controller. The deployment guide +// leaves the location open ("it does not matter where your control plane +// Kubernetes cluster is located, you can of course use a cluster managed +// by a hyperscaler") and asks only that the partitions can reach it, so a +// plan says which of the two it is: 'kaas', a managed service that orders +// no hardware, or 'on-prem', dedicated nodes this plan has to buy. One +// control plane serves the whole installation, so it is plan-level rather +// than part of a Partition. Its nodes run the Kubernetes cluster and are +// never metal-stack-managed machines: they stay out of the machine pool +// (planNodes) and the hardware compatibility list does not apply to them. +export const ControlPlaneHostingSchema = z.enum(['kaas', 'on-prem']) +export type ControlPlaneHosting = z.infer + +/** 'central-rack': the nodes join the host partition's central rack and + * attach to its exit switches. 'own-rack': a dedicated rack with its own + * leaf pair, uplinked to the spines like a compute rack. */ +export const ControlPlanePlacementSchema = z.enum(['central-rack', 'own-rack']) +export type ControlPlanePlacement = z.infer + +/** Geometry of the control-plane rack, mirroring what a compute rack keeps + * in its Advanced section. */ +export const ControlPlaneRackSchema = z.object({ + name: z.string().default('Control plane rack'), + heightUnits: z.number().int().positive().default(42), + maxPowerWatts: z.number().int().positive().default(12000), + leafModelId: z.string().default('switch-as7726'), + leafCount: z.number().int().min(0).default(2), +}) +export type ControlPlaneRack = z.infer + +export const ControlPlaneSchema = z.object({ + hosting: ControlPlaneHostingSchema.default('kaas'), + /** Nodes of the Kubernetes cluster (on-prem); three for etcd quorum. */ + nodeCount: z.number().int().min(0).default(3), + nodeModelId: z.string().default('server-mgmt-121h'), + uplink: UplinkSpeedSchema.default('2x25G'), + /** Partition whose site hosts the nodes; empty means the first one. */ + partitionId: z.string().default(''), + placement: ControlPlanePlacementSchema.default('central-rack'), + rack: ControlPlaneRackSchema.default({ + name: 'Control plane rack', + heightUnits: 42, + maxPowerWatts: 12000, + leafModelId: 'switch-as7726', + leafCount: 2, + }), +}) +export type ControlPlane = z.infer + +export function defaultControlPlane(): ControlPlane { + return ControlPlaneSchema.parse({}) +} + export const PlanSchema = z.object({ schemaVersion: z.literal(SCHEMA_VERSION), id: z.string(), @@ -161,5 +216,7 @@ export const PlanSchema = z.object({ sparesPerLine: z.number().int().min(0).default(2), /** IPv4/IPv6 address plan (IPs tab). */ ipPlan: IpPlanSchema.default(defaultIpPlan), + /** Where the metal-stack control plane runs (one per installation). */ + controlPlane: ControlPlaneSchema.default(defaultControlPlane), }) export type Plan = z.infer diff --git a/src/model/templates.ts b/src/model/templates.ts index 2863f60..7842ed2 100644 --- a/src/model/templates.ts +++ b/src/model/templates.ts @@ -86,8 +86,14 @@ export const templates: PlanTemplate[] = [ id: 'production', name: 'Production', description: - 'One partition, redundant management network, two rack groups with 224 workers and 3 storage servers.', - build: () => plan('Production', [productionPartition('Partition 1')]), + 'One partition, redundant management network, two rack groups with 224 workers and 3 storage servers, control plane on three on-prem nodes.', + build: () => { + const p = plan('Production', [productionPartition('Partition 1')]) + // A production install that owns its control plane: three nodes in + // the partition's central rack. + p.controlPlane = { ...p.controlPlane, hosting: 'on-prem', nodeCount: 3 } + return p + }, }, { id: 'three-partitions', diff --git a/src/store/planStore.ts b/src/store/planStore.ts index 2fc955e..f417327 100644 --- a/src/store/planStore.ts +++ b/src/store/planStore.ts @@ -14,6 +14,8 @@ import { type Rack, type RackDefaults, type ServerGroup, + type ControlPlane, + type ControlPlaneRack, } from '../model/plan' export type View = 'plan' | 'topology' | 'racks' | 'ips' | 'bom' @@ -24,6 +26,8 @@ interface PlannerState { setActiveView: (view: View) => void setPlanName: (name: string) => void setSparesPerLine: (sparesPerLine: number) => void + patchControlPlane: (patch: Partial) => void + patchControlPlaneRack: (patch: Partial) => void patchIpFamily: (family: IpFamilyKey, patch: Partial) => void patchIpInfra: (patch: Partial) => void setIpv6Enabled: (enabled: boolean) => void @@ -96,6 +100,19 @@ export const usePlanStore = create()( }, }), })), + patchControlPlane: (patch) => + set((s) => ({ + plan: touched(s.plan, { controlPlane: { ...s.plan.controlPlane, ...patch } }), + })), + patchControlPlaneRack: (patch) => + set((s) => ({ + plan: touched(s.plan, { + controlPlane: { + ...s.plan.controlPlane, + rack: { ...s.plan.controlPlane.rack, ...patch }, + }, + }), + })), patchIpInfra: (patch) => set((s) => ({ plan: touched(s.plan, { diff --git a/src/views/PlanView.tsx b/src/views/PlanView.tsx index e0f0f7a..d4d073b 100644 --- a/src/views/PlanView.tsx +++ b/src/views/PlanView.tsx @@ -5,6 +5,7 @@ import { exportPlanJson, importPlanJson } from '../io/json' import { downloadText } from '../io/download' import { usePlanStore } from '../store/planStore' import { useToastStore } from '../store/toastStore' +import ControlPlaneSection from './plan/ControlPlaneSection' import ExternalNetworksSection from './plan/ExternalNetworksSection' import CentralRackSection from './plan/CentralRackSection' import RackSection from './plan/RackSection' @@ -175,6 +176,9 @@ export default function PlanView() { Add partition + {/* Plan-level, like the external networks, and rarely touched: it + sits with them at the end rather than above the hardware. */} + diff --git a/src/views/RackLayoutView.tsx b/src/views/RackLayoutView.tsx index dbd93aa..248578f 100644 --- a/src/views/RackLayoutView.tsx +++ b/src/views/RackLayoutView.tsx @@ -25,6 +25,11 @@ const SLOT_STYLE: Record = { 'mgmt-leaf': NetworkSwitch, 'mgmt-server': ServerCog, 'server-group': Server, + 'control-plane': MetalStack, 'external-network': Globe, } @@ -147,6 +149,7 @@ export const SLOT_ICON: Record = { mgmt: ServerCog, server: Server, storage: HardDrive, + 'control-plane': MetalStack, } /** A decorative icon at text size (16 px by default). */ diff --git a/src/views/plan/ControlPlaneSection.tsx b/src/views/plan/ControlPlaneSection.tsx new file mode 100644 index 0000000..530ce81 --- /dev/null +++ b/src/views/plan/ControlPlaneSection.tsx @@ -0,0 +1,182 @@ +import { itemLabel, serversForUsage, switchesForRole } from '../../model/catalog' +import type { ControlPlane, Plan } from '../../model/plan' +import { controlPlaneFootprint } from '../../derive/controlPlane' +import { formatPower } from '../../derive/rackLayout' +import type { Issue } from '../../derive/validate' +import { usePlanStore } from '../../store/planStore' +import { DOCS } from './docs' +import { NumberField, SelectField } from './fields' +import InfoBubble from './InfoBubble' +import IssueBadges from './IssueBadges' +import { CONTROL_PLANE_ANCHOR } from './navigate' +import { Icon, SECTION_ICON } from '../icons' +import { optionLabel } from './options' + +/** Where the metal-stack control plane runs. The deployment guide leaves + * the location open and asks only that the partitions can reach it, so + * this section is a hosting choice first: a managed Kubernetes service + * orders nothing, on-prem nodes are hardware in a rack. */ +export default function ControlPlaneSection({ plan, issues }: { plan: Plan; issues: Issue[] }) { + const patch = usePlanStore((s) => s.patchControlPlane) + const patchRack = usePlanStore((s) => s.patchControlPlaneRack) + const cp = plan.controlPlane + const own = issues.filter((i) => i.target.section === 'control-plane') + const hasErrors = own.some((i) => i.severity === 'error') + const onPrem = cp.hosting === 'on-prem' + const ownRack = onPrem && cp.placement === 'own-rack' + const footprint = controlPlaneFootprint(cp) + + const summary = onPrem + ? `${cp.nodeCount} × ${itemLabel(cp.nodeModelId)} · ${footprint.units} U, ${formatPower(footprint.watts)}` + : 'managed Kubernetes' + + return ( +
+
+

+ + Control plane for metal-stack.io installation + +

+ + + + {summary} + + +
+ +
+ patch({ hosting: v as ControlPlane['hosting'] })} + /> + {onPrem && ( + <> + p.id === cp.partitionId) ? cp.partitionId : ''} + options={[ + { value: '', label: plan.partitions[0]?.name ?? 'First partition' }, + ...plan.partitions.map((p) => ({ value: p.id, label: p.name })), + ]} + onChange={(v) => patch({ partitionId: v })} + /> + patch({ placement: v as ControlPlane['placement'] })} + /> + ({ + value: i.id, + label: optionLabel(i), + }))} + onChange={(v) => patch({ nodeModelId: v })} + /> + patch({ nodeCount: n })} + /> + patch({ uplink: v as ControlPlane['uplink'] })} + /> + + )} +
+ + {ownRack && ( +
+ + Advanced · the control plane rack + +
+ + ({ value: i.id, label: optionLabel(i) }))} + onChange={(v) => patchRack({ leafModelId: v })} + /> + patchRack({ leafCount: n })} + /> + patchRack({ heightUnits: n })} + /> + patchRack({ maxPowerWatts: Math.round(kw * 1000) })} + /> +
+
+ )} +
+ ) +} diff --git a/src/views/plan/SidePanel.tsx b/src/views/plan/SidePanel.tsx index dccd359..6ad9039 100644 --- a/src/views/plan/SidePanel.tsx +++ b/src/views/plan/SidePanel.tsx @@ -1,7 +1,7 @@ import { formatBandwidth, formatRatio, rackBandwidth } from '../../derive/bandwidth' import { deriveBom } from '../../derive/bom' import { formatTally, planNodes } from '../../derive/nodes' -import { deriveRackLayout, formatPower, physicalRackCount } from '../../derive/rackLayout' +import { deriveRackLayout, formatPower } from '../../derive/rackLayout' import { deriveTopology, filterTopology } from '../../derive/topology' import type { Issue } from '../../derive/validate' import type { Plan } from '../../model/plan' @@ -54,6 +54,8 @@ function planTotals(plan: Plan) { } } return { + // Every physical rack the layout knows, the control-plane rack included. + rackCount: racks.length, usedU: racks.reduce((u, r) => u + r.usedU, 0), totalU: racks.reduce((u, r) => u + r.heightUnits, 0), powerWatts: racks.reduce((w, r) => w + r.powerWatts, 0), @@ -87,7 +89,7 @@ export default function SidePanel({ plan, issues }: { plan: Plan; issues: Issue[ : 0 const complete = bom.every((l) => lineTotal({ currency, prices }, l) !== undefined) // Physical racks, central racks included (a rack group counts as three). - const racks = plan.partitions.reduce((n, p) => n + physicalRackCount(p), 0) + const racks = totals.rackCount const graph = filterTopology(deriveTopology(plan), 'production') const hasTopology = graph.partitions.some( (p) => p.racks.length > 0 || p.central.spines.length > 0, @@ -104,6 +106,14 @@ export default function SidePanel({ plan, issues }: { plan: Plan; issues: Issue[ {plural(partitions, 'partition')}, {plural(racks, 'rack')},{' '} {plural(sumCategory(plan, 'server'), 'server chassis', 'server chassis')}

+

+ Control plane:{' '} + {plan.controlPlane.hosting === 'kaas' + ? 'managed Kubernetes' + : `${plural(plan.controlPlane.nodeCount, 'node')} on-prem, ${ + plan.controlPlane.placement === 'own-rack' ? 'own rack' : 'central rack' + }`} +

`fabric-${partitionId}` +export const CONTROL_PLANE_ANCHOR = 'control-plane' export const rackAnchor = (rackId: string) => `rack-${rackId}` /** Anchor of an IPs-tab field ("ipv4.shootPodCidr" → "ip-ipv4-shootPodCidr"). */ @@ -13,6 +15,10 @@ export const ipFieldAnchor = (field: string) => `ip-${field.replace(/\./g, '-')} export function anchorFor(target: IssueTarget): string | undefined { if (target.section === 'ips') return ipFieldAnchor(target.field ?? 'top') + if (target.section === 'control-plane') return CONTROL_PLANE_ANCHOR + // The control-plane rack is not a plan rack: its box in the diagram + // points at the control plane section. + if (target.rackId === CONTROL_PLANE_RACK_ID) return CONTROL_PLANE_ANCHOR if (target.rackId) return rackAnchor(target.rackId) if (target.partitionId) return fabricAnchor(target.partitionId) return undefined diff --git a/src/views/topology/Diagram.tsx b/src/views/topology/Diagram.tsx index 7306683..7886324 100644 --- a/src/views/topology/Diagram.tsx +++ b/src/views/topology/Diagram.tsx @@ -220,28 +220,33 @@ function layoutPartition( // superspines/exits, then spines) and management (mgmt servers over mgmt // spines, aligned to the bottom rows). const hasRouters = central.routers.length > 0 + // On-prem control-plane nodes in the central rack share the top row with + // the routers; a managed cluster is a capsule above the rack instead. + const cpBox = partition.controlPlane?.managed ? undefined : partition.controlPlane?.node + const row0 = [...central.routers, ...(cpBox ? [cpBox] : [])] const prodRow1W = rowWidth(central.superspines.length) + rowWidth(central.exits.length) + 56 - const prodW = Math.max( - prodRow1W, - rowWidth(central.spines.length), - rowWidth(central.routers.length), - ) + const prodW = Math.max(prodRow1W, rowWidth(central.spines.length), rowWidth(row0.length)) const mgmtW = Math.max(rowWidth(central.mgmtServers.length), rowWidth(central.mgmtSpines.length)) const columnGap = prodW > 0 && mgmtW > 0 ? COLUMN_GAP : 0 const centralInnerW = prodW + columnGap + mgmtW const fabricW = Math.max(racksW + storageW, centralInnerW + 2 * RACK_PAD, 300) const cx = fabricW / 2 - // External networks sit above the central rack, centered over the exits. - const extH = centralNets.length > 0 ? EXT_H + 28 : 0 + // The networks that attach in the central rack, and a managed control + // plane, sit above it, centered over the exits. + const capsules = [ + ...centralNets, + ...(partition.controlPlane?.managed ? [partition.controlPlane.node] : []), + ] + const extH = capsules.length > 0 ? EXT_H + 28 : 0 const boxTop = y0 + 24 + extH const row0Y = boxTop + RACK_HEAD - const row1Y = hasRouters ? row0Y + NODE_H + ROW_GAP : row0Y + const row1Y = row0.length > 0 ? row0Y + NODE_H + ROW_GAP : row0Y const row2Y = row1Y + NODE_H + ROW_GAP const innerLeft = cx - centralInnerW / 2 const prodCx = innerLeft + prodW / 2 const mgmtCx = innerLeft + prodW + columnGap + mgmtW / 2 - if (hasRouters) placeRow(rects, central.routers, prodCx, row0Y) + if (row0.length > 0) placeRow(rects, row0, prodCx, row0Y) placeGroupedRow(rects, central.superspines, central.exits, prodCx, row1Y) placeRow(rects, central.spines, prodCx, row2Y) placeRow(rects, central.mgmtServers, mgmtCx, row1Y) @@ -253,14 +258,14 @@ function layoutPartition( h: row2Y + NODE_H + RACK_PAD - boxTop, } layout.boxes.push({ rect: boxRect, name: 'Central rack', target: { partitionId: partition.id } }) - if (centralNets.length > 0) { + if (capsules.length > 0) { const anchors = hasRouters ? central.routers : central.exits const exitRects = anchors.map((e) => rects.get(e.id)).filter((r): r is Rect => !!r) const ecx = exitRects.length > 0 ? exitRects.reduce((sum, r) => sum + r.x + r.w / 2, 0) / exitRects.length : prodCx - placeCapsules(rects, centralNets, ecx, y0 + 24) + placeCapsules(rects, capsules, ecx, y0 + 24) } // Compute racks and the storage box below. The physical racks of a @@ -378,10 +383,12 @@ function linkGeometry(from: Rect, to: Rect): { a: Pt; b: Pt; path: string } { return { a, b, path: `M ${a.x} ${a.y} C ${a.x} ${my}, ${b.x} ${my}, ${b.x} ${b.y}` } } -function NodeBox({ node, r }: { node: TopoNode; r: Rect }) { +/** `capsule` draws the node as a dashed pill: external networks, and a + * managed control plane, which is somewhere else just the same. */ +function NodeBox({ node, r, capsule }: { node: TopoNode; r: Rect; capsule?: boolean }) { const isServer = node.kind === 'server-group' || node.kind === 'mgmt-server' || node.kind === 'router' - const isExternal = node.kind === 'external-network' + const isExternal = capsule ?? node.kind === 'external-network' const isMgmt = node.kind === 'mgmt-spine' || node.kind === 'mgmt-leaf' || node.kind === 'mgmt-server' @@ -478,10 +485,19 @@ export default function Diagram({ ...p.central.mgmtSpines, ...p.central.mgmtServers, ...p.storageLeaves, + ...(p.controlPlane ? [p.controlPlane.node] : []), ...p.racks.flatMap((r) => [...r.leaves, ...r.mgmtLeaves, ...r.serverGroups]), ]), ] + // Nodes drawn as dashed pills above the central rack. + const capsuleIds = new Set( + graph.partitions.flatMap((p) => [ + ...p.externalNetworks.map((n) => n.id), + ...(p.controlPlane?.managed ? [p.controlPlane.node.id] : []), + ]), + ) + const showPartitionLabels = graph.partitions.length > 1 return ( @@ -570,7 +586,7 @@ export default function Diagram({ onMouseEnter={interactive ? () => setHover(node.id) : undefined} onMouseLeave={interactive ? () => setHover(null) : undefined} > - + ) : null })}