From 5bb0877571b329ac8c14f0b27a23086ce663060b Mon Sep 17 00:00:00 2001 From: Rohan Adwankar <39285979+RohanAdwankar@users.noreply.github.com> Date: Wed, 7 Oct 2026 16:05:14 -0700 Subject: [PATCH 1/2] Add Helm-derived deployment footprint to installation builder Generate resource catalogs from tracked Helm manifests and reuse module and inference selections. Show partial requests, limits and storage with missing-resource warnings, and validate composition and documentation builds. Signed-off-by: Rohan Adwankar <39285979+RohanAdwankar@users.noreply.github.com> --- .github/workflows/ci.yml | 3 + .github/workflows/docs.yml | 15 +++++ docs/_static/docs.css | 3 + docs/_static/install-builder.mjs | 89 ++++++++++++++++++++++++++++- docs/building.md | 4 +- docs/conf.py | 78 +++++++++++++++++++++++++ docs/installation_builder.md | 4 ++ docs/pyproject.toml | 1 + docs/uv.lock | 2 + tests/docs/install-builder.test.mjs | 50 +++++++++++++++- 10 files changed, 246 insertions(+), 3 deletions(-) diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml index 75bf3ff..19c8aae 100644 --- a/.github/workflows/ci.yml +++ b/.github/workflows/ci.yml @@ -32,6 +32,9 @@ jobs: with: node-version: 24 + - uses: astral-sh/setup-uv@94527f2e458b27549849d47d273a16bec83a01e9 + with: + version: 0.12.6 - name: Prepare chart test dependencies run: | set -euo pipefail diff --git a/.github/workflows/docs.yml b/.github/workflows/docs.yml index 6591651..a5496b6 100644 --- a/.github/workflows/docs.yml +++ b/.github/workflows/docs.yml @@ -26,8 +26,23 @@ jobs: - uses: astral-sh/setup-uv@94527f2e458b27549849d47d273a16bec83a01e9 # v7 with: version: 0.12.6 + - uses: actions/setup-node@49933ea5288caeca8642d1e84afbd3f7d6820020 + with: + node-version: 24 + - name: Prepare public chart footprint dependencies + run: | + chart=helm/mosaic-stack + mkdir -p "$chart/charts" + openshell_version="$(helm dependency list "$chart" | awk '$1 == "helm-chart" { print $2 }')" + iraop_version="$(helm dependency list "$chart" | awk '$1 == "iraop-stack" { print $2 }')" + helm pull oci://ghcr.io/nvidia/openshell/helm-chart --version "$openshell_version" --destination "$chart/charts" + stub="$(mktemp -d)" + printf 'apiVersion: v2\nname: iraop-stack\nversion: %s\n' "$iraop_version" >"$stub/Chart.yaml" + helm package "$stub" --destination "$chart/charts" - name: Build documentation run: uv run --project docs --locked sphinx-build -W --keep-going -b html -c docs . docs/_build/html + - name: Validate footprint composition + run: node --test tests/docs/*.test.mjs - uses: actions/upload-pages-artifact@7b1f4a764d45c48632c6b24a0339c27f5614fb0b # v4 with: path: docs/_build/html diff --git a/docs/_static/docs.css b/docs/_static/docs.css index 77958f7..1a98215 100644 --- a/docs/_static/docs.css +++ b/docs/_static/docs.css @@ -31,3 +31,6 @@ pre, .admonition, .search-button { border-radius: 0; } @media (max-width: 600px) { #install-command-builder .builder-settings, .builder-modules { grid-template-columns: 1fr; } } #install-command-builder .builder-settings > label:has(input[type=checkbox]) { flex-direction: row; align-items: center; grid-column: 1 / -1; } +.builder-sizing-table { overflow-x: auto; } +.builder-sizing table { width: 100%; min-width: 720px; font-size: .85rem; } +.builder-sizing th, .builder-sizing td { padding: .4rem; text-align: left; } diff --git a/docs/_static/install-builder.mjs b/docs/_static/install-builder.mjs index 92afde6..bef52f1 100644 --- a/docs/_static/install-builder.mjs +++ b/docs/_static/install-builder.mjs @@ -114,7 +114,13 @@ if (root) { const copy = document.createElement('button'); copy.type = 'button'; copy.textContent = 'Copy command'; const status = document.createElement('span'); status.setAttribute('role', 'status'); copy.addEventListener('click', async () => {try {await navigator.clipboard.writeText(code.textContent); status.textContent = 'Copied';} catch {status.textContent = 'Select the command text to copy it.';}}); - root.replaceChildren(form, heading, copy, status, output); + const sizing = document.createElement('section'); sizing.className = 'builder-sizing'; + let catalog; + root.replaceChildren(form, sizing, heading, copy, status, output); + fetch(new URL('sizing-data.json', import.meta.url)).then(response => { + if (!response.ok) throw new Error(`HTTP ${response.status}`); + return response.json(); + }).then(data => {catalog = data; update(false);}).catch(error => {sizing.textContent = `Deployment footprint unavailable: ${error.message}`;}); function update(renderSettings) { const enabled = enabledModules(state); for (const [name, input] of inputs) {input.checked = enabled.has(name); input.disabled = enabled.has(name) && !state.modules.includes(name);} @@ -144,6 +150,87 @@ if (root) { if (enabled.has('research')) notes.push('Create iraop-secrets with NVIDIA_API_KEY, NVIDIA_CHAT_API_KEY and IRAOP_API_KEY; prepare the corpus directory.'); if (state.inference !== 'external') notes.push('Verify GPU, memory and model cache requirements in the on-prem inference section.'); prerequisites.textContent = notes.join(' '); code.textContent = buildCommand(state); status.textContent = ''; + if (catalog) renderSizing(sizing, catalog, state, enabled); } update(true); } +export function selectedComponents(catalog, state, enabled) { + const selections = [...enabled].filter(name => name !== 'slurm'); + if (enabled.has('slurm')) selections.push(state.slurmBackend === 'vanilla' ? 'slurm-vanilla' : 'slurm'); + if (!state.sandboxInstalled) selections.push('sandbox'); + if (state.inference !== 'external') selections.push(state.inference); + const components = {}; + for (const group of [catalog.base, ...selections.map(name => catalog.variants[name])]) { + for (const [key, component] of Object.entries(group)) { + const previous = components[key]; + components[key] = {...component}; + for (const field of ['containers', 'initContainers']) { + if (component[field]) components[key][field] = Object.values(Object.fromEntries([...(previous?.[field] || []), ...component[field]].map(container => [container.name, container]))); + } + } + } + return Object.values(components); +} + +export function quantity(value, resource) { + const match = String(value).match(/^([0-9.]+)([a-zA-Z]*)$/); + if (!match) throw new Error(`Unsupported resource quantity: ${value}`); + const factors = resource === 'cpu' ? {'': 1, m: .001, u: .000001, n: .000000001} : {'': 1, Ki: 1024, Mi: 1024 ** 2, Gi: 1024 ** 3, Ti: 1024 ** 4, k: 1000, M: 1000 ** 2, G: 1000 ** 3, T: 1000 ** 4}; + if (!(match[2] in factors)) throw new Error(`Unsupported resource unit: ${match[2]}`); + return Number(match[1]) * factors[match[2]]; +} + +export function footprint(component, field, resource) { + const amount = container => quantity(container.resources[field]?.[resource] || 0, resource); + const running = (component.containers || []).reduce((sum, container) => sum + amount(container), 0); + const initializing = Math.max(0, ...(component.initContainers || []).map(amount)); + return Math.max(running, initializing) * component.count; +} + +export function renderSizing(panel, catalog, state, enabled) { + panel.replaceChildren(); + const title = document.createElement('h3'); title.textContent = 'Deployment footprint'; panel.append(title); + const note = document.createElement('p'); + note.textContent = `Rendered chart ${catalog.version}, source ${catalog.revision.slice(0, 12)}. Configured resources, not measured capacity. Only valid for this source version; channel, pinned-version and site overrides may differ.`; + panel.append(note); + const components = selectedComponents(catalog, state, enabled); + const table = document.createElement('table'); + const header = table.createTHead().insertRow(); + for (const text of ['Component', 'CPU request / limit', 'Memory request / limit', 'GPU request / limit', 'Storage']) {const cell = document.createElement('th'); cell.textContent = text; header.append(cell);} + const body = table.createTBody(); + const format = (amount, resource) => resource === 'memory' ? `${Number((amount / 1024 ** 3).toFixed(3))} GiB` : `${Number(amount.toFixed(3))}`; + const warnings = ['Dynamic execution sandboxes, existing inference services, observability backends and corpus storage are not included. GPU memory compatibility and workload concurrency require separate validation.']; + for (const component of components) { + const row = body.insertRow(); + row.insertCell().textContent = `${component.name}${component.kind === 'DaemonSet' ? ' (per matching node)' : component.kind === 'Job' ? ' (setup job)' : ''}`; + for (const resource of ['cpu', 'memory', 'nvidia.com/gpu']) { + const values = ['requests', 'limits'].map(field => { + const containers = [...(component.containers || []), ...(component.initContainers || [])]; + const amount = footprint(component, field, resource); + if (resource !== 'nvidia.com/gpu' && containers.some(container => container.resources[field]?.[resource] === undefined)) return amount ? `≥ ${format(amount, resource)}` : 'Unspecified'; + return format(amount, resource); + }); + row.insertCell().textContent = component.kind === 'PersistentVolumeClaim' ? '—' : values.join(' / '); + } + const storage = typeof component.storage === 'string' ? [component.storage] : component.storage || []; + row.insertCell().textContent = storage.length ? format(storage.reduce((sum, value) => sum + quantity(value, 'memory'), 0) * component.count, 'memory') : '—'; + if (Object.keys(component.nodeSelector || {}).length) warnings.push(`${component.name} placement: ${JSON.stringify(component.nodeSelector)}.`); + } + const subtotal = body.insertRow(); subtotal.insertCell().textContent = 'Known fixed-workload subtotal (partial)'; + const fixed = components.filter(component => ['Deployment', 'StatefulSet', 'Sandbox'].includes(component.kind)); + for (const resource of ['cpu', 'memory', 'nvidia.com/gpu']) { + subtotal.insertCell().textContent = ['requests', 'limits'].map(field => `≥ ${format(fixed.reduce((sum, component) => sum + footprint(component, field, resource), 0), resource)}`).join(' / '); + } + subtotal.insertCell().textContent = format(components.reduce((sum, component) => { + const storage = typeof component.storage === 'string' ? [component.storage] : component.storage || []; + return sum + storage.reduce((amount, value) => amount + quantity(value, 'memory'), 0) * component.count; + }, 0), 'memory'); + warnings.push('Subtotal excludes setup jobs and per-node collectors; unspecified requests/limits are not counted. PVC subtotal excludes pre-existing and dynamically allocated storage.'); + if (enabled.has('research') && !catalog.researchAvailable) warnings.push('Research dependency workloads are unavailable in this documentation build; the footprint is incomplete.'); + if (state.inference !== 'external') warnings.push('Model-server GPU requests must fit on one compatible node. Storage includes model-cache PVC capacity, not download size.'); + const summary = document.createElement('p'); + summary.textContent = `Known fixed-workload requests: ≥ ${format(fixed.reduce((sum, component) => sum + footprint(component, 'requests', 'cpu'), 0), 'cpu')} CPU cores; ≥ ${format(fixed.reduce((sum, component) => sum + footprint(component, 'requests', 'memory'), 0), 'memory')} memory; ≥ ${format(fixed.reduce((sum, component) => sum + footprint(component, 'requests', 'nvidia.com/gpu'), 0), 'nvidia.com/gpu')} GPUs. Partial footprint; see component details below.`; + const details = document.createElement('div'); details.className = 'builder-sizing-table'; details.append(table); + panel.append(summary, details); + const caveat = document.createElement('p'); caveat.textContent = warnings.join(' '); panel.append(caveat); +} diff --git a/docs/building.md b/docs/building.md index 010288e..de4d66a 100644 --- a/docs/building.md +++ b/docs/building.md @@ -7,7 +7,7 @@ The documentation site uses NVIDIA's Sphinx theme and renders the Markdown guide ## Build and preview -Install [uv](https://docs.astral.sh/uv/getting-started/installation/), then run from the repository root: +Install [uv](https://docs.astral.sh/uv/getting-started/installation/), Node.js and Helm, and prepare the chart dependencies with `helm dependency build helm/mosaic-stack` (private registry access is required for the research dependency). Then run from the repository root: ```bash uv run --project docs --locked sphinx-build -W --keep-going -b html -c docs . docs/_build/html @@ -18,6 +18,8 @@ Open `http://localhost:8080`. The build treats warnings as errors, including bro ## GitHub Pages +The footprint catalog is generated during each HTML build by rendering the tracked chart with the builder's own commands. The public documentation workflow does not authenticate to the private research registry; it uses an empty research dependency solely for rendering the public chart. Research workloads are consequently marked unavailable in that build. With an authorized real dependency installed locally, its manifests are included. Never deploy the documentation/test placeholder dependency. + The documentation workflow builds pull requests without publishing. After a change lands on `main`, it publishes the generated static site to GitHub Pages using the `github-pages` environment. A repository administrator must first enable **Settings → Pages → Build and deployment → Source: GitHub Actions**. The intended URL is `https://nvidia.github.io/AI-Factory-Operations-Agent/`. Until Pages is enabled and a deployment succeeds, that URL will not serve the documentation. No registry or inference credentials are needed for the documentation build. diff --git a/docs/conf.py b/docs/conf.py index 8f92dc5..527b566 100644 --- a/docs/conf.py +++ b/docs/conf.py @@ -1,6 +1,13 @@ # SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved. # SPDX-License-Identifier: Apache-2.0 +import json +import os +import subprocess +from pathlib import Path + +import yaml + project = "AI Factory Operations Agent" copyright = "2026, NVIDIA Corporation" extensions = ["myst_parser", "sphinx_copybutton"] @@ -35,3 +42,74 @@ def render_dco(app, docname, source): def setup(app): app.connect("source-read", render_dco) + app.connect("build-finished", write_catalog) + + +ROOT = Path(__file__).resolve().parents[1] +CHART = ROOT / "helm/mosaic-stack" +MODULES = ["kubernetes", "observability", "grafana", "bcm", "slurm", "diagnostics", "research", "terminal", "edit", "clusters"] + + +def render_footprint(state): + script = "import {buildCommand} from './docs/_static/install-builder.mjs'; console.log(JSON.stringify(buildCommand(JSON.parse(process.argv[1]))));" + command = json.loads(subprocess.check_output(["node", "--input-type=module", "-e", script, json.dumps(state)], cwd=ROOT, text=True)) + shell = 'trap \'[ -z "${MOSAIC_CHART_WORKDIR:-}" ] || rm -rf "$MOSAIC_CHART_WORKDIR"\' EXIT; helm() { if [ "$1" = pull ]; then ln -s "$CHART" "$MOSAIC_CHART_WORKDIR/mosaic-stack"; else printf \'%s\\0\' "$@"; fi; }; ' + arguments = subprocess.check_output(["bash", "-c", shell + command], env={**os.environ, "CHART": str(CHART)}, text=True).split("\0")[4:-1] + rendered = ["template", "mosaic", str(CHART)] + arguments = iter(arguments) + for argument in arguments: + if argument in {"--atomic", "--wait", "--devel"}: + continue + if argument in {"--timeout", "--version"}: + next(arguments) + elif argument == "-f": + next(arguments) + profile = "vllm-ultra-4gpu.yaml" if state["inference"] == "ultra" else "vllm-super-1gpu.yaml" + rendered.extend(["-f", str(CHART / "profiles" / profile)]) + else: + rendered.append(argument) + output = subprocess.check_output(["helm", *rendered], text=True) + components = {} + for document in yaml.safe_load_all(output): + if not document: + continue + kind = document.get("kind") + spec = document.get("spec", {}) + name = document["metadata"]["name"] + key = f"{kind}/{document['metadata'].get('namespace', 'mosaic')}/{name}" + if kind == "PersistentVolumeClaim": + components[key] = {"name": name, "kind": kind, "storage": spec["resources"]["requests"]["storage"], "count": 1} + elif kind in {"Deployment", "StatefulSet", "DaemonSet", "Job", "Sandbox"}: + pod = spec.get("template", {}).get("spec", {}) + if not pod.get("containers"): + continue + components[key] = { + "name": name, "kind": kind, "count": spec.get("replicas", 1), + "containers": [{"name": container["name"], "resources": container.get("resources", {})} for container in pod["containers"]], + "initContainers": [{"name": container["name"], "resources": container.get("resources", {})} for container in pod.get("initContainers", [])], + "storage": [claim["spec"]["resources"]["requests"]["storage"] for claim in spec.get("volumeClaimTemplates", [])], + "nodeSelector": pod.get("nodeSelector", {}), + } + return components + + +def generate(): + defaults = {"inference": "external", "modules": [], "sandboxInstalled": True, "baseUrl": "https://inference.example.com/v1", "model": "served-model", "bcmHead": "head.example.com", "evidenceRoot": "/slurm", "evidenceNode": "collector-node", "corpus": "/corpus"} + base = render_footprint(defaults) + variants = {} + selections = {"sandbox": {"sandboxInstalled": False}, "super": {"inference": "super"}, "ultra": {"inference": "ultra"}} + selections.update({module: {"modules": [module]} for module in MODULES}) + selections["slurm-vanilla"] = {"modules": ["slurm"], "slurmBackend": "vanilla"} + for name, selection in selections.items(): + rendered = render_footprint({**defaults, **selection}) + variants[name] = {key: value for key, value in rendered.items() if value != base.get(key)} + chart = yaml.safe_load((CHART / "Chart.yaml").read_text()) + revision = subprocess.check_output(["git", "rev-parse", "HEAD"], cwd=ROOT, text=True).strip() + research_available = any(key not in base and component["kind"] in {"Deployment", "StatefulSet"} for key, component in variants["research"].items()) + return {"version": chart["version"], "revision": revision, "base": base, "variants": variants, "researchAvailable": research_available} + + +def write_catalog(app, exception): + if exception is None and app.builder.format == "html": + target = Path(app.outdir) / "_static/sizing-data.json" + target.write_text(json.dumps(generate(), indent=2) + "\n") diff --git a/docs/installation_builder.md b/docs/installation_builder.md index 3be2f3f..5cd435f 100644 --- a/docs/installation_builder.md +++ b/docs/installation_builder.md @@ -2,6 +2,10 @@ Select your inference path and modules to generate an installation command. Complete [namespace and registry access setup](installation.md#1-create-the-namespace-and-registry-access) and create the indicated Secrets first. For an existing installation, follow the [upgrade instructions](installation.md#upgrade-an-existing-installation) to preserve its site configuration. +The deployment footprint updates with the selections above. It is generated from this documentation revision's rendered Helm manifests, including module dependencies, init containers and persistent volumes. Requests and limits are configuration values, not production capacity recommendations. Unspecified resources and per-node collectors prevent a complete cluster-wide total. Dynamic sandboxes and existing infrastructure are excluded. + +The calculator does not fetch arbitrary published chart versions. A different chart channel, pinned release, site override or GPU profile can change the footprint. Verify the target chart and model compatibility before deployment. + ```{raw} html
``` diff --git a/docs/pyproject.toml b/docs/pyproject.toml index b5eacbe..0a216ea 100644 --- a/docs/pyproject.toml +++ b/docs/pyproject.toml @@ -6,6 +6,7 @@ name = "ai-factory-operations-agent-docs" version = "0.1.0" requires-python = ">=3.12" dependencies = [ + "pyyaml==6.0.3", "sphinx==9.1.0", "nvidia-sphinx-theme==0.0.9.post1", "myst-parser==5.1.0", diff --git a/docs/uv.lock b/docs/uv.lock index 91237b5..659ad36 100644 --- a/docs/uv.lock +++ b/docs/uv.lock @@ -21,6 +21,7 @@ source = { virtual = "." } dependencies = [ { name = "myst-parser" }, { name = "nvidia-sphinx-theme" }, + { name = "pyyaml" }, { name = "sphinx" }, { name = "sphinx-copybutton" }, ] @@ -29,6 +30,7 @@ dependencies = [ requires-dist = [ { name = "myst-parser", specifier = "==5.1.0" }, { name = "nvidia-sphinx-theme", specifier = "==0.0.9.post1" }, + { name = "pyyaml", specifier = "==6.0.3" }, { name = "sphinx", specifier = "==9.1.0" }, { name = "sphinx-copybutton", specifier = "==0.5.2" }, ] diff --git a/tests/docs/install-builder.test.mjs b/tests/docs/install-builder.test.mjs index c157091..484f9bc 100644 --- a/tests/docs/install-builder.test.mjs +++ b/tests/docs/install-builder.test.mjs @@ -4,7 +4,7 @@ import {test} from 'node:test'; import assert from 'node:assert/strict'; import {spawnSync} from 'node:child_process'; import {fileURLToPath} from 'node:url'; -import {buildCommand, enabledModules} from '../../docs/_static/install-builder.mjs'; +import {buildCommand, enabledModules, footprint, quantity, selectedComponents} from '../../docs/_static/install-builder.mjs'; const chart = fileURLToPath(new URL('../../helm/mosaic-stack', import.meta.url)); function argumentsFor(state) { const shell = `trap '[ -z "\${MOSAIC_CHART_WORKDIR:-}" ] || rm -rf "$MOSAIC_CHART_WORKDIR"' EXIT; helm() { if [ "$1" = pull ]; then ln -s '${chart}' "$MOSAIC_CHART_WORKDIR/mosaic-stack"; else printf '%s\\0' "$@"; fi; };\n${buildCommand(state)}\nprintf '%s\\0' "\${MOSAIC_CHART_WORKDIR:-}"`; @@ -51,3 +51,51 @@ test('pinning replaces the development selector and storage overrides are option assert.ok(args.includes('openclaw.pvc.storageClassName=workspace-storage')); assert.ok(args.includes('mosaicUi.auditPvc.storageClassName=workspace-storage')); }); +test('Kubernetes quantities preserve decimal and binary units', () => { + assert.equal(quantity('250m', 'cpu'), .25); + assert.equal(quantity('2Gi', 'memory'), 2 * 1024 ** 3); + assert.equal(quantity('500M', 'memory'), 500000000); + assert.throws(() => quantity('2invalid', 'memory')); +}); + +test('pod scheduling footprint sums app containers and takes init peak before replicas', () => { + for (const cores of [1, 3, 5]) { + const container = cpu => ({resources: {requests: {cpu: String(cpu)}}}); + const component = {count: 2, containers: [container(cores), container(cores)], initContainers: [container(cores * 3)]}; + assert.equal(footprint(component, 'requests', 'cpu'), cores * 6); + } +}); + +test('selection deduplicates dependencies and merges shared workload init containers', () => { + const base = {kind: 'Deployment', containers: [{name: 'runtime', resources: {}}], initContainers: [], count: 1}; + const catalog = {base: {runtime: base}, variants: { + bcm: {runtime: {...base, initContainers: [{name: 'adapter', resources: {}}]}, adapter: {name: 'adapter'}}, + diagnostics: {runtime: {...base, initContainers: [{name: 'diagnostics', resources: {}}]}, adapter: {name: 'adapter'}}, + }}; + const components = selectedComponents(catalog, {inference: 'external', sandboxInstalled: true}, new Set(['bcm', 'diagnostics'])); + assert.equal(components.length, 2); + assert.equal(components[0].initContainers.length, 2); + assert.equal(components[0].containers.length, 1); +}); + +test('composed footprints match direct Helm renders across inference and module selections', () => { + const script = `import sys,json +sys.path.insert(0,'docs') +from conf import generate,render_footprint,MODULES +catalog=generate() +cases=[] +for inference in ['external','super','ultra']: + for backend in ['bcm','vanilla']: + for modules in [MODULES, MODULES[::2], MODULES[1::2]]: + state={'inference':inference,'modules':modules,'slurmBackend':backend,'baseUrl':'https://inference.example.com/v1','model':'served-model','bcmHead':'head.example.com','evidenceRoot':'/slurm','evidenceNode':'collector-node','corpus':'/corpus'} + cases.append({'state':state,'components':list(render_footprint(state).values())}) +print(json.dumps({'catalog':catalog,'cases':cases}))`; + const result = spawnSync('uv', ['run', '--project', 'docs', '--locked', 'python', '-c', script], {cwd: fileURLToPath(new URL('../../', import.meta.url)), encoding: 'utf8', maxBuffer: 8 * 1024 * 1024}); + assert.equal(result.status, 0, result.stderr); + const {catalog, cases} = JSON.parse(result.stdout); + const normalize = components => Object.fromEntries(components.map(component => [`${component.kind}/${component.name}`, {...component, + containers: [...(component.containers || [])].sort((left, right) => left.name.localeCompare(right.name)), + initContainers: [...(component.initContainers || [])].sort((left, right) => left.name.localeCompare(right.name)), + }])); + for (const {state, components} of cases) assert.deepEqual(normalize(selectedComponents(catalog, state, enabledModules(state))), normalize(components)); +}); From 8d107da8cb9e46e8837dd814202f82a64fa66ada Mon Sep 17 00:00:00 2001 From: Rohan Adwankar <39285979+RohanAdwankar@users.noreply.github.com> Date: Wed, 7 Oct 2026 16:14:39 -0700 Subject: [PATCH 2/2] Clarify selection-driven deployment resources Make cloud endpoint zero-GPU behavior explicit, explain absent chart resource settings, remove the visible source revision and collapse component details. Signed-off-by: Rohan Adwankar <39285979+RohanAdwankar@users.noreply.github.com> --- docs/_static/install-builder.mjs | 22 +++++++++++++--------- docs/installation_builder.md | 2 +- 2 files changed, 14 insertions(+), 10 deletions(-) diff --git a/docs/_static/install-builder.mjs b/docs/_static/install-builder.mjs index bef52f1..a5bb01e 100644 --- a/docs/_static/install-builder.mjs +++ b/docs/_static/install-builder.mjs @@ -189,9 +189,10 @@ export function footprint(component, field, resource) { export function renderSizing(panel, catalog, state, enabled) { panel.replaceChildren(); - const title = document.createElement('h3'); title.textContent = 'Deployment footprint'; panel.append(title); + const title = document.createElement('h3'); title.textContent = 'Resources for your selections'; panel.append(title); const note = document.createElement('p'); - note.textContent = `Rendered chart ${catalog.version}, source ${catalog.revision.slice(0, 12)}. Configured resources, not measured capacity. Only valid for this source version; channel, pinned-version and site overrides may differ.`; + const inference = state.inference === 'external' ? 'Existing endpoint (cloud or separately hosted)' : state.inference === 'super' ? 'On-prem Nemotron Super' : 'On-prem Nemotron Ultra'; + note.textContent = `Updates automatically from the form above. Inference: ${inference}. Enabled modules, including dependencies: ${[...enabled].join(', ') || 'none'}.`; panel.append(note); const components = selectedComponents(catalog, state, enabled); const table = document.createElement('table'); @@ -207,7 +208,7 @@ export function renderSizing(panel, catalog, state, enabled) { const values = ['requests', 'limits'].map(field => { const containers = [...(component.containers || []), ...(component.initContainers || [])]; const amount = footprint(component, field, resource); - if (resource !== 'nvidia.com/gpu' && containers.some(container => container.resources[field]?.[resource] === undefined)) return amount ? `≥ ${format(amount, resource)}` : 'Unspecified'; + if (resource !== 'nvidia.com/gpu' && containers.some(container => container.resources[field]?.[resource] === undefined)) return amount ? `≥ ${format(amount, resource)}` : 'Not set in chart'; return format(amount, resource); }); row.insertCell().textContent = component.kind === 'PersistentVolumeClaim' ? '—' : values.join(' / '); @@ -216,21 +217,24 @@ export function renderSizing(panel, catalog, state, enabled) { row.insertCell().textContent = storage.length ? format(storage.reduce((sum, value) => sum + quantity(value, 'memory'), 0) * component.count, 'memory') : '—'; if (Object.keys(component.nodeSelector || {}).length) warnings.push(`${component.name} placement: ${JSON.stringify(component.nodeSelector)}.`); } - const subtotal = body.insertRow(); subtotal.insertCell().textContent = 'Known fixed-workload subtotal (partial)'; + const subtotal = body.insertRow(); subtotal.insertCell().textContent = 'Configured subtotal (partial)'; const fixed = components.filter(component => ['Deployment', 'StatefulSet', 'Sandbox'].includes(component.kind)); for (const resource of ['cpu', 'memory', 'nvidia.com/gpu']) { - subtotal.insertCell().textContent = ['requests', 'limits'].map(field => `≥ ${format(fixed.reduce((sum, component) => sum + footprint(component, field, resource), 0), resource)}`).join(' / '); + subtotal.insertCell().textContent = ['requests', 'limits'].map(field => `${resource === 'nvidia.com/gpu' ? '' : '≥ '}${format(fixed.reduce((sum, component) => sum + footprint(component, field, resource), 0), resource)}`).join(' / '); } subtotal.insertCell().textContent = format(components.reduce((sum, component) => { const storage = typeof component.storage === 'string' ? [component.storage] : component.storage || []; return sum + storage.reduce((amount, value) => amount + quantity(value, 'memory'), 0) * component.count; }, 0), 'memory'); - warnings.push('Subtotal excludes setup jobs and per-node collectors; unspecified requests/limits are not counted. PVC subtotal excludes pre-existing and dynamically allocated storage.'); + warnings.push('CPU and memory are partial configured totals, not recommended production sizing. “Not set in chart” means the chart does not declare that request or limit; it does not mean zero usage. Subtotal excludes setup jobs and per-node collectors. PVC subtotal excludes pre-existing and dynamically allocated storage. Different chart versions and site overrides may change these values.'); if (enabled.has('research') && !catalog.researchAvailable) warnings.push('Research dependency workloads are unavailable in this documentation build; the footprint is incomplete.'); if (state.inference !== 'external') warnings.push('Model-server GPU requests must fit on one compatible node. Storage includes model-cache PVC capacity, not download size.'); const summary = document.createElement('p'); - summary.textContent = `Known fixed-workload requests: ≥ ${format(fixed.reduce((sum, component) => sum + footprint(component, 'requests', 'cpu'), 0), 'cpu')} CPU cores; ≥ ${format(fixed.reduce((sum, component) => sum + footprint(component, 'requests', 'memory'), 0), 'memory')} memory; ≥ ${format(fixed.reduce((sum, component) => sum + footprint(component, 'requests', 'nvidia.com/gpu'), 0), 'nvidia.com/gpu')} GPUs. Partial footprint; see component details below.`; + const gpus = fixed.reduce((sum, component) => sum + footprint(component, 'requests', 'nvidia.com/gpu'), 0); + summary.textContent = `Local GPU requests: ${format(gpus, 'nvidia.com/gpu')}${state.inference === 'external' ? ' (no local model server)' : ''}. Configured CPU requests: ≥ ${format(fixed.reduce((sum, component) => sum + footprint(component, 'requests', 'cpu'), 0), 'cpu')} cores. Configured memory requests: ≥ ${format(fixed.reduce((sum, component) => sum + footprint(component, 'requests', 'memory'), 0), 'memory')}. CPU and memory totals are incomplete where the chart has no resource settings.`; const details = document.createElement('div'); details.className = 'builder-sizing-table'; details.append(table); - panel.append(summary, details); - const caveat = document.createElement('p'); caveat.textContent = warnings.join(' '); panel.append(caveat); + const expanded = document.createElement('details'); + const label = document.createElement('summary'); label.textContent = 'Component resource settings'; expanded.append(label, details); + panel.append(summary, expanded); + const caveat = document.createElement('p'); caveat.textContent = warnings.join(' '); expanded.append(caveat); } diff --git a/docs/installation_builder.md b/docs/installation_builder.md index 5cd435f..6ff8457 100644 --- a/docs/installation_builder.md +++ b/docs/installation_builder.md @@ -2,7 +2,7 @@ Select your inference path and modules to generate an installation command. Complete [namespace and registry access setup](installation.md#1-create-the-namespace-and-registry-access) and create the indicated Secrets first. For an existing installation, follow the [upgrade instructions](installation.md#upgrade-an-existing-installation) to preserve its site configuration. -The deployment footprint updates with the selections above. It is generated from this documentation revision's rendered Helm manifests, including module dependencies, init containers and persistent volumes. Requests and limits are configuration values, not production capacity recommendations. Unspecified resources and per-node collectors prevent a complete cluster-wide total. Dynamic sandboxes and existing infrastructure are excluded. +Resources update automatically with the selections above. An existing endpoint, including a cloud LLM, requires no local model-server GPUs. Expand **Component resource settings** to see the selected modules and their dependencies. Values come from rendered Helm manifests, including init containers and persistent volumes. “Not set in chart” means a CPU or memory request or limit is absent, so totals are partial configuration values rather than production sizing recommendations. Per-node collectors, dynamic sandboxes and existing infrastructure are excluded from the workload subtotal. The calculator does not fetch arbitrary published chart versions. A different chart channel, pinned release, site override or GPU profile can change the footprint. Verify the target chart and model compatibility before deployment.