Skip to content
Open
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
3 changes: 3 additions & 0 deletions .github/workflows/ci.yml
Original file line number Diff line number Diff line change
Expand Up @@ -32,6 +32,9 @@ jobs:
with:
node-version: 24

- uses: astral-sh/setup-uv@94527f2e458b27549849d47d273a16bec83a01e9
with:
version: 0.12.6
- name: Prepare chart test dependencies
run: |
set -euo pipefail
Expand Down
15 changes: 15 additions & 0 deletions .github/workflows/docs.yml
Original file line number Diff line number Diff line change
Expand Up @@ -26,8 +26,23 @@ jobs:
- uses: astral-sh/setup-uv@94527f2e458b27549849d47d273a16bec83a01e9 # v7
with:
version: 0.12.6
- uses: actions/setup-node@49933ea5288caeca8642d1e84afbd3f7d6820020
with:
node-version: 24
- name: Prepare public chart footprint dependencies
run: |
chart=helm/mosaic-stack
mkdir -p "$chart/charts"
openshell_version="$(helm dependency list "$chart" | awk '$1 == "helm-chart" { print $2 }')"
iraop_version="$(helm dependency list "$chart" | awk '$1 == "iraop-stack" { print $2 }')"
helm pull oci://ghcr.io/nvidia/openshell/helm-chart --version "$openshell_version" --destination "$chart/charts"
stub="$(mktemp -d)"
printf 'apiVersion: v2\nname: iraop-stack\nversion: %s\n' "$iraop_version" >"$stub/Chart.yaml"
helm package "$stub" --destination "$chart/charts"
- name: Build documentation
run: uv run --project docs --locked sphinx-build -W --keep-going -b html -c docs . docs/_build/html
- name: Validate footprint composition
run: node --test tests/docs/*.test.mjs
- uses: actions/upload-pages-artifact@7b1f4a764d45c48632c6b24a0339c27f5614fb0b # v4
with:
path: docs/_build/html
Expand Down
3 changes: 3 additions & 0 deletions docs/_static/docs.css
Original file line number Diff line number Diff line change
Expand Up @@ -31,3 +31,6 @@ pre, .admonition, .search-button { border-radius: 0; }
@media (max-width: 600px) { #install-command-builder .builder-settings, .builder-modules { grid-template-columns: 1fr; } }

#install-command-builder .builder-settings > label:has(input[type=checkbox]) { flex-direction: row; align-items: center; grid-column: 1 / -1; }
.builder-sizing-table { overflow-x: auto; }
.builder-sizing table { width: 100%; min-width: 720px; font-size: .85rem; }
.builder-sizing th, .builder-sizing td { padding: .4rem; text-align: left; }
93 changes: 92 additions & 1 deletion docs/_static/install-builder.mjs
Original file line number Diff line number Diff line change
Expand Up @@ -114,7 +114,13 @@ if (root) {
const copy = document.createElement('button'); copy.type = 'button'; copy.textContent = 'Copy command';
const status = document.createElement('span'); status.setAttribute('role', 'status');
copy.addEventListener('click', async () => {try {await navigator.clipboard.writeText(code.textContent); status.textContent = 'Copied';} catch {status.textContent = 'Select the command text to copy it.';}});
root.replaceChildren(form, heading, copy, status, output);
const sizing = document.createElement('section'); sizing.className = 'builder-sizing';
let catalog;
root.replaceChildren(form, sizing, heading, copy, status, output);
fetch(new URL('sizing-data.json', import.meta.url)).then(response => {
if (!response.ok) throw new Error(`HTTP ${response.status}`);
return response.json();
}).then(data => {catalog = data; update(false);}).catch(error => {sizing.textContent = `Deployment footprint unavailable: ${error.message}`;});
function update(renderSettings) {
const enabled = enabledModules(state);
for (const [name, input] of inputs) {input.checked = enabled.has(name); input.disabled = enabled.has(name) && !state.modules.includes(name);}
Expand Down Expand Up @@ -144,6 +150,91 @@ if (root) {
if (enabled.has('research')) notes.push('Create iraop-secrets with NVIDIA_API_KEY, NVIDIA_CHAT_API_KEY and IRAOP_API_KEY; prepare the corpus directory.');
if (state.inference !== 'external') notes.push('Verify GPU, memory and model cache requirements in the on-prem inference section.');
prerequisites.textContent = notes.join(' '); code.textContent = buildCommand(state); status.textContent = '';
if (catalog) renderSizing(sizing, catalog, state, enabled);
}
update(true);
}
export function selectedComponents(catalog, state, enabled) {
const selections = [...enabled].filter(name => name !== 'slurm');
if (enabled.has('slurm')) selections.push(state.slurmBackend === 'vanilla' ? 'slurm-vanilla' : 'slurm');
if (!state.sandboxInstalled) selections.push('sandbox');
if (state.inference !== 'external') selections.push(state.inference);
const components = {};
for (const group of [catalog.base, ...selections.map(name => catalog.variants[name])]) {
for (const [key, component] of Object.entries(group)) {
const previous = components[key];
components[key] = {...component};
for (const field of ['containers', 'initContainers']) {
if (component[field]) components[key][field] = Object.values(Object.fromEntries([...(previous?.[field] || []), ...component[field]].map(container => [container.name, container])));
}
}
}
return Object.values(components);
}

export function quantity(value, resource) {
const match = String(value).match(/^([0-9.]+)([a-zA-Z]*)$/);
if (!match) throw new Error(`Unsupported resource quantity: ${value}`);
const factors = resource === 'cpu' ? {'': 1, m: .001, u: .000001, n: .000000001} : {'': 1, Ki: 1024, Mi: 1024 ** 2, Gi: 1024 ** 3, Ti: 1024 ** 4, k: 1000, M: 1000 ** 2, G: 1000 ** 3, T: 1000 ** 4};
if (!(match[2] in factors)) throw new Error(`Unsupported resource unit: ${match[2]}`);
return Number(match[1]) * factors[match[2]];
}

export function footprint(component, field, resource) {
const amount = container => quantity(container.resources[field]?.[resource] || 0, resource);
const running = (component.containers || []).reduce((sum, container) => sum + amount(container), 0);
const initializing = Math.max(0, ...(component.initContainers || []).map(amount));
return Math.max(running, initializing) * component.count;
}

export function renderSizing(panel, catalog, state, enabled) {
panel.replaceChildren();
const title = document.createElement('h3'); title.textContent = 'Resources for your selections'; panel.append(title);
const note = document.createElement('p');
const inference = state.inference === 'external' ? 'Existing endpoint (cloud or separately hosted)' : state.inference === 'super' ? 'On-prem Nemotron Super' : 'On-prem Nemotron Ultra';
note.textContent = `Updates automatically from the form above. Inference: ${inference}. Enabled modules, including dependencies: ${[...enabled].join(', ') || 'none'}.`;
panel.append(note);
const components = selectedComponents(catalog, state, enabled);
const table = document.createElement('table');
const header = table.createTHead().insertRow();
for (const text of ['Component', 'CPU request / limit', 'Memory request / limit', 'GPU request / limit', 'Storage']) {const cell = document.createElement('th'); cell.textContent = text; header.append(cell);}
const body = table.createTBody();
const format = (amount, resource) => resource === 'memory' ? `${Number((amount / 1024 ** 3).toFixed(3))} GiB` : `${Number(amount.toFixed(3))}`;
const warnings = ['Dynamic execution sandboxes, existing inference services, observability backends and corpus storage are not included. GPU memory compatibility and workload concurrency require separate validation.'];
for (const component of components) {
const row = body.insertRow();
row.insertCell().textContent = `${component.name}${component.kind === 'DaemonSet' ? ' (per matching node)' : component.kind === 'Job' ? ' (setup job)' : ''}`;
for (const resource of ['cpu', 'memory', 'nvidia.com/gpu']) {
const values = ['requests', 'limits'].map(field => {
const containers = [...(component.containers || []), ...(component.initContainers || [])];
const amount = footprint(component, field, resource);
if (resource !== 'nvidia.com/gpu' && containers.some(container => container.resources[field]?.[resource] === undefined)) return amount ? `≥ ${format(amount, resource)}` : 'Not set in chart';
return format(amount, resource);
});
row.insertCell().textContent = component.kind === 'PersistentVolumeClaim' ? '—' : values.join(' / ');
}
const storage = typeof component.storage === 'string' ? [component.storage] : component.storage || [];
row.insertCell().textContent = storage.length ? format(storage.reduce((sum, value) => sum + quantity(value, 'memory'), 0) * component.count, 'memory') : '—';
if (Object.keys(component.nodeSelector || {}).length) warnings.push(`${component.name} placement: ${JSON.stringify(component.nodeSelector)}.`);
}
const subtotal = body.insertRow(); subtotal.insertCell().textContent = 'Configured subtotal (partial)';
const fixed = components.filter(component => ['Deployment', 'StatefulSet', 'Sandbox'].includes(component.kind));
for (const resource of ['cpu', 'memory', 'nvidia.com/gpu']) {
subtotal.insertCell().textContent = ['requests', 'limits'].map(field => `${resource === 'nvidia.com/gpu' ? '' : '≥ '}${format(fixed.reduce((sum, component) => sum + footprint(component, field, resource), 0), resource)}`).join(' / ');
}
subtotal.insertCell().textContent = format(components.reduce((sum, component) => {
const storage = typeof component.storage === 'string' ? [component.storage] : component.storage || [];
return sum + storage.reduce((amount, value) => amount + quantity(value, 'memory'), 0) * component.count;
}, 0), 'memory');
warnings.push('CPU and memory are partial configured totals, not recommended production sizing. “Not set in chart” means the chart does not declare that request or limit; it does not mean zero usage. Subtotal excludes setup jobs and per-node collectors. PVC subtotal excludes pre-existing and dynamically allocated storage. Different chart versions and site overrides may change these values.');
if (enabled.has('research') && !catalog.researchAvailable) warnings.push('Research dependency workloads are unavailable in this documentation build; the footprint is incomplete.');
if (state.inference !== 'external') warnings.push('Model-server GPU requests must fit on one compatible node. Storage includes model-cache PVC capacity, not download size.');
const summary = document.createElement('p');
const gpus = fixed.reduce((sum, component) => sum + footprint(component, 'requests', 'nvidia.com/gpu'), 0);
summary.textContent = `Local GPU requests: ${format(gpus, 'nvidia.com/gpu')}${state.inference === 'external' ? ' (no local model server)' : ''}. Configured CPU requests: ≥ ${format(fixed.reduce((sum, component) => sum + footprint(component, 'requests', 'cpu'), 0), 'cpu')} cores. Configured memory requests: ≥ ${format(fixed.reduce((sum, component) => sum + footprint(component, 'requests', 'memory'), 0), 'memory')}. CPU and memory totals are incomplete where the chart has no resource settings.`;
const details = document.createElement('div'); details.className = 'builder-sizing-table'; details.append(table);
const expanded = document.createElement('details');
const label = document.createElement('summary'); label.textContent = 'Component resource settings'; expanded.append(label, details);
panel.append(summary, expanded);
const caveat = document.createElement('p'); caveat.textContent = warnings.join(' '); expanded.append(caveat);
}
4 changes: 3 additions & 1 deletion docs/building.md
Original file line number Diff line number Diff line change
Expand Up @@ -7,7 +7,7 @@ The documentation site uses NVIDIA's Sphinx theme and renders the Markdown guide

## Build and preview

Install [uv](https://docs.astral.sh/uv/getting-started/installation/), then run from the repository root:
Install [uv](https://docs.astral.sh/uv/getting-started/installation/), Node.js and Helm, and prepare the chart dependencies with `helm dependency build helm/mosaic-stack` (private registry access is required for the research dependency). Then run from the repository root:

```bash
uv run --project docs --locked sphinx-build -W --keep-going -b html -c docs . docs/_build/html
Expand All @@ -18,6 +18,8 @@ Open `http://localhost:8080`. The build treats warnings as errors, including bro

## GitHub Pages

The footprint catalog is generated during each HTML build by rendering the tracked chart with the builder's own commands. The public documentation workflow does not authenticate to the private research registry; it uses an empty research dependency solely for rendering the public chart. Research workloads are consequently marked unavailable in that build. With an authorized real dependency installed locally, its manifests are included. Never deploy the documentation/test placeholder dependency.

The documentation workflow builds pull requests without publishing. After a change lands on `main`, it publishes the generated static site to GitHub Pages using the `github-pages` environment. A repository administrator must first enable **Settings → Pages → Build and deployment → Source: GitHub Actions**.

The intended URL is `https://nvidia.github.io/AI-Factory-Operations-Agent/`. Until Pages is enabled and a deployment succeeds, that URL will not serve the documentation. No registry or inference credentials are needed for the documentation build.
78 changes: 78 additions & 0 deletions docs/conf.py
Original file line number Diff line number Diff line change
@@ -1,6 +1,13 @@
# SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved.
# SPDX-License-Identifier: Apache-2.0

import json
import os
import subprocess
from pathlib import Path

import yaml

project = "AI Factory Operations Agent"
copyright = "2026, NVIDIA Corporation"
extensions = ["myst_parser", "sphinx_copybutton"]
Expand Down Expand Up @@ -35,3 +42,74 @@ def render_dco(app, docname, source):

def setup(app):
app.connect("source-read", render_dco)
app.connect("build-finished", write_catalog)


ROOT = Path(__file__).resolve().parents[1]
CHART = ROOT / "helm/mosaic-stack"
MODULES = ["kubernetes", "observability", "grafana", "bcm", "slurm", "diagnostics", "research", "terminal", "edit", "clusters"]


def render_footprint(state):
script = "import {buildCommand} from './docs/_static/install-builder.mjs'; console.log(JSON.stringify(buildCommand(JSON.parse(process.argv[1]))));"
command = json.loads(subprocess.check_output(["node", "--input-type=module", "-e", script, json.dumps(state)], cwd=ROOT, text=True))
shell = 'trap \'[ -z "${MOSAIC_CHART_WORKDIR:-}" ] || rm -rf "$MOSAIC_CHART_WORKDIR"\' EXIT; helm() { if [ "$1" = pull ]; then ln -s "$CHART" "$MOSAIC_CHART_WORKDIR/mosaic-stack"; else printf \'%s\\0\' "$@"; fi; }; '
arguments = subprocess.check_output(["bash", "-c", shell + command], env={**os.environ, "CHART": str(CHART)}, text=True).split("\0")[4:-1]
rendered = ["template", "mosaic", str(CHART)]
arguments = iter(arguments)
for argument in arguments:
if argument in {"--atomic", "--wait", "--devel"}:
continue
if argument in {"--timeout", "--version"}:
next(arguments)
elif argument == "-f":
next(arguments)
profile = "vllm-ultra-4gpu.yaml" if state["inference"] == "ultra" else "vllm-super-1gpu.yaml"
rendered.extend(["-f", str(CHART / "profiles" / profile)])
else:
rendered.append(argument)
output = subprocess.check_output(["helm", *rendered], text=True)
components = {}
for document in yaml.safe_load_all(output):
if not document:
continue
kind = document.get("kind")
spec = document.get("spec", {})
name = document["metadata"]["name"]
key = f"{kind}/{document['metadata'].get('namespace', 'mosaic')}/{name}"
if kind == "PersistentVolumeClaim":
components[key] = {"name": name, "kind": kind, "storage": spec["resources"]["requests"]["storage"], "count": 1}
elif kind in {"Deployment", "StatefulSet", "DaemonSet", "Job", "Sandbox"}:
pod = spec.get("template", {}).get("spec", {})
if not pod.get("containers"):
continue
components[key] = {
"name": name, "kind": kind, "count": spec.get("replicas", 1),
"containers": [{"name": container["name"], "resources": container.get("resources", {})} for container in pod["containers"]],
"initContainers": [{"name": container["name"], "resources": container.get("resources", {})} for container in pod.get("initContainers", [])],
"storage": [claim["spec"]["resources"]["requests"]["storage"] for claim in spec.get("volumeClaimTemplates", [])],
"nodeSelector": pod.get("nodeSelector", {}),
}
return components


def generate():
defaults = {"inference": "external", "modules": [], "sandboxInstalled": True, "baseUrl": "https://inference.example.com/v1", "model": "served-model", "bcmHead": "head.example.com", "evidenceRoot": "/slurm", "evidenceNode": "collector-node", "corpus": "/corpus"}
base = render_footprint(defaults)
variants = {}
selections = {"sandbox": {"sandboxInstalled": False}, "super": {"inference": "super"}, "ultra": {"inference": "ultra"}}
selections.update({module: {"modules": [module]} for module in MODULES})
selections["slurm-vanilla"] = {"modules": ["slurm"], "slurmBackend": "vanilla"}
for name, selection in selections.items():
rendered = render_footprint({**defaults, **selection})
variants[name] = {key: value for key, value in rendered.items() if value != base.get(key)}
chart = yaml.safe_load((CHART / "Chart.yaml").read_text())
revision = subprocess.check_output(["git", "rev-parse", "HEAD"], cwd=ROOT, text=True).strip()
research_available = any(key not in base and component["kind"] in {"Deployment", "StatefulSet"} for key, component in variants["research"].items())
return {"version": chart["version"], "revision": revision, "base": base, "variants": variants, "researchAvailable": research_available}


def write_catalog(app, exception):
if exception is None and app.builder.format == "html":
target = Path(app.outdir) / "_static/sizing-data.json"
target.write_text(json.dumps(generate(), indent=2) + "\n")
4 changes: 4 additions & 0 deletions docs/installation_builder.md
Original file line number Diff line number Diff line change
Expand Up @@ -2,6 +2,10 @@

Select your inference path and modules to generate an installation command. Complete [namespace and registry access setup](installation.md#1-create-the-namespace-and-registry-access) and create the indicated Secrets first. For an existing installation, follow the [upgrade instructions](installation.md#upgrade-an-existing-installation) to preserve its site configuration.

Resources update automatically with the selections above. An existing endpoint, including a cloud LLM, requires no local model-server GPUs. Expand **Component resource settings** to see the selected modules and their dependencies. Values come from rendered Helm manifests, including init containers and persistent volumes. “Not set in chart” means a CPU or memory request or limit is absent, so totals are partial configuration values rather than production sizing recommendations. Per-node collectors, dynamic sandboxes and existing infrastructure are excluded from the workload subtotal.

The calculator does not fetch arbitrary published chart versions. A different chart channel, pinned release, site override or GPU profile can change the footprint. Verify the target chart and model compatibility before deployment.

```{raw} html
<div id="install-command-builder"><noscript>Enable JavaScript to use the builder, or follow the <a href="installation.html">installation guide</a>.</noscript></div>
```
1 change: 1 addition & 0 deletions docs/pyproject.toml
Original file line number Diff line number Diff line change
Expand Up @@ -6,6 +6,7 @@ name = "ai-factory-operations-agent-docs"
version = "0.1.0"
requires-python = ">=3.12"
dependencies = [
"pyyaml==6.0.3",
"sphinx==9.1.0",
"nvidia-sphinx-theme==0.0.9.post1",
"myst-parser==5.1.0",
Expand Down
2 changes: 2 additions & 0 deletions docs/uv.lock

Some generated files are not rendered by default. Learn more about how customized files appear on GitHub.

Loading
Loading