diff --git a/.github/workflows/monitor_unit_test.yml b/.github/workflows/monitor_unit_test.yml index 4eafa23e..c1f0f794 100644 --- a/.github/workflows/monitor_unit_test.yml +++ b/.github/workflows/monitor_unit_test.yml @@ -15,6 +15,8 @@ on: - "pyproject.toml" - "rl_insight/client/**" - "rl_insight/collector/**" + - "rl_insight/grafana/**" + - "rl_insight/config/services/grafana/**" - "rl_insight/utils/**" - "tests/monitor/ut/**" - ".github/workflows/monitor_unit_test.yml" diff --git a/pyproject.toml b/pyproject.toml index eb972f3c..d2b97dd5 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -22,6 +22,10 @@ dependencies = [ "opentelemetry-exporter-otlp-proto-http", "pyyaml", "psutil", + # In-process Jsonnet evaluator used by rl_insight.grafana.renderer. rjsonnet + # publishes CPython ABI3 wheels for Windows, macOS and Linux, so no Jsonnet + # CLI, Go toolchain or compiler is required at install time. + "rjsonnet>=0.5.6", ] [project.optional-dependencies] @@ -64,6 +68,11 @@ rl_insight = [ "config/services/grafana/provisioning/dashboards/*.yml", "config/services/grafana/dashboards/*.json", "config/services/grafana/dashboards/*/*.json", + # Jsonnet mechanism sources the runtime renders at startup: the entrypoint, + # the composition registry and the generic framework assets. + "config/services/grafana/jsonnet/*.jsonnet", + "config/services/grafana/jsonnet/*.libsonnet", + "config/services/grafana/jsonnet/framework/*.libsonnet", ] recipe = [ "config/*.yaml", diff --git a/rl_insight/config/config.yaml b/rl_insight/config/config.yaml index 848b18dd..a2d1c7d9 100644 --- a/rl_insight/config/config.yaml +++ b/rl_insight/config/config.yaml @@ -42,6 +42,12 @@ grafana: port: 3000 config_file: ${service_root}/services/grafana/grafana.ini provisioning_dir: ${service_root}/services/grafana/provisioning + # Explicit Jsonnet composition config rendered at startup. Empty (the + # default) renders the Jsonnet entrypoint bundled in the installed package. + # An unreadable path or a Jsonnet error stops startup before Grafana runs. + dashboard_config: "" + # Legacy static directory. Copying it is only used when it points somewhere + # other than the bundled defaults, and then it replaces the built-in render. dashboards_dir: ${service_root}/services/grafana/dashboards diff --git a/rl_insight/config/services/grafana/jsonnet/dashboard_compositions.libsonnet b/rl_insight/config/services/grafana/jsonnet/dashboard_compositions.libsonnet new file mode 100644 index 00000000..115ce551 --- /dev/null +++ b/rl_insight/config/services/grafana/jsonnet/dashboard_compositions.libsonnet @@ -0,0 +1,16 @@ +// The composition registry consumed by `dashboards.jsonnet`. +// +// The generic mechanism ships this file empty: nothing here knows about any +// concrete dashboard, so this change can be merged and started on its own while +// the runtime keeps staging the bundled static JSON dashboards. +// +// A dependent change that adds real content specializes this registry: +// +// local myDashboard = import 'compositions/my_dashboard.libsonnet'; +// { compositions: { my_dashboard: myDashboard } } +// +// Each entry provides `modules` (the ordered modules to compose) and `dashboard` +// (metadata, title, tags, chrome `spec`, `variableOrder`, `rowOrder`). +{ + compositions: {}, +} diff --git a/rl_insight/config/services/grafana/jsonnet/dashboards.jsonnet b/rl_insight/config/services/grafana/jsonnet/dashboards.jsonnet new file mode 100644 index 00000000..7030c84e --- /dev/null +++ b/rl_insight/config/services/grafana/jsonnet/dashboards.jsonnet @@ -0,0 +1,13 @@ +// Thin stable entrypoint: composes every dashboard registered in the +// production composition registry. New dashboards only need a registry entry; +// this file should not need to change. +local composer = import 'framework/composer.libsonnet'; +local registry = import 'dashboard_compositions.libsonnet'; + +{ + [name]: composer.compose( + registry.compositions[name].modules, + registry.compositions[name].dashboard + ) + for name in std.objectFields(registry.compositions) +} diff --git a/rl_insight/config/services/grafana/jsonnet/framework/composer.libsonnet b/rl_insight/config/services/grafana/jsonnet/framework/composer.libsonnet new file mode 100644 index 00000000..307de06c --- /dev/null +++ b/rl_insight/config/services/grafana/jsonnet/framework/composer.libsonnet @@ -0,0 +1,274 @@ +// Generic Grafana dashboard composer. +// +// Takes an ordered list of sub-dashboard modules (A, B, C, D, ...) plus a +// dashboard-level composition record, and renders one Grafana Dashboard v2 +// resource. The composer knows nothing about any concrete dashboard: modules +// are plain data, and every merge decision follows the explicit rules +// documented in README.md ("Composition rules"). +// +// Module schema (all fields plain data, no behavioural code): +// panels: [{ key, outputKey, id, title, queries, description?, links?, +// vizBase?, vizPatch?, transformations?, queryOptions? }] +// rows?: { : } (panels referenced by `key`) +// rowItems?: { : [GridLayoutItem, ...] } +// additive extension of an existing row, no rows created +// variables?: { : } +// tags?: [tag, ...] +// +// Dashboard schema (owned by the composition config, not by modules): +// metadata, title, tags, spec (dashboard chrome), variableOrder, rowOrder +local viz = import 'viz.libsonnet'; + +local has(object, field) = std.objectHas(object, field); + +// Guard helper: returns `value` when `condition` holds, otherwise fails +// evaluation with `message`. Threading the value through the guard keeps the +// check eager despite Jsonnet's lazy evaluation. +local require(condition, message, value) = + if condition then value else error message; + +// Apply an RFC 7396-style merge patch. Modules only describe the fields that +// differ from the shared visualization defaults in viz.libsonnet. +local mergePatch(base, patch) = + if std.type(base) != 'object' || std.type(patch) != 'object' then patch + else + { + [field]: base[field] + for field in std.objectFields(base) + if !has(patch, field) + } + { + [field]: + if has(base, field) then mergePatch(base[field], patch[field]) + else patch[field] + for field in std.objectFields(patch) + if patch[field] != null + }; + +local prometheusQuery(query) = { + kind: 'DataQuery', + group: 'prometheus', + version: 'v0', + spec: { + editorMode: if has(query, 'editorMode') then query.editorMode else 'builder', + expr: query.expr, + legendFormat: if has(query, 'legend') then query.legend else '{{__name__}}', + range: true, + } + (if has(query, 'extra') then query.extra else {}), +} + (if has(query, 'labels') then { labels: query.labels } else {}); + +local panelQuery(query, index) = { + kind: 'PanelQuery', + spec: { + query: if has(query, 'raw') then query.raw else prometheusQuery(query), + refId: if has(query, 'refId') then query.refId else std.char(std.codepoint('A') + index), + hidden: if has(query, 'hidden') then query.hidden else false, + }, +}; + +local panel(record) = { + kind: 'Panel', + spec: { + id: record.id, + title: record.title, + description: if has(record, 'description') then record.description else '', + links: if has(record, 'links') then record.links else [], + data: { + kind: 'QueryGroup', + spec: { + queries: std.mapWithIndex(function(index, query) panelQuery(query, index), record.queries), + transformations: if has(record, 'transformations') then record.transformations else [], + queryOptions: if has(record, 'queryOptions') then record.queryOptions else {}, + }, + }, + vizConfig: mergePatch( + viz[if has(record, 'vizBase') then record.vizBase else 'timeseries'], + if has(record, 'vizPatch') then record.vizPatch else {} + ), + }, +}; + +// Rewrite `ElementReference` nodes inside layout specs: layouts address panels +// by module-unique `key`, while dashboard elements are keyed by `outputKey`. +// Resolving late keeps module files free of naming collisions. +local replaceReferences(value, outputKeys) = + if std.type(value) == 'array' then + [replaceReferences(item, outputKeys) for item in value] + else if std.type(value) == 'object' then + if has(value, 'kind') && value.kind == 'ElementReference' then + value { name: outputKeys[value.name] } + else + { [field]: replaceReferences(value[field], outputKeys) for field in std.objectFields(value) } + else value; + +local duplicates(items) = + local counts = std.foldl( + function(acc, item) acc { [item + '']: std.get(acc, item + '', 0) + 1 }, + items, + {} + ); + [field for field in std.objectFields(counts) if counts[field] > 1]; + +// Returns `value` unless `items` contains duplicates, in which case evaluation +// fails and names the offending entries. +local requireUnique(items, what, value) = + local dups = duplicates(items); + require( + std.length(dups) == 0, + 'duplicate ' + what + '(s): ' + std.join(', ', dups), + value + ); + +local mergedField(modules, field) = + std.foldl( + function(acc, module) + acc + (if has(module, field) then module[field] else {}), + modules, + {} + ); + +// Compose `modules` into one dashboard following the rules in README.md. +// `dashboard` carries the dashboard-level decisions: metadata, title, tags, +// chrome `spec`, and the variable/row ordering. +local compose(modules, dashboard) = + local checkedModules = + require( + std.type(modules) == 'array' && std.length(modules) > 0, + 'compose() needs a non-empty module list', + modules + ); + local recordsRaw = std.flattenArrays([module.panels for module in checkedModules]); + local recordsKeyed = requireUnique( + [record.key for record in recordsRaw], + 'panel key', + recordsRaw + ); + local recordsNamed = requireUnique( + [record.outputKey for record in recordsKeyed], + 'panel outputKey', + recordsKeyed + ); + local records = requireUnique( + [record.id + '' for record in recordsNamed], + 'panel id', + recordsNamed + ); + // Object merge (`+`) collapses same-named fields silently, so duplicates + // must be detected on the per-module field lists, not on the merged object. + local rowsMerged = mergedField(checkedModules, 'rows'); + local rows = requireUnique( + std.flattenArrays([ + if has(module, 'rows') then std.objectFields(module.rows) else [] + for module in checkedModules + ]), + 'row name', + rowsMerged + ); + // `rowItems` is additive: modules append GridLayoutItems to rows owned by + // other (or their own) modules without creating rows. Extensions are folded + // in composition order, so items appended by earlier modules come first. + // Appended `ElementReference` names are resolved together with the row's + // own references at render time (late key -> outputKey resolution). + local rowItemExtensions = [ + if has(module, 'rowItems') then module.rowItems else {} + for module in checkedModules + ]; + // Every extension target is validated eagerly: unknown names and rows whose + // structure cannot take items fail evaluation instead of being ignored. + // (Object-merge fields are lazy, so validation must flow through a value + // the dashboard actually uses.) + local extensionTargets = + std.flattenArrays([std.objectFields(ext) for ext in rowItemExtensions]); + local validatedExtensions = [ + require( + std.objectHas(rows, name), + 'unknown row extension target: ' + name, + require( + has(rows[name], 'kind') + && rows[name].kind == 'RowsLayoutRow' + && has(rows[name], 'spec') + && has(rows[name].spec, 'layout') + && has(rows[name].spec.layout, 'kind') + && rows[name].spec.layout.kind == 'GridLayout' + && has(rows[name].spec.layout, 'spec') + && has(rows[name].spec.layout.spec, 'items'), + 'unsupported row extension target: expected a RowsLayoutRow with a ' + + 'GridLayout layout carrying spec.items', + rows[name] + ) + ) + for name in extensionTargets + ]; + local extensionDigest = std.toString(validatedExtensions); + assert std.type(extensionDigest) == 'string' : 'rowItems validation failed'; + local rowsExtended = std.foldl( + function(acc, extensions) + acc + { + [name]: + acc[name] + { + spec: acc[name].spec { + layout: acc[name].spec.layout { + spec: { items: acc[name].spec.layout.spec.items + extensions[name] }, + }, + }, + } + for name in std.objectFields(extensions) + }, + rowItemExtensions, + rows + ); + local variablesMerged = mergedField(checkedModules, 'variables'); + local variables = requireUnique( + std.flattenArrays([ + if has(module, 'variables') then std.objectFields(module.variables) else [] + for module in checkedModules + ]), + 'variable name', + variablesMerged + ); + local outputKeys = { [record.key]: record.outputKey for record in records }; + local variablesOrdered = require( + std.length([name for name in dashboard.variableOrder if !std.objectHas(variables, name)]) == 0, + 'unknown variable(s) in variableOrder: ' + + std.join( + ', ', + [name for name in dashboard.variableOrder if !std.objectHas(variables, name)] + ), + variables + ); + local rowsOrdered = require( + std.length([name for name in dashboard.rowOrder if !std.objectHas(rowsExtended, name)]) == 0, + 'unknown row(s) in rowOrder: ' + + std.join(', ', [name for name in dashboard.rowOrder if !std.objectHas(rowsExtended, name)]), + rowsExtended + ); + local moduleTags = std.flattenArrays([ + if has(module, 'tags') then module.tags else [] + for module in checkedModules + ]); + local tags = std.foldl( + function(acc, tag) if std.member(acc, tag) then acc else acc + [tag], + dashboard.tags + moduleTags, + [] + ); + { + apiVersion: 'dashboard.grafana.app/v2', + kind: 'Dashboard', + metadata: dashboard.metadata, + spec: dashboard.spec { + elements: { [record.outputKey]: panel(record) for record in records }, + layout: { + kind: 'RowsLayout', + spec: { + rows: [replaceReferences(rowsOrdered[name], outputKeys) for name in dashboard.rowOrder], + }, + }, + title: dashboard.title, + variables: [variablesOrdered[name] for name in dashboard.variableOrder], + tags: tags, + }, + }; + +{ + compose: compose, +} diff --git a/rl_insight/config/services/grafana/jsonnet/framework/viz.libsonnet b/rl_insight/config/services/grafana/jsonnet/framework/viz.libsonnet new file mode 100644 index 00000000..6f4f5e9d --- /dev/null +++ b/rl_insight/config/services/grafana/jsonnet/framework/viz.libsonnet @@ -0,0 +1,396 @@ +// Generated once from the existing dashboards; maintained as Jsonnet source. +{ + heatmap: { + group: 'heatmap', + kind: 'VizConfig', + spec: { + fieldConfig: { + defaults: { + custom: { + hideFrom: { + legend: false, + tooltip: false, + viz: false, + }, + scaleDistribution: { + type: 'linear', + }, + }, + }, + overrides: [], + }, + options: { + annotations: { + clustering: -1, + multiLane: false, + }, + calculate: false, + cellGap: 1, + cellValues: { + unit: 'none', + }, + color: { + exponent: 0.5, + fill: 'dark-orange', + min: 0, + mode: 'scheme', + reverse: false, + scale: 'exponential', + scheme: 'Spectral', + steps: 64, + }, + exemplars: { + color: 'rgba(255,0,255,0.7)', + }, + filterValues: { + le: 1e-09, + }, + legend: { + show: true, + showLegend: true, + }, + rowsFrame: { + layout: 'auto', + value: 'Request count', + }, + tooltip: { + mode: 'single', + showColorScale: false, + yHistogram: true, + }, + yAxis: { + axisPlacement: 'left', + reverse: false, + unit: 'none', + }, + }, + }, + version: '13.0.2', + }, + stat: { + group: 'stat', + kind: 'VizConfig', + spec: { + fieldConfig: { + defaults: { + color: { + mode: 'thresholds', + }, + thresholds: { + mode: 'absolute', + steps: [ + { + color: 'green', + value: 0, + }, + ], + }, + unit: 's', + }, + overrides: [], + }, + options: { + colorMode: 'value', + graphMode: 'none', + justifyMode: 'auto', + orientation: 'auto', + percentChangeColorMode: 'standard', + reduceOptions: { + calcs: [ + 'lastNotNull', + ], + fields: '', + values: false, + }, + showPercentChange: false, + text: {}, + textMode: 'auto', + wideLayout: true, + }, + }, + version: '13.0.2', + }, + bargauge: { + group: 'bargauge', + kind: 'VizConfig', + spec: { + fieldConfig: { + defaults: { + color: { + mode: 'continuous-GrYlRd', + }, + max: 1, + min: 0, + thresholds: { + mode: 'absolute', + steps: [ + { + color: 'green', + value: 0, + }, + { + color: 'yellow', + value: 0.7, + }, + { + color: 'red', + value: 0.9, + }, + ], + }, + unit: 'percentunit', + }, + overrides: [], + }, + options: { + displayMode: 'gradient', + legend: { + calcs: [], + displayMode: 'list', + placement: 'bottom', + showLegend: false, + }, + maxVizHeight: 300, + minVizHeight: 10, + minVizWidth: 0, + namePlacement: 'auto', + orientation: 'horizontal', + reduceOptions: { + calcs: [ + 'lastNotNull', + ], + fields: '', + values: false, + }, + showUnfilled: true, + sizing: 'auto', + valueMode: 'color', + }, + }, + version: '13.0.2', + }, + 'state-timeline': { + group: 'state-timeline', + kind: 'VizConfig', + spec: { + fieldConfig: { + defaults: { + color: { + mode: 'thresholds', + }, + custom: { + axisPlacement: 'auto', + fillOpacity: 70, + hideFrom: { + legend: false, + tooltip: false, + viz: false, + }, + insertNulls: false, + lineWidth: 0, + spanNulls: false, + }, + mappings: [ + { + options: { + pattern: '.*compute_log_prob.*', + result: { + color: 'yellow', + index: 0, + }, + }, + type: 'regex', + }, + { + options: { + pattern: '.*generate.*', + result: { + color: 'blue', + index: 1, + }, + }, + type: 'regex', + }, + { + options: { + pattern: '.*compute_ref_log_prob.*', + result: { + color: 'purple', + index: 2, + }, + }, + type: 'regex', + }, + ], + thresholds: { + mode: 'absolute', + steps: [ + { + color: 'green', + value: 0, + }, + ], + }, + }, + overrides: [], + }, + options: { + alignValue: 'left', + annotations: { + clustering: -1, + multiLane: false, + }, + legend: { + displayMode: 'list', + placement: 'bottom', + showLegend: true, + }, + mergeValues: true, + perPage: 64, + rowHeight: 0.9, + showValue: 'auto', + tooltip: { + hideZeros: false, + mode: 'single', + sort: 'none', + }, + }, + }, + version: '13.0.2', + }, + gauge: { + group: 'gauge', + kind: 'VizConfig', + spec: { + fieldConfig: { + defaults: { + color: { + mode: 'thresholds', + }, + thresholds: { + mode: 'absolute', + steps: [ + { + color: 'green', + value: 0, + }, + { + color: 'red', + value: 80, + }, + ], + }, + }, + overrides: [], + }, + options: { + barShape: 'flat', + barWidthFactor: 0.5, + effects: { + barGlow: false, + centerGlow: false, + gradient: true, + }, + endpointMarker: 'point', + minVizHeight: 75, + minVizWidth: 75, + orientation: 'auto', + reduceOptions: { + calcs: [ + 'lastNotNull', + ], + fields: '', + values: false, + }, + segmentCount: 1, + segmentSpacing: 0.3, + shape: 'gauge', + showThresholdLabels: false, + showThresholdMarkers: true, + sizing: 'auto', + sparkline: true, + textMode: 'auto', + }, + }, + version: '13.0.2', + }, + timeseries: { + group: 'timeseries', + kind: 'VizConfig', + spec: { + fieldConfig: { + defaults: { + color: { + mode: 'palette-classic', + }, + custom: { + axisBorderShow: false, + axisCenteredZero: false, + axisColorMode: 'text', + axisLabel: '', + axisPlacement: 'auto', + barAlignment: 0, + barWidthFactor: 0.6, + drawStyle: 'line', + fillOpacity: 8, + gradientMode: 'none', + hideFrom: { + legend: false, + tooltip: false, + viz: false, + }, + insertNulls: false, + lineInterpolation: 'linear', + lineWidth: 2, + pointSize: 5, + scaleDistribution: { + type: 'linear', + }, + showPoints: 'never', + showValues: false, + spanNulls: false, + stacking: { + group: 'A', + mode: 'none', + }, + thresholdsStyle: { + mode: 'off', + }, + }, + thresholds: { + mode: 'absolute', + steps: [ + { + color: 'green', + value: 0, + }, + { + color: 'red', + value: 80, + }, + ], + }, + }, + overrides: [], + }, + options: { + annotations: { + clustering: -1, + multiLane: false, + }, + legend: { + calcs: [], + displayMode: 'list', + placement: 'bottom', + showLegend: true, + }, + tooltip: { + hideZeros: false, + mode: 'single', + sort: 'none', + }, + }, + }, + version: '13.0.2', + }, +} diff --git a/rl_insight/grafana/__init__.py b/rl_insight/grafana/__init__.py new file mode 100644 index 00000000..e3aaa167 --- /dev/null +++ b/rl_insight/grafana/__init__.py @@ -0,0 +1,37 @@ +# Copyright (c) 2026 verl-project authors. +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. + +"""Generic Grafana dashboard rendering shipped with the ``rl_insight`` package. + +The public API lives in :mod:`rl_insight.grafana.renderer`; this package only +re-exports it so callers can ``from rl_insight.grafana import render_dashboards``. +""" + +from rl_insight.grafana.renderer import ( + FRAMEWORK_DIR, + JsonnetRenderError, + generated_text, + materialize_dashboards, + render_dashboards, + stale_dashboards, +) + +__all__ = [ + "FRAMEWORK_DIR", + "JsonnetRenderError", + "generated_text", + "materialize_dashboards", + "render_dashboards", + "stale_dashboards", +] diff --git a/rl_insight/grafana/renderer.py b/rl_insight/grafana/renderer.py new file mode 100644 index 00000000..34e08948 --- /dev/null +++ b/rl_insight/grafana/renderer.py @@ -0,0 +1,157 @@ +# Copyright (c) 2026 verl-project authors. +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. + +"""Render Grafana dashboard JSON from a Jsonnet composition config. + +A composition config is a Jsonnet file that imports the generic composer and +evaluates to ``{ "": }``, where each +dashboard resource is the object returned by ``composer.compose(modules, +dashboard)`` for that dashboard. + +This module is the single runtime core for that rendering: it evaluates the +config in-process through the ``rjsonnet`` binding and serializes the result +deterministically. Nothing here shells out to a Jsonnet CLI or to another +Python entry point, so importing :mod:`rl_insight.grafana.renderer` is enough +to render dashboards on every supported platform. + +Determinism: the composer emits object fields in sorted order and +:func:`generated_text` uses a fixed indentation, so two runs over the same +sources always produce byte-identical files. +""" + +from __future__ import annotations + +import json +from pathlib import Path +from typing import Any + +import rjsonnet + +#: Directory holding the generic Jsonnet framework assets that ship inside the +#: installed package (``composer.libsonnet``, ``viz.libsonnet``). Composition +#: configs import the composer from here, so there is exactly one source of +#: truth for the framework, shared by the library and by the optional CLI. +FRAMEWORK_DIR = ( + Path(__file__).resolve().parent.parent + / "config" + / "services" + / "grafana" + / "jsonnet" + / "framework" +) + + +class JsonnetRenderError(RuntimeError): + """A composition config could not be evaluated or serialized. + + The message always names the offending config path and, for evaluation + failures, keeps the Jsonnet stack trace produced by the evaluator. + """ + + +def _evaluate(config: Path) -> str: + """Evaluate ``config`` with the in-process ``rjsonnet`` binding.""" + try: + return rjsonnet.evaluate_file(str(config)) + except Exception as error: + raise JsonnetRenderError( + f"Jsonnet evaluation failed for config {config}: {error}" + ) from error + + +def render_dashboards(config: Path) -> dict[str, Any]: + """Evaluate ``config`` and return its ``{name: dashboard}`` mapping. + + Raises :class:`JsonnetRenderError` when the config is missing, fails to + evaluate, does not produce JSON, or does not produce an object of + dashboards. An empty object is valid: the registry ships empty, so a config + registering no composition renders zero dashboards and the runtime keeps + staging the bundled static dashboards. + """ + config = Path(config) + if not config.is_file(): + raise JsonnetRenderError(f"composition config not found: {config}") + + rendered = _evaluate(config) + + try: + dashboards = json.loads(rendered) + except json.JSONDecodeError as error: + raise JsonnetRenderError( + f"Jsonnet evaluation of {config} did not produce JSON: {error}" + ) from error + + if not isinstance(dashboards, dict): + raise JsonnetRenderError(f"{config} must evaluate to an object of dashboards") + return dashboards + + +def generated_text(dashboard: Any) -> str: + """Serialize one dashboard deterministically (stable bytes across runs).""" + return json.dumps(dashboard, ensure_ascii=False, indent=2) + "\n" + + +def materialize_dashboards( + config: Path, output_dir: Path, *, overwrite: bool = True +) -> list[Path]: + """Render ``config`` and write one ``.json`` per dashboard. + + Returns the written paths in the order the config declares them. The + output directory is created when missing. + + With ``overwrite=False`` the rendering is additive: every target path is + computed up front and, if any of them already exists, nothing at all is + written and :class:`JsonnetRenderError` names the conflicting paths. The + check happens before the first write, so a collision never leaves a + partially materialized set behind. + """ + dashboards = render_dashboards(config) + output_dir = Path(output_dir) + targets = [(name, output_dir / f"{name}.json") for name in dashboards] + + if not overwrite: + collisions = [path for _, path in targets if path.exists()] + if collisions: + raise JsonnetRenderError( + f"refusing to overwrite existing dashboard file(s) rendered " + f"from config {config}: " + ", ".join(str(path) for path in collisions) + ) + + output_dir.mkdir(parents=True, exist_ok=True) + + written: list[Path] = [] + for name, path in targets: + path.write_text(generated_text(dashboards[name]), encoding="utf-8") + written.append(path) + return written + + +def stale_dashboards(config: Path, expected_dir: Path) -> list[Path]: + """Return the expected files that do not match ``config``'s rendering. + + A file is stale when it is missing or differs byte-for-byte from the + deterministic rendering. An empty result means the checked-in dashboards + are up to date. + """ + dashboards = render_dashboards(config) + expected_dir = Path(expected_dir) + + stale: list[Path] = [] + for name, dashboard in dashboards.items(): + path = expected_dir / f"{name}.json" + if not path.exists() or path.read_text(encoding="utf-8") != generated_text( + dashboard + ): + stale.append(path) + return stale diff --git a/rl_insight/server/runtime.py b/rl_insight/server/runtime.py index 9f3ce0b5..2ef79422 100644 --- a/rl_insight/server/runtime.py +++ b/rl_insight/server/runtime.py @@ -39,7 +39,9 @@ import yaml from omegaconf import DictConfig, OmegaConf +from ..grafana.renderer import JsonnetRenderError, materialize_dashboards from ..utils.constants import ( + MonitorPaths, PrometheusScrape, prometheus_targets_file_from_config, ) @@ -126,7 +128,7 @@ def prepare_files( ) if bool(OmegaConf.select(self.conf, "grafana.enable", default=True)): grafana_config = _render_grafana_config(self.conf, runtime_dir, data_dir) - dashboard_path = _stage_grafana_dashboards(self.conf, runtime_dir) + dashboard_path = _prepare_grafana_dashboards(self.conf, runtime_dir) _render_grafana_provisioning(self.conf, runtime_dir, dashboard_path) return RuntimeFiles( @@ -920,18 +922,116 @@ def _render_grafana_provisioning( ) -def _stage_grafana_dashboards(conf: DictConfig, runtime_dir: Path) -> Path: - source = Path(str(OmegaConf.select(conf, "grafana.dashboards_dir"))).resolve() +def _prepare_grafana_dashboards(conf: DictConfig, runtime_dir: Path) -> Path: + """Materialize the dashboards Grafana loads into ``/dashboards``. + + The runtime directory is rebuilt from scratch on every start, so a removed + or renamed source file never leaves a stale copy behind. The base source is + then chosen by this precedence (see ``tools/grafana/framework/README.md``): + + 1. a non-empty ``grafana.dashboard_config`` renders that Jsonnet config; a + missing file or a Jsonnet error stops startup before Grafana runs; + 2. otherwise a ``grafana.dashboards_dir`` that points somewhere other than + the bundled default directory is copied as-is (legacy static workflow), + and no built-in Jsonnet runs; + 3. otherwise the Jsonnet entrypoint bundled in the installed package is + rendered — no CLI, no generated files on disk to keep in sync. + + In cases 1 and 3 the bundled static dashboards are staged first and the + Jsonnet dashboards are added alongside them as separate resources: the + Jsonnet render never overwrites a staged file, so a composition that would + land on an existing path aborts the startup instead. In case 2 only the + configured directory is copied. + + ``grafana.extra_dashboard_dir`` is merged on top of whichever base source + applies, last, and keeps its recursive copy, invalid-path and filename + collision behavior. + """ target = (runtime_dir / "dashboards").resolve() - target.mkdir(parents=True, exist_ok=True) - if source != target and source.exists(): + if target.exists(): shutil.rmtree(target) - target.mkdir(parents=True) + target.mkdir(parents=True) + + dashboard_config = OmegaConf.select(conf, "grafana.dashboard_config") + configured = str(dashboard_config).strip() if dashboard_config is not None else "" + + if configured: + _stage_bundled_dashboards(target) + _materialize_dashboards( + Path(configured).expanduser(), _jsonnet_output_dir(target) + ) + else: + dashboards_dir = ( + Path(str(OmegaConf.select(conf, "grafana.dashboards_dir"))) + .expanduser() + .resolve() + ) + bundled_dir = MonitorPaths.GRAFANA_DASHBOARDS_DIR.resolve() + if dashboards_dir == bundled_dir: + _stage_bundled_dashboards(target) + _materialize_dashboards( + MonitorPaths.GRAFANA_JSONNET_ENTRYPOINT, _jsonnet_output_dir(target) + ) + else: + if not dashboards_dir.exists(): + raise RuntimeError( + f"Grafana dashboards directory {str(dashboards_dir)!r} " + "does not exist." + ) + if not dashboards_dir.is_dir(): + raise RuntimeError( + f"Grafana dashboards path {str(dashboards_dir)!r} " + "is not a directory." + ) + _copy_dashboard_directory(dashboards_dir, target) + + _merge_extra_dashboards(conf, target) + return target + + +def _jsonnet_output_dir(target: Path) -> Path: + """Folder the Jsonnet compositions are rendered into inside ``target``.""" + return target / MonitorPaths.GRAFANA_JSONNET_OUTPUT_SUBDIR + + +def _stage_bundled_dashboards(target: Path) -> None: + """Stage the dashboards that ship inside the installed package. + + The whole bundled directory is copied, including the ``verl`` folder that + holds the committed static dashboards. Those stay the official dashboards + users already have; the Jsonnet render adds further dashboards next to them + instead of replacing them, so nothing here may be skipped or overwritten. + """ + source = MonitorPaths.GRAFANA_DASHBOARDS_DIR.resolve() + if source != target and source.is_dir(): _copy_dashboard_directory(source, target) + +def _materialize_dashboards(config: Path, target: Path) -> None: + """Render a Jsonnet composition config into the runtime dashboards dir. + + The rendering is additive (``overwrite=False``): the bundled static + dashboards are already staged, so a composition whose output filename + collides with one of them fails the startup before Grafana runs instead of + replacing a committed dashboard. + + The renderer keeps the config path and the Jsonnet evaluation context in + its message, so a broken config fails the startup with an actionable error + instead of starting Grafana with an empty or partial dashboard set. + """ + if not config.is_file(): + raise RuntimeError(f"Grafana dashboard config '{config}' does not exist.") + try: + materialize_dashboards(config, target, overwrite=False) + except JsonnetRenderError as error: + raise RuntimeError(f"Grafana dashboard generation failed: {error}") from error + + +def _merge_extra_dashboards(conf: DictConfig, target: Path) -> None: + """Copy ``grafana.extra_dashboard_dir`` onto the staged base source.""" extra = OmegaConf.select(conf, "grafana.extra_dashboard_dir") if extra is None: - return target + return planned_json = { path.relative_to(target) @@ -940,7 +1040,7 @@ def _stage_grafana_dashboards(conf: DictConfig, runtime_dir: Path) -> Path: } extra_source = Path(str(extra)).resolve() if extra_source == target: - return target + return if not extra_source.exists(): raise RuntimeError( f"Extra Grafana dashboard directory {str(extra_source)!r} does not exist." @@ -961,7 +1061,6 @@ def _stage_grafana_dashboards(conf: DictConfig, runtime_dir: Path) -> Path: planned_json.add(relative_path) _copy_dashboard_directory(extra_source, target) - return target def _copy_dashboard_directory(source: Path, target: Path) -> None: diff --git a/rl_insight/utils/constants.py b/rl_insight/utils/constants.py index 016fcc17..e559037c 100644 --- a/rl_insight/utils/constants.py +++ b/rl_insight/utils/constants.py @@ -35,6 +35,18 @@ class MonitorPaths: GRAFANA_CONFIG_FILE = SERVICES_DIR / "grafana" / "grafana.ini" GRAFANA_PROVISIONING_DIR = SERVICES_DIR / "grafana" / "provisioning" GRAFANA_DASHBOARDS_DIR = SERVICES_DIR / "grafana" / "dashboards" + #: Production Jsonnet sources shipped inside the installed package. The + #: runtime renders :data:`GRAFANA_JSONNET_ENTRYPOINT` at startup and adds + #: the rendered dashboards beside the committed static JSON under + #: :data:`GRAFANA_DASHBOARDS_DIR`, so a Grafana folder holds both a bundled + #: static dashboard and its startup-generated ``_jsonnet`` counterpart. + GRAFANA_JSONNET_DIR = SERVICES_DIR / "grafana" / "jsonnet" + GRAFANA_JSONNET_ENTRYPOINT = GRAFANA_JSONNET_DIR / "dashboards.jsonnet" + #: Grafana folder the rendered compositions are written into. The runtime + #: renders into the same folder the bundled static dashboards have always + #: used, so the folder structure Grafana shows does not change. + GRAFANA_JSONNET_OUTPUT_SUBDIR = "verl" + GRAFANA_JSONNET_OUTPUT_DIR = GRAFANA_DASHBOARDS_DIR / GRAFANA_JSONNET_OUTPUT_SUBDIR class MonitorRayActor: diff --git a/tests/monitor/ut/test_grafana_dashboards.py b/tests/monitor/ut/test_grafana_dashboards.py index e3adce51..5cc17a7d 100644 --- a/tests/monitor/ut/test_grafana_dashboards.py +++ b/tests/monitor/ut/test_grafana_dashboards.py @@ -12,25 +12,185 @@ # See the License for the specific language governing permissions and # limitations under the License. -"""Unit tests for Grafana dashboard directory staging.""" +"""Unit tests for Grafana dashboard preparation at server startup. + +The runtime evaluates the Jsonnet sources shipped inside the installed package +in process, so no test here runs a generate/Jsonnet CLI. The composition +registry ships empty in this change, which is exactly what the empty-registry +and static-compatibility tests below exercise; the production compositions and +their business tests live in the change that depends on this one. +""" from __future__ import annotations +import json +import shutil from pathlib import Path import pytest from omegaconf import OmegaConf from rl_insight.server import runtime as runtime_module +from rl_insight.utils.constants import MonitorPaths + +BUNDLED_DASHBOARDS_DIR = MonitorPaths.GRAFANA_DASHBOARDS_DIR +JSONNET_DIR = MonitorPaths.GRAFANA_JSONNET_DIR +JSONNET_OUTPUT_DIR = MonitorPaths.GRAFANA_JSONNET_OUTPUT_DIR +#: Source files the runtime must never write to. +MONITORED_SOURCE_FILES = ( + MonitorPaths.GRAFANA_JSONNET_ENTRYPOINT, + JSONNET_DIR / "dashboard_compositions.libsonnet", + JSONNET_OUTPUT_DIR / "verl_tainer_v1_with_vllm_engine.json", + JSONNET_OUTPUT_DIR / "verl_tainer_v1_with_sglang_engine.json", + MonitorPaths.CONFIG_FILE, +) -def _conf(builtin: Path, extra: Path | None = None): - grafana = {"dashboards_dir": str(builtin)} + +def _conf(builtin: Path | None = None, extra: Path | None = None, dashboard_config=""): + grafana = { + "dashboards_dir": str( + builtin if builtin is not None else BUNDLED_DASHBOARDS_DIR + ) + } + if dashboard_config: + grafana["dashboard_config"] = dashboard_config if extra is not None: grafana["extra_dashboard_dir"] = str(extra) return OmegaConf.create({"grafana": grafana}) +def _renderer(): + """Import the packaged renderer, skipping when the framework is absent.""" + return pytest.importorskip("rl_insight.grafana.renderer") + + +def _write_custom_config(directory: Path, name: str = "custom_board") -> Path: + """Write a user composition config next to a copy of the framework assets. + + The assets are copied instead of addressed by a relative path because the + checkout and the temporary directory can be on different Windows drives, + where ``os.path.relpath`` raises. + """ + for asset in ("composer.libsonnet", "viz.libsonnet"): + shutil.copy(JSONNET_DIR / "framework" / asset, directory / asset) + config = directory / "custom.jsonnet" + config.write_text( + "local composer = import 'composer.libsonnet';\n" + f"{{ {name}: composer.compose([{{\n" + " panels: [{ key: 'toy.panel', outputKey: 'toy-panel', id: 1,\n" + " title: 'Toy panel', queries: [{ expr: 'toy_metric' }] }],\n" + "}], {\n" + f" metadata: {{ name: '{name}', labels: {{}}, annotations: {{}} }},\n" + f" title: '{name}', tags: ['RL-Insight'], spec: {{}},\n" + " variableOrder: [], rowOrder: [],\n" + "}) }\n", + encoding="utf-8", + ) + return config + + +def _read_json(path: Path) -> dict: + return json.loads(path.read_text(encoding="utf-8")) + + +def test_prepare_stages_static_dashboards_with_an_empty_composition_set( + tmp_path, +) -> None: + _renderer() + # An explicitly empty composition set is valid: nothing is rendered and the + # bundled static dashboards are still staged, byte for byte. + config = tmp_path / "empty.jsonnet" + config.write_text("{}\n", encoding="utf-8") + + staged = runtime_module._prepare_grafana_dashboards( + _conf(dashboard_config=str(config)), tmp_path / "runtime" + ) + + assert list(staged.glob("*/*_jsonnet.json")) == [] + for name in ( + "verl_tainer_v1_with_vllm_engine", + "verl_tainer_v1_with_sglang_engine", + ): + staged_file = staged / "verl" / f"{name}.json" + staged_source = JSONNET_OUTPUT_DIR / f"{name}.json" + assert staged_file.read_bytes() == staged_source.read_bytes() + assert (staged / "quick_start_demo" / "quick_start_demo.json").is_file() + assert (staged / "agent_loop_trajectory" / "agent_loop_trajectory.json").is_file() + assert ( + staged / "verl-omni" / "verl_omni_trainer_v1_with_vllm_omni_engine.json" + ).is_file() + + +def test_prepare_writes_only_into_the_runtime_directory(tmp_path) -> None: + _renderer() + before = {path: path.read_bytes() for path in MONITORED_SOURCE_FILES} + + runtime_module._prepare_grafana_dashboards(_conf(), tmp_path / "runtime") + + assert {path: path.read_bytes() for path in MONITORED_SOURCE_FILES} == before + + +def test_prepare_renders_explicit_dashboard_config(tmp_path) -> None: + _renderer() + config = _write_custom_config(tmp_path) + + staged = runtime_module._prepare_grafana_dashboards( + _conf(dashboard_config=str(config)), tmp_path / "runtime" + ) + + written = _read_json( + staged / MonitorPaths.GRAFANA_JSONNET_OUTPUT_SUBDIR / "custom_board.json" + ) + assert written["spec"]["title"] == "custom_board" + assert written["metadata"]["name"] == "custom_board" + # The bundled static dashboards are staged as well: the custom composition + # replaces the bundled entrypoint, not the other bundled dashboards. + assert (staged / "verl" / "verl_tainer_v1_with_vllm_engine.json").is_file() + assert (staged / "quick_start_demo" / "quick_start_demo.json").is_file() + + +@pytest.mark.parametrize( + ("config_name", "body", "expected_error"), + [ + ("nope.jsonnet", None, "does not exist"), + ("broken.jsonnet", "{ this is not jsonnet", "generation failed"), + ], +) +def test_prepare_rejects_bad_dashboard_config( + tmp_path, config_name: str, body: str | None, expected_error: str +) -> None: + _renderer() + config = tmp_path / config_name + if body is not None: + config.write_text(body, encoding="utf-8") + + with pytest.raises(RuntimeError, match=expected_error) as error: + runtime_module._prepare_grafana_dashboards( + _conf(dashboard_config=str(config)), tmp_path / "runtime" + ) + + assert str(config) in str(error.value) + + +def test_prepare_rejects_a_custom_config_that_hits_a_static_dashboard(tmp_path) -> None: + _renderer() + static = JSONNET_OUTPUT_DIR / "verl_tainer_v1_with_vllm_engine.json" + before = static.read_bytes() + config = _write_custom_config(tmp_path, name="verl_tainer_v1_with_vllm_engine") + runtime_dir = tmp_path / "runtime" + + with pytest.raises(RuntimeError, match="refusing to overwrite"): + runtime_module._prepare_grafana_dashboards( + _conf(dashboard_config=str(config)), runtime_dir + ) + + # Nothing was overwritten: neither the packaged source nor the staged copy. + staged = runtime_dir / "dashboards" / "verl" / static.name + assert static.read_bytes() == before + assert staged.read_bytes() == before + + def test_stage_copies_builtin_and_extra_without_parsing_json(tmp_path) -> None: builtin = tmp_path / "builtin" extra = tmp_path / "extra" @@ -39,7 +199,7 @@ def test_stage_copies_builtin_and_extra_without_parsing_json(tmp_path) -> None: (builtin / "verl" / "board.json").write_text("{}", encoding="utf-8") (extra / "custom" / "router.json").write_text("not valid json", encoding="utf-8") - staged = runtime_module._stage_grafana_dashboards( + staged = runtime_module._prepare_grafana_dashboards( _conf(builtin, extra), tmp_path / "runtime" ) @@ -60,10 +220,7 @@ def test_stage_rejects_json_collision_before_copying_extra(tmp_path) -> None: runtime_dir = tmp_path / "runtime" with pytest.raises(RuntimeError, match="already exists in runtime dashboards"): - runtime_module._stage_grafana_dashboards( - _conf(builtin, extra), - runtime_dir, - ) + runtime_module._prepare_grafana_dashboards(_conf(builtin, extra), runtime_dir) staged = runtime_dir / "dashboards" assert (staged / "shared" / "board.json").read_text(encoding="utf-8") == "builtin" @@ -80,15 +237,36 @@ def test_stage_refreshes_extra_in_shared_builtin_directory(tmp_path) -> None: extra_dashboard.write_text("first", encoding="utf-8") runtime_dir = tmp_path / "runtime" - runtime_module._stage_grafana_dashboards(_conf(builtin, extra), runtime_dir) + runtime_module._prepare_grafana_dashboards(_conf(builtin, extra), runtime_dir) extra_dashboard.write_text("second", encoding="utf-8") - staged = runtime_module._stage_grafana_dashboards( + staged = runtime_module._prepare_grafana_dashboards( _conf(builtin, extra), runtime_dir ) assert (staged / "shared" / "extra.json").read_text(encoding="utf-8") == "second" +def test_stage_drops_stale_files_between_starts(tmp_path) -> None: + builtin = tmp_path / "builtin" + (builtin / "verl").mkdir(parents=True) + board = builtin / "verl" / "board.json" + board.write_text("{}", encoding="utf-8") + runtime_dir = tmp_path / "runtime" + + runtime_module._prepare_grafana_dashboards(_conf(builtin), runtime_dir) + board.unlink() + staged = runtime_module._prepare_grafana_dashboards(_conf(builtin), runtime_dir) + + assert not (staged / "verl" / "board.json").exists() + + +def test_prepare_rejects_missing_legacy_dashboards_dir(tmp_path) -> None: + with pytest.raises(RuntimeError, match="does not exist"): + runtime_module._prepare_grafana_dashboards( + _conf(tmp_path / "absent"), tmp_path / "runtime" + ) + + @pytest.mark.parametrize( ("entry_type", "expected_error"), [("missing", "does not exist"), ("file", "is not a directory")], @@ -103,6 +281,39 @@ def test_stage_rejects_invalid_extra_directory( extra.write_text("not a directory", encoding="utf-8") with pytest.raises(RuntimeError, match=expected_error): - runtime_module._stage_grafana_dashboards( + runtime_module._prepare_grafana_dashboards( _conf(builtin, extra), tmp_path / "runtime" ) + + +def test_prepare_generated_and_extra_dashboards_coexist(tmp_path) -> None: + _renderer() + extra = tmp_path / "extra" + (extra / "custom").mkdir(parents=True) + (extra / "custom" / "router.json").write_text("{}", encoding="utf-8") + config = _write_custom_config(tmp_path) + + staged = runtime_module._prepare_grafana_dashboards( + _conf(extra=extra, dashboard_config=str(config)), tmp_path / "runtime" + ) + + assert (staged / "custom" / "router.json").is_file() + assert (staged / "verl" / "custom_board.json").is_file() + + +@pytest.mark.parametrize("name", ["verl_tainer_v1_with_vllm_engine", "custom_board"]) +def test_prepare_keeps_collision_detection_against_staged_dashboards( + tmp_path, name: str +) -> None: + _renderer() + extra = tmp_path / "extra" + (extra / "verl").mkdir(parents=True) + (extra / "verl" / f"{name}.json").write_text("{}", encoding="utf-8") + # The first name collides with a bundled static dashboard, the second with + # a dashboard rendered by the custom config; both are detected the same way. + conf = _conf(extra=extra) + if name == "custom_board": + conf = _conf(extra=extra, dashboard_config=str(_write_custom_config(tmp_path))) + + with pytest.raises(RuntimeError, match="already exists in runtime dashboards"): + runtime_module._prepare_grafana_dashboards(conf, tmp_path / "runtime") diff --git a/tests/monitor/ut/test_grafana_framework.py b/tests/monitor/ut/test_grafana_framework.py new file mode 100644 index 00000000..57ab6e9c --- /dev/null +++ b/tests/monitor/ut/test_grafana_framework.py @@ -0,0 +1,536 @@ +# Copyright (c) 2026 verl-project authors. +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. + +"""Tests for the generic Grafana dashboard composition framework. + +Every test renders a minimal toy composition it writes into ``tmp_path``, never +a production dashboard, through ``rl_insight.grafana.renderer``. The optional CLI +is covered only to prove it is a thin wrapper over that same core. +""" + +from __future__ import annotations + +import json +import shutil +import subprocess +import sys +from pathlib import Path + +import pytest + +from rl_insight.grafana import renderer + +REPO_ROOT = Path(__file__).resolve().parents[3] +GENERATE = REPO_ROOT / "tools" / "grafana" / "framework" / "generate.py" +FRAMEWORK_DIR = renderer.FRAMEWORK_DIR +COMPOSER = FRAMEWORK_DIR / "composer.libsonnet" +IMPORT_COMPOSER = f"local composer = import '{COMPOSER.as_posix()}';" + +#: A minimal toy module: one panel, one row, one variable. JSON is a Jsonnet +#: subset, so the templates below stay valid for any prefix. +MODULE = """local %(p)s = { + panels: [{ key: '%(p)s.panel', outputKey: '%(p)s-panel', id: %(id)d, + title: '%(p)s panel', queries: [{ expr: 'toy_%(p)s_metric' }] }], + rows: { '%(p)s-row': { kind: 'RowsLayoutRow', spec: { title: '%(p)s row', + collapse: false, layout: { kind: 'GridLayout', spec: { items: [{ + kind: 'GridLayoutItem', spec: { x: 0, y: 0, width: 12, height: 8, + element: { kind: 'ElementReference', name: '%(p)s.panel' } } }] } } } } }, + variables: { zone_%(p)s: { kind: 'ConstantVariable', + spec: { name: 'zone_%(p)s', label: 'Zone', value: 'z1', hide: 'dontHide' } } }, + %(tags)s +}; +""" + +#: An additive module: one panel plus `rowItems` extending a row it does not own. +EXTENSION = """local %(p)s = { + panels: [{ key: '%(p)s.panel', outputKey: '%(p)s-panel', id: %(id)d, + title: '%(p)s panel', queries: [{ expr: 'toy_%(p)s_metric' }] }], + rowItems: { '%(target)s': [{ kind: 'GridLayoutItem', spec: { x: 12, y: 0, + width: 12, height: 8, element: { kind: 'ElementReference', name: '%(key)s' } } }] }, +}; +""" + +DASHBOARD = ( + "{ metadata: { name: 'toy', uid: 'toy' }, title: 'Toy dashboard', " + "tags: ['example'], spec: {}, variableOrder: %s, rowOrder: %s }" +) +TWO_MODULES: list[tuple[str, int, list[str] | None]] = [ + ("m1", 1, None), + ("m2", 2, None), +] +FULL_DASHBOARD = DASHBOARD % ("['zone_m1']", "['m1-row', 'm2-row']") + + +def module_source(prefix: str, panel_id: int, tags: list[str] | None = None) -> str: + tags_line = "" if tags is None else "tags: " + json.dumps(tags) + "," + return MODULE % {"p": prefix, "id": panel_id, "tags": tags_line} + + +def extension_module_source( + prefix: str, panel_id: int, target_row: str, panel_key: str +) -> str: + return EXTENSION % { + "p": prefix, + "id": panel_id, + "target": target_row, + "key": panel_key, + } + + +def write_config(tmp_path: Path, body: str) -> Path: + tmp_path.mkdir(parents=True, exist_ok=True) + path = tmp_path / "compose.jsonnet" + path.write_text(body, encoding="utf-8") + return path + + +def compose_body( + modules: list[tuple[str, int, list[str] | None]], dashboard: str +) -> str: + return ( + "".join(module_source(*module) for module in modules) + + IMPORT_COMPOSER + + "\n{ compose: composer.compose([" + + ", ".join(prefix for prefix, _, _ in modules) + + "], " + + dashboard + + ") }\n" + ) + + +def compose_config( + tmp_path: Path, + modules: list[tuple[str, int, list[str] | None]], + dashboard: str, + replacements: list[tuple[str, str]] | None = None, +) -> Path: + """Write a config composing the given toy modules into ``tmp_path``.""" + body = compose_body(modules, dashboard) + for old, new in replacements or []: + body = body.replace(old, new) + return write_config(tmp_path, body) + + +def two_dashboard_config(tmp_path: Path) -> Path: + """A config rendering two dashboards, ``first`` before ``second``.""" + body = "".join(module_source(*module) for module in TWO_MODULES) + IMPORT_COMPOSER + body += "\n{\n" + for name, prefix in (("first", "m1"), ("second", "m2")): + body += ( + f" {name}: composer.compose([{prefix}], " + + DASHBOARD % (f"['zone_{prefix}']", f"['{prefix}-row']") + + "),\n" + ) + return write_config(tmp_path, body + "}\n") + + +def render(tmp_path: Path, config: Path, out: str) -> dict: + """Render through the package core and read the composed dashboard back.""" + renderer.materialize_dashboards(config, tmp_path / out) + return json.loads((tmp_path / out / "compose.json").read_text(encoding="utf-8")) + + +def run_generate(*args: str, expect: int = 0) -> subprocess.CompletedProcess[str]: + proc = subprocess.run( + [sys.executable, str(GENERATE), *args], capture_output=True, text=True + ) + assert proc.returncode == expect, f"{proc.returncode=}\n{proc.stderr}" + return proc + + +def test_render_is_byte_deterministic(tmp_path: Path) -> None: + config = compose_config(tmp_path, TWO_MODULES, FULL_DASHBOARD) + + renderer.materialize_dashboards(config, tmp_path / "run1") + renderer.materialize_dashboards(config, tmp_path / "run2") + + assert (tmp_path / "run1" / "compose.json").read_bytes() == ( + tmp_path / "run2" / "compose.json" + ).read_bytes() + + +def test_empty_registry_renders_no_dashboard(tmp_path: Path) -> None: + # The registry ships empty: an empty composition set is valid, renders zero + # dashboards and writes nothing, so the runtime can keep its static ones. + config = write_config(tmp_path, "{}\n") + output_dir = tmp_path / "out" + + assert renderer.render_dashboards(config) == {} + assert renderer.materialize_dashboards(config, output_dir) == [] + assert list(output_dir.glob("*.json")) == [] + + +def test_check_mode_passes_when_in_sync_and_reports_stale_files( + tmp_path: Path, +) -> None: + config = compose_config(tmp_path, TWO_MODULES, FULL_DASHBOARD) + expected_dir = tmp_path / "expected" + renderer.materialize_dashboards(config, expected_dir) + assert renderer.stale_dashboards(config, expected_dir) == [] + + stale = json.loads((expected_dir / "compose.json").read_text(encoding="utf-8")) + stale["spec"]["title"] = "tampered" + (expected_dir / "compose.json").write_text( + json.dumps(stale, ensure_ascii=False, indent=2) + "\n", encoding="utf-8" + ) + + assert renderer.stale_dashboards(config, expected_dir) == [ + expected_dir / "compose.json" + ] + + +def test_materialize_writes_dashboards_and_honours_overwrite(tmp_path: Path) -> None: + config = two_dashboard_config(tmp_path) + output_dir = tmp_path / "nested" / "out" + + # Additive mode writes every dashboard and creates the output directory. + written = renderer.materialize_dashboards(config, output_dir, overwrite=False) + assert written == [output_dir / "first.json", output_dir / "second.json"] + for path in written: + assert json.loads(path.read_text(encoding="utf-8"))["spec"]["title"] == ( + "Toy dashboard" + ) + + # The default stays a plain write that replaces the target path. + for path in written: + path.write_text("old\n", encoding="utf-8") + assert renderer.materialize_dashboards(config, output_dir) == written + assert "Toy dashboard" in written[0].read_text(encoding="utf-8") + + +def test_materialize_without_overwrite_rejects_existing_targets( + tmp_path: Path, +) -> None: + # Every target is checked before the first write: a collision on the second + # dashboard leaves the first one unwritten and both paths are reported. + config = two_dashboard_config(tmp_path) + output_dir = tmp_path / "out" + output_dir.mkdir() + for name in ("first", "second"): + (output_dir / f"{name}.json").write_text("pre-existing\n", encoding="utf-8") + + with pytest.raises(renderer.JsonnetRenderError) as excinfo: + renderer.materialize_dashboards(config, output_dir, overwrite=False) + + message = str(excinfo.value) + assert str(output_dir / "first.json") in message + assert str(output_dir / "second.json") in message + assert (output_dir / "first.json").read_text(encoding="utf-8") == "pre-existing\n" + + +@pytest.mark.parametrize( + ("body", "expected_error"), + [(None, "composition config not found"), ("[1, 2, 3]\n", "object of dashboards")], +) +def test_invalid_config_is_rejected_with_its_path( + tmp_path: Path, body: str | None, expected_error: str +) -> None: + config = ( + tmp_path / "absent.jsonnet" if body is None else write_config(tmp_path, body) + ) + + with pytest.raises(renderer.JsonnetRenderError) as excinfo: + renderer.render_dashboards(config) + + assert expected_error in str(excinfo.value) + assert str(config) in str(excinfo.value) + + +def test_evaluation_failure_names_the_config_and_the_jsonnet_context( + tmp_path: Path, +) -> None: + config = write_config(tmp_path, "{ compose: error 'toy composition failure' }\n") + + with pytest.raises(renderer.JsonnetRenderError) as excinfo: + renderer.render_dashboards(config) + + message = str(excinfo.value) + assert str(config) in message + assert "runtime error: toy composition failure" in message + + +def test_rendering_never_shells_out( + tmp_path: Path, monkeypatch: pytest.MonkeyPatch +) -> None: + # The core must evaluate in-process: no Jsonnet CLI, no subprocess, no + # gojsonnet binding. Break every escape hatch and render anyway. + config = compose_config(tmp_path, TWO_MODULES, FULL_DASHBOARD) + source = Path(renderer.__file__).read_text(encoding="utf-8") + assert "subprocess" not in source + assert "gojsonnet" not in source + + def explode(*args: object, **kwargs: object) -> None: + raise AssertionError("the renderer must not spawn a subprocess") + + monkeypatch.setattr(subprocess, "run", explode) + monkeypatch.setattr(subprocess, "Popen", explode) + monkeypatch.setattr(shutil, "which", lambda name: None) + + assert set(renderer.render_dashboards(config)) == {"compose"} + + +def test_framework_assets_ship_inside_the_installed_package() -> None: + assert FRAMEWORK_DIR.is_dir() + for name in ("composer.libsonnet", "viz.libsonnet"): + assert (FRAMEWORK_DIR / name).is_file() + # No second copy of the framework outside the package. + assert not ( + REPO_ROOT / "tools" / "grafana" / "framework" / "composer.libsonnet" + ).exists() + + +def test_layout_references_resolve_across_modules(tmp_path: Path) -> None: + # The second module's layout row references the first module's panel by + # `key`; the composer rewrites it to the panel's `outputKey` when rendering. + config = compose_config( + tmp_path, + TWO_MODULES, + FULL_DASHBOARD, + replacements=[("name: 'm2.panel'", "name: 'm1.panel'")], + ) + + composed = render(tmp_path, config, "out")["spec"] + + assert set(composed["elements"]) == {"m1-panel", "m2-panel"} + second_row = composed["layout"]["spec"]["rows"][1] + referenced = second_row["spec"]["layout"]["spec"]["items"][0]["spec"]["element"] + assert referenced == {"kind": "ElementReference", "name": "m1-panel"} + + +CONFLICT_CASES = [ + ("duplicate panel key(s)", [("key: 'm2.panel'", "key: 'm1.panel'")]), + ("duplicate panel outputKey(s)", [("m2-panel", "m1-panel")]), + ("duplicate panel id(s)", [("id: 2,", "id: 1,")]), + ("duplicate row name(s)", [("m2-row", "m1-row")]), + ("duplicate variable name(s)", [("zone_m2", "zone_m1")]), +] + + +@pytest.mark.parametrize(("expected_error", "replacements"), CONFLICT_CASES) +def test_composition_conflicts_are_rejected( + tmp_path: Path, expected_error: str, replacements: list[tuple[str, str]] +) -> None: + config = compose_config( + tmp_path, + TWO_MODULES, + DASHBOARD % ("['zone_m1']", "['m1-row']"), + replacements=replacements, + ) + + with pytest.raises(renderer.JsonnetRenderError) as excinfo: + renderer.render_dashboards(config) + + assert expected_error in str(excinfo.value) + + +@pytest.mark.parametrize( + ("dashboard", "expected_error"), + [ + ( + DASHBOARD % ("['zone_m1', 'missing_var']", "['m1-row']"), + "unknown variable(s) in variableOrder: missing_var", + ), + ( + DASHBOARD % ("['zone_m1']", "['m1-row', 'missing_row']"), + "unknown row(s) in rowOrder: missing_row", + ), + ], +) +def test_unknown_order_entries_are_rejected( + tmp_path: Path, dashboard: str, expected_error: str +) -> None: + config = compose_config(tmp_path, TWO_MODULES, dashboard) + + with pytest.raises(renderer.JsonnetRenderError) as excinfo: + renderer.render_dashboards(config) + + assert expected_error in str(excinfo.value) + + +def test_composition_order_decides_tag_and_variable_order(tmp_path: Path) -> None: + def composed( + modules: list[tuple[str, int, list[str] | None]], + variable_order: str, + out: str, + ) -> dict: + config = compose_config( + tmp_path / f"src-{out}", + modules, + DASHBOARD % (variable_order, "['m1-row']"), + ) + return render(tmp_path, config, f"out-{out}")["spec"] + + tagged: list[tuple[str, int, list[str] | None]] = [ + ("m1", 1, ["one"]), + ("m2", 2, ["two"]), + ] + forward = composed(tagged, "['zone_m1', 'zone_m2']", "fwd") + reversed_modules = composed( + [("m2", 2, ["two"]), ("m1", 1, ["one"])], "['zone_m2', 'zone_m1']", "rev" + ) + + assert forward["tags"] == ["example", "one", "two"] + assert reversed_modules["tags"] == ["example", "two", "one"] + assert [item["spec"]["name"] for item in forward["variables"]] == [ + "zone_m1", + "zone_m2", + ] + assert [item["spec"]["name"] for item in reversed_modules["variables"]] == [ + "zone_m2", + "zone_m1", + ] + + +def row_item_names(composed: dict, row_name: str) -> list[str]: + row = next( + candidate + for candidate in composed["spec"]["layout"]["spec"]["rows"] + if candidate["spec"]["title"] == row_name + ) + return [ + item["spec"]["element"]["name"] + for item in row["spec"]["layout"]["spec"]["items"] + ] + + +def extension_config( + tmp_path: Path, module_order: list[str], extra_sources: str +) -> Path: + """A config whose base module ``m1`` owns the row the extensions append to.""" + body = ( + module_source("m1", 1) + + extra_sources + + IMPORT_COMPOSER + + "\n{ compose: composer.compose([" + + ", ".join(module_order) + + "], " + + DASHBOARD % ("['zone_m1']", "['m1-row']") + + ") }\n" + ) + return write_config(tmp_path, body) + + +def test_row_items_single_extension_appends_and_resolves(tmp_path: Path) -> None: + # The extension module owns no rows (panels + rowItems only) and appends one + # item to the row owned by the base module; the appended ElementReference + # resolves through key -> outputKey like any other. + config = extension_config( + tmp_path, + ["m1", "m1x"], + extension_module_source("m1x", 11, "m1-row", "m1x.panel"), + ) + + composed = render(tmp_path, config, "out") + + assert set(composed["spec"]["elements"]) == {"m1-panel", "m1x-panel"} + assert row_item_names(composed, "m1 row") == ["m1-panel", "m1x-panel"] + # rowItems never creates rows: the only row is the one m1 owns. + assert len(composed["spec"]["layout"]["spec"]["rows"]) == 1 + + +def test_row_items_append_in_composition_order(tmp_path: Path) -> None: + sources = extension_module_source("e1", 11, "m1-row", "e1.panel") + ( + extension_module_source("e2", 12, "m1-row", "e2.panel") + ) + + forward = render( + tmp_path, + extension_config(tmp_path / "src-fwd", ["m1", "e1", "e2"], sources), + "out-fwd", + ) + reversed_order = render( + tmp_path, + extension_config(tmp_path / "src-rev", ["m1", "e2", "e1"], sources), + "out-rev", + ) + + assert row_item_names(forward, "m1 row") == ["m1-panel", "e1-panel", "e2-panel"] + assert row_item_names(reversed_order, "m1 row") == [ + "m1-panel", + "e2-panel", + "e1-panel", + ] + + +@pytest.mark.parametrize( + ("target", "break_layout", "expected_error"), + [ + ("missing-row", False, "unknown row extension target: missing-row"), + ("m1-row", True, "unsupported row extension target"), + ], +) +def test_row_items_invalid_targets_are_rejected( + tmp_path: Path, target: str, break_layout: bool, expected_error: str +) -> None: + # A target row that does not exist, or whose layout is not a GridLayout, + # must fail loudly instead of silently ignoring the extension. + base = module_source("m1", 1) + if break_layout: + base = base.replace("kind: 'GridLayout'", "kind: 'RowsLayout'") + body = ( + base + + extension_module_source("e1", 11, target, "m1.panel") + + IMPORT_COMPOSER + + "\n{ compose: composer.compose([m1, e1], " + + DASHBOARD % ("['zone_m1']", "['m1-row']") + + ") }\n" + ) + config = write_config(tmp_path, body) + + with pytest.raises(renderer.JsonnetRenderError) as excinfo: + renderer.render_dashboards(config) + + assert expected_error in str(excinfo.value) + + +def test_cli_writes_exactly_what_the_core_renders(tmp_path: Path) -> None: + config = compose_config(tmp_path, TWO_MODULES, FULL_DASHBOARD) + cli_out = tmp_path / "cli-out" + + run_generate("--config", str(config), "--out-dir", str(cli_out)) + + expected = renderer.generated_text(renderer.render_dashboards(config)["compose"]) + assert (cli_out / "compose.json").read_text(encoding="utf-8") == expected + + +def test_cli_check_matches_the_core_and_reports_stale_files(tmp_path: Path) -> None: + config = compose_config(tmp_path, TWO_MODULES, FULL_DASHBOARD) + expected_dir = tmp_path / "expected" + + run_generate("--config", str(config), "--out-dir", str(expected_dir)) + run_generate( + "--config", str(config), "--check", "--expected-dir", str(expected_dir) + ) + + (expected_dir / "compose.json").write_text("{}\n", encoding="utf-8") + proc = run_generate( + "--config", + str(config), + "--check", + "--expected-dir", + str(expected_dir), + expect=1, + ) + assert "compose.json" in proc.stderr + + +def test_cli_reports_evaluation_failures_with_the_config_path(tmp_path: Path) -> None: + config = write_config(tmp_path, "{ compose: error 'toy composition failure' }\n") + + proc = run_generate( + "--config", str(config), "--out-dir", str(tmp_path / "out"), expect=2 + ) + + assert str(config) in proc.stderr + assert "runtime error: toy composition failure" in proc.stderr diff --git a/tools/grafana/framework/README.md b/tools/grafana/framework/README.md new file mode 100644 index 00000000..da5ff5b1 --- /dev/null +++ b/tools/grafana/framework/README.md @@ -0,0 +1,96 @@ +# Grafana dashboard framework + +RL-Insight builds its Grafana dashboards from **Jsonnet** sources shipped inside +the `rl_insight` package and materializes them when the service starts. Grafana +keeps reading plain JSON, and nobody runs a generator or commits generated +files. + +This change is the mechanism layer: the composition registry ships **empty**, so +the runtime stages the bundled static dashboards and registers no Jsonnet one. +The production compositions come with the change that depends on this one. + +## How it works + +![Jsonnet dashboard file architecture](diagrams/jsonnet-files.svg) +![RL-Insight dashboard startup flow](diagrams/server-start.svg) + +`rl-insight server start` rebuilds `/dashboards` from scratch: the +bundled static dashboards are staged byte for byte, then +`rl_insight.grafana.renderer` evaluates the entrypoint in process through the +`rjsonnet` binding and writes one `.json` per registered +composition. Rendering is additive — an existing file is never overwritten and a +collision fails startup. The runtime never writes to the package sources, the +config or the repository, Grafana provisioning keeps reading plain JSON from the +runtime directory, and no Jsonnet CLI, Go toolchain or compiler is required. + +### Source layout + +| Path | Purpose | +| --- | --- | +| `jsonnet/dashboards.jsonnet` | Sole entrypoint; composes every registered composition. Normally unchanged. | +| `jsonnet/dashboard_compositions.libsonnet` | Composition registry; empty in this change. | +| `jsonnet/dashboards/*.libsonnet` | Reusable content modules (`panels`, `rows`, `rowItems`, `variables`, `tags`). | +| `jsonnet/framework/composer.libsonnet` | Ordered merge, conflict detection, late layout reference resolution. | +| `jsonnet/framework/viz.libsonnet` | Shared visualization defaults and RFC 7396 viz patches. | +| `rl_insight/grafana/renderer.py` | `render_dashboards`, `materialize_dashboards`, `stale_dashboards`. | +| `tools/grafana/framework/generate.py` | Optional CLI wrapper over the same renderer (debug/CI only). | + +### Configuration precedence + +| `config.yaml` key | Effect | +| --- | --- | +| `grafana.dashboard_config` | Non-empty: render that Jsonnet config instead of the bundled entrypoint. A missing file or Jsonnet error stops startup before Grafana runs. | +| `grafana.dashboards_dir` | Legacy static directory. When it is not the bundled default it is copied as-is and no Jsonnet runs. | +| `grafana.extra_dashboard_dir` | Merged last, with recursive copy and filename-collision failure. | + +## Extending the framework (example only) + +```jsonnet +// jsonnet/dashboards/foo.libsonnet: plain data, no behaviour +{ + panels: [{ + key: 'foo.panel', outputKey: 'panel-foo', id: 900, title: 'Foo metric', + queries: [{ expr: 'foo_metric' }], + }], + rows: { 'foo row': { kind: 'RowsLayoutRow', spec: { title: 'foo row', + collapse: false, layout: { kind: 'GridLayout', spec: { items: [{ + kind: 'GridLayoutItem', spec: { x: 0, y: 0, width: 24, height: 8, + element: { kind: 'ElementReference', name: 'foo.panel' } } }] } } } } }, + variables: { foo_instance: { kind: 'ConstantVariable', spec: { + name: 'foo_instance', label: 'Instance', value: 'i0', hide: 'dontHide' } } }, +} +``` + +```jsonnet +// jsonnet/dashboard_compositions.libsonnet: specialize the empty registry +local foo = import 'dashboards/foo.libsonnet'; + +{ compositions: { foo_dashboard: { + modules: [foo], + dashboard: { + metadata: { name: 'foo-dashboard-id', labels: {}, annotations: {} }, + title: 'foo_dashboard', tags: ['RL-Insight', 'foo'], spec: {}, + variableOrder: ['foo_instance'], rowOrder: ['foo row'], + }, +} } } +``` + +`rowItems` lets a module append panels to a row another module owns; see +`composer.libsonnet` for the full module schema and composition rules. + +## Optional CLI + +`tools/grafana/framework/generate.py` renders through the same core for local +debugging and CI. The server never calls it: + +```bash +python tools/grafana/framework/generate.py --config --out-dir +python tools/grafana/framework/generate.py --config --check --expected-dir +``` + +Exit codes: `0` success, `1` stale files found by `--check`, `2` rendering +failed. + +## Tests + +`tests/monitor/ut/` covers the composer, renderer, CLI and startup preparation. diff --git a/tools/grafana/framework/diagrams/jsonnet-files.archify.json b/tools/grafana/framework/diagrams/jsonnet-files.archify.json new file mode 100644 index 00000000..78e615c1 --- /dev/null +++ b/tools/grafana/framework/diagrams/jsonnet-files.archify.json @@ -0,0 +1 @@ +{"schema_version":1,"diagram_type":"architecture","meta":{"title":"Jsonnet dashboard file architecture","output":"jsonnet-files.html","locale":"en","quality_profile":"showcase","repository":{"url":"https://github.com/verl-project/rl-insight.git","provider":"github","link_mode":"web","revision":"acd34577f9040b7636da3a89539da869c011a221"}},"components":[{"id":"modules","type":"backend","label":"dashboards/*.libsonnet","sublabel":"Reusable panel, row and variable modules","pos":[40,80],"size":[210,72],"sources":[{"path":"rl_insight/config/services/grafana/jsonnet/dashboards.jsonnet","line":8}]},{"id":"composition","type":"backend","label":"Composition","sublabel":"One dashboard: modules + dashboard config","pos":[320,80],"size":[220,72],"sources":[{"path":"rl_insight/config/services/grafana/jsonnet/dashboard_compositions.libsonnet","line":12}]},{"id":"registry","type":"database","label":"dashboard_compositions.libsonnet","sublabel":"#174 empty; #173 registers production","pos":[620,80],"size":[240,72],"sources":[{"path":"rl_insight/config/services/grafana/jsonnet/dashboard_compositions.libsonnet","line":3}]},{"id":"entry","type":"backend","label":"dashboards.jsonnet","sublabel":"Stable sole entrypoint","pos":[760,240],"size":[240,72],"sources":[{"path":"rl_insight/config/services/grafana/jsonnet/dashboards.jsonnet","line":4}]},{"id":"composer","type":"backend","label":"framework/composer.libsonnet","sublabel":"Generic ordered composition","pos":[320,240],"size":[240,72],"sources":[{"path":"rl_insight/config/services/grafana/jsonnet/dashboards.jsonnet","line":8}]},{"id":"viz","type":"backend","label":"framework/viz.libsonnet","sublabel":"Visualization defaults","pos":[0,240],"size":[210,72],"sources":[{"path":"rl_insight/config/services/grafana/jsonnet/framework/viz.libsonnet","line":1}]},{"id":"renderer","type":"backend","label":"renderer.py","sublabel":"In-process rjsonnet materialization","pos":[620,400],"size":[240,72],"sources":[{"path":"rl_insight/grafana/renderer.py","line":22}]},{"id":"runtime","type":"external","label":"runtime.py","sublabel":"Startup integration","pos":[320,400],"size":[240,72],"sources":[{"path":"rl_insight/server/runtime.py","line":925}]},{"id":"production","type":"external","label":"#173 production modules","sublabel":"Adds real vLLM/SGLang compositions","pos":[980,80],"size":[240,72],"sources":[{"path":"rl_insight/config/services/grafana/jsonnet/dashboard_compositions.libsonnet","line":7}]}],"connections":[{"id":"modules-composition","from":"modules","to":"composition","label":"selected by"},{"id":"composition-registry","from":"composition","to":"registry","label":"registered as"},{"id":"production-registry","from":"production","to":"registry","label":"extends","variant":"dashed"},{"id":"registry-entry","from":"registry","to":"entry","label":"imported by"},{"id":"entry-composer","from":"entry","to":"composer","label":"compose"},{"id":"composer-viz","from":"composer","to":"viz","label":"defaults"},{"id":"entry-renderer","from":"entry","to":"renderer","label":"evaluated by"},{"id":"runtime-renderer","from":"runtime","to":"renderer","label":"imports"}]} diff --git a/tools/grafana/framework/diagrams/jsonnet-files.svg b/tools/grafana/framework/diagrams/jsonnet-files.svg new file mode 100644 index 00000000..5e01df97 --- /dev/null +++ b/tools/grafana/framework/diagrams/jsonnet-files.svg @@ -0,0 +1 @@ +Jsonnet dashboard file architectureAn architecture diagram generated by Archify.dashboards/*.libsonnet · Reusable panel, row and variable modules · Architecture componentdashboards/*.libsonnetReusable panel, row and variable modulesComposition · One dashboard: modules + dashboard config · Architecture componentCompositionOne dashboard: modules + dashboard configdashboard_compositions.libsonnet · #174 empty; #173 registers production · Architecture componentdashboard_compositions.libsonnet#174 empty; #173 registers productiondashboards.jsonnet · Stable sole entrypoint · Architecture componentdashboards.jsonnetStable sole entrypointframework/composer.libsonnet · Generic ordered composition · Architecture componentframework/composer.libsonnetGeneric ordered compositionframework/viz.libsonnet · Visualization defaults · Architecture componentframework/viz.libsonnetVisualization defaultsrenderer.py · In-process rjsonnet materialization · Architecture componentrenderer.pyIn-process rjsonnet materializationruntime.py · Startup integration · Architecture componentruntime.pyStartup integration#173 production modules · Adds real vLLM/SGLang compositions · Architecture component#173 production modulesAdds real vLLM/SGLang compositionsselected byregistered asextendsimported bycomposedefaultsevaluated byimportsLegendBackendDatabaseExternal \ No newline at end of file diff --git a/tools/grafana/framework/diagrams/server-start.archify.json b/tools/grafana/framework/diagrams/server-start.archify.json new file mode 100644 index 00000000..08f48e96 --- /dev/null +++ b/tools/grafana/framework/diagrams/server-start.archify.json @@ -0,0 +1 @@ +{"schema_version":1,"diagram_type":"architecture","meta":{"title":"RL-Insight dashboard startup flow","output":"server-start.html","locale":"en","quality_profile":"showcase","repository":{"url":"https://github.com/verl-project/rl-insight.git","provider":"github","link_mode":"web","revision":"acd34577f9040b7636da3a89539da869c011a221"}},"components":[{"id":"command","type":"external","label":"rl-insight server start","sublabel":"Normal user workflow","pos":[40,220],"size":[220,72],"sources":[{"path":"rl_insight/server/runtime.py","line":131}]},{"id":"runtime","type":"backend","label":"runtime.py","sublabel":"prepare_grafana_dashboards()","pos":[320,220],"size":[220,72],"sources":[{"path":"rl_insight/server/runtime.py","line":925}]},{"id":"static","type":"database","label":"Bundled static Dashboard JSON","sublabel":"Staged byte-for-byte first","pos":[320,80],"size":[220,72],"sources":[{"path":"rl_insight/server/runtime.py","line":997}]},{"id":"renderer","type":"backend","label":"renderer.py","sublabel":"rjsonnet in process","pos":[600,220],"size":[220,72],"sources":[{"path":"rl_insight/grafana/renderer.py","line":22}]},{"id":"registry","type":"database","label":"Jsonnet registry","sublabel":"#174 empty; #173 adds compositions","pos":[600,80],"size":[220,72],"sources":[{"path":"rl_insight/config/services/grafana/jsonnet/dashboard_compositions.libsonnet","line":3}]},{"id":"runtime-json","type":"database","label":"/dashboards/*.json","sublabel":"Static + generated JSON coexist","pos":[880,220],"size":[240,72],"sources":[{"path":"rl_insight/server/runtime.py","line":959}]},{"id":"provisioning","type":"backend","label":"Grafana provisioning","sublabel":"File provider scans runtime directory","pos":[880,360],"size":[240,72],"sources":[{"path":"rl_insight/server/runtime.py","line":131}]},{"id":"grafana","type":"external","label":"Grafana","sublabel":"Reads JSON only","pos":[880,500],"size":[240,72],"sources":[{"path":"rl_insight/server/runtime.py","line":131}]},{"id":"branch","type":"external","label":"#173 registered compositions","sublabel":"Adds Jsonnet dashboards alongside static","pos":[40,80],"size":[220,72],"sources":[{"path":"rl_insight/config/services/grafana/jsonnet/dashboard_compositions.libsonnet","line":7}]}],"connections":[{"id":"start-runtime","from":"command","to":"runtime","label":"prepare"},{"id":"stage-static","from":"static","to":"runtime","label":"copy first","variant":"dashed"},{"id":"registry-render","from":"registry","to":"renderer","label":"evaluate entrypoint","variant":"dashed"},{"id":"branch-registry","from":"branch","to":"registry","label":"#173 only","variant":"dashed"},{"id":"runtime-render","from":"runtime","to":"renderer","label":"materialize"},{"id":"runtime-output","from":"renderer","to":"runtime-json","label":"write JSON"},{"id":"static-output","from":"runtime","to":"runtime-json","label":"stage static JSON","variant":"dashed"},{"id":"output-provision","from":"runtime-json","to":"provisioning","label":"file provider"},{"id":"provision-grafana","from":"provisioning","to":"grafana","label":"load JSON"}]} diff --git a/tools/grafana/framework/diagrams/server-start.svg b/tools/grafana/framework/diagrams/server-start.svg new file mode 100644 index 00000000..be757510 --- /dev/null +++ b/tools/grafana/framework/diagrams/server-start.svg @@ -0,0 +1 @@ +RL-Insight dashboard startup flowAn architecture diagram generated by Archify.rl-insight server start · Normal user workflow · Architecture componentrl-insight server startNormal user workflowruntime.py · prepare_grafana_dashboards() · Architecture componentruntime.pyprepare_grafana_dashboards()Bundled static Dashboard JSON · Staged byte-for-byte first · Architecture componentBundled static Dashboard JSONStaged byte-for-byte firstrenderer.py · rjsonnet in process · Architecture componentrenderer.pyrjsonnet in processJsonnet registry · #174 empty; #173 adds compositions · Architecture componentJsonnet registry#174 empty; #173 adds compositions<runtime_dir>/dashboards/*.json · Static + generated JSON coexist · Architecture component<runtime_dir>/dashboards/*.jsonStatic + generated JSON coexistGrafana provisioning · File provider scans runtime directory · Architecture componentGrafana provisioningFile provider scans runtime directoryGrafana · Reads JSON only · Architecture componentGrafanaReads JSON only#173 registered compositions · Adds Jsonnet dashboards alongside static · Architecture component#173 registered compositionsAdds Jsonnet dashboards alongside staticpreparecopy firstevaluate entrypoint#173 onlymaterializewrite JSONstage static JSONfile providerload JSONLegendBackendDatabaseExternal \ No newline at end of file diff --git a/tools/grafana/framework/generate.py b/tools/grafana/framework/generate.py new file mode 100644 index 00000000..96d4e7e4 --- /dev/null +++ b/tools/grafana/framework/generate.py @@ -0,0 +1,100 @@ +# Copyright (c) 2026 verl-project authors. +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. + +"""Optional CLI over the installed-package Grafana dashboard renderer. + +Thin wrapper: evaluation, serialization and the ``--check`` comparison all come +from :mod:`rl_insight.grafana.renderer`, so the CLI and any in-process caller +render exactly the same bytes. See ``README.md`` for usage; exit codes are ``0`` +success, ``1`` stale files found by ``--check`` and ``2`` a config that could +not be rendered. +""" + +from __future__ import annotations + +import argparse +import sys +from pathlib import Path + +_REPO_ROOT = Path(__file__).resolve().parents[3] +if (_REPO_ROOT / "rl_insight").is_dir() and str(_REPO_ROOT) not in sys.path: + # Running straight from a source checkout without an installed package. + sys.path.insert(0, str(_REPO_ROOT)) + +from rl_insight.grafana.renderer import ( # noqa: E402 + JsonnetRenderError, + materialize_dashboards, + stale_dashboards, +) + + +def main() -> int: + parser = argparse.ArgumentParser(description=__doc__) + parser.add_argument( + "--config", type=Path, required=True, help="composition config (.jsonnet)" + ) + parser.add_argument( + "--out-dir", + type=Path, + help="directory to write .json files into", + ) + parser.add_argument( + "--check", + action="store_true", + help="compare against expected files instead of writing (exit 1 on mismatch)", + ) + parser.add_argument( + "--expected-dir", + type=Path, + help="directory holding the expected files for --check (default: --out-dir)", + ) + args = parser.parse_args() + + if not args.check and args.out_dir is None: + parser.error("--out-dir is required unless --check is used") + expected_dir = args.expected_dir if args.expected_dir is not None else args.out_dir + + if args.check: + if expected_dir is None: + print( + "error: --check requires --expected-dir (or --out-dir)", file=sys.stderr + ) + return 2 + try: + stale = stale_dashboards(args.config, expected_dir) + except JsonnetRenderError as error: + print(f"error: {error}", file=sys.stderr) + return 2 + if stale: + print( + "Generated dashboards do not match the expected files:", file=sys.stderr + ) + for path in stale: + print(f" {path}", file=sys.stderr) + return 1 + return 0 + + assert args.out_dir is not None + try: + written = materialize_dashboards(args.config, args.out_dir) + except JsonnetRenderError as error: + print(f"error: {error}", file=sys.stderr) + return 2 + for path in written: + print(path) + return 0 + + +if __name__ == "__main__": + raise SystemExit(main())