From 16bcf88a3d1d39f51beb72aba8bd678ed0f35372 Mon Sep 17 00:00:00 2001 From: Jackson Weber Date: Fri, 25 Sep 2026 15:04:36 -0700 Subject: [PATCH 1/3] feat(perf): report SDK benchmarks as named OTLP log events Measure genuine SDK throughput and signed memory deltas, preserve raw artifacts, and generate named OTLP log results with explicitly configured export. Co-authored-by: Copilot App <223556219+Copilot@users.noreply.github.com> --- CHANGELOG.md | 1 + CONTRIBUTING.md | 81 ++++ perf/benchmark-sdk.mjs | 146 +++++++ perf/benchmark.mjs | 273 ++++++------ perf/export-events.mjs | 106 +++++ perf/memory-worker.mjs | 58 +++ perf/report-results.mjs | 181 ++++++++ test/integration/performance-events.test.mjs | 413 +++++++++++++++++++ 8 files changed, 1122 insertions(+), 137 deletions(-) create mode 100644 perf/benchmark-sdk.mjs create mode 100644 perf/export-events.mjs create mode 100644 perf/memory-worker.mjs create mode 100644 perf/report-results.mjs create mode 100644 test/integration/performance-events.test.mjs diff --git a/CHANGELOG.md b/CHANGELOG.md index 2384515b..9462db28 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -7,6 +7,7 @@ - Add GenAI v1.42 InvokeAgent request, response, cache-token, and provider attribute capture for manual A365 scopes. [#239](https://github.com/microsoft/opentelemetry-distro-javascript/pull/239) ### Other Changes +- Add offline SDK throughput and signed memory benchmarks with raw artifacts and explicitly configured named OTLP log result export. - Consolidate Dependabot updates for Vitest 4.1.11, Hono 4.13.7, qs 6.16.0, fast-uri 3.1.7, actions/deploy-pages 5.0.1, and actions/checkout 7.0.1. - Document local npm lockfile regeneration for contributors who cannot access the Microsoft package proxy, while retaining the proxy-generated lockfile. [#245](https://github.com/microsoft/opentelemetry-distro-javascript/pull/245) diff --git a/CONTRIBUTING.md b/CONTRIBUTING.md index d8452d7b..e18db763 100644 --- a/CONTRIBUTING.md +++ b/CONTRIBUTING.md @@ -63,3 +63,84 @@ the required proxy rather than using this workflow to bypass its restrictions. - Link related issues when applicable. - Update documentation when public behavior or setup changes. - Keep the repository planning and README documents aligned with the implementation. + +## SDK performance benchmarks + +After building, run the standalone harness on Node.js 22 or later. It measures +recording span creation (with and without an attribute), counter aggregation, +and log emission through the built SDK. Non-exporting processors and a metric +reader keep serialization, network transport, and exporter batching out of the +measurement. Recording/aggregation probes fail rather than measuring no-op +providers. This is not an end-to-end exporter or application benchmark. + +```sh +node --expose-gc perf/benchmark.mjs --output tmp/perf/raw.json --iterations 100000 --rounds 12 --memory-iterations 10000 --memory-trials 5 +node perf/export-events.mjs --input tmp/perf/raw.json --output tmp/perf/events.json +``` + +Both commands are offline by default. The benchmark disables inherited +OpenTelemetry/Application Insights exporter and sampler configuration, +SDKStats, automatic instrumentation, and network resource discovery in its +dedicated processes. The VM resource detector is temporarily replaced during +SDK startup because the distro invokes it independently of the detector +environment setting. None of these changes affect normal library usage. +Use a clean Node process without instrumentation preloads. + +`--package-root` defaults to the current directory and must contain the tested +`@microsoft/opentelemetry` manifest and `dist/esm/index.js`. It can also point to +an extracted npm package, with dependencies installed in that directory or an +ancestor. The manifest supplies the measured package name/version. Optional +`--revision` must identify that tested package's source revision, not the CI +orchestration repository; omit it if unknown. Optional `--run-id` supplies an +explicit execution correlation shared by the result events. Neither identifier +is inferred from a host, collector, timestamp, or session. + +The options shown above are the defaults. Each throughput scenario has 20,000 +warmup operations followed by the requested number of measured rounds, with GC +before each round. Raw output retains nanosecond durations, per-round operation +rates, and the existing `ns/op` samples/median consumed by `perf/compare.mjs`. +Only the original two span cases are regression-gating in that comparison. +Throughput telemetry is the **median of per-round rates**, in `operations/s`, +not the reciprocal of the median duration. + +Each memory trial uses a fresh `--expose-gc` worker, warms up +`min(memory-iterations, 1000)` operations, settles GC three times, captures a +baseline, performs the requested operations, then captures immediate and +post-GC snapshots. The raw artifact retains all snapshots, counts, timestamps, +package identity, runtime/OS/architecture, and supplied provenance. +Memory results are medians of **signed deltas**, without clamping or noise +thresholds: immediate heap-used, post-GC retained heap, and immediate RSS minus +baseline, in `By`; heap and retained-heap per-operation deltas are in +`By/{operation}`. These are noisy process observations, not total allocated +bytes or a leak diagnosis. Negative and zero observations remain valid. + +The second command validates the raw observations and writes native OTLP JSON +named log events (`microsoft.opentelemetry.benchmark.result`). Each event has +`test.case.name`, `test.suite.name`, `benchmark.metric`, numeric `benchmark.value`, +`benchmark.unit`, `benchmark.statistic`, and actual iteration/round or +memory-trial counts. Case names are distinct from scenario labels. Metrics are +`microsoft.opentelemetry.benchmark.throughput` and +`microsoft.opentelemetry.benchmark.memory.{heap_used_delta,heap_used_delta_per_operation,retained_heap_delta,retained_heap_delta_per_operation,rss_delta}`. +The custom JSON reporting harness is identified by `telemetry.sdk.*` and +`service.name`, separately from the measured `package.name`/`package.version`. +Runtime/OS/architecture are measured resource attributes; optional provenance +uses `vcs.ref.head.revision` and `benchmark.run_id`. + +Sending is a separate, explicit action for a trusted CI job or operator: + +```sh +node perf/export-events.mjs --input tmp/perf/raw.json --output tmp/perf/events.json --endpoint https://YOUR-COLLECTOR/otlp/v1/logs +``` + +There is no default endpoint or environment-variable fallback. Keep the real +collector URL in private CI configuration, and gate submission separately from +offline benchmark execution (never submit untrusted PR results). HTTPS is +required except for loopback test servers. Credentials, query strings, and +redirects are not accepted. The request uses `Content-Type: application/json`; +timestamps/int64 attributes are strings and measurement `doubleValue` fields +are finite numbers. The default request timeout is 20,000 ms, configurable with +`--timeout-ms` up to 120,000 ms. Non-2xx responses, malformed success responses, +and partial success/error bodies fail the command. There are no automatic +retries because delivery may be uncertain. Both raw and generated payload files +remain available after export failure; archive them even on failed CI runs. +HTTP success alone does not prove downstream ingestion or dashboard refresh. diff --git a/perf/benchmark-sdk.mjs b/perf/benchmark-sdk.mjs new file mode 100644 index 00000000..709933da --- /dev/null +++ b/perf/benchmark-sdk.mjs @@ -0,0 +1,146 @@ +// Copyright (c) Microsoft Corporation. +// Licensed under the MIT License. + +import assert from "node:assert/strict"; +import { createRequire } from "node:module"; +import { join } from "node:path"; +import { pathToFileURL } from "node:url"; + +export const scenarios = [ + { name: "span", test: "span_creation", category: "span" }, + { + name: "span_with_attribute", + test: "span_creation_with_attribute", + category: "span", + }, + { name: "counter_add", test: "metric_counter_add", category: "metric" }, + { name: "logger_emit", test: "log_emit", category: "log" }, +]; + +export async function startBenchmarkSdk(packageRoot) { + // This helper runs only in dedicated benchmark processes, never in the library. + for (const key of Object.keys(process.env)) { + if ( + /^(OTEL_|APPLICATIONINSIGHTS_|AZURE_MONITOR_|MICROSOFT_OTEL_|A365_|ENABLE_A365_)/i.test(key) + ) { + delete process.env[key]; + } + } + process.env.MICROSOFT_OTEL_SDKSTATS_DISABLED = "true"; + process.env.APPLICATIONINSIGHTS_STATSBEAT_DISABLED_ALL = "true"; + process.env.OTEL_NODE_RESOURCE_DETECTORS = "none"; + process.env.OTEL_TRACES_SAMPLER = "always_on"; + + const requireFromPackage = createRequire(join(packageRoot, "package.json")); + const { metrics, trace } = requireFromPackage("@opentelemetry/api"); + const { logs } = requireFromPackage("@opentelemetry/api-logs"); + const { MetricReader } = requireFromPackage("@opentelemetry/sdk-metrics"); + const { azureVmDetector } = requireFromPackage("@opentelemetry/resource-detector-azure"); + // InternalConfig invokes this detector even with OTEL_NODE_RESOURCE_DETECTORS=none. + // Resource discovery/startup is not measured; suppress its metadata HTTP request. + const originalDetect = azureVmDetector.detect; + azureVmDetector.detect = () => ({ attributes: {} }); + + class NonExportingMetricReader extends MetricReader { + async onForceFlush() {} + async onShutdown() {} + } + const metricReader = new NonExportingMetricReader(); + const logRecordProcessor = { + enabled: () => true, + forceFlush: async () => {}, + onEmit: () => {}, + shutdown: async () => {}, + }; + const spanProcessor = { + forceFlush: async () => {}, + onStart: () => {}, + onEnd: () => {}, + shutdown: async () => {}, + }; + let shutdown; + try { + const sdk = await import(pathToFileURL(join(packageRoot, "dist", "esm", "index.js")).href); + shutdown = sdk.shutdownMicrosoftOpenTelemetry; + sdk.useMicrosoftOpenTelemetry({ + azureMonitor: { enabled: false }, + a365: { enabled: false }, + enableConsoleExporters: false, + instrumentationOptions: Object.fromEntries( + [ + "azureSdk", + "bunyan", + "console", + "http", + "langchain", + "mongoDb", + "mySql", + "openaiAgents", + "postgreSql", + "redis", + "redis4", + "winston", + ].map((name) => [name, { enabled: false }]), + ), + logRecordProcessors: [logRecordProcessor], + metricReaders: [metricReader], + samplingRatio: 1, + spanProcessors: [spanProcessor], + tracesPerSecond: 0, + }); + const tracer = trace.getTracer("performance-test"); + const logger = logs.getLogger("performance-test"); + const counter = metrics.getMeter("performance-test").createCounter("benchmark-counter"); + const probe = tracer.startSpan("benchmark-probe"); + assert(probe.isRecording(), "Benchmark requires a recording tracer"); + probe.end(); + let emitted = false; + logRecordProcessor.onEmit = () => { + emitted = true; + }; + logger.emit({ body: "benchmark-probe" }); + logRecordProcessor.onEmit = () => {}; + assert(emitted, "Benchmark requires a recording logger"); + counter.add(1); + const collected = await metricReader.collect(); + assert.equal(collected.errors.length, 0, "Benchmark metric collection failed"); + assert( + collected.resourceMetrics.scopeMetrics.some((scope) => + scope.metrics.some( + (metric) => + metric.descriptor.name === "benchmark-counter" && + metric.dataPoints.some((point) => point.value === 1), + ), + ), + "Benchmark requires an aggregating counter", + ); + const operations = [ + () => { + tracer.startSpan("benchmark-span").end(); + }, + () => { + const span = tracer.startSpan("benchmark-span"); + span.setAttribute("benchmark.attribute", 1); + span.end(); + }, + () => { + counter.add(1); + }, + () => { + logger.emit({ body: "benchmark-log" }); + }, + ]; + return { + scenarios: scenarios.map((scenario, index) => ({ + ...scenario, + operation: operations[index], + })), + shutdown, + }; + } catch (error) { + await shutdown?.(); + throw error; + } finally { + azureVmDetector.detect = originalDetect; + } +} diff --git a/perf/benchmark.mjs b/perf/benchmark.mjs index fdb83e2c..4940a4d7 100644 --- a/perf/benchmark.mjs +++ b/perf/benchmark.mjs @@ -1,157 +1,156 @@ // Copyright (c) Microsoft Corporation. // Licensed under the MIT License. -import { writeFile } from "node:fs/promises"; -import { createRequire } from "node:module"; -import { isAbsolute, join, resolve } from "node:path"; -import { pathToFileURL } from "node:url"; - -const DEFAULT_ITERATIONS = 100_000; -const DEFAULT_ROUNDS = 12; -const WARMUP_ITERATIONS = 20_000; - -function readArgument(name, fallback) { - const index = process.argv.indexOf(name); - if (index === -1) { - return fallback; - } - const value = process.argv[index + 1]; - if (!value) { - throw new Error(`Missing value for ${name}`); +import { execFile } from "node:child_process"; +import { readFile, mkdir, writeFile } from "node:fs/promises"; +import { release } from "node:os"; +import { dirname, join, resolve } from "node:path"; +import { fileURLToPath } from "node:url"; +import { parseArgs, promisify } from "node:util"; +import { startBenchmarkSdk } from "./benchmark-sdk.mjs"; +import { median } from "./report-results.mjs"; + +const { values } = parseArgs({ + options: Object.fromEntries( + [ + ["package-root", process.cwd()], + ["output", undefined], + ["iterations", "100000"], + ["rounds", "12"], + ["memory-iterations", "10000"], + ["memory-trials", "5"], + ["run-id", undefined], + ["revision", undefined], + ].map(([name, defaultValue]) => [name, { type: "string", default: defaultValue }]), + ), +}); +function count(name) { + const value = Number(values[name]); + if (!Number.isSafeInteger(value) || value <= 0) { + throw new Error(`--${name} must be a positive safe integer`); } return value; } - -function median(values) { - const sorted = [...values].sort((left, right) => left - right); - const midpoint = Math.floor(sorted.length / 2); - return sorted.length % 2 === 0 ? (sorted[midpoint - 1] + sorted[midpoint]) / 2 : sorted[midpoint]; +const iterations = count("iterations"); +const rounds = count("rounds"); +const memoryIterations = count("memory-iterations"); +const memoryTrials = count("memory-trials"); +const warmupIterations = 20_000; +if (typeof globalThis.gc !== "function") { + throw new Error("Benchmark requires Node --expose-gc"); } - -function runIterations(operation, iterations) { - const start = process.hrtime.bigint(); - for (let index = 0; index < iterations; index += 1) { - operation(); +for (const option of ["run-id", "revision"]) { + if (values[option] !== undefined && !values[option].trim()) { + throw new Error(`--${option} must not be empty`); } - return Number(process.hrtime.bigint() - start) / iterations; } - -async function benchmark(name, operation, iterations, rounds) { - runIterations(operation, WARMUP_ITERATIONS); - const samples = []; - - for (let round = 0; round < rounds; round += 1) { - globalThis.gc?.(); - samples.push(runIterations(operation, iterations)); - await new Promise((resolveRound) => setImmediate(resolveRound)); - } - - const result = { - gating: true, - name, - samples, - stats: { median: median(samples) }, - unit: "ns/op", - }; - console.log(`${name}: ${result.stats.median.toFixed(1)} ns/op`); - return result; -} - -const packageRootArgument = readArgument("--package-root", process.cwd()); -const packageRoot = isAbsolute(packageRootArgument) - ? packageRootArgument - : resolve(process.cwd(), packageRootArgument); -const outputArgument = readArgument("--output"); -const iterations = Number(readArgument("--iterations", String(DEFAULT_ITERATIONS))); -const rounds = Number(readArgument("--rounds", String(DEFAULT_ROUNDS))); - -if (!Number.isInteger(iterations) || iterations <= 0) { - throw new Error("--iterations must be a positive integer"); -} -if (!Number.isInteger(rounds) || rounds <= 0) { - throw new Error("--rounds must be a positive integer"); -} - -const distroEntryPoint = pathToFileURL(join(packageRoot, "dist", "esm", "index.js")).href; -const requireFromPackage = createRequire(join(packageRoot, "package.json")); -const { trace } = requireFromPackage("@opentelemetry/api"); -process.env.MICROSOFT_OTEL_SDKSTATS_DISABLED = "true"; -const { shutdownMicrosoftOpenTelemetry, useMicrosoftOpenTelemetry } = await import( - distroEntryPoint -); -const spanProcessor = { - forceFlush: () => Promise.resolve(), - onEnd: () => {}, - onStart: () => {}, - shutdown: () => Promise.resolve(), -}; - -useMicrosoftOpenTelemetry({ - azureMonitor: { enabled: false }, - enableConsoleExporters: false, - instrumentationOptions: { - azureSdk: { enabled: false }, - bunyan: { enabled: false }, - console: { enabled: false }, - http: { enabled: false }, - langchain: { enabled: false }, - mongoDb: { enabled: false }, - mySql: { enabled: false }, - openaiAgents: { enabled: false }, - postgreSql: { enabled: false }, - redis: { enabled: false }, - redis4: { enabled: false }, - winston: { enabled: false }, - }, - samplingRatio: 1, - spanProcessors: [spanProcessor], - tracesPerSecond: 0, -}); - -const tracer = trace.getTracer("performance-test"); -const probeSpan = tracer.startSpan("benchmark-probe"); -if (!probeSpan.isRecording()) { - throw new Error(`Benchmark tracer for ${packageRoot} is not backed by a recording provider`); +const packageRoot = resolve(values["package-root"]); +const manifest = JSON.parse(await readFile(join(packageRoot, "package.json"), "utf8")); +if ( + manifest.name !== "@microsoft/opentelemetry" || + typeof manifest.version !== "string" || + !manifest.version +) { + throw new Error("--package-root must identify a built @microsoft/opentelemetry package"); } -probeSpan.end(); +const startedAt = new Date().toISOString(); const benchmarks = []; - +function runIterations(operation, count) { + const start = process.hrtime.bigint(); + for (let index = 0; index < count; index += 1) { + operation(); + } + return Number(process.hrtime.bigint() - start); +} +const { scenarios, shutdown } = await startBenchmarkSdk(packageRoot); try { - benchmarks.push( - await benchmark( - "span", - () => { - tracer.startSpan("benchmark-span").end(); - }, - iterations, - rounds, - ), - ); - benchmarks.push( - await benchmark( - "span_with_attribute", - () => { - const span = tracer.startSpan("benchmark-span"); - span.setAttribute("benchmark.attribute", 1); - span.end(); - }, - iterations, - rounds, - ), - ); + for (const scenario of scenarios) { + runIterations(scenario.operation, warmupIterations); + const durationsNs = []; + for (let round = 0; round < rounds; round += 1) { + globalThis.gc(); + durationsNs.push(runIterations(scenario.operation, iterations)); + await new Promise((resolveRound) => setImmediate(resolveRound)); + } + const samples = durationsNs.map((duration) => duration / iterations); + benchmarks.push({ + name: scenario.name, + test: scenario.test, + category: scenario.category, + gating: scenario.category === "span", + samples, + durationsNs, + operationsPerSecond: durationsNs.map((duration) => (iterations * 1e9) / duration), + stats: { median: median(samples) }, + unit: "ns/op", + completedAt: new Date().toISOString(), + }); + } } finally { - await shutdownMicrosoftOpenTelemetry(); + await shutdown(); +} +const memory = []; +const execute = promisify(execFile); +for (const scenario of scenarios) { + const trials = []; + for (let trial = 0; trial < memoryTrials; trial += 1) { + const { stdout } = await execute( + process.execPath, + [ + "--expose-gc", + fileURLToPath(new URL("./memory-worker.mjs", import.meta.url)), + "--package-root", + packageRoot, + "--scenario", + scenario.name, + "--iterations", + String(memoryIterations), + ], + { timeout: 120_000, maxBuffer: 1024 * 1024, windowsHide: true }, + ); + const lines = stdout.split(/\r?\n/).filter((line) => line.startsWith("MEMORY_RESULT:")); + if (lines.length !== 1) { + throw new Error(`Memory worker ${scenario.name} did not return exactly one result`); + } + trials.push(JSON.parse(lines[0].slice("MEMORY_RESULT:".length))); + } + memory.push({ + name: scenario.name, + test: scenario.test, + category: scenario.category, + trials, + completedAt: new Date().toISOString(), + }); } - const result = { - benchmarks, - iterations, + schemaVersion: 1, + package: { name: manifest.name, version: manifest.version }, packageRoot, + harness: { name: "microsoft-opentelemetry-benchmark", version: "1" }, + environment: { + runtimeName: "nodejs", + runtimeVersion: process.versions.node, + osType: { win32: "windows", darwin: "darwin" }[process.platform] ?? process.platform, + osVersion: release(), + architecture: + { x64: "amd64", ia32: "x86", arm: "arm32", arm64: "arm64" }[process.arch] ?? process.arch, + }, + ...(values["run-id"] === undefined ? {} : { runId: values["run-id"] }), + ...(values.revision === undefined ? {} : { revision: values.revision }), + startedAt, + completedAt: new Date().toISOString(), + iterations, rounds, + warmupIterations, + memoryIterations, + memoryTrials, + benchmarks, + memory, }; - -if (outputArgument) { - await writeFile(outputArgument, `${JSON.stringify(result, null, 2)}\n`, "utf8"); +const json = `${JSON.stringify(result, null, 2)}\n`; +if (values.output) { + await mkdir(dirname(resolve(values.output)), { recursive: true }); + await writeFile(values.output, json, "utf8"); } else { - console.log(JSON.stringify(result, null, 2)); + process.stdout.write(json); } diff --git a/perf/export-events.mjs b/perf/export-events.mjs new file mode 100644 index 00000000..72f285b0 --- /dev/null +++ b/perf/export-events.mjs @@ -0,0 +1,106 @@ +// Copyright (c) Microsoft Corporation. +// Licensed under the MIT License. + +import { mkdir, readFile, writeFile } from "node:fs/promises"; +import { dirname, resolve } from "node:path"; +import { pathToFileURL } from "node:url"; +import { parseArgs } from "node:util"; +import { createEvents } from "./report-results.mjs"; + +export async function exportEvents(payload, endpoint, timeoutMs = 20_000) { + const url = new URL(endpoint); + const loopback = ["localhost", "127.0.0.1", "[::1]"].includes(url.hostname); + if ( + (url.protocol !== "https:" && !(url.protocol === "http:" && loopback)) || + url.pathname !== "/otlp/v1/logs" || + url.username || + url.password || + url.search || + url.hash + ) { + throw new Error( + "Endpoint must be an explicit HTTPS /otlp/v1/logs URL without credentials or query (HTTP allowed only on loopback)", + ); + } + if (!Number.isSafeInteger(timeoutMs) || timeoutMs <= 0 || timeoutMs > 120_000) { + throw new Error("--timeout-ms must be a positive integer at most 120000"); + } + const body = JSON.stringify(payload); + if (Buffer.byteLength(body) > 4 * 1024 * 1024) { + throw new Error("OTLP payload exceeds 4 MiB"); + } + let response; + let responseBody = ""; + try { + response = await fetch(url, { + method: "POST", + headers: { "Content-Type": "application/json" }, + body, + signal: AbortSignal.timeout(timeoutMs), + redirect: "error", + }); + if (response.body) { + for await (const chunk of response.body) { + responseBody += Buffer.from(chunk).toString("utf8"); + if (Buffer.byteLength(responseBody) > 64 * 1024) { + throw new Error("Response exceeds 64 KiB"); + } + } + } + } catch { + throw new Error( + "OTLP export failed: network error, redirect, oversized response, or timeout; delivery may be unknown. No retry was attempted.", + ); + } + if (!response.ok) { + throw new Error(`OTLP export failed: HTTP ${response.status}; no retry was attempted`); + } + let reply; + try { + reply = JSON.parse(responseBody); + } catch { + throw new Error("OTLP export failed: malformed JSON response"); + } + if ( + reply === null || + Array.isArray(reply) || + typeof reply !== "object" || + Object.keys(reply).length !== 0 + ) { + throw new Error( + "OTLP export failed: expected an empty success object, received partialSuccess, error, or unexpected fields", + ); + } +} + +async function main() { + const { values } = parseArgs({ + options: { + input: { type: "string" }, + output: { type: "string" }, + endpoint: { type: "string" }, + "timeout-ms": { type: "string", default: "20000" }, + }, + }); + if (!values.input || !values.output) throw new Error("--input and --output are required"); + const input = resolve(values.input); + const output = resolve(values.output); + if ( + process.platform === "win32" ? input.toLowerCase() === output.toLowerCase() : input === output + ) { + throw new Error("--output must not overwrite the raw --input artifact"); + } + const payload = createEvents(JSON.parse(await readFile(input, "utf8"))); + await mkdir(dirname(output), { recursive: true }); + await writeFile(output, `${JSON.stringify(payload, null, 2)}\n`, "utf8"); + if (values.endpoint !== undefined) { + await exportEvents(payload, values.endpoint, Number(values["timeout-ms"])); + } +} + +if (process.argv[1] && pathToFileURL(resolve(process.argv[1])).href === import.meta.url) { + main().catch((error) => { + console.error(error.message); + process.exitCode = 1; + }); +} diff --git a/perf/memory-worker.mjs b/perf/memory-worker.mjs new file mode 100644 index 00000000..d83b96ba --- /dev/null +++ b/perf/memory-worker.mjs @@ -0,0 +1,58 @@ +// Copyright (c) Microsoft Corporation. +// Licensed under the MIT License. + +import { parseArgs } from "node:util"; +import { startBenchmarkSdk } from "./benchmark-sdk.mjs"; + +const { values } = parseArgs({ + options: { + "package-root": { type: "string" }, + scenario: { type: "string" }, + iterations: { type: "string" }, + }, +}); +const iterations = Number(values.iterations); +if (!values["package-root"] || !Number.isSafeInteger(iterations) || iterations <= 0) { + throw new Error("Memory worker requires --package-root and positive integer --iterations"); +} +if (typeof globalThis.gc !== "function") { + throw new Error("Memory benchmark requires Node --expose-gc"); +} + +async function settle() { + for (let round = 0; round < 3; round += 1) { + globalThis.gc(); + await new Promise((resolve) => setImmediate(resolve)); + } +} + +const { scenarios, shutdown } = await startBenchmarkSdk(values["package-root"]); +try { + const scenario = scenarios.find((candidate) => candidate.name === values.scenario); + if (!scenario) { + throw new Error("Unknown memory scenario"); + } + const warmupIterations = Math.min(iterations, 1_000); + for (let index = 0; index < warmupIterations; index += 1) { + scenario.operation(); + } + await settle(); + const baseline = process.memoryUsage(); + for (let index = 0; index < iterations; index += 1) { + scenario.operation(); + } + const immediate = process.memoryUsage(); + await settle(); + const retained = process.memoryUsage(); + process.stdout.write( + `MEMORY_RESULT:${JSON.stringify({ + baseline, + immediate, + retained, + warmupIterations, + completedAt: new Date().toISOString(), + })}\n`, + ); +} finally { + await shutdown(); +} diff --git a/perf/report-results.mjs b/perf/report-results.mjs new file mode 100644 index 00000000..4a6254ca --- /dev/null +++ b/perf/report-results.mjs @@ -0,0 +1,181 @@ +// Copyright (c) Microsoft Corporation. +// Licensed under the MIT License. + +import assert from "node:assert/strict"; +import { scenarios } from "./benchmark-sdk.mjs"; + +export function median(values) { + assert(values.length > 0 && values.every(Number.isFinite), "Expected finite observations"); + const sorted = [...values].sort((left, right) => left - right); + const midpoint = Math.floor(sorted.length / 2); + return sorted.length % 2 === 0 + ? sorted[midpoint - 1] / 2 + sorted[midpoint] / 2 + : sorted[midpoint]; +} + +function text(value, name) { + assert(typeof value === "string" && value.trim().length > 0, `Invalid ${name}`); + return value; +} + +function count(value, name) { + assert(Number.isSafeInteger(value) && value > 0, `Invalid ${name}`); + return value; +} + +function timestamp(value) { + text(value, "measurement timestamp"); + const milliseconds = Date.parse(value); + assert(Number.isFinite(milliseconds) && milliseconds > 0, "Invalid measurement timestamp"); + const nanoseconds = BigInt(milliseconds) * 1_000_000n; + assert(nanoseconds <= 18_446_744_073_709_551_615n, "Measurement timestamp exceeds uint64"); + return nanoseconds.toString(); +} + +function attributes(values) { + return Object.entries(values).map(([key, value]) => ({ + key, + value: typeof value === "number" ? { intValue: String(value) } : { stringValue: value }, + })); +} + +function event(benchmark, metric, value, unit, statistic, counts) { + assert(Number.isFinite(value), `Non-finite ${metric}`); + return { + eventName: "microsoft.opentelemetry.benchmark.result", + timeUnixNano: timestamp(benchmark.completedAt), + attributes: [ + ...attributes({ + "test.case.name": text(benchmark.test, "test case"), + "test.suite.name": "microsoft-opentelemetry-sdk", + "benchmark.scenario": text(benchmark.name, "scenario"), + "benchmark.category": text(benchmark.category, "category"), + "benchmark.metric": `microsoft.opentelemetry.benchmark.${metric}`, + "benchmark.unit": unit, + "benchmark.statistic": statistic, + ...counts, + }), + { key: "benchmark.value", value: { doubleValue: value } }, + ], + }; +} + +export function createEvents(result) { + assert.equal(result.schemaVersion, 1, "Unsupported raw benchmark schema"); + assert.equal(result.package?.name, "@microsoft/opentelemetry", "Unexpected measured package"); + text(result.package.version, "measured package version"); + assert.equal(result.harness?.name, "microsoft-opentelemetry-benchmark", "Unexpected harness"); + assert.equal(result.harness.version, "1", "Unsupported harness version"); + const iterations = count(result.iterations, "iterations"); + const rounds = count(result.rounds, "rounds"); + const memoryIterations = count(result.memoryIterations, "memory iterations"); + const memoryTrials = count(result.memoryTrials, "memory trials"); + count(result.warmupIterations, "warmup iterations"); + timestamp(result.startedAt); + timestamp(result.completedAt); + const env = result.environment; + assert(env && env.runtimeName === "nodejs", "Expected Node.js benchmark environment"); + const resource = { + "package.name": result.package.name, + "package.version": result.package.version, + "service.name": result.harness.name, + "telemetry.sdk.name": result.harness.name, + "telemetry.sdk.version": result.harness.version, + "telemetry.sdk.language": "javascript", + "user_agent.synthetic.type": "test", + "process.runtime.name": env.runtimeName, + "process.runtime.version": text(env.runtimeVersion, "runtime version"), + "os.type": text(env.osType, "OS type"), + "os.version": text(env.osVersion, "OS version"), + "host.arch": text(env.architecture, "architecture"), + }; + if (result.runId !== undefined) resource["benchmark.run_id"] = text(result.runId, "run ID"); + if (result.revision !== undefined) + resource["vcs.ref.head.revision"] = text(result.revision, "revision"); + const logRecords = []; + for (const collection of [result.benchmarks, result.memory]) { + assert(Array.isArray(collection), "Missing benchmark observations"); + assert.deepEqual( + collection.map((item) => item.name).sort(), + scenarios.map((item) => item.name).sort(), + "Expected one result for each SDK scenario", + ); + } + for (const benchmark of result.benchmarks) { + assert.equal(benchmark.unit, "ns/op", "Unexpected timing unit"); + assert.equal(benchmark.durationsNs?.length, rounds, "Timing round count mismatch"); + assert( + benchmark.durationsNs.every((duration) => Number.isSafeInteger(duration) && duration > 0), + "Invalid raw duration", + ); + const samples = benchmark.durationsNs.map((duration) => duration / iterations); + assert.deepEqual(benchmark.samples, samples, "Raw timing samples mismatch"); + assert.equal(benchmark.stats?.median, median(samples), "Raw timing median mismatch"); + const rates = benchmark.durationsNs.map((duration) => (iterations * 1e9) / duration); + assert.deepEqual(benchmark.operationsPerSecond, rates, "Raw operation rates mismatch"); + logRecords.push( + event(benchmark, "throughput", median(rates), "operations/s", "median", { + "benchmark.iterations": iterations, + "benchmark.rounds": rounds, + "benchmark.warmup_iterations": result.warmupIterations, + }), + ); + } + for (const benchmark of result.memory) { + const throughput = result.benchmarks.find((item) => item.name === benchmark.name); + assert.equal(benchmark.test, throughput.test, "Memory test identity mismatch"); + assert.equal(benchmark.category, throughput.category, "Memory category mismatch"); + assert.equal(benchmark.trials?.length, memoryTrials, "Memory trial count mismatch"); + for (const trial of benchmark.trials) { + timestamp(trial.completedAt); + assert.equal( + trial.warmupIterations, + Math.min(memoryIterations, 1_000), + "Memory warmup mismatch", + ); + for (const snapshot of [trial.baseline, trial.immediate, trial.retained]) { + for (const field of ["rss", "heapUsed", "heapTotal", "external", "arrayBuffers"]) { + assert( + Number.isSafeInteger(snapshot?.[field]) && snapshot[field] >= 0, + `Invalid memory snapshot ${field}`, + ); + } + } + } + const heap = median( + benchmark.trials.map((trial) => trial.immediate.heapUsed - trial.baseline.heapUsed), + ); + const retained = median( + benchmark.trials.map((trial) => trial.retained.heapUsed - trial.baseline.heapUsed), + ); + const rss = median(benchmark.trials.map((trial) => trial.immediate.rss - trial.baseline.rss)); + for (const [metric, value, unit] of [ + ["heap_used_delta", heap, "By"], + ["heap_used_delta_per_operation", heap / memoryIterations, "By/{operation}"], + ["retained_heap_delta", retained, "By"], + ["retained_heap_delta_per_operation", retained / memoryIterations, "By/{operation}"], + ["rss_delta", rss, "By"], + ]) { + logRecords.push( + event(benchmark, `memory.${metric}`, value, unit, "median_delta", { + "benchmark.iterations": memoryIterations, + "benchmark.memory_trials": memoryTrials, + "benchmark.warmup_iterations": Math.min(memoryIterations, 1_000), + }), + ); + } + } + return { + resourceLogs: [ + { + resource: { attributes: attributes(resource) }, + scopeLogs: [ + { + scope: { name: result.harness.name, version: result.harness.version }, + logRecords, + }, + ], + }, + ], + }; +} diff --git a/test/integration/performance-events.test.mjs b/test/integration/performance-events.test.mjs new file mode 100644 index 00000000..15331043 --- /dev/null +++ b/test/integration/performance-events.test.mjs @@ -0,0 +1,413 @@ +// Copyright (c) Microsoft Corporation. +// Licensed under the MIT License. + +import assert from "node:assert/strict"; +import { execFile } from "node:child_process"; +import { once } from "node:events"; +import { mkdtemp, readFile, rm, writeFile } from "node:fs/promises"; +import { createServer } from "node:http"; +import { tmpdir } from "node:os"; +import { join, resolve } from "node:path"; +import { test } from "node:test"; +import { pathToFileURL } from "node:url"; +import { promisify } from "node:util"; +import { scenarios } from "../../perf/benchmark-sdk.mjs"; +import { exportEvents } from "../../perf/export-events.mjs"; +import { createEvents, median } from "../../perf/report-results.mjs"; + +const execute = promisify(execFile); +const root = resolve(import.meta.dirname, "../.."); +const run = (args, options = {}) => + execute(process.execPath, args, { cwd: root, timeout: 60_000, ...options }); +const asMap = (attributes) => Object.fromEntries(attributes.map(({ key, value }) => [key, value])); +const records = (payload) => payload.resourceLogs[0].scopeLogs[0].logRecords; + +// Deliberately synthetic arithmetic fixtures, never submitted outside a loopback test server. +function fixture() { + const completedAt = "2026-01-01T00:00:00.000Z"; + const snapshot = (heapUsed, rss) => ({ + heapUsed, + rss, + heapTotal: 100, + external: 0, + arrayBuffers: 0, + }); + return { + schemaVersion: 1, + package: { name: "@microsoft/opentelemetry", version: "1.2.3-test" }, + harness: { name: "microsoft-opentelemetry-benchmark", version: "1" }, + environment: { + runtimeName: "nodejs", + runtimeVersion: "22.0.0", + osType: "windows", + osVersion: "test", + architecture: "amd64", + }, + startedAt: completedAt, + completedAt, + iterations: 2, + rounds: 2, + warmupIterations: 20_000, + memoryIterations: 2, + memoryTrials: 2, + benchmarks: scenarios.map((scenario) => ({ + ...scenario, + completedAt, + unit: "ns/op", + durationsNs: [1_000_000_000, 2_000_000_000], + samples: [500_000_000, 1_000_000_000], + operationsPerSecond: [2, 1], + stats: { median: 750_000_000 }, + })), + memory: scenarios.map((scenario) => ({ + ...scenario, + completedAt, + trials: [-8, -4].map((delta) => ({ + baseline: snapshot(20, 50), + immediate: snapshot(20 + delta, 50), + retained: snapshot(18, 50), + warmupIterations: 2, + completedAt, + })), + })), + }; +} + +async function serverFor(t, handler) { + const server = createServer(handler); + server.listen(0, "127.0.0.1"); + await once(server, "listening"); + t.after(async () => { + server.closeAllConnections(); + await new Promise((resolve) => server.close(resolve)); + }); + return `http://127.0.0.1:${server.address().port}/otlp/v1/logs`; +} + +async function tempFor(t) { + const directory = await mkdtemp(join(tmpdir(), "sdk-perf-test-")); + t.after(() => rm(directory, { recursive: true, force: true })); + return directory; +} + +test("named logs preserve measured identity, exact units, optional run and signed memory", () => { + const raw = fixture(); + const payload = createEvents(raw); + const resource = asMap(payload.resourceLogs[0].resource.attributes); + assert.deepEqual(resource["package.name"], { stringValue: "@microsoft/opentelemetry" }); + assert.deepEqual(resource["package.version"], { stringValue: "1.2.3-test" }); + assert.deepEqual(resource["telemetry.sdk.name"], { stringValue: raw.harness.name }); + assert.notEqual( + resource["telemetry.sdk.version"].stringValue, + resource["package.version"].stringValue, + ); + assert.equal(resource["benchmark.run_id"], undefined); + assert.equal(resource["service.instance.id"], undefined); + assert.equal(resource["vcs.ref.head.revision"], undefined); + assert.deepEqual(resource["os.type"], { stringValue: "windows" }); + assert.deepEqual(resource["host.arch"], { stringValue: "amd64" }); + assert.equal(records(payload).length, 24); + const byMetric = new Map(); + for (const record of records(payload)) { + assert.equal(record.eventName, "microsoft.opentelemetry.benchmark.result"); + assert.equal(record.timeUnixNano, "1767225600000000000"); + const attrs = asMap(record.attributes); + assert.equal(typeof attrs["benchmark.value"].doubleValue, "number"); + assert.equal(attrs["benchmark.iterations"].intValue, "2"); + assert.equal(attrs["benchmark.source"], undefined); + assert.equal(attrs["benchmark.test"], undefined); + assert.equal(attrs["benchmark.name"], undefined); + byMetric.set(attrs["benchmark.metric"].stringValue, attrs); + } + const throughput = byMetric.get("microsoft.opentelemetry.benchmark.throughput"); + assert.equal( + throughput["benchmark.value"].doubleValue, + 1.5, + "median of rates, not reciprocal median duration", + ); + assert.equal(throughput["benchmark.unit"].stringValue, "operations/s"); + assert.equal(throughput["benchmark.rounds"].intValue, "2"); + for (const [suffix, value, unit] of [ + ["heap_used_delta", -6, "By"], + ["heap_used_delta_per_operation", -3, "By/{operation}"], + ["retained_heap_delta", -2, "By"], + ["retained_heap_delta_per_operation", -1, "By/{operation}"], + ["rss_delta", 0, "By"], + ]) { + const attrs = byMetric.get(`microsoft.opentelemetry.benchmark.memory.${suffix}`); + assert.equal(attrs["benchmark.value"].doubleValue, value); + assert.equal(attrs["benchmark.unit"].stringValue, unit); + assert.equal(attrs["benchmark.statistic"].stringValue, "median_delta"); + assert.equal(attrs["benchmark.memory_trials"].intValue, "2"); + } + const first = asMap(records(payload)[0].attributes); + assert.equal(first["test.case.name"].stringValue, "span_creation"); + assert.equal(first["benchmark.scenario"].stringValue, "span"); + raw.runId = "explicit-execution"; + raw.revision = "tested-package-revision"; + const correlated = asMap(createEvents(raw).resourceLogs[0].resource.attributes); + assert.equal(correlated["benchmark.run_id"].stringValue, raw.runId); + assert.equal(correlated["vcs.ref.head.revision"].stringValue, raw.revision); +}); + +test("raw artifact validation rejects missing, non-finite, mismatched or fabricated summaries", () => { + const mutations = [ + (raw) => { + raw.iterations = 0; + }, + (raw) => { + raw.rounds = 3; + }, + (raw) => { + raw.memoryTrials = 1; + }, + (raw) => { + raw.package.version = ""; + }, + (raw) => { + raw.package.name = "reporter-not-tested-package"; + }, + (raw) => { + raw.runId = ""; + }, + (raw) => { + raw.benchmarks[0].durationsNs[0] = Infinity; + }, + (raw) => { + raw.benchmarks[0].samples[0] = NaN; + }, + (raw) => { + raw.benchmarks[0].stats.median = 1; + }, + (raw) => { + raw.benchmarks[0].operationsPerSecond[0] = 5; + }, + (raw) => { + raw.benchmarks[0].completedAt = "unknown"; + }, + (raw) => { + raw.benchmarks[0].completedAt = "3000-01-01T00:00:00.000Z"; + }, + (raw) => { + raw.benchmarks[0].test = ""; + }, + (raw) => { + raw.memory[0].trials[0].baseline.heapUsed = undefined; + }, + (raw) => { + raw.memory[0].trials[0].immediate.rss = -1; + }, + (raw) => { + raw.memory[0].trials[0].warmupIterations = 0; + }, + (raw) => { + raw.benchmarks.pop(); + }, + (raw) => { + raw.memory[0].name = raw.memory[1].name; + }, + ]; + for (const mutate of mutations) { + const raw = fixture(); + mutate(raw); + assert.throws(() => createEvents(raw)); + } + assert.throws(() => median([])); + assert.throws(() => median([NaN])); +}); + +test("explicit export posts native OTLP JSON to the exact logs path once", async (t) => { + const received = []; + const endpoint = await serverFor(t, async (request, response) => { + const chunks = []; + for await (const chunk of request) chunks.push(chunk); + received.push({ + path: request.url, + method: request.method, + headers: request.headers, + body: JSON.parse(Buffer.concat(chunks)), + }); + response.writeHead(200, { "content-type": "application/json" }); + response.end("{}"); + }); + const payload = createEvents(fixture()); + await exportEvents(payload, endpoint); + assert.equal(received.length, 1); + assert.equal(received[0].path, "/otlp/v1/logs"); + assert.equal(received[0].method, "POST"); + assert.equal(received[0].headers["content-type"], "application/json"); + assert.equal(received[0].headers.authorization, undefined); + assert.deepEqual(received[0].body, payload); +}); + +test("export rejects non-2xx, redirects, malformed responses, partial success and timeout without retries", async (t) => { + for (const [status, body] of [ + [400, "{}"], + [429, "{}"], + [500, "{}"], + [302, "{}"], + [200, ""], + [200, "invalid"], + [200, "null"], + [200, "[]"], + [200, '{"partialSuccess":{"rejectedLogRecords":"1","errorMessage":"rejected"}}'], + [200, '{"partialSuccess":{"rejectedLogRecords":"0"}}'], + [200, '{"error":"failed"}'], + [200, '{"code":3,"message":"invalid"}'], + ]) { + await t.test(`${status} ${body}`, async (t) => { + let requests = 0; + const endpoint = await serverFor(t, (_request, response) => { + requests += 1; + response.writeHead(status, { + location: "/must-not-follow", + "content-type": "application/json", + }); + response.end(body); + }); + await assert.rejects(exportEvents(createEvents(fixture()), endpoint), /OTLP export failed/); + assert.equal(requests, 1); + }); + } + await t.test("timeout", async (t) => { + let requests = 0; + const endpoint = await serverFor(t, () => { + requests += 1; + }); + await assert.rejects(exportEvents(createEvents(fixture()), endpoint, 100), /timeout/); + assert.equal(requests, 1); + }); +}); + +test("unsafe or wrong endpoints are rejected before making a request", async () => { + for (const endpoint of [ + "http://example.invalid/otlp/v1/logs", + "https://example.invalid/v1/logs", + "https://example.invalid/otlp/v1/metrics", + "https://user:secret@example.invalid/otlp/v1/logs", + "https://example.invalid/otlp/v1/logs?token=secret", + "https://example.invalid/otlp/v1/logs#fragment", + ]) { + await assert.rejects(exportEvents(createEvents(fixture()), endpoint), /Endpoint must/); + } + await assert.rejects(exportEvents({}, "https://example.invalid/otlp/v1/logs", 0), /timeout-ms/); +}); + +test("offline CLI ignores exporter environment and preserves both artifacts on export failure", async (t) => { + const directory = await tempFor(t); + const input = join(directory, "raw.json"); + const output = join(directory, "events.json"); + const raw = JSON.stringify(fixture()); + await writeFile(input, raw); + let requests = 0; + const endpoint = await serverFor(t, (_request, response) => { + requests += 1; + response.writeHead(503); + response.end("{}"); + }); + const args = ["perf/export-events.mjs", "--input", input, "--output", output]; + await run(args, { + env: { + ...process.env, + OTEL_EXPORTER_OTLP_ENDPOINT: endpoint, + OTEL_EXPORTER_OTLP_LOGS_ENDPOINT: endpoint, + }, + }); + assert.equal(requests, 0); + assert.equal(records(JSON.parse(await readFile(output))).length, 24); + await assert.rejects(run([...args, "--endpoint", endpoint]), /HTTP 503/); + assert.equal(requests, 1); + assert.equal(await readFile(input, "utf8"), raw); + assert.equal(records(JSON.parse(await readFile(output))).length, 24); + await assert.rejects( + run(["perf/export-events.mjs", "--input", input, "--output", input]), + /must not overwrite/, + ); + assert.equal(await readFile(input, "utf8"), raw); +}); + +test("built SDK benchmark records genuine workloads offline despite inherited telemetry configuration", async (t) => { + const directory = await tempFor(t); + const output = join(directory, "raw.json"); + const networkGuard = join(directory, "deny-network.mjs"); + await writeFile( + networkGuard, + ` + import net from "node:net"; + net.Socket.prototype.connect = function () { + process.stderr.write("Unexpected network connection in offline benchmark\\n"); + process.exit(73); + }; + `, + ); + let requests = 0; + const endpoint = await serverFor(t, (_request, response) => { + requests += 1; + response.end("{}"); + }); + const env = { + ...process.env, + NODE_OPTIONS: `--import="${pathToFileURL(networkGuard).href}"`, + OTEL_EXPORTER_OTLP_ENDPOINT: endpoint, + OTEL_EXPORTER_OTLP_TRACES_ENDPOINT: endpoint, + OTEL_EXPORTER_OTLP_LOGS_ENDPOINT: endpoint, + OTEL_EXPORTER_OTLP_METRICS_ENDPOINT: endpoint, + OTEL_RESOURCE_ATTRIBUTES: "secret=must-not-leak", + OTEL_TRACES_SAMPLER: "always_off", + OTEL_SDK_DISABLED: "true", + MICROSOFT_OTEL_SDKSTATS_DISABLED: "false", + APPLICATIONINSIGHTS_STATS_CONNECTION_STRING: `InstrumentationKey=00000000-0000-0000-0000-000000000001;IngestionEndpoint=${endpoint}`, + APPLICATIONINSIGHTS_CONNECTION_STRING: `InstrumentationKey=00000000-0000-0000-0000-000000000001;IngestionEndpoint=${endpoint}`, + }; + await run( + [ + "--expose-gc", + "perf/benchmark.mjs", + "--package-root", + root, + "--output", + output, + "--iterations", + "20", + "--rounds", + "2", + "--memory-iterations", + "20", + "--memory-trials", + "1", + ], + { env }, + ); + assert.equal(requests, 0, "workload must not export telemetry"); + const rawText = await readFile(output, "utf8"); + assert(!rawText.includes("must-not-leak")); + assert(!rawText.includes("00000000-0000-0000-0000-000000000001")); + const raw = JSON.parse(rawText); + assert.equal(raw.runId, undefined); + assert.equal(raw.revision, undefined); + assert.equal(raw.environment.runtimeVersion, process.versions.node); + assert.equal(raw.environment.osType, process.platform === "win32" ? "windows" : process.platform); + const manifest = JSON.parse(await readFile(join(root, "package.json"))); + assert.deepEqual(raw.package, { name: manifest.name, version: manifest.version }); + assert.equal(raw.benchmarks.length, 4); + assert.equal(raw.memory.length, 4); + assert.equal(records(createEvents(raw)).length, 24); + await run(["perf/compare.mjs", "--baseline", output, "--candidate", output]); +}); + +test("invalid benchmark arguments fail before SDK startup", async () => { + for (const [flag, value] of [ + ["--iterations", "0"], + ["--rounds", "1.5"], + ["--memory-trials", "-1"], + ["--memory-iterations", "NaN"], + ["--iterations", "9007199254740992"], + ]) { + await assert.rejects( + run(["--expose-gc", "perf/benchmark.mjs", `${flag}=${value}`]), + /positive safe integer/, + ); + } + await assert.rejects(run(["perf/benchmark.mjs"]), /requires Node --expose-gc/); + await assert.rejects(run(["--expose-gc", "perf/benchmark.mjs", "--typo", "1"]), /Unknown option/); +}); From 74c015e77ade59c5545694234612b562a6490fba Mon Sep 17 00:00:00 2001 From: Jackson Weber Date: Fri, 25 Sep 2026 16:22:20 -0700 Subject: [PATCH 2/3] fix(perf): preserve raw artifacts across output aliases Reject input/output file identity aliases and atomically replace generated payloads without writing through output links. Co-authored-by: Copilot App <223556219+Copilot@users.noreply.github.com> --- CONTRIBUTING.md | 2 + perf/export-events.mjs | 29 +++++++++- test/integration/performance-events.test.mjs | 60 +++++++++++++++++++- 3 files changed, 87 insertions(+), 4 deletions(-) diff --git a/CONTRIBUTING.md b/CONTRIBUTING.md index e18db763..b9296a58 100644 --- a/CONTRIBUTING.md +++ b/CONTRIBUTING.md @@ -143,4 +143,6 @@ are finite numbers. The default request timeout is 20,000 ms, configurable with and partial success/error bodies fail the command. There are no automatic retries because delivery may be uncertain. Both raw and generated payload files remain available after export failure; archive them even on failed CI runs. +Output files are replaced atomically, and existing aliases of the raw input +(including hardlinks and symlinks) are rejected rather than overwritten. HTTP success alone does not prove downstream ingestion or dashboard refresh. diff --git a/perf/export-events.mjs b/perf/export-events.mjs index 72f285b0..b1fa5b51 100644 --- a/perf/export-events.mjs +++ b/perf/export-events.mjs @@ -1,8 +1,8 @@ // Copyright (c) Microsoft Corporation. // Licensed under the MIT License. -import { mkdir, readFile, writeFile } from "node:fs/promises"; -import { dirname, resolve } from "node:path"; +import { mkdir, mkdtemp, readFile, realpath, rename, rm, stat, writeFile } from "node:fs/promises"; +import { basename, dirname, join, resolve } from "node:path"; import { pathToFileURL } from "node:url"; import { parseArgs } from "node:util"; import { createEvents } from "./report-results.mjs"; @@ -92,7 +92,30 @@ async function main() { } const payload = createEvents(JSON.parse(await readFile(input, "utf8"))); await mkdir(dirname(output), { recursive: true }); - await writeFile(output, `${JSON.stringify(payload, null, 2)}\n`, "utf8"); + const directory = await realpath(dirname(output)); + const destination = join(directory, basename(output)); + const temporaryDirectory = await mkdtemp(join(directory, ".sdk-perf-")); + try { + const temporaryFile = join(temporaryDirectory, "events.json"); + await writeFile(temporaryFile, `${JSON.stringify(payload, null, 2)}\n`, { + encoding: "utf8", + flag: "wx", + }); + const inputStat = await stat(input, { bigint: true }); + let outputStat; + try { + outputStat = await stat(destination, { bigint: true }); + } catch (error) { + if (error.code !== "ENOENT") throw error; + } + if (outputStat && inputStat.dev === outputStat.dev && inputStat.ino === outputStat.ino) { + throw new Error("--output must not overwrite the raw --input artifact"); + } + // Replace the directory entry, not its target, even if an output link changes after stat. + await rename(temporaryFile, destination); + } finally { + await rm(temporaryDirectory, { recursive: true, force: true }); + } if (values.endpoint !== undefined) { await exportEvents(payload, values.endpoint, Number(values["timeout-ms"])); } diff --git a/test/integration/performance-events.test.mjs b/test/integration/performance-events.test.mjs index 15331043..7b6c0e5e 100644 --- a/test/integration/performance-events.test.mjs +++ b/test/integration/performance-events.test.mjs @@ -4,7 +4,7 @@ import assert from "node:assert/strict"; import { execFile } from "node:child_process"; import { once } from "node:events"; -import { mkdtemp, readFile, rm, writeFile } from "node:fs/promises"; +import { link, mkdtemp, readFile, readdir, rm, symlink, writeFile } from "node:fs/promises"; import { createServer } from "node:http"; import { tmpdir } from "node:os"; import { join, resolve } from "node:path"; @@ -326,6 +326,64 @@ test("offline CLI ignores exporter environment and preserves both artifacts on e assert.equal(await readFile(input, "utf8"), raw); }); +test("export CLI preserves raw input through filesystem aliases", async (t) => { + for (const kind of ["hardlink", "symlink", "directory alias", "case variant"]) { + await t.test(kind, async (t) => { + if (kind === "case variant" && process.platform !== "win32") { + t.skip("Case-insensitive path regression applies to Windows"); + return; + } + const directory = await tempFor(t); + const input = join(directory, "raw.json"); + const output = + kind === "directory alias" + ? join(directory, "alias", "raw.json") + : join(directory, kind === "case variant" ? "RAW.JSON" : "events.json"); + const raw = JSON.stringify(fixture()); + await writeFile(input, raw); + if (kind === "hardlink") await link(input, output); + if (kind === "directory alias") { + await symlink( + directory, + join(directory, "alias"), + process.platform === "win32" ? "junction" : "dir", + ); + } + if (kind === "symlink") { + try { + await symlink(input, output, "file"); + } catch (error) { + if (process.platform !== "win32" || error.code !== "EPERM") throw error; + t.skip("Creating file symlinks requires Windows developer mode or elevation"); + return; + } + } + await assert.rejects( + run(["perf/export-events.mjs", "--input", input, "--output", output]), + /must not overwrite/, + ); + assert.equal(await readFile(input, "utf8"), raw); + assert(!(await readdir(directory)).some((name) => name.startsWith(".sdk-perf-"))); + }); + } +}); + +test("export CLI replaces legitimate existing output without modifying its other hardlinks", async (t) => { + const directory = await tempFor(t); + const input = join(directory, "raw.json"); + const output = join(directory, "events.json"); + const previous = join(directory, "previous.json"); + const raw = JSON.stringify(fixture()); + await writeFile(input, raw); + await writeFile(previous, "old output"); + await link(previous, output); + await run(["perf/export-events.mjs", "--input", input, "--output", output]); + assert.equal(await readFile(input, "utf8"), raw); + assert.equal(await readFile(previous, "utf8"), "old output"); + assert.equal(records(JSON.parse(await readFile(output))).length, 24); + assert.deepEqual((await readdir(directory)).sort(), ["events.json", "previous.json", "raw.json"]); +}); + test("built SDK benchmark records genuine workloads offline despite inherited telemetry configuration", async (t) => { const directory = await tempFor(t); const output = join(directory, "raw.json"); From faac8e53874ac3f6b1b48bd53833ace349a1f9e9 Mon Sep 17 00:00:00 2001 From: Jackson Weber Date: Mon, 28 Sep 2026 13:06:57 -0700 Subject: [PATCH 3/3] fix(perf): validate canonical scenario identities Co-authored-by: Copilot App <223556219+Copilot@users.noreply.github.com> --- CONTRIBUTING.md | 6 ++- perf/report-results.mjs | 13 ++++-- test/integration/performance-events.test.mjs | 49 ++++++++++++++++++++ 3 files changed, 63 insertions(+), 5 deletions(-) diff --git a/CONTRIBUTING.md b/CONTRIBUTING.md index b9296a58..6b5776e9 100644 --- a/CONTRIBUTING.md +++ b/CONTRIBUTING.md @@ -115,8 +115,10 @@ baseline, in `By`; heap and retained-heap per-operation deltas are in bytes or a leak diagnosis. Negative and zero observations remain valid. The second command validates the raw observations and writes native OTLP JSON -named log events (`microsoft.opentelemetry.benchmark.result`). Each event has -`test.case.name`, `test.suite.name`, `benchmark.metric`, numeric `benchmark.value`, +named log events (`microsoft.opentelemetry.benchmark.result`). Both throughput +and memory observations must match the canonical scenario names, test cases, +and categories defined in `perf/benchmark-sdk.mjs`, regardless of collection order. +Each event has `test.case.name`, `test.suite.name`, `benchmark.metric`, numeric `benchmark.value`, `benchmark.unit`, `benchmark.statistic`, and actual iteration/round or memory-trial counts. Case names are distinct from scenario labels. Metrics are `microsoft.opentelemetry.benchmark.throughput` and diff --git a/perf/report-results.mjs b/perf/report-results.mjs index 4a6254ca..f5a07260 100644 --- a/perf/report-results.mjs +++ b/perf/report-results.mjs @@ -100,6 +100,16 @@ export function createEvents(result) { scenarios.map((item) => item.name).sort(), "Expected one result for each SDK scenario", ); + for (const benchmark of collection) { + const scenario = scenarios.find((item) => item.name === benchmark.name); + for (const field of ["test", "category"]) { + assert.equal( + benchmark[field], + scenario[field], + `SDK scenario ${field} identity mismatch for ${benchmark.name}`, + ); + } + } } for (const benchmark of result.benchmarks) { assert.equal(benchmark.unit, "ns/op", "Unexpected timing unit"); @@ -122,9 +132,6 @@ export function createEvents(result) { ); } for (const benchmark of result.memory) { - const throughput = result.benchmarks.find((item) => item.name === benchmark.name); - assert.equal(benchmark.test, throughput.test, "Memory test identity mismatch"); - assert.equal(benchmark.category, throughput.category, "Memory category mismatch"); assert.equal(benchmark.trials?.length, memoryTrials, "Memory trial count mismatch"); for (const trial of benchmark.trials) { timestamp(trial.completedAt); diff --git a/test/integration/performance-events.test.mjs b/test/integration/performance-events.test.mjs index 7b6c0e5e..b49da69b 100644 --- a/test/integration/performance-events.test.mjs +++ b/test/integration/performance-events.test.mjs @@ -150,6 +150,43 @@ test("named logs preserve measured identity, exact units, optional run and signe assert.equal(correlated["vcs.ref.head.revision"].stringValue, raw.revision); }); +test("raw artifact validation rejects noncanonical scenario identities in either or both collections", async (t) => { + for (const collections of [["benchmarks"], ["memory"], ["benchmarks", "memory"]]) { + for (const [index, scenario] of scenarios.entries()) { + for (const field of ["test", "category"]) { + for (const label of [ + `relabeled_${scenario[field]}`, + scenarios.find((other) => other[field] !== scenario[field])[field], + ]) { + await t.test(`${collections.join(" and ")}: ${scenario.name} ${field}=${label}`, () => { + const raw = fixture(); + for (const collection of collections) { + raw[collection][index][field] = label; + } + assert.throws(() => createEvents(raw), { + name: "AssertionError", + message: new RegExp( + `^SDK scenario ${field} identity mismatch for ${scenario.name}\\b`, + ), + }); + }); + } + } + } + } +}); + +test("raw artifact validation accepts independently reordered scenarios without changing events", () => { + const raw = fixture(); + const expected = createEvents(raw); + raw.benchmarks.reverse(); + raw.memory.push(raw.memory.shift()); + const payload = createEvents(raw); + assert.deepEqual(payload.resourceLogs[0].resource, expected.resourceLogs[0].resource); + assert.equal(records(payload).length, records(expected).length); + assert.deepEqual(new Set(records(payload)), new Set(records(expected))); +}); + test("raw artifact validation rejects missing, non-finite, mismatched or fabricated summaries", () => { const mutations = [ (raw) => { @@ -203,9 +240,21 @@ test("raw artifact validation rejects missing, non-finite, mismatched or fabrica (raw) => { raw.benchmarks.pop(); }, + (raw) => { + raw.memory.pop(); + }, + (raw) => { + raw.benchmarks[0].name = raw.benchmarks[1].name; + }, (raw) => { raw.memory[0].name = raw.memory[1].name; }, + (raw) => { + raw.benchmarks[0].name = "unknown"; + }, + (raw) => { + raw.memory[0].name = "unknown"; + }, ]; for (const mutate of mutations) { const raw = fixture();