diff --git a/.github/CODEOWNERS b/.github/CODEOWNERS index 12048a44de..b66a4b72ef 100644 --- a/.github/CODEOWNERS +++ b/.github/CODEOWNERS @@ -9,6 +9,9 @@ # subdirectories. /packages/cache samarpan.bhattacharya@sourcefuse.com +# Owner of the benchmarking helpers package and any of its subdirectories. +/packages/benchmarking @sf-sahil-jassal + # In this example, owner owns any files in the services/audit-service # directory at the root of the repository and any of its # subdirectories. diff --git a/lerna.json b/lerna.json index 0e3ab24966..4793d8b3d4 100644 --- a/lerna.json +++ b/lerna.json @@ -3,6 +3,7 @@ "packages/core", "packages/cli", "packages/cache", + "packages/benchmarking", "packages/file-utils", "packages/feature-toggle", "packages/custom-sf-changelog/", diff --git a/package-lock.json b/package-lock.json index 9e140a0c95..dffea8d7a3 100644 --- a/package-lock.json +++ b/package-lock.json @@ -10,6 +10,7 @@ "packages/core/", "packages/cli/", "packages/cache/", + "packages/benchmarking/", "packages/feature-toggle/", "packages/observability/", "packages/file-utils/", @@ -10630,6 +10631,10 @@ "resolved": "services/authentication-service", "link": true }, + "node_modules/@sourceloop/benchmarking": { + "resolved": "packages/benchmarking", + "link": true + }, "node_modules/@sourceloop/bpmn-service": { "resolved": "services/bpmn-service", "link": true @@ -43593,6 +43598,15 @@ "node": ">=6" } }, + "node_modules/tinybench": { + "version": "5.1.0", + "resolved": "https://registry.npmjs.org/tinybench/-/tinybench-5.1.0.tgz", + "integrity": "sha512-LXKNtFualiKOm6gADe1UXPtf8+Nfn1CtPMEHAT33Fd2YjQatrujkDcK0+4wRC1X6t7fxUDXUs6BsvuIgfkDgDg==", + "license": "MIT", + "engines": { + "node": ">=20.0.0" + } + }, "node_modules/tinyexec": { "version": "1.3.0", "resolved": "https://registry.npmjs.org/tinyexec/-/tinyexec-1.3.0.tgz", @@ -50016,6 +50030,50 @@ "zod": "^3.25.28 || ^4" } }, + "packages/benchmarking": { + "name": "@sourceloop/benchmarking", + "version": "0.1.0", + "license": "MIT", + "dependencies": { + "tinybench": "^5.1.0", + "tslib": "^2.8.1" + }, + "devDependencies": { + "@istanbuljs/nyc-config-typescript": "^1.0.2", + "@loopback/build": "^12.0.11", + "@loopback/eslint-config": "^16.0.1", + "@loopback/testlab": "^8.0.11", + "@types/mocha": "^10.0.10", + "@types/node": "^24.19.0", + "eslint": "^8.57.1", + "mocha": "^11.7.5", + "source-map-support": "^0.5.21", + "typescript": "^5.9.3" + }, + "engines": { + "node": "22 || 24" + }, + "peerDependencies": { + "mocha": ">=10" + } + }, + "packages/benchmarking/node_modules/@types/node": { + "version": "24.19.0", + "resolved": "https://registry.npmjs.org/@types/node/-/node-24.19.0.tgz", + "integrity": "sha512-zY+5tKxXdhGh1PYI0ac+7juvEu4OI6vWtVVoj5i2m42jxAY1U+zHGt6QCyOFwykdP62sM3MJ9stoYYUw5aCWew==", + "dev": true, + "license": "MIT", + "dependencies": { + "undici-types": ">=7.24.0 <7.24.7" + } + }, + "packages/benchmarking/node_modules/undici-types": { + "version": "7.24.6", + "resolved": "https://registry.npmjs.org/undici-types/-/undici-types-7.24.6.tgz", + "integrity": "sha512-WRNW+sJgj5OBN4/0JpHFqtqzhpbnV0GuB+OozA9gCL7a993SmU+1JBZCzLNxYsbMfIeDL+lTsphD5jN5N+n0zg==", + "dev": true, + "license": "MIT" + }, "packages/cache": { "name": "@sourceloop/cache", "version": "7.1.1", diff --git a/package.json b/package.json index cf0a3da873..fcd191bd6b 100644 --- a/package.json +++ b/package.json @@ -76,6 +76,7 @@ "packages/core/", "packages/cli/", "packages/cache/", + "packages/benchmarking/", "packages/feature-toggle/", "packages/observability/", "packages/file-utils/", diff --git a/packages/benchmarking/.eslintignore b/packages/benchmarking/.eslintignore new file mode 100644 index 0000000000..38423d3797 --- /dev/null +++ b/packages/benchmarking/.eslintignore @@ -0,0 +1,5 @@ +node_modules/ +dist/ +coverage/ +.eslintrc.js +mochawesome-report diff --git a/packages/benchmarking/.eslintrc.js b/packages/benchmarking/.eslintrc.js new file mode 100644 index 0000000000..4e5330c272 --- /dev/null +++ b/packages/benchmarking/.eslintrc.js @@ -0,0 +1,12 @@ +module.exports = { + extends: '@loopback/eslint-config', + rules: { + 'no-extra-boolean-cast': 'off', + '@typescript-eslint/interface-name-prefix': 'off', + 'no-prototype-builtins': 'off', + }, + parserOptions: { + project: './tsconfig.json', + tsconfigRootDir: __dirname, + }, +}; diff --git a/packages/benchmarking/.gitignore b/packages/benchmarking/.gitignore new file mode 100644 index 0000000000..c8c7a8c5ba --- /dev/null +++ b/packages/benchmarking/.gitignore @@ -0,0 +1,66 @@ +# Logs +logs +*.log +npm-debug.log* +yarn-debug.log* +yarn-error.log* + +# Runtime data +pids +*.pid +*.seed +*.pid.lock + +# Directory for instrumented libs generated by jscoverage/JSCover +lib-cov + +# Coverage directory used by tools like istanbul +coverage + +# nyc test coverage +.nyc_output + +# Grunt intermediate storage (http://gruntjs.com/creating-plugins#storing-task-files) +.grunt + +# Bower dependency directory (https://bower.io/) +bower_components + +# node-waf configuration +.lock-wscript + +# Compiled binary addons (http://nodejs.org/api/addons.html) +build/Release + +# Dependency directories +node_modules/ +jspm_packages/ + +# Typescript v1 declaration files +typings/ + +# Optional npm cache directory +.npm + +# Optional eslint cache +.eslintcache + +# Optional REPL history +.node_repl_history + +# Output of 'npm pack' +*.tgz + +# Yarn Integrity file +.yarn-integrity + +# dotenv environment variables file +.env + +# Transpiled JavaScript files from Typescript +/dist + +# Cache used by TypeScript's incremental build +*.tsbuildinfo + +mochawesome-report \ No newline at end of file diff --git a/packages/benchmarking/.mocharc.json b/packages/benchmarking/.mocharc.json new file mode 100644 index 0000000000..7d5887c8aa --- /dev/null +++ b/packages/benchmarking/.mocharc.json @@ -0,0 +1,6 @@ +{ + "exit": true, + "recursive": true, + "require": "source-map-support/register", + "timeout": 10000 +} diff --git a/packages/benchmarking/.nycrc b/packages/benchmarking/.nycrc new file mode 100644 index 0000000000..26f53635e7 --- /dev/null +++ b/packages/benchmarking/.nycrc @@ -0,0 +1,9 @@ +{ + "include": ["dist"], + "exclude": ["dist/__tests__/"], + "extension": [".js", ".ts"], + "reporter": ["text", "html", "json"], + "exclude-after-remap": false, + "check-coverage": true, + "lines": 75 +} diff --git a/packages/benchmarking/.prettierignore b/packages/benchmarking/.prettierignore new file mode 100644 index 0000000000..64187888dd --- /dev/null +++ b/packages/benchmarking/.prettierignore @@ -0,0 +1,4 @@ +dist +*.json +mochawesome-report +coverage \ No newline at end of file diff --git a/packages/benchmarking/.prettierrc b/packages/benchmarking/.prettierrc new file mode 100644 index 0000000000..2e48c76c31 --- /dev/null +++ b/packages/benchmarking/.prettierrc @@ -0,0 +1,7 @@ +{ + "bracketSpacing": false, + "singleQuote": true, + "printWidth": 80, + "trailingComma": "all", + "arrowParens": "avoid" +} diff --git a/packages/benchmarking/.yo-rc.json b/packages/benchmarking/.yo-rc.json new file mode 100644 index 0000000000..7ac3e079ac --- /dev/null +++ b/packages/benchmarking/.yo-rc.json @@ -0,0 +1,6 @@ +{ + "@sourceloop/cli": { + "packageManager": "npm", + "version": "4.1.0" + } +} diff --git a/packages/benchmarking/LICENSE b/packages/benchmarking/LICENSE new file mode 100644 index 0000000000..75ac5c32e8 --- /dev/null +++ b/packages/benchmarking/LICENSE @@ -0,0 +1,21 @@ +MIT License + +Copyright (c) [2020-2023] [SourceFuse] + +Permission is hereby granted, free of charge, to any person obtaining a copy +of this software and associated documentation files (the "Software"), to deal +in the Software without restriction, including without limitation the rights +to use, copy, modify, merge, publish, distribute, sublicense, and/or sell +copies of the Software, and to permit persons to whom the Software is +furnished to do so, subject to the following conditions: + +The above copyright notice and this permission notice shall be included in all +copies or substantial portions of the Software. + +THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR +IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, +FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE +AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER +LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, +OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE +SOFTWARE. diff --git a/packages/benchmarking/README.md b/packages/benchmarking/README.md new file mode 100644 index 0000000000..fe6f7d7118 --- /dev/null +++ b/packages/benchmarking/README.md @@ -0,0 +1,283 @@ +# @sourceloop/benchmarking + +Benchmark helpers for Mocha. We write a benchmark the same way we write a test, and the build fails when throughput drops further than we allow against a tracked baseline. + +The package wraps [tinybench](https://github.com/tinylibs/tinybench) and keeps a rolling history of results in a JSON file that we commit. Nothing else to run, no separate benchmark runner, no dashboard. + +## What this is, and what it is not + +This is a regression gate on the latency of one logical operation, run serially. + +It is not a load test. The harness runs one callback at a time and does not model arrival rates or concurrent users, so it will not tell us how the service behaves at 500 people hitting it at once. If our callback opens ten connections in parallel, that is our code's concurrency, and the numbers shift because of it, see the `Promise.all` note below. For arrival rate or real concurrent load, reach for k6 or Artillery. What we get here is narrower, and still useful: did the change we just made slow down this one operation. + +We picked that scope on purpose. A serial latency gate is cheap enough to run on every pull request. A real load test is not. + +## Install + +```sh +npm install --save-dev @sourceloop/benchmarking +``` + +Node 22.12 or later, or Node 24. The floor is higher than the rest of this monorepo, which allows Node 22.0 upward. tinybench 5 ships as ESM only, and this package compiles to CommonJS, so loading it depends on `require(esm)`. That landed as stable in 22.12 and is flagged as experimental before it. We would rather state an honest floor than have the package fail on a Node version we claimed to support. + +Mocha is a peer dependency. We do not pull our own copy in, we use the one already in the project. + +## Quickstart + +```ts +import {bench} from '@sourceloop/benchmarking'; +import {OrderService} from '../services'; + +bench.describe('order service', () => { + let service: OrderService; + + before(async () => { + service = await givenOrderService(); + }); + + bench.it('creates an order', async () => { + await service.create(orderFixture); + }); +}); +``` + +Run it: + +```sh +BENCH_UPDATE_BASELINE=1 npm run test:benchmark +``` + +The first few runs record samples and pass without judging anything. Once the history holds `BENCH_MIN_SAMPLES` entries, the gate turns on. + +`bench.describe` and `bench.it` both prefix the title with `benchmark:`. That prefix is what lets a normal test run skip all of this: + +```jsonc +{ + "scripts": { + "test": "lb-mocha --grep 'benchmark:' --invert \"dist/__tests__\"", + "test:benchmark": "lb-mocha \"dist/__tests__/acceptance\"", + }, +} +``` + +Mocha matches `--grep` against the full title path, so the prefix on either the suite or the test is enough. + +## Settings + +Everything is an environment variable. There is no config file and no options object, because these values change per environment, not per call site. + +| Variable | Default | Range | What it does | +| ----------------------- | ---------------------- | --------------------------- | ---------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | +| `BENCH_ITERATIONS` | `10` | 1 to 1000 | How many times tinybench runs the callback. | +| `BENCH_OPS_TO_TRACK` | `10` | 1 to 100 | Size of the rolling history window. | +| `BENCH_MIN_SAMPLES` | `3` | 1 to `BENCH_OPS_TO_TRACK` | Samples needed before the gate starts failing builds. The window never holds more than `BENCH_OPS_TO_TRACK`, so a higher floor is clamped down rather than leaving the gate off forever. | +| `BENCH_THRESHOLD` | `40` | 0 to 100 | Percent drop we tolerate before the run fails. `0` fails on any drop at all, it does not turn the gate off. | +| `BENCH_WARMUP` | on | `false` to disable | Runs one unmeasured warmup iteration. | +| `BENCH_UPDATE_BASELINE` | off | `1` to enable | Writes the result file. Without this, nothing is saved. | +| `BENCH_REPORT_FILE` | `./.bench/report.json` | any path inside the project | Where the history lives. | +| `BENCH_FORCE_BASELINE` | off | `1` to enable | On a run that regressed, throws the window away and restarts from that run. Ignored otherwise. | +| `CI` | off | `true` to enable | Prints a line per benchmark with throughput, latency and deviation. | + +Out of range numbers get clamped rather than rejected, so `BENCH_ITERATIONS=5000` becomes 1000. Anything that is not a whole number falls back to the default instead, including a fraction: `BENCH_ITERATIONS=7.5` runs ten iterations, not seven. `BENCH_THRESHOLD` is the one exception and takes a fraction such as `7.5`. A report path containing `..` is rejected outright, so a stray variable cannot write outside the project. + +The default threshold of 40 percent is loose, and that is deliberate. CodSpeed measured around 2.7 percent run to run variation on GitHub hosted runners, which is enough to make a 2 percent gate fail roughly half the time on code that did not change. A gate nobody trusts gets disabled within a month. Start loose, tighten once the history shows what the real noise is. + +## How the gate decides + +We keep a list of past throughput means per benchmark. Each run compares against the average of the last `BENCH_OPS_TO_TRACK` entries: + +``` +deviation = ((current - average) / average) * 100 +``` + +Negative means slower. The run fails when `deviation < -BENCH_THRESHOLD`. + +Three things about this are worth knowing before trusting the output. + +We compare against a window average, not against the single previous run. github-action-benchmark compares to the previous result by default, which means one lucky fast run quietly raises the bar for everything after it. An average over ten runs absorbs that. + +Below `BENCH_MIN_SAMPLES` the gate stays quiet. A new benchmark records its result and passes, and with `CI=true` it prints how many runs it has recorded against how many it needs. Failing a build on a comparison against one prior sample is a coin flip, not a signal. + +When a run regresses, we do not add it to the history. The window stays exactly as it was. This one is our own choice and it has no precedent in bencher, github-action-benchmark or CodSpeed, all of which append every result regardless of outcome. Their way is statistically cleaner. Ours prevents a slow bleed, where three merges each lose 15 percent, none of them trips a 40 percent gate on its own, and the baseline drifts down with them until the service is half as fast and the benchmark is still green. Our baseline means last known good, not most recent. + +`BENCH_FORCE_BASELINE=1` is the way back out. It only does something on a run that actually regressed, and there it replaces the whole window with that run. A slowdown small enough to pass the gate needs no reset, because it is already in the history. + +## Writing benchmarks that mean something + +A benchmark that measures the wrong thing is worse than no benchmark. It costs CI minutes, it fails on unrelated changes, and people learn to rerun it until it passes. + +### The callback must be async + +```ts +// good +bench.it('creates an order', async () => { + await service.create(fixture); +}); + +// bad +bench.it('creates an order', () => service.create(fixture)); +``` + +Both return a promise, but tinybench 5 works out whether a plain function is async by calling it once and looking at the result. That extra call is not measured and it is not cleaned up. Declaring `async` skips the probe. + +### Measure one operation + +```ts +// good +bench.it('validates an order payload', async () => { + await validator.validate(payload); +}); + +// bad +bench.it('order flow', async () => { + const user = await createUser(); + const cart = await fillCart(user); + await checkout(cart); + await sendConfirmationEmail(cart); +}); +``` + +When the second one regresses by 45 percent, we have learned nothing except that something in four operations got slower. Then we spend an afternoon bisecting by hand. Split it into four benchmarks and the failing one names itself. + +### Keep setup out of the measurement + +```ts +// good +let payload: OrderPayload; +before(() => { + payload = buildLargeOrderPayload(); +}); +bench.it('validates an order payload', async () => { + await validator.validate(payload); +}); + +// bad +bench.it('validates an order payload', async () => { + const payload = buildLargeOrderPayload(); + await validator.validate(payload); +}); +``` + +Fixture construction lands in the number. If someone later adds a field to the fixture builder, the benchmark regresses and the validator never changed. + +### Never touch the network or a shared database + +```ts +// bad +bench.it('fetches exchange rates', async () => { + await fetch('https://api.example.com/rates'); +}); +``` + +This measures somebody else's uptime. It will fail on a Friday for reasons nobody in the repo can fix. Point benchmarks at in-memory fakes or a container the run owns. + +The same goes for a shared database. If two CI jobs hit the same instance, each one shows up in the other's numbers. + +### Each iteration must leave the world as it found it + +```ts +// bad +const queue: Order[] = []; +bench.it('enqueues an order', async () => { + queue.push(await buildOrder()); +}); +``` + +Iteration 500 works on a much bigger array than iteration 1. We are measuring array growth. Reset shared state in the callback, or use a structure where the operation costs the same every time. + +### Do not assert on absolute time + +```ts +// bad +bench.it('creates an order', async () => { + const start = Date.now(); + await service.create(fixture); + expect(Date.now() - start).to.be.lessThan(50); +}); +``` + +50 milliseconds on a developer laptop is 300 on a shared runner. The whole point of the baseline is that it is relative to the same machine class over time. Let the gate do the judging, keep assertions out of the callback. + +### Do not benchmark a mock + +```ts +// bad +const repo = createStubInstance(OrderRepository); +repo.create.resolves(fixture); +bench.it('creates an order', async () => { + await new OrderService(repo).create(fixture); +}); +``` + +If the expensive part is stubbed, the number tracks sinon's dispatch cost. Mock at the network edge if we must, and keep the code under measurement real. + +### Watch what the number counts + +```ts +bench.it('creates 100 orders', async () => { + await Promise.all(orders.map(o => service.create(o))); +}); +``` + +This is legitimate, as long as we read it correctly. Throughput here is batches per second, not orders per second. Divide by 100 for per order cost. The name should say so, otherwise the next person reads the dashboard wrong. + +## Where the noise comes from + +These numbers carry real noise, so treat a failure as a prompt to go and look rather than proof of a regression. V8 optimises hot loops and deoptimises when object shapes change, so a benchmark can measure a fast path that production never hits. A GC pause inside one iteration drags the mean with it, and CI runners add a few percent on their own from noisy neighbours and CPU throttling. + +That is why the threshold defaults to a loose 40 percent and why we compare against a window rather than the last run. Reproduce a failure locally before reverting anything. + +## Running in CI + +The result file is committed. That is what makes the history survive between runs, and it also means a regression shows up in review as a diff on `.bench/report.json`. Keep an eye on it in review, because an unreadable file resets the history and the gate prints a warning and passes. + +The history is keyed by the full Mocha title path, so renaming a suite or a benchmark orphans its samples and starts a fresh baseline. That is worth a second of thought before a tidy-up rename. + +Mocha's `--parallel` mode is not supported. Every benchmark reads and rewrites the same file, so two workers would overwrite each other. + +Only the baseline branch should write it. On a pull request we want the gate to judge, not to record: + +```yaml +- name: Benchmarks + run: npm run test:benchmark + env: + CI: 'true' + BENCH_UPDATE_BASELINE: ${{ github.ref == 'refs/heads/master' && '1' || '' }} + +- name: Commit the baseline + if: github.ref == 'refs/heads/master' + run: | + git add .bench/report.json + git diff --staged --quiet || git commit -m 'chore: update benchmark baseline [skip ci]' + git push +``` + +One job, not two, so the sample we record is the one the gate just judged. An empty `BENCH_UPDATE_BASELINE` reads as off, because only the exact string `1` enables it. If a pull request wrote to the baseline, every branch would be measuring against itself and the gate would never fire. And the commit step matters: without it the file is written on a runner that gets thrown away, and the history never grows. + +When a benchmark fails and we decide the slowdown is worth it, say a new validation pass that halves throughput, reset the window on the baseline branch: + +```sh +BENCH_UPDATE_BASELINE=1 BENCH_FORCE_BASELINE=1 npm run test:benchmark +``` + +That only has an effect on a run the gate rejected, which is exactly the case we are recovering from. Then commit the file and say in the commit message why we accepted it. The next person to find a cliff in the history will want that sentence. + +## Timeouts + +Mocha's default timeout is two seconds, which most benchmarks blow through. `bench.it` sets its own timeout from the iteration count: five seconds, plus two for every iteration and one more for warmup. At the default ten iterations that is 27 seconds. A `.mocharc` timeout is not needed for benchmarks and will not override this. + +## API + +`bench.describe(title, callback)` declares a suite and prefixes the title. The callback must be synchronous, because Mocha collects the tests inside it while the file loads. + +`bench.it(title, callback)` declares a benchmark, runs it, compares it, and fails the test on a regression. The callback must be async. + +Both are reached through `bench` and nowhere else. The package exposes one root entry point, so there is no second name for either of them and no way to reach past the barrel into a module. + +`config` is a live view of the settings. Every property is a getter that reads `process.env` when accessed, so a test can change a variable after the module is loaded. + +The errors all extend `BenchmarkingError`. `PerformanceRegressionError` carries `deviation` and `threshold` as fields so nothing has to parse a message. `BenchmarkError` means the run itself failed, which is different from a regression and should be triaged differently. `ReporterError` wraps a failure to read or write the result file and keeps the original on `cause`. `BenchmarkConfigError` means a setting was refused, currently only a report path pointing outside the project. + +## License + +MIT. See [LICENSE](./LICENSE). diff --git a/packages/benchmarking/package.json b/packages/benchmarking/package.json new file mode 100644 index 0000000000..fc05dbed96 --- /dev/null +++ b/packages/benchmarking/package.json @@ -0,0 +1,90 @@ +{ + "name": "@sourceloop/benchmarking", + "version": "0.1.0", + "description": "Benchmark helpers for Mocha that fail a build when throughput regresses against a tracked baseline", + "keywords": [ + "loopback-extension", + "loopback", + "benchmark", + "performance", + "mocha" + ], + "main": "dist/index.js", + "types": "dist/index.d.ts", + "exports": { + ".": { + "types": "./dist/index.d.ts", + "default": "./dist/index.js" + }, + "./package.json": "./package.json" + }, + "engines": { + "node": "22 || 24" + }, + "files": [ + "README.md", + "dist", + "src", + "!*/__tests__" + ], + "scripts": { + "build": "lb-tsc", + "build:watch": "lb-tsc --watch", + "lint": "npm run eslint && npm run prettier:check", + "lint:fix": "npm run eslint:fix && npm run prettier:fix", + "prettier:cli": "prettier \"**/*.ts\" \"**/*.js\"", + "prettier:check": "npm run prettier:cli -- -l", + "prettier:fix": "npm run prettier:cli -- --write", + "eslint": "eslint --report-unused-disable-directives .", + "eslint:fix": "npm run eslint -- --fix", + "pretest": "npm run rebuild", + "test": "lb-mocha --grep 'benchmark:' --invert --allow-console-logs \"dist/__tests__\"", + "test:benchmark": "lb-mocha --allow-console-logs \"dist/__tests__/acceptance\"", + "test:dev": "lb-mocha --allow-console-logs dist/__tests__/**/*.js", + "clean": "lb-clean dist *.tsbuildinfo .eslintcache", + "rebuild": "npm run clean && npm run build", + "prune": "npm prune --production", + "coverage": "nyc npm run test" + }, + "repository": { + "type": "git", + "url": "https://github.com/sourcefuse/loopback4-microservice-catalog.git", + "directory": "packages/benchmarking" + }, + "author": "Sourcefuse", + "license": "MIT", + "peerDependencies": { + "mocha": ">=10" + }, + "dependencies": { + "tinybench": "^5.1.0", + "tslib": "^2.8.1" + }, + "devDependencies": { + "@istanbuljs/nyc-config-typescript": "^1.0.2", + "@loopback/build": "^12.0.11", + "@loopback/eslint-config": "^16.0.1", + "@loopback/testlab": "^8.0.11", + "@types/mocha": "^10.0.10", + "@types/node": "^24.19.0", + "eslint": "^8.57.1", + "mocha": "^11.7.5", + "source-map-support": "^0.5.21", + "typescript": "^5.9.3" + }, + "publishConfig": { + "registry": "https://registry.npmjs.org/", + "access": "public" + }, + "typedoc": { + "config": { + "entryPoints": [ + "src/index.ts" + ], + "out": "packages/benchmarking", + "plugin": [ + "typedoc-plugin-markdown" + ] + } + } +} diff --git a/packages/benchmarking/src/__tests__/acceptance/self.benchmark.ts b/packages/benchmarking/src/__tests__/acceptance/self.benchmark.ts new file mode 100644 index 0000000000..6450f35754 --- /dev/null +++ b/packages/benchmarking/src/__tests__/acceptance/self.benchmark.ts @@ -0,0 +1,46 @@ +// Copyright (c) 2026 Sourcefuse Technologies +// +// This software is released under the MIT License. +// https://opensource.org/licenses/MIT +import {expect} from '@loopback/testlab'; +import {mkdtempSync, readFileSync, rmSync} from 'node:fs'; +import {tmpdir} from 'node:os'; +import {join} from 'node:path'; +import {bench} from '../../functions'; +import type {BaselineData} from '../../types'; + +bench.describe('the package itself', () => { + let workDir: string; + let reportFile: string; + const saved = {...process.env}; + + // Scoped to this suite so a combined run cannot inherit these settings, and + // so loading the file without running it leaves no temporary directory. + before(() => { + workDir = mkdtempSync(join(tmpdir(), 'benchmarking-self-')); + reportFile = join(workDir, 'report.json'); + process.env.BENCH_ITERATIONS = '2'; + process.env.BENCH_REPORT_FILE = reportFile; + process.env.BENCH_UPDATE_BASELINE = '1'; + }); + + after(() => { + // The benchmark below asserts nothing on its own, so the proof that the + // whole path ran is the sample it left behind. + const report = JSON.parse( + readFileSync(reportFile, 'utf-8'), + ) as BaselineData; + // The suite key keeps the prefix that bench.describe adds to the title. + const entry = + report['benchmark: the package itself']['measures an async callback']; + expect(entry.history).to.have.length(1); + expect(entry.latest.throughput.mean).to.be.above(0); + + process.env = {...saved}; + rmSync(workDir, {recursive: true, force: true}); + }); + + bench.it('measures an async callback', async () => { + await Promise.resolve(); + }); +}); diff --git a/packages/benchmarking/src/__tests__/unit/bench-it.function.unit.ts b/packages/benchmarking/src/__tests__/unit/bench-it.function.unit.ts new file mode 100644 index 0000000000..9803509f6e --- /dev/null +++ b/packages/benchmarking/src/__tests__/unit/bench-it.function.unit.ts @@ -0,0 +1,190 @@ +// Copyright (c) 2026 Sourcefuse Technologies +// +// This software is released under the MIT License. +// https://opensource.org/licenses/MIT +import {expect} from '@loopback/testlab'; +import {mkdtempSync, readFileSync, rmSync, writeFileSync} from 'node:fs'; +import {tmpdir} from 'node:os'; +import {join} from 'node:path'; +import {PerformanceRegressionError} from '../../errors'; +import {benchIt, resultLine} from '../../functions/bench-it.function'; +import type {BaselineData, PerformanceMetrics} from '../../types'; + +/** + * `benchIt` registers a real Mocha test, so we declare one inside a nested + * suite and drive its callback by hand. The registered test never runs on its + * own: the `benchmark:` prefix is what `npm test` filters out. + */ +let declared: Mocha.Test; +describe('orders', () => { + describe('writes', () => { + declared = benchIt('sample operation', async () => { + await Promise.resolve(); + }); + }); +}); + +const SUITE_KEY = 'orders > writes'; + +/** Mocha context stub, enough for the two calls the callback makes. */ +function givenContext() { + const timeouts: number[] = []; + return { + timeouts, + ctx: {timeout: (ms: number) => timeouts.push(ms), test: declared}, + }; +} + +async function run(context: object): Promise { + const fn = declared.fn as (this: object) => Promise; + await fn.call(context); +} + +function givenMetrics(): PerformanceMetrics { + return { + throughput: {mean: 1234.5678, min: 1000, max: 1500}, + latency: {mean: 0.81, min: 0.7, max: 0.9}, + samples: 12, + }; +} + +describe('benchIt', () => { + const saved = {...process.env}; + let workDir: string; + let reportFile: string; + + const readReport = () => + JSON.parse(readFileSync(reportFile, 'utf-8')) as BaselineData; + + function seedUnreachableBaseline() { + writeFileSync( + reportFile, + JSON.stringify({ + [SUITE_KEY]: { + 'sample operation': { + history: [Number.MAX_SAFE_INTEGER, Number.MAX_SAFE_INTEGER], + }, + }, + }), + ); + } + + beforeEach(() => { + workDir = mkdtempSync(join(tmpdir(), 'benchmarking-bench-it-')); + reportFile = join(workDir, 'report.json'); + process.env.BENCH_REPORT_FILE = reportFile; + process.env.BENCH_UPDATE_BASELINE = '1'; + process.env.BENCH_ITERATIONS = '2'; + process.env.BENCH_MIN_SAMPLES = '1'; + delete process.env.BENCH_FORCE_BASELINE; + delete process.env.CI; + }); + + afterEach(() => { + rmSync(workDir, {recursive: true, force: true}); + process.env = {...saved}; + }); + + it('prefixes the title so a normal test run skips it', () => { + expect(declared.title).to.equal('benchmark: sample operation'); + }); + + it('allows more time than the Mocha default', async () => { + const {ctx, timeouts} = givenContext(); + + await run(ctx); + + // 5000 base, plus 2000 for each of the 2 iterations and the warmup. + expect(timeouts).to.eql([11000]); + }); + + it('files the history under the joined suite title path', async () => { + const {ctx} = givenContext(); + + await run(ctx); + + expect(readReport()[SUITE_KEY]['sample operation'].history).to.have.length( + 1, + ); + }); + + it('records a sample and passes while the baseline is building', async () => { + // A drop this far below the seeded baseline would fail the gate, but the + // window holds 2 of the 5 samples the gate waits for. + seedUnreachableBaseline(); + process.env.BENCH_MIN_SAMPLES = '5'; + const {ctx} = givenContext(); + + await run(ctx); + + expect(readReport()[SUITE_KEY]['sample operation'].history).to.have.length( + 3, + ); + }); + + it('fails the test when throughput drops past the threshold', async () => { + seedUnreachableBaseline(); + const {ctx} = givenContext(); + + await expect(run(ctx)).to.be.rejectedWith(PerformanceRegressionError); + }); + + it('holds the baseline when a run regresses', async () => { + seedUnreachableBaseline(); + const {ctx} = givenContext(); + + await expect(run(ctx)).to.be.rejectedWith(PerformanceRegressionError); + + expect(readReport()[SUITE_KEY]['sample operation'].history).to.eql([ + Number.MAX_SAFE_INTEGER, + Number.MAX_SAFE_INTEGER, + ]); + }); +}); + +describe('resultLine', () => { + const saved = {...process.env}; + + afterEach(() => { + process.env = {...saved}; + }); + + it('says how many more samples a new benchmark needs', () => { + process.env.BENCH_MIN_SAMPLES = '5'; + + const line = resultLine('creates an order', givenMetrics(), { + deviation: 0, + buildingBaseline: true, + regressed: false, + recorded: 2, + }); + + expect(line).to.equal( + ' creates an order: building baseline, 1234.57 ops/s, 2 of 5 runs recorded', + ); + }); + + it('reports the deviation with throughput and latency', () => { + const line = resultLine('creates an order', givenMetrics(), { + deviation: -12.3456, + buildingBaseline: false, + regressed: false, + recorded: 10, + }); + + expect(line).to.equal( + ' creates an order: STABLE -12.35%, 1234.57 ops/s, 0.810 ms latency over 12 samples', + ); + }); + + it('labels a regression as one', () => { + const line = resultLine('creates an order', givenMetrics(), { + deviation: -80, + buildingBaseline: false, + regressed: true, + recorded: 10, + }); + + expect(line).to.match(/^ {2}creates an order: REGRESSION -80\.00%/); + }); +}); diff --git a/packages/benchmarking/src/__tests__/unit/config.unit.ts b/packages/benchmarking/src/__tests__/unit/config.unit.ts new file mode 100644 index 0000000000..3e84298979 --- /dev/null +++ b/packages/benchmarking/src/__tests__/unit/config.unit.ts @@ -0,0 +1,144 @@ +// Copyright (c) 2026 Sourcefuse Technologies +// +// This software is released under the MIT License. +// https://opensource.org/licenses/MIT +import {expect} from '@loopback/testlab'; +import {config} from '../../config'; +import {BenchmarkConfigError} from '../../errors'; + +const BENCH_ENV_KEYS = [ + 'BENCH_OPS_TO_TRACK', + 'BENCH_ITERATIONS', + 'BENCH_WARMUP', + 'BENCH_THRESHOLD', + 'BENCH_MIN_SAMPLES', + 'BENCH_REPORT_FILE', + 'BENCH_FORCE_BASELINE', + 'BENCH_UPDATE_BASELINE', + 'CI', +] as const; + +describe('config', () => { + const saved = new Map(); + + beforeEach(() => { + for (const key of BENCH_ENV_KEYS) { + saved.set(key, process.env[key]); + delete process.env[key]; + } + }); + + afterEach(() => { + for (const [key, value] of saved) { + if (value === undefined) { + delete process.env[key]; + } else { + process.env[key] = value; + } + } + saved.clear(); + }); + + it('falls back to defaults when nothing is set', () => { + expect(config.opsToTrack).to.equal(10); + expect(config.iterations).to.equal(10); + expect(config.threshold).to.equal(40); + expect(config.minSamples).to.equal(3); + expect(config.warmup).to.be.true(); + expect(config.updateBaseline).to.be.false(); + expect(config.isCI).to.be.false(); + expect(config.forceBaseline).to.be.false(); + expect(config.reportFile).to.equal('./.bench/report.json'); + }); + + it('reads a variable set after the module was imported', () => { + process.env.BENCH_ITERATIONS = '25'; + expect(config.iterations).to.equal(25); + }); + + it('caps the minimum sample count at the window size', () => { + process.env.BENCH_OPS_TO_TRACK = '2'; + process.env.BENCH_MIN_SAMPLES = '9'; + // A floor above the window would leave the gate building a baseline for + // ever, because the window can never hold that many samples. + expect(config.minSamples).to.equal(2); + }); + + it('lowers the default minimum to fit a small window', () => { + process.env.BENCH_OPS_TO_TRACK = '2'; + expect(config.minSamples).to.equal(2); + }); + + it('clamps a value above the allowed maximum', () => { + process.env.BENCH_ITERATIONS = '5000'; + process.env.BENCH_OPS_TO_TRACK = '250'; + expect(config.iterations).to.equal(1000); + expect(config.opsToTrack).to.equal(100); + }); + + it('clamps a value below the allowed minimum', () => { + process.env.BENCH_ITERATIONS = '0'; + process.env.BENCH_THRESHOLD = '-15'; + expect(config.iterations).to.equal(1); + expect(config.threshold).to.equal(0); + }); + + it('falls back to the default when a value is not a number', () => { + process.env.BENCH_ITERATIONS = 'ten'; + process.env.BENCH_THRESHOLD = 'high'; + expect(config.iterations).to.equal(10); + expect(config.threshold).to.equal(40); + }); + + it('falls back to the default rather than reading a partial number', () => { + process.env.BENCH_ITERATIONS = '10abc'; + expect(config.iterations).to.equal(10); + + process.env.BENCH_ITERATIONS = '7.5'; + expect(config.iterations).to.equal(10); + }); + + it('reads exponent notation as the whole number it means', () => { + process.env.BENCH_ITERATIONS = '1e2'; + expect(config.iterations).to.equal(100); + }); + + it('falls back to the default when a value is blank', () => { + process.env.BENCH_ITERATIONS = ' '; + process.env.BENCH_THRESHOLD = ''; + expect(config.iterations).to.equal(10); + expect(config.threshold).to.equal(40); + }); + + it('accepts a fractional threshold', () => { + process.env.BENCH_THRESHOLD = '7.5'; + expect(config.threshold).to.equal(7.5); + }); + + it('disables warmup only for the exact string "false"', () => { + process.env.BENCH_WARMUP = 'false'; + expect(config.warmup).to.be.false(); + + process.env.BENCH_WARMUP = '0'; + expect(config.warmup).to.be.true(); + }); + + it('enables baseline writes only for the exact string "1"', () => { + process.env.BENCH_UPDATE_BASELINE = 'true'; + expect(config.updateBaseline).to.be.false(); + + process.env.BENCH_UPDATE_BASELINE = '1'; + expect(config.updateBaseline).to.be.true(); + }); + + it('rejects a report path that escapes the project', () => { + process.env.BENCH_REPORT_FILE = '../../etc/report.json'; + expect(() => config.reportFile).to.throw(BenchmarkConfigError); + expect(() => config.reportFile).to.throw(/Path traversal detected/); + }); + + it('accepts a report path inside the project', () => { + process.env.BENCH_REPORT_FILE = './reports/bench.json'; + expect(config.reportFile).to.equal('./reports/bench.json'); + }); +}); diff --git a/packages/benchmarking/src/__tests__/unit/metrics.function.unit.ts b/packages/benchmarking/src/__tests__/unit/metrics.function.unit.ts new file mode 100644 index 0000000000..1cdc5da118 --- /dev/null +++ b/packages/benchmarking/src/__tests__/unit/metrics.function.unit.ts @@ -0,0 +1,73 @@ +// Copyright (c) 2026 Sourcefuse Technologies +// +// This software is released under the MIT License. +// https://opensource.org/licenses/MIT +import {expect} from '@loopback/testlab'; +import type {Task} from 'tinybench'; +import {BenchmarkError} from '../../errors'; +import {deviationLabel, toMetrics} from '../../functions/metrics.function'; + +function givenTask(result: unknown): Task { + return {result} as Task; +} + +const fullResult = { + throughput: {mean: 100, min: 90, max: 110}, + latency: {mean: 10, min: 9, max: 11, samples: [1, 2, 3, 4]}, +}; + +describe('toMetrics', () => { + it('keeps the throughput and latency range', () => { + const metrics = toMetrics(givenTask(fullResult)); + + expect(metrics.throughput).to.eql({mean: 100, min: 90, max: 110}); + expect(metrics.latency).to.eql({mean: 10, min: 9, max: 11}); + }); + + it('counts the latency samples', () => { + expect(toMetrics(givenTask(fullResult)).samples).to.equal(4); + }); + + it('rejects a task that never ran', () => { + expect(() => toMetrics(undefined)).to.throw(BenchmarkError); + expect(() => toMetrics(undefined)).to.throw(/produced no result/); + }); + + it('surfaces the message when the callback threw', () => { + const task = givenTask({...fullResult, error: new Error('boom')}); + + expect(() => toMetrics(task)).to.throw(BenchmarkError); + expect(() => toMetrics(task)).to.throw( + /The benchmark callback threw: boom/, + ); + }); + + it('rejects a result missing throughput', () => { + const task = givenTask({latency: fullResult.latency}); + + expect(() => toMetrics(task)).to.throw(/missing throughput or latency/); + }); + + it('rejects a result missing latency', () => { + const task = givenTask({throughput: fullResult.throughput}); + + expect(() => toMetrics(task)).to.throw(/missing throughput or latency/); + }); +}); + +describe('deviationLabel', () => { + it('follows the gate rather than judging the number itself', () => { + expect(deviationLabel(-41, true)).to.equal('REGRESSION'); + // A big drop while the baseline is still building is not a regression. + expect(deviationLabel(-41, false)).to.equal('STABLE'); + }); + + it('calls a gain past the threshold an improvement', () => { + expect(deviationLabel(41, false)).to.equal('IMPROVEMENT'); + }); + + it('calls anything inside the threshold stable', () => { + expect(deviationLabel(0, false)).to.equal('STABLE'); + expect(deviationLabel(40, false)).to.equal('STABLE'); + }); +}); diff --git a/packages/benchmarking/src/__tests__/unit/reporter.function.unit.ts b/packages/benchmarking/src/__tests__/unit/reporter.function.unit.ts new file mode 100644 index 0000000000..ccc8261d05 --- /dev/null +++ b/packages/benchmarking/src/__tests__/unit/reporter.function.unit.ts @@ -0,0 +1,421 @@ +// Copyright (c) 2026 Sourcefuse Technologies +// +// This software is released under the MIT License. +// https://opensource.org/licenses/MIT +import {expect} from '@loopback/testlab'; +import { + existsSync, + mkdtempSync, + readFileSync, + rmSync, + writeFileSync, +} from 'node:fs'; +import {tmpdir} from 'node:os'; +import {join} from 'node:path'; +import type {BaselineData, PerformanceMetrics} from '../../types'; +import {BenchmarkConfigError, ReporterError} from '../../errors'; +import {reporter} from '../../functions/reporter.function'; + +const SUITE = 'orders'; +const TEST = 'creates an order'; + +function metricsWithThroughput(throughputMean: number): PerformanceMetrics { + return { + throughput: { + mean: throughputMean, + min: throughputMean, + max: throughputMean, + }, + latency: {mean: 1, min: 1, max: 1}, + samples: 10, + }; +} + +describe('reporter', () => { + const savedEnv = new Map(); + let workDir: string; + let reportFile: string; + + function setEnv(key: string, value: string | undefined) { + if (!savedEnv.has(key)) { + savedEnv.set(key, process.env[key]); + } + if (value === undefined) { + delete process.env[key]; + } else { + process.env[key] = value; + } + } + + function writeHistory(history: number[]) { + const data: BaselineData = { + [SUITE]: { + [TEST]: { + history, + latest: { + ...metricsWithThroughput(history[history.length - 1] ?? 0), + latestDeviation: 0, + }, + }, + }, + }; + writeFileSync(reportFile, JSON.stringify(data)); + } + + function readReport(): BaselineData { + return JSON.parse(readFileSync(reportFile, 'utf-8')) as BaselineData; + } + + beforeEach(() => { + workDir = mkdtempSync(join(tmpdir(), 'benchmarking-')); + reportFile = join(workDir, 'report.json'); + setEnv('BENCH_REPORT_FILE', reportFile); + setEnv('BENCH_UPDATE_BASELINE', '1'); + setEnv('BENCH_THRESHOLD', '40'); + setEnv('BENCH_MIN_SAMPLES', '3'); + setEnv('BENCH_OPS_TO_TRACK', '10'); + setEnv('BENCH_FORCE_BASELINE', undefined); + }); + + afterEach(() => { + for (const [key, value] of savedEnv) { + if (value === undefined) { + delete process.env[key]; + } else { + process.env[key] = value; + } + } + savedEnv.clear(); + rmSync(workDir, {recursive: true, force: true}); + }); + + describe('while the baseline is still building', () => { + it('reports the first ever run as building, with no deviation', () => { + const result = reporter(SUITE, TEST, metricsWithThroughput(100)); + + expect(result.buildingBaseline).to.be.true(); + expect(result.deviation).to.equal(0); + }); + + it('keeps reporting as building below the minimum sample count', () => { + writeHistory([100, 100]); + + const result = reporter(SUITE, TEST, metricsWithThroughput(10)); + + expect(result.buildingBaseline).to.be.true(); + }); + + it('records the sample even though it does not gate', () => { + writeHistory([100, 100]); + + reporter(SUITE, TEST, metricsWithThroughput(10)); + + expect(readReport()[SUITE][TEST].history).to.eql([100, 100, 10]); + }); + + it('stops reporting as building once the minimum is reached', () => { + writeHistory([100, 100, 100]); + + const result = reporter(SUITE, TEST, metricsWithThroughput(100)); + + expect(result.buildingBaseline).to.be.false(); + }); + }); + + describe('deviation', () => { + it('is zero for a run matching the baseline average', () => { + writeHistory([100, 100, 100]); + + const result = reporter(SUITE, TEST, metricsWithThroughput(100)); + + expect(result.deviation).to.equal(0); + }); + + it('is positive for a faster run', () => { + writeHistory([100, 100, 100]); + + const result = reporter(SUITE, TEST, metricsWithThroughput(150)); + + expect(result.deviation).to.equal(50); + }); + + it('is negative for a slower run', () => { + writeHistory([100, 100, 100]); + + const result = reporter(SUITE, TEST, metricsWithThroughput(75)); + + expect(result.deviation).to.equal(-25); + }); + + it('averages the window rather than using the last run', () => { + writeHistory([50, 100, 150]); + + const result = reporter(SUITE, TEST, metricsWithThroughput(100)); + + expect(result.deviation).to.equal(0); + }); + + it('only looks at the most recent BENCH_OPS_TO_TRACK samples', () => { + setEnv('BENCH_OPS_TO_TRACK', '3'); + writeHistory([1000, 1000, 100, 100, 100]); + + const result = reporter(SUITE, TEST, metricsWithThroughput(100)); + + expect(result.deviation).to.equal(0); + }); + + it('falls back to zero when the baseline average is zero', () => { + writeHistory([0, 0, 0]); + + const result = reporter(SUITE, TEST, metricsWithThroughput(100)); + + expect(result.deviation).to.equal(0); + }); + }); + + describe('history updates', () => { + it('appends a passing run', () => { + writeHistory([100, 100, 100]); + + reporter(SUITE, TEST, metricsWithThroughput(110)); + + expect(readReport()[SUITE][TEST].history).to.eql([100, 100, 100, 110]); + }); + + it('drops the oldest sample past the window size', () => { + setEnv('BENCH_OPS_TO_TRACK', '3'); + writeHistory([100, 200, 300]); + + reporter(SUITE, TEST, metricsWithThroughput(400)); + + expect(readReport()[SUITE][TEST].history).to.eql([200, 300, 400]); + }); + + it('holds the window when a run regresses', () => { + writeHistory([100, 100, 100]); + + reporter(SUITE, TEST, metricsWithThroughput(10)); + + expect(readReport()[SUITE][TEST].history).to.eql([100, 100, 100]); + }); + + it('restarts from the regressed run when forced', () => { + setEnv('BENCH_FORCE_BASELINE', '1'); + writeHistory([100, 100, 100]); + + reporter(SUITE, TEST, metricsWithThroughput(10)); + + expect(readReport()[SUITE][TEST].history).to.eql([10]); + }); + + it('leaves a passing run untouched by the force flag', () => { + setEnv('BENCH_FORCE_BASELINE', '1'); + writeHistory([100, 100, 100]); + + reporter(SUITE, TEST, metricsWithThroughput(110)); + + expect(readReport()[SUITE][TEST].history).to.eql([100, 100, 100, 110]); + }); + }); + + describe('the regression flag', () => { + it('stays down for a run that is faster than the baseline', () => { + writeHistory([100, 100, 100]); + + expect( + reporter(SUITE, TEST, metricsWithThroughput(150)).regressed, + ).to.be.false(); + }); + + it('stays down for a drop inside the threshold', () => { + writeHistory([100, 100, 100]); + + expect( + reporter(SUITE, TEST, metricsWithThroughput(61)).regressed, + ).to.be.false(); + }); + + it('stays down for a drop landing exactly on the threshold', () => { + writeHistory([100, 100, 100]); + + expect( + reporter(SUITE, TEST, metricsWithThroughput(60)).regressed, + ).to.be.false(); + }); + + it('goes up for a drop past the threshold', () => { + writeHistory([100, 100, 100]); + + expect( + reporter(SUITE, TEST, metricsWithThroughput(59)).regressed, + ).to.be.true(); + }); + + it('stays down while the baseline is still building', () => { + writeHistory([100]); + + expect( + reporter(SUITE, TEST, metricsWithThroughput(1)).regressed, + ).to.be.false(); + }); + + it('follows a threshold changed at run time', () => { + writeHistory([100, 100, 100]); + setEnv('BENCH_THRESHOLD', '10'); + + expect( + reporter(SUITE, TEST, metricsWithThroughput(85)).regressed, + ).to.be.true(); + }); + + it('records the latest metrics alongside the history', () => { + writeHistory([100, 100, 100]); + + reporter(SUITE, TEST, metricsWithThroughput(150)); + + const latest = readReport()[SUITE][TEST].latest; + expect(latest.latestDeviation).to.equal(50); + expect(latest.throughput.mean).to.equal(150); + }); + }); + + describe('file handling', () => { + it('writes nothing unless BENCH_UPDATE_BASELINE is set', () => { + setEnv('BENCH_UPDATE_BASELINE', undefined); + writeHistory([100, 100, 100]); + + reporter(SUITE, TEST, metricsWithThroughput(150)); + + expect(readReport()[SUITE][TEST].history).to.eql([100, 100, 100]); + }); + + it('creates the report directory when it is missing', () => { + const nested = join(workDir, 'nested', 'deep', 'report.json'); + setEnv('BENCH_REPORT_FILE', nested); + + reporter(SUITE, TEST, metricsWithThroughput(100)); + + expect(existsSync(nested)).to.be.true(); + }); + + it('keeps other suites in the file', () => { + writeFileSync( + reportFile, + JSON.stringify({ + billing: { + 'charges a card': { + history: [5], + latest: {...metricsWithThroughput(5), latestDeviation: 0}, + }, + }, + }), + ); + + reporter(SUITE, TEST, metricsWithThroughput(100)); + + expect(readReport().billing['charges a card'].history).to.eql([5]); + }); + + it('keeps other tests in the same suite', () => { + writeFileSync( + reportFile, + JSON.stringify({ + [SUITE]: { + 'reads an order': { + history: [5], + latest: {...metricsWithThroughput(5), latestDeviation: 0}, + }, + }, + }), + ); + + reporter(SUITE, TEST, metricsWithThroughput(100)); + + expect(readReport()[SUITE]['reads an order'].history).to.eql([5]); + }); + + it('starts fresh when the file holds invalid JSON', () => { + writeFileSync(reportFile, 'not json at all'); + + const result = reporter(SUITE, TEST, metricsWithThroughput(100)); + + expect(result.buildingBaseline).to.be.true(); + expect(readReport()[SUITE][TEST].history).to.eql([100]); + }); + + it('starts fresh when the history is not an array', () => { + writeFileSync( + reportFile, + JSON.stringify({[SUITE]: {[TEST]: {history: 'corrupt'}}}), + ); + + const result = reporter(SUITE, TEST, metricsWithThroughput(100)); + + expect(result.deviation).to.equal(0); + expect(readReport()[SUITE][TEST].history).to.eql([100]); + }); + + it('starts fresh when the file holds JSON that is not an object', () => { + for (const content of ['null', '[1, 2]', '5']) { + writeFileSync(reportFile, content); + + const result = reporter(SUITE, TEST, metricsWithThroughput(100)); + + expect(result.deviation).to.equal(0); + expect(readReport()[SUITE][TEST].history).to.eql([100]); + } + }); + + it('drops history entries that are not finite numbers', () => { + writeFileSync( + reportFile, + JSON.stringify({[SUITE]: {[TEST]: {history: [100, 'x', null]}}}), + ); + + const result = reporter(SUITE, TEST, metricsWithThroughput(100)); + + expect(result.deviation).to.equal(0); + expect(readReport()[SUITE][TEST].history).to.eql([100, 100]); + }); + + it('reports the run just recorded in `recorded`', () => { + expect( + reporter(SUITE, TEST, metricsWithThroughput(100)).recorded, + ).to.equal(1); + }); + + it('reports nothing recorded when the baseline is not updated', () => { + setEnv('BENCH_UPDATE_BASELINE', undefined); + + expect( + reporter(SUITE, TEST, metricsWithThroughput(100)).recorded, + ).to.equal(0); + }); + + it('leaves no temporary file behind after a write', () => { + reporter(SUITE, TEST, metricsWithThroughput(100)); + + expect(existsSync(`${reportFile}.tmp`)).to.be.false(); + }); + + it('wraps a write failure in a ReporterError naming the test', () => { + setEnv('BENCH_REPORT_FILE', join(workDir, 'report.json', 'nope.json')); + writeFileSync(reportFile, '{}'); + + expect(() => reporter(SUITE, TEST, metricsWithThroughput(100))).to.throw( + ReporterError, + ); + expect(() => reporter(SUITE, TEST, metricsWithThroughput(100))).to.throw( + /orders\/creates an order/, + ); + }); + + it('passes a bad setting through instead of calling it an I/O failure', () => { + setEnv('BENCH_REPORT_FILE', '../escape/report.json'); + + expect(() => reporter(SUITE, TEST, metricsWithThroughput(100))).to.throw( + BenchmarkConfigError, + ); + }); + }); +}); diff --git a/packages/benchmarking/src/config.ts b/packages/benchmarking/src/config.ts new file mode 100644 index 0000000000..53fdc97299 --- /dev/null +++ b/packages/benchmarking/src/config.ts @@ -0,0 +1,131 @@ +// Copyright (c) 2026 Sourcefuse Technologies +// +// This software is released under the MIT License. +// https://opensource.org/licenses/MIT +import {BenchmarkConfigError} from './errors/benchmark-config.error'; + +const DEFAULT_OPS_TO_TRACK = 10; +const DEFAULT_ITERATIONS = 10; +const DEFAULT_THRESHOLD = 40; +const DEFAULT_MIN_SAMPLES = 3; + +const MAX_OPS_TO_TRACK = 100; +const MAX_ITERATIONS = 1000; +const MIN_THRESHOLD = 0; +const MAX_THRESHOLD = 100; + +/** + * Every property is `readonly`, because the live object is built from getters. + * An assignment would be silently dropped at run time, so we make it a compile + * error instead. + */ +export type BenchmarkConfig = { + readonly opsToTrack: number; + readonly iterations: number; + readonly warmup: boolean; + readonly threshold: number; + readonly minSamples: number; + readonly updateBaseline: boolean; + readonly reportFile: string; + readonly isCI: boolean; + readonly forceBaseline: boolean; +}; + +/** + * Reads a whole number. `Number` is used rather than `parseInt`, because + * `parseInt` reads `1e3` as 1 and `10abc` as 10. We would rather fall back to + * the default than run with a number the caller did not mean. + */ +function readInt( + value: string | undefined, + fallback: number, + min: number, + max: number, +): number { + if (!value?.trim()) { + return fallback; + } + const parsed = Number(value); + if (!Number.isInteger(parsed)) { + return fallback; + } + return Math.min(Math.max(parsed, min), max); +} + +/** + * Rejects any path containing `..`, to stop a traversal out of the project. + * Absolute paths pass: these variables are set by the developer who already + * controls the process. + */ +function readPath(value: string | undefined, fallback: string): string { + // Trimmed, because a path pasted into a CI `env:` block often carries a + // trailing newline, and that would create a directory named after it. + const path = value?.trim() ?? ''; + const resolved = path || fallback; + if (resolved.includes('..')) { + throw new BenchmarkConfigError( + `Path traversal detected in "${resolved}". Use a path inside the project.`, + ); + } + return resolved; +} + +/** + * Live view of the benchmark settings. Every property is a getter, so + * `process.env` is read at access time rather than at import time. That is what + * lets a test set a variable after the module is already loaded. + */ +export const config: BenchmarkConfig = { + get opsToTrack(): number { + return readInt( + process.env.BENCH_OPS_TO_TRACK, + DEFAULT_OPS_TO_TRACK, + 1, + MAX_OPS_TO_TRACK, + ); + }, + get iterations(): number { + return readInt( + process.env.BENCH_ITERATIONS, + DEFAULT_ITERATIONS, + 1, + MAX_ITERATIONS, + ); + }, + get warmup(): boolean { + return process.env.BENCH_WARMUP !== 'false'; + }, + get threshold(): number { + const raw = process.env.BENCH_THRESHOLD; + const parsed = Number(raw ?? ''); + if (!raw?.trim() || !Number.isFinite(parsed)) { + return DEFAULT_THRESHOLD; + } + return Math.min(Math.max(parsed, MIN_THRESHOLD), MAX_THRESHOLD); + }, + /** + * Capped at `opsToTrack`, because the window never holds more than that many + * samples. A higher floor would leave the gate building a baseline forever. + */ + get minSamples(): number { + const ceiling = this.opsToTrack; + return readInt( + process.env.BENCH_MIN_SAMPLES, + Math.min(DEFAULT_MIN_SAMPLES, ceiling), + 1, + ceiling, + ); + }, + get updateBaseline(): boolean { + return process.env.BENCH_UPDATE_BASELINE === '1'; + }, + get reportFile(): string { + return readPath(process.env.BENCH_REPORT_FILE, './.bench/report.json'); + }, + get isCI(): boolean { + return process.env.CI === 'true'; + }, + get forceBaseline(): boolean { + return process.env.BENCH_FORCE_BASELINE === '1'; + }, +}; diff --git a/packages/benchmarking/src/describe-error.ts b/packages/benchmarking/src/describe-error.ts new file mode 100644 index 0000000000..04431db0ff --- /dev/null +++ b/packages/benchmarking/src/describe-error.ts @@ -0,0 +1,17 @@ +// Copyright (c) 2026 Sourcefuse Technologies +// +// This software is released under the MIT License. +// https://opensource.org/licenses/MIT +/** + * Turns an unknown throw into something worth reading. `String()` on an object + * gives `[object Object]`, which hides the detail exactly when it is needed. + * + * Deliberately not re-exported from the package root: it explains our own + * errors and is no use to a consumer. + */ +export function describeError(error: unknown): string { + if (error instanceof Error) { + return error.message; + } + return typeof error === 'string' ? error : JSON.stringify(error); +} diff --git a/packages/benchmarking/src/errors/benchmark-config.error.ts b/packages/benchmarking/src/errors/benchmark-config.error.ts new file mode 100644 index 0000000000..7f27dea457 --- /dev/null +++ b/packages/benchmarking/src/errors/benchmark-config.error.ts @@ -0,0 +1,11 @@ +// Copyright (c) 2026 Sourcefuse Technologies +// +// This software is released under the MIT License. +// https://opensource.org/licenses/MIT +import {BenchmarkingError} from './benchmarking.error'; + +/** + * Thrown when an environment variable holds a value that cannot be accepted, + * eg. a path traversal attempt in `BENCH_REPORT_FILE`. + */ +export class BenchmarkConfigError extends BenchmarkingError {} diff --git a/packages/benchmarking/src/errors/benchmark.error.ts b/packages/benchmarking/src/errors/benchmark.error.ts new file mode 100644 index 0000000000..d7a721fa94 --- /dev/null +++ b/packages/benchmarking/src/errors/benchmark.error.ts @@ -0,0 +1,24 @@ +// Copyright (c) 2026 Sourcefuse Technologies +// +// This software is released under the MIT License. +// https://opensource.org/licenses/MIT +import {BenchmarkingError} from './benchmarking.error'; + +/** + * Thrown when a tinybench run gives unusable results: no result at all, a + * task-level error, or missing throughput/latency data. Callers can catch this + * class to tell harness failures apart from a real regression. + */ +export class BenchmarkError extends BenchmarkingError { + /** + * Carries the error the benchmark callback threw, so its stack survives. + * A declared field rather than the ES2022 `ErrorOptions`, because the + * inherited tsconfig pins `lib` to es2020. + */ + constructor( + message: string, + public readonly cause?: unknown, + ) { + super(message); + } +} diff --git a/packages/benchmarking/src/errors/benchmarking.error.ts b/packages/benchmarking/src/errors/benchmarking.error.ts new file mode 100644 index 0000000000..0688e58d54 --- /dev/null +++ b/packages/benchmarking/src/errors/benchmarking.error.ts @@ -0,0 +1,14 @@ +// Copyright (c) 2026 Sourcefuse Technologies +// +// This software is released under the MIT License. +// https://opensource.org/licenses/MIT +/** + * Base for every error thrown by `@sourceloop/benchmarking`. Subclasses inherit + * the correct `name` from `new.target`, so it cannot drift from the class name. + */ +export class BenchmarkingError extends Error { + constructor(message: string) { + super(message); + this.name = new.target.name; + } +} diff --git a/packages/benchmarking/src/errors/index.ts b/packages/benchmarking/src/errors/index.ts new file mode 100644 index 0000000000..aa54a77c13 --- /dev/null +++ b/packages/benchmarking/src/errors/index.ts @@ -0,0 +1,9 @@ +// Copyright (c) 2026 Sourcefuse Technologies +// +// This software is released under the MIT License. +// https://opensource.org/licenses/MIT +export * from './benchmarking.error'; +export * from './benchmark.error'; +export * from './benchmark-config.error'; +export * from './performance-regression.error'; +export * from './reporter.error'; diff --git a/packages/benchmarking/src/errors/performance-regression.error.ts b/packages/benchmarking/src/errors/performance-regression.error.ts new file mode 100644 index 0000000000..9190bbd156 --- /dev/null +++ b/packages/benchmarking/src/errors/performance-regression.error.ts @@ -0,0 +1,23 @@ +// Copyright (c) 2026 Sourcefuse Technologies +// +// This software is released under the MIT License. +// https://opensource.org/licenses/MIT +import {BenchmarkingError} from './benchmarking.error'; + +const DECIMAL_PRECISION = 2; + +/** + * Thrown when a run drops further below the baseline than the configured + * threshold allows. Deviation and threshold stay available as fields, so + * callers do not have to parse the message. + */ +export class PerformanceRegressionError extends BenchmarkingError { + constructor( + public readonly deviation: number, + public readonly threshold: number, + ) { + super( + `Performance regression detected: deviation of ${deviation.toFixed(DECIMAL_PRECISION)}% exceeds the threshold of -${threshold}%`, + ); + } +} diff --git a/packages/benchmarking/src/errors/reporter.error.ts b/packages/benchmarking/src/errors/reporter.error.ts new file mode 100644 index 0000000000..09f1f4ed2f --- /dev/null +++ b/packages/benchmarking/src/errors/reporter.error.ts @@ -0,0 +1,26 @@ +// Copyright (c) 2026 Sourcefuse Technologies +// +// This software is released under the MIT License. +// https://opensource.org/licenses/MIT +import {describeError} from '../describe-error'; +import {BenchmarkingError} from './benchmarking.error'; + +/** + * Thrown when the reporter cannot read or write baseline data. The original + * failure stays available on `cause`. + * + * `cause` is a declared field instead of the ES2022 `ErrorOptions` argument + * because the inherited tsconfig pins `lib: ["es2020"]`, which predates that + * constructor overload. + */ +export class ReporterError extends BenchmarkingError { + constructor( + suite: string, + test: string, + public readonly cause: unknown, + ) { + super( + `Failed to report benchmark data for "${suite}/${test}": ${describeError(cause)}`, + ); + } +} diff --git a/packages/benchmarking/src/functions/bench-describe.function.ts b/packages/benchmarking/src/functions/bench-describe.function.ts new file mode 100644 index 0000000000..83713338b9 --- /dev/null +++ b/packages/benchmarking/src/functions/bench-describe.function.ts @@ -0,0 +1,20 @@ +// Copyright (c) 2026 Sourcefuse Technologies +// +// This software is released under the MIT License. +// https://opensource.org/licenses/MIT +/** + * Declares a benchmark suite. Reach it through {@link bench.describe}. + * + * The title is prefixed with `benchmark:` so a normal `npm test` can skip + * every benchmark with `--grep 'benchmark:' --invert`. + * + * @param title - Suite name, without the prefix. + * @param callback - Suite body. It must be synchronous, as Mocha collects the + * tests inside it while the file is loaded. + */ +export function benchDescribe( + title: string, + callback: () => void, +): Mocha.Suite { + return describe(`benchmark: ${title}`, callback); +} diff --git a/packages/benchmarking/src/functions/bench-it.function.ts b/packages/benchmarking/src/functions/bench-it.function.ts new file mode 100644 index 0000000000..491ed9e144 --- /dev/null +++ b/packages/benchmarking/src/functions/bench-it.function.ts @@ -0,0 +1,95 @@ +// Copyright (c) 2026 Sourcefuse Technologies +// +// This software is released under the MIT License. +// https://opensource.org/licenses/MIT +import {Bench} from 'tinybench'; +import {config} from '../config'; +import {type PerformanceMetrics} from '../types'; +import {PerformanceRegressionError} from '../errors/performance-regression.error'; +import {deviationLabel, toMetrics} from './metrics.function'; +import {reporter, type ReporterResult} from './reporter.function'; + +const THROUGHPUT_PRECISION = 2; +const LATENCY_PRECISION = 3; +const TIMEOUT_BASE_MS = 5000; +const TIMEOUT_PER_ITERATION_MS = 2000; + +/** + * Builds the one line we print per benchmark. Kept apart from the printing so + * the wording can be asserted without capturing the console. + * + * @internal + */ +export function resultLine( + test: string, + metrics: PerformanceMetrics, + result: ReporterResult, +): string { + const throughput = metrics.throughput.mean.toFixed(THROUGHPUT_PRECISION); + const latency = metrics.latency.mean.toFixed(LATENCY_PRECISION); + + if (result.buildingBaseline) { + // Counts recorded runs, not the iterations inside this one. Those are two + // different numbers and reading one as the other is misleading. + return ` ${test}: building baseline, ${throughput} ops/s, ${result.recorded} of ${config.minSamples} runs recorded`; + } + + const label = deviationLabel(result.deviation, result.regressed); + return ` ${test}: ${label} ${result.deviation.toFixed(THROUGHPUT_PRECISION)}%, ${throughput} ops/s, ${latency} ms latency over ${metrics.samples} samples`; +} + +/** + * Declares a single benchmark and fails the test when throughput regresses past + * the threshold. Reach it through {@link bench.it}. + * + * The callback must be declared `async`. tinybench detects a plain function + * that returns a promise by calling it once before measuring, which adds an + * extra unmeasured run. + * + * The callback should cover one logical operation. Concurrency and volume are + * yours to set up: with a `Promise.all` body, `throughput.mean` counts batches + * per second, not requests per second. + * + * @param title - Benchmark name, without the prefix. + * @param callback - The operation to measure. + */ +export function benchIt( + title: string, + callback: () => Promise, +): Mocha.Test { + return it(`benchmark: ${title}`, async function () { + // eslint-disable-next-line @typescript-eslint/no-invalid-this + this.timeout( + TIMEOUT_BASE_MS + (config.iterations + 1) * TIMEOUT_PER_ITERATION_MS, + ); + // The full title path keys the history, so renaming a suite or a benchmark + // orphans its samples and starts a new baseline. Mocha always gives a + // running test a parent, the root suite at the very least. + // eslint-disable-next-line @typescript-eslint/no-invalid-this + const suite = this.test!.parent!.titlePath().join(' > '); + + const bench = new Bench({ + iterations: config.iterations, + warmup: config.warmup, + warmupIterations: 1, + warmupTime: 0, + // `0` drops the time based stopping rule, so the iteration count alone + // decides when the run ends. See tinylibs/tinybench#83. + time: 0, + }); + + bench.add(title, callback); + await bench.run(); + + const metrics = toMetrics(bench.tasks[0]); + const result = reporter(suite, title, metrics); + + if (config.isCI) { + console.info(resultLine(title, metrics, result)); + } + + if (result.regressed) { + throw new PerformanceRegressionError(result.deviation, config.threshold); + } + }); +} diff --git a/packages/benchmarking/src/functions/index.ts b/packages/benchmarking/src/functions/index.ts new file mode 100644 index 0000000000..9f59f6b70a --- /dev/null +++ b/packages/benchmarking/src/functions/index.ts @@ -0,0 +1,27 @@ +// Copyright (c) 2026 Sourcefuse Technologies +// +// This software is released under the MIT License. +// https://opensource.org/licenses/MIT +import {benchDescribe} from './bench-describe.function'; +import {benchIt} from './bench-it.function'; + +/** + * The only way in. The two functions behind it stay unexported on purpose, so + * there is one name for each idea rather than two. + * + * It mirrors Mocha, so a benchmark file reads like any other test file: + * + * ```ts + * import {bench} from '@sourceloop/benchmarking'; + * + * bench.describe('order service', () => { + * bench.it('creates an order', async () => { + * await service.create(fixture); + * }); + * }); + * ``` + */ +export const bench = { + describe: benchDescribe, + it: benchIt, +} as const; diff --git a/packages/benchmarking/src/functions/metrics.function.ts b/packages/benchmarking/src/functions/metrics.function.ts new file mode 100644 index 0000000000..843c2abb4a --- /dev/null +++ b/packages/benchmarking/src/functions/metrics.function.ts @@ -0,0 +1,63 @@ +// Copyright (c) 2026 Sourcefuse Technologies +// +// This software is released under the MIT License. +// https://opensource.org/licenses/MIT +import {type Task} from 'tinybench'; +import {config} from '../config'; +import {type PerformanceMetrics} from '../types'; +import {BenchmarkError} from '../errors/benchmark.error'; + +/** + * Describes a deviation for the CI log. Presentation only, it gates nothing. + * `regressed` comes from the gate rather than being recomputed here, so the + * label can never disagree with the outcome of the run. + */ +export function deviationLabel(deviation: number, regressed: boolean): string { + if (regressed) { + return 'REGRESSION'; + } + return deviation > config.threshold ? 'IMPROVEMENT' : 'STABLE'; +} + +/** + * Narrows a tinybench task down to the numbers we keep. + * + * @throws BenchmarkError when the run produced nothing usable. That is a broken + * benchmark rather than a slow one, so it must not read as a regression. + */ +export function toMetrics(task: Task | undefined): PerformanceMetrics { + const result = task?.result; + + if (!result) { + throw new BenchmarkError( + 'The benchmark produced no result. Check that the callback returns and does not hang.', + ); + } + + if (result.error) { + throw new BenchmarkError( + `The benchmark callback threw: ${result.error.message}`, + result.error, + ); + } + + if (!result.throughput || !result.latency) { + throw new BenchmarkError( + 'The benchmark result is missing throughput or latency data. This usually means every iteration failed.', + ); + } + + return { + throughput: { + mean: result.throughput.mean, + min: result.throughput.min, + max: result.throughput.max, + }, + latency: { + mean: result.latency.mean, + min: result.latency.min, + max: result.latency.max, + }, + samples: result.latency.samples.length, + }; +} diff --git a/packages/benchmarking/src/functions/reporter.function.ts b/packages/benchmarking/src/functions/reporter.function.ts new file mode 100644 index 0000000000..f2de2bd390 --- /dev/null +++ b/packages/benchmarking/src/functions/reporter.function.ts @@ -0,0 +1,161 @@ +// Copyright (c) 2026 Sourcefuse Technologies +// +// This software is released under the MIT License. +// https://opensource.org/licenses/MIT +import { + existsSync, + mkdirSync, + readFileSync, + renameSync, + rmSync, + writeFileSync, +} from 'node:fs'; +import {dirname} from 'node:path'; +import {config} from '../config'; +import {describeError} from '../describe-error'; +import {type BaselineData, type PerformanceMetrics} from '../types'; +import {BenchmarkingError} from '../errors/benchmarking.error'; +import {ReporterError} from '../errors/reporter.error'; + +const PERCENT = 100; + +function readBaseline(): BaselineData { + if (!existsSync(config.reportFile)) { + return {}; + } + try { + const parsed: unknown = JSON.parse( + readFileSync(config.reportFile, 'utf-8'), + ); + if ( + typeof parsed !== 'object' || + parsed === null || + Array.isArray(parsed) + ) { + throw new TypeError('expected a JSON object'); + } + return parsed as BaselineData; + } catch (error) { + // A corrupt file must not block the run. Say so loudly though: a silent + // fresh start turns the gate off for the next few runs. + console.info( + `Baseline file "${config.reportFile}" is unreadable, starting a fresh history: ${describeError(error)}`, + ); + return {}; + } +} + +/** + * Replaces the file in one step, so an interrupted run cannot leave a truncated + * baseline behind. A truncated file parses as empty and would silently drop + * every recorded history. + */ +function writeBaseline(data: BaselineData): void { + mkdirSync(dirname(config.reportFile), {recursive: true}); + const temporary = `${config.reportFile}.tmp`; + try { + writeFileSync(temporary, JSON.stringify(data, null, 2)); + renameSync(temporary, config.reportFile); + } catch (error) { + rmSync(temporary, {force: true}); + throw error; + } +} + +/** + * Picks the window to store. A regressed run is held out of the history, so a + * bad merge cannot drag the rolling average down. `BENCH_FORCE_BASELINE` is the + * way back: it restarts the window from the run that failed. + */ +function nextHistory( + recent: number[], + throughput: number, + regressed: boolean, +): number[] { + if (regressed && config.forceBaseline) { + return [throughput]; + } + if (regressed) { + return recent; + } + return [...recent, throughput].slice(-config.opsToTrack); +} + +function mean(values: number[]): number { + return values.reduce((a, b) => a + b, 0) / values.length; +} + +export type ReporterResult = { + /** Percentage change of this run against the baseline average. */ + deviation: number; + /** True while the history holds fewer samples than `BENCH_MIN_SAMPLES`. */ + buildingBaseline: boolean; + /** Runs held in the stored window once this run has been handled. */ + recorded: number; + /** + * True when this run dropped further below the baseline than the threshold + * allows. The caller fails the test on it, and the history is held back for + * the same reason, so both decisions are made here and cannot drift apart. + */ + regressed: boolean; +}; + +/** + * Compares a run against the recorded history and records the new sample. + * + * The history is a rolling window of throughput means rather than a single + * previous run, because one CI sample is far too noisy to gate on. + * + * When a run regresses the history is held as it is, so a bad merge cannot drag + * the rolling average down and quietly widen the accepted range. This is our own + * choice, not a pattern copied from another tool. `BENCH_FORCE_BASELINE=1` + * restarts the window from a regressed run, which is the escape hatch after a + * deliberate trade-off. + */ +export function reporter( + suite: string, + test: string, + metrics: PerformanceMetrics, +): ReporterResult { + try { + const baseline = readBaseline(); + const history = baseline[suite]?.[test]?.history ?? []; + const recent = (Array.isArray(history) ? history : []) + .filter(value => Number.isFinite(value)) + .slice(-config.opsToTrack); + + const buildingBaseline = recent.length < config.minSamples; + // An empty window gives NaN, which falls through the same guard that + // catches a zero average. + const average = mean(recent); + const rawDeviation = + ((metrics.throughput.mean - average) / average) * PERCENT; + const deviation = Number.isFinite(rawDeviation) ? rawDeviation : 0; + + const regressed = !buildingBaseline && deviation < -config.threshold; + + const stored = config.updateBaseline + ? nextHistory(recent, metrics.throughput.mean, regressed) + : recent; + + if (config.updateBaseline) { + baseline[suite] = { + ...baseline[suite], + [test]: { + history: stored, + latest: {...metrics, latestDeviation: deviation}, + }, + }; + writeBaseline(baseline); + } + + return {deviation, buildingBaseline, regressed, recorded: stored.length}; + } catch (error) { + // A bad setting is the caller's problem and already says so. Only wrap the + // failures that are really about reading or writing the baseline. + if (error instanceof BenchmarkingError) { + throw error; + } + throw new ReporterError(suite, test, error); + } +} diff --git a/packages/benchmarking/src/index.ts b/packages/benchmarking/src/index.ts new file mode 100644 index 0000000000..12b8e5a3da --- /dev/null +++ b/packages/benchmarking/src/index.ts @@ -0,0 +1,8 @@ +// Copyright (c) 2026 Sourcefuse Technologies +// +// This software is released under the MIT License. +// https://opensource.org/licenses/MIT +export * from './config'; +export * from './types'; +export * from './errors'; +export * from './functions'; diff --git a/packages/benchmarking/src/types.ts b/packages/benchmarking/src/types.ts new file mode 100644 index 0000000000..fe7aa068a9 --- /dev/null +++ b/packages/benchmarking/src/types.ts @@ -0,0 +1,30 @@ +// Copyright (c) 2026 Sourcefuse Technologies +// +// This software is released under the MIT License. +// https://opensource.org/licenses/MIT +/** + * Shape of the report file and of a single measured run. These describe what we + * store on disk, not how the run is configured. + */ +export type MetricRange = { + mean: number; + min: number; + max: number; +}; + +export type PerformanceMetrics = { + throughput: MetricRange; + latency: MetricRange; + samples: number; +}; + +export type BaselineEntry = { + history: number[]; + latest: PerformanceMetrics & {latestDeviation: number}; +}; + +export type BaselineData = { + [suiteName: string]: { + [testName: string]: BaselineEntry; + }; +}; diff --git a/packages/benchmarking/tsconfig.json b/packages/benchmarking/tsconfig.json new file mode 100644 index 0000000000..5da414b90c --- /dev/null +++ b/packages/benchmarking/tsconfig.json @@ -0,0 +1,10 @@ +{ + "$schema": "http://json.schemastore.org/tsconfig", + "extends": "@loopback/build/config/tsconfig.common.json", + "compilerOptions": { + "outDir": "dist", + "rootDir": "src", + "types": ["node", "mocha"] + }, + "include": ["src"] +}