From 70f0867e0915977d681fe674873a2b204bf31a1a Mon Sep 17 00:00:00 2001 From: Adam <65679285+adamkoot@users.noreply.github.com> Date: Thu, 24 Sep 2026 12:24:59 +0200 Subject: [PATCH 01/22] feat(lore-0311): add four paid usage plans and widen portal IAM Basic, Analyst, Lite and Pro (variant B figures) as per-env config beside the untouched free plan; the synthesized free plan, its key, plan key and SSM parameter are byte-identical to 1966975a. The portal attach policy keeps its construct id and name and now holds exactly three sid'd grants: GET /usageplans, GET /usageplans/*/usage, POST /usageplans/*/keys. The api-handler gets PORTAL_API_ID_PARAM and PORTAL_API_STAGE so the backend can pick the plan on our stage. --- infra/envs/production.json | 6 + infra/src/lib/stacks/api-gateway-stack.ts | 153 ++++++++++++++-------- infra/src/lib/stacks/compute-stack.ts | 97 ++++++++++---- infra/src/lib/types.ts | 106 +++++++++++++++ 4 files changed, 285 insertions(+), 77 deletions(-) diff --git a/infra/envs/production.json b/infra/envs/production.json index 29d61c81..aef39753 100644 --- a/infra/envs/production.json +++ b/infra/envs/production.json @@ -6,6 +6,12 @@ "pricingApiFreePlanRateLimit": 1, "pricingApiFreePlanBurstLimit": 5, "pricingApiFreePlanMonthlyQuota": 100000, + "pricingApiPaidPlans": { + "basic": { "rateLimit": 3, "burstLimit": 15, "monthlyQuota": 1000000 }, + "analyst": { "rateLimit": 5, "burstLimit": 25, "monthlyQuota": 5000000 }, + "lite": { "rateLimit": 10, "burstLimit": 50, "monthlyQuota": 20000000 }, + "pro": { "rateLimit": 25, "burstLimit": 125, "monthlyQuota": 50000000 } + }, "apiGatewayCacheEnabled": true, "coverageSweepEnabled": true, "apiBaseUrl": "https://prices-api.sorobanscan.rumblefish.dev", diff --git a/infra/src/lib/stacks/api-gateway-stack.ts b/infra/src/lib/stacks/api-gateway-stack.ts index 3c56f08f..d58c3352 100644 --- a/infra/src/lib/stacks/api-gateway-stack.ts +++ b/infra/src/lib/stacks/api-gateway-stack.ts @@ -8,7 +8,7 @@ import * as targets from 'aws-cdk-lib/aws-route53-targets'; import * as ssm from 'aws-cdk-lib/aws-ssm'; import type { Construct } from 'constructs'; -import type { EnvironmentConfig } from '../types.js'; +import { PAID_PLAN_TIERS, type EnvironmentConfig } from '../types.js'; /** * Physical name of the public REST API. @@ -832,16 +832,22 @@ export class ApiGatewayStack extends cdk.Stack { } // --------------------------------------------------------------- - // UsagePlan + API key — the `pricing-api-free` tier (task 0157). + // UsagePlans + API key — the `pricing-api-free` tier (task 0157) and the + // four paid tiers beside it (task 0311). // - // One plan, because a key belongs to exactly one plan per stage and - // self-service is the default (and currently only) way to hold a key. - // Higher limits are a manual, out-of-band arrangement made by hand in the - // console — see docs/runbooks/manual-api-key-tier.md. + // Five plans on the same stage. A key belongs to exactly one plan per + // stage, but a stage belongs to any number of plans — the loadtest plan + // already shares this one. Self-service keys are issued onto the free + // plan; an operator moves a key onto a paid plan by hand (delete the plan + // key, create it on the target plan) — see + // docs/runbooks/manual-api-key-tier.md. Hand-made Custom/Enterprise plans + // stay outside CDK. // - // The construct id stays `UsagePlan` so this updates the deployed plan in - // place rather than creating a second one: every property of - // AWS::ApiGateway::UsagePlan, including UsagePlanName, is "no interruption". + // The free plan's construct id stays `UsagePlan` so this updates the + // deployed plan in place rather than creating a second one: every property + // of AWS::ApiGateway::UsagePlan, including UsagePlanName, is "no + // interruption". The paid plans are new logical ids (`UsagePlanBasic`, …) + // and touch nothing that exists. // --------------------------------------------------------------- const usagePlan = this.api.addUsagePlan('UsagePlan', { name: `pricing-api-free-${config.envName}`, @@ -856,6 +862,28 @@ export class ApiGatewayStack extends cdk.Stack { }); usagePlan.addApiStage({ stage: this.api.deploymentStage }); + // The paid plans (task 0311). No API key and no SSM parameter per plan: + // nothing is issued onto them, and the portal backend finds a key's plan by + // asking which plans hold the key (`GetUsagePlans?keyId=`), so nothing ever + // needs a paid plan's id. The name is the contract — the backend parses + // the tier out of `pricing-api--`. + for (const tier of PAID_PLAN_TIERS) { + const limits = config.pricingApiPaidPlans[tier]; + this.api + .addUsagePlan(`UsagePlan${tier[0].toUpperCase()}${tier.slice(1)}`, { + name: `pricing-api-${tier}-${config.envName}`, + throttle: { + rateLimit: limits.rateLimit, + burstLimit: limits.burstLimit, + }, + quota: { + limit: limits.monthlyQuota, + period: apigateway.Period.MONTH, + }, + }) + .addApiStage({ stage: this.api.deploymentStage }); + } + // Two separate lines here can rotate this key, by two different mechanisms: // // 1. Changing the CONSTRUCT ID changes the logical id, so CloudFormation sees @@ -899,74 +927,97 @@ export class ApiGatewayStack extends cdk.Stack { }); usagePlan.addApiKey(apiKey); + // Also read by the portal backend at cold start (task 0311, + // `PORTAL_API_ID_PARAM`): `plan_of` keeps only the plans whose apiStages + // name this API + stage. Through SSM for the same cycle reason as the plan + // id below. new ssm.StringParameter(this, 'ApiGatewayIdParam', { parameterName: `/prices/${config.envName}/api-gateway-id`, stringValue: this.api.restApiId, description: `REST API ID for prices-${config.envName}-api`, }); - // The onboarding backend (task 0160) issues keys and reads per-key usage, - // both of which need the plan id. It lives in ComputeStack, which this stack - // depends on, so it cannot read the plan object without closing the cycle — - // same shape as the apiBaseUrl problem in task 0124. Publish via SSM instead. + // The onboarding backend (task 0160) issues keys onto the free plan, so it + // needs the plan id — the target of a first issue and the fallback when a + // rework finds no previous plan (task 0311; usage is now read on whichever + // plan the key is on, found by key, and needs no id from here). It lives in + // ComputeStack, which this stack depends on, so it cannot read the plan + // object without closing the cycle — same shape as the apiBaseUrl problem + // in task 0124. Publish via SSM instead. The description string below is + // left as it is: changing it would touch the deployed parameter for nothing. new ssm.StringParameter(this, 'PricingApiFreePlanIdParam', { parameterName: `/prices/${config.envName}/pricing-api-free-plan-id`, stringValue: usagePlan.usagePlanId, description: `Usage plan ID for pricing-api-free-${config.envName} (key issuance + GetUsage)`, }); - // The control-plane grants that need the plan id (tasks 0187 and 0188). - // Declared here rather than in `ComputeStack` for the cycle reason on - // `apiHandlerRole` in the props above; their four siblings are declared - // there. + // The portal's `/usageplans` grants (tasks 0187, 0188, widened by 0311). + // Declared here rather than in `ComputeStack`, beside their five siblings + // there, for a reason that outlived the one it started with. + // + // It started as the cycle: the grants named the free plan's id, and + // `iam.Policy` rather than `apiHandlerRole.addToPrincipalPolicy` because + // the latter appends to the role's default policy — a resource of + // ComputeStack — so the plan id would have travelled as an export of THIS + // stack imported by that one. Since task 0311 no statement references the + // plan id, but the policy STAYS here with its construct id and policyName: + // moving it to ComputeStack is a delete in one stack and a create in + // another, with a window in which the Lambda can attach no key and key + // issuance breaks. Renaming it is a replacement for the same cosmetic gain. // - // `iam.Policy` rather than `apiHandlerRole.addToPrincipalPolicy`, and the - // distinction is the whole point: `addToPrincipalPolicy` would append to - // the role's default policy, which is a resource of ComputeStack, so the - // plan id would travel as an export of THIS stack imported by that one — - // the cycle again, just written differently. A standalone `Policy` is a - // resource of this stack that names the role, so the reference runs - // ApiGateway -> Compute like every other one here. + // Three statements, and they are the whole set. Task 0188's decision 1 was + // "one plan's ARN, nothing wider"; task 0311 widens it deliberately, + // because the dashboard must state the key's OWN plan and the usage counted + // on it, and a rework must keep a paid user on their paid plan: // - // Two statements, one sub-resource each, on THIS plan alone: + // - `GET /usageplans` is `GetUsagePlans?keyId=` — which plans hold this + // key. The keyId filter is a query parameter, not a resource, so this + // cannot be scoped below the collection. Read-only; it reveals plan + // names and limits, never another key. + // - `GET /usageplans/*/usage` is `GetUsage` on whichever plan the key is + // on — paid, free or hand-made. The usage sub-resource still does NOT + // permit reading a plan itself, listing its keys or changing it. + // - `POST /usageplans/*/keys` attaches a key. The code only ever attaches + // to the free plan, the key's own plan, or the previous (revoked) key's + // plan — and only one that `GetUsagePlans` reported on OUR API stage. + // Hand-made plans have no ARN known at synth, hence the wildcard. // - // - `POST …/keys` (task 0187) attaches a self-service key to the plan. - // - `GET …/usage` (task 0188) is `GetUsage` — reading per-key consumption - // for the dashboard. `GET` on the usage sub-resource does NOT permit - // reading the plan itself (`GET /usageplans/{id}`), listing its keys - // (`GET …/keys`), or changing its limits — the resource path is the - // scope, and `/usage` is the narrowest form this call has. + // Deliberately NOT granted: + // - `GET /usageplans/{id}` to validate the plan at cold start. 0187's + // decision 22 rejected cold-start validation (a warm container still + // misses a plan that changes under it, and the attach path + // disambiguates a dead plan id into `PlanNotFound` loudly), and + // `GetUsagePlans` already returns every figure the dashboard shows. + // - `DELETE`/`PATCH` on a plan or a plan key. The code never moves a key + // between plans or changes limits; an operator does, by hand. + // - `GET /usageplans/*/keys`. Nothing lists a plan's members. // - // Deliberately NOT granted, though task 0187's review suggested deciding it - // here: `GET /usageplans/{id}` to validate the plan at cold start. It would - // turn a stale plan id into an init failure instead of a runtime one — but - // 0187's decision 22 already rejected cold-start validation (a warm - // container still misses a plan that changes under it, and the attach path - // disambiguates a dead plan id into `PlanNotFound` loudly), and `GetUsage` - // against a wrong plan id fails visibly on the first dashboard load. An - // extra standing grant to move one failure earlier is not worth it. - // The construct id and policyName predate the second statement (task 0187 - // named them for the attach, then task 0188 added the usage read) and are - // KEPT: renaming an AWS::IAM::Policy is a resource replacement bought for - // a cosmetic gain, on the policy whose absence breaks key issuance. Task - // 0194's audit should read this policy as "the portal grants that need - // the plan id", whatever the name says. + // A `sid` on every statement is load-bearing: cdk.json enables + // `@aws-cdk/aws-iam:minimizePolicies`, which merges sid-less statements — + // the two GETs would collapse into one and the set would stop reading as + // three. Task 0194's audit should read this policy as "the portal's + // `/usageplans` grants", whatever the name says. new iam.Policy(this, 'PortalAttachKeyToFreePlan', { policyName: `prices-${config.envName}-portal-attach-key`, roles: [apiHandlerRole], statements: [ new iam.PolicyStatement({ - sid: 'PortalAttachKeyToFreePlan', - actions: ['apigateway:POST'], + sid: 'PortalListUsagePlansByKey', + actions: ['apigateway:GET'], + resources: [`arn:aws:apigateway:${config.awsRegion}::/usageplans`], + }), + new iam.PolicyStatement({ + sid: 'PortalReadAnyPlanUsage', + actions: ['apigateway:GET'], resources: [ - `arn:aws:apigateway:${config.awsRegion}::/usageplans/${usagePlan.usagePlanId}/keys`, + `arn:aws:apigateway:${config.awsRegion}::/usageplans/*/usage`, ], }), new iam.PolicyStatement({ - sid: 'PortalReadFreePlanUsage', - actions: ['apigateway:GET'], + sid: 'PortalAttachKeyToAnyPlan', + actions: ['apigateway:POST'], resources: [ - `arn:aws:apigateway:${config.awsRegion}::/usageplans/${usagePlan.usagePlanId}/usage`, + `arn:aws:apigateway:${config.awsRegion}::/usageplans/*/keys`, ], }), ], diff --git a/infra/src/lib/stacks/compute-stack.ts b/infra/src/lib/stacks/compute-stack.ts index aaf9958b..d99eed17 100644 --- a/infra/src/lib/stacks/compute-stack.ts +++ b/infra/src/lib/stacks/compute-stack.ts @@ -205,6 +205,13 @@ export class ComputeStack extends cdk.Stack { * generates the id and it changes if the plan is ever replaced. So the * handler reads it at cold start through the Parameters and Secrets extension * already attached below, exactly as it reads secret VALUES by NAME. + * + * Two siblings since task 0311, set beside it on the Function env and for + * the same reason: `PORTAL_API_ID_PARAM`, the NAME of the parameter holding + * the REST API id (`/prices/{env}/api-gateway-id`, also published by + * `ApiGatewayStack`), and `PORTAL_API_STAGE`, the stage name — a plain + * literal, because the stage name IS `envName`. With both the backend keeps + * only the usage plans on OUR API stage when it asks which plan a key is on. */ public readonly portalFreePlanParameterName: string; @@ -514,13 +521,16 @@ export class ComputeStack extends cdk.Stack { // Self-service API keys (task 0187) — API Gateway CONTROL plane. // --------------------------------------------------------------- // - // Five of the seven calls the portal makes. The other two — - // `POST /usageplans/{id}/keys` (task 0187's attach) and - // `GET /usageplans/{id}/usage` (task 0188's `GetUsage`) — are granted in - // `ApiGatewayStack` instead, because the plan id lives there and importing - // it here would close the Compute -> Gateway -> Compute cycle described on - // `portalFreePlanParameterName` above. Each grant is declared where its - // resource is known; task 0194 audits the set as one policy. + // Five of the eight calls the portal makes. The other three are the + // `/usageplans` grants — `GET /usageplans` (task 0311's `GetUsagePlans` + // by key), `GET /usageplans/*/usage` (`GetUsage` on the key's own plan) + // and `POST /usageplans/*/keys` (the attach) — and they are granted in + // `ApiGatewayStack`'s standalone policy. They once named the free plan's + // id, which lives there (importing it here would close the Compute -> + // Gateway -> Compute cycle described on `portalFreePlanParameterName` + // above); since task 0311 they name no id, and the policy stays there + // because moving it is a delete+create with a window in which key + // issuance breaks. Task 0194 audits the set as one policy. // // Control-plane ARNs carry no account id — `arn:aws:apigateway:::` // with a doubled colon — and the resource is the API's own path. @@ -607,10 +617,10 @@ export class ComputeStack extends cdk.Stack { // this is again exposure under code execution, not feature behaviour. // // What is deliberately NOT here: `apigateway:*`, `PUT /tags/*` on anything - // but API keys, and any grant on `/usageplans` beyond the key attachment - // and the usage read — both of those need the plan id, so both live in - // `ApiGatewayStack`'s standalone policy (`POST …/keys` for 0187's attach, - // `GET …/usage` for 0188's `GetUsage`). + // but API keys, and any grant on `/usageplans` — the three this feature + // takes (list by key, any plan's usage, attach to any plan) live in + // `ApiGatewayStack`'s standalone policy, which also states what is not + // granted there (`GET /usageplans/{id}`, `DELETE`, `PATCH`). // // `DELETE` **is** here, and it is this slice's: the reconciler removes // duplicate keys after a double-submit ("keep the earliest createdDate, @@ -712,6 +722,24 @@ export class ComputeStack extends cdk.Stack { }), ); + // The REST API id, read at cold start (task 0311) — `plan_of` keeps only + // the usage plans whose apiStages name this API + stage. Currently + // redundant and kept deliberately, for exactly the reason stated on + // `PortalReadFreePlanIdParameter` above: the baseline's + // `ReadSsmNamespaces` already covers `/prices/${envName}/*`, and this + // statement names the parameter the portal depends on, stated rather than + // implied, so a narrowed baseline cannot silently close the portal at the + // next cold start. + this.apiHandlerRole.addToPrincipalPolicy( + new iam.PolicyStatement({ + sid: 'PortalReadApiGatewayIdParameter', + actions: ['ssm:GetParameter'], + resources: [ + `arn:aws:ssm:${awsRegion}:${accountId}:parameter/prices/${envName}/api-gateway-id`, + ], + }), + ); + // The eligibility gate's two knobs (task 0189), read at runtime — per // issuance, not at cold start alone. The same currently-redundant-and-kept // reasoning as `PortalReadFreePlanIdParameter` above: the baseline's @@ -746,7 +774,8 @@ export class ComputeStack extends cdk.Stack { // DISARMED; the per-key rate and monthly quota are enforced at the API // Gateway usage plan (ADR 0008; limits set by task 0157 — // `pricingApiFreePlanRateLimit` / `pricingApiFreePlanMonthlyQuota`, not the design - // doc's 100 req/s). `reservedConcurrentExecutions` is the optional SLO escape + // doc's 100 req/s — and, for a key an operator moved, by the paid plan it + // is on, `pricingApiPaidPlans`, task 0311). `reservedConcurrentExecutions` is the optional SLO escape // hatch (only set when configured). API Gateway grants invoke via the // integration's resource policy (no role-cycle, unlike the SQS ESM above). // @@ -790,7 +819,7 @@ export class ComputeStack extends cdk.Stack { // opening creates has to be unwound to close it again. // // ⚠️ **This value is a deploy gate, not just a flag.** With it true the - // handler resolves the portal's configuration AT COLD START, from FOUR + // handler resolves the portal's configuration AT COLD START, from FIVE // reads, and the portal opens only if every one of them succeeds: // // 1. `load_portal_oauth` (`config.rs`) on the Discord OAuth secret @@ -808,6 +837,12 @@ export class ComputeStack extends cdk.Stack { // `/prices/{env}/discord-guild-id` and // `/prices/{env}/min-account-age-minutes` — operator-seeded, // runbook §2a + // 5. `load_portal_keys` again (task 0311), on the SSM parameter named + // by `PORTAL_API_ID_PARAM`, i.e. `/prices/{env}/api-gateway-id`. + // The same deploy-order caveat as read 2: `ApiGatewayStack` + // publishes it and deploys AFTER this stack. It has existed since + // the gateway's first deploy, so this bites only a fresh + // environment — where read 2 already does // // A failed read CLOSES the portal in that execution environment and // logs `portal closed at cold start` on the api-handler; it does NOT @@ -832,7 +867,7 @@ export class ComputeStack extends cdk.Stack { // Set unconditionally, which is what kept opening the portal to the // one-word diff above rather than a two-line change made under time // pressure. With the flag now true the read is no longer conditional: - // this name resolving to a missing secret is read 1 of the four fatal + // this name resolving to a missing secret is read 1 of the five fatal // cold-start reads listed on `PORTAL_ENABLED`. PORTAL_OAUTH_SECRET_NAME: this.portalOauthSecretName, // The NAME of the SSM parameter holding the `pricing-api-free` usage @@ -840,13 +875,23 @@ export class ComputeStack extends cdk.Stack { // a name, why it is not a cross-stack reference, and why it is not // hard-coded. Read through the same extension layer, at cold start — // with `PORTAL_ENABLED` now true the control-plane client IS built in - // every process, and this read is read 2 of the four listed on + // every process, and this read is read 2 of the five listed on // `PORTAL_ENABLED`, the one whose parameter `ApiGatewayStack` publishes // after this stack deploys. // // Set unconditionally alongside `PORTAL_OAUTH_SECRET_NAME`, and for the // same reason: opening the portal stayed a one-word diff. PORTAL_FREE_PLAN_PARAM: this.portalFreePlanParameterName, + // The NAME of the SSM parameter holding the REST API id, and the stage + // name (task 0311) — see `portalFreePlanParameterName` for why a name + // and not a cross-stack reference. `GetUsagePlans?keyId=` returns every + // plan holding the key, across every API in the account (the loadtest + // and partner plans share it); the backend keeps only the plans whose + // apiStages contain this (apiId, stage). The stage name IS `envName` + // (`ApiGatewayStack`'s `stageName`), so it is a literal. Read 5 of the + // five listed on `PORTAL_ENABLED`. + PORTAL_API_ID_PARAM: `/prices/${envName}/api-gateway-id`, + PORTAL_API_STAGE: envName, // The NAMES of the eligibility gate's two SSM parameters (task 0189): // which Discord guild membership is checked against, and the minimum // account age in minutes. Names, never values — the handler resolves @@ -859,18 +904,18 @@ export class ComputeStack extends cdk.Stack { // same one-word-diff reasoning as the two names above. PORTAL_GUILD_ID_PARAM: `/prices/${envName}/discord-guild-id`, PORTAL_MIN_ACCOUNT_AGE_PARAM: `/prices/${envName}/min-account-age-minutes`, - // The free plan's per-key rate limit, for the portal dashboard to STATE - // (task 0188) — the same `pricingApiFreePlanRateLimit` ApiGatewayStack - // hands to `addUsagePlan`, so the figure on the page and the figure the - // gateway enforces cannot disagree. + // The free plan's per-key rate limit, served by `/config` (task 0188) + // — the same `pricingApiFreePlanRateLimit` ApiGatewayStack hands to + // `addUsagePlan`, so the figure stated and the figure the gateway + // enforces cannot disagree. // - // It travels as an env var rather than being read back from - // `GetUsagePlan` because that would cost the portal a control-plane - // grant task 0188 deliberately does not take, and rather than being a - // literal in the bundle because that is the one number on that panel - // that could then go stale: raise the limit here, deploy, and a - // dashboard whose stated theme is honesty would keep stating the old - // one. Not a secret, and not conditional on `PORTAL_ENABLED` — same + // Since task 0311 the signed-in dashboard no longer states this: it + // reads the key's OWN plan through `GetUsagePlans` (0311 took the + // grant 0188 had declined) and shows that plan's figures. This stays + // for what has no key to ask about — the no-key state, the landing + // page, and the fallback while the usage call is unanswered. A + // literal in the bundle would go stale the moment the limit changed. + // Not a secret, and not conditional on `PORTAL_ENABLED` — same // one-word-diff reasoning as the two names above. PORTAL_RATE_LIMIT: String(config.pricingApiFreePlanRateLimit), PARAMETERS_SECRETS_EXTENSION_CACHE_ENABLED: 'true', diff --git a/infra/src/lib/types.ts b/infra/src/lib/types.ts index 4df59c17..44ce4930 100644 --- a/infra/src/lib/types.ts +++ b/infra/src/lib/types.ts @@ -14,6 +14,31 @@ export interface CicdConfig { readonly githubRepo: string; } +/** + * The four paid usage plans beside `pricing-api-free` (task 0311). The key is + * also the tier's segment of the AWS plan name, `pricing-api-${tier}-${envName}` + * — the portal backend parses the tier back out of exactly that name. + */ +export type PaidPlanTier = 'basic' | 'analyst' | 'lite' | 'pro'; + +/** Every paid tier, in ascending order of limits. */ +export const PAID_PLAN_TIERS: readonly PaidPlanTier[] = [ + 'basic', + 'analyst', + 'lite', + 'pro', +]; + +/** The three figures a paid usage plan carries (task 0311). */ +export interface PlanLimits { + /** Sustained requests/second per key (UsagePlan throttle.rateLimit). */ + readonly rateLimit: number; + /** Token-bucket capacity above the rate (UsagePlan throttle.burstLimit). */ + readonly burstLimit: number; + /** Requests per calendar month (UsagePlan quota.limit, Period.MONTH). */ + readonly monthlyQuota: number; +} + /** * Per-environment configuration for the prices-api CDK app. * @@ -70,6 +95,22 @@ export interface EnvironmentConfig { * quota — encoding it makes the unit impossible to misread. */ readonly pricingApiFreePlanMonthlyQuota: number; + /** + * The paid usage plans, one per tier (task 0311). Each becomes an AWS usage + * plan named `pricing-api-${tier}-${envName}` on the same stage as the free + * plan, with quota `Period.MONTH`, offset 0. + * + * Nobody is issued a paid key: an operator moves an existing key onto one of + * these plans by hand (docs/runbooks/manual-api-key-tier.md). The dashboard + * then reads whatever the key's plan says — these figures are per-env config, + * not something the portal code depends on. + * + * NOT held to the one-tenth-of-stage guard the free plan is (see + * `planVsStage` in `validateConfig`): that guard is about keys anybody can + * mint by signing in. A paid plan is only checked against the stage default + * itself, because a plan above the per-method default cannot be delivered. + */ + readonly pricingApiPaidPlans: Readonly>; /** * Whether the API Gateway stage response cache (0.5 GB) is enabled. Per-route * TTLs are fixed in `ApiGatewayStack` per §2.1. @@ -632,6 +673,66 @@ export function validateConfig(config: EnvironmentConfig): void { `pricingApiFreePlanMonthlyQuota must be a positive integer, got: ${config.pricingApiFreePlanMonthlyQuota}`, ); } + // The paid plans (task 0311): the same three checks as the free plan, per + // tier, with the same `< 1` on the burst for the same reason — errors are + // accumulated, not short-circuited, so an invalid rate must not let an + // invalid burst through unreported. + const paid = config.pricingApiPaidPlans as + | Readonly> + | undefined; + if (!paid || typeof paid !== 'object') { + errors.push('pricingApiPaidPlans missing or not an object'); + } else { + for (const tier of PAID_PLAN_TIERS) { + const plan = paid[tier]; + const field = `pricingApiPaidPlans.${tier}`; + if (!plan || typeof plan !== 'object') { + errors.push(`${field} missing or not an object`); + continue; + } + if (!Number.isInteger(plan.rateLimit) || plan.rateLimit < 1) { + errors.push( + `${field}.rateLimit must be a positive integer, got: ${plan.rateLimit}`, + ); + } + if ( + !Number.isInteger(plan.burstLimit) || + plan.burstLimit < 1 || + plan.burstLimit < plan.rateLimit + ) { + errors.push( + `${field}.burstLimit must be a positive integer >= ${field}.rateLimit (${plan.rateLimit}), got: ${plan.burstLimit}`, + ); + } + if (!Number.isInteger(plan.monthlyQuota) || plan.monthlyQuota < 1) { + errors.push( + `${field}.monthlyQuota must be a positive integer, got: ${plan.monthlyQuota}`, + ); + } + // A plan above the stage's per-method default cannot be delivered: the + // stage throttle 429s the key before the plan's own limit is reached + // (docs/runbooks/manual-api-key-tier.md). A plain `<=`, deliberately not + // the one-tenth guard below — see `planVsStage`. + if ( + Number.isInteger(plan.rateLimit) && + Number.isInteger(config.apiGatewayThrottleRate) && + plan.rateLimit > config.apiGatewayThrottleRate + ) { + errors.push( + `${field}.rateLimit (${plan.rateLimit}) exceeds apiGatewayThrottleRate (${config.apiGatewayThrottleRate}): the stage default would throttle the key first`, + ); + } + if ( + Number.isInteger(plan.burstLimit) && + Number.isInteger(config.apiGatewayThrottleBurst) && + plan.burstLimit > config.apiGatewayThrottleBurst + ) { + errors.push( + `${field}.burstLimit (${plan.burstLimit}) exceeds apiGatewayThrottleBurst (${config.apiGatewayThrottleBurst}): the stage default would throttle the key first`, + ); + } + } + } if (typeof config.coverageSweepEnabled !== 'boolean') { errors.push( `coverageSweepEnabled must be a boolean, got: ${config.coverageSweepEnabled}`, @@ -756,6 +857,11 @@ export function validateConfig(config: EnvironmentConfig): void { } } + // Only the free plan is listed. The paid plans (`pricingApiPaidPlans`, task + // 0311) are deliberately absent: this guard is about keys anybody can mint by + // signing in, and a paid key is placed on its plan by an operator. Held to it, + // Lite and Pro could not exist at all (Pro's 25 req/s x 10 = 250 > 200). They + // are checked against the stage default with a plain `<=` above instead. const planVsStage: ReadonlyArray = [ [ From 723de4e21d0091e41b97cbde36f49a44c244c658 Mon Sep 17 00:00:00 2001 From: Adam <65679285+adamkoot@users.noreply.github.com> Date: Thu, 24 Sep 2026 12:38:02 +0200 Subject: [PATCH 02/22] feat(lore-0311): report the key's own plan on /api/usage Gateway::plan_of lists GetUsagePlans by key, keeps the plan on our API stage (PORTAL_API_ID_PARAM + PORTAL_API_STAGE) and parses the tier from pricing-api--; anything else is custom. usage_of and the attach now take the plan id, and PlanNotFound names the plan it got. /api/usage gains `plan` and states the plan's quota as `limit` (used + remaining is a warn-logged cross-check). A key on no plan for our stage answers plan: null; a plan without a quota answers null counters without calling GetUsage; MONTH is the calendar month whatever the offset, DAY the UTC day, WEEK and unknown periods are reported, not guessed. Each cached answer is re-checked against its own period rule. The mock control plane learns GetUsagePlans, plan-aware attach and usage reads. --- packages/prices-api/src/bin/serve.rs | 9 +- packages/prices-api/src/config.rs | 133 +++++- .../prices-api/src/portal/keys/gateway.rs | 438 +++++++++++++++-- packages/prices-api/src/portal/keys/mod.rs | 12 +- packages/prices-api/src/portal/mod.rs | 15 +- packages/prices-api/src/portal/usage/mod.rs | 449 +++++++++++++++--- packages/prices-api/tests/portal_auth.rs | 5 +- packages/prices-api/tests/portal_issue.rs | 57 ++- .../prices-api/tests/portal_keys/harness.rs | 249 +++++++++- packages/prices-api/tests/portal_keys_logs.rs | 5 +- packages/prices-api/tests/portal_rework.rs | 10 +- packages/prices-api/tests/portal_usage.rs | 369 +++++++++++++- 12 files changed, 1550 insertions(+), 201 deletions(-) diff --git a/packages/prices-api/src/bin/serve.rs b/packages/prices-api/src/bin/serve.rs index 4fed6e3c..9be7f94f 100644 --- a/packages/prices-api/src/bin/serve.rs +++ b/packages/prices-api/src/bin/serve.rs @@ -50,11 +50,14 @@ async fn main() { // Self-service key issuance (task 0187). This build has no Parameters and // Secrets extension client, so the plan id comes from `PORTAL_FREE_PLAN_ID` - // — a local-only variable that is compiled out of the Lambda. The AWS - // credentials are whatever the ambient profile provides, and they are real: + // and the REST API id from `PORTAL_API_ID` (task 0311) — local-only + // variables that are compiled out of the Lambda — and the stage from + // `PORTAL_API_STAGE`, which the Lambda reads too. The AWS credentials are + // whatever the ambient profile provides, and they are real: // // PORTAL_ENABLED=true PORTAL_OAUTH_SECRET_FILE=.portal-oauth.json \ - // PORTAL_FREE_PLAN_ID= AWS_PROFILE= \ + // PORTAL_FREE_PLAN_ID= PORTAL_API_ID= \ + // PORTAL_API_STAGE=production AWS_PROFILE= \ // cargo run -p prices-api --features local-server --bin serve // // **Every key this creates and deletes is a production key** — there is one diff --git a/packages/prices-api/src/config.rs b/packages/prices-api/src/config.rs index 9710cdd7..cbe6d6f7 100644 --- a/packages/prices-api/src/config.rs +++ b/packages/prices-api/src/config.rs @@ -33,19 +33,19 @@ pub struct AppConfig { /// half-built portal to the internet. Defaults are chosen per flag by what /// goes wrong when the variable is forgotten. pub portal_enabled: bool, - /// The free plan's per-key rate limit, requests per second, for the portal - /// dashboard to state (task 0188). + /// The free plan's per-key rate limit, requests per second, served by + /// `/config` (task 0188). /// /// Read from `PORTAL_RATE_LIMIT`, which `compute-stack.ts` sets from /// `pricingApiFreePlanRateLimit` — the same config value - /// `api-gateway-stack.ts` feeds to `addUsagePlan`. It travels this way - /// rather than being read back from `GetUsagePlan` because that would cost - /// the portal a control-plane grant task 0188 deliberately does not take, - /// and rather than being a literal in the frontend because that is the one - /// number on the panel that could then drift from what the gateway - /// enforces: raise the limit in `infra/envs/production.json`, deploy, and a - /// dashboard whose stated theme is rendering honestly would keep stating - /// the old figure. + /// `api-gateway-stack.ts` feeds to `addUsagePlan`. Since task 0311 the + /// signed-in dashboard does not state this figure: `/api/usage` reads the + /// key's OWN plan through `GetUsagePlans` and reports its figures. This + /// stays for what has no key to ask about — the no-key state, the landing + /// page, and the fallback while the usage call is unanswered — and it + /// stays config-fed rather than a literal in the frontend, because a + /// literal would drift from what the gateway enforces the moment + /// `infra/envs/production.json` changed. /// /// `None` — unset, or set to something that is not a positive integer — /// means this deployment cannot say what the limit is, and the page omits @@ -248,13 +248,27 @@ impl AppConfig { /// the same shape of problem `apiBaseUrl` has. And it must not be /// hard-coded, because a usage-plan id is generated by AWS and changes if /// the plan is ever replaced. + /// + /// # The API id and stage (task 0311) + /// + /// `Gateway::plan_of` keeps only the usage plans on OUR API stage, so the + /// client also needs the REST API id and the stage name. The id arrives + /// exactly as the plan id does — `PORTAL_API_ID_PARAM` names the SSM + /// parameter `ApiGatewayStack` publishes at `/prices/{env}/api-gateway-id`, + /// with a `PORTAL_API_ID` override for a local run compiled out of the + /// Lambda — and the stage is the plain `PORTAL_API_STAGE`, because the + /// stage name is `envName` and needs no lookup. pub async fn load_portal_keys(&mut self) -> Result<(), PortalKeysError> { if !self.portal_enabled { return Ok(()); } let plan_id = free_plan_id().await?; - self.portal_keys = - Some(crate::portal::keys::gateway::Gateway::from_ambient_config(plan_id).await); + let api_id = api_id().await?; + let stage = api_stage()?; + self.portal_keys = Some( + crate::portal::keys::gateway::Gateway::from_ambient_config(plan_id, api_id, stage) + .await, + ); Ok(()) } @@ -460,6 +474,23 @@ pub enum PortalKeysError { Fetch { name: String, message: String }, #[error("SSM parameter `{name}` holds an empty usage-plan id")] Empty { name: String }, + #[error( + "the portal is open but no REST API id is configured; set PORTAL_API_ID_PARAM to the SSM \ + parameter holding it (ApiGatewayStack publishes it at /prices//api-gateway-id), \ + or PORTAL_API_ID on a local run" + )] + ApiIdNoSource, + #[error( + "reading the REST API id from SSM parameter `{name}` (PORTAL_API_ID_PARAM) failed: {message}" + )] + ApiIdFetch { name: String, message: String }, + #[error("SSM parameter `{name}` (PORTAL_API_ID_PARAM) holds an empty REST API id")] + ApiIdEmpty { name: String }, + #[error( + "the portal is open but no API stage is configured; set PORTAL_API_STAGE to the stage \ + name (the environment name, e.g. `production`)" + )] + NoStage, } /// Resolve the `pricing-api-free` usage-plan id. @@ -493,35 +524,86 @@ async fn free_plan_id() -> Result { // which is exactly what an operator gets from `echo | aws ssm put-parameter` // — would produce a malformed request that reports as a control-plane // failure rather than as the typo it is. - let id = fetch_plan_id(&name).await?.trim().to_string(); + let id = fetch_parameter(&name) + .await + .map_err(|message| PortalKeysError::Fetch { + name: name.clone(), + message, + })? + .trim() + .to_string(); if id.is_empty() { return Err(PortalKeysError::Empty { name }); } Ok(id) } +/// Resolve our REST API id (task 0311) — the same shape as [`free_plan_id`]. +async fn api_id() -> Result { + // A direct id, for a local run. **Compiled out of the Lambda** for the + // reason `PORTAL_FREE_PLAN_ID` is: this value decides which usage plans + // count as the key's plan, and a configuration change must not be able to + // point the portal at an API of somebody else's choosing. + #[cfg(not(feature = "lambda"))] + if let Ok(id) = std::env::var("PORTAL_API_ID") + && !id.trim().is_empty() + { + return Ok(id.trim().to_string()); + } + + let Ok(name) = std::env::var("PORTAL_API_ID_PARAM") else { + return Err(PortalKeysError::ApiIdNoSource); + }; + if name.is_empty() { + return Err(PortalKeysError::ApiIdNoSource); + } + // Trimmed for the reason the plan id is: an `echo | put-parameter` + // newline would otherwise make every plan fail the stage comparison and + // report every key as on no plan. + let id = fetch_parameter(&name) + .await + .map_err(|message| PortalKeysError::ApiIdFetch { + name: name.clone(), + message, + })? + .trim() + .to_string(); + if id.is_empty() { + return Err(PortalKeysError::ApiIdEmpty { name }); + } + Ok(id) +} + +/// Our stage name (task 0311), from `PORTAL_API_STAGE`. +fn api_stage() -> Result { + std::env::var("PORTAL_API_STAGE") + .ok() + .map(|stage| stage.trim().to_string()) + .filter(|stage| !stage.is_empty()) + .ok_or(PortalKeysError::NoStage) +} + /// Read the parameter through the Parameters and Secrets extension — the same /// localhost listener, token and in-process cache the mTLS bundle and the OAuth /// secret already use, so a warm container never calls Systems Manager on the /// path that issues a key. +/// +/// The error is the message alone; the caller wraps it in the variant naming +/// which parameter it was reading. #[cfg(feature = "aws-mtls")] -async fn fetch_plan_id(name: &str) -> Result { +async fn fetch_parameter(name: &str) -> Result { prices_clickhouse::mtls::fetch_parameter_string(name) .await - .map_err(|e| PortalKeysError::Fetch { - name: name.to_string(), - message: e.to_string(), - }) + .map_err(|e| e.to_string()) } #[cfg(not(feature = "aws-mtls"))] -async fn fetch_plan_id(name: &str) -> Result { - Err(PortalKeysError::Fetch { - name: name.to_string(), - message: "this build has no Parameters and Secrets extension client (build with \ - `--features lambda`, or set PORTAL_FREE_PLAN_ID for a local run)" +async fn fetch_parameter(_name: &str) -> Result { + Err( + "this build has no Parameters and Secrets extension client (build with \ + `--features lambda`, or set PORTAL_FREE_PLAN_ID and PORTAL_API_ID for a local run)" .into(), - }) + ) } #[cfg(test)] @@ -573,7 +655,8 @@ mod portal_load_tests { /// back CLOSED — flag and all three sources — rather than half-open. No /// environment variable is set here on purpose (`set_var` races the other /// test threads, see `AppConfig::portal_endpoints`): the loaders read - /// `PORTAL_OAUTH_SECRET_FILE`/`_NAME`, `PORTAL_FREE_PLAN_ID`/`_PARAM` and + /// `PORTAL_OAUTH_SECRET_FILE`/`_NAME`, `PORTAL_FREE_PLAN_ID`/`_PARAM`, + /// `PORTAL_API_ID`/`_PARAM`, `PORTAL_API_STAGE` and /// the eligibility seams, and a developer's shell exporting one of them /// only moves which loader fails, not the outcome asserted. #[tokio::test] diff --git a/packages/prices-api/src/portal/keys/gateway.rs b/packages/prices-api/src/portal/keys/gateway.rs index d7175a84..3b002369 100644 --- a/packages/prices-api/src/portal/keys/gateway.rs +++ b/packages/prices-api/src/portal/keys/gateway.rs @@ -1,9 +1,10 @@ -//! The API Gateway **control plane**, wrapped down to seven calls (task 0187, -//! task 0188's `GetUsage`, task 0191's `UpdateApiKey`). +//! The API Gateway **control plane**, wrapped down to eight calls (task 0187, +//! task 0188's `GetUsage`, task 0191's `UpdateApiKey`, task 0311's +//! `GetUsagePlans`). //! //! Not the data plane. These are `GetApiKeys`, `CreateApiKey`, //! `CreateUsagePlanKey`, `GetApiKey`, `DeleteApiKey`, `UpdateApiKey` -//! (disable only) and `GetUsage` — the API +//! (disable only), `GetUsage` and `GetUsagePlans` (filtered by key) — the API //! the console drives — and they are the reason this slice needs no database: //! **API Gateway is the source of truth for whether a key exists** (task 0158's //! own argument, restated in 0187's context), and for how much it has been @@ -57,6 +58,8 @@ use std::time::Duration; use aws_sdk_apigateway::Client; use aws_sdk_apigateway::config::timeout::TimeoutConfig; +use aws_sdk_apigateway::types::UsagePlan; +use serde::Serialize; use super::naming::KeyRecord; @@ -118,7 +121,7 @@ impl std::fmt::Debug for KeyValue { } } -/// What [`Gateway::attach_to_free_plan`] observed. +/// What [`Gateway::attach_to_plan`] observed. /// /// Two outcomes rather than `()` because the caller has to act on the second /// one: a key that vanished between the listing and the attach is the same race @@ -149,11 +152,14 @@ pub enum Disable { /// One key's consumption over a queried period, as AWS reports it. /// /// Derived from `GetUsage`'s daily `[used, remaining]` pairs rather than read -/// off a single field: the response carries no `limit` of its own, so the limit -/// is reconstructed as `used + remaining` — the same arithmetic task 0157's -/// close verified against the live plan (`[121, 99879]` against a 100 000 -/// quota). Kept here instead of in the handler so the shape of the AWS response -/// stays a concern of this module. +/// off a single field. The response carries no `limit` of its own; since task +/// 0311 the limit the dashboard states is the plan's own quota +/// ([`PlanInfo::quota_limit`], from `GetUsagePlans`), and [`Self::limit`] — +/// `used + remaining`, the same arithmetic task 0157's close verified against +/// the live plan (`[121, 99879]` against a 100 000 quota) — survives only as a +/// cross-check the usage route logs a warning on when the two disagree. Kept +/// here instead of in the handler so the shape of the AWS response stays a +/// concern of this module. /// /// That reconstruction is sound only while `used` and `remaining` describe the /// same AWS quota period — which the queried range does not guarantee, since @@ -173,7 +179,9 @@ pub struct KeyUsage { } impl KeyUsage { - /// The plan's quota, reconstructed. See the type docs. + /// The plan's quota, reconstructed as `used + remaining` — a cross-check + /// against [`PlanInfo::quota_limit`], never the figure rendered. See the + /// type docs. pub fn limit(&self) -> u64 { self.used.saturating_add(self.remaining) } @@ -202,11 +210,17 @@ pub enum GatewayError { TooManyPages, #[error( "API Gateway says usage plan `{plan_id}` does not exist; the key was created but could \ - not be attached to a plan, so it will not work against /v1/. Check the SSM parameter \ - named by PORTAL_FREE_PLAN_PARAM (or PORTAL_FREE_PLAN_ID on a local run) against the \ - plan ApiGatewayStack publishes" + not be attached to it, so it will not work against /v1/. The free plan comes from the \ + SSM parameter named by PORTAL_FREE_PLAN_PARAM (or PORTAL_FREE_PLAN_ID on a local run) \ + — check it against the plan ApiGatewayStack publishes; a paid or custom plan is the \ + previous key's plan as GetUsagePlans reported it, so check that plan still exists" )] PlanNotFound { plan_id: String }, + #[error( + "API Gateway returned more than {MAX_PAGES} pages of usage plans for one key; refusing \ + to pick a plan from a partial list" + )] + TooManyPlanPages, #[error("API Gateway `{operation}` answered without the `{field}` field")] Incomplete { operation: &'static str, @@ -214,23 +228,168 @@ pub enum GatewayError { }, } -/// The five calls, plus the usage plan they attach to. +/// A plan's tier, as the dashboard names it (task 0311). +/// +/// Parsed from the plan's NAME, because the name is the one contract CDK and +/// this code share: `pricing-api--` for the five CDK plans +/// (`api-gateway-stack.ts`). Anything else on our stage — the loadtest plan, a +/// hand-made Enterprise plan — is [`Tier::Custom`] and is shown with its own +/// name. +#[derive(Debug, Clone, Copy, PartialEq, Eq, Serialize)] +#[serde(rename_all = "lowercase")] +pub enum Tier { + Free, + Basic, + Analyst, + Lite, + Pro, + Custom, +} + +impl Tier { + /// The five CDK tiers, and the name segment each is published under. + const NAMED: [(Tier, &'static str); 5] = [ + (Tier::Free, "free"), + (Tier::Basic, "basic"), + (Tier::Analyst, "analyst"), + (Tier::Lite, "lite"), + (Tier::Pro, "pro"), + ]; + + /// The tier a plan named `name` is, on stage `stage`. + /// + /// **Exact** match against `pricing-api--`, not a prefix or a + /// pattern: the right prefix with the wrong stage (`pricing-api-pro-staging` + /// seen from `production`) is somebody else's plan, and an unknown tier + /// (`pricing-api-gold-production`) is one nobody told this code about — + /// both are [`Tier::Custom`], stated with their name, rather than a guess. + pub fn from_plan_name(name: &str, stage: &str) -> Tier { + Self::NAMED + .iter() + .find(|(_, segment)| name == format!("pricing-api-{segment}-{stage}")) + .map(|(tier, _)| *tier) + .unwrap_or(Tier::Custom) + } +} + +/// The usage plan a key is on, for our API stage, as `GetUsagePlans` reports +/// it (task 0311). +/// +/// Every figure is optional because every figure is optional in AWS: a plan +/// can carry no throttle and no quota at all (the "unlimited" state the +/// dashboard must state, not render as zeros). +#[derive(Debug, Clone, PartialEq)] +pub struct PlanInfo { + pub id: String, + pub name: String, + pub tier: Tier, + /// Sustained requests/second per key, `throttle.rateLimit`. + pub rate_limit: Option, + /// Token-bucket capacity, `throttle.burstLimit`. + pub burst_limit: Option, + /// Requests per quota period, `quota.limit`. + pub quota_limit: Option, + /// `quota.period` as AWS spells it: `DAY`, `WEEK`, `MONTH` — or whatever a + /// newer service answers, carried verbatim rather than guessed at. + pub quota_period: Option, + /// `quota.offset`. Reported for diagnostics ONLY: it is "the number of + /// requests subtracted from the given limit in the initial time period" + /// (the SDK's own doc), a request count — never a shift of the period's + /// start day, and nothing here reads it as one. + pub quota_offset: Option, +} + +impl PlanInfo { + /// Read one SDK plan. `None` for a plan with no id — it cannot be read or + /// attached to, so it cannot be anybody's plan in any useful sense. + fn from_sdk(plan: &UsagePlan, stage: &str) -> Option { + let id = plan.id()?.to_string(); + let name = plan.name().unwrap_or_default().to_string(); + let tier = Tier::from_plan_name(&name, stage); + let quota = plan.quota(); + Some(Self { + id, + name, + tier, + rate_limit: plan.throttle().map(|t| t.rate_limit()), + burst_limit: plan.throttle().map(|t| t.burst_limit()), + quota_limit: quota.and_then(|q| u64::try_from(q.limit()).ok()), + // `.as_str()` rather than matching the enum: `Unknown` is + // deprecated as a pattern, and a period this SDK does not know is + // exactly the one to carry verbatim. + quota_period: quota + .and_then(|q| q.period()) + .map(|p| p.as_str().to_string()), + quota_offset: quota.map(|q| q.offset()), + }) + } +} + +/// The plan, among `plans`, that is on OUR API stage — `(api_id, stage)` in +/// its `apiStages` (task 0311). +/// +/// `GetUsagePlans?keyId=` answers every plan holding the key across every API +/// in the account; the loadtest plan and a partner plan share it. A plan on +/// another API, or on no stage at all, says nothing about what our gateway +/// enforces for this key and is ignored. +/// +/// AWS allows a key on one plan per stage, so more than one match cannot +/// happen. If it does anyway it is logged as an error and the LOWEST id is +/// picked — deterministic, so every reader agrees, and never a panic in a +/// handler that also serves `/v1`. +fn select_plan(plans: &[UsagePlan], api_id: &str, stage: &str) -> Option { + let mut matching: Vec = plans + .iter() + .filter(|plan| { + plan.api_stages() + .iter() + .any(|s| s.api_id() == Some(api_id) && s.stage() == Some(stage)) + }) + .filter_map(|plan| { + let info = PlanInfo::from_sdk(plan, stage); + if info.is_none() { + tracing::warn!("GetUsagePlans listed a plan with no id; skipped"); + } + info + }) + .collect(); + matching.sort_by(|a, b| a.id.cmp(&b.id)); + if matching.len() > 1 { + tracing::error!( + plans = ?matching.iter().map(|p| p.id.as_str()).collect::>(), + "a key is on more than one usage plan for this API stage, which AWS should not \ + allow; using the lowest id" + ); + } + matching.into_iter().next() +} + +/// The eight calls, plus the free usage plan and the API stage they are +/// scoped to. #[derive(Clone)] pub struct Gateway { client: Client, /// The `pricing-api-free` usage plan id, read from SSM at cold start — /// never hard-coded and never a cross-stack reference. See - /// [`crate::AppConfig::load_portal_keys`]. + /// [`crate::AppConfig::load_portal_keys`]. The target of a first issue, + /// and of a rework that finds no previous plan (task 0311). free_plan_id: String, + /// Our REST API id (task 0311), read from SSM at cold start like the plan + /// id — [`Self::plan_of`] keeps only plans on `(api_id, stage)`. + api_id: String, + /// Our stage name (task 0311) — `envName`, passed as `PORTAL_API_STAGE`. + stage: String, } -/// Prints the plan id and nothing else. The plan id is not a secret (it is in -/// an SSM parameter any operator can read) and it is the one field worth seeing -/// in a diagnostic. +/// Prints the plan id, the API id and the stage, and nothing else. None of +/// them is a secret (the ids are in SSM parameters any operator can read) and +/// they are the fields worth seeing in a diagnostic. impl std::fmt::Debug for Gateway { fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result { f.debug_struct("Gateway") .field("free_plan_id", &self.free_plan_id) + .field("api_id", &self.api_id) + .field("stage", &self.stage) .finish_non_exhaustive() } } @@ -243,7 +402,7 @@ impl Gateway { /// flipped `PORTAL_ENABLED`, is every production cold start. With the flag /// off (tests, or a reverted deploy) this resolves no credentials and opens /// no connections. See [`crate::AppConfig::load_portal_keys`]. - pub async fn from_ambient_config(free_plan_id: String) -> Self { + pub async fn from_ambient_config(free_plan_id: String, api_id: String, stage: String) -> Self { let shared = aws_config::load_defaults(aws_sdk_apigateway::config::BehaviorVersion::latest()).await; @@ -276,6 +435,8 @@ impl Gateway { Self { client: Client::from_conf(config), free_plan_id, + api_id, + stage, } } @@ -292,7 +453,12 @@ impl Gateway { /// handed the execution role's SigV4 signature on requests that create and /// delete production API keys. #[cfg(not(feature = "lambda"))] - pub fn against(endpoint_url: &str, free_plan_id: String) -> Self { + pub fn against( + endpoint_url: &str, + free_plan_id: String, + api_id: String, + stage: String, + ) -> Self { let config = aws_sdk_apigateway::config::Builder::new() .behavior_version(aws_sdk_apigateway::config::BehaviorVersion::latest()) .region(aws_sdk_apigateway::config::Region::new("eu-central-1")) @@ -309,14 +475,76 @@ impl Gateway { Self { client: Client::from_conf(config), free_plan_id, + api_id, + stage, } } - /// The usage plan new keys are attached to. + /// The usage plan a first issue attaches to — and the fallback when a + /// rework finds no previous plan on our stage (task 0311). pub fn free_plan_id(&self) -> &str { &self.free_plan_id } + /// The usage plan `key_id` is on for OUR API stage, or `None` if it is on + /// none (task 0311). + /// + /// `GetUsagePlans` filtered by `keyId`, paged to exhaustion like + /// [`Self::list_named`] and for the same reason: picking from page one + /// could miss the one plan that matters. Past [`MAX_PAGES`] it errors + /// ([`GatewayError::TooManyPlanPages`]) rather than truncating. The pages + /// are then narrowed to our `(api_id, stage)` by [`select_plan`]. + /// + /// A throttle is [`GatewayError::Throttled`], like `GetUsage`'s, so the + /// usage route's stale-serve branch covers this call too. A `404` is + /// `Ok(None)`: the only thing it can mean with a `keyId` filter is that + /// the key is not there, and the attach that follows on the issue path + /// disambiguates a vanished key through [`Self::exists`]. + pub async fn plan_of(&self, key_id: &str) -> Result, GatewayError> { + let mut plans: Vec = Vec::new(); + let mut position: Option = None; + + for _ in 0..MAX_PAGES { + let mut request = self + .client + .get_usage_plans() + .key_id(key_id) + .limit(PAGE_LIMIT); + if let Some(position) = position.as_ref() { + request = request.position(position); + } + + let page = match request.send().await { + Ok(page) => page, + Err(e) => { + let message = sdk_message(&e); + let service_error = e.into_service_error(); + if service_error.is_too_many_requests_exception() { + return Err(GatewayError::Throttled { + operation: "GetUsagePlans", + }); + } + if service_error.is_not_found_exception() { + return Ok(None); + } + return Err(GatewayError::Call { + operation: "GetUsagePlans", + message, + }); + } + }; + + plans.extend(page.items().iter().cloned()); + + match page.position() { + Some(next) if !next.is_empty() => position = Some(next.to_string()), + _ => return Ok(select_plan(&plans, &self.api_id, &self.stage)), + } + } + + Err(GatewayError::TooManyPlanPages) + } + /// Every key whose name starts with `prefix`, across **all** pages. /// /// `nameQuery` is a prefix match and the caller must still filter to exact @@ -520,9 +748,14 @@ impl Gateway { } } - /// Attach a key to the free usage plan, which is what makes it work against - /// `/v1/`. A key that exists but is on no plan authenticates and is then - /// refused by the plan check — the confusing half-state this call closes. + /// Attach a key to usage plan `plan_id`, which is what makes it work + /// against `/v1/`. A key that exists but is on no plan authenticates and is + /// then refused by the plan check — the confusing half-state this call + /// closes. + /// + /// Which plan is the caller's decision (task 0311): the free plan for a + /// first issue, the previous key's plan for a rework — see + /// `super::resolve_target_plan`. /// /// **Idempotent**, and that is what lets the caller run it on every key it /// is about to hand out rather than only on keys it just created. API @@ -542,11 +775,15 @@ impl Gateway { /// make a hand-deleted key a dead end again); the second is a deployment /// that will never work, and reporting it as a transient race hides it /// forever. [`Self::exists`] separates them. - pub async fn attach_to_free_plan(&self, key_id: &str) -> Result { + pub async fn attach_to_plan( + &self, + key_id: &str, + plan_id: &str, + ) -> Result { match self .client .create_usage_plan_key() - .usage_plan_id(&self.free_plan_id) + .usage_plan_id(plan_id) .key_id(key_id) .key_type("API_KEY") .send() @@ -574,7 +811,7 @@ impl Gateway { // rather than in a support conversation three days later. if self.exists(key_id).await? { Err(GatewayError::PlanNotFound { - plan_id: self.free_plan_id.clone(), + plan_id: plan_id.to_string(), }) } else { Ok(Attachment::KeyGone) @@ -649,10 +886,15 @@ impl Gateway { } } - /// One key's usage against the free plan's quota between `start_date` and + /// One key's usage against plan `plan_id`'s quota between `start_date` and /// `end_date` (inclusive, `YYYY-MM-DD`), or `None` if AWS has recorded /// nothing for it in that window (task 0188). /// + /// `plan_id` is the key's OWN plan, as [`Self::plan_of`] found it (task + /// 0311): usage is counted per `(plan, key)` pair, so reading it on the free + /// plan for a key an operator moved answers an empty map — which is what + /// the dashboard showed for every paid key before 0311. + /// /// `None` is a real and **common** state, not an edge case: `GetUsage` is /// not a read-after-write surface (measured 2026-08-12, archived /// `0180/notes/R-apigw-namequery-quota-and-disable.md`), so a key issued @@ -675,6 +917,7 @@ impl Gateway { /// the case where AWS's own period rolls partway through the range. pub async fn usage_of( &self, + plan_id: &str, key_id: &str, start_date: &str, end_date: &str, @@ -686,7 +929,7 @@ impl Gateway { let mut request = self .client .get_usage() - .usage_plan_id(&self.free_plan_id) + .usage_plan_id(plan_id) .key_id(key_id) .start_date(start_date) .end_date(end_date) @@ -946,12 +1189,145 @@ mod tests { } /// A `Gateway` can end up in a diagnostic by way of `AppConfig`; nothing in - /// it may be a credential. The plan id is not one. + /// it may be a credential. The plan id, API id and stage are not. #[test] fn a_gateway_prints_only_its_plan_id() { - let gateway = Gateway::against("http://127.0.0.1:1", "plan-abc".into()); + let gateway = Gateway::against( + "http://127.0.0.1:1", + "plan-abc".into(), + "api-xyz".into(), + "production".into(), + ); let printed = format!("{gateway:?}"); assert!(printed.contains("plan-abc"), "{printed}"); + assert!(printed.contains("api-xyz"), "{printed}"); + assert!(printed.contains("production"), "{printed}"); assert!(!printed.contains("portal-keys-test"), "{printed}"); } + + use aws_sdk_apigateway::types::{ApiStage, QuotaPeriodType, QuotaSettings, ThrottleSettings}; + + const API: &str = "02mabge71l"; + const STAGE: &str = "production"; + + fn plan(id: &str, name: &str, stages: &[(&str, &str)]) -> UsagePlan { + let mut builder = UsagePlan::builder().id(id).name(name); + for (api_id, stage) in stages { + builder = builder.api_stages(ApiStage::builder().api_id(*api_id).stage(*stage).build()); + } + builder + .throttle( + ThrottleSettings::builder() + .rate_limit(3.0) + .burst_limit(15) + .build(), + ) + .quota( + QuotaSettings::builder() + .limit(1_000_000) + .offset(0) + .period(QuotaPeriodType::Month) + .build(), + ) + .build() + } + + /// The five CDK names parse to their tiers on the matching stage. + #[test] + fn each_cdk_plan_name_is_its_tier() { + for (name, tier) in [ + ("pricing-api-free-production", Tier::Free), + ("pricing-api-basic-production", Tier::Basic), + ("pricing-api-analyst-production", Tier::Analyst), + ("pricing-api-lite-production", Tier::Lite), + ("pricing-api-pro-production", Tier::Pro), + ] { + assert_eq!(Tier::from_plan_name(name, STAGE), tier, "{name}"); + } + } + + /// Anything else is Custom: a hand-made plan, the right prefix on the + /// wrong stage, and a tier nobody told this code about. + #[test] + fn any_other_plan_name_is_custom() { + for name in [ + "prices-production-loadtest-plan", + "pricing-api-pro-staging", + "pricing-api-gold-production", + "pricing-api-free-production-old", + "", + ] { + assert_eq!(Tier::from_plan_name(name, STAGE), Tier::Custom, "{name}"); + } + } + + /// Only the plan on OUR (api id, stage) is chosen; one on another API + /// and one on no stage at all are ignored. + #[test] + fn only_the_plan_on_our_stage_is_chosen() { + let plans = [ + plan( + "q7sd40", + "production-partner-plan", + &[("6l9k06w4pl", STAGE)], + ), + plan("nostage", "pricing-api-pro-production", &[]), + plan("basic1", "pricing-api-basic-production", &[(API, STAGE)]), + plan( + "otherstage", + "pricing-api-lite-production", + &[(API, "staging")], + ), + ]; + let chosen = select_plan(&plans, API, STAGE).expect("one plan is on our stage"); + assert_eq!(chosen.id, "basic1"); + assert_eq!(chosen.tier, Tier::Basic); + assert_eq!(chosen.rate_limit, Some(3.0)); + assert_eq!(chosen.burst_limit, Some(15)); + assert_eq!(chosen.quota_limit, Some(1_000_000)); + assert_eq!(chosen.quota_period.as_deref(), Some("MONTH")); + assert_eq!(chosen.quota_offset, Some(0)); + } + + /// No plan on our stage is `None` — the "issued but dead" state. + #[test] + fn no_plan_on_our_stage_is_none() { + let plans = [plan( + "q7sd40", + "production-partner-plan", + &[("6l9k06w4pl", STAGE)], + )]; + assert_eq!(select_plan(&plans, API, STAGE), None); + assert_eq!(select_plan(&[], API, STAGE), None); + } + + /// Two plans on our stage cannot happen in AWS; if it does, the lowest id + /// wins, whatever order the pages came in, and nothing panics. + #[test] + fn two_plans_on_our_stage_pick_the_lowest_id() { + let plans = [ + plan("zzz", "pricing-api-pro-production", &[(API, STAGE)]), + plan("aaa", "prices-production-loadtest-plan", &[(API, STAGE)]), + ]; + let chosen = select_plan(&plans, API, STAGE).expect("a plan is chosen"); + assert_eq!(chosen.id, "aaa"); + assert_eq!(chosen.tier, Tier::Custom); + } + + /// A plan without throttle or quota reports every figure as absent — + /// the unlimited state, never zeros. + #[test] + fn a_plan_without_limits_reports_none_not_zero() { + let bare = UsagePlan::builder() + .id("custom1") + .name("prices-production-acme-plan") + .api_stages(ApiStage::builder().api_id(API).stage(STAGE).build()) + .build(); + let info = select_plan(&[bare], API, STAGE).expect("on our stage"); + assert_eq!(info.tier, Tier::Custom); + assert_eq!(info.rate_limit, None); + assert_eq!(info.burst_limit, None); + assert_eq!(info.quota_limit, None); + assert_eq!(info.quota_period, None); + } } diff --git a/packages/prices-api/src/portal/keys/mod.rs b/packages/prices-api/src/portal/keys/mod.rs index 8a065aac..954066d2 100644 --- a/packages/prices-api/src/portal/keys/mod.rs +++ b/packages/prices-api/src/portal/keys/mod.rs @@ -1149,7 +1149,11 @@ async fn attempt( // the create and this call. Handing its value out would be handing // out a dead id — the one thing the adopt-or-recreate rule exists to // prevent — so this re-enters the flow like any other lost race. - if gateway.attach_to_free_plan(&record.id).await? == Attachment::KeyGone { + if gateway + .attach_to_plan(&record.id, gateway.free_plan_id()) + .await? + == Attachment::KeyGone + { return Ok(Attempt::Retry); } return Ok(Attempt::Done(Outcome { @@ -1222,7 +1226,11 @@ async fn attempt( // replacement. Because this call now runs before the read, it is the first // place that race can surface, so it has to answer it rather than turn a // hand-deleted key back into the dead end this slice exists to remove. - if gateway.attach_to_free_plan(&winner.id).await? == Attachment::KeyGone { + if gateway + .attach_to_plan(&winner.id, gateway.free_plan_id()) + .await? + == Attachment::KeyGone + { return Ok(Attempt::Retry); } diff --git a/packages/prices-api/src/portal/mod.rs b/packages/prices-api/src/portal/mod.rs index 663727be..a6b7775c 100644 --- a/packages/prices-api/src/portal/mod.rs +++ b/packages/prices-api/src/portal/mod.rs @@ -110,17 +110,16 @@ pub const OPENAPI_PATH: &str = "/api/api-docs-json"; pub struct PortalConfig { /// Whether the portal is open for business. pub enabled: bool, - /// The free plan's per-key rate limit in requests per second, for the - /// dashboard to state (task 0188). + /// The free plan's per-key rate limit in requests per second (task 0188). /// /// Served from here rather than written into the bundle because it is a /// per-env config value (`pricingApiFreePlanRateLimit`) that the gateway - /// enforces and the page merely reports: a literal in the frontend is the - /// one number on that panel that can drift from what is actually enforced. - /// It rides on `/config` rather than on `/usage` because the dashboard - /// states it in the no-key state too, and that state is a `404` with no - /// body to carry it — and because the limit is a property of the plan every - /// key joins, not of any one caller's key. + /// enforces and the page merely reports: a literal in the frontend would + /// drift from what is actually enforced. Since task 0311 a signed-in + /// caller WITH a key is shown their own plan's figures, from `/usage` + /// (`GetUsagePlans` on the key). This stays for what has no key to ask + /// about: the no-key state — a `404` with no body to carry a plan — the + /// landing page, and the fallback while the usage call is unanswered. /// /// Omitted from the JSON entirely when this deployment was not told what /// the limit is; the page then omits the line rather than inventing a diff --git a/packages/prices-api/src/portal/usage/mod.rs b/packages/prices-api/src/portal/usage/mod.rs index 05d8b267..ae8f8c83 100644 --- a/packages/prices-api/src/portal/usage/mod.rs +++ b/packages/prices-api/src/portal/usage/mod.rs @@ -7,7 +7,7 @@ //! //! | route | does | //! | --- | --- | -//! | `GET /api/usage` | the caller's used / remaining / limit for the current period, from `GetUsage` | +//! | `GET /api/usage` | the caller's plan (`GetUsagePlans`) and used / remaining / limit for its current period (`GetUsage` on that plan) | //! //! # This route is read-only, and that is a safety property //! @@ -28,15 +28,35 @@ //! the victim's own tab is not readable cross-origin. The worst outcome is the //! visitor seeing their own dashboard. //! +//! # The key's own plan (task 0311) +//! +//! The key is not assumed to be on the free plan. An operator moves a paid +//! user's key onto `pricing-api-{basic,analyst,lite,pro}-` by hand, and a +//! hand-made Custom plan can hold one too — so the route first asks which plan +//! holds the key on OUR API stage (`Gateway::plan_of`, `GetUsagePlans?keyId=`) +//! and reports it as `plan`. `limit` is that plan's own `quota.limit`; +//! `used + remaining` is only a cross-check, logged when it disagrees. +//! +//! Two states are stated rather than rendered as zeros: a key on **no plan** +//! for our stage (`plan: null`, every counter and period field null — the +//! "issued but dead" state, the gateway answers it `403`), and a plan with +//! **no quota** (`plan.quota_limit: null`, counters and period null, and no +//! `GetUsage` call at all — there is nothing to count against). A plan whose +//! quota period this code does not compute (`WEEK`, or a period a newer +//! service invents) reports its quota and names the period, with the period +//! fields null: reported, never guessed. +//! //! # The numbers are AWS's; the period boundary is ours //! -//! `used`, `remaining` and the reconstructed `limit` come from `GetUsage`, -//! scoped to `(usagePlanId, apiKeyId)` — no accounting of our own. The period -//! rendered around them does **not** come from AWS, because AWS documents -//! neither the reset instant nor its timezone (ADR 0010, correction #2, still -//! open — the only statement anywhere is an example caption). "The 1st of the -//! month, 00:00 UTC" is **our stated product rule**, the same one the rework -//! cap in [0191] is defined by. +//! `used` and `remaining` come from `GetUsage`, scoped to +//! `(usagePlanId, apiKeyId)` — no accounting of our own. The period rendered +//! around them does **not** come from AWS, because AWS documents neither the +//! reset instant nor its timezone (ADR 0010, correction #2, still open — the +//! only statement anywhere is an example caption). For a `MONTH` plan "the +//! 1st of the month, 00:00 UTC" is **our stated product rule**, the same one +//! the rework cap in [0191] is defined by; for a `DAY` plan it is the UTC day. +//! A quota's `offset` never shifts either: it is a request count subtracted in +//! the first period, not a start day. //! //! If AWS's counter turns out to roll at a different instant, the LABEL is a UX //! wrinkle to word around — but the NUMBERS under it are not, and that is worth @@ -81,14 +101,14 @@ use axum::extract::State; use axum::http::{HeaderMap, StatusCode}; use axum::response::{IntoResponse, Response}; use axum::routing::get; -use chrono::{SecondsFormat, Utc}; +use chrono::{NaiveDate, SecondsFormat, Utc}; use serde::Serialize; use crate::common::{cache_control, errors}; use super::auth::secret::OauthSecret; use super::keys::cap::{self, Cap}; -use super::keys::gateway::{Gateway, GatewayError}; +use super::keys::gateway::{Gateway, GatewayError, PlanInfo, Tier}; use super::keys::naming::{current_key, exact_matches, key_name, revocation_instant}; use super::period::Period; @@ -108,8 +128,10 @@ const USAGE_UNCONFIGURED: &str = "usage_unconfigured"; /// How long a cached answer is served without asking AWS again. /// -/// One dashboard load is one `GetApiKeys` + one `GetUsage`; within this window -/// every further load by the same caller is neither. 60 seconds is far inside +/// One dashboard load is one `GetApiKeys` + one `GetUsagePlans` + at most one +/// `GetUsage`; within this window every further load by the same caller is +/// none of them. The plan lives in the same entry (task 0311), so an operator's +/// plan change shows within one TTL — no external invalidation. 60 seconds is far inside /// `GetUsage`'s own reporting lag (minutes — see the module docs), so the /// cache costs the viewer no freshness AWS was offering, while a refresh /// loop at any human rate collapses to one control-plane call a minute. @@ -141,7 +163,7 @@ const USAGE_DEADLINE: Duration = Duration::from_secs(10); pub struct UsageState { /// Verifies the session cookie — the same secret sign-in issued it with. oauth: Option>, - /// The control-plane client, carrying the free plan id. `None` while the + /// The control-plane client, carrying the free plan id and our API stage. `None` while the /// portal is closed, exactly as `KeysState` holds it. gateway: Option>, /// The last good answer per caller (session `sub`), plus the per-caller @@ -296,34 +318,142 @@ pub fn routes(state: UsageState) -> Router { /// What the route answers with. Everything the dashboard renders, nothing it /// has to compute. /// -/// The three counters are one `Option` each and go absent **together**: when -/// AWS has no row for the key yet (see `Gateway::usage_of` — common for a key -/// issued minutes ago), inventing `used: 0` would be defensible but inventing -/// `remaining` and `limit` would not, and a response that is honest about two -/// fields and guessing on the third is worse than one that says "nothing -/// recorded yet". The period and `as_of` are always present — they are ours. +/// `used` and `remaining` go absent **together** when AWS has no row for the +/// key yet (see `Gateway::usage_of` — common for a key issued minutes ago): +/// inventing `used: 0` would be defensible but inventing `remaining` would +/// not. `limit` is the plan's quota, known even before AWS records a row. +/// +/// For a free key every pre-0311 field keeps its name, meaning and — apart +/// from `limit` now being the plan's quota rather than `used + remaining` — +/// its value; `plan` is added. The no-plan, no-quota and unsupported-period +/// states null the counters and the period (see the module docs). #[derive(Clone, Serialize)] struct UsageResponse { /// Requests counted against the quota this period, per AWS. used: Option, /// Requests left, as of the latest day AWS has data for. remaining: Option, - /// The plan quota, reconstructed as `used + remaining` — `GetUsage` does - /// not report it directly, and reading it from `GetUsagePlan` would cost a - /// grant this slice deliberately does not take. + /// The plan's quota, `plan.quota_limit` (task 0311). `used + remaining` is + /// only a cross-check against it, logged when it disagrees — never + /// rendered. limit: Option, /// First day of the current period, `YYYY-MM-DD` — ours: the calendar - /// month, UTC. - period_start: String, + /// month, UTC, for a `MONTH` plan; the UTC day for a `DAY` plan; null when + /// there is no plan, no quota, or a period this code does not compute. + period_start: Option, /// Last day of the current period, inclusive, `YYYY-MM-DD`. - period_end: String, - /// When the quota resets under our stated rule: the 1st of the next month, - /// 00:00 UTC, as an RFC 3339 instant. - resets_at: String, - /// When the `GetUsage` behind this answer was made, RFC 3339. The "last + period_end: Option, + /// When the quota resets under our stated rule, RFC 3339: the 1st of the + /// next month (`MONTH`) or the next day (`DAY`), 00:00 UTC. + resets_at: Option, + /// When the lookup behind this answer was made, RFC 3339. The "last /// updated" line renders this — for a cached or stale-served answer it is /// the fetch time, not now, which is the point. as_of: String, + /// The key's plan on our API stage (task 0311); `null` when it is on none. + plan: Option, +} + +/// The plan as the dashboard needs it (task 0311): the pill (`tier`, and +/// `name` for a Custom plan), the Rate Limit card's figures and the Monthly +/// Usage card's quota. Every figure is optional because AWS makes it so, and +/// an absent one is stated ("Unlimited"), never rendered as zero. +#[derive(Clone, Serialize)] +struct PlanWire { + tier: Tier, + name: String, + rate_limit_per_second: Option, + burst_limit: Option, + quota_limit: Option, + /// `MONTH`, `DAY`, `WEEK` — or whatever else AWS answers, verbatim. + quota_period: Option, +} + +impl From<&PlanInfo> for PlanWire { + fn from(plan: &PlanInfo) -> Self { + Self { + tier: plan.tier, + name: plan.name.clone(), + rate_limit_per_second: plan.rate_limit, + burst_limit: plan.burst_limit, + quota_limit: plan.quota_limit, + quota_period: plan.quota_period.clone(), + } + } +} + +/// The current period under a plan's `quota.period` (task 0311). +/// +/// `MONTH` is today's calendar-month rule, **whatever the quota's offset** — +/// the offset is a request count, not a start day. `DAY` is the UTC day, which +/// is trivial. Anything else (`WEEK`: which weekday? AWS does not say) is +/// [`PeriodRule::Unsupported`] and is reported by name, not guessed. +#[derive(Debug, Clone, PartialEq, Eq)] +enum PeriodRule { + Month(Period), + Day(NaiveDate), + Unsupported, +} + +/// Which rule an answer was built under — what the cache re-checks it +/// against (see [`answers_for_period`]). `None` is a period-independent +/// answer: no plan, no quota, or an unsupported period. +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +enum PeriodKind { + Month, + Day, + None, +} + +fn ymd(date: NaiveDate) -> String { + date.format("%Y-%m-%d").to_string() +} + +impl PeriodRule { + fn for_period(quota_period: Option<&str>, today: NaiveDate) -> Self { + match quota_period { + Some("MONTH") => PeriodRule::Month(Period::containing(today)), + Some("DAY") => PeriodRule::Day(today), + _ => PeriodRule::Unsupported, + } + } + + fn kind(&self) -> PeriodKind { + match self { + PeriodRule::Month(_) => PeriodKind::Month, + PeriodRule::Day(_) => PeriodKind::Day, + PeriodRule::Unsupported => PeriodKind::None, + } + } + + /// First day of the period, `YYYY-MM-DD` — also the query's `startDate`. + fn period_start(&self) -> Option { + match self { + PeriodRule::Month(period) => Some(period.start_ymd()), + PeriodRule::Day(day) => Some(ymd(*day)), + PeriodRule::Unsupported => None, + } + } + + /// Last day of the period, inclusive. + fn period_end(&self) -> Option { + match self { + PeriodRule::Month(period) => Some(period.end_ymd()), + PeriodRule::Day(day) => Some(ymd(*day)), + PeriodRule::Unsupported => None, + } + } + + /// The instant the period ends, RFC 3339. + fn resets_at(&self) -> Option { + match self { + PeriodRule::Month(period) => Some(period.resets_at()), + PeriodRule::Day(day) => day + .succ_opt() + .map(|next| format!("{}T00:00:00Z", ymd(next))), + PeriodRule::Unsupported => None, + } + } } /// What the cache remembers for one caller. @@ -331,9 +461,21 @@ struct UsageResponse { /// "No key" is cached alongside real answers, deliberately: the lookup for a /// keyless caller costs the same `GetApiKeys` as anyone else's, and a keyless /// caller pressing refresh is the same loop as anyone else pressing refresh. +/// +/// A usage answer carries the period rule it was built under (task 0311), so +/// the cache can tell when it stops describing "now" — at the month roll for a +/// `MONTH` plan, at midnight UTC for a `DAY` plan, never for a +/// period-independent answer. The plan lives in the same entry, so a plan an +/// operator changed shows within one TTL. +/// +/// The body is boxed: it is ~240 bytes beside a unit `NoKey`, and every cache +/// entry would otherwise pay for the larger variant. #[derive(Clone)] enum CachedAnswer { - Usage(UsageResponse), + Usage { + body: Box, + rule: PeriodKind, + }, NoKey, } @@ -370,17 +512,17 @@ async fn usage(State(state): State, headers: HeaderMap) -> Response )); }; - // Computed once per request and used to validate cache entries as well as - // to build the query: an entry answering for a different period_start is - // last month's answer wearing this month's label, and the minute after a - // month boundary is exactly when a viewer checks whether the reset - // happened. - let period_start = Period::now().start_ymd(); + // Computed once per request and used to validate cache entries: an entry + // answering for a different period_start is last period's answer wearing + // this period's label, and the minute after a boundary is exactly when a + // viewer checks whether the reset happened. Each entry is checked against + // its OWN rule (month or day) — see `answers_for_period`. + let today = Utc::now().date_naive(); // Fresh cache hit: no control-plane call of any kind. This is the // "repeated dashboard loads do not produce one GetUsage call each" // acceptance criterion, in one branch. - if let Some(entry) = cached(&state, &session.sub, state.ttl, &period_start) { + if let Some(entry) = cached(&state, &session.sub, state.ttl, today) { return answer(entry.answer); } @@ -399,7 +541,7 @@ async fn usage(State(state): State, headers: HeaderMap) -> Response // timing — so it gets the same answer as the throttle arm below: // the last good answer (re-stamped, so the next TTL of loads // leaves the struggling control plane alone) beats the error page. - if let Some(entry) = cached(&state, &session.sub, STALE_KEEP, &period_start) { + if let Some(entry) = cached(&state, &session.sub, STALE_KEEP, today) { tracing::warn!( deadline_secs = state.deadline.as_secs_f32(), "portal usage lookup ran out of time; serving the cached answer" @@ -430,7 +572,7 @@ async fn usage(State(state): State, headers: HeaderMap) -> Response // happening and invite a retry; the entry the next success writes ends // the condition. Err(GatewayError::Throttled { operation }) => { - if let Some(entry) = cached(&state, &session.sub, STALE_KEEP, &period_start) { + if let Some(entry) = cached(&state, &session.sub, STALE_KEEP, today) { tracing::warn!( operation, "control plane is throttling; serving the cached usage answer" @@ -474,7 +616,26 @@ async fn usage(State(state): State, headers: HeaderMap) -> Response } } -/// Look the key up (read-only) and read its usage. +/// An answer with every counter and period field null — the no-plan, +/// no-quota and unsupported-period states (task 0311). `limit` carries the +/// plan's quota when there is one. +fn without_counters(plan: Option<&PlanInfo>) -> CachedAnswer { + CachedAnswer::Usage { + body: Box::new(UsageResponse { + used: None, + remaining: None, + limit: plan.and_then(|p| p.quota_limit), + period_start: None, + period_end: None, + resets_at: None, + as_of: Utc::now().to_rfc3339_opts(SecondsFormat::Secs, true), + plan: plan.map(PlanWire::from), + }), + rule: PeriodKind::None, + } +} + +/// Look the key up (read-only), find its plan, and read its usage on it. async fn fetch(gateway: &Gateway, name: &str) -> Result { // The same list → exact filter → rank as the reveal, so the usage shown is // the usage of the key the reveal hands out — and nothing more: no create, @@ -497,33 +658,69 @@ async fn fetch(gateway: &Gateway, name: &str) -> Result Option { let cache = state .cache @@ -542,7 +739,7 @@ fn cached( .entries .get(sub) .filter(|entry| entry.fetched_at.elapsed() < max_age) - .filter(|entry| answers_for_period(&entry.answer, current_period_start)) + .filter(|entry| answers_for_period(&entry.answer, today)) .cloned() } @@ -563,17 +760,27 @@ fn epoch_of(state: &UsageState, sub: &str) -> Option { cache.epochs.get(sub).map(|mark| mark.value) } -/// Whether a cached answer still describes the current period. +/// Whether a cached answer still describes the current period — under the +/// answer's OWN rule (task 0311), not the calendar month for everyone. /// /// An entry cached before midnight on the last of the month and served after /// it would render last month's `period_start`/`period_end` and a `resets_at` /// already in the past, labelled "this period" — a minute a month under the /// TTL, up to [`STALE_KEEP`] under the throttle fallback, and precisely when a -/// viewer looks to see whether the reset happened. "No key" carries no period -/// and stays valid across the boundary. -fn answers_for_period(answer: &CachedAnswer, current_period_start: &str) -> bool { +/// viewer looks to see whether the reset happened. A `DAY` answer has the same +/// hazard at every midnight UTC, and is served from the cache (and the stale +/// fallback) only within its own day. "No key", and an answer with no period +/// (no plan, no quota, an unsupported period), stay valid across any boundary. +fn answers_for_period(answer: &CachedAnswer, today: NaiveDate) -> bool { match answer { - CachedAnswer::Usage(body) => body.period_start == current_period_start, + CachedAnswer::Usage { body, rule } => { + let current_start = match rule { + PeriodKind::Month => Period::containing(today).start_ymd(), + PeriodKind::Day => ymd(today), + PeriodKind::None => return true, + }; + body.period_start.as_deref() == Some(current_start.as_str()) + } CachedAnswer::NoKey => true, } } @@ -635,7 +842,7 @@ fn remember(state: &UsageState, sub: &str, answer: CachedAnswer, epoch: Option Response { match cached { - CachedAnswer::Usage(body) => no_store(Json(body).into_response()), + CachedAnswer::Usage { body, .. } => no_store(Json(body).into_response()), // A real portal `404` with the JSON envelope — deliberately // distinguishable from the gate's empty one: the portal is open, the // caller is signed in, and the honest answer is "you have no key", @@ -695,16 +902,28 @@ mod tests { assert!(!rest.contains('/')); } + fn date(y: i32, m: u32, d: u32) -> NaiveDate { + NaiveDate::from_ymd_opt(y, m, d).unwrap() + } + + fn answer_with(period_start: Option<&str>, rule: PeriodKind) -> CachedAnswer { + CachedAnswer::Usage { + body: Box::new(UsageResponse { + used: Some(1), + remaining: Some(2), + limit: Some(3), + period_start: period_start.map(str::to_string), + period_end: Some("2026-08-31".to_string()), + resets_at: Some("2026-09-01T00:00:00Z".to_string()), + as_of: "2026-08-19T10:00:00Z".to_string(), + plan: None, + }), + rule, + } + } + fn usage_answer(period_start: &str) -> CachedAnswer { - CachedAnswer::Usage(UsageResponse { - used: Some(1), - remaining: Some(2), - limit: Some(3), - period_start: period_start.to_string(), - period_end: "2026-08-31".to_string(), - resets_at: "2026-09-01T00:00:00Z".to_string(), - as_of: "2026-08-19T10:00:00Z".to_string(), - }) + answer_with(Some(period_start), PeriodKind::Month) } /// A cached answer survives the cache only inside its own period: last @@ -714,21 +933,93 @@ mod tests { fn a_cached_answer_dies_at_the_month_boundary() { assert!(answers_for_period( &usage_answer("2026-08-01"), - "2026-08-01" + date(2026, 8, 31) )); assert!(!answers_for_period( &usage_answer("2026-08-01"), - "2026-09-01" + date(2026, 9, 1) )); } + /// A DAY answer lives within its own UTC day — cache hits and the stale + /// fallback included — and dies at midnight (task 0311). + #[test] + fn a_day_answer_dies_at_midnight_utc() { + let answer = answer_with(Some("2026-09-24"), PeriodKind::Day); + assert!(answers_for_period(&answer, date(2026, 9, 24))); + assert!(!answers_for_period(&answer, date(2026, 9, 25))); + } + + /// No plan, no quota or an unsupported period: nothing in the answer is + /// tied to a period, so only the TTL retires it (task 0311). + #[test] + fn a_periodless_answer_is_period_independent() { + let answer = answer_with(None, PeriodKind::None); + assert!(answers_for_period(&answer, date(2026, 9, 24))); + assert!(answers_for_period(&answer, date(2027, 1, 1))); + } + /// "No key" carries no period and stays valid across the boundary — a /// month rolling over does not conjure a key into existence. #[test] fn no_key_is_period_independent() { - assert!(answers_for_period(&CachedAnswer::NoKey, "2026-09-01")); + assert!(answers_for_period(&CachedAnswer::NoKey, date(2026, 9, 1))); + } + + /// MONTH is the calendar month, UTC — the pre-0311 rule byte for byte. + #[test] + fn a_month_plan_is_the_calendar_month() { + let rule = PeriodRule::for_period(Some("MONTH"), date(2026, 9, 24)); + assert_eq!(rule.kind(), PeriodKind::Month); + assert_eq!(rule.period_start().as_deref(), Some("2026-09-01")); + assert_eq!(rule.period_end().as_deref(), Some("2026-09-30")); + assert_eq!(rule.resets_at().as_deref(), Some("2026-10-01T00:00:00Z")); } + /// The offset is a request count, not a start day: a plan with offset 7 + /// gets exactly the period a plan with offset 0 does. The rule is not + /// even given the offset, which is the point — this test pins that the + /// period depends on `quota.period` alone. + #[test] + fn a_month_offset_never_shifts_the_period() { + let today = date(2026, 12, 31); + let offset_zero = PeriodRule::for_period(Some("MONTH"), today); + let offset_seven = PeriodRule::for_period(Some("MONTH"), today); + assert_eq!(offset_zero, offset_seven); + assert_eq!(offset_seven.period_start().as_deref(), Some("2026-12-01")); + assert_eq!( + offset_seven.resets_at().as_deref(), + Some("2027-01-01T00:00:00Z") + ); + } + + /// DAY is the UTC day, resetting at the next midnight UTC. + #[test] + fn a_day_plan_is_the_utc_day() { + let rule = PeriodRule::for_period(Some("DAY"), date(2026, 2, 28)); + assert_eq!(rule.kind(), PeriodKind::Day); + assert_eq!(rule.period_start().as_deref(), Some("2026-02-28")); + assert_eq!(rule.period_end().as_deref(), Some("2026-02-28")); + assert_eq!(rule.resets_at().as_deref(), Some("2026-03-01T00:00:00Z")); + } + + /// WEEK, an unknown period and no period at all are reported, not + /// guessed: every field is absent. + #[test] + fn week_and_unknown_periods_are_unsupported() { + for period in [Some("WEEK"), Some("FORTNIGHT"), None] { + let rule = PeriodRule::for_period(period, date(2026, 9, 24)); + assert_eq!(rule, PeriodRule::Unsupported, "{period:?}"); + assert_eq!(rule.kind(), PeriodKind::None); + assert_eq!(rule.period_start(), None); + assert_eq!(rule.period_end(), None); + assert_eq!(rule.resets_at(), None); + } + } + + fn any_day() -> NaiveDate { + date(2026, 8, 19) + } const ANY_PERIOD: &str = "2026-08-01"; /// The write-after-eviction race, replayed step by step: a "no key" @@ -748,13 +1039,13 @@ mod tests { // The lookup finishes with its stale keyless snapshot. remember(&state, sub, CachedAnswer::NoKey, epoch_at_lookup_start); assert!( - cached(&state, sub, CACHE_TTL, ANY_PERIOD).is_none(), + cached(&state, sub, CACHE_TTL, any_day()).is_none(), "a pre-eviction 'no key' must not be cached" ); // Whereas a lookup that STARTED after the eviction stores normally. remember(&state, sub, CachedAnswer::NoKey, epoch_of(&state, sub)); - assert!(cached(&state, sub, CACHE_TTL, ANY_PERIOD).is_some()); + assert!(cached(&state, sub, CACHE_TTL, any_day()).is_some()); } /// What `STALE_KEEP` pruning does to one caller's mark, without waiting @@ -824,7 +1115,7 @@ mod tests { remember(&state, sub, CachedAnswer::NoKey, epoch_at_lookup_start); assert!( - cached(&state, sub, CACHE_TTL, ANY_PERIOD).is_some(), + cached(&state, sub, CACHE_TTL, any_day()).is_some(), "no mark means no eviction happened, so this 'no key' is good" ); } @@ -872,11 +1163,11 @@ mod tests { let stale_epoch = epoch_of(&state, sub); state.cache_handle().invalidate_no_key(sub); remember(&state, sub, usage_answer(ANY_PERIOD), stale_epoch); - assert!(cached(&state, sub, CACHE_TTL, ANY_PERIOD).is_some()); + assert!(cached(&state, sub, CACHE_TTL, any_day()).is_some()); // And invalidating again does not evict it — only "no key" is the // handle's to remove. state.cache_handle().invalidate_no_key(sub); - assert!(cached(&state, sub, CACHE_TTL, ANY_PERIOD).is_some()); + assert!(cached(&state, sub, CACHE_TTL, any_day()).is_some()); } } diff --git a/packages/prices-api/tests/portal_auth.rs b/packages/prices-api/tests/portal_auth.rs index 7d627683..0fc5ed2c 100644 --- a/packages/prices-api/tests/portal_auth.rs +++ b/packages/prices-api/tests/portal_auth.rs @@ -38,8 +38,7 @@ use mock_discord::{GRANTED_SCOPE, MemberReply, MockDiscord, USER_ID}; // `USER_ID` of its own, and this file already has both. #[path = "portal_keys/harness.rs"] mod harness; -use harness::{MockGateway, PLAN_ID}; -use prices_api::portal::keys::gateway::Gateway; +use harness::{MockGateway, test_gateway}; // --------------------------------------------------------------------------- // Router under test @@ -624,7 +623,7 @@ fn app_with_keys_and( api_base: discord.base.clone(), ..Endpoints::default() }, - portal_keys: Some(Gateway::against(&gateway.base, PLAN_ID.to_string())), + portal_keys: Some(test_gateway(&gateway.base)), portal_eligibility: Some(eligibility), portal_rate_limit: None, portal_web_origin: None, diff --git a/packages/prices-api/tests/portal_issue.rs b/packages/prices-api/tests/portal_issue.rs index 3905ddce..48ce7093 100644 --- a/packages/prices-api/tests/portal_issue.rs +++ b/packages/prices-api/tests/portal_issue.rs @@ -32,7 +32,6 @@ use harness::*; use mock_discord::{GRANTED_SCOPE, MemberReply, MockDiscord}; use prices_api::portal::auth::discord::Endpoints; use prices_api::portal::auth::{cookies, session::Session, state_token}; -use prices_api::portal::keys::gateway::Gateway; use prices_api::portal::usage::USAGE_PATH; /// Milliseconds since the Discord epoch, shifted into snowflake position — @@ -58,7 +57,7 @@ fn issue_app_with( ) -> Router { build_app_with( true, - Some(Gateway::against(&gateway.base, PLAN_ID.to_string())), + Some(test_gateway(&gateway.base)), Endpoints { api_base: discord.base.clone(), ..Endpoints::default() @@ -86,7 +85,7 @@ async fn everything_including_issue_is_an_empty_404_while_the_portal_is_closed() let gateway = MockGateway::start().await; let closed = build_app_with( false, - Some(Gateway::against(&gateway.base, PLAN_ID.to_string())), + Some(test_gateway(&gateway.base)), Endpoints { api_base: discord.base.clone(), ..Endpoints::default() @@ -985,3 +984,55 @@ async fn a_duplicate_that_will_not_delete_does_not_withhold_the_key() { "the winner is still the earliest, and it is what the reveal hands out" ); } + +// --------------------------------------------------------------------------- +// Attaching to a plan other than free (task 0311) +// --------------------------------------------------------------------------- + +/// A plan the control plane does not know is `PlanNotFound` NAMING that plan +/// — a paid or custom plan id comes from the previous key's `GetUsagePlans` +/// answer, not from the SSM parameter, so a message that only pointed at +/// `PORTAL_FREE_PLAN_PARAM` would send the operator to the wrong place. +#[tokio::test] +async fn attaching_to_an_unknown_plan_names_that_plan() { + use prices_api::portal::keys::gateway::GatewayError; + + let gateway = MockGateway::start().await; + let key = gateway.with(|s| s.seed(&key_name(), 100)); + + let error = test_gateway(&gateway.base) + .attach_to_plan(&key, "vanishedplan9") + .await + .expect_err("the mock knows no plan `vanishedplan9`"); + assert!( + matches!(&error, GatewayError::PlanNotFound { plan_id } if plan_id == "vanishedplan9"), + "{error:?}" + ); + let message = error.to_string(); + assert!(message.contains("vanishedplan9"), "{message}"); + assert!(message.contains("GetUsagePlans"), "{message}"); + assert!(message.contains("PORTAL_FREE_PLAN_PARAM"), "{message}"); + assert!(gateway.with(|s| s.plan_keys.is_empty())); +} + +/// And a known paid plan is attached to like the free one. +#[tokio::test] +async fn attaching_to_a_paid_plan_puts_the_key_on_it() { + use prices_api::portal::keys::gateway::Attachment; + + let gateway = MockGateway::start().await; + let key = gateway.with(|s| { + s.plans.push(StoredPlan::basic()); + s.seed(&key_name(), 100) + }); + + let attached = test_gateway(&gateway.base) + .attach_to_plan(&key, BASIC_PLAN_ID) + .await + .expect("basic1 exists"); + assert_eq!(attached, Attachment::OnPlan); + assert_eq!( + gateway.with(|s| s.plan_keys.clone()), + vec![(BASIC_PLAN_ID.to_string(), key)] + ); +} diff --git a/packages/prices-api/tests/portal_keys/harness.rs b/packages/prices-api/tests/portal_keys/harness.rs index e0934c7c..73d14d81 100644 --- a/packages/prices-api/tests/portal_keys/harness.rs +++ b/packages/prices-api/tests/portal_keys/harness.rs @@ -70,11 +70,114 @@ pub struct StoredKey { pub last_updated_at: Option, } +/// One usage plan the mock knows (task 0311) — what `GetUsagePlans` lists +/// and what `CreateUsagePlanKey` / `GetUsage` accept in their `{plan}` path. +#[derive(Clone, Debug)] +pub struct StoredPlan { + pub id: String, + pub name: String, + /// `(apiId, stage)` pairs — the plan's `apiStages`. + pub api_stages: Vec<(String, String)>, + /// `(rateLimit, burstLimit)`, or no throttle at all. + pub throttle: Option<(f64, i32)>, + /// `(limit, offset, period)`, or no quota at all. + pub quota: Option<(i32, i32, &'static str)>, +} + +impl StoredPlan { + /// A plan on OUR stage (`API_ID`, `STAGE`) with the given figures. + pub fn on_our_stage( + id: &str, + name: &str, + throttle: Option<(f64, i32)>, + quota: Option<(i32, i32, &'static str)>, + ) -> Self { + Self { + id: id.to_string(), + name: name.to_string(), + api_stages: vec![(API_ID.to_string(), STAGE.to_string())], + throttle, + quota, + } + } + + /// The free plan every mock starts with: `pricing-api-free-production`, + /// 1 req/s, burst 5, 100 000 a MONTH — so every pre-0311 test keeps the + /// free-plan semantics it was written against. + pub fn free() -> Self { + Self::on_our_stage( + PLAN_ID, + "pricing-api-free-production", + Some((1.0, 5)), + Some((100_000, 0, "MONTH")), + ) + } + + /// `pricing-api-basic-production`, 3 req/s, burst 15, 1 000 000 a MONTH. + pub fn basic() -> Self { + Self::on_our_stage( + BASIC_PLAN_ID, + "pricing-api-basic-production", + Some((3.0, 15)), + Some((1_000_000, 0, "MONTH")), + ) + } + + /// A hand-made plan on our stage with no throttle and no quota — + /// the unlimited Custom state. + pub fn custom_unlimited() -> Self { + Self::on_our_stage(CUSTOM_PLAN_ID, "prices-production-acme-plan", None, None) + } + + fn json(&self) -> Value { + let mut plan = json!({ + "id": self.id, + "name": self.name, + "apiStages": self + .api_stages + .iter() + .map(|(api_id, stage)| json!({ "apiId": api_id, "stage": stage })) + .collect::>(), + }); + if let Some((rate, burst)) = self.throttle { + plan["throttle"] = json!({ "rateLimit": rate, "burstLimit": burst }); + } + if let Some((limit, offset, period)) = self.quota { + plan["quota"] = json!({ "limit": limit, "offset": offset, "period": period }); + } + plan + } +} + #[derive(Default)] pub struct Store { pub keys: Vec, /// `(usage_plan_id, key_id)` pairs, in the order they were attached. pub plan_keys: Vec<(String, String)>, + /// Every usage plan the mock knows (task 0311). Seeded with + /// [`StoredPlan::free`] by [`MockGateway::start`]; a test adds paid, + /// custom, other-API or WEEK plans with `s.plans.push(..)`. + pub plans: Vec, + /// How many plans one `GetUsagePlans` page holds. + pub plans_page_size: usize, + /// How many `GetUsagePlans` HTTP calls arrived. + pub plans_calls: usize, + /// Every `keyId` `GetUsagePlans` was asked about. + pub plans_queries: Vec>, + /// Answer every `GetUsagePlans` with `429 TooManyRequestsException`. + /// Sticky, for the reason `throttle_usage` is: the SDK's own backoff + /// retries a 429, so only a throttle that persists reaches the handler. + pub throttle_plans: bool, + /// Every `{plan}` segment `GetUsage` was asked for, in order (task 0311) — + /// which plan's counter was read. Separate from `usage_queries`, whose + /// 3-tuple shape the pre-0311 tests destructure. + pub usage_plan_queries: Vec, + /// Answer the next `CreateUsagePlanKey` with `400 BadRequestException` + /// and change nothing, then clear (task 0311) — an attach that did not + /// happen after a create that did: the crash-between-create-and-attach + /// window. A 400 rather than a 500 because the SDK retries a 500, so a + /// one-shot 500 is not observable from a handler at all. + pub fail_next_attach: bool, /// How many keys one `GetApiKeys` page holds. Small numbers force the /// pagination path the reconciler must walk to exhaustion. pub page_size: usize, @@ -231,6 +334,20 @@ impl Store { id } + /// [`Self::seed`], attached to `plan` — what an issued key looks like + /// (task 0311): since the usage route reads the key's own plan, a key on + /// no plan is the "issued but dead" state, not the ordinary one. + pub fn seed_on_plan(&mut self, name: &str, created_at: u64, plan: &str) -> String { + let id = self.seed(name, created_at); + self.plan_keys.push((plan.to_string(), id.clone())); + id + } + + /// [`Self::seed_on_plan`] on the free plan. + pub fn seed_on_free_plan(&mut self, name: &str, created_at: u64) -> String { + self.seed_on_plan(name, created_at, PLAN_ID) + } + /// A key the owner revoked at `revoked_at` (task 0191): disabled, with /// `lastUpdatedDate` set to the revocation instant. pub fn seed_revoked(&mut self, name: &str, created_at: u64, revoked_at: u64) -> String { @@ -279,6 +396,8 @@ impl MockGateway { pub async fn start() -> Self { let store = Arc::new(Mutex::new(Store { page_size: 100, + plans: vec![StoredPlan::free()], + plans_page_size: 100, ..Store::default() })); @@ -288,6 +407,7 @@ impl MockGateway { "/apikeys/{id}", get(read_key).delete(delete_key).patch(update_key), ) + .route("/usageplans", get(list_plans)) .route("/usageplans/{plan}/keys", post(attach_key)) .route("/usageplans/{plan}/usage", get(read_usage)) .with_state(store.clone()); @@ -512,6 +632,76 @@ pub async fn update_key( Json(api_key_json(&snapshot)).into_response() } +#[derive(serde::Deserialize)] +pub struct PlansQuery { + #[serde(rename = "keyId")] + pub key_id: Option, + pub position: Option, + #[allow(dead_code)] + pub limit: Option, +} + +/// `GET /usageplans` — `GetUsagePlans`, task 0311's one new call. +/// +/// Answers under **`item`**, not `items`: the SDK's deserializer matches +/// `"item"` (`shape_get_usage_plans.rs`), and the CLI's `items` is a rename. +/// A mock that answered `items` would hand every test an empty page and the +/// no-plan branch would cover everything, vacuously — the same trap +/// [`read_usage`] documents for `values`. +/// +/// With `keyId`, only the plans that key is attached to (per `plan_keys`), +/// exactly as the service filters; paged by `plans_page_size`. +pub async fn list_plans( + State(store): State>>, + Query(query): Query, +) -> Response { + let mut store = store.lock().unwrap(); + store.plans_calls += 1; + store.plans_queries.push(query.key_id.clone()); + if store.throttle_plans { + return throttled(); + } + + let matched: Vec = store + .plans + .iter() + .filter(|plan| match query.key_id.as_deref() { + Some(key_id) => store + .plan_keys + .iter() + .any(|(p, k)| p == &plan.id && k == key_id), + None => true, + }) + .cloned() + .collect(); + let start: usize = query + .position + .as_deref() + .and_then(|p| p.parse().ok()) + .unwrap_or(0); + let start = start.min(matched.len()); + let end = (start + store.plans_page_size.max(1)).min(matched.len()); + + let mut body = json!({ + "item": matched[start..end].iter().map(StoredPlan::json).collect::>(), + }); + if end < matched.len() { + body["position"] = json!(end.to_string()); + } + Json(body).into_response() +} + +/// The `400` shape the SDK maps to `BadRequestException` — not retried by +/// the SDK, so one of them is observable from a handler. +pub fn bad_request() -> Response { + ( + StatusCode::BAD_REQUEST, + [("x-amzn-errortype", "BadRequestException")], + Json(json!({ "message": "Bad Request" })), + ) + .into_response() +} + pub async fn attach_key( State(store): State>>, Path(plan): Path, @@ -520,6 +710,9 @@ pub async fn attach_key( let mut store = store.lock().unwrap(); store.attach_calls += 1; let key_id = body["keyId"].as_str().unwrap_or_default().to_string(); + if std::mem::take(&mut store.fail_next_attach) { + return bad_request(); + } if store.attach_always_404 { return not_found(); } @@ -530,15 +723,27 @@ pub async fn attach_key( if !store.keys.iter().any(|k| k.id == key_id) { return not_found(); } - // Already on this plan → `409 ConflictException`, exactly as the service - // answers. The reconciler attaches every key it is about to hand out, so - // this is the ordinary case rather than an edge one, and a mock that - // silently accepted a re-attach would let a handler that treats the + // An unknown plan is the same `404` as an unknown key (task 0311) — + // which is what makes `PlanNotFound` for a paid plan testable. + let Some(target) = store.plans.iter().find(|p| p.id == plan).cloned() else { + return not_found(); + }; + // Already on this plan — or on any plan sharing one of its stages, since + // a key belongs to one plan per stage — → `409 ConflictException`, as the + // service answers. The reconciler attaches every key it is about to hand + // out, so this is the ordinary case rather than an edge one, and a mock + // that silently accepted a re-attach would let a handler that treats the // conflict as a failure pass every test here. + let shares_a_stage = |other: &str| { + other == plan + || store.plans.iter().any(|p| { + p.id == other && p.api_stages.iter().any(|s| target.api_stages.contains(s)) + }) + }; if store .plan_keys .iter() - .any(|(p, k)| p == &plan && k == &key_id) + .any(|(p, k)| k == &key_id && shares_a_stage(p)) { return conflict(); } @@ -573,11 +778,12 @@ pub struct UsageQuery { /// no-rows path would cover everything, vacuously. pub async fn read_usage( State(store): State>>, - Path(_plan): Path, + Path(plan): Path, Query(query): Query, ) -> Response { let mut store = store.lock().unwrap(); store.usage_calls += 1; + store.usage_plan_queries.push(plan.clone()); let key_id = query.key_id.clone().unwrap_or_default(); store.usage_queries.push(( key_id.clone(), @@ -607,7 +813,7 @@ pub async fn read_usage( let end = (start + page).min(days.len()); let mut body = json!({ - "usagePlanId": "freeplan1", + "usagePlanId": plan, "startDate": query.start_date, "endDate": query.end_date, }); @@ -657,7 +863,7 @@ pub fn not_found() -> Response { /// The `409` shape the SDK maps to `ConflictException`. /// /// Sibling of [`not_found`] and load-bearing for the same reason: the header is -/// what restJson1 matches on, and `Gateway::attach_to_free_plan` reads that +/// what restJson1 matches on, and `Gateway::attach_to_plan` reads that /// mapping to decide that a key already on the plan is the desired state rather /// than a failure. pub fn conflict() -> Response { @@ -687,6 +893,24 @@ pub fn api_key_json(key: &StoredKey) -> Value { pub const SIGNING_KEY: &str = "0123456789abcdef0123456789abcdef0123456789abcdef0123456789abcdef"; pub const PLAN_ID: &str = "freeplan1"; +/// Our REST API id and stage, as the mock's plans carry them (task 0311). +pub const API_ID: &str = "02mabge71l"; +pub const STAGE: &str = "production"; +/// The mock's `pricing-api-basic-production` plan id ([`StoredPlan::basic`]). +pub const BASIC_PLAN_ID: &str = "basic1"; +/// The mock's hand-made unlimited plan id ([`StoredPlan::custom_unlimited`]). +pub const CUSTOM_PLAN_ID: &str = "custom1"; + +/// A `Gateway` pointed at the mock at `base`, scoped to the free plan and our +/// API stage — the one constructor every test uses. +pub fn test_gateway(base: &str) -> Gateway { + Gateway::against( + base, + PLAN_ID.to_string(), + API_ID.to_string(), + STAGE.to_string(), + ) +} pub const USER_ID: &str = "308994132968210433"; pub const USER_NAME: &str = "adam"; @@ -763,10 +987,7 @@ pub const GUILD_ID: &str = "897514728459468821"; /// A router with the portal open, sign-in configured, and the control plane /// pointed at `mock`. pub fn app_against(mock: &MockGateway) -> Router { - build_app( - true, - Some(Gateway::against(&mock.base, PLAN_ID.to_string())), - ) + build_app(true, Some(test_gateway(&mock.base))) } /// The key routes alone, with a shortened reconciliation deadline. @@ -780,7 +1001,7 @@ pub fn keys_router_with_deadline(mock: &MockGateway, deadline: std::time::Durati prices_api::portal::keys::routes( prices_api::portal::keys::KeysState::new( Some(oauth_secret()), - Some(Gateway::against(&mock.base, PLAN_ID.to_string())), + Some(test_gateway(&mock.base)), ) .with_deadline(deadline), ) @@ -873,7 +1094,7 @@ pub fn usage_router_with( prices_api::portal::usage::routes( prices_api::portal::usage::UsageState::new( Some(oauth_secret()), - Some(Gateway::against(&mock.base, PLAN_ID.to_string())), + Some(test_gateway(&mock.base)), ) .with_ttl(ttl) .with_deadline(deadline), diff --git a/packages/prices-api/tests/portal_keys_logs.rs b/packages/prices-api/tests/portal_keys_logs.rs index e0aadc4d..9f14a131 100644 --- a/packages/prices-api/tests/portal_keys_logs.rs +++ b/packages/prices-api/tests/portal_keys_logs.rs @@ -107,10 +107,7 @@ async fn no_key_value_ever_reaches_the_logs() { let discord = MockDiscord::start(GRANTED_SCOPE, None).await; let app = build_app_with( true, - Some(prices_api::portal::keys::gateway::Gateway::against( - &mock.base, - PLAN_ID.to_string(), - )), + Some(test_gateway(&mock.base)), prices_api::portal::auth::discord::Endpoints { api_base: discord.base.clone(), ..Default::default() diff --git a/packages/prices-api/tests/portal_rework.rs b/packages/prices-api/tests/portal_rework.rs index 9f636cea..91c007d5 100644 --- a/packages/prices-api/tests/portal_rework.rs +++ b/packages/prices-api/tests/portal_rework.rs @@ -33,14 +33,13 @@ use chrono::{Datelike, NaiveDate, Utc}; use harness::*; use mock_discord::{GRANTED_SCOPE, MemberReply, MockDiscord}; use prices_api::portal::auth::discord::Endpoints; -use prices_api::portal::keys::gateway::Gateway; use prices_api::portal::keys::{KEY_PATH, PORTAL_REQUEST_HEADER, REWORK_PATH}; use prices_api::portal::usage::USAGE_PATH; fn app_with_discord(discord: &MockDiscord, gateway: &MockGateway) -> Router { build_app_with( true, - Some(Gateway::against(&gateway.base, PLAN_ID.to_string())), + Some(test_gateway(&gateway.base)), Endpoints { api_base: discord.base.clone(), ..Endpoints::default() @@ -129,10 +128,7 @@ async fn reveal_via(router: &Router) -> Reply { async fn revoke_is_an_empty_404_while_the_portal_is_closed() { let gateway = MockGateway::start().await; seed_attached(&gateway, 1_000); - let closed = build_app( - false, - Some(Gateway::against(&gateway.base, PLAN_ID.to_string())), - ); + let closed = build_app(false, Some(test_gateway(&gateway.base))); let reply = revoke(&closed, Some(&session_cookie(USER_ID))).await; assert_eq!(reply.status, StatusCode::NOT_FOUND); @@ -1029,7 +1025,7 @@ async fn a_same_site_revoke_is_accepted_from_the_configured_origin_only() { seed_attached(&gateway, 1_000); let app = build_app_on( true, - Some(Gateway::against(&gateway.base, PLAN_ID.to_string())), + Some(test_gateway(&gateway.base)), Endpoints { api_base: discord.base.clone(), ..Endpoints::default() diff --git a/packages/prices-api/tests/portal_usage.rs b/packages/prices-api/tests/portal_usage.rs index 5a066156..b776aea5 100644 --- a/packages/prices-api/tests/portal_usage.rs +++ b/packages/prices-api/tests/portal_usage.rs @@ -44,7 +44,7 @@ async fn usage(mock: &MockGateway, sub: &str) -> Reply { /// the exact pair of days task 0157's close read off the live plan. fn seed_key_with_usage(mock: &MockGateway) -> String { mock.with(|s| { - let id = s.seed(&format!("discord-{USER_ID}-key"), 100); + let id = s.seed_on_free_plan(&format!("discord-{USER_ID}-key"), 100); s.usage .insert(id.clone(), vec![vec![121, 99_879], vec![0, 99_879]]); id @@ -202,13 +202,15 @@ async fn no_session_is_401_not_signed_in() { } /// A key AWS has no rows for yet — the ordinary state minutes after issuance, -/// because `GetUsage` is not a read-after-write surface. The three counters go -/// absent **together**; the period and `as_of` are ours and stay. +/// because `GetUsage` is not a read-after-write surface. `used` and +/// `remaining` go absent **together**; the period and `as_of` are ours and +/// stay, and since task 0311 so does `limit` — it is the plan's quota, known +/// from `GetUsagePlans` whether or not AWS has counted anything yet. #[tokio::test] async fn a_key_with_no_rows_reports_nothing_recorded_rather_than_zeros() { let mock = MockGateway::start().await; mock.with(|s| { - s.seed(&format!("discord-{USER_ID}-key"), 100); + s.seed_on_free_plan(&format!("discord-{USER_ID}-key"), 100); }); let reply = usage(&mock, USER_ID).await; @@ -216,7 +218,8 @@ async fn a_key_with_no_rows_reports_nothing_recorded_rather_than_zeros() { let body = reply.json(); assert_eq!(body["used"], Value::Null, "{body}"); assert_eq!(body["remaining"], Value::Null, "{body}"); - assert_eq!(body["limit"], Value::Null, "{body}"); + assert_eq!(body["limit"], 100_000, "{body}"); + assert_eq!(body["plan"]["tier"], "free", "{body}"); assert!(body["period_start"].is_string()); assert!(body["as_of"].is_string()); } @@ -228,7 +231,7 @@ async fn a_key_with_no_rows_reports_nothing_recorded_rather_than_zeros() { async fn usage_pages_are_summed_to_exhaustion() { let mock = MockGateway::start().await; mock.with(|s| { - let id = s.seed(&format!("discord-{USER_ID}-key"), 100); + let id = s.seed_on_free_plan(&format!("discord-{USER_ID}-key"), 100); s.usage.insert( id, vec![ @@ -261,14 +264,15 @@ async fn usage_pages_are_summed_to_exhaustion() { /// documents neither the reset instant nor its timezone. /// /// Summing the range blindly would answer `used = 40` beside `remaining = 90` -/// and reconstruct a **130** limit on a 100 plan: a quota no plan has, rendered -/// as fact on the panel whose stated theme is honesty. Counting from the reset -/// is what makes the two figures describe one period. +/// — a pair no period has, rendered as fact on the panel whose stated theme is +/// honesty. Counting from the reset is what makes the two figures describe one +/// period. (`limit` is the plan's quota since task 0311; `used + remaining` +/// disagreeing with it is logged, not rendered.) #[tokio::test] async fn a_quota_reset_inside_the_queried_range_does_not_inflate_the_numbers() { let mock = MockGateway::start().await; mock.with(|s| { - let id = s.seed(&format!("discord-{USER_ID}-key"), 100); + let id = s.seed_on_free_plan(&format!("discord-{USER_ID}-key"), 100); s.usage.insert( id, vec![ @@ -287,7 +291,7 @@ async fn a_quota_reset_inside_the_queried_range_does_not_inflate_the_numbers() { let body = reply.json(); assert_eq!(body["used"], 10, "{body}"); assert_eq!(body["remaining"], 90, "{body}"); - assert_eq!(body["limit"], 100, "{body}"); + assert_eq!(body["limit"], 100_000, "{body}"); } /// The other half of that rule, and the one a naive "restart on any change" @@ -298,7 +302,7 @@ async fn a_quota_reset_inside_the_queried_range_does_not_inflate_the_numbers() { async fn an_idle_day_mid_period_still_counts_the_days_before_it() { let mock = MockGateway::start().await; mock.with(|s| { - let id = s.seed(&format!("discord-{USER_ID}-key"), 100); + let id = s.seed_on_free_plan(&format!("discord-{USER_ID}-key"), 100); s.usage .insert(id, vec![vec![10, 99_990], vec![0, 99_990], vec![5, 99_985]]); }); @@ -321,9 +325,9 @@ async fn an_idle_day_mid_period_still_counts_the_days_before_it() { async fn a_prefix_neighbour_is_not_the_callers_key() { let mock = MockGateway::start().await; let (own, _imposter) = mock.with(|s| { - let own = s.seed(&format!("discord-{USER_ID}-key"), 200); + let own = s.seed_on_free_plan(&format!("discord-{USER_ID}-key"), 200); // An exact-name EXTENSION — what a console-created copy looks like. - let imposter = s.seed(&format!("discord-{USER_ID}-key-old"), 100); + let imposter = s.seed_on_free_plan(&format!("discord-{USER_ID}-key-old"), 100); s.usage.insert(own.clone(), vec![vec![7, 99_993]]); s.usage.insert(imposter.clone(), vec![vec![555, 0]]); (own, imposter) @@ -349,8 +353,8 @@ async fn a_prefix_neighbour_is_not_the_callers_key() { async fn duplicates_resolve_to_the_reveals_winner_without_deleting_the_loser() { let mock = MockGateway::start().await; let (earliest, later) = mock.with(|s| { - let later = s.seed(&format!("discord-{USER_ID}-key"), 500); - let earliest = s.seed(&format!("discord-{USER_ID}-key"), 100); + let later = s.seed_on_free_plan(&format!("discord-{USER_ID}-key"), 500); + let earliest = s.seed_on_free_plan(&format!("discord-{USER_ID}-key"), 100); s.usage.insert(earliest.clone(), vec![vec![11, 99_989]]); s.usage.insert(later.clone(), vec![vec![999, 0]]); (earliest, later) @@ -409,7 +413,7 @@ async fn the_cache_is_per_caller() { seed_key_with_usage(&mock); let other = "999999999999999999"; mock.with(|s| { - let id = s.seed(&format!("discord-{other}-key"), 100); + let id = s.seed_on_free_plan(&format!("discord-{other}-key"), 100); s.usage.insert(id, vec![vec![3, 99_997]]); }); let router = app_against(&mock); @@ -583,7 +587,7 @@ async fn a_reveal_that_finds_a_key_evicts_a_cached_no_key() { // The key appears — 0189's callback created it in another request. mock.with(|s| { - let id = s.seed(&format!("discord-{USER_ID}-key"), 1_000); + let id = s.seed_on_free_plan(&format!("discord-{USER_ID}-key"), 1_000); s.plan_keys.push((PLAN_ID.to_string(), id)); }); @@ -687,7 +691,7 @@ async fn a_stalled_lookup_serves_the_last_good_answer() { async fn a_malformed_daily_row_is_skipped_not_defaulted() { let mock = MockGateway::start().await; mock.with(|s| { - let id = s.seed(&format!("discord-{USER_ID}-key"), 100); + let id = s.seed_on_free_plan(&format!("discord-{USER_ID}-key"), 100); s.usage.insert(id, vec![vec![5, 99_995], vec![121]]); }); @@ -700,19 +704,21 @@ async fn a_malformed_daily_row_is_skipped_not_defaulted() { } /// Every row malformed degrades to "nothing recorded" — the same honest shape -/// as no rows at all — never to invented zeros or an invented limit. +/// as no rows at all — never to invented zeros. The limit is the plan's quota +/// (task 0311), which is known, not invented. #[tokio::test] async fn all_rows_malformed_degrades_to_nothing_recorded() { let mock = MockGateway::start().await; mock.with(|s| { - let id = s.seed(&format!("discord-{USER_ID}-key"), 100); + let id = s.seed_on_free_plan(&format!("discord-{USER_ID}-key"), 100); s.usage.insert(id, vec![vec![]]); }); let reply = usage(&mock, USER_ID).await; assert_eq!(reply.status, StatusCode::OK); assert_eq!(reply.json()["used"], Value::Null); - assert_eq!(reply.json()["limit"], Value::Null); + assert_eq!(reply.json()["remaining"], Value::Null); + assert_eq!(reply.json()["limit"], 100_000); } /// The portal open with no control-plane client wired answers `503` and says @@ -746,3 +752,322 @@ async fn usage_accepts_only_get() { .await; assert_eq!(reply.status, StatusCode::METHOD_NOT_ALLOWED); } + +// --------------------------------------------------------------------------- +// The key's own plan (task 0311) +// --------------------------------------------------------------------------- + +/// A plan on another REST API — the partner plan (`q7sd40`, a DAY quota) that +/// shares the account on production. +fn other_api_plan() -> StoredPlan { + StoredPlan { + id: "q7sd40".to_string(), + name: "production-partner-plan".to_string(), + api_stages: vec![("6l9k06w4pl".to_string(), STAGE.to_string())], + throttle: Some((50.0, 100)), + quota: Some((10_000, 0, "DAY")), + } +} + +/// A free key: every pre-0311 field as before, plus the plan — and the usage +/// was read ON the free plan, with the key's plans looked up by key. +#[tokio::test] +async fn a_free_key_reports_the_free_plan() { + let mock = MockGateway::start().await; + let key_id = seed_key_with_usage(&mock); + + let body = usage(&mock, USER_ID).await.json(); + assert_eq!(body["used"], 121, "{body}"); + assert_eq!(body["remaining"], 99_879, "{body}"); + assert_eq!(body["limit"], 100_000, "{body}"); + assert_eq!(body["plan"]["tier"], "free", "{body}"); + assert_eq!( + body["plan"]["name"], "pricing-api-free-production", + "{body}" + ); + assert_eq!(body["plan"]["rate_limit_per_second"], 1.0, "{body}"); + assert_eq!(body["plan"]["burst_limit"], 5, "{body}"); + assert_eq!(body["plan"]["quota_limit"], 100_000, "{body}"); + assert_eq!(body["plan"]["quota_period"], "MONTH", "{body}"); + mock.with(|s| { + assert_eq!(s.usage_plan_queries, vec![PLAN_ID.to_string()]); + assert_eq!(s.plans_queries, vec![Some(key_id.clone())]); + }); +} + +/// A key an operator moved to Basic: the usage is read on `basic1`, not on +/// the free plan, and the answer states Basic's figures. +#[tokio::test] +async fn a_paid_key_reads_usage_on_its_own_plan() { + let mock = MockGateway::start().await; + mock.with(|s| { + s.plans.push(StoredPlan::basic()); + let id = s.seed_on_plan(&format!("discord-{USER_ID}-key"), 100, BASIC_PLAN_ID); + s.usage.insert(id, vec![vec![40, 999_960]]); + }); + + let body = usage(&mock, USER_ID).await.json(); + assert_eq!(body["plan"]["tier"], "basic", "{body}"); + assert_eq!(body["plan"]["rate_limit_per_second"], 3.0, "{body}"); + assert_eq!(body["plan"]["burst_limit"], 15, "{body}"); + assert_eq!(body["limit"], 1_000_000, "{body}"); + assert_eq!(body["used"], 40, "{body}"); + assert_eq!(body["remaining"], 999_960, "{body}"); + assert!(body["resets_at"].is_string(), "{body}"); + mock.with(|s| assert_eq!(s.usage_plan_queries, vec![BASIC_PLAN_ID.to_string()])); +} + +/// A hand-made plan with neither throttle nor quota: Custom, with its own +/// name, every counter and period field null, and `GetUsage` not called — +/// there is nothing to count against, and zeros would be a lie. +#[tokio::test] +async fn an_unlimited_custom_plan_states_no_counters_and_reads_no_usage() { + let mock = MockGateway::start().await; + mock.with(|s| { + s.plans.push(StoredPlan::custom_unlimited()); + s.seed_on_plan(&format!("discord-{USER_ID}-key"), 100, CUSTOM_PLAN_ID); + }); + + let reply = usage(&mock, USER_ID).await; + assert_eq!(reply.status, StatusCode::OK); + let body = reply.json(); + assert_eq!(body["plan"]["tier"], "custom", "{body}"); + assert_eq!( + body["plan"]["name"], "prices-production-acme-plan", + "{body}" + ); + for field in [ + "rate_limit_per_second", + "burst_limit", + "quota_limit", + "quota_period", + ] { + assert_eq!(body["plan"][field], Value::Null, "plan.{field}: {body}"); + } + for field in [ + "used", + "remaining", + "limit", + "period_start", + "period_end", + "resets_at", + ] { + assert_eq!(body[field], Value::Null, "{field}: {body}"); + } + assert!(body["as_of"].is_string(), "{body}"); + mock.with(|s| { + assert_eq!(s.usage_calls, 0, "an unlimited plan has nothing to count"); + assert!(s.usage_plan_queries.is_empty()); + }); +} + +/// A WEEK quota: reported by name with its limit, the period left null — +/// which weekday AWS rolls on is not documented, so it is not guessed — and +/// no `GetUsage`, because the window it would need is the guess. +#[tokio::test] +async fn a_week_plan_reports_the_period_by_name_not_a_guess() { + let mock = MockGateway::start().await; + mock.with(|s| { + s.plans.push(StoredPlan::on_our_stage( + "weekly1", + "prices-production-weekly-plan", + Some((2.0, 10)), + Some((5_000, 0, "WEEK")), + )); + s.seed_on_plan(&format!("discord-{USER_ID}-key"), 100, "weekly1"); + }); + + let body = usage(&mock, USER_ID).await.json(); + assert_eq!(body["plan"]["tier"], "custom", "{body}"); + assert_eq!(body["plan"]["quota_period"], "WEEK", "{body}"); + assert_eq!(body["plan"]["quota_limit"], 5_000, "{body}"); + assert_eq!(body["limit"], 5_000, "{body}"); + for field in [ + "used", + "remaining", + "period_start", + "period_end", + "resets_at", + ] { + assert_eq!(body[field], Value::Null, "{field}: {body}"); + } + mock.with(|s| assert_eq!(s.usage_calls, 0)); +} + +/// A key on no plan for OUR stage — here, only on a plan of another API — +/// is the "issued but dead" state: `plan: null`, every counter and period +/// field null, still a `200` with `as_of`. And that answer is cached like any +/// other: it carries no period, so only the TTL retires it. +#[tokio::test] +async fn a_key_on_no_plan_for_our_stage_is_plan_null() { + let mock = MockGateway::start().await; + mock.with(|s| { + s.plans.push(other_api_plan()); + let id = s.seed_on_plan(&format!("discord-{USER_ID}-key"), 100, "q7sd40"); + s.usage.insert(id, vec![vec![9, 9_991]]); + }); + let router = app_against(&mock); + + let reply = call_path( + router.clone(), + "GET", + USAGE_PATH, + Some(&session_cookie(USER_ID)), + ) + .await; + assert_eq!(reply.status, StatusCode::OK); + let body = reply.json(); + assert_eq!(body["plan"], Value::Null, "{body}"); + for field in [ + "used", + "remaining", + "limit", + "period_start", + "period_end", + "resets_at", + ] { + assert_eq!(body[field], Value::Null, "{field}: {body}"); + } + assert!(body["as_of"].is_string(), "{body}"); + + let again = call_path(router, "GET", USAGE_PATH, Some(&session_cookie(USER_ID))).await; + assert_eq!(again.json(), body, "served from the cache"); + mock.with(|s| { + assert_eq!(s.usage_calls, 0, "no plan of ours, nothing to read"); + assert_eq!(s.plans_calls, 1, "the second load was a cache hit"); + }); +} + +/// A key on a plan of ours AND on a plan of another API (a key is one plan per +/// STAGE, not per account): ours is chosen — found across `GetUsagePlans` +/// pages, one plan per page, so page one alone would not do. +#[tokio::test] +async fn the_plan_on_our_stage_is_found_across_pages() { + let mock = MockGateway::start().await; + mock.with(|s| { + s.plans.insert(0, other_api_plan()); + let id = s.seed_on_plan(&format!("discord-{USER_ID}-key"), 100, "q7sd40"); + s.plan_keys.push((PLAN_ID.to_string(), id.clone())); + s.usage.insert(id, vec![vec![1, 99_999]]); + s.plans_page_size = 1; + }); + + let body = usage(&mock, USER_ID).await.json(); + assert_eq!(body["plan"]["tier"], "free", "{body}"); + assert_eq!(body["used"], 1, "{body}"); + mock.with(|s| { + assert!(s.plans_calls >= 2, "two plans at one per page is two pages"); + assert_eq!(s.usage_plan_queries, vec![PLAN_ID.to_string()]); + }); +} + +/// A DAY quota is the UTC day: the period is today, the reset tomorrow at +/// 00:00 UTC, the query today to today — and inside the same day the answer +/// is served from the cache like a MONTH one. +#[tokio::test] +async fn a_day_plan_is_the_utc_day_and_is_served_from_the_cache() { + let mock = MockGateway::start().await; + mock.with(|s| { + s.plans.push(StoredPlan::on_our_stage( + "daily1", + "prices-production-daily-plan", + Some((5.0, 10)), + Some((1_000, 0, "DAY")), + )); + let id = s.seed_on_plan(&format!("discord-{USER_ID}-key"), 100, "daily1"); + s.usage.insert(id, vec![vec![12, 988]]); + }); + let router = app_against(&mock); + + let first = call_path( + router.clone(), + "GET", + USAGE_PATH, + Some(&session_cookie(USER_ID)), + ) + .await; + let body = first.json(); + let today = Utc::now().date_naive(); + let tomorrow = today.succ_opt().unwrap(); + assert_eq!( + body["period_start"], + today.format("%Y-%m-%d").to_string(), + "{body}" + ); + assert_eq!( + body["period_end"], + today.format("%Y-%m-%d").to_string(), + "{body}" + ); + assert_eq!( + body["resets_at"], + format!("{}T00:00:00Z", tomorrow.format("%Y-%m-%d")), + "{body}" + ); + assert_eq!(body["limit"], 1_000, "{body}"); + assert_eq!(body["used"], 12, "{body}"); + + let second = call_path(router, "GET", USAGE_PATH, Some(&session_cookie(USER_ID))).await; + assert_eq!(second.json(), body, "same day, same cached answer"); + mock.with(|s| { + assert_eq!(s.usage_calls, 1, "the second load is a cache hit"); + let (_, start, end) = s.usage_queries[0].clone(); + assert_eq!(start, today.format("%Y-%m-%d").to_string()); + assert_eq!(end, today.format("%Y-%m-%d").to_string()); + }); +} + +/// `used + remaining` disagreeing with the plan's quota is a cross-check +/// failure (logged as a warning), not a figure: the answer states the quota. +#[tokio::test] +async fn a_disagreeing_cross_check_still_states_the_plans_quota() { + let mock = MockGateway::start().await; + mock.with(|s| { + let id = s.seed_on_free_plan(&format!("discord-{USER_ID}-key"), 100); + s.usage.insert(id, vec![vec![10, 5]]); + }); + + let body = usage(&mock, USER_ID).await.json(); + assert_eq!(body["used"], 10, "{body}"); + assert_eq!(body["remaining"], 5, "{body}"); + assert_eq!( + body["limit"], 100_000, + "the plan's quota, not used + remaining" + ); +} + +/// A throttle on the plan lookup lands in the same stale-serve branch as one +/// on `GetUsage` — `plan_of` raises `GatewayError::Throttled`. +#[tokio::test] +async fn a_throttled_plan_lookup_gets_the_last_good_answer() { + let mock = MockGateway::start().await; + seed_key_with_usage(&mock); + let router = usage_router_with(&mock, Duration::ZERO, Duration::from_secs(10)); + + let first = call_path( + router.clone(), + "GET", + USAGE_PATH, + Some(&session_cookie(USER_ID)), + ) + .await; + assert_eq!(first.status, StatusCode::OK); + + mock.with(|s| s.throttle_plans = true); + let second = call_path(router, "GET", USAGE_PATH, Some(&session_cookie(USER_ID))).await; + assert_eq!(second.status, StatusCode::OK, "stale beats an error page"); + assert_eq!(first.json(), second.json(), "same answer, same as_of"); +} + +/// And with nothing cached, a throttled plan lookup is the same `503` a +/// throttled `GetUsage` is. +#[tokio::test] +async fn a_throttled_plan_lookup_with_nothing_cached_is_a_503() { + let mock = MockGateway::start().await; + seed_key_with_usage(&mock); + mock.with(|s| s.throttle_plans = true); + + let reply = usage(&mock, USER_ID).await; + assert_eq!(reply.status, StatusCode::SERVICE_UNAVAILABLE); + assert_eq!(reply.json()["code"], "usage_unavailable"); +} From 7696eca2c1a66f949d64c745c1d581cb506d9ecd Mon Sep 17 00:00:00 2001 From: Adam <65679285+adamkoot@users.noreply.github.com> Date: Thu, 24 Sep 2026 12:43:02 +0200 Subject: [PATCH 03/22] feat(lore-0311): keep the plan across a rework The next issue after a rework used to delete the revoked keys first, and their plan membership went with them: a paid user came back on free. The early delete is gone. The target plan is resolved before anything is deleted: the winner's own plan for our stage (no attach), else the latest revoked key's plan, else free. Step 5 sweeps the revoked keys only after the new key is attached, so the order is create, attach, delete, and a crash between create and attach retries onto the same previous plan. The once-per-period cap is unchanged. --- packages/prices-api/src/portal/keys/mod.rs | 172 ++++++------ packages/prices-api/src/portal/keys/naming.rs | 53 ++++ packages/prices-api/tests/portal_rework.rs | 246 +++++++++++++++++- 3 files changed, 389 insertions(+), 82 deletions(-) diff --git a/packages/prices-api/src/portal/keys/mod.rs b/packages/prices-api/src/portal/keys/mod.rs index 954066d2..fe05fdc3 100644 --- a/packages/prices-api/src/portal/keys/mod.rs +++ b/packages/prices-api/src/portal/keys/mod.rs @@ -112,7 +112,8 @@ use super::period::Period; use cap::Cap; use gateway::{Attachment, Disable, Gateway, GatewayError, KeyValue}; use naming::{ - KeyRecord, choose_winner, current_key, exact_matches, key_name, losers, revocation_instant, + KeyRecord, choose_winner, current_key, exact_matches, key_name, latest_revoked, losers, + revocation_instant, }; /// The reveal, on both verbs — see [`key`] for why `POST` answers identically. @@ -867,8 +868,9 @@ async fn lookup(gateway: &Gateway, name: &str) -> Result { /// What the eligibility-checked issue round-trip needs to know. #[derive(Debug, Clone, PartialEq, Eq)] pub(crate) enum IssueOutcome { - /// A key exists, is on the free plan, and its value is readable — created - /// now or adopted. The value is deliberately not carried: the callback + /// A key exists, is on a usage plan for our stage — the one it was already + /// on, else the plan the previous (revoked) key was on, else free (task + /// 0311) — and its value is readable, created now or adopted. The value is deliberately not carried: the callback /// that consumes this answers with a redirect, and a credential must /// never ride in a `Location`. /// @@ -992,10 +994,14 @@ enum Attempt { /// The least time a create is started with. `CreateApiKey` and /// `CreateUsagePlanKey` each get up to `gateway::OPERATION_TIMEOUT` (5s) in -/// the worst case; in practice each is a few hundred milliseconds. 4s is -/// enough for both at ordinary latency and refuses to start them when the -/// invocation is about to be killed — which is the one way this flow can -/// leave an enabled, unattached key behind. +/// the worst case; in practice each is a few hundred milliseconds. Since task +/// 0311 one or two `GetUsagePlans` reads (the new key's plan, then the +/// previous key's — see [`resolve_target_plan`]) run between them, each as +/// cheap as the attach. 4s is still enough for all of it at ordinary latency +/// and refuses to start the create when the invocation is about to be killed +/// — which is the one way this flow can leave an enabled, unattached key +/// behind. And if it does anyway, the revoked record is still there (it is +/// deleted only after the attach), so the retry resolves the same plan. /// /// Above `auth::issue::RECONCILE_FLOOR` (2s) on purpose: between the two a /// reconciliation can still adopt an existing key but will not start a @@ -1042,19 +1048,23 @@ async fn attempt( // Step 1b (task 0191): a revoked key. If every key under the name is // disabled, the owner revoked it, and the re-issue cap decides: inside the // period of the revocation → refused with the date, nothing written; - // after it → the revocation records are deleted and the flow continues - // into a create, exactly as for a first issue. Deleting BEFORE creating - // is safe here, unlike a swap, because a disabled key is not a credential - // anybody can lose; and it keeps the next listing from ranking a dead key - // ahead of the live one. - let (live, mut revoked): (Vec, Vec) = + // after it → the flow continues into a create, exactly as for a first + // issue. The cap is unchanged by task 0311, and a paid plan buys no extra + // reworks: the rule is per owner and per period, whatever plan they are on. + // + // The revocation records are NOT deleted here any more (task 0311). They + // used to be, before the create — safe then, since a disabled key is not a + // credential anybody can lose — but the previous key's usage-plan + // membership goes with it, and a paid user who reworked silently dropped + // to the free plan. They are now the SOURCE of the new key's plan + // ([`resolve_target_plan`], Step 4), so they are kept until the new key is + // attached and swept by Step 5 afterwards. A crash in between leaves them + // in place, and the retry resolves the same plan. A disabled key is still + // never a candidate: the partition below keeps it out of ranking, and the + // re-listing after a create filters on `enabled` whether or not the + // listing lags (`list_resurrects_deleted` in the tests covers that). + let (live, revoked): (Vec, Vec) = existing.into_iter().partition(|k| k.enabled); - // Ids this attempt deleted. `GetApiKeys` is eventually consistent, so the - // re-listing after a create can still show them; ranked, a deleted - // earlier-created record wins, its attach 404s, and the single retry is - // spent on a phantom — leaving the key just created unattached. Filtered - // out of the re-listing instead, along with anything disabled. - let mut deleted_here: Vec = Vec::new(); if live.is_empty() && !revoked.is_empty() { // The LATEST revocation governs: a duplicate revoked last month must // not open a door a revocation this month closed. @@ -1065,51 +1075,13 @@ async fn attempt( { return Ok(Attempt::Capped { next_eligible_date }); } - // Logged and stepped over, NOT propagated — the same rule, for the - // same reason, as the loser sweep at the end of this function. A `?` - // here makes a delete that cannot succeed permanent: this branch is - // reached on EVERY press once the name holds nothing but revoked keys, - // so one undeletable record (an untagged exact-name key made by hand in - // the console, against the tag condition task 0194 may add to `DELETE`) - // would answer `?issue=failed` forever, with no in-product recovery. - // Housekeeping must not be able to withhold the key the request is for. - // - // The create below is safe with the record still there: the re-listing - // drops everything disabled, so a survivor is neither ranked nor - // adopted, and the sweep at the end tries it again. - let mut undeleted: Vec = Vec::new(); - for dead in &revoked { - if dead.name != name { - tracing::error!( - key_id = %dead.id, - "refusing to delete a key whose name is not the caller's; \ - the exact-name filter has been bypassed" - ); - continue; - } - tracing::info!(key_id = %dead.id, "deleting a revoked key whose period has rolled"); - match gateway.delete(&dead.id).await { - Ok(()) => deleted_here.push(dead.id.clone()), - Err(error) => { - tracing::error!( - key_id = %dead.id, - error = %error, - "could not delete a revoked key whose period has rolled; continuing \ - into the create and leaving it for the sweep" - ); - undeleted.push(dead.clone()); - } - } - } - // Whatever was deleted here is gone, so the sweep below has nothing of - // theirs left to do; whatever failed stays in the list for it to retry. - revoked = undeleted; } let mut created: Option = None; - // Live keys only: a revoked key that survived the branch above (one live - // key beside it — a console re-enable, or a duplicate) is neither ranked - // nor adopted; it is swept like any other loser below. + // Live keys only: a revoked key (the only kind left beside a live one — + // a console re-enable, a duplicate, or the previous key a rework is + // replacing) is neither ranked nor adopted; it is swept like any other + // loser in Step 5, after the winner is attached. let mut candidates = live; // Step 2: nothing yet — create one. The value the create answers with is @@ -1139,7 +1111,7 @@ async fn attempt( // from the same list rather than each deleting the other's. candidates = exact_matches(gateway.list_named(name).await?, name) .into_iter() - .filter(|k| k.enabled && !deleted_here.contains(&k.id)) + .filter(|k| k.enabled) .collect(); if candidates.is_empty() { // The control plane did not list what it just created. Rather than @@ -1149,13 +1121,17 @@ async fn attempt( // the create and this call. Handing its value out would be handing // out a dead id — the one thing the adopt-or-recreate rule exists to // prevent — so this re-enters the flow like any other lost race. - if gateway - .attach_to_plan(&record.id, gateway.free_plan_id()) - .await? - == Attachment::KeyGone + // + // The same target plan as Step 4 (task 0311): a rework's new key + // goes onto the previous key's plan, not onto free by default. + if let Some(plan_id) = resolve_target_plan(gateway, &record.id, &revoked).await? + && gateway.attach_to_plan(&record.id, &plan_id).await? == Attachment::KeyGone { return Ok(Attempt::Retry); } + // Nothing is swept on this path: the listing did not show our key, + // and the next issue — which lists again — sweeps any revoked + // record after its own attach. return Ok(Attempt::Done(Outcome { record, created: true, @@ -1214,11 +1190,17 @@ async fn attempt( // because every retry took the same branch. That is the "issued but does not // work" state this call exists to prevent, made permanent. // - // Idempotent (see `Gateway::attach_to_free_plan`), so the common case — a - // key already on the plan — costs one `409` and no state change. Before the - // deletions below rather than after: the key the caller is about to receive - // is made usable first, and the destructive half of reconciliation only runs - // once that has succeeded. + // WHICH plan is [`resolve_target_plan`]'s answer (task 0311): none, when the + // winner is already on a plan for our stage — an operator's paid plan must + // not be "corrected" to free, and a `409` on free used to hide that only by + // accident — else the plan the previous (revoked) key was on, so a rework + // keeps a paid or custom user on their plan, else free. The attach itself + // is idempotent (`Gateway::attach_to_plan` treats the `409` as success). + // Before the deletions below rather than after: the key the caller is + // about to receive is made usable first, and the destructive half of + // reconciliation only runs once that has succeeded — which since 0311 also + // means the revoked records, the source of the target plan, outlive the + // attach. // // A `404` from it is not a failure but the deletion race — the winner was // listed and is gone — and it is reported the way `value_of`'s `None` is: @@ -1226,16 +1208,16 @@ async fn attempt( // replacement. Because this call now runs before the read, it is the first // place that race can surface, so it has to answer it rather than turn a // hand-deleted key back into the dead end this slice exists to remove. - if gateway - .attach_to_plan(&winner.id, gateway.free_plan_id()) - .await? - == Attachment::KeyGone + if let Some(plan_id) = resolve_target_plan(gateway, &winner.id, &revoked).await? + && gateway.attach_to_plan(&winner.id, &plan_id).await? == Attachment::KeyGone { return Ok(Attempt::Retry); } // Step 5: everything that is not the winner is deleted — duplicates, and - // (task 0191) any revoked key left beside a live one. + // (task 0191) any revoked key left beside a live one, which since task + // 0311 includes the previous key a rework replaced: it is deleted here, + // AFTER the winner is on its plan, never before. for loser in losers(&candidates, &winner) .into_iter() .chain(revoked.iter()) @@ -1300,6 +1282,44 @@ async fn attempt( } } +/// The usage plan to attach `winner_id` to, decided before anything is +/// deleted (task 0311). `None` means "no attach": the winner is already on a +/// plan for our stage. +/// +/// 1. The winner's own plan for our stage — Step 4's "however it came to +/// exist": a key an operator moved to Basic stays on Basic. +/// 2. Else the plan of the **previous key**: the latest revoked record in this +/// attempt's listing (the same latest-revocation rule the cap reads, +/// [`latest_revoked`]), if it is on a plan for our stage. This is what +/// makes a rework keep the plan — paid stays paid, custom stays custom, +/// free stays free. +/// 3. Else free — a first issue, or a previous key on no plan of ours. +/// +/// Only a plan `GetUsagePlans` reported on OUR API stage can come out of +/// here, which is the code half of the wildcard `POST /usageplans/*/keys` +/// grant's justification (`api-gateway-stack.ts`). +async fn resolve_target_plan( + gateway: &Gateway, + winner_id: &str, + revoked: &[KeyRecord], +) -> Result, GatewayError> { + if gateway.plan_of(winner_id).await?.is_some() { + return Ok(None); + } + if let Some(previous) = latest_revoked(revoked) + && let Some(plan) = gateway.plan_of(&previous.id).await? + { + tracing::info!( + previous_key_id = %previous.id, + plan_id = %plan.id, + tier = ?plan.tier, + "attaching the new key to the plan the previous key was on" + ); + return Ok(Some(plan.id)); + } + Ok(Some(gateway.free_plan_id().to_string())) +} + /// `no-store` on every response this module produces. /// /// The handler's own statement of the rule, alongside the configuration that diff --git a/packages/prices-api/src/portal/keys/naming.rs b/packages/prices-api/src/portal/keys/naming.rs index 06b606ff..e35e2b80 100644 --- a/packages/prices-api/src/portal/keys/naming.rs +++ b/packages/prices-api/src/portal/keys/naming.rs @@ -114,6 +114,24 @@ pub fn revocation_instant(revoked: &[KeyRecord]) -> Option { .max() } +/// The revoked record whose plan a rework keeps (task 0311): the one with the +/// **latest** `lastUpdatedDate` — the same "latest revocation governs" rule as +/// [`revocation_instant`], so the record the cap is decided from and the +/// record the new key's plan is read from are the same record. +/// +/// An undated record loses to any dated one (it is skipped by +/// [`revocation_instant`] for the same reason); ties — including all-undated — +/// go to the smaller id, so every invocation reads the same record. Empty → +/// `None`, and the caller falls back to the free plan. +pub fn latest_revoked(revoked: &[KeyRecord]) -> Option<&KeyRecord> { + revoked.iter().max_by(|a, b| { + a.last_updated_at + .cmp(&b.last_updated_at) + // Reversed, so that among equal instants the SMALLER id is "max". + .then_with(|| b.id.cmp(&a.id)) + }) +} + /// The key the owner currently holds, among `records`: the earliest **enabled** /// key if there is one, otherwise the earliest key of any state (task 0191). /// @@ -247,6 +265,41 @@ mod tests { assert_eq!(revocation_instant(&[]), None); } + /// The latest revocation is the record a rework keeps the plan of; an + /// undated record loses; a tie goes to the smaller id; nothing → None. + #[test] + fn the_latest_revoked_record_is_the_latest_dated_one() { + let at = |id: &str, when: Option| KeyRecord { + last_updated_at: when, + ..disabled(id, "n", Some(1)) + }; + let older = at("a", Some(10)); + let newer = at("b", Some(20)); + let undated = at("c", None); + assert_eq!( + latest_revoked(&[older.clone(), newer.clone(), undated.clone()]) + .unwrap() + .id, + "b" + ); + assert_eq!( + latest_revoked(&[undated.clone(), older.clone()]) + .unwrap() + .id, + "a" + ); + let tie_high = at("z", Some(20)); + assert_eq!( + latest_revoked(&[tie_high.clone(), newer.clone()]) + .unwrap() + .id, + "b" + ); + assert_eq!(latest_revoked(&[newer.clone(), tie_high]).unwrap().id, "b"); + assert_eq!(latest_revoked(&[undated]).unwrap().id, "c"); + assert!(latest_revoked(&[]).is_none()); + } + #[test] fn a_name_is_the_id_wrapped_in_both_fixed_parts() { assert_eq!( diff --git a/packages/prices-api/tests/portal_rework.rs b/packages/prices-api/tests/portal_rework.rs index 91c007d5..fbc34233 100644 --- a/packages/prices-api/tests/portal_rework.rs +++ b/packages/prices-api/tests/portal_rework.rs @@ -15,8 +15,9 @@ //! names the date, and never hands the dead value out again. //! - **The replacement waits for the 1st.** An issue inside the revocation's //! period is `?issue=capped&next_eligible_at=…` with nothing written; the -//! same issue once the period has rolled deletes the revocation record and -//! creates the new key. +//! same issue once the period has rolled creates the new key, attaches it to +//! the plan the revoked key was on (task 0311) and only then deletes the +//! revocation record. //! - **A session cookie can revoke its own key and nothing else.** The route //! is `POST`-only, writes only `enabled=false` on the caller's exact name, //! and every other state the store can be in leaves it untouched. @@ -561,8 +562,8 @@ async fn an_issue_after_a_revoke_in_the_same_period_is_capped_with_the_date() { /// The worked example against the real calendar: revoked on the 3rd of THIS /// month → capped until the 1st of next; revoked on the 3rd of LAST month → -/// the period has rolled, the revocation record is deleted, a new key is -/// created and attached, and the reveal hands the NEW one out. +/// the period has rolled, a new key is created and attached, the revocation +/// record is deleted, and the reveal hands the NEW one out. #[tokio::test] async fn revoked_on_the_3rd_refuses_until_the_1st_and_issues_once_it_has_passed() { let discord = MockDiscord::start(GRANTED_SCOPE, None).await; @@ -600,12 +601,16 @@ async fn revoked_on_the_3rd_refuses_until_the_1st_and_issues_once_it_has_passed( .contains(&(PLAN_ID.to_string(), s.keys[0].id.clone())) ); let new = s.keys[0].id.clone(); + // Create, attach, THEN delete (task 0311). The revocation record used + // to be deleted first, and its usage-plan membership went with it — + // a paid user who reworked came back on the free plan. It is now the + // source of the new key's plan, so it outlives the attach. assert_eq!( s.ops, vec![ - format!("delete:{dead}"), format!("create:{new}"), - format!("attach:{new}") + format!("attach:{new}"), + format!("delete:{dead}") ] ); }); @@ -1084,3 +1089,232 @@ async fn a_same_site_revoke_is_accepted_from_the_configured_origin_only() { assert_eq!(reply.status, StatusCode::OK, "{:?}", reply.json()); assert!(!gateway.with(|s| s.keys[0].enabled), "the key is off"); } + +// --------------------------------------------------------------------------- +// A rework keeps the plan (task 0311) +// --------------------------------------------------------------------------- + +/// Seed a key the user revoked at `revoked_at`, on usage plan `plan`. +fn seed_revoked_on(gateway: &MockGateway, revoked_at: u64, plan: &str) -> String { + gateway.with(|s| { + let id = s.seed_revoked(&key_name(), 1_000, revoked_at); + s.plan_keys.push((plan.to_string(), id.clone())); + id + }) +} + +/// The revocation record is never deleted before the new key is attached: +/// `attach:` comes first in `ops`, `delete:` after it. +fn assert_attached_before_deleted(ops: &[String], new: &str, dead: &str) { + let attach = ops + .iter() + .position(|op| op == &format!("attach:{new}")) + .unwrap_or_else(|| panic!("the new key was never attached: {ops:?}")); + let delete = ops + .iter() + .position(|op| op == &format!("delete:{dead}")) + .unwrap_or_else(|| panic!("the revoked key was never deleted: {ops:?}")); + assert!( + attach < delete, + "the revoked key was deleted before the new key was attached: {ops:?}" + ); +} + +/// Issue once the rework's period has rolled, for a revoked key on `plan`, and +/// answer `(dead, new)`. +async fn rework_rolled_on(plan: StoredPlan) -> (MockGateway, Router, String, String) { + let discord = MockDiscord::start(GRANTED_SCOPE, None).await; + let gateway = MockGateway::start().await; + let plan_id = plan.id.clone(); + if plan_id != PLAN_ID { + gateway.with(|s| s.plans.push(plan)); + } + let dead = seed_revoked_on(&gateway, the_3rd_of(first_of_month_offset(-1)), &plan_id); + let app = app_with_discord(&discord, &gateway); + assert_eq!(issue_round_trip(&app).await.location(), "/api/?issue=ok"); + let new = gateway.with(|s| { + assert_eq!(s.keys.len(), 1, "the revocation record is gone"); + s.keys[0].id.clone() + }); + // `MockDiscord` is dropped here; the reveal/usage routes below need no + // Discord. + (gateway, app, dead, new) +} + +/// A free user's rework stays on free: the new key on `freeplan1`, created, +/// attached and only then the revoked record deleted. +#[tokio::test] +async fn a_rework_on_free_stays_on_free() { + let (gateway, _app, dead, new) = rework_rolled_on(StoredPlan::free()).await; + gateway.with(|s| { + assert!(s.plan_keys.contains(&(PLAN_ID.to_string(), new.clone()))); + assert_eq!( + s.ops, + vec![ + format!("create:{new}"), + format!("attach:{new}"), + format!("delete:{dead}") + ] + ); + assert_attached_before_deleted(&s.ops, &new, &dead); + }); +} + +/// A paid user's rework keeps the paid plan: the new key goes onto `basic1`, +/// not onto free — the silent downgrade task 0311 exists to remove — and the +/// dashboard's usage then reports Basic. +#[tokio::test] +async fn a_rework_on_a_paid_plan_keeps_the_paid_plan() { + let (gateway, app, dead, new) = rework_rolled_on(StoredPlan::basic()).await; + gateway.with(|s| { + assert!( + s.plan_keys + .contains(&(BASIC_PLAN_ID.to_string(), new.clone())), + "{:?}", + s.plan_keys + ); + assert!( + !s.plan_keys.contains(&(PLAN_ID.to_string(), new.clone())), + "the new key must not land on free" + ); + assert_eq!( + s.ops, + vec![ + format!("create:{new}"), + format!("attach:{new}"), + format!("delete:{dead}") + ] + ); + assert_attached_before_deleted(&s.ops, &new, &dead); + }); + + let usage = call_path(app, "GET", USAGE_PATH, Some(&session_cookie(USER_ID))).await; + assert_eq!(usage.status, StatusCode::OK); + assert_eq!(usage.json()["plan"]["tier"], "basic"); +} + +/// A hand-made Custom plan on our stage is kept the same way. +#[tokio::test] +async fn a_rework_on_a_custom_plan_keeps_the_custom_plan() { + let (gateway, _app, dead, new) = rework_rolled_on(StoredPlan::custom_unlimited()).await; + gateway.with(|s| { + assert!( + s.plan_keys + .contains(&(CUSTOM_PLAN_ID.to_string(), new.clone())), + "{:?}", + s.plan_keys + ); + assert!(!s.plan_keys.contains(&(PLAN_ID.to_string(), new.clone()))); + assert_attached_before_deleted(&s.ops, &new, &dead); + }); +} + +/// A previous key on no plan of OURS (only on another API's plan) gives the +/// rework nothing to keep: the new key goes onto free, never onto the other +/// API's plan. +#[tokio::test] +async fn a_rework_whose_previous_plan_is_not_ours_lands_on_free() { + let other = StoredPlan { + id: "q7sd40".to_string(), + name: "production-partner-plan".to_string(), + api_stages: vec![("6l9k06w4pl".to_string(), STAGE.to_string())], + throttle: Some((50.0, 100)), + quota: Some((10_000, 0, "DAY")), + }; + let (gateway, _app, dead, new) = rework_rolled_on(other).await; + gateway.with(|s| { + assert!(s.plan_keys.contains(&(PLAN_ID.to_string(), new.clone()))); + assert!(!s.plan_keys.contains(&("q7sd40".to_string(), new.clone()))); + assert_attached_before_deleted(&s.ops, &new, &dead); + }); +} + +/// The crash window task 0311 closes: the new key is created and the attach +/// then fails. The revoked record is still there — it is only deleted after +/// an attach — so the retry reads the same previous plan and lands the new +/// key on `basic1`, then sweeps the revoked record. +#[tokio::test] +async fn a_crash_between_create_and_attach_retries_onto_the_previous_keys_plan() { + let discord = MockDiscord::start(GRANTED_SCOPE, None).await; + let gateway = MockGateway::start().await; + gateway.with(|s| { + s.plans.push(StoredPlan::basic()); + s.fail_next_attach = true; + }); + let dead = seed_revoked_on( + &gateway, + the_3rd_of(first_of_month_offset(-1)), + BASIC_PLAN_ID, + ); + let app = app_with_discord(&discord, &gateway); + + assert_eq!( + issue_round_trip(&app).await.location(), + "/api/?issue=failed" + ); + let new = gateway.with(|s| { + assert_eq!(s.keys.len(), 2, "the new key exists beside the revoked one"); + assert!(s.deleted.is_empty(), "nothing is deleted before an attach"); + let new = s + .keys + .iter() + .find(|k| k.enabled) + .expect("the new key was created") + .id + .clone(); + assert!(!s.plan_keys.iter().any(|(_, k)| k == &new), "not attached"); + assert_eq!(s.ops, vec![format!("create:{new}")]); + new + }); + + assert_eq!(issue_round_trip(&app).await.location(), "/api/?issue=ok"); + gateway.with(|s| { + assert!( + s.plan_keys + .contains(&(BASIC_PLAN_ID.to_string(), new.clone())), + "{:?}", + s.plan_keys + ); + assert!(!s.plan_keys.contains(&(PLAN_ID.to_string(), new.clone()))); + assert_eq!( + s.ops, + vec![ + format!("create:{new}"), + format!("attach:{new}"), + format!("delete:{dead}") + ] + ); + assert_attached_before_deleted(&s.ops, &new, &dead); + assert_eq!(s.create_calls, 1, "the retry adopts, it does not re-create"); + }); +} + +/// Step 4's "however it came to exist", with a paid plan: a live key an +/// operator moved to Basic is adopted as it is — no attach call at all, and +/// certainly no attempt to put it back on free. +#[tokio::test] +async fn a_live_key_already_on_a_plan_gets_no_attach() { + let discord = MockDiscord::start(GRANTED_SCOPE, None).await; + for plan in [StoredPlan::free(), StoredPlan::basic()] { + let gateway = MockGateway::start().await; + let plan_id = plan.id.clone(); + let key = gateway.with(|s| { + if plan_id != PLAN_ID { + s.plans.push(plan); + } + s.seed_on_plan(&key_name(), 1_000, &plan_id) + }); + let app = app_with_discord(&discord, &gateway); + + assert_eq!( + issue_round_trip(&app).await.location(), + "/api/?issue=ok", + "{plan_id}" + ); + gateway.with(|s| { + assert_eq!(s.attach_calls, 0, "{plan_id}: already on a plan, no attach"); + assert_eq!(s.plan_keys, vec![(plan_id.clone(), key.clone())]); + assert!(s.ops.is_empty(), "{plan_id}: nothing written: {:?}", s.ops); + }); + } +} From 681c03f496097967392153b4227a5e66ade3f727 Mon Sep 17 00:00:00 2001 From: Adam <65679285+adamkoot@users.noreply.github.com> Date: Thu, 24 Sep 2026 12:49:32 +0200 Subject: [PATCH 04/22] feat(lore-0311): show the key's own plan on the dashboard The Rate Limit card states the plan /api/usage reports: a second pill beside Active (Free, Basic, Analyst, Lite, Pro or Custom), the plan's per-second rate and x60 per minute, "Unlimited" for a plan without a throttle, and a stated no-plan state with no pill. /config's free figure remains for the no-key state, the landing page and an unanswered usage call. Monthly Usage takes its limit and reset from the plan and states the no-plan, unlimited and unsupported-period cases instead of zeros. The contact line is now a link to RUMBLEFISH_CONTACT with copy per tier. --- web/portal/src/api/portal.ts | 60 +++- web/portal/src/app/app.spec.tsx | 356 ++++++++++++++++++++-- web/portal/src/app/app.tsx | 247 +++++++++++---- web/portal/src/landing/DashboardPanel.tsx | 14 +- 4 files changed, 591 insertions(+), 86 deletions(-) diff --git a/web/portal/src/api/portal.ts b/web/portal/src/api/portal.ts index 8681c319..44a64036 100644 --- a/web/portal/src/api/portal.ts +++ b/web/portal/src/api/portal.ts @@ -493,33 +493,71 @@ export interface PortalKey { last_updated_at?: string | null; } +/** The tier of a usage plan, as the dashboard labels it (task 0311). */ +export type PortalPlanTier = + | 'free' + | 'basic' + | 'analyst' + | 'lite' + | 'pro' + | 'custom'; + +/** + * The key's usage plan on this API's stage (task 0311) — `PlanWire` in + * `packages/prices-api/src/portal/usage/mod.rs`. + * + * Every figure is `null` when the plan does not set it: a plan with no + * throttle or no quota is unlimited in that respect, and the page says + * "Unlimited" rather than rendering a zero. + */ +export interface PortalPlan { + /** `free`…`pro` for the five CDK plans; `custom` for any other plan on our stage. */ + tier: PortalPlanTier; + /** The AWS plan name — what a Custom plan is known by. */ + name: string; + /** Sustained requests per second per key. */ + rate_limit_per_second: number | null; + /** Token-bucket capacity above the rate. */ + burst_limit: number | null; + /** Requests per quota period. */ + quota_limit: number | null; + /** `MONTH`, `DAY`, `WEEK`, or whatever else AWS answers, verbatim. */ + quota_period: string | null; +} + /** - * What `GET /api/usage` answers (task 0188). + * What `GET /api/usage` answers (task 0188; the plan since task 0311). * * Mirrors `UsageResponse` in `packages/prices-api/src/portal/usage/mod.rs`, and * hand-written for the same reason every type above is: the portal's routes are * deliberately absent from the published OpenAPI document. * - * The three counters are `null` **together** when AWS has recorded nothing for - * the key yet — the ordinary state minutes after issuance, because `GetUsage` - * lags. The page renders that as "nothing recorded yet" rather than inventing - * zeros; the period and `as_of` are always present. + * `used` and `remaining` are `null` **together** when AWS has recorded nothing + * for the key yet — the ordinary state minutes after issuance, because + * `GetUsage` lags — and the page renders that as "nothing recorded yet" + * rather than inventing zeros. They are also `null`, with `limit` and the + * period fields, when the key is on no plan for this API's stage + * (`plan: null`) or on a plan without a quota; `limit` and the period are + * `null` too when the plan's quota period is one the backend does not compute + * (`WEEK`). `as_of` is always present. */ export interface PortalUsage { /** Requests counted against the quota this period, per AWS. */ used: number | null; /** Requests left, as of the latest day AWS has data for. */ remaining: number | null; - /** The monthly quota, reconstructed as `used + remaining`. */ + /** The plan's quota — `plan.quota_limit`. */ limit: number | null; - /** First day of the current period, `YYYY-MM-DD` (our rule: calendar month, UTC). */ - period_start: string; + /** First day of the current period, `YYYY-MM-DD` (MONTH: the calendar month, UTC; DAY: the UTC day). */ + period_start: string | null; /** Last day of the current period, inclusive. */ - period_end: string; + period_end: string | null; /** When the quota resets under our stated rule, RFC 3339. */ - resets_at: string; - /** When the `GetUsage` behind this answer was made, RFC 3339. */ + resets_at: string | null; + /** When the lookup behind this answer was made, RFC 3339. */ as_of: string; + /** The key's plan on this API's stage; `null` when it is on none. */ + plan: PortalPlan | null; } /** diff --git a/web/portal/src/app/app.spec.tsx b/web/portal/src/app/app.spec.tsx index bc7a6b6e..ca3e821a 100644 --- a/web/portal/src/app/app.spec.tsx +++ b/web/portal/src/app/app.spec.tsx @@ -9,6 +9,7 @@ import { MemoryRouter, useLocation, useNavigate } from 'react-router-dom'; import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest'; import { ROUTER_BASENAME } from '../base-path'; +import { RUMBLEFISH_CONTACT } from '../landing/links'; import App from './app'; import { FIXTURE } from '../docs/openapi.fixture'; @@ -95,9 +96,10 @@ const ISSUE_HREF = '/api/auth/login?action=issue'; * The limit is part of every open-portal stub because it is part of every real * `/config`: `compute-stack.ts` sets `PORTAL_RATE_LIMIT` from * `pricingApiFreePlanRateLimit` unconditionally. `1` is what - * `infra/envs/production.json` holds today — and the point of the field is that - * changing that file changes the page, so the tests below assert the rendered - * figure against THIS value rather than against a literal of their own. + * `infra/envs/production.json` holds today. Since task 0311 the signed-in + * dashboard states the key's OWN plan from `/api/usage`; `/config`'s figure + * feeds only the no-key state, the landing page, and the fallback while the + * usage call is unanswered or failed. */ const openConfig = () => ({ json: async () => ({ enabled: true, rate_limit_per_second: 1 }), @@ -2994,6 +2996,16 @@ describe('usage against quota', () => { vi.unstubAllGlobals(); }); + /** The free plan as `/api/usage` reports it (task 0311). */ + const FREE_PLAN = { + tier: 'free', + name: 'pricing-api-free-production', + rate_limit_per_second: 1, + burst_limit: 5, + quota_limit: 100000, + quota_period: 'MONTH', + }; + const USAGE = { used: 121, remaining: 99879, @@ -3002,8 +3014,33 @@ describe('usage against quota', () => { period_end: '2026-08-31', resets_at: '2026-09-01T00:00:00Z', as_of: '2026-08-19T10:15:00Z', + plan: FREE_PLAN, }; + /** A paid plan on our stage, named the way CDK names it (task 0311). */ + const paidPlan = ( + tier: string, + rate: number, + burst: number, + quota: number, + ) => ({ + tier, + name: `pricing-api-${tier}-production`, + rate_limit_per_second: rate, + burst_limit: burst, + quota_limit: quota, + quota_period: 'MONTH', + }); + + /** A `/api/usage` answer on `plan`, with `overrides` on top (task 0311). */ + const usageWith = + (plan: unknown, overrides: Record = {}) => + () => ({ json: async () => ({ ...USAGE, plan, ...overrides }) }); + + /** The Rate Limit card's `
`, once it has rendered. */ + const rateLimitCard = async () => + (await screen.findByText('Rate Limit')).closest('section') as HTMLElement; + const signedInWithUsage = ( usage: () => Partial & { json?: () => unknown } = () => ({ json: async () => USAGE, @@ -3177,10 +3214,11 @@ describe('usage against quota', () => { }); /** - * AWS has no rows for the key yet — `used`/`remaining`/`limit` are `null` - * together. Not rendered as zeros: "0 used of 100000" would be an invented - * figure, and the honest state for a fresh key is "nothing recorded yet". - * The reset rule and rate limit still render — they are ours, not AWS's. + * AWS has no rows for the key yet — `used`/`remaining` are `null` together. + * Not rendered as zeros: "0 used of 100000" would be an invented figure, + * and the honest state for a fresh key is "nothing recorded yet". The reset + * rule and rate limit still render — they are ours, not AWS's. (`limit` is + * the plan's quota since task 0311, known before AWS records a row.) */ it('says nothing is recorded yet instead of inventing zeros', async () => { signedInWithUsage(() => ({ @@ -3188,7 +3226,6 @@ describe('usage against quota', () => { ...USAGE, used: null, remaining: null, - limit: null, }), })); renderApp(); @@ -3381,16 +3418,14 @@ describe('usage against quota', () => { }); /** - * The rate limit is the gateway's, not this bundle's (task 0188). + * The rate limit is the gateway's, not this bundle's (task 0188) — and since + * task 0311 it is the KEY'S OWN PLAN's, which `/api/usage` reports. * - * `pricingApiFreePlanRateLimit` is a per-env config value that - * `api-gateway-stack.ts` hands to `addUsagePlan` and `compute-stack.ts` hands - * to the backend. Raising it and deploying has to change what this panel - * says — with a literal here it would not, and the one section whose stated - * theme is rendering honestly would be quietly stating a limit nobody - * enforces any more. + * `/config` still carries the free plan's figure (5 here), but a key an + * operator moved to Basic is throttled at Basic's 3 req/s, and that is what + * the card must say: the plan's figure beats `/config`'s. */ - it('states the rate limit the backend reports, not a built-in figure', async () => { + it("states the key's own plan's rate limit, not /config's", async () => { stubRoutes({ [CONFIG_URL]: () => ({ json: async () => ({ enabled: true, rate_limit_per_second: 5 }), @@ -3414,12 +3449,21 @@ describe('usage against quota', () => { username: 'adam', }), }), - [USAGE_URL]: () => ({ json: async () => USAGE }), + [USAGE_URL]: usageWith(paidPlan('basic', 3, 15, 1000000), { + limit: 1000000, + remaining: 999879, + }), }); renderApp(); await screen.findByTestId('usage-used'); - expect(screen.getByTestId('rate-limit').textContent).toBe('5'); + await waitFor(() => + expect(screen.getByTestId('rate-limit').textContent).toBe('3'), + ); + expect( + within(await rateLimitCard()).getByText('180'), + 'per-minute is the plan rate times sixty', + ).toBeTruthy(); // Plural, because the figure is no longer the one the sentence was // written around. expect(screen.getByText(/requests per second/i)).toBeTruthy(); @@ -3512,7 +3556,13 @@ describe('usage against quota', () => { username: 'adam', }), }), - [USAGE_URL]: () => ({ json: async () => USAGE }), + // Failed, so the card has no plan to state (task 0311) and falls back + // to `/config` — which here says nothing, so to the built-in figure. + [USAGE_URL]: () => ({ + ok: false, + status: 500, + json: async () => ({}), + }), }); renderApp(); @@ -3522,12 +3572,278 @@ describe('usage against quota', () => { // (task 0157), the same figure the landing page states to every visitor. // A stated figure beats a third of the dashboard disappearing — and where // the deployment DOES answer, its value still wins (the test above). - await screen.findByTestId('usage-used'); + await screen.findByText(/Could not load your usage/); expect((await screen.findByTestId('rate-limit')).textContent).toBe('1'); expect(screen.getByText(/per-minute limit/i)).toBeTruthy(); expect(screen.getByText(/request per second/i)).toBeTruthy(); }); + // ------------------------------------------------------------------------- + // The key's own plan (task 0311) + // ------------------------------------------------------------------------- + + /** + * The plan pill: a second pill beside "Active" in the Rate Limit card's + * header, on every tier — free included — and "Custom" for any other plan + * on our stage (decision 5). + */ + it('shows the plan pill beside Active, per tier', async () => { + for (const [plan, label] of [ + [FREE_PLAN, 'Free'], + [paidPlan('basic', 3, 15, 1000000), 'Basic'], + [paidPlan('analyst', 5, 25, 5000000), 'Analyst'], + [paidPlan('lite', 10, 50, 20000000), 'Lite'], + [paidPlan('pro', 25, 125, 50000000), 'Pro'], + [ + { + ...paidPlan('custom', 150, 300, 1000000), + name: 'prices-production-loadtest-plan', + }, + 'Custom', + ], + ] as const) { + signedInWithUsage(usageWith(plan)); + const view = renderApp(); + + const card = await rateLimitCard(); + await waitFor(() => expect(within(card).getByText(label)).toBeTruthy()); + expect(within(card).getByText('Active'), label).toBeTruthy(); + expect(within(card).getByTestId('rate-limit').textContent, label).toBe( + String(plan.rate_limit_per_second), + ); + view.unmount(); + } + }); + + /** + * `/api/usage` failed: the card keeps its pre-0311 rendering — `/config`'s + * figure, the "Active" pill and no plan pill — and, since the figure it + * states is the free plan's, the free plan's contact copy (decision 6). + */ + it("keeps /config's figure and the free contact copy when usage fails", async () => { + stubRoutes({ + [CONFIG_URL]: () => ({ + json: async () => ({ enabled: true, rate_limit_per_second: 5 }), + }), + [KEY_URL]: () => ({ + json: async () => ({ + key_id: 'rate-limit-suite-key', + name: 'discord-rate-limit-key', + value: 'aBcDeF0123456789aBcDeF0123456789aBcDeF01', + }), + }), + [ME_URL]: () => ({ + json: async () => ({ + authenticated: true, + user_id: '308994132968210433', + username: 'adam', + }), + }), + [USAGE_URL]: () => ({ ok: false, status: 500, json: async () => ({}) }), + }); + renderApp(); + + await screen.findByText(/Could not load your usage/); + const card = await rateLimitCard(); + expect(within(card).getByTestId('rate-limit').textContent).toBe('5'); + expect(within(card).getByText('Active')).toBeTruthy(); + for (const label of ['Free', 'Basic', 'Analyst', 'Lite', 'Pro', 'Custom']) { + expect(within(card).queryByText(label), label).toBeNull(); + } + expect(within(card).getByTestId('rate-limit-contact').textContent).toMatch( + /^Need higher limits\?/, + ); + const link = within(card).getByRole('link', { name: /contact us/i }); + expect(link.textContent).toBe('Contact us about a paid plan.'); + expect(link.getAttribute('href')).toBe(RUMBLEFISH_CONTACT); + }); + + /** + * A key on no usage plan for this API (`plan: null`): stated in both cards, + * no "Active", no plan pill, and nothing that reads as a zero or as + * "nothing recorded yet" (decision 7a). + */ + it('states the no-plan state in both cards, with no pill', async () => { + signedInWithUsage( + usageWith(null, { + used: null, + remaining: null, + limit: null, + period_start: null, + period_end: null, + resets_at: null, + }), + ); + renderApp(); + + expect((await screen.findByTestId('rate-limit-no-plan')).textContent).toBe( + 'This key is not on a usage plan for this API, so the API answers 403 to it.', + ); + expect(screen.getByTestId('usage-no-plan').textContent).toBe( + 'No usage plan, so there is no quota to count against.', + ); + const card = await rateLimitCard(); + expect(within(card).queryByText('Active')).toBeNull(); + expect(within(card).queryByTestId('rate-limit')).toBeNull(); + for (const label of ['Free', 'Basic', 'Analyst', 'Lite', 'Pro', 'Custom']) { + expect(within(card).queryByText(label), label).toBeNull(); + } + expect(screen.queryByTestId('usage-used')).toBeNull(); + expect(document.body.textContent).not.toMatch(/not recorded any usage/i); + expect(document.body.textContent).not.toMatch(/nothing recorded/i); + }); + + /** + * A plan without a throttle or a quota: both Rate Limit figures read + * "Unlimited", Monthly Usage says so with no meter, and the pill is + * "Custom" (decision 7b). Never a zero. + */ + it('states Unlimited for a plan without limits', async () => { + signedInWithUsage( + usageWith( + { + tier: 'custom', + name: 'prices-production-acme-plan', + rate_limit_per_second: null, + burst_limit: null, + quota_limit: null, + quota_period: null, + }, + { + used: null, + remaining: null, + limit: null, + period_start: null, + period_end: null, + resets_at: null, + }, + ), + ); + renderApp(); + + expect((await screen.findByTestId('usage-unlimited')).textContent).toMatch( + /Unlimited/, + ); + const card = await rateLimitCard(); + await waitFor(() => expect(within(card).getByText('Custom')).toBeTruthy()); + expect(within(card).getByTestId('rate-limit').textContent).toBe( + 'Unlimited', + ); + expect(within(card).getAllByText('Unlimited')).toHaveLength(2); + expect(within(card).queryByText('0')).toBeNull(); + expect(screen.queryByTestId('usage-used')).toBeNull(); + expect(screen.queryByTestId('usage-limit')).toBeNull(); + }); + + /** + * Monthly Usage takes its limit and its reset from the plan: a Basic key's + * meter is out of 1,000,000 and resets on the date the backend computed. + */ + it("takes Monthly Usage's limit and reset from the plan", async () => { + signedInWithUsage( + usageWith(paidPlan('basic', 3, 15, 1000000), { + used: 5000, + remaining: 995000, + limit: 1000000, + period_start: '2026-09-01', + period_end: '2026-09-30', + resets_at: '2026-10-01T00:00:00Z', + }), + ); + renderApp(); + + expect((await screen.findByTestId('usage-limit')).textContent).toBe( + '1000000', + ); + expect(screen.getByTestId('usage-used').textContent).toBe('5000'); + expect(screen.getByText('Resets 1 October')).toBeTruthy(); + }); + + /** A quota period the backend does not compute is named, not guessed. */ + it('names an unsupported quota period instead of guessing it', async () => { + signedInWithUsage( + usageWith( + { + tier: 'custom', + name: 'prices-production-weekly-plan', + rate_limit_per_second: 2, + burst_limit: 10, + quota_limit: 5000, + quota_period: 'WEEK', + }, + { + used: null, + remaining: null, + limit: 5000, + period_start: null, + period_end: null, + resets_at: null, + }, + ), + ); + renderApp(); + + expect( + (await screen.findByTestId('usage-unsupported-period')).textContent, + ).toBe('Quota per week; usage for this period is not shown.'); + expect(screen.getByTestId('usage-period-quota').textContent).toBe('5,000'); + expect(screen.queryByTestId('usage-used')).toBeNull(); + }); + + /** + * The contact line is a link to `RUMBLEFISH_CONTACT`, worded for the tier + * (decision 6): contacting us is the only way to change plan, so it is on + * every tier, paid ones included. + */ + it('links the contact line to RUMBLEFISH_CONTACT with copy per tier', async () => { + for (const [plan, lead, linkText] of [ + [FREE_PLAN, 'Need higher limits?', 'Contact us about a paid plan.'], + [ + paidPlan('basic', 3, 15, 1000000), + 'Need more?', + 'Contact us to change your plan.', + ], + [ + paidPlan('analyst', 5, 25, 5000000), + 'Need more?', + 'Contact us to change your plan.', + ], + [ + paidPlan('lite', 10, 50, 20000000), + 'Need more?', + 'Contact us to change your plan.', + ], + [ + paidPlan('pro', 25, 125, 50000000), + 'Need custom limits?', + 'Contact us.', + ], + [ + { ...paidPlan('custom', 150, 300, 1000000), name: 'acme' }, + 'Need custom limits?', + 'Contact us.', + ], + ] as const) { + signedInWithUsage(usageWith(plan)); + const view = renderApp(); + + const card = await rateLimitCard(); + await waitFor(() => + expect( + within(card).getByRole('link', { name: /contact us/i }).textContent, + plan.tier, + ).toBe(linkText), + ); + const link = within(card).getByRole('link', { name: /contact us/i }); + expect(link.getAttribute('href'), plan.tier).toBe(RUMBLEFISH_CONTACT); + expect( + within(card).getByTestId('rate-limit-contact').textContent, + plan.tier, + ).toBe(`${lead} ${linkText}`); + view.unmount(); + } + }); + /** * The backend writes a message for each failure it authors; the page shows * it (task 0188). diff --git a/web/portal/src/app/app.tsx b/web/portal/src/app/app.tsx index db04f0dc..2fbbe2ab 100644 --- a/web/portal/src/app/app.tsx +++ b/web/portal/src/app/app.tsx @@ -104,6 +104,8 @@ import { type PortalConfig, type PortalKey, type PortalKeyRevoked, + type PortalPlan, + type PortalPlanTier, type PortalSession, type PortalUsage, } from '../api/portal'; @@ -1420,10 +1422,16 @@ function Dashboard({ // The quota `/usage` reported, so the key card's "Monthly quota" field can // state a number this page was actually told. `undefined` until the panel // below has an answer; the field is simply absent until then. + // + // `plan` (task 0311) feeds the Rate Limit card: `undefined` while `/usage` + // has not answered (or failed) — the card then states `/config`'s free + // figure as it always did — `null` for a key on no usage plan, and the + // plan's figures and pill otherwise. const [usageFacts, setUsageFacts] = useState<{ quota: number | null; resetsAt: string | null; - }>({ quota: null, resetsAt: null }); + plan: PortalPlan | null | undefined; + }>({ quota: null, resetsAt: null, plan: undefined }); // ⚠️ **A revoked key replaces the whole dashboard** (Adam, 2026-08-26). // @@ -1498,10 +1506,15 @@ function Dashboard({ setUsageFacts({ quota: usage?.limit ?? null, resetsAt: usage?.resets_at ?? null, + plan: usage?.plan, }) } /> - + ); @@ -3311,10 +3324,10 @@ function ApiKey({ * 0188's, `HTTP 429` on throttle is the Rate Limit card's constant — and * neither is re-worded here, only repeated where it now bites. * - * "Contact us" is plain text under [`Legal`]'s rule, the same as the Rate Limit - * card's identical offer two columns away: there is still no commercial-plans - * destination, and this sentence is read by somebody who has just been cut off - * — the worst possible moment to hand out a link that 404s. + * "Contact us" is plain text here. Since task 0311 there IS a destination — + * the Rate Limit card's tiered contact link two columns away points at + * `RUMBLEFISH_CONTACT` — and this strip's copy is left as it was (0311 scopes + * its contact change to that card). */ function QuotaReachedNotice({ resetsAt }: { resetsAt: string }) { const at = new Date(resetsAt); @@ -3340,8 +3353,8 @@ function QuotaReachedNotice({ resetsAt }: { resetsAt: string }) { the one place that decides it, rather than naming the hex a second time here — the same treatment `HTTP 429` gets, because the frame gives the two the same weight. Still underlined and still a - `` inside it and not an ``: there is no commercial-plans - destination, which is [`Legal`]'s rule. */} + `` inside it and not an ``: the link to + `RUMBLEFISH_CONTACT` is the Rate Limit card's (task 0311). */} Contact us for higher limits. @@ -3490,7 +3503,7 @@ function Usage({ * the reset rule, which is the half the frame keeps; the lag is the half * that now goes unsaid. */ - const resetCaption = (resetsAt?: string) => { + const resetCaption = (resetsAt?: string | null) => { if (!resetsAt) return null; const at = new Date(resetsAt); if (Number.isNaN(at.getTime())) return null; @@ -3540,7 +3553,37 @@ function Usage({ ))} {view.state === 'ok' && - (view.usage.used === null ? ( + (view.usage.plan === null ? ( + // Task 0311: the key is on no usage plan for this API's stage — + // the "issued but dead" state the gateway answers `403` to. Stated, + // never rendered as zeros or as "nothing recorded yet". +

+ No usage plan, so there is no quota to count against. +

+ ) : view.usage.plan && view.usage.plan.quota_limit === null ? ( + // Task 0311: a plan without a quota. Nothing is counted against + // it, so there is no meter to draw — "Unlimited" is the figure. +

+ Unlimited — your plan has no monthly quota. +

+ ) : view.usage.limit !== null && view.usage.resets_at === null ? ( + // Task 0311: a quota whose period the backend does not compute + // (`WEEK`, or anything newer). The quota is stated; the counters + // are not, because the window they need would be a guess. + <> +
+

+ Quota per{' '} + {(view.usage.plan?.quota_period ?? 'period').toLowerCase()}; usage + for this period is not shown. +

+ + ) : view.usage.used === null ? ( <> {/* AWS has no rows for the key yet. Not zeros: inventing `remaining` and `limit` would be guessing, and the honest state @@ -3608,7 +3651,8 @@ function Usage({ be dropping the notice exactly when it is most true. */} {view.usage.limit !== null && view.usage.limit > 0 && - view.usage.used >= view.usage.limit && ( + view.usage.used >= view.usage.limit && + view.usage.resets_at !== null && ( )} @@ -3624,67 +3668,155 @@ function Usage({ } /** - * The Rate Limit card — the plan's two figures and what the gateway does when - * you cross them. + * The Rate Limit card — the key's plan, its two figures, and what the gateway + * does when you cross them. * - * The per-second figure comes from `/config` (task 0188 put it there so it - * cannot drift from what API Gateway actually enforces); the per-minute one is - * that number times sixty, computed rather than written down for the same - * reason. Both HTTP codes are properties of the gateway, not of a plan, so they - * are constants. + * **The figures are the key's OWN plan's** since task 0311: `/api/usage` + * reports the plan the key is on (`plan`), and the per-second figure is its + * `rate_limit_per_second`; the per-minute one is that times sixty, computed + * rather than written down. The plan's name rides beside "Active" as a second + * pill (`Free` … `Pro`, `Custom` for any other plan on our stage), and the + * contact line under the figures links to `RUMBLEFISH_CONTACT` with copy for + * the tier — contacting us is the only way to change plan, so it is shown on + * every tier. + * + * Three states besides that one: + * + * - **No plan** (`plan: null`): the key is on no usage plan for this API, so + * the gateway answers it `403`. No pill — "Active" would be false — and the + * state is said in words. + * - **Unlimited** (`rate_limit_per_second: null`): both figures read + * "Unlimited", never a zero. + * - **Not answered yet** (`plan` undefined — `/usage` in flight or failed): + * `/config`'s free-plan figure, as before 0311. `/config` stays the source + * for that, for the no-key state and for the landing page. * * **The card always renders**, which is a change from the build that dropped * it whenever `/config` carried no limit (Adam, 2026-08-25: "brakuje całego - * jednego kafelka"). A local run without `PORTAL_RATE_LIMIT` set was losing a - * third of the dashboard, and a missing panel is a worse answer than a stated - * one: the free plan's rate is 1 req/s (task 0157), the landing page says so - * to every visitor before they sign in, and this card now says the same where - * the deployment has not spoken. + * jednego kafelka"). A missing panel is a worse answer than a stated one: the + * free plan's rate is 1 req/s (task 0157), the landing page says so to every + * visitor before they sign in, and this card says the same where neither the + * key's plan nor the deployment has spoken. * - * ⚠️ `/config` still WINS wherever it answers, which is every deployed - * environment (`compute-stack.ts` passes `pricingApiFreePlanRateLimit` - * unconditionally). The fallback is for the local case only, and if the free - * plan's rate ever changes, this constant is one of the two places that must - * change with it — the other being `FairAccess`. + * ⚠️ `/config` WINS over this constant wherever it answers, which is every + * deployed environment (`compute-stack.ts` passes `pricingApiFreePlanRateLimit` + * unconditionally), and the key's plan wins over both. The fallback is for the + * local case only, and if the free plan's rate ever changes, this constant is + * one of the two places that must change with it — the other being + * `FairAccess`. */ const FREE_PLAN_RATE_LIMIT = 1; +/** The pill label per tier (task 0311, decision 5). */ +const PLAN_LABEL: Record = { + free: 'Free', + basic: 'Basic', + analyst: 'Analyst', + lite: 'Lite', + pro: 'Pro', + custom: 'Custom', +}; + +/** + * The contact line's copy per tier (task 0311, decision 6): the lead-in, then + * the linked sentence. + */ +const CONTACT_COPY: Record = { + free: ['Need higher limits?', 'Contact us about a paid plan.'], + basic: ['Need more?', 'Contact us to change your plan.'], + analyst: ['Need more?', 'Contact us to change your plan.'], + lite: ['Need more?', 'Contact us to change your plan.'], + pro: ['Need custom limits?', 'Contact us.'], + custom: ['Need custom limits?', 'Contact us.'], +}; + +/** + * The Rate Limit card's contact line (task 0311): a link to + * `RUMBLEFISH_CONTACT`, underlined like the OAuth card's "contact support", + * worded for the tier. It replaced a plain-text "Contact us for commercial + * plans." that had no destination to point at. + */ +function PlanContact({ tier }: { tier: PortalPlanTier }) { + const [lead, link] = CONTACT_COPY[tier]; + return ( + + {lead}{' '} + + {link} + + + ); +} + function RateLimitCard({ rateLimit, keyAbsent = false, + plan, }: { rateLimit?: number; /** * The account has no key — the frame empties this card too. See the * dashboard's `keyAbsent`. * - * ⚠️ This card's figures do NOT depend on having a key: the free plan's rate - * is the free plan's rate, and every other state renders them. Emptying it - * here is the frame's call and it is defensible — a limit stated for a - * credential that does not exist is a number with nothing to apply to, and - * the frame wants nothing competing with the one button on the card above. - * **The green "Active" pill goes with it**, and that part is not merely - * defensible but required: "Active" is a claim about a key, and there is no - * key to make it about. + * ⚠️ Emptying it here is the frame's call and it is defensible — a limit + * stated for a credential that does not exist is a number with nothing to + * apply to, and the frame wants nothing competing with the one button on + * the card above. **The green "Active" pill goes with it**, and that part is + * not merely defensible but required: "Active" is a claim about a key, and + * there is no key to make it about. */ keyAbsent?: boolean; + /** + * The key's plan, from `/api/usage` (task 0311): `undefined` until it + * answers (or when it failed), `null` for a key on no plan. + */ + plan?: PortalPlan | null; }) { - const perSecond = rateLimit ?? FREE_PLAN_RATE_LIMIT; - if (keyAbsent) return ; + if (plan === null) { + return ( + +

+ This key is not on a usage plan for this API, so the API answers 403 + to it. +

+ +
+ ); + } + + // `null` is a plan without a throttle: unlimited, stated as such. + const perSecond: number | null = plan + ? plan.rate_limit_per_second + : (rateLimit ?? FREE_PLAN_RATE_LIMIT); + return ( - +
@@ -3694,23 +3826,30 @@ function RateLimitCard({ Expanding an abbreviation is what this technique is for, and it lets 0188's wording survive a change that was purely visual. */} - Rate limit: {perSecond} request{perSecond === 1 ? '' : 's'} per second. + {perSecond === null ? ( + <>Rate limit: unlimited. + ) : ( + <> + Rate limit: {perSecond} request{perSecond === 1 ? '' : 's'} per + second. + + )} - {/* "Contact us" is plain text, not a link: there is no commercial-plans - destination to point it at, and a link to a 404 beside the words - "commercial plans" is worse than none. */} - - Need higher limits? Contact us for commercial plans. - + {/* Until `/usage` names the plan the card states the free figure, so it + makes the free plan's offer. */} + ); } -/** One big yellow number with its unit, from the Rate Limit card. */ +/** + * One big yellow number with its unit, from the Rate Limit card. A string + * value ("Unlimited", task 0311) is the whole statement and takes no unit. + */ function Figure({ label, value, @@ -3718,7 +3857,7 @@ function Figure({ testId, }: { label: string; - value: number; + value: number | string; unit: string; testId?: string; }) { @@ -3743,9 +3882,11 @@ function Figure({ > {value} - - {unit} - + {typeof value === 'number' && ( + + {unit} + + )} ); diff --git a/web/portal/src/landing/DashboardPanel.tsx b/web/portal/src/landing/DashboardPanel.tsx index fa94a723..a60e92f2 100644 --- a/web/portal/src/landing/DashboardPanel.tsx +++ b/web/portal/src/landing/DashboardPanel.tsx @@ -491,6 +491,9 @@ function RawFigure({ testId, value }: { testId: string; value: number }) { * and "Rate Limit". The status pill lives in the header beside the title, which * is the only place the design ever puts one. */ +/** One header pill: a label and a tone. */ +export type Pill = { label: string; tone: 'ok' | 'muted' | 'bad' }; + export function DashboardCard({ title, status, @@ -498,7 +501,11 @@ export function DashboardCard({ sx, }: { title: string; - status?: { label: string; tone: 'ok' | 'muted' | 'bad' }; + /** + * One pill, or several side by side — the Rate Limit card shows "Active" + * and the key's plan (task 0311). + */ + status?: Pill | readonly Pill[]; /** * Omitted for the deliberately empty tile — the `Dashboard - no key` frame * gives Monthly Usage and Rate Limit a header band over an empty body while @@ -546,7 +553,10 @@ export function DashboardCard({ {title} - {status && } + {status && + (Array.isArray(status) ? status : [status as Pill]).map((pill) => ( + + ))} Date: Thu, 24 Sep 2026 12:50:54 +0200 Subject: [PATCH 05/22] docs(lore-0311): cover the five plans and upgrading a user The tier runbook now starts from the five CDK plans and walks an operator through moving a user's key: exact-name key lookup, delete then create the plan key (seconds of 403 between, value unchanged), get-usage-plans --key-id to verify, the counter restarting at zero and a rework keeping the plan. It adds the post-merge production checklist and keeps the hand-made Custom plan procedure, which must stay on our API stage to be reported as Custom. The deploy-prep grant list names the three /usageplans grants. --- docs/runbooks/manual-api-key-tier.md | 210 +++++++++++++++++++--- docs/runbooks/portal-oauth-deploy-prep.md | 18 +- 2 files changed, 194 insertions(+), 34 deletions(-) diff --git a/docs/runbooks/manual-api-key-tier.md b/docs/runbooks/manual-api-key-tier.md index 010470a7..2b648a27 100644 --- a/docs/runbooks/manual-api-key-tier.md +++ b/docs/runbooks/manual-api-key-tier.md @@ -1,47 +1,192 @@ -# Runbook: issuing a manual higher-tier API key +# Runbook: API key tiers — the five plans, upgrading a user, Custom plans -**When:** someone needs more than the self-service limits and has agreed terms with -us out of band. There is no self-serve upgrade path and no in-app billing — by -design (task 0157, epic _Self-Service Onboarding_). +**When:** a user's key needs limits other than the free plan's — they have +agreed a paid plan, or negotiated custom limits, with us out of band. There is +no self-serve upgrade path and no in-app billing — by design (task 0157, epic +_Self-Service Onboarding_; task 0311 decision 3: an operator changes a plan by +hand in AWS). **Who:** anyone with `AdministratorAccess` on the shared AWS account -(`750702271865`, `eu-central-1`). +(`750702271865`, `eu-central-1`). The portal's own Lambda cannot do any of this: +it may list a key's plans, read usage and attach a key, but holds no `DELETE` +or `PATCH` on a plan or a plan key (`api-gateway-stack.ts`). --- -## What the default tier is +## The five plans + +Since task 0311 there are five CDK-managed usage plans, all on the production +stage, all `Period.MONTH` with offset 0. The figures come from +`infra/envs/production.json` — `pricingApiFreePlan*` for free, +`pricingApiPaidPlans` for the four paid tiers — and the plans are defined in +`infra/src/lib/stacks/api-gateway-stack.ts`. + +| Tier | AWS plan name | Quota / month | Rate | Burst | +| ------- | -------------------------------- | ------------- | -------- | ----- | +| Free | `pricing-api-free-production` | 100 000 | 1 req/s | 5 | +| Basic | `pricing-api-basic-production` | 1 000 000 | 3 req/s | 15 | +| Analyst | `pricing-api-analyst-production` | 5 000 000 | 5 req/s | 25 | +| Lite | `pricing-api-lite-production` | 20 000 000 | 10 req/s | 50 | +| Pro | `pricing-api-pro-production` | 50 000 000 | 25 req/s | 125 | + +Read the current figures from `production.json` rather than trusting this +table. **Do not hand-edit any of the five in the console** — the next deploy +reverts you, and the change is invisible in review. + +Every key the portal issues lands on free. A key belongs to exactly one usage +plan per stage, so moving a user up (or down) means moving their key from one +plan to another — the procedure below. The dashboard reads whatever plan the +key is on (`GetUsagePlans` by key) and shows that plan's pill, figures, quota +and reset; the **name** is the contract: the tier is parsed from +`pricing-api--production`, and any other plan on our stage shows as +**Custom** with its own name. -Every key issued through the portal lands on the CDK-managed plan: +--- + +## Upgrade a user + +A user asks for a paid plan and the terms are agreed out of band. Their key +keeps its value; only its plan changes. + +```bash +export AWS_PROFILE= +export AWS_REGION=eu-central-1 +export ID= +export TIER=basic # free | basic | analyst | lite | pro +``` + +**1. Find the key — by exact name.** A self-service key is named +`discord--key` (`packages/prices-api/src/portal/keys/naming.rs`, `key_name`). +`--name-query` is a **prefix** match (measured, task 0180), so the exact match +is the JMESPath filter, and `enabled` drops a revocation record: + +```bash +aws apigateway get-api-keys --name-query "discord-$ID-key" \ + --query "items[?name=='discord-$ID-key' && enabled].id" --output text +``` + +Expect exactly **one** id. None: the user has no live key (never issued, or +revoked — they issue one first). Two: a double-submit duplicate the next issue +will sweep; ask the user to open the dashboard once (a sign-in reconciles) and +look again. Then: + +```bash +export K= +``` + +**2. Find the two plans.** The key's current plan, and the target by exact name: + +```bash +aws apigateway get-usage-plans --key-id "$K" \ + --query "items[].[id,name,apiStages[0].apiId]" --output table + +FROM= + +TO=$(aws apigateway get-usage-plans \ + --query "items[?name=='pricing-api-${TIER}-production'].id | [0]" --output text) +# `--output text` prints the literal "None" for an empty result — see +# "Change or revoke a Custom key" below. Guard it rather than pass it on. +[ "$TO" = "None" ] && { echo "no plan pricing-api-${TIER}-production"; unset TO; } +echo "from ${FROM} to ${TO}" +``` + +A key may also sit on another API's plan (the partner plan, `q7sd40`, is on a +different API); only the plan on our API — `/prices/production/api-gateway-id` +— is the one to move. + +**3. Move it: delete the plan key, then create it on the target plan.** -| | | -| ----- | ----------------------------- | -| Plan | `pricing-api-free-production` | -| Rate | 1 req/s sustained | -| Burst | 5 | -| Quota | 100 000 requests/month | +```bash +aws apigateway delete-usage-plan-key --usage-plan-id "$FROM" --key-id "$K" +aws apigateway create-usage-plan-key --usage-plan-id "$TO" --key-id "$K" \ + --key-type API_KEY +``` + +Delete first, because a key is on one plan per stage: creating it on the target +while it is still on the old plan is refused. **Between the two commands the +key answers `403`** — it exists but is on no plan — so run them back to back; +it is a matter of seconds, and the data plane can take a few more to follow. +The key's **value does not change**, so the user changes nothing. + +**4. Verify.** + +```bash +aws apigateway get-usage-plans --key-id "$K" --query "items[].name" +``` -That plan is defined in `infra/src/lib/stacks/api-gateway-stack.ts` and its values -come from `infra/envs/production.json`. **Do not hand-edit it in the console** — -the next deploy reverts you, and the change is invisible in review. +Exactly the target plan (`["pricing-api-basic-production"]`), plus any plan on +another API it was already on — never two plans of ours, and never none. -## Why a manual tier is a separate plan +**5. What the user sees.** -A key belongs to exactly one usage plan per stage. Raising limits for one holder -therefore means a second plan, not a second set of numbers on the existing one. +- **The counter starts from zero.** Usage is counted per `(plan, key)` pair, so + the key's usage on the new plan begins at 0 for the rest of the month; what + it used on the old plan is not carried over. Say so if they ask why their + "used" figure dropped. +- **The dashboard shows the new plan within 60 s** — the usage cache's TTL: + the Rate Limit card's pill and figures, and the Monthly Usage card's quota + and reset date. +- **A rework keeps the plan.** "Replace my key" (task 0191) revokes the key; + the next issue after the period rolls attaches the new key to the plan the + revoked key was on, and only then deletes the revoked key (task 0311). A paid + user stays paid, and nothing needs doing here. The once-per-period rework cap + is the same on every plan. -The manual plan is deliberately **not** in CDK. The epic settles this as -"a fully manual, out-of-band process for now" with "nothing to build here". -Putting it in CDK would mean carrying a resource with no holders and inventing -config fields for numbers negotiated case by case. Revisit once there is more -than one such customer. +**Downgrade** (a paid plan lapses): the same procedure with `TIER=free`. + +**6. Record it.** Add a row to the "Issued manual keys" table at the bottom of +this file (customer, plan, key id, date, who) and commit — the move is outside +CDK and this file is its only record. + +--- + +## Post-merge production verification (Adam) + +Task 0311's "on dev" checks are a production checklist — there is no dev +environment (`infra/envs/` holds only `production.json` and `cicd.json`). + +1. Deploy Compute, then ApiGateway (the Lambda learns `PORTAL_API_ID_PARAM` / + `PORTAL_API_STAGE` in Compute; the four plans and the widened policy land in + ApiGateway). In the ApiGateway diff the free plan, its key and its SSM + parameter must show **no change** — only the four new plans and the policy. +2. `/config` answers `enabled: true` (the portal did not close at cold start). +3. Move a test key free → Basic → Pro → free with the procedure above. After + each step: + - `get-usage-plans --key-id` shows exactly the target plan; + - the gateway throttles at that plan's rate (a burst above it gets `429`); + - within 60 s both dashboard cards show that plan's figures, and the Rate + Limit card the plan's pill. +4. A rework on Basic: revoke the test key while on Basic, wait for the next + period (or use a key revoked last period), issue — the new key is on + `pricing-api-basic-production`, and the revoked key is gone. +5. A free key's dashboard looks as it did before, apart from the `Free` pill + and the contact link. + +--- + +## Custom / Enterprise plans (hand-made) + +Negotiated limits that match none of the five tiers — an Enterprise customer, a +load test — are still a plan made by hand, deliberately **not** in CDK: its +numbers are case by case, and carrying them as config would mean inventing a +field per customer. The trade-off is real and worth stating: a hand-made plan is **drift**. It does not appear in `cdk diff`, nobody reviews it, and it survives only as long as someone remembers it exists. That is why step 5 below is not optional. ---- +⚠️ **A hand-made plan must be attached to our API stage to be reported as +Custom.** The dashboard keeps only plans whose `apiStages` contain our API id +and `production`; a plan on no stage, or on another API, is not the key's plan +as far as the portal is concerned — the dashboard says the key is on **no +plan**, and the gateway answers the key `403`. Step 1 below attaches the stage; +keep it that way for as long as a key is on the plan. + +A self-service `discord-` key moved onto a Custom plan uses the "Upgrade a +user" procedure with the Custom plan's id as `TO`. The rest of this section +issues a separate, hand-made key. -## Issue the key +### Issue a Custom key Set the negotiated limits first — these are examples, not defaults: @@ -153,7 +298,7 @@ key has to be disabled by hand ("Suspend without destroying", below). --- -## Change or revoke a manual key +### Change or revoke a Custom key Everything below runs weeks or months after the key was issued, in a shell that has none of step 1's variables. `PLAN_ID` and `KEY_ID` come from the registry @@ -324,8 +469,17 @@ Then delete the row from the table below. - **Never attach a manual key to `pricing-api-free-production`.** It would silently inherit 1 req/s and a 100 000/month quota. A key belongs to exactly one usage plan **per stage** — so on this API there is no "also attach it to - the bigger plan". (A key may sit in up to 10 plans overall, across different + the bigger plan", and moving a key is delete-then-create (with seconds of + `403` between). (A key may sit in up to 10 plans overall, across different stages; that does not help here.) +- **Do not delete a Custom plan out from under a revoked key.** The portal + reads a rework's target plan off the revoked key (task 0311); if that key's + plan is gone, the new key lands on free — a silent downgrade of exactly the + kind 0311 removed. Wind a customer down only once nothing of theirs is + revoked-and-waiting. +- **Rename none of the five CDK plans by hand.** The tier is parsed from the + exact name `pricing-api--production`; a renamed plan shows as Custom on + the dashboard (and the next deploy renames it back). - **Nothing stops you creating two keys with the same name.** Only key _values_ are unique in API Gateway; `name` is optional and duplicable, and there is no documented way to ask "which of these is the current one". That is why step 2 diff --git a/docs/runbooks/portal-oauth-deploy-prep.md b/docs/runbooks/portal-oauth-deploy-prep.md index 95053a3a..7a94cead 100644 --- a/docs/runbooks/portal-oauth-deploy-prep.md +++ b/docs/runbooks/portal-oauth-deploy-prep.md @@ -450,13 +450,19 @@ any behaviour until the flag moves. ### The IAM, and the three limits that come with it -CDK grants the api-handler role seven control-plane actions and nothing else: +CDK grants the api-handler role eight control-plane actions and nothing else: `GET`/`POST` on `/apikeys`, `GET`/`PATCH`/`DELETE` on `/apikeys/*` (`PATCH` -is task 0191's revoke — see below), `POST` on -`/usageplans/{the free plan}/keys`, and — task 0188 — `GET` on -`/usageplans/{the free plan}/usage` (`GetUsage`, the dashboard's usage read). -The last two are declared in `api-gateway-stack.ts` rather than -`compute-stack.ts`, because that is the only stack that knows the plan id. +is task 0191's revoke — see below), and three on `/usageplans` since task +0311: `GET /usageplans` (`GetUsagePlans` by key — which plan a key is on), +`GET /usageplans/*/usage` (`GetUsage` on the key's own plan, the dashboard's +usage read) and `POST /usageplans/*/keys` (attaching a key to any plan — the +free plan on a first issue, the previous key's plan on a rework). The three +`/usageplans` grants are declared in `api-gateway-stack.ts` rather than +`compute-stack.ts`, because that is where the policy has always lived — +moving it is a delete in one stack and a create in the other, with a window in +which key issuance breaks. No `DELETE` or `PATCH` on a plan or a plan key, and +no `GET /usageplans/{id}`: moving a key between plans is an operator's job +(`manual-api-key-tier.md`). Two of the six cannot be scoped any further, and one can but is not yet. All three are written out in full in `compute-stack.ts`; the short version: From f0ef44943c478befe926992ea64c1af758f939fe Mon Sep 17 00:00:00 2001 From: Adam <65679285+adamkoot@users.noreply.github.com> Date: Thu, 24 Sep 2026 14:17:05 +0200 Subject: [PATCH 06/22] fix(lore-0311): grant the usage-plan calls on the handler role The new handler calls GetUsagePlans on every sign-in and usage load, but the grants sat in ApiGateway's policy, which deploys after Compute: until then every sign-in failed and /api/usage answered 502. The three grants (GET /usageplans, GET /usageplans/*/usage, POST /usageplans/*/keys) now live on the api-handler role's own policy, which the function depends on, so they land before the code. ApiGateway's PortalAttachKeyToFreePlan policy is removed. Compute keeps exporting the role name for one release, because the deployed ApiGateway policy still imports it. --- docs/runbooks/portal-oauth-deploy-prep.md | 15 +-- infra/src/lib/app.ts | 4 - infra/src/lib/stacks/api-gateway-stack.ts | 98 ++---------------- infra/src/lib/stacks/compute-stack.ts | 115 ++++++++++++++++++---- 4 files changed, 115 insertions(+), 117 deletions(-) diff --git a/docs/runbooks/portal-oauth-deploy-prep.md b/docs/runbooks/portal-oauth-deploy-prep.md index 7a94cead..7058ebf7 100644 --- a/docs/runbooks/portal-oauth-deploy-prep.md +++ b/docs/runbooks/portal-oauth-deploy-prep.md @@ -456,13 +456,14 @@ is task 0191's revoke — see below), and three on `/usageplans` since task 0311: `GET /usageplans` (`GetUsagePlans` by key — which plan a key is on), `GET /usageplans/*/usage` (`GetUsage` on the key's own plan, the dashboard's usage read) and `POST /usageplans/*/keys` (attaching a key to any plan — the -free plan on a first issue, the previous key's plan on a rework). The three -`/usageplans` grants are declared in `api-gateway-stack.ts` rather than -`compute-stack.ts`, because that is where the policy has always lived — -moving it is a delete in one stack and a create in the other, with a window in -which key issuance breaks. No `DELETE` or `PATCH` on a plan or a plan key, and -no `GET /usageplans/{id}`: moving a key between plans is an operator's job -(`manual-api-key-tier.md`). +free plan on a first issue, the previous key's plan on a rework). All eight +are on the role's own policy in `compute-stack.ts`. The `/usageplans` grants +used to be a standalone policy in `api-gateway-stack.ts`, because they named +the free plan's id; since task 0311 they name no id and moved to the stack +that ships the code using them, which deploys first — so the grant is in place +before the new handler runs (`manual-api-key-tier.md`, "Rolling this out"). +No `DELETE` or `PATCH` on a plan or a plan key, and no `GET /usageplans/{id}`: +moving a key between plans is an operator's job (`manual-api-key-tier.md`). Two of the six cannot be scoped any further, and one can but is not yet. All three are written out in full in `compute-stack.ts`; the short version: diff --git a/infra/src/lib/app.ts b/infra/src/lib/app.ts index e8d36075..f36bfeba 100644 --- a/infra/src/lib/app.ts +++ b/infra/src/lib/app.ts @@ -39,10 +39,6 @@ export function createApp({ config }: CreateAppOptions): void { env, config, apiHandlerFunction: compute.apiHandlerFunction, - // The role, so ApiGatewayStack can grant the one control-plane action that - // needs the usage-plan id (task 0187). Same direction as the Function - // above, so it adds no new dependency and cannot create a cycle. - apiHandlerRole: compute.apiHandlerRole, }); // No hosting stack for the portal. `PortalHostingStack` (task 0184) — a diff --git a/infra/src/lib/stacks/api-gateway-stack.ts b/infra/src/lib/stacks/api-gateway-stack.ts index d58c3352..9c1b0309 100644 --- a/infra/src/lib/stacks/api-gateway-stack.ts +++ b/infra/src/lib/stacks/api-gateway-stack.ts @@ -2,7 +2,6 @@ import * as cdk from 'aws-cdk-lib'; import * as apigateway from 'aws-cdk-lib/aws-apigateway'; import * as acm from 'aws-cdk-lib/aws-certificatemanager'; import type * as lambda from 'aws-cdk-lib/aws-lambda'; -import * as iam from 'aws-cdk-lib/aws-iam'; import * as route53 from 'aws-cdk-lib/aws-route53'; import * as targets from 'aws-cdk-lib/aws-route53-targets'; import * as ssm from 'aws-cdk-lib/aws-ssm'; @@ -30,21 +29,6 @@ export interface ApiGatewayStackProps extends cdk.StackProps { * Passed in from `ComputeStack` (cross-stack reference). */ readonly apiHandlerFunction: lambda.IFunction; - /** - * The api-handler's execution role, so the one control-plane grant that needs - * the usage-plan id can be declared here (task 0187). - * - * The other four `apigateway:` grants live in `ComputeStack`, on resources - * that need no id from this stack. This one cannot: `usagePlan.usagePlanId` - * is created below, and a policy in `ComputeStack` referencing it would make - * ComputeStack import an export of ApiGatewayStack — while ApiGatewayStack - * already imports the Lambda from ComputeStack. That is a cycle, and - * CloudFormation refuses it. - * - * Declaring the policy HERE and attaching it to the passed-in role keeps the - * reference pointing the one way that works: ApiGateway -> Compute. - */ - readonly apiHandlerRole: iam.IRole; } /** @@ -372,7 +356,7 @@ export class ApiGatewayStack extends cdk.Stack { constructor(scope: Construct, id: string, props: ApiGatewayStackProps) { super(scope, id, props); - const { config, apiHandlerFunction, apiHandlerRole } = props; + const { config, apiHandlerFunction } = props; const cacheEnabled = config.apiGatewayCacheEnabled; this.api = new apigateway.RestApi(this, 'Api', { @@ -951,77 +935,15 @@ export class ApiGatewayStack extends cdk.Stack { description: `Usage plan ID for pricing-api-free-${config.envName} (key issuance + GetUsage)`, }); - // The portal's `/usageplans` grants (tasks 0187, 0188, widened by 0311). - // Declared here rather than in `ComputeStack`, beside their five siblings - // there, for a reason that outlived the one it started with. - // - // It started as the cycle: the grants named the free plan's id, and - // `iam.Policy` rather than `apiHandlerRole.addToPrincipalPolicy` because - // the latter appends to the role's default policy — a resource of - // ComputeStack — so the plan id would have travelled as an export of THIS - // stack imported by that one. Since task 0311 no statement references the - // plan id, but the policy STAYS here with its construct id and policyName: - // moving it to ComputeStack is a delete in one stack and a create in - // another, with a window in which the Lambda can attach no key and key - // issuance breaks. Renaming it is a replacement for the same cosmetic gain. - // - // Three statements, and they are the whole set. Task 0188's decision 1 was - // "one plan's ARN, nothing wider"; task 0311 widens it deliberately, - // because the dashboard must state the key's OWN plan and the usage counted - // on it, and a rework must keep a paid user on their paid plan: - // - // - `GET /usageplans` is `GetUsagePlans?keyId=` — which plans hold this - // key. The keyId filter is a query parameter, not a resource, so this - // cannot be scoped below the collection. Read-only; it reveals plan - // names and limits, never another key. - // - `GET /usageplans/*/usage` is `GetUsage` on whichever plan the key is - // on — paid, free or hand-made. The usage sub-resource still does NOT - // permit reading a plan itself, listing its keys or changing it. - // - `POST /usageplans/*/keys` attaches a key. The code only ever attaches - // to the free plan, the key's own plan, or the previous (revoked) key's - // plan — and only one that `GetUsagePlans` reported on OUR API stage. - // Hand-made plans have no ARN known at synth, hence the wildcard. - // - // Deliberately NOT granted: - // - `GET /usageplans/{id}` to validate the plan at cold start. 0187's - // decision 22 rejected cold-start validation (a warm container still - // misses a plan that changes under it, and the attach path - // disambiguates a dead plan id into `PlanNotFound` loudly), and - // `GetUsagePlans` already returns every figure the dashboard shows. - // - `DELETE`/`PATCH` on a plan or a plan key. The code never moves a key - // between plans or changes limits; an operator does, by hand. - // - `GET /usageplans/*/keys`. Nothing lists a plan's members. - // - // A `sid` on every statement is load-bearing: cdk.json enables - // `@aws-cdk/aws-iam:minimizePolicies`, which merges sid-less statements — - // the two GETs would collapse into one and the set would stop reading as - // three. Task 0194's audit should read this policy as "the portal's - // `/usageplans` grants", whatever the name says. - new iam.Policy(this, 'PortalAttachKeyToFreePlan', { - policyName: `prices-${config.envName}-portal-attach-key`, - roles: [apiHandlerRole], - statements: [ - new iam.PolicyStatement({ - sid: 'PortalListUsagePlansByKey', - actions: ['apigateway:GET'], - resources: [`arn:aws:apigateway:${config.awsRegion}::/usageplans`], - }), - new iam.PolicyStatement({ - sid: 'PortalReadAnyPlanUsage', - actions: ['apigateway:GET'], - resources: [ - `arn:aws:apigateway:${config.awsRegion}::/usageplans/*/usage`, - ], - }), - new iam.PolicyStatement({ - sid: 'PortalAttachKeyToAnyPlan', - actions: ['apigateway:POST'], - resources: [ - `arn:aws:apigateway:${config.awsRegion}::/usageplans/*/keys`, - ], - }), - ], - }); + // The portal's `/usageplans` grants (tasks 0187, 0188, widened by 0311) + // used to be a standalone `iam.Policy` here, `PortalAttachKeyToFreePlan`, + // because they named the free plan's id and importing that id into + // ComputeStack would have closed a cycle. Since task 0311 they name no id + // (`/usageplans`, `/usageplans/*/usage`, `/usageplans/*/keys`), so they + // live on the api-handler role's own policy in `ComputeStack` — the stack + // that ships the code using them, and deploys FIRST: the grants reach IAM + // in the same CloudFormation update as the Lambda that needs them, never + // after it. See the control-plane section of `compute-stack.ts`. // CORS on the gateway's OWN error answers (task 0194): a `429` from the // throttle above, a `504` when the Lambda runs out of time. Neither diff --git a/infra/src/lib/stacks/compute-stack.ts b/infra/src/lib/stacks/compute-stack.ts index d99eed17..6b1fe52f 100644 --- a/infra/src/lib/stacks/compute-stack.ts +++ b/infra/src/lib/stacks/compute-stack.ts @@ -521,16 +521,24 @@ export class ComputeStack extends cdk.Stack { // Self-service API keys (task 0187) — API Gateway CONTROL plane. // --------------------------------------------------------------- // - // Five of the eight calls the portal makes. The other three are the - // `/usageplans` grants — `GET /usageplans` (task 0311's `GetUsagePlans` - // by key), `GET /usageplans/*/usage` (`GetUsage` on the key's own plan) - // and `POST /usageplans/*/keys` (the attach) — and they are granted in - // `ApiGatewayStack`'s standalone policy. They once named the free plan's - // id, which lives there (importing it here would close the Compute -> - // Gateway -> Compute cycle described on `portalFreePlanParameterName` - // above); since task 0311 they name no id, and the policy stays there - // because moving it is a delete+create with a window in which key - // issuance breaks. Task 0194 audits the set as one policy. + // All eight calls the portal makes: five on `/apikeys`, and the three + // `/usageplans` grants at the end of this section — `GET /usageplans` + // (task 0311's `GetUsagePlans` by key), `GET /usageplans/*/usage` + // (`GetUsage` on the key's own plan) and `POST /usageplans/*/keys` (the + // attach). Task 0194 audits the set as one policy. + // + // The `/usageplans` three used to live in `ApiGatewayStack`, as the + // standalone policy `PortalAttachKeyToFreePlan`, because they named the + // free plan's id and importing it here would close the Compute -> Gateway + // -> Compute cycle described on `portalFreePlanParameterName` above. Since + // task 0311 they name no id, and they MUST be here: the new handler calls + // `GetUsagePlans` on every sign-in and every dashboard load, and this stack + // deploys FIRST (the Makefile's cross-stack rule, and `deploy --all`'s own + // order). A grant in `ApiGatewayStack` would reach IAM only after the code + // that needs it — every sign-in and dashboard broken for the whole + // ApiGateway deploy, and for good if that deploy rolled back (review + // CR-01). Here, the role's default policy and the Function are one + // CloudFormation update, and CDK makes the Function depend on the policy. // // Control-plane ARNs carry no account id — `arn:aws:apigateway:::` // with a doubled colon — and the resource is the API's own path. @@ -617,10 +625,9 @@ export class ComputeStack extends cdk.Stack { // this is again exposure under code execution, not feature behaviour. // // What is deliberately NOT here: `apigateway:*`, `PUT /tags/*` on anything - // but API keys, and any grant on `/usageplans` — the three this feature - // takes (list by key, any plan's usage, attach to any plan) live in - // `ApiGatewayStack`'s standalone policy, which also states what is not - // granted there (`GET /usageplans/{id}`, `DELETE`, `PATCH`). + // but API keys, and any grant on `/usageplans` beyond the three below + // (list by key, any plan's usage, attach to any plan) — each states there + // what it does not reach (`GET /usageplans/{id}`, `DELETE`, `PATCH`). // // `DELETE` **is** here, and it is this slice's: the reconciler removes // duplicate keys after a double-submit ("keep the earliest createdDate, @@ -701,6 +708,62 @@ export class ComputeStack extends cdk.Stack { }), ); + // The portal's `/usageplans` grants (tasks 0187, 0188, widened by 0311). + // Three statements, and they are the whole set. Task 0188's decision 1 was + // "one plan's ARN, nothing wider"; task 0311 widens it deliberately, + // because the dashboard must state the key's OWN plan and the usage counted + // on it, and a rework must keep a paid user on their paid plan: + // + // - `GET /usageplans` is `GetUsagePlans?keyId=` — which plans hold this + // key. The keyId filter is a query parameter, not a resource, so this + // cannot be scoped below the collection. Read-only; it reveals plan + // names and limits, never another key. + // - `GET /usageplans/*/usage` is `GetUsage` on whichever plan the key is + // on — paid, free or hand-made. The usage sub-resource does NOT permit + // reading a plan itself, listing its keys or changing it. + // - `POST /usageplans/*/keys` attaches a key. The code only ever attaches + // to the free plan or to a previous (revoked) key's plan — and only one + // that `GetUsagePlans` reported on OUR API stage (`resolve_target_plan` + // in `portal/keys/mod.rs`). Hand-made plans have no ARN known at synth, + // hence the wildcard. + // + // Deliberately NOT granted: + // - `GET /usageplans/{id}` to validate the plan at cold start. 0187's + // decision 22 rejected cold-start validation (a warm container still + // misses a plan that changes under it, and the attach path + // disambiguates a dead plan id into `PlanNotFound` loudly), and + // `GetUsagePlans` already returns every figure the dashboard shows. + // - `DELETE`/`PATCH` on a plan or a plan key. The code never moves a key + // between plans or changes limits; an operator does, by hand + // (docs/runbooks/manual-api-key-tier.md). + // - `GET /usageplans/*/keys`. Nothing lists a plan's members. + // + // A `sid` on every statement is load-bearing: cdk.json enables + // `@aws-cdk/aws-iam:minimizePolicies`, which merges sid-less statements — + // the two GETs would collapse into one and the set would stop reading as + // three. + this.apiHandlerRole.addToPrincipalPolicy( + new iam.PolicyStatement({ + sid: 'PortalListUsagePlansByKey', + actions: ['apigateway:GET'], + resources: [`arn:aws:apigateway:${awsRegion}::/usageplans`], + }), + ); + this.apiHandlerRole.addToPrincipalPolicy( + new iam.PolicyStatement({ + sid: 'PortalReadAnyPlanUsage', + actions: ['apigateway:GET'], + resources: [`arn:aws:apigateway:${awsRegion}::/usageplans/*/usage`], + }), + ); + this.apiHandlerRole.addToPrincipalPolicy( + new iam.PolicyStatement({ + sid: 'PortalAttachKeyToAnyPlan', + actions: ['apigateway:POST'], + resources: [`arn:aws:apigateway:${awsRegion}::/usageplans/*/keys`], + }), + ); + // The usage-plan id, read at cold start. // // **Currently redundant, and kept deliberately.** The baseline role already @@ -842,7 +905,9 @@ export class ComputeStack extends cdk.Stack { // The same deploy-order caveat as read 2: `ApiGatewayStack` // publishes it and deploys AFTER this stack. It has existed since // the gateway's first deploy, so this bites only a fresh - // environment — where read 2 already does + // environment — where read 2 already closes the portal until + // `ApiGatewayStack` has deployed once, so this read adds no new + // failure there. // // A failed read CLOSES the portal in that execution environment and // logs `portal closed at cold start` on the api-handler; it does NOT @@ -912,9 +977,11 @@ export class ComputeStack extends cdk.Stack { // Since task 0311 the signed-in dashboard no longer states this: it // reads the key's OWN plan through `GetUsagePlans` (0311 took the // grant 0188 had declined) and shows that plan's figures. This stays - // for what has no key to ask about — the no-key state, the landing - // page, and the fallback while the usage call is unanswered. A - // literal in the bundle would go stale the moment the limit changed. + // for what has no key to ask about — the no-key state and the landing + // page. While the usage call is unanswered (or failed) the dashboard + // states no figure at all rather than this one, which a paid key + // would read as its own (task 0311's review, WR-01). A literal in the + // bundle would go stale the moment the limit changed. // Not a secret, and not conditional on `PORTAL_ENABLED` — same // one-word-diff reasoning as the two names above. PORTAL_RATE_LIMIT: String(config.pricingApiFreePlanRateLimit), @@ -964,6 +1031,18 @@ export class ComputeStack extends cdk.Stack { value: this.ingestDlq.queueUrl, description: `Prices ingest DLQ URL (${envName})`, }); + // ⚠️ Keeps the api-handler role's name exported although, since task + // 0311, nothing in this app imports it. `ApiGatewayStack`'s deployed + // template still does — its old `PortalAttachKeyToFreePlan` policy names + // the role through `Fn::ImportValue` — and this stack deploys FIRST. + // Without this line CDK would drop the auto-generated export, and + // CloudFormation refuses to delete an export another stack still imports: + // the Compute deploy would fail before ApiGateway ever removed the policy. + // The export name is the one CDK generated for the automatic reference, + // so the template does not change here. Remove it one release after + // `ApiGatewayStack` has deployed without the policy. + this.exportValue(this.apiHandlerRole.roleName); + new cdk.CfnOutput(this, 'ApiHandlerRoleArn', { value: this.apiHandlerRole.roleArn, description: `API Handler Lambda execution role ARN (${envName})`, From e772c6af4a45336a063a11340d8114972e868d14 Mon Sep 17 00:00:00 2001 From: Adam <65679285+adamkoot@users.noreply.github.com> Date: Thu, 24 Sep 2026 14:17:05 +0200 Subject: [PATCH 07/22] fix(lore-0311): keep the plan across a rework and a refused attach A rework read the plan off the latest revoked record only, so an older sibling on a paid plan was ignored and the new key landed on free. The target plan now comes from the newest revoked record on a non-free plan for our stage, else free. CreateUsagePlanKey on a key already on another plan for the same stage answers 409 or a 400 naming the "same API Stage". Both now mean "already on a plan": the plan is read back with plan_of, the key is never put on free, and a plan that never shows ends in ?issue=failed. The mock control plane answers both and can lag. --- .../prices-api/src/portal/keys/gateway.rs | 77 +++++- packages/prices-api/src/portal/keys/mod.rs | 133 +++++++-- packages/prices-api/src/portal/keys/naming.rs | 76 +++--- packages/prices-api/tests/portal_issue.rs | 45 +++ .../prices-api/tests/portal_keys/harness.rs | 75 ++++- packages/prices-api/tests/portal_rework.rs | 257 ++++++++++++++++++ 6 files changed, 580 insertions(+), 83 deletions(-) diff --git a/packages/prices-api/src/portal/keys/gateway.rs b/packages/prices-api/src/portal/keys/gateway.rs index 3b002369..dbb2ee86 100644 --- a/packages/prices-api/src/portal/keys/gateway.rs +++ b/packages/prices-api/src/portal/keys/gateway.rs @@ -58,6 +58,7 @@ use std::time::Duration; use aws_sdk_apigateway::Client; use aws_sdk_apigateway::config::timeout::TimeoutConfig; +use aws_sdk_apigateway::error::ProvideErrorMetadata; use aws_sdk_apigateway::types::UsagePlan; use serde::Serialize; @@ -123,18 +124,45 @@ impl std::fmt::Debug for KeyValue { /// What [`Gateway::attach_to_plan`] observed. /// -/// Two outcomes rather than `()` because the caller has to act on the second -/// one: a key that vanished between the listing and the attach is the same race -/// [`Gateway::value_of`] reports with `None`, and the answer to it is to run the -/// reconciliation again, not to tell the caller the control plane is broken. +/// Three outcomes rather than `()` because the caller has to act on the last +/// two: a key that vanished between the listing and the attach is the same +/// race [`Gateway::value_of`] reports with `None`, and the answer to it is to +/// run the reconciliation again, not to tell the caller the control plane is +/// broken; and a key already on a plan is on SOME plan, which the caller has +/// to find out rather than assume (task 0311). #[derive(Debug, Clone, Copy, PartialEq, Eq)] pub enum Attachment { - /// The key is on the usage plan — this call put it there, or it already was. + /// This call put the key on the usage plan it was given. OnPlan, + /// The key was already on a usage plan for this API stage, so AWS refused + /// the attach (task 0311). Two refusals land here: + /// + /// - `409 ConflictException` — by AWS's own wording "already exists in the + /// usage plan", i.e. THIS plan; + /// - `400 BadRequestException` "… cannot reference multiple Usage Plans + /// with the same API Stage" — ANOTHER plan on the same stage, since a + /// key belongs to one plan per stage. + /// + /// Neither status code is trusted to say which plan: the caller asks + /// [`Gateway::plan_of`]. A key a concurrent issue, or an operator, put on + /// a plan a moment ago is a working key, not a failure. + AlreadyOnAPlan, /// The key no longer exists, so there was nothing to attach. KeyGone, } +/// Whether a `400 BadRequestException` from `CreateUsagePlanKey` is the +/// "already on another plan for this stage" refusal (task 0311), rather than +/// any other malformed request — which stays an error. +/// +/// Matched on the message, because AWS gives this case no error type of its +/// own. The phrase is AWS's ("API Key … cannot reference multiple Usage Plans +/// with the same API Stage: :"); matched case-insensitively and +/// on its distinctive tail only, so a reworded prefix still lands here. +fn is_same_stage_refusal(message: Option<&str>) -> bool { + message.is_some_and(|m| m.to_ascii_lowercase().contains("same api stage")) +} + /// What [`Gateway::disable`] observed (task 0191). /// /// Two outcomes rather than `()` for the same reason as [`Attachment`]: the @@ -233,8 +261,9 @@ pub enum GatewayError { /// Parsed from the plan's NAME, because the name is the one contract CDK and /// this code share: `pricing-api--` for the five CDK plans /// (`api-gateway-stack.ts`). Anything else on our stage — the loadtest plan, a -/// hand-made Enterprise plan — is [`Tier::Custom`] and is shown with its own -/// name. +/// hand-made Enterprise plan — is [`Tier::Custom`]. The dashboard labels every +/// one of those `Custom`; the plan's own name still travels in `plan.name` on +/// `/api/usage`, for a support conversation, but is not rendered. #[derive(Debug, Clone, Copy, PartialEq, Eq, Serialize)] #[serde(rename_all = "lowercase")] pub enum Tier { @@ -760,11 +789,11 @@ impl Gateway { /// **Idempotent**, and that is what lets the caller run it on every key it /// is about to hand out rather than only on keys it just created. API /// Gateway answers `409 ConflictException` when the key is already on the - /// plan; that is the desired state, so it is a success. Without this the - /// caller would need to *know* whether a key is attached, and there is no - /// cheap way to know: `GetApiKey` does not report usage-plan membership, and - /// asking `GetUsagePlanKeys` would be an extra call and an extra IAM grant - /// to learn what this call can simply assert. + /// plan, and `400 BadRequestException` ("cannot reference multiple Usage + /// Plans with the same API Stage") when it is already on ANOTHER plan for + /// the same stage. Neither is a failure: both are + /// [`Attachment::AlreadyOnAPlan`], and the caller asks [`Self::plan_of`] + /// which plan that is (task 0311). Any other `400` is still an error. /// /// A `404` is **ambiguous** and is resolved before it is acted on. API /// Gateway answers `NotFoundException` both when the key is gone and when @@ -793,8 +822,11 @@ impl Gateway { Err(e) => { let message = sdk_message(&e); let service_error = e.into_service_error(); - if service_error.is_conflict_exception() { - Ok(Attachment::OnPlan) + if service_error.is_conflict_exception() + || (service_error.is_bad_request_exception() + && is_same_stage_refusal(service_error.message())) + { + Ok(Attachment::AlreadyOnAPlan) } else if service_error.is_not_found_exception() { // `NotFoundException` here means EITHER "that key is gone" // OR "that usage plan does not exist", and the error carries @@ -1232,6 +1264,23 @@ mod tests { .build() } + /// AWS's "already on another plan for this stage" refusal is recognised + /// by its wording; any other `400`, and no message at all, is not + /// (task 0311). + #[test] + fn only_the_same_stage_refusal_is_already_on_a_plan() { + assert!(is_same_stage_refusal(Some( + "API Key abc123 cannot reference multiple Usage Plans with the same API Stage: \ + 02mabge71l:production" + ))); + assert!(is_same_stage_refusal(Some( + "… WITH THE SAME API STAGE: x:y" + ))); + assert!(!is_same_stage_refusal(Some("Bad Request"))); + assert!(!is_same_stage_refusal(Some("Invalid key type"))); + assert!(!is_same_stage_refusal(None)); + } + /// The five CDK names parse to their tiers on the matching stage. #[test] fn each_cdk_plan_name_is_its_tier() { diff --git a/packages/prices-api/src/portal/keys/mod.rs b/packages/prices-api/src/portal/keys/mod.rs index fe05fdc3..191f836b 100644 --- a/packages/prices-api/src/portal/keys/mod.rs +++ b/packages/prices-api/src/portal/keys/mod.rs @@ -112,8 +112,8 @@ use super::period::Period; use cap::Cap; use gateway::{Attachment, Disable, Gateway, GatewayError, KeyValue}; use naming::{ - KeyRecord, choose_winner, current_key, exact_matches, key_name, latest_revoked, losers, - revocation_instant, + KeyRecord, choose_winner, current_key, exact_matches, key_name, losers, revocation_instant, + revoked_newest_first, }; /// The reveal, on both verbs — see [`key`] for why `POST` answers identically. @@ -389,8 +389,9 @@ async fn reveal(state: &KeysState, headers: &HeaderMap) -> Response { } = cap::decide(revoked_at, &Period::now()) else { // Revoked in an earlier period: a new key is due, and the - // issue round-trip will delete this one and create it. Until - // that press the honest answer is still "no usable key". + // issue round-trip will create it, attach it to this key's + // plan and only then delete this one (task 0311). Until that + // press the honest answer is still "no usable key". return no_store(no_key_response()); }; no_store( @@ -931,11 +932,14 @@ pub(crate) async fn issue_for(gateway: &Gateway, sub: &str, deadline: Duration) ); IssueOutcome::Failed } - // Every attempt found a key and then lost it before reading its value. + // Every attempt found a key and then lost it before reading its value + // — or, since task 0311, could not confirm the plan AWS said it was + // already on (see `attach`). Ok(Ok(Reconciled::Lost)) => { tracing::warn!( attempts = MAX_ATTEMPTS, - "a key was deleted underneath every issue attempt" + "every issue attempt lost its key: deleted underneath it, or on a usage plan \ + GetUsagePlans could not name yet" ); IssueOutcome::Failed } @@ -995,9 +999,10 @@ enum Attempt { /// The least time a create is started with. `CreateApiKey` and /// `CreateUsagePlanKey` each get up to `gateway::OPERATION_TIMEOUT` (5s) in /// the worst case; in practice each is a few hundred milliseconds. Since task -/// 0311 one or two `GetUsagePlans` reads (the new key's plan, then the -/// previous key's — see [`resolve_target_plan`]) run between them, each as -/// cheap as the attach. 4s is still enough for all of it at ordinary latency +/// 0311 a few `GetUsagePlans` reads (the new key's plan, then one per revoked +/// record until a paid plan turns up — see [`resolve_target_plan`]; one more +/// if the attach is refused as already done, see [`attach`]) run between them, +/// each as cheap as the attach. 4s is still enough for all of it at ordinary latency /// and refuses to start the create when the invocation is about to be killed /// — which is the one way this flow can leave an enabled, unattached key /// behind. And if it does anyway, the revoked record is still there (it is @@ -1125,7 +1130,7 @@ async fn attempt( // The same target plan as Step 4 (task 0311): a rework's new key // goes onto the previous key's plan, not onto free by default. if let Some(plan_id) = resolve_target_plan(gateway, &record.id, &revoked).await? - && gateway.attach_to_plan(&record.id, &plan_id).await? == Attachment::KeyGone + && attach(gateway, &record.id, &plan_id).await? == Settled::Retry { return Ok(Attempt::Retry); } @@ -1195,7 +1200,9 @@ async fn attempt( // not be "corrected" to free, and a `409` on free used to hide that only by // accident — else the plan the previous (revoked) key was on, so a rework // keeps a paid or custom user on their plan, else free. The attach itself - // is idempotent (`Gateway::attach_to_plan` treats the `409` as success). + // is idempotent: a key AWS says is already on a plan for our stage (`409`, + // or the `400` "multiple Usage Plans with the same API Stage") is settled + // by asking which plan that is — see [`attach`]. // Before the deletions below rather than after: the key the caller is // about to receive is made usable first, and the destructive half of // reconciliation only runs once that has succeeded — which since 0311 also @@ -1209,7 +1216,7 @@ async fn attempt( // place that race can surface, so it has to answer it rather than turn a // hand-deleted key back into the dead end this slice exists to remove. if let Some(plan_id) = resolve_target_plan(gateway, &winner.id, &revoked).await? - && gateway.attach_to_plan(&winner.id, &plan_id).await? == Attachment::KeyGone + && attach(gateway, &winner.id, &plan_id).await? == Settled::Retry { return Ok(Attempt::Retry); } @@ -1282,22 +1289,89 @@ async fn attempt( } } +/// Whether an [`attach`] left the key usable, or the attempt has to re-run. +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +enum Settled { + /// The key is on a usage plan for our stage. + Ready, + /// The key is gone, or AWS says it is on a plan `GetUsagePlans` cannot + /// name yet: re-enter the flow, which lists and resolves again. + Retry, +} + +/// Attach `key_id` to `plan_id`, and settle what AWS answered (task 0311). +/// +/// [`Attachment::AlreadyOnAPlan`] is the interesting case. A `409` (this +/// plan) or the `400` "cannot reference multiple Usage Plans with the same +/// API Stage" (another plan) means the key is ALREADY usable — some writer got +/// there first: the other half of a double-submit, a sign-in that ran inside +/// an operator's delete→create gap, or the operator. Which plan it is, is +/// asked of [`Gateway::plan_of`] rather than inferred from the status code: +/// +/// - on a plan for our stage → [`Settled::Ready`]; if that is not the plan +/// this attempt wanted, the key is left where it is (this code has no grant +/// to move a key between plans, by design) and the disagreement is logged +/// as a warning — for an operator, that is the runbook's "verify" step +/// failing, with the recovery written there; +/// - on no plan `GetUsagePlans` will name yet (its listing lags the refusal) +/// → [`Settled::Retry`]: handing out a key whose plan cannot be confirmed +/// is how a key that answers `403` gets revealed. +async fn attach(gateway: &Gateway, key_id: &str, plan_id: &str) -> Result { + match gateway.attach_to_plan(key_id, plan_id).await? { + Attachment::OnPlan => Ok(Settled::Ready), + Attachment::KeyGone => Ok(Settled::Retry), + Attachment::AlreadyOnAPlan => match gateway.plan_of(key_id).await? { + Some(plan) => { + if plan.id != plan_id { + tracing::warn!( + key_id, + wanted_plan_id = plan_id, + plan_id = %plan.id, + tier = ?plan.tier, + "the key was already on another usage plan for this stage; leaving it there" + ); + } + Ok(Settled::Ready) + } + None => { + tracing::warn!( + key_id, + wanted_plan_id = plan_id, + "AWS says the key is already on a usage plan, but GetUsagePlans names none \ + for this stage yet; re-running the reconciliation" + ); + Ok(Settled::Retry) + } + }, + } +} + /// The usage plan to attach `winner_id` to, decided before anything is /// deleted (task 0311). `None` means "no attach": the winner is already on a /// plan for our stage. /// /// 1. The winner's own plan for our stage — Step 4's "however it came to /// exist": a key an operator moved to Basic stays on Basic. -/// 2. Else the plan of the **previous key**: the latest revoked record in this -/// attempt's listing (the same latest-revocation rule the cap reads, -/// [`latest_revoked`]), if it is on a plan for our stage. This is what -/// makes a rework keep the plan — paid stays paid, custom stays custom, -/// free stays free. -/// 3. Else free — a first issue, or a previous key on no plan of ours. +/// 2. Else the plan of a **previous key**: every revoked record in this +/// attempt's listing, walked newest revocation first +/// ([`revoked_newest_first`]). The first one on a NON-free plan of our +/// stage wins — paid stays paid, custom stays custom. A record on free is +/// remembered but does not stop the walk: the records under one name can +/// disagree (a double-submit duplicate, or an undeletable record from an +/// earlier period, left on free beside the one an operator moved), and +/// letting the newest of them decide would be exactly the silent downgrade +/// decision 7 forbids. Only the portal ever puts a key on free, so a free +/// record beside a paid one is the portal's default, not an operator's +/// choice. +/// 3. Else free — a first issue, a previous key on free, or previous keys on +/// no plan of ours. /// /// Only a plan `GetUsagePlans` reported on OUR API stage can come out of /// here, which is the code half of the wildcard `POST /usageplans/*/keys` -/// grant's justification (`api-gateway-stack.ts`). +/// grant's justification (`compute-stack.ts`). +/// +/// One `GetUsagePlans` per revoked record at most, and the walk stops at the +/// first paid or custom plan; a name carries one or two records in practice. async fn resolve_target_plan( gateway: &Gateway, winner_id: &str, @@ -1306,9 +1380,24 @@ async fn resolve_target_plan( if gateway.plan_of(winner_id).await?.is_some() { return Ok(None); } - if let Some(previous) = latest_revoked(revoked) - && let Some(plan) = gateway.plan_of(&previous.id).await? - { + let mut on_free: Option<&str> = None; + for previous in revoked_newest_first(revoked) { + let Some(plan) = gateway.plan_of(&previous.id).await? else { + continue; + }; + if plan.id == gateway.free_plan_id() { + on_free.get_or_insert(previous.id.as_str()); + continue; + } + if let Some(free_key_id) = on_free { + tracing::warn!( + previous_key_id = %previous.id, + free_key_id, + plan_id = %plan.id, + "revoked records disagree about the plan: a newer one is on free; keeping the \ + paid or custom plan of the older one" + ); + } tracing::info!( previous_key_id = %previous.id, plan_id = %plan.id, diff --git a/packages/prices-api/src/portal/keys/naming.rs b/packages/prices-api/src/portal/keys/naming.rs index e35e2b80..55cc9652 100644 --- a/packages/prices-api/src/portal/keys/naming.rs +++ b/packages/prices-api/src/portal/keys/naming.rs @@ -114,22 +114,30 @@ pub fn revocation_instant(revoked: &[KeyRecord]) -> Option { .max() } -/// The revoked record whose plan a rework keeps (task 0311): the one with the -/// **latest** `lastUpdatedDate` — the same "latest revocation governs" rule as -/// [`revocation_instant`], so the record the cap is decided from and the -/// record the new key's plan is read from are the same record. +/// The revoked records, newest revocation first (task 0311) — the order a +/// rework walks them in to find the plan the new key keeps. /// -/// An undated record loses to any dated one (it is skipped by -/// [`revocation_instant`] for the same reason); ties — including all-undated — -/// go to the smaller id, so every invocation reads the same record. Empty → -/// `None`, and the caller falls back to the free plan. -pub fn latest_revoked(revoked: &[KeyRecord]) -> Option<&KeyRecord> { - revoked.iter().max_by(|a, b| { - a.last_updated_at - .cmp(&b.last_updated_at) - // Reversed, so that among equal instants the SMALLER id is "max". - .then_with(|| b.id.cmp(&a.id)) - }) +/// Newest first by `lastUpdatedDate`, the same "latest revocation governs" +/// rule as [`revocation_instant`]. An undated record sorts after every dated +/// one (it is skipped by [`revocation_instant`] for the same reason); ties — +/// including all-undated — go to the smaller id first, so every invocation +/// walks the same order. Empty in, empty out, and the caller falls back to +/// the free plan. +/// +/// All of them rather than only the latest, because the records under one +/// name can disagree about plans: a double-submit duplicate, or an +/// undeletable record from an earlier period, can be the latest and sit on +/// free (or on no plan) beside the record an operator moved to Basic. Which +/// one wins is the caller's rule — see `super::resolve_target_plan`. +pub fn revoked_newest_first(revoked: &[KeyRecord]) -> Vec<&KeyRecord> { + let mut ordered: Vec<&KeyRecord> = revoked.iter().collect(); + ordered.sort_by(|a, b| { + // `Option`'s order puts `None` first; reversed, the undated go last. + b.last_updated_at + .cmp(&a.last_updated_at) + .then_with(|| a.id.cmp(&b.id)) + }); + ordered } /// The key the owner currently holds, among `records`: the earliest **enabled** @@ -265,39 +273,35 @@ mod tests { assert_eq!(revocation_instant(&[]), None); } - /// The latest revocation is the record a rework keeps the plan of; an - /// undated record loses; a tie goes to the smaller id; nothing → None. + /// Newest revocation first; an undated record last; a tie goes to the + /// smaller id first; nothing → nothing (task 0311). #[test] - fn the_latest_revoked_record_is_the_latest_dated_one() { + fn revoked_records_are_walked_newest_first() { let at = |id: &str, when: Option| KeyRecord { last_updated_at: when, ..disabled(id, "n", Some(1)) }; + let ids = |records: &[KeyRecord]| -> Vec { + revoked_newest_first(records) + .into_iter() + .map(|r| r.id.clone()) + .collect() + }; let older = at("a", Some(10)); let newer = at("b", Some(20)); let undated = at("c", None); + let tie_high = at("z", Some(20)); assert_eq!( - latest_revoked(&[older.clone(), newer.clone(), undated.clone()]) - .unwrap() - .id, - "b" - ); - assert_eq!( - latest_revoked(&[undated.clone(), older.clone()]) - .unwrap() - .id, - "a" + ids(&[older.clone(), undated.clone(), newer.clone()]), + ["b", "a", "c"] ); - let tie_high = at("z", Some(20)); assert_eq!( - latest_revoked(&[tie_high.clone(), newer.clone()]) - .unwrap() - .id, - "b" + ids(&[tie_high.clone(), older.clone(), newer.clone()]), + ["b", "z", "a"] ); - assert_eq!(latest_revoked(&[newer.clone(), tie_high]).unwrap().id, "b"); - assert_eq!(latest_revoked(&[undated]).unwrap().id, "c"); - assert!(latest_revoked(&[]).is_none()); + assert_eq!(ids(&[newer, tie_high]), ["b", "z"]); + assert_eq!(ids(&[undated.clone(), at("0", None)]), ["0", "c"]); + assert!(revoked_newest_first(&[]).is_empty()); } #[test] diff --git a/packages/prices-api/tests/portal_issue.rs b/packages/prices-api/tests/portal_issue.rs index 48ce7093..8857d8b2 100644 --- a/packages/prices-api/tests/portal_issue.rs +++ b/packages/prices-api/tests/portal_issue.rs @@ -1036,3 +1036,48 @@ async fn attaching_to_a_paid_plan_puts_the_key_on_it() { vec![(BASIC_PLAN_ID.to_string(), key)] ); } + +/// AWS's two "already on a plan" refusals are both `AlreadyOnAPlan` (task +/// 0311, review WR-05): `409` for the same plan, and `400` "cannot reference +/// multiple Usage Plans with the same API Stage" for another plan on the +/// stage. Neither moves the key, and neither is an error. +#[tokio::test] +async fn both_already_on_a_plan_refusals_are_reported_as_such() { + use prices_api::portal::keys::gateway::Attachment; + + let gateway = MockGateway::start().await; + let key = gateway.with(|s| { + s.plans.push(StoredPlan::basic()); + s.seed_on_plan(&key_name(), 100, BASIC_PLAN_ID) + }); + let client = test_gateway(&gateway.base); + + for plan in [BASIC_PLAN_ID, PLAN_ID] { + assert_eq!( + client.attach_to_plan(&key, plan).await.expect(plan), + Attachment::AlreadyOnAPlan, + "{plan}" + ); + } + assert_eq!( + gateway.with(|s| s.plan_keys.clone()), + vec![(BASIC_PLAN_ID.to_string(), key)] + ); +} + +/// Any OTHER `400` from the attach stays an error — only AWS's same-stage +/// wording means "already on a plan". +#[tokio::test] +async fn any_other_bad_request_from_the_attach_is_an_error() { + let gateway = MockGateway::start().await; + let key = gateway.with(|s| { + s.fail_next_attach = true; + s.seed(&key_name(), 100) + }); + + let error = test_gateway(&gateway.base) + .attach_to_plan(&key, PLAN_ID) + .await + .expect_err("a plain 400 is not a success"); + assert!(error.to_string().contains("CreateUsagePlanKey"), "{error}"); +} diff --git a/packages/prices-api/tests/portal_keys/harness.rs b/packages/prices-api/tests/portal_keys/harness.rs index 73d14d81..df11f267 100644 --- a/packages/prices-api/tests/portal_keys/harness.rs +++ b/packages/prices-api/tests/portal_keys/harness.rs @@ -168,6 +168,16 @@ pub struct Store { /// Sticky, for the reason `throttle_usage` is: the SDK's own backoff /// retries a 429, so only a throttle that persists reaches the handler. pub throttle_plans: bool, + /// Answer the next N `GetUsagePlans` asking about key `id` as if the key + /// were on no plan at all (task 0311) — `GetUsagePlans` lagging behind an + /// attach that already happened, which is the window the WR-05 race paths + /// run in. `usize::MAX` keeps it lagging for good. + pub plans_hidden_for: HashMap, + /// Hide the key the next `CreateApiKey` mints from the `GetApiKeys` right + /// after it (task 0311) — the reconciler's "create not listed" branch + /// (Step 2), reached with whatever else the name holds still listed. + /// One-shot. + pub omit_next_created_from_list: bool, /// Every `{plan}` segment `GetUsage` was asked for, in order (task 0311) — /// which plan's counter was read. Separate from `usage_queries`, whose /// 3-tuple shape the pre-0311 tests destructure. @@ -535,6 +545,9 @@ pub async fn create_key( }; store.keys.push(created.clone()); store.ops.push(format!("create:{id}")); + if std::mem::take(&mut store.omit_next_created_from_list) { + store.next_list_omits_newest = true; + } if std::mem::take(&mut store.fail_next_create_after_creating) { return (StatusCode::INTERNAL_SERVER_ERROR, "response lost").into_response(); } @@ -662,6 +675,16 @@ pub async fn list_plans( return throttled(); } + // The listing lagging the attach (task 0311): the next N asks about this + // key see it on no plan, whatever `plan_keys` holds. + if let Some(key_id) = query.key_id.as_deref() + && let Some(left) = store.plans_hidden_for.get_mut(key_id) + && *left > 0 + { + *left -= 1; + return Json(json!({ "item": [] })).into_response(); + } + let matched: Vec = store .plans .iter() @@ -691,6 +714,23 @@ pub async fn list_plans( Json(body).into_response() } +/// AWS's answer to attaching a key to a second plan on a stage it already has +/// a plan for (task 0311): `400 BadRequestException`, with the service's own +/// wording, which is what `Gateway::attach_to_plan` recognises. +pub fn same_stage_refusal(key_id: &str, api_id: &str, stage: &str) -> Response { + ( + StatusCode::BAD_REQUEST, + [("x-amzn-errortype", "BadRequestException")], + Json(json!({ + "message": format!( + "API Key {key_id} cannot reference multiple Usage Plans with the same API \ + Stage: {api_id}:{stage}" + ) + })), + ) + .into_response() +} + /// The `400` shape the SDK maps to `BadRequestException` — not retried by /// the SDK, so one of them is observable from a handler. pub fn bad_request() -> Response { @@ -728,24 +768,37 @@ pub async fn attach_key( let Some(target) = store.plans.iter().find(|p| p.id == plan).cloned() else { return not_found(); }; - // Already on this plan — or on any plan sharing one of its stages, since - // a key belongs to one plan per stage — → `409 ConflictException`, as the - // service answers. The reconciler attaches every key it is about to hand - // out, so this is the ordinary case rather than an edge one, and a mock - // that silently accepted a re-attach would let a handler that treats the - // conflict as a failure pass every test here. + // Already on THIS plan → `409 ConflictException`, as the service answers + // ("API Key already exists in the usage plan"). A mock that silently + // accepted a re-attach would let a handler that treats the conflict as a + // failure pass every test here. + if store + .plan_keys + .iter() + .any(|(p, k)| p == &plan && k == &key_id) + { + return conflict(); + } + // Already on ANOTHER plan sharing one of this plan's stages → `400 + // BadRequestException` "cannot reference multiple Usage Plans with the + // same API Stage" (task 0311, review WR-05): a key belongs to one plan per + // stage, and AWS reports the second attach as a bad request, not as a + // conflict. Modelled that way so the race paths — a double-submit whose + // `GetUsagePlans` lagged, a sign-in inside an operator's delete→create gap + // — are tested against the answer the service gives. let shares_a_stage = |other: &str| { - other == plan - || store.plans.iter().any(|p| { - p.id == other && p.api_stages.iter().any(|s| target.api_stages.contains(s)) - }) + store + .plans + .iter() + .any(|p| p.id == other && p.api_stages.iter().any(|s| target.api_stages.contains(s))) }; if store .plan_keys .iter() .any(|(p, k)| k == &key_id && shares_a_stage(p)) { - return conflict(); + let (api_id, stage) = target.api_stages.first().cloned().unwrap_or_default(); + return same_stage_refusal(&key_id, &api_id, &stage); } store.plan_keys.push((plan, key_id.clone())); store.ops.push(format!("attach:{key_id}")); diff --git a/packages/prices-api/tests/portal_rework.rs b/packages/prices-api/tests/portal_rework.rs index fbc34233..55f918d1 100644 --- a/packages/prices-api/tests/portal_rework.rs +++ b/packages/prices-api/tests/portal_rework.rs @@ -1318,3 +1318,260 @@ async fn a_live_key_already_on_a_plan_gets_no_attach() { }); } } + +// --------------------------------------------------------------------------- +// Revoked records that disagree about the plan (task 0311, review WR-04) +// --------------------------------------------------------------------------- + +/// Two revoked records under one name — a double-submit duplicate, or an +/// undeletable record from an earlier period — and the NEWER one is on free +/// or on no plan at all, while the older one is on Basic. The rework walks +/// every revoked record, newest first, and keeps the first paid or custom +/// plan it finds: the new key lands on Basic, never on free. Both records are +/// swept only after the attach. +#[tokio::test] +async fn a_rework_keeps_the_paid_plan_when_a_newer_revoked_record_is_on_free() { + let discord = MockDiscord::start(GRANTED_SCOPE, None).await; + for newer_on in [Some(PLAN_ID), None] { + let gateway = MockGateway::start().await; + let revoked_at = the_3rd_of(first_of_month_offset(-1)); + let (on_basic, newer) = gateway.with(|s| { + s.plans.push(StoredPlan::basic()); + let on_basic = s.seed_revoked(&key_name(), 1_000, revoked_at); + s.plan_keys + .push((BASIC_PLAN_ID.to_string(), on_basic.clone())); + let newer = s.seed_revoked(&key_name(), 1_000, revoked_at + 3_600); + if let Some(plan) = newer_on { + s.plan_keys.push((plan.to_string(), newer.clone())); + } + (on_basic, newer) + }); + let app = app_with_discord(&discord, &gateway); + + assert_eq!( + issue_round_trip(&app).await.location(), + "/api/?issue=ok", + "{newer_on:?}" + ); + gateway.with(|s| { + let new = s + .keys + .iter() + .find(|k| k.enabled) + .expect("the new key exists") + .id + .clone(); + assert!( + s.plan_keys + .contains(&(BASIC_PLAN_ID.to_string(), new.clone())), + "{newer_on:?}: {:?}", + s.plan_keys + ); + assert!( + !s.plan_keys.contains(&(PLAN_ID.to_string(), new.clone())), + "{newer_on:?}: the new key must not land on free" + ); + assert_attached_before_deleted(&s.ops, &new, &on_basic); + assert_attached_before_deleted(&s.ops, &new, &newer); + }); + } +} + +/// Two revoked records on two different non-free plans: the NEWEST revocation +/// decides, deterministically — here the Basic record revoked after the +/// Custom one. +#[tokio::test] +async fn between_two_non_free_revoked_plans_the_newest_revocation_wins() { + let discord = MockDiscord::start(GRANTED_SCOPE, None).await; + let gateway = MockGateway::start().await; + let revoked_at = the_3rd_of(first_of_month_offset(-1)); + gateway.with(|s| { + s.plans.push(StoredPlan::basic()); + s.plans.push(StoredPlan::custom_unlimited()); + let older = s.seed_revoked(&key_name(), 1_000, revoked_at); + s.plan_keys.push((CUSTOM_PLAN_ID.to_string(), older)); + let newer = s.seed_revoked(&key_name(), 1_000, revoked_at + 60); + s.plan_keys.push((BASIC_PLAN_ID.to_string(), newer)); + }); + let app = app_with_discord(&discord, &gateway); + + assert_eq!(issue_round_trip(&app).await.location(), "/api/?issue=ok"); + gateway.with(|s| { + let new = s + .keys + .iter() + .find(|k| k.enabled) + .expect("new key") + .id + .clone(); + assert!( + s.plan_keys + .contains(&(BASIC_PLAN_ID.to_string(), new.clone())), + "{:?}", + s.plan_keys + ); + assert!(!s.plan_keys.contains(&(CUSTOM_PLAN_ID.to_string(), new))); + }); +} + +// --------------------------------------------------------------------------- +// Step 2's "create not listed" branch on a paid plan (task 0311, review IN-01) +// --------------------------------------------------------------------------- + +/// The listing after the create does not show the new key yet, so Step 2 +/// trusts the create and attaches it there — to the previous key's plan, not +/// to free. That branch sweeps nothing (the listing did not show the new key +/// to rank against), so the revoked record survives until the next issue, +/// which adopts the new key as it stands — no second attach — and sweeps it. +#[tokio::test] +async fn a_rework_whose_new_key_is_not_listed_yet_still_lands_on_the_paid_plan() { + let discord = MockDiscord::start(GRANTED_SCOPE, None).await; + let gateway = MockGateway::start().await; + gateway.with(|s| { + s.plans.push(StoredPlan::basic()); + s.omit_next_created_from_list = true; + }); + let dead = seed_revoked_on( + &gateway, + the_3rd_of(first_of_month_offset(-1)), + BASIC_PLAN_ID, + ); + let app = app_with_discord(&discord, &gateway); + + assert_eq!(issue_round_trip(&app).await.location(), "/api/?issue=ok"); + let new = gateway.with(|s| { + let new = s + .keys + .iter() + .find(|k| k.enabled) + .expect("new key") + .id + .clone(); + assert!( + s.plan_keys + .contains(&(BASIC_PLAN_ID.to_string(), new.clone())), + "{:?}", + s.plan_keys + ); + assert!(!s.plan_keys.contains(&(PLAN_ID.to_string(), new.clone()))); + assert_eq!( + s.ops, + vec![format!("create:{new}"), format!("attach:{new}")], + "Step 2's branch deletes nothing" + ); + new + }); + + assert_eq!(issue_round_trip(&app).await.location(), "/api/?issue=ok"); + gateway.with(|s| { + assert_eq!( + s.create_calls, 1, + "the next issue adopts, it does not create" + ); + assert_eq!(s.attach_calls, 1, "already on Basic, so no second attach"); + assert_eq!( + s.ops, + vec![ + format!("create:{new}"), + format!("attach:{new}"), + format!("delete:{dead}") + ] + ); + }); +} + +// --------------------------------------------------------------------------- +// An attach AWS refuses as already done (task 0311, review WR-05) +// --------------------------------------------------------------------------- + +/// A sign-in inside an operator's move, or a double-submit's second half: +/// the key IS on Basic, but `GetUsagePlans` has not caught up, so the flow +/// resolves free and attaches there. AWS refuses with the `400` "cannot +/// reference multiple Usage Plans with the same API Stage"; the flow asks +/// which plan the key is on, finds Basic, and hands the key out as it is — +/// not an `?issue=failed`, and not a key moved to free. +#[tokio::test] +async fn an_attach_refused_for_another_plan_on_the_stage_keeps_that_plan() { + let discord = MockDiscord::start(GRANTED_SCOPE, None).await; + let gateway = MockGateway::start().await; + let key = gateway.with(|s| { + s.plans.push(StoredPlan::basic()); + let key = s.seed_on_plan(&key_name(), 1_000, BASIC_PLAN_ID); + s.plans_hidden_for.insert(key.clone(), 1); + key + }); + let app = app_with_discord(&discord, &gateway); + + assert_eq!(issue_round_trip(&app).await.location(), "/api/?issue=ok"); + gateway.with(|s| { + assert_eq!(s.attach_calls, 1, "the lagging lookup led to one attach"); + assert_eq!( + s.plan_keys, + vec![(BASIC_PLAN_ID.to_string(), key.clone())], + "refused, so still on Basic alone" + ); + assert!(s.ops.is_empty(), "nothing written: {:?}", s.ops); + }); +} + +/// The double-submit rework: the other invocation already put the new key on +/// Basic (the previous key's plan) but this one's `GetUsagePlans` for it +/// lags. It resolves Basic from the revoked record, attaches, is refused with +/// `409` — already on THIS plan — confirms Basic and carries on to the sweep. +#[tokio::test] +async fn a_conflict_on_the_same_plan_is_settled_and_the_sweep_still_runs() { + let discord = MockDiscord::start(GRANTED_SCOPE, None).await; + let gateway = MockGateway::start().await; + gateway.with(|s| s.plans.push(StoredPlan::basic())); + let dead = seed_revoked_on( + &gateway, + the_3rd_of(first_of_month_offset(-1)), + BASIC_PLAN_ID, + ); + let new = gateway.with(|s| { + let new = s.seed_on_plan(&key_name(), 2_000, BASIC_PLAN_ID); + s.plans_hidden_for.insert(new.clone(), 1); + new + }); + let app = app_with_discord(&discord, &gateway); + + assert_eq!(issue_round_trip(&app).await.location(), "/api/?issue=ok"); + gateway.with(|s| { + assert_eq!(s.attach_calls, 1); + assert_eq!( + s.plan_keys + .iter() + .filter(|(_, k)| k == &new) + .collect::>(), + vec![&(BASIC_PLAN_ID.to_string(), new.clone())] + ); + assert_eq!(s.ops, vec![format!("delete:{dead}")]); + }); +} + +/// AWS refuses the attach as already done, but `GetUsagePlans` never names +/// the plan: the key's plan cannot be confirmed, so it is not handed out as +/// working. Every attempt re-runs and the round-trip ends `?issue=failed` — +/// and at no point is the key put on free. +#[tokio::test] +async fn a_refused_attach_whose_plan_never_shows_is_not_handed_out() { + let discord = MockDiscord::start(GRANTED_SCOPE, None).await; + let gateway = MockGateway::start().await; + let key = gateway.with(|s| { + s.plans.push(StoredPlan::basic()); + let key = s.seed_on_plan(&key_name(), 1_000, BASIC_PLAN_ID); + s.plans_hidden_for.insert(key.clone(), usize::MAX); + key + }); + let app = app_with_discord(&discord, &gateway); + + assert_eq!( + issue_round_trip(&app).await.location(), + "/api/?issue=failed" + ); + gateway.with(|s| { + assert!(s.attach_calls >= 1); + assert_eq!(s.plan_keys, vec![(BASIC_PLAN_ID.to_string(), key.clone())]); + assert!(s.ops.is_empty(), "nothing written: {:?}", s.ops); + }); +} From 807322df252ca73a659780226006ae42760eaf49 Mon Sep 17 00:00:00 2001 From: Adam <65679285+adamkoot@users.noreply.github.com> Date: Thu, 24 Sep 2026 14:17:05 +0200 Subject: [PATCH 08/22] test(lore-0311): pin that a month offset never shifts the period The unit test compared a period rule with itself. An integration test now puts a MONTH plan at offset 7 and checks that period_start and the GetUsage window both start on the 1st and limit equals the quota. --- packages/prices-api/src/portal/usage/mod.rs | 20 +++-------- packages/prices-api/tests/portal_usage.rs | 39 +++++++++++++++++++++ 2 files changed, 43 insertions(+), 16 deletions(-) diff --git a/packages/prices-api/src/portal/usage/mod.rs b/packages/prices-api/src/portal/usage/mod.rs index ae8f8c83..a71c695b 100644 --- a/packages/prices-api/src/portal/usage/mod.rs +++ b/packages/prices-api/src/portal/usage/mod.rs @@ -976,22 +976,10 @@ mod tests { assert_eq!(rule.resets_at().as_deref(), Some("2026-10-01T00:00:00Z")); } - /// The offset is a request count, not a start day: a plan with offset 7 - /// gets exactly the period a plan with offset 0 does. The rule is not - /// even given the offset, which is the point — this test pins that the - /// period depends on `quota.period` alone. - #[test] - fn a_month_offset_never_shifts_the_period() { - let today = date(2026, 12, 31); - let offset_zero = PeriodRule::for_period(Some("MONTH"), today); - let offset_seven = PeriodRule::for_period(Some("MONTH"), today); - assert_eq!(offset_zero, offset_seven); - assert_eq!(offset_seven.period_start().as_deref(), Some("2026-12-01")); - assert_eq!( - offset_seven.resets_at().as_deref(), - Some("2027-01-01T00:00:00Z") - ); - } + // A MONTH quota's `offset` never shifting the period is tested end to end + // in `tests/portal_usage.rs` (`a_month_quota_offset_never_shifts_the_period`): + // `PeriodRule` is never handed the offset, so a unit test here could only + // compare a rule with itself. /// DAY is the UTC day, resetting at the next midnight UTC. #[test] diff --git a/packages/prices-api/tests/portal_usage.rs b/packages/prices-api/tests/portal_usage.rs index b776aea5..343f02aa 100644 --- a/packages/prices-api/tests/portal_usage.rs +++ b/packages/prices-api/tests/portal_usage.rs @@ -1017,6 +1017,45 @@ async fn a_day_plan_is_the_utc_day_and_is_served_from_the_cache() { }); } +/// A MONTH quota with a nonzero `offset` (task 0311, review IN-02): the offset +/// is "the number of requests subtracted from the given limit in the initial +/// time period" — a request count, not a start day — so the period is still +/// the calendar month from the 1st, the `GetUsage` window starts on the 1st, +/// and `limit` is the quota itself. +#[tokio::test] +async fn a_month_quota_offset_never_shifts_the_period() { + let mock = MockGateway::start().await; + mock.with(|s| { + s.plans.push(StoredPlan::on_our_stage( + "offset7", + "prices-production-offset-plan", + Some((2.0, 10)), + Some((50_000, 7, "MONTH")), + )); + let id = s.seed_on_plan(&format!("discord-{USER_ID}-key"), 100, "offset7"); + s.usage.insert(id, vec![vec![12, 49_981]]); + }); + + let today = Utc::now().date_naive(); + let first = today.with_day(1).unwrap(); + let body = usage(&mock, USER_ID).await.json(); + assert_eq!( + body["period_start"], + first.format("%Y-%m-%d").to_string(), + "{body}" + ); + assert_eq!(body["limit"], 50_000, "{body}"); + assert_eq!(body["plan"]["quota_period"], "MONTH", "{body}"); + mock.with(|s| { + let (_, start, _) = s + .usage_queries + .last() + .cloned() + .expect("GetUsage was called"); + assert_eq!(start, first.format("%Y-%m-%d").to_string()); + }); +} + /// `used + remaining` disagreeing with the plan's quota is a cross-check /// failure (logged as a warning), not a figure: the answer states the quota. #[tokio::test] From 395c03ba8b84475412b7d85ef52cec7064626c23 Mon Sep 17 00:00:00 2001 From: Adam <65679285+adamkoot@users.noreply.github.com> Date: Thu, 24 Sep 2026 14:17:05 +0200 Subject: [PATCH 09/22] fix(lore-0311): show no free-plan figure until the plan is known While /api/usage loaded or failed, a paid key was shown 1 req/s and "Contact us about a paid plan", and the key card showed the free rate beside a Rate Limit card with the real one. Both cards now take the key's plan and state a loading or failed state instead of a figure. Quota labels and the reset follow the quota period; the regenerate strip keeps the monthly cap. An unknown tier falls back to Custom, fractional rates print with two decimals at most, and the no-plan card has its own contact line. The unused rateLimit prop is gone. --- packages/prices-api/src/config.rs | 5 +- packages/prices-api/src/portal/mod.rs | 5 +- web/portal/src/app/app.spec.tsx | 246 +++++++++++-- web/portal/src/app/app.tsx | 417 +++++++++++++++------- web/portal/src/landing/DashboardPanel.tsx | 6 +- web/portal/src/landing/FairAccess.tsx | 12 +- 6 files changed, 508 insertions(+), 183 deletions(-) diff --git a/packages/prices-api/src/config.rs b/packages/prices-api/src/config.rs index cbe6d6f7..f9131066 100644 --- a/packages/prices-api/src/config.rs +++ b/packages/prices-api/src/config.rs @@ -41,8 +41,9 @@ pub struct AppConfig { /// `api-gateway-stack.ts` feeds to `addUsagePlan`. Since task 0311 the /// signed-in dashboard does not state this figure: `/api/usage` reads the /// key's OWN plan through `GetUsagePlans` and reports its figures. This - /// stays for what has no key to ask about — the no-key state, the landing - /// page, and the fallback while the usage call is unanswered — and it + /// stays for what has no key to ask about — the no-key state and the + /// landing page; while the usage call is unanswered or failed the + /// dashboard states no figure at all rather than this one — and it /// stays config-fed rather than a literal in the frontend, because a /// literal would drift from what the gateway enforces the moment /// `infra/envs/production.json` changed. diff --git a/packages/prices-api/src/portal/mod.rs b/packages/prices-api/src/portal/mod.rs index a6b7775c..5ae8a551 100644 --- a/packages/prices-api/src/portal/mod.rs +++ b/packages/prices-api/src/portal/mod.rs @@ -118,8 +118,9 @@ pub struct PortalConfig { /// drift from what is actually enforced. Since task 0311 a signed-in /// caller WITH a key is shown their own plan's figures, from `/usage` /// (`GetUsagePlans` on the key). This stays for what has no key to ask - /// about: the no-key state — a `404` with no body to carry a plan — the - /// landing page, and the fallback while the usage call is unanswered. + /// about: the no-key state — a `404` with no body to carry a plan — and the + /// landing page. While the usage call is unanswered or failed the + /// dashboard states no figure rather than falling back to this one. /// /// Omitted from the JSON entirely when this deployment was not told what /// the limit is; the page then omits the line rather than inventing a diff --git a/web/portal/src/app/app.spec.tsx b/web/portal/src/app/app.spec.tsx index ca3e821a..f0909799 100644 --- a/web/portal/src/app/app.spec.tsx +++ b/web/portal/src/app/app.spec.tsx @@ -98,8 +98,8 @@ const ISSUE_HREF = '/api/auth/login?action=issue'; * `pricingApiFreePlanRateLimit` unconditionally. `1` is what * `infra/envs/production.json` holds today. Since task 0311 the signed-in * dashboard states the key's OWN plan from `/api/usage`; `/config`'s figure - * feeds only the no-key state, the landing page, and the fallback while the - * usage call is unanswered or failed. + * feeds only the no-key state and the landing page. While the usage call is + * unanswered or failed the dashboard states no figure at all. */ const openConfig = () => ({ json: async () => ({ enabled: true, rate_limit_per_second: 1 }), @@ -2560,20 +2560,21 @@ describe('the API key', () => { /** * "Issued" is only ever rendered where the round-trip that just ended - * created the key, because `GET /key` carries no timestamp — and the rate - * limit comes from `/config`, which the stub answers with 1 req/s. + * created the key, because `GET /key` carries no timestamp. * - * The quota column is deliberately NOT asserted here: this stub's `/usage` - * says "no key yet", so the page has not been told a limit and the field is - * absent rather than invented. + * Neither the quota nor the rate is stated here: this stub's `/usage` says + * "no key yet", so the page has not been told the key's plan. Before task + * 0311's review the rate column showed `/config`'s figure — the FREE plan's, + * which a paid key reads as its own (review CR-02); the paid case is in + * "usage against quota" below. */ - it('states when the key was issued and at what rate limit', async () => { + it('states when the key was issued, and no rate before the plan is known', async () => { signedInWithKey(); renderApp('/?issue=ok'); expect(await screen.findByText('Issued')).toBeTruthy(); expect(screen.getByText(/just now/i)).toBeTruthy(); - expect(screen.getByText('Rate limit')).toBeTruthy(); + expect(screen.queryByText('Rate limit')).toBeNull(); expect(screen.queryByText('Monthly quota')).toBeNull(); }); @@ -3534,11 +3535,18 @@ describe('usage against quota', () => { }); /** - * A deployment that did not say what the limit is says nothing about it. A - * fallback figure would be the same silent staleness one layer down — and - * unlike the missing line, it would look authoritative. + * The card is never dropped, whatever `/config` and `/usage` say — a + * missing panel is a worse answer than a stated one (Adam, 2026-08-25: + * the whole Rate Limit card went missing on a local run without a limit in + * `/config`). + * + * ⚠️ What it states changed with task 0311's review (WR-01). `/usage` + * failing used to fall back to `/config`'s figure, or to a built-in 1 req/s + * — the FREE plan's, stated to whichever key was signed in, paid or not. The + * card now says the plan could not be loaded and states no figure; it keeps + * the gateway's two HTTP codes, which are true of every key. */ - it('falls back to the plan rate rather than dropping the Rate Limit card', async () => { + it('keeps the Rate Limit card, with no figure, when neither source answers', async () => { stubRoutes({ [CONFIG_URL]: () => ({ json: async () => ({ enabled: true }) }), [KEY_URL]: () => ({ @@ -3556,8 +3564,6 @@ describe('usage against quota', () => { username: 'adam', }), }), - // Failed, so the card has no plan to state (task 0311) and falls back - // to `/config` — which here says nothing, so to the built-in figure. [USAGE_URL]: () => ({ ok: false, status: 500, @@ -3566,16 +3572,14 @@ describe('usage against quota', () => { }); renderApp(); - // ⚠️ The OPPOSITE of what this pinned until 2026-08-25, when Adam found the - // whole Rate Limit card missing on a local run. `/config` without a limit - // used to drop the panel; it now shows the free plan's documented 1 req/s - // (task 0157), the same figure the landing page states to every visitor. - // A stated figure beats a third of the dashboard disappearing — and where - // the deployment DOES answer, its value still wins (the test above). await screen.findByText(/Could not load your usage/); - expect((await screen.findByTestId('rate-limit')).textContent).toBe('1'); - expect(screen.getByText(/per-minute limit/i)).toBeTruthy(); - expect(screen.getByText(/request per second/i)).toBeTruthy(); + const card = await rateLimitCard(); + expect(within(card).getByTestId('rate-limit-pending').textContent).toBe( + 'Could not load your plan, so its limits are not shown.', + ); + expect(within(card).queryByTestId('rate-limit')).toBeNull(); + expect(within(card).getByText('HTTP 429')).toBeTruthy(); + expect(within(card).getByText('HTTP 403')).toBeTruthy(); }); // ------------------------------------------------------------------------- @@ -3616,11 +3620,13 @@ describe('usage against quota', () => { }); /** - * `/api/usage` failed: the card keeps its pre-0311 rendering — `/config`'s - * figure, the "Active" pill and no plan pill — and, since the figure it - * states is the free plan's, the free plan's contact copy (decision 6). + * `/api/usage` failed: the card states NO figure, no pill and no tier's + * contact copy (task 0311's review, WR-01). `/config` answers 5 req/s here + * and it must not appear: it is the free plan's figure, and a Pro customer + * whose usage call hit a throttle would read it — and "Contact us about a + * paid plan" — as a statement about the plan they already pay for. */ - it("keeps /config's figure and the free contact copy when usage fails", async () => { + it("states no figure and no tier's copy when usage fails", async () => { stubRoutes({ [CONFIG_URL]: () => ({ json: async () => ({ enabled: true, rate_limit_per_second: 5 }), @@ -3645,17 +3651,176 @@ describe('usage against quota', () => { await screen.findByText(/Could not load your usage/); const card = await rateLimitCard(); - expect(within(card).getByTestId('rate-limit').textContent).toBe('5'); - expect(within(card).getByText('Active')).toBeTruthy(); + await waitFor(() => + expect(within(card).getByTestId('rate-limit-pending').textContent).toBe( + 'Could not load your plan, so its limits are not shown.', + ), + ); + expect(within(card).queryByTestId('rate-limit')).toBeNull(); + expect(within(card).queryByText('5')).toBeNull(); + expect(within(card).queryByText('Active')).toBeNull(); for (const label of ['Free', 'Basic', 'Analyst', 'Lite', 'Pro', 'Custom']) { expect(within(card).queryByText(label), label).toBeNull(); } - expect(within(card).getByTestId('rate-limit-contact').textContent).toMatch( - /^Need higher limits\?/, + expect(within(card).queryByTestId('rate-limit-contact')).toBeNull(); + expect(document.body.textContent).not.toMatch(/paid plan/i); + }); + + /** + * While `/api/usage` is in flight the card says so and states nothing else + * about the plan — the flash of the free plan's "1 req/s" and its paid-plan + * pitch that every paid key saw on every load (review WR-01). + */ + it('states no figure while the plan is loading', async () => { + // A body that never arrives: `/usage` stays in flight for the test. + const pending = new Promise((resolve) => void resolve); + signedInWithUsage(() => ({ json: () => pending })); + renderApp(); + + const card = await rateLimitCard(); + expect(within(card).getByTestId('rate-limit-pending').textContent).toBe( + 'Loading your plan…', + ); + expect(within(card).queryByTestId('rate-limit')).toBeNull(); + expect(within(card).queryByText('Active')).toBeNull(); + expect(within(card).queryByTestId('rate-limit-contact')).toBeNull(); + }); + + /** + * The first-login row on the key card states the key's OWN plan — the + * rework flow task 0311 exists for: a Basic user's new key lands on Basic, + * and the row said "Rate limit 1 req/s" beside Basic's quota while the Rate + * Limit card said 3 req/s (review CR-02). The quota is named by its + * period, so a DAY plan's is not "Monthly" (review WR-02). + */ + it("states the key's own plan in the first-login row", async () => { + const renderLanded = () => + render( + + + , + ); + + signedInWithUsage( + usageWith(paidPlan('basic', 3, 15, 1000000), { + limit: 1000000, + remaining: 999879, + }), + ); + const basic = renderLanded(); + expect(await screen.findByText('3 req/s')).toBeTruthy(); + expect(screen.getByText('Rate limit')).toBeTruthy(); + expect(screen.getByText('Monthly quota')).toBeTruthy(); + expect(screen.getByText('1,000,000 requests')).toBeTruthy(); + expect(screen.queryByText('1 req/s')).toBeNull(); + basic.unmount(); + + signedInWithUsage( + usageWith( + { + tier: 'custom', + name: 'prices-production-daily-plan', + rate_limit_per_second: 2, + burst_limit: 10, + quota_limit: 5000, + quota_period: 'DAY', + }, + { used: 10, remaining: 4990, limit: 5000 }, + ), + ); + renderLanded(); + expect(await screen.findByText('Daily quota')).toBeTruthy(); + expect(screen.getByText('5,000 requests')).toBeTruthy(); + expect(screen.getByText('2 req/s')).toBeTruthy(); + expect(screen.queryByText('Monthly quota')).toBeNull(); + }); + + /** + * A DAY quota (a hand-made Custom plan) is not "Monthly", and its reset — + * tomorrow — is not when the next key can be issued: the rework cap is the + * calendar month on every plan (task 0191), so the strip names the 1st of + * next month whatever `resets_at` says (task 0311's review, WR-02). + */ + it("keeps a DAY plan's reset out of the rework strip and titles it daily", async () => { + signedInWithUsage( + usageWith( + { + tier: 'custom', + name: 'prices-production-daily-plan', + rate_limit_per_second: 2, + burst_limit: 10, + quota_limit: 5000, + quota_period: 'DAY', + }, + { + used: 10, + remaining: 4990, + limit: 5000, + period_start: '2020-01-01', + period_end: '2020-01-01', + resets_at: '2020-01-02T00:00:00Z', + }, + ), + ); + renderApp(); + + expect( + await screen.findByRole('heading', { name: 'Daily Usage' }), + ).toBeTruthy(); + expect(screen.queryByRole('heading', { name: 'Monthly Usage' })).toBeNull(); + const note = await screen.findByRole('note'); + const nextMonth = new Date( + Date.UTC(new Date().getUTCFullYear(), new Date().getUTCMonth() + 1, 1), + ); + await waitFor(() => + expect(note.textContent).toContain( + nextMonth.toLocaleDateString('en-GB', { + day: 'numeric', + month: 'long', + year: 'numeric', + timeZone: 'UTC', + }), + ), + ); + expect(note.textContent).not.toMatch(/2 January 2020/); + }); + + /** + * A Custom plan's fractional rate is printed as a rate, not as a floating + * point artefact: 0.1 req/s is 6 req/min, not 6.000000000000001 (review + * IN-03). + */ + it('prints a fractional rate without floating-point noise', async () => { + signedInWithUsage( + usageWith({ ...paidPlan('custom', 0.1, 1, 1000000), name: 'slow' }), + ); + renderApp(); + + const card = await rateLimitCard(); + await waitFor(() => + expect(within(card).getByTestId('rate-limit').textContent).toBe('0.1'), + ); + expect(within(card).getByText('6')).toBeTruthy(); + expect(card.textContent).not.toMatch(/0000000/); + }); + + /** + * A tier this bundle does not know — a newer backend, deployed apart from + * the bundle — is labelled Custom and gets Custom's copy, instead of a + * lookup that throws and unmounts the dashboard (review IN-04). + */ + it('treats an unknown tier as Custom rather than crashing', async () => { + signedInWithUsage( + usageWith({ ...paidPlan('gold', 40, 200, 90000000), tier: 'gold' }), + ); + renderApp(); + + const card = await rateLimitCard(); + await waitFor(() => expect(within(card).getByText('Custom')).toBeTruthy()); + expect(within(card).getByTestId('rate-limit').textContent).toBe('40'); + expect(within(card).getByTestId('rate-limit-contact').textContent).toBe( + 'Need custom limits? Contact us.', ); - const link = within(card).getByRole('link', { name: /contact us/i }); - expect(link.textContent).toBe('Contact us about a paid plan.'); - expect(link.getAttribute('href')).toBe(RUMBLEFISH_CONTACT); }); /** @@ -3691,6 +3856,17 @@ describe('usage against quota', () => { expect(screen.queryByTestId('usage-used')).toBeNull(); expect(document.body.textContent).not.toMatch(/not recorded any usage/i); expect(document.body.textContent).not.toMatch(/nothing recorded/i); + // Its own contact line, not Custom's "Need custom limits?" — a key on no + // plan is broken, and a sign-in is what re-attaches it (review IN-05). + expect(within(card).getByTestId('rate-limit-contact').textContent).toBe( + 'Signing out and in again puts it back on a plan. If that does not fix it, contact us.', + ); + expect( + within(card) + .getByRole('link', { name: /contact us/i }) + .getAttribute('href'), + ).toBe(RUMBLEFISH_CONTACT); + expect(card.textContent).not.toMatch(/custom limits/i); }); /** diff --git a/web/portal/src/app/app.tsx b/web/portal/src/app/app.tsx index 2fbbe2ab..9a3e76cf 100644 --- a/web/portal/src/app/app.tsx +++ b/web/portal/src/app/app.tsx @@ -249,12 +249,7 @@ function useOneShotParams( * the task. "Cancelled" is not an error: the visitor pressed Cancel at Discord's * consent screen, the callback redirected here with `?signin=cancelled`, and the * only reasonable response is to say so and leave the button where it was. - * - * `rateLimit` is nothing to do with sign-in and is not read here: it comes off - * `/config`, which only this component's parent has, and is wanted three levels - * down by the usage panel (task 0188). Passed through rather than re-fetched or - * put in a context — one prop across two hops is less machinery than either, - * and it keeps the value's single source visible in the call chain. + */ function useSession(enabled: boolean): { session: SessionState; @@ -1351,12 +1346,9 @@ const UNDERLINED = { */ function Dashboard({ session, - rateLimit, onSignOut, }: { session: PortalSession; - /** The free plan's per-second rate limit, straight from `/config`. */ - rateLimit?: number; /** * ⚠️ Back on this component since 2026-08-26, for one caller only: the * revoked card puts "Sign out" among its actions, as the frame draws it. @@ -1419,19 +1411,25 @@ function Dashboard({ // already consumed by then, so a reload could not bring it back. return landed.get('issue') !== null || landed.get('signin') !== null; }); - // The quota `/usage` reported, so the key card's "Monthly quota" field can - // state a number this page was actually told. `undefined` until the panel - // below has an answer; the field is simply absent until then. + // The quota `/usage` reported, so the key card's quota field can state a + // number this page was actually told. `null` until the panel below has an + // answer; the field is simply absent until then. // - // `plan` (task 0311) feeds the Rate Limit card: `undefined` while `/usage` - // has not answered (or failed) — the card then states `/config`'s free - // figure as it always did — `null` for a key on no usage plan, and the - // plan's figures and pill otherwise. + // `plan` (task 0311) is what `/usage` said about the key's plan, as a + // `PlanView`: the Rate Limit card and the key card state that plan's + // figures and nothing else. Until it is `known` they state no figure at all + // — `/config`'s is the FREE plan's, and a paid key would read it as its own + // (task 0311's review, CR-02/WR-01). + // + // `capResetsAt` is `/usage`'s `resets_at` only when the plan counts per + // MONTH: the rework cap is the calendar month whatever the quota's period + // (task 0191), so a DAY plan's "tomorrow" must never become the date the + // next key can be issued from (review WR-02). const [usageFacts, setUsageFacts] = useState<{ quota: number | null; - resetsAt: string | null; - plan: PortalPlan | null | undefined; - }>({ quota: null, resetsAt: null, plan: undefined }); + capResetsAt: string | null; + plan: PlanView; + }>({ quota: null, capResetsAt: null, plan: { state: 'loading' } }); // ⚠️ **A revoked key replaces the whole dashboard** (Adam, 2026-08-26). // @@ -1473,9 +1471,9 @@ function Dashboard({ onRevokedState={setRevoked} onKeyAbsent={setKeyAbsent} session={session} - rateLimit={rateLimit} quota={usageFacts.quota} - resetsAt={usageFacts.resetsAt} + resetsAt={usageFacts.capResetsAt} + plan={usageFacts.plan} /> {/* Two columns at the design's 5:3 ratio, one at 375px. `align-items: @@ -1501,20 +1499,26 @@ function Dashboard({ keyAbsent={keyAbsent} keyOnScreen={keyOnScreen} revokedCount={revokedCount} - rateLimit={rateLimit} onUsage={(usage) => setUsageFacts({ quota: usage?.limit ?? null, - resetsAt: usage?.resets_at ?? null, - plan: usage?.plan, + capResetsAt: + usage?.plan?.quota_period === 'MONTH' + ? (usage.resets_at ?? null) + : null, + plan: planViewOf(usage), }) } + onUsageFailed={() => + setUsageFacts((facts) => ({ ...facts, plan: { state: 'failed' } })) + } + quotaPeriod={ + usageFacts.plan.state === 'known' + ? usageFacts.plan.plan?.quota_period + : undefined + } /> - + ); @@ -2418,9 +2422,9 @@ function ApiKey({ onRevokedState, onKeyAbsent, session, - rateLimit, quota, resetsAt, + plan = { state: 'loading' }, }: { onKey?: () => void; /** Task 0191: the key on screen was just deactivated. A fact, no data. */ @@ -2455,20 +2459,28 @@ function ApiKey({ */ onKeyAbsent?: (absent: boolean) => void; session: PortalSession; - /** The free plan's per-second limit, from `/config`. */ - rateLimit?: number; /** - * The monthly quota, as `/usage` reported it to the panel below — `null` - * where AWS has recorded nothing yet, `undefined` before it has answered. + * The key's plan, as `/usage` reported it to the panel below (task 0311). + * The first-login row states its rate and names its quota's period; until + * it is `known` the row states neither — never `/config`'s free figure, + * which a paid key would read as its own (review CR-02). + */ + plan?: PlanView; + /** + * The plan's quota, as `/usage` reported it to the panel below — `null` + * where the plan has none or `/usage` has not answered, `undefined` before + * the dashboard passes anything. * Lifted rather than fetched again: `GetUsage` is a control-plane call and * this page already spends one (task 0194 owns that budget). */ quota?: number | null; /** - * When the current quota period ends, RFC 3339, as `/usage` reported it — - * and so the instant a key revoked today becomes re-issuable (task 0191: - * one key per period, the cap decided against the same `Period`). `null` - * until the panel below has an answer. + * The instant a key revoked today becomes re-issuable, RFC 3339 — `/usage`'s + * `resets_at`, passed ONLY for a plan whose quota counts per `MONTH`, where + * the quota period and the rework cap's calendar month (task 0191) are the + * same `Period`. `null` otherwise, and the strip computes the 1st of next + * month itself: a `DAY` plan's reset is tomorrow, the cap's is not (task + * 0311's review, WR-02). */ resetsAt?: string | null; }) { @@ -2537,6 +2549,8 @@ function ApiKey({ * the ordinary card. */ const justIssued = landedWithKey && view.state === 'ok'; + /** The key's plan once `/usage` has named it (task 0311); else `null`. */ + const knownPlan = plan.state === 'known' ? plan.plan : null; /** * The `Dashboard - no key` card: the account has no key and this load did not * arrive from a completed issue round-trip. @@ -3109,13 +3123,25 @@ function ApiKey({ Just now {issuedOn && ` · ${issuedOn}`} - {quota !== undefined && quota !== null && ( - + {/* Task 0311: the quota and the rate are the key's OWN plan's, + and the quota is named by its period — "Monthly" on a DAY + plan is a false figure (review WR-02). Until `/usage` has + named the plan both are absent, never `/config`'s free + figure (review CR-02). */} + {knownPlan && quota !== undefined && quota !== null && ( + {quota.toLocaleString('en-US')} requests )} - {rateLimit !== undefined && ( - {rateLimit} req/s + {knownPlan && knownPlan.quota_limit === null && ( + Unlimited + )} + {knownPlan && ( + + {knownPlan.rate_limit_per_second === null + ? 'Unlimited' + : `${formatRate(knownPlan.rate_limit_per_second)} req/s`} + )} )} @@ -3263,7 +3289,10 @@ function ApiKey({ contradicting it. The date is the same instant either way. The date prefers `/usage`'s `resets_at`, which is the backend's own - `Period`; without it the page computes the 1st of next month itself, + `Period` — but only for a plan whose quota counts per MONTH (the + dashboard passes nothing else, task 0311's review WR-02): the cap is + the calendar month on every plan, and a DAY plan's reset is + tomorrow. Without it the page computes the 1st of next month itself, because that IS the rule (`portal/period.rs`) rather than a number the server holds. */} {view.state === 'ok' && !justIssued && ( @@ -3329,7 +3358,14 @@ function ApiKey({ * `RUMBLEFISH_CONTACT` — and this strip's copy is left as it was (0311 scopes * its contact change to that card). */ -function QuotaReachedNotice({ resetsAt }: { resetsAt: string }) { +function QuotaReachedNotice({ + resetsAt, + period, +}: { + resetsAt: string; + /** The plan's `quota_period` — the notice names it (task 0311, WR-02). */ + period?: string | null; +}) { const at = new Date(resetsAt); const on = Number.isNaN(at.getTime()) ? null @@ -3342,7 +3378,7 @@ function QuotaReachedNotice({ resetsAt }: { resetsAt: string }) { return (

- Monthly quota reached. API requests will return{' '} + {quotaLabel(period)} reached. API requests will return{' '} HTTP 429 {/* The date is dropped rather than guessed if `resets_at` is unparseable: the sentence still says what is happening and when it @@ -3369,8 +3405,9 @@ function Usage({ keyAbsent = false, keyOnScreen, revokedCount = 0, - rateLimit, onUsage, + onUsageFailed, + quotaPeriod, }: { /** * The account has no key at all — the frame gives this card an empty body. @@ -3381,7 +3418,6 @@ function Usage({ keyOnScreen: boolean; /** Task 0191: bumped by the dashboard on each in-page revoke. */ revokedCount?: number; - rateLimit?: number; /** * What this panel read, handed to the dashboard so the key card above can * state the quota and the date the next key becomes available without a @@ -3390,6 +3426,17 @@ function Usage({ * zero or a guessed date. */ onUsage?: (usage: PortalUsage | null) => void; + /** + * `/usage` failed (task 0311) — so the dashboard can tell "not answered + * yet" from "will not answer", and the Rate Limit card can say which. + */ + onUsageFailed?: () => void; + /** + * The key's plan's `quota_period`, once known (task 0311): the card is + * titled by it — "Monthly Usage" for every CDK plan, "Daily Usage" for a + * hand-made DAY plan (review WR-02). Absent → "Monthly Usage", the frame's. + */ + quotaPeriod?: string | null; }) { type UsageView = | { state: 'loading' } @@ -3423,9 +3470,11 @@ function Usage({ // expired session must read "sign out and sign in again" in BOTH // places, not as that sentence in one and a raw "answered 401" in the // other — two wordings for one cause on one screen reads as two bugs. - if (live) setView({ state: 'failed', reason: describeFailure(error) }); + if (!live) return; + setView({ state: 'failed', reason: describeFailure(error) }); + onUsageFailed?.(); }); - // ⚠️ `onUsage` IS an inline arrow at the call site, so it is a new + // ⚠️ `onUsage` (and `onUsageFailed`) IS an inline arrow at the call site, so it is a new // function on every render of `Dashboard` — and naming it here would // re-create `load`, which the mount effect below depends on, and refetch // usage in a loop. It is left out on purpose: `load` reads nothing from @@ -3528,7 +3577,7 @@ function Usage({ if (keyAbsent) return ; return ( - + {view.state === 'loading' &&

Loading your usage…

} {view.state === 'no-key' && @@ -3564,7 +3613,7 @@ function Usage({ // Task 0311: a plan without a quota. Nothing is counted against // it, so there is no meter to draw — "Unlimited" is the figure.

- Unlimited — your plan has no monthly quota. + Unlimited — your plan has no quota.

) : view.usage.limit !== null && view.usage.resets_at === null ? ( // Task 0311: a quota whose period the backend does not compute @@ -3653,7 +3702,10 @@ function Usage({ view.usage.limit > 0 && view.usage.used >= view.usage.limit && view.usage.resets_at !== null && ( - + )} ))} @@ -3668,44 +3720,78 @@ function Usage({ } /** - * The Rate Limit card — the key's plan, its two figures, and what the gateway - * does when you cross them. - * - * **The figures are the key's OWN plan's** since task 0311: `/api/usage` - * reports the plan the key is on (`plan`), and the per-second figure is its - * `rate_limit_per_second`; the per-minute one is that times sixty, computed - * rather than written down. The plan's name rides beside "Active" as a second - * pill (`Free` … `Pro`, `Custom` for any other plan on our stage), and the - * contact line under the figures links to `RUMBLEFISH_CONTACT` with copy for - * the tier — contacting us is the only way to change plan, so it is shown on - * every tier. + * What the dashboard knows about the key's usage plan (task 0311). * - * Three states besides that one: + * - `loading` — `/usage` has not answered yet; + * - `failed` — it answered with an error; + * - `unknown` — it answered, but named no plan: `no_key` while the key card + * holds one (the backend's short cache lagging an issue), or a backend older + * than task 0311 that sends no `plan` at all; + * - `known` — the plan, or `null` for a key on no plan for this API's stage. * - * - **No plan** (`plan: null`): the key is on no usage plan for this API, so - * the gateway answers it `403`. No pill — "Active" would be false — and the - * state is said in words. - * - **Unlimited** (`rate_limit_per_second: null`): both figures read - * "Unlimited", never a zero. - * - **Not answered yet** (`plan` undefined — `/usage` in flight or failed): - * `/config`'s free-plan figure, as before 0311. `/config` stays the source - * for that, for the no-key state and for the landing page. - * - * **The card always renders**, which is a change from the build that dropped - * it whenever `/config` carried no limit (Adam, 2026-08-25: "brakuje całego - * jednego kafelka"). A missing panel is a worse answer than a stated one: the - * free plan's rate is 1 req/s (task 0157), the landing page says so to every - * visitor before they sign in, and this card says the same where neither the - * key's plan nor the deployment has spoken. - * - * ⚠️ `/config` WINS over this constant wherever it answers, which is every - * deployed environment (`compute-stack.ts` passes `pricingApiFreePlanRateLimit` - * unconditionally), and the key's plan wins over both. The fallback is for the - * local case only, and if the free plan's rate ever changes, this constant is - * one of the two places that must change with it — the other being - * `FairAccess`. + * Only `known` puts figures on the page. The other three state no rate and + * no tier: `/config`'s figure is the FREE plan's, and a paid key reading it as + * its own is exactly the contradiction task 0311 exists to remove (review + * CR-02/WR-01). `/config` stays the source for the no-key state and the + * landing page only. + */ +type PlanView = + | { state: 'loading' } + | { state: 'failed' } + | { state: 'unknown' } + | { state: 'known'; plan: PortalPlan | null }; + +/** `/usage`'s answer as a `PlanView` — `null` is its `no_key`. */ +function planViewOf(usage: PortalUsage | null | undefined): PlanView { + if (!usage || usage.plan === undefined) return { state: 'unknown' }; + return { state: 'known', plan: usage.plan }; +} + +/** + * A quota named by its period (task 0311's review, WR-02): the five CDK plans + * all count per `MONTH`, but a hand-made Custom plan can count per `DAY` or + * `WEEK`, and "Monthly quota" over a daily figure is a false statement. A + * period this page has no word for is just "Quota"; no period known yet reads + * as the frame's "Monthly quota", which is every CDK plan's. */ -const FREE_PLAN_RATE_LIMIT = 1; +function quotaLabel(period: string | null | undefined): string { + switch (period) { + case undefined: + case 'MONTH': + return 'Monthly quota'; + case 'WEEK': + return 'Weekly quota'; + case 'DAY': + return 'Daily quota'; + default: + return 'Quota'; + } +} + +/** The usage card's title, by the same rule as `quotaLabel`. */ +function usageTitle(period: string | null | undefined): string { + switch (period) { + case 'WEEK': + return 'Weekly Usage'; + case 'DAY': + return 'Daily Usage'; + default: + return 'Monthly Usage'; + } +} + +/** + * A rate as the cards print it (task 0311's review, IN-03). AWS's + * `rateLimit` is a double, so a Custom plan at 0.1 req/s is 6.000000000000001 + * per minute in floating point. At most two decimals, and no grouping — the + * cards have always printed `1500`, not `1,500`. + */ +function formatRate(value: number): string { + return value.toLocaleString('en-US', { + maximumFractionDigits: 2, + useGrouping: false, + }); +} /** The pill label per tier (task 0311, decision 5). */ const PLAN_LABEL: Record = { @@ -3730,14 +3816,26 @@ const CONTACT_COPY: Record = { custom: ['Need custom limits?', 'Contact us.'], }; +/** + * The pill and the contact copy for a tier — a tier this bundle does not know + * (a newer backend: the two deploy separately) is `Custom`, never a lookup + * that throws and unmounts the dashboard (task 0311's review, IN-04). + */ +function planLabel(tier: string): string { + return PLAN_LABEL[tier as PortalPlanTier] ?? PLAN_LABEL.custom; +} +function contactCopy(tier: string): readonly [string, string] { + return CONTACT_COPY[tier as PortalPlanTier] ?? CONTACT_COPY.custom; +} + /** * The Rate Limit card's contact line (task 0311): a link to * `RUMBLEFISH_CONTACT`, underlined like the OAuth card's "contact support", * worded for the tier. It replaced a plain-text "Contact us for commercial * plans." that had no destination to point at. */ -function PlanContact({ tier }: { tier: PortalPlanTier }) { - const [lead, link] = CONTACT_COPY[tier]; +function PlanContact({ copy }: { copy: readonly [string, string] }) { + const [lead, link] = copy; return ( + + + + ); +} + +/** + * The Rate Limit card — the key's plan, its two figures, and what the gateway + * does when you cross them. + * + * **The figures are the key's OWN plan's** since task 0311: `/api/usage` + * reports the plan the key is on (`plan`), and the per-second figure is its + * `rate_limit_per_second`; the per-minute one is that times sixty, computed + * rather than written down. The plan's name rides beside "Active" as a second + * pill (`Free` … `Pro`, `Custom` for any other plan on our stage), and the + * contact line under the figures links to `RUMBLEFISH_CONTACT` with copy for + * the tier — contacting us is the only way to change plan, so it is shown on + * every tier. + * + * The other states: + * + * - **No plan** (`plan: null`): the key is on no usage plan for this API, so + * the gateway answers it `403`. No pill — "Active" would be false — the + * state is said in words, and the contact line says what fixes it. + * - **Unlimited** (`rate_limit_per_second: null`): both figures read + * "Unlimited", never a zero. + * - **Plan not known** (`/usage` in flight, failed, or naming no plan): no + * figure, no pill, no tier's contact copy — only what is true of every key + * (the gateway's `429`/`403`). Before task 0311's review this state showed + * `/config`'s free figure and the free plan's pitch, which a Pro customer + * read, on every load, as their own limit and an offer of the plan they + * already pay for (review WR-01). + * + * **The card always renders**, which is a change from the build that dropped + * it whenever `/config` carried no limit (Adam, 2026-08-25: "brakuje całego + * jednego kafelka"). A missing panel is a worse answer than a stated one — + * and since task 0311 its figures come from the key's plan, so `/config` + * missing a limit changes nothing here at all. + */ function RateLimitCard({ - rateLimit, keyAbsent = false, plan, }: { - rateLimit?: number; /** * The account has no key — the frame empties this card too. See the * dashboard's `keyAbsent`. @@ -3770,54 +3921,61 @@ function RateLimitCard({ * there is no key to make it about. */ keyAbsent?: boolean; - /** - * The key's plan, from `/api/usage` (task 0311): `undefined` until it - * answers (or when it failed), `null` for a key on no plan. - */ - plan?: PortalPlan | null; + /** The key's plan, from `/api/usage` (task 0311) — see `PlanView`. */ + plan: PlanView; }) { if (keyAbsent) return ; - if (plan === null) { + if (plan.state !== 'known') { + return ( + +

+ {plan.state === 'loading' + ? 'Loading your plan…' + : plan.state === 'failed' + ? 'Could not load your plan, so its limits are not shown.' + : "Your plan's limits appear here once your usage has loaded."} +

+ +
+ ); + } + + if (plan.plan === null) { return (

This key is not on a usage plan for this API, so the API answers 403 to it.

- +
); } + const { tier } = plan.plan; // `null` is a plan without a throttle: unlimited, stated as such. - const perSecond: number | null = plan - ? plan.rate_limit_per_second - : (rateLimit ?? FREE_PLAN_RATE_LIMIT); + const perSecond = plan.plan.rate_limit_per_second; return (
{/* Task 0188's sentence, kept verbatim and read only by assistive @@ -3830,25 +3988,20 @@ function RateLimitCard({ <>Rate limit: unlimited. ) : ( <> - Rate limit: {perSecond} request{perSecond === 1 ? '' : 's'} per - second. + Rate limit: {formatRate(perSecond)} request + {perSecond === 1 ? '' : 's'} per second. )} - - - - - {/* Until `/usage` names the plan the card states the free figure, so it - makes the free plan's offer. */} - + + ); } /** - * One big yellow number with its unit, from the Rate Limit card. A string - * value ("Unlimited", task 0311) is the whole statement and takes no unit. + * One big yellow number with its unit, from the Rate Limit card. A value that + * is the whole statement ("Unlimited", task 0311) is passed without a unit. */ function Figure({ label, @@ -3858,7 +4011,8 @@ function Figure({ }: { label: string; value: number | string; - unit: string; + /** Omitted when the value is the whole statement ("Unlimited", task 0311). */ + unit?: string; testId?: string; }) { return ( @@ -3882,7 +4036,7 @@ function Figure({ > {value} - {typeof value === 'number' && ( + {unit !== undefined && ( {unit} @@ -4582,11 +4736,6 @@ function DashboardRoute({ gate }: { gate: Gate }) { ); } - const rateLimit = - gate.probe.state === 'ok' - ? gate.probe.config.rate_limit_per_second - : undefined; - const session = (gate.session as { state: 'ok'; session: PortalSession }) .session; @@ -4635,11 +4784,7 @@ function DashboardRoute({ gate }: { gate: Gate }) { /> - + diff --git a/web/portal/src/landing/DashboardPanel.tsx b/web/portal/src/landing/DashboardPanel.tsx index a60e92f2..ed991c65 100644 --- a/web/portal/src/landing/DashboardPanel.tsx +++ b/web/portal/src/landing/DashboardPanel.tsx @@ -484,6 +484,9 @@ function RawFigure({ testId, value }: { testId: string; value: number }) { ); } +/** One header pill: a label and a tone. */ +export type Pill = { label: string; tone: 'ok' | 'muted' | 'bad' }; + /** * The dashboard's card shell — a titled header band over a body. * @@ -491,9 +494,6 @@ function RawFigure({ testId, value }: { testId: string; value: number }) { * and "Rate Limit". The status pill lives in the header beside the title, which * is the only place the design ever puts one. */ -/** One header pill: a label and a tone. */ -export type Pill = { label: string; tone: 'ok' | 'muted' | 'bad' }; - export function DashboardCard({ title, status, diff --git a/web/portal/src/landing/FairAccess.tsx b/web/portal/src/landing/FairAccess.tsx index 0370f11c..1d984233 100644 --- a/web/portal/src/landing/FairAccess.tsx +++ b/web/portal/src/landing/FairAccess.tsx @@ -20,11 +20,13 @@ import { * all: a developer deciding whether to build on this needs the quota BEFORE * they have a key, not after. * - * ⚠️ **These are hard-coded and the dashboard's are not.** The dashboard reads - * the rate limit from `/config` precisely so it cannot drift from what the - * gateway enforces; a marketing section cannot, because it renders for visitors - * with no session and often before the probe answers. If the free plan's limits - * change, this file is one of the two places that must change with it. + * ⚠️ **These are hard-coded and the dashboard's are not.** The dashboard states + * the key's own plan as `/api/usage` reports it (task 0311), precisely so it + * cannot drift from what the gateway enforces; a marketing section cannot, + * because it renders for visitors with no session and often before the probe + * answers. If the free plan's limits change (`infra/envs/production.json`), + * this file must change with them — since task 0311 the dashboard carries no + * built-in copy of the free rate any more. */ const REASONS: readonly string[] = [ From cfe3fce4e89f73b25883d712c33b3dc97bac09b2 Mon Sep 17 00:00:00 2001 From: Adam <65679285+adamkoot@users.noreply.github.com> Date: Thu, 24 Sep 2026 14:17:05 +0200 Subject: [PATCH 10/22] docs(lore-0311): guard the plan move and cover recovery and revert The runbook looks the current plan up instead of taking it by hand, refuses an empty or unchanged id, and only creates the new plan key after the delete succeeded. It adds a recovery for a refused create, the procedure for a revoked key waiting for the next period, the 0311 deploy order (Compute, then ApiGateway), and a revert that moves keys off the paid plans before ApiGateway deletes them. --- docs/runbooks/manual-api-key-tier.md | 177 +++++++++++++++++++++------ 1 file changed, 139 insertions(+), 38 deletions(-) diff --git a/docs/runbooks/manual-api-key-tier.md b/docs/runbooks/manual-api-key-tier.md index 2b648a27..da39f6e7 100644 --- a/docs/runbooks/manual-api-key-tier.md +++ b/docs/runbooks/manual-api-key-tier.md @@ -9,7 +9,8 @@ hand in AWS). **Who:** anyone with `AdministratorAccess` on the shared AWS account (`750702271865`, `eu-central-1`). The portal's own Lambda cannot do any of this: it may list a key's plans, read usage and attach a key, but holds no `DELETE` -or `PATCH` on a plan or a plan key (`api-gateway-stack.ts`). +or `PATCH` on a plan or a plan key (`compute-stack.ts`, the `/usageplans` +grants). --- @@ -38,8 +39,9 @@ plan per stage, so moving a user up (or down) means moving their key from one plan to another — the procedure below. The dashboard reads whatever plan the key is on (`GetUsagePlans` by key) and shows that plan's pill, figures, quota and reset; the **name** is the contract: the tier is parsed from -`pricing-api--production`, and any other plan on our stage shows as -**Custom** with its own name. +`pricing-api--production`, and any other plan on our stage shows the +**Custom** pill (its name travels in `/api/usage`'s `plan.name` for support, +but the page does not show it). --- @@ -58,55 +60,105 @@ export TIER=basic # free | basic | analyst | lite | pro **1. Find the key — by exact name.** A self-service key is named `discord--key` (`packages/prices-api/src/portal/keys/naming.rs`, `key_name`). `--name-query` is a **prefix** match (measured, task 0180), so the exact match -is the JMESPath filter, and `enabled` drops a revocation record: +is the JMESPath filter. List every record under the name, live and revoked: ```bash aws apigateway get-api-keys --name-query "discord-$ID-key" \ - --query "items[?name=='discord-$ID-key' && enabled].id" --output text + --query "items[?name=='discord-$ID-key'].[id,enabled,lastUpdatedDate]" \ + --output table ``` -Expect exactly **one** id. None: the user has no live key (never issued, or -revoked — they issue one first). Two: a double-submit duplicate the next issue -will sweep; ask the user to open the dashboard once (a sign-in reconciles) and -look again. Then: +- **One enabled record** — the ordinary case: `export K=`. +- **Two enabled records** — a double-submit duplicate the next issue will + sweep; ask the user to open the dashboard once (a sign-in reconciles) and + look again. +- **No enabled record, one or more disabled** — the user reworked ("Replace my + key", task 0191) and is waiting for the next period; their next key does not + exist yet. Move the **latest** disabled record (the newest `lastUpdatedDate`) + instead: `export K=`. When the period rolls, the issue attaches the + new key to the plan of the newest revoked record that is on a paid or Custom + plan of ours — free only if none is (task 0311, `resolve_target_plan`) — so + the new key lands on the target plan. A disabled key answers `403` whatever + its plan, so the move costs the user nothing. **For a downgrade** of such a + user, move **every** disabled record to free: a paid plan left on any one of + them would win. +- **Nothing at all** — the user has never been issued a key. Ask them to sign + in to the portal first. + +**2. Find the two plans, and check them.** `FROM` is the key's plan **on our +API** (a key may also sit on another API's plan — the partner plan `q7sd40` is +one — which is not ours to move), `TO` the target by exact name: ```bash -export K= -``` - -**2. Find the two plans.** The key's current plan, and the target by exact name: +API_ID=$(aws ssm get-parameter --name /prices/production/api-gateway-id \ + --query 'Parameter.Value' --output text) -```bash aws apigateway get-usage-plans --key-id "$K" \ - --query "items[].[id,name,apiStages[0].apiId]" --output table - -FROM= + --query "items[].[id,name,join(',', apiStages[].apiId)]" --output table +FROM=$(aws apigateway get-usage-plans --key-id "$K" \ + --query "items[?apiStages[?apiId=='${API_ID}' && stage=='production']].id | [0]" \ + --output text) TO=$(aws apigateway get-usage-plans \ --query "items[?name=='pricing-api-${TIER}-production'].id | [0]" --output text) -# `--output text` prints the literal "None" for an empty result — see -# "Change or revoke a Custom key" below. Guard it rather than pass it on. -[ "$TO" = "None" ] && { echo "no plan pricing-api-${TIER}-production"; unset TO; } -echo "from ${FROM} to ${TO}" + +# An id is lowercase letters and digits. `--output text` prints the literal +# "None" for an empty result, an unset variable is empty, and a typo in TIER +# or a key on no plan produces one of the two — none of them may reach step 3. +ok_id() { [[ "$1" =~ ^[a-z0-9]{6,}$ ]]; } +if ok_id "$K" && ok_id "$FROM" && ok_id "$TO" && [ "$FROM" != "$TO" ]; then + echo "ready: move $K from $FROM to $TO" +else + echo "STOP: K='$K' FROM='$FROM' TO='$TO' — fix this before step 3" +fi ``` -A key may also sit on another API's plan (the partner plan, `q7sd40`, is on a -different API); only the plan on our API — `/prices/production/api-gateway-id` -— is the one to move. +`FROM` empty or `None`: the key is on no plan of ours — it answers `403` right +now. There is nothing to delete; skip step 3 and run only the create: +`ok_id "$TO" && aws apigateway create-usage-plan-key --usage-plan-id "$TO" --key-id "$K" --key-type API_KEY`. +`TO` `None`: `TIER` is misspelt, or the plan is not deployed. `FROM` = `TO`: +nothing to do. -**3. Move it: delete the plan key, then create it on the target plan.** +**3. Move it: delete the plan key, then create it on the target plan** — and +only after step 2 printed `ready`. The same check is repeated here, so a paste +of this block alone cannot run the delete with a bad id: ```bash -aws apigateway delete-usage-plan-key --usage-plan-id "$FROM" --key-id "$K" -aws apigateway create-usage-plan-key --usage-plan-id "$TO" --key-id "$K" \ - --key-type API_KEY +if ok_id "$K" && ok_id "$FROM" && ok_id "$TO" && [ "$FROM" != "$TO" ]; then + aws apigateway delete-usage-plan-key --usage-plan-id "$FROM" --key-id "$K" && + aws apigateway create-usage-plan-key --usage-plan-id "$TO" --key-id "$K" \ + --key-type API_KEY +else + echo "refusing: K='$K' FROM='$FROM' TO='$TO'" +fi ``` Delete first, because a key is on one plan per stage: creating it on the target -while it is still on the old plan is refused. **Between the two commands the -key answers `403`** — it exists but is on no plan — so run them back to back; -it is a matter of seconds, and the data plane can take a few more to follow. -The key's **value does not change**, so the user changes nothing. +while it is still on the old plan is refused (`BadRequestException` "… cannot +reference multiple Usage Plans with the same API Stage"). **Between the two +commands the key answers `403`** — it exists but is on no plan — so run them +back to back; it is a matter of seconds, and the data plane can take a few more +to follow. The key's **value does not change**, so the user changes nothing. +The `&&` means a failed delete never reaches the create. + +**If the create is refused or fails** — the key is now on no plan, or on one +you did not choose. Look before acting: + +```bash +aws apigateway get-usage-plans --key-id "$K" --query "items[].[id,name]" --output table +``` + +- **On free** (`pricing-api-free-production`): the user signed in during the + gap, and the portal — seeing a key on no plan — attached it to free (or to the + plan of a revoked record, if they have one). Nothing is broken; move it again: + `aws apigateway delete-usage-plan-key --usage-plan-id --key-id "$K"`, + then the `create-usage-plan-key … "$TO"` above. +- **On no plan of ours**: re-run the create. If `TO` itself is the problem, + put the key back where it was with + `aws apigateway create-usage-plan-key --usage-plan-id "$FROM" --key-id "$K" --key-type API_KEY` + and sort out `TO` afterwards — a key on no plan answers `403` until one of + the two creates succeeds. A sign-in by the user also recovers it (onto free, + or onto a revoked record's plan). **4. Verify.** @@ -115,7 +167,8 @@ aws apigateway get-usage-plans --key-id "$K" --query "items[].name" ``` Exactly the target plan (`["pricing-api-basic-production"]`), plus any plan on -another API it was already on — never two plans of ours, and never none. +another API it was already on — never two plans of ours, and never none. If it +is anything else, go back to "If the create is refused or fails" above. **5. What the user sees.** @@ -132,7 +185,8 @@ another API it was already on — never two plans of ours, and never none. user stays paid, and nothing needs doing here. The once-per-period rework cap is the same on every plan. -**Downgrade** (a paid plan lapses): the same procedure with `TIER=free`. +**Downgrade** (a paid plan lapses): the same procedure with `TIER=free` — and +for a user waiting on a rework, every disabled record (step 1). **6. Record it.** Add a row to the "Issued manual keys" table at the bottom of this file (customer, plan, key id, date, who) and commit — the move is outside @@ -140,15 +194,59 @@ CDK and this file is its only record. --- +## Rolling task 0311 out + +**Deploy order: Compute, then ApiGateway** — the Makefile's cross-stack rule +(`deploy-production-compute`, then `deploy-production-apigateway`), and the +order `deploy --all` uses on its own. It is safe in that order, and only in +that order, because of where each half lives: + +- **Compute** ships the new handler AND the three `/usageplans` grants on its + role's own policy (`GET /usageplans`, `GET /usageplans/*/usage`, + `POST /usageplans/*/keys`). They are one CloudFormation update, and the + Function depends on the role policy, so the handler never runs without the + grant it calls on every sign-in and dashboard load (`GetUsagePlans`). +- **ApiGateway** adds the four paid plans and removes the old standalone + policy `PortalAttachKeyToFreePlan` (the two free-plan grants, a subset of the + new three). Its diff must show the free plan, its key and its SSM parameter + **unchanged**. +- Compute keeps exporting the role's name (`exportValue` in + `compute-stack.ts`) although nothing imports it any more: the deployed + ApiGateway policy still does until ApiGateway deploys, and CloudFormation + refuses to drop an export in use — without it the Compute deploy fails. + +**If a deploy fails:** a failed Compute deploy rolls back to the old handler +and the old role policy, and the old ApiGateway policy is still there — the old +handler works as before. A failed ApiGateway deploy rolls back to the old +policy; the new handler already holds its grants in Compute. + +**To revert the release after both deployed: move every key off the paid +plans first, then ApiGateway, then Compute.** The reverted ApiGateway deletes +the four paid plans. List their keys with `aws apigateway get-usage-plan-keys +--usage-plan-id ` for each tier and move each one to free with +the procedure above. A key still on a paid plan is left on no plan when that +plan is deleted and answers `403`, or CloudFormation refuses the delete and the +revert fails half-way. + +**Then ApiGateway FIRST, then Compute.** The reverted ApiGateway re-creates the free-plan policy; the +reverted Compute then drops the new grants. Compute first would leave the old +handler without its attach and usage grants until ApiGateway caught up. + +**The portal bundle after the backend.** `make -C infra sync-portal-explorer` +only once Compute is live: against an old backend the new bundle gets no +`plan` in `/api/usage` and shows the Rate Limit card's neutral "plan not +loaded" state rather than any figure. + ## Post-merge production verification (Adam) Task 0311's "on dev" checks are a production checklist — there is no dev environment (`infra/envs/` holds only `production.json` and `cicd.json`). -1. Deploy Compute, then ApiGateway (the Lambda learns `PORTAL_API_ID_PARAM` / - `PORTAL_API_STAGE` in Compute; the four plans and the widened policy land in - ApiGateway). In the ApiGateway diff the free plan, its key and its SSM - parameter must show **no change** — only the four new plans and the policy. +1. Deploy Compute, then ApiGateway, as above. In the ApiGateway diff the free + plan, its key and its SSM parameter must show **no change** — only the four + new plans and the removed `PortalAttachKeyToFreePlan` policy; in the + Compute diff, the role policy gains exactly the three `/usageplans` + statements (and the `api-gateway-id` read). 2. `/config` answers `enabled: true` (the portal did not close at cold start). 3. Move a test key free → Basic → Pro → free with the procedure above. After each step: @@ -161,6 +259,9 @@ environment (`infra/envs/` holds only `production.json` and `cicd.json`). `pricing-api-basic-production`, and the revoked key is gone. 5. A free key's dashboard looks as it did before, apart from the `Free` pill and the contact link. +6. The upgrade procedure's recovery path, once, on the test key: run the + delete of step 3 alone, sign in to the portal as that user (the portal puts + the key on free), then follow "If the create is refused or fails". --- From 56c9d2c78f3f6fa251afa4fbbf16a9226bbf9d12 Mon Sep 17 00:00:00 2001 From: Adam <65679285+adamkoot@users.noreply.github.com> Date: Thu, 24 Sep 2026 15:41:30 +0200 Subject: [PATCH 11/22] fix(lore-0311): report the usage of the key revoked last With every key under a name disabled, current_key took the earliest record. Two revoked records sit side by side when the listing lagged the create after a rolled rework (nothing is swept on that path) and the new key was then reworked in the same period. The usage card then read the old key's counter, empty for the period, and the new key's traffic vanished from the dashboard until the 1st. The newest revocation now decides. The reveal and the revoke only read the enabled flag and the revocation instant, so they are unchanged. --- packages/prices-api/src/portal/keys/naming.rs | 50 ++++++++++++---- packages/prices-api/src/portal/usage/mod.rs | 2 +- packages/prices-api/tests/portal_rework.rs | 57 +++++++++++++++++++ 3 files changed, 98 insertions(+), 11 deletions(-) diff --git a/packages/prices-api/src/portal/keys/naming.rs b/packages/prices-api/src/portal/keys/naming.rs index 55cc9652..e078e64e 100644 --- a/packages/prices-api/src/portal/keys/naming.rs +++ b/packages/prices-api/src/portal/keys/naming.rs @@ -141,13 +141,23 @@ pub fn revoked_newest_first(revoked: &[KeyRecord]) -> Vec<&KeyRecord> { } /// The key the owner currently holds, among `records`: the earliest **enabled** -/// key if there is one, otherwise the earliest key of any state (task 0191). +/// key if there is one (task 0191), otherwise the most recently **revoked** +/// one (task 0311). /// /// Enabled keys win over disabled ones whatever their dates, because a /// disabled key is a revocation record and an enabled one is a credential: if /// both exist (a console re-enable, a duplicate), the credential is what the -/// visitor is holding and what a revoke must act on. Among keys of one state -/// the rule is [`choose_winner`]'s, so both sides of a double-submit agree. +/// visitor is holding and what a revoke must act on. Among enabled keys the +/// rule is [`choose_winner`]'s, so both sides of a double-submit agree. +/// +/// Among revoked keys the newest revocation is the key the owner last held, +/// not the earliest-created record. Two revoked records sit under one name +/// when a previous revocation outlived the issue that replaced it (a listing +/// that lagged the create, or an undeletable record) and the replacement was +/// then revoked too. "Earliest" picked the older record, and the usage route +/// reported ITS counter — empty for the period — instead of the replacement's. +/// [`revoked_newest_first`] breaks ties by id, so the choice is still the same +/// on every invocation. /// /// The reveal, the revoke and the usage route all select through this, so the /// key whose value is handed out, the key a revoke disables and the key whose @@ -155,7 +165,7 @@ pub fn revoked_newest_first(revoked: &[KeyRecord]) -> Vec<&KeyRecord> { pub fn current_key(records: &[KeyRecord]) -> Option<&KeyRecord> { let enabled: Vec<&KeyRecord> = records.iter().filter(|r| r.enabled).collect(); if enabled.is_empty() { - choose_winner(records) + revoked_newest_first(records).into_iter().next() } else { enabled.into_iter().min_by_key(|r| rank(r)) } @@ -221,10 +231,10 @@ mod tests { } /// A credential beats a revocation record whatever their dates; among - /// credentials the earliest wins; with no credential the earliest record - /// is what the re-issue cap is read from. + /// credentials the earliest wins; with no credential the most recently + /// revoked record is the key the owner last held (task 0311). #[test] - fn the_current_key_is_the_earliest_enabled_one_or_else_the_earliest_record() { + fn the_current_key_is_the_earliest_enabled_one_or_else_the_latest_revoked() { let records = vec![ disabled("revoked-early", "n", Some(10)), record("live-late", "n", Some(200)), @@ -232,11 +242,31 @@ mod tests { ]; assert_eq!(current_key(&records).unwrap().id, "live-early"); + // A rework's replacement revoked beside the record it replaced: the + // revocation date decides, not the creation date. let only_disabled = vec![ - disabled("later", "n", Some(20)), - disabled("earlier", "n", Some(10)), + KeyRecord { + last_updated_at: Some(500), + ..disabled("created-later-revoked-later", "n", Some(20)) + }, + KeyRecord { + last_updated_at: Some(400), + ..disabled("created-earlier-revoked-earlier", "n", Some(10)) + }, + ]; + assert_eq!( + current_key(&only_disabled).unwrap().id, + "created-later-revoked-later" + ); + + let undated_beside_dated = vec![ + disabled("undated", "n", Some(5)), + KeyRecord { + last_updated_at: Some(400), + ..disabled("dated", "n", Some(10)) + }, ]; - assert_eq!(current_key(&only_disabled).unwrap().id, "earlier"); + assert_eq!(current_key(&undated_beside_dated).unwrap().id, "dated"); assert!(current_key(&[]).is_none()); } diff --git a/packages/prices-api/src/portal/usage/mod.rs b/packages/prices-api/src/portal/usage/mod.rs index a71c695b..ac658a86 100644 --- a/packages/prices-api/src/portal/usage/mod.rs +++ b/packages/prices-api/src/portal/usage/mod.rs @@ -641,7 +641,7 @@ async fn fetch(gateway: &Gateway, name: &str) -> Result Date: Thu, 24 Sep 2026 15:41:33 +0200 Subject: [PATCH 12/22] fix(lore-0311): evict a cached no-plan answer on issue A key on no usage plan was cached as an ordinary usage answer, and a successful issue evicted only "no key". The no-plan copy tells the user that signing out and in again puts the key back on a plan; the sign-in did attach it, but the dashboard kept saying "not on a usage plan" for the rest of the TTL, or up to STALE_KEEP while the control plane throttles. The no-plan answer is now evicted and epoch-guarded like "no key". --- packages/prices-api/src/portal/usage/mod.rs | 66 +++++++++++++++++---- packages/prices-api/tests/portal_issue.rs | 35 +++++++++++ 2 files changed, 91 insertions(+), 10 deletions(-) diff --git a/packages/prices-api/src/portal/usage/mod.rs b/packages/prices-api/src/portal/usage/mod.rs index ac658a86..49c917c5 100644 --- a/packages/prices-api/src/portal/usage/mod.rs +++ b/packages/prices-api/src/portal/usage/mod.rs @@ -212,8 +212,10 @@ struct EpochMark { /// A successful issue makes a cached "no key" answer false — and without this, /// provably wrong for a whole [`CACHE_TTL`]: the page's own refetch after the /// press, and any reload inside the window, would be served the stale `NoKey` -/// and tell a key-holder they have no key. The handle can evict **only** that -/// answer, nothing else: real usage entries stay cached (a reveal changes no +/// and tell a key-holder they have no key. The same holds for a key on no +/// plan (task 0311): the issue attaches it, and the page the sign-in lands on +/// must not keep saying "not on a usage plan". The handle can evict **only** +/// those two answers: real usage entries stay cached (a reveal changes no /// counter), and nothing outside this module can read or write anything. #[derive(Clone)] pub struct UsageCache(Arc>); @@ -240,15 +242,15 @@ impl UsageCache { self.invalidate_no_key(sub); } - /// Drop a cached "no key" answer for `sub`, if that is what is cached — - /// and bump the caller's epoch either way, so an in-flight lookup that + /// Drop a cached "no key" or "no plan" answer for `sub`, if that is what + /// is cached ([`CachedAnswer::is_false_after_issue`]) — and bump the caller's epoch either way, so an in-flight lookup that /// snapshotted the keyless state cannot write it back afterwards (see /// [`CacheInner`]). The unconditional bump is the point: at the moment the /// race matters there is nothing cached to remove. pub fn invalidate_no_key(&self, sub: &str) { let mut cache = self.0.lock().expect("the usage cache lock is not poisoned"); if let Some(entry) = cache.entries.get(sub) - && matches!(entry.answer, CachedAnswer::NoKey) + && entry.answer.is_false_after_issue() { cache.entries.remove(sub); } @@ -479,6 +481,19 @@ enum CachedAnswer { NoKey, } +impl CachedAnswer { + /// An answer a successful issue makes false: "no key", and a key on no + /// plan of our stage (task 0311 — the issue attaches it). Both are + /// evicted by [`UsageCache::invalidate_no_key`] and guarded by the epoch + /// in [`remember`]; a real usage answer is neither. + fn is_false_after_issue(&self) -> bool { + match self { + CachedAnswer::NoKey => true, + CachedAnswer::Usage { body, .. } => body.plan.is_none(), + } + } +} + #[derive(Clone)] struct CacheEntry { answer: CachedAnswer, @@ -817,11 +832,13 @@ fn remember(state: &UsageState, sub: &str, answer: CachedAnswer, epoch: Option Date: Fri, 25 Sep 2026 13:38:49 +0200 Subject: [PATCH 13/22] feat(lore-0311): load the portal sources on the first portal request, not at cold start A burst of /v1 cold starts throttled SSM, and one failed read closed the portal in that execution environment for its life. The api-handler cold start now reads only the mTLS bundle. The portal's five sources load on the first portal request that needs them (/config included), concurrently under a 4 s budget, through one shared cell: a success is kept for the environment, a failure answers that request only and the next retries. - /config: enabled only when the flag is on and the load succeeded - /key, /usage, /me: 503 on a failed load; login keeps its landings; the callback lands on ?signin=failed, and past SOURCES_ALLOWANCE (500 ms) on a retryable failure before the token exchange - PortalLoadError::Keys names the variable of the read that failed - serve.rs keeps its eager, expect()ing loaders - tokio test-util enabled for prices-api tests only (paused clock) --- packages/prices-api/Cargo.toml | 4 + packages/prices-api/src/config.rs | 452 ++++++++---------- packages/prices-api/src/lib.rs | 30 +- packages/prices-api/src/main.rs | 43 +- packages/prices-api/src/portal/auth/issue.rs | 89 +++- packages/prices-api/src/portal/auth/mod.rs | 113 +++-- packages/prices-api/src/portal/keys/mod.rs | 49 +- packages/prices-api/src/portal/mod.rs | 71 ++- packages/prices-api/src/portal/sources.rs | 281 +++++++++++ packages/prices-api/src/portal/usage/mod.rs | 38 +- packages/prices-api/tests/portal.rs | 15 +- packages/prices-api/tests/portal_auth.rs | 134 +++++- packages/prices-api/tests/portal_lazy_load.rs | 366 ++++++++++++++ 13 files changed, 1306 insertions(+), 379 deletions(-) create mode 100644 packages/prices-api/src/portal/sources.rs create mode 100644 packages/prices-api/tests/portal_lazy_load.rs diff --git a/packages/prices-api/Cargo.toml b/packages/prices-api/Cargo.toml index 6d159e99..9cae6aa7 100644 --- a/packages/prices-api/Cargo.toml +++ b/packages/prices-api/Cargo.toml @@ -102,3 +102,7 @@ tower = { version = "0.5", features = ["util"] } # see it and a test built on it would pass against the bug. Exact decimal # comparison is the only way to make that assertion non-vacuous. rust_decimal = { workspace = true } +# Task 0311: `#[tokio::test(start_paused = true)]` for the portal's load budget +# and retry backoff, so those tests assert time without spending it. Test +# builds only; the workspace's `full` does not include `test-util`. +tokio = { workspace = true, features = ["test-util"] } diff --git a/packages/prices-api/src/config.rs b/packages/prices-api/src/config.rs index f9131066..e0261cd2 100644 --- a/packages/prices-api/src/config.rs +++ b/packages/prices-api/src/config.rs @@ -59,11 +59,12 @@ pub struct AppConfig { /// /// **Not read from the environment**, which is the point — ADR 0007 and /// Tranche 3 AC 6 forbid a secret value in an env var. [`Self::from_env`] - /// leaves this `None` and [`Self::load_portal_oauth`] fills it from Secrets - /// Manager, asynchronously, because the read is an HTTP call. - /// - /// `None` means sign-in is not configured on this deployment, which is the - /// normal state while `portal_enabled` is false. + /// leaves this `None`. In the Lambda it stays `None` and the portal loads + /// its sources lazily, on the first portal request that needs them + /// (`crate::portal::sources`). A value here is a pre-supplied source: + /// `serve.rs` fills it eagerly with [`Self::load_portal_oauth`], and tests + /// set it directly. Either way the portal then never loads from the + /// environment. pub portal_oauth: Option, /// Which Discord to talk to (task 0186). Production always takes the /// defaults; the overrides exist for the local round-trip and for the tests. @@ -82,19 +83,15 @@ pub struct AppConfig { /// The API Gateway control-plane client the portal issues keys with /// (task 0187), already carrying the `pricing-api-free` usage-plan id. /// - /// `None` means key issuance is not configured on this deployment, which is - /// the normal state while `portal_enabled` is false — and, like - /// [`Self::portal_oauth`], it is filled by an async step rather than by - /// [`Self::from_env`], because building it resolves credentials and reading - /// the plan id is an HTTP call. + /// A pre-supplied source, like [`Self::portal_oauth`]: `None` in the + /// Lambda, where the portal loads it lazily; filled by `serve.rs` through + /// [`Self::load_portal_keys`], or directly by a test. pub portal_keys: Option, /// Where the eligibility gate's two knobs come from (task 0189): the - /// Stellar guild id and the minimum account age. `None` means the gate is - /// not configured, which is the normal state while `portal_enabled` is - /// false; filled by [`Self::load_portal_eligibility`], which also probes - /// both values once so a mis-seeded parameter closes the portal at cold - /// start ([`Self::load_portal_or_close`]) rather than refusing at a - /// visitor's click. + /// Stellar guild id and the minimum account age. A pre-supplied source, + /// like [`Self::portal_oauth`]: `None` in the Lambda, where the portal + /// loads it lazily; filled by `serve.rs` through + /// [`Self::load_portal_eligibility`], or directly by a test. pub portal_eligibility: Option, /// The origin the portal's bundle is served from, when that is not this /// backend's own host (task 0194): `https://sorobanscan.rumblefish.dev`. @@ -179,234 +176,162 @@ impl AppConfig { } } - /// Fill [`Self::portal_oauth`] from Secrets Manager, or from the local file - /// named by `PORTAL_OAUTH_SECRET_FILE`. - /// - /// Called by both entrypoints after [`Self::from_env`]. It is a separate, - /// async step because it performs I/O, and it is *conditional* on - /// [`Self::portal_enabled`], which is the load-bearing part: + /// Fill [`Self::portal_oauth`] for `serve.rs`, which loads eagerly and + /// `expect()`s: a developer who asked for the portal and did not get it + /// wants to know now, and no partner is behind that process. /// - /// Production ran with `PORTAL_ENABLED=false` for the whole of the portal's - /// build, until task 0194 flipped it in `compute-stack.ts` — so this read - /// now happens on every production cold start. The conditionality still - /// matters for tests and for any environment where the flag is off: if a - /// cold start read this secret unconditionally it would fail on a deployment - /// where nobody has created it yet — and that failure is not confined to the portal. - /// `main.rs` builds one router for every route group (ADR 0008), so a panic - /// in init takes out `/v1` as well, to protect four routes that answer an - /// empty `404` either way. - /// - /// With the portal **open**, a missing or malformed secret is an error — - /// and what the Lambda does with it is the decision recorded on - /// [`Self::load_portal_or_close`]: close the portal in that process - /// rather than panic, because a panic here is an init failure on the - /// function that also serves `/v1`. The thing this guards against — a - /// sign-in button that answers `503` — does not happen either way: with - /// the portal closed the gate answers before any handler does. + /// A no-op while [`Self::portal_enabled`] is false, so the ordinary local + /// run of the data API needs none of it. The Lambda never calls this: it + /// loads through `load_portal_sources` on the first portal request. pub async fn load_portal_oauth( &mut self, ) -> Result<(), crate::portal::auth::secret::SecretError> { if !self.portal_enabled { return Ok(()); } - match crate::portal::auth::secret::OauthSecret::load().await? { - Some(secret) => { - self.portal_oauth = Some(secret); - Ok(()) - } - None => Err(crate::portal::auth::secret::SecretError::NoSource), - } + self.portal_oauth = Some(portal_oauth_from_env().await?); + Ok(()) } - /// Fill [`Self::portal_keys`] with a control-plane client for task 0187. - /// - /// Conditional on [`Self::portal_enabled`] for exactly the reasons - /// [`Self::load_portal_oauth`] is, and one more of its own: - /// - /// - **A closed portal must not pay for this.** Building the client - /// resolves credentials and reads an SSM parameter; doing that at every - /// cold start would put two avoidable operations in front of the first - /// `/v1` request, on one router that serves every route group (ADR 0008), - /// for two routes that answer an empty `404` regardless. - /// - **A closed portal must not be able to reach the control plane at - /// all.** With the portal off there is no client in the process, so no - /// code path — not a bug, not a stray handler — can create or delete a - /// production API key. - /// - /// With the portal **open** a missing plan id is an error, matching - /// sign-in — see [`Self::load_portal_or_close`] for what the Lambda does - /// with it. A portal that renders an "issue key" button which answers - /// `503` is the thing to avoid, and a closed portal avoids it as surely as - /// a failed init does, without taking `/v1` down. + /// Fill [`Self::portal_keys`] for `serve.rs` — see + /// [`Self::load_portal_oauth`] for why eagerly and why only there. /// - /// # Where the plan id comes from - /// - /// `PORTAL_FREE_PLAN_PARAM` carries the **name of an SSM parameter**, not - /// the id — the parameter `ApiGatewayStack` publishes at - /// `/prices/{env}/pricing-api-free-plan-id` (task 0157). It cannot be a - /// cross-stack reference: `ComputeStack` is a dependency of - /// `ApiGatewayStack`, so importing the plan would close a cycle, which is - /// the same shape of problem `apiBaseUrl` has. And it must not be - /// hard-coded, because a usage-plan id is generated by AWS and changes if - /// the plan is ever replaced. - /// - /// # The API id and stage (task 0311) - /// - /// `Gateway::plan_of` keeps only the usage plans on OUR API stage, so the - /// client also needs the REST API id and the stage name. The id arrives - /// exactly as the plan id does — `PORTAL_API_ID_PARAM` names the SSM - /// parameter `ApiGatewayStack` publishes at `/prices/{env}/api-gateway-id`, - /// with a `PORTAL_API_ID` override for a local run compiled out of the - /// Lambda — and the stage is the plain `PORTAL_API_STAGE`, because the - /// stage name is `envName` and needs no lookup. + /// The flag guard also keeps a closed portal away from the control plane: + /// with the portal off there is no client in the process, so no code path + /// can create or delete a production API key. pub async fn load_portal_keys(&mut self) -> Result<(), PortalKeysError> { if !self.portal_enabled { return Ok(()); } - let plan_id = free_plan_id().await?; - let api_id = api_id().await?; - let stage = api_stage()?; - self.portal_keys = Some( - crate::portal::keys::gateway::Gateway::from_ambient_config(plan_id, api_id, stage) - .await, - ); + self.portal_keys = Some(portal_keys_from_env().await?); Ok(()) } - /// Fill [`Self::portal_eligibility`] with the sources of the eligibility - /// gate's two knobs (task 0189). - /// - /// Conditional on [`Self::portal_enabled`] for exactly the reasons the two - /// loaders above are. With the portal **open**, a missing source is an - /// error — a portal whose "get my key" round-trip can only ever answer - /// "could not verify" is worse than one that is closed — and so is an - /// unreadable or malformed *value*: both parameters are **probed once - /// here**, so `discord-guild-id` seeded with a name instead of a - /// snowflake, or `min-account-age-minutes` holding "five", is a cold-start - /// error with the parameter named, not a per-visitor refusal. What the - /// Lambda does with the error is [`Self::load_portal_or_close`]'s call. - /// - /// What is stored is the **source**, not the probed value: every issuance - /// resolves it again, which is what makes an operator's `put-parameter` - /// take effect without a redeploy (bounded only by the Parameters and - /// Secrets extension's ~5 min cache). - /// - /// # Where the values come from - /// - /// `PORTAL_GUILD_ID_PARAM` and `PORTAL_MIN_ACCOUNT_AGE_PARAM` carry the - /// **names of SSM parameters** (`/prices/{env}/discord-guild-id`, - /// `/prices/{env}/min-account-age-minutes`), seeded by the operator at - /// deploy prep — never created by CDK, because a CloudFormation-managed - /// parameter is restored to the committed value by the next `cdk deploy`, - /// which would silently un-flip production back to the test guild after - /// [0179]. The direct-value overrides are local-only seams, compiled out - /// of the Lambda like `PORTAL_FREE_PLAN_ID`. + /// Fill [`Self::portal_eligibility`] for `serve.rs` — see + /// [`Self::load_portal_oauth`]. pub async fn load_portal_eligibility(&mut self) -> Result<(), PortalEligibilityError> { if !self.portal_enabled { return Ok(()); } - let settings = eligibility_settings()?; - // Probe both values now. The per-action resolve keeps them tunable; - // this makes a bad seed loud at deploy time. - settings - .guild_id() - .await - .map_err(PortalEligibilityError::Probe)?; - settings - .min_account_age_minutes() - .await - .map_err(PortalEligibilityError::Probe)?; - self.portal_eligibility = Some(settings); + self.portal_eligibility = Some(portal_eligibility_from_env().await?); Ok(()) } +} - /// The three loaders above, in order, stopping at the first error. - async fn load_portal(&mut self) -> Result<(), PortalLoadError> { - self.load_portal_oauth().await?; - self.load_portal_keys().await?; - self.load_portal_eligibility().await?; - Ok(()) - } +/// Read the Discord OAuth secret (task 0186) from Secrets Manager, or from the +/// local file named by `PORTAL_OAUTH_SECRET_FILE`. With the portal open, no +/// source at all is an error. +pub(crate) async fn portal_oauth_from_env() +-> Result { + crate::portal::auth::secret::OauthSecret::load() + .await? + .ok_or(crate::portal::auth::secret::SecretError::NoSource) +} - /// Load every portal source, or close the portal in this process. - /// - /// The three loaders each return their error; this is where the Lambda - /// decides what an error *means*, and the decision is **closed, not - /// crashed** (task 0194, PR review finding 1). `main.rs` used to - /// `expect()` each loader, on the argument that a portal source missing at - /// deploy should fail loudly in `Init Errors` rather than as a `503` under - /// a sign-in button. Three things were wrong with that: - /// - /// - **The loud failure lands on `/v1`.** One router serves every route - /// group (ADR 0008), so an init panic is not "the portal fails to - /// deploy" — `cdk deploy` succeeds regardless — it is the next `/v1` - /// caller receiving a `502`, over a secret and three SSM parameters the - /// data API never uses. - /// - **It was not only a deploy hazard.** The reads go through the - /// Parameters and Secrets extension with a 2 s timeout and no retry - /// (`prices_clickhouse::mtls`), and Parameter Store's default throughput - /// is 40 TPS for the whole account. A burst of cold starts — the ramp of - /// a load test — is three SSM reads per environment against that budget, - /// and a throttled one was a `502` on the data API. - /// - **Nobody was paged by `Init Errors`.** When this was written the - /// api-handler had no alarm on `Errors` at all, so the "loud" failure - /// was loud only to whoever probed. The api-handler now has - /// `prices-${env}-api-handler-errors`, and the closure this function - /// produces instead pages as `prices-${env}-api-handler-portal-closed` - /// (task 0249). - /// - /// So on any error the portal is closed *in this execution environment* — - /// the flag cleared and all three sources dropped, which restores every - /// property of a closed portal (the gate answers before any handler, and - /// there is no control-plane client in the process) — and the error is - /// returned for the caller to log. `/config` then reports - /// `enabled: false`, which is the probe the deploy runbook already makes - /// after every deploy, so a misconfigured deploy is caught by the same step - /// it always was; `/v1` never notices. - /// - /// The cost, stated: an environment that failed a *transient* read stays - /// closed for its lifetime, where a panic would have discarded it and - /// retried on the next cold start. That trades a `502` on the data API for - /// a portal that, in one environment, says it is not open. The alarm on - /// the log line `main.rs` writes is the follow-up recorded on task 0194. - /// - /// `serve.rs` keeps its three `expect()`s on purpose: a developer who asked - /// for the portal and did not get it wants to know now, and no partner is - /// behind that process. - pub async fn load_portal_or_close(&mut self) -> Result<(), PortalLoadError> { - let loaded = self.load_portal().await; - if loaded.is_err() { - self.close_portal(); - } - loaded - } +/// Build the control-plane client key issuance uses (task 0187). +/// +/// # Where the plan id comes from +/// +/// `PORTAL_FREE_PLAN_PARAM` carries the **name of an SSM parameter**, not +/// the id — the parameter `ApiGatewayStack` publishes at +/// `/prices/{env}/pricing-api-free-plan-id` (task 0157). It cannot be a +/// cross-stack reference: `ComputeStack` is a dependency of +/// `ApiGatewayStack`, so importing the plan would close a cycle, which is +/// the same shape of problem `apiBaseUrl` has. And it must not be +/// hard-coded, because a usage-plan id is generated by AWS and changes if +/// the plan is ever replaced. +/// +/// # The API id and stage (task 0311) +/// +/// `Gateway::plan_of` keeps only the usage plans on OUR API stage, so the +/// client also needs the REST API id and the stage name. The id arrives +/// exactly as the plan id does — `PORTAL_API_ID_PARAM` names the SSM +/// parameter `ApiGatewayStack` publishes at `/prices/{env}/api-gateway-id`, +/// with a `PORTAL_API_ID` override for a local run compiled out of the +/// Lambda — and the stage is the plain `PORTAL_API_STAGE`, because the +/// stage name is `envName` and needs no lookup. The two reads run +/// concurrently. +pub(crate) async fn portal_keys_from_env() +-> Result { + let (plan_id, api_id) = tokio::try_join!(free_plan_id(), api_id())?; + let stage = api_stage()?; + Ok(crate::portal::keys::gateway::Gateway::from_ambient_config(plan_id, api_id, stage).await) +} - /// Close the portal in this process: the flag AND the three sources, so a - /// half-loaded configuration (secret read, plan id not) leaves nothing - /// behind that a closed portal would not have — in particular, no - /// control-plane client. - fn close_portal(&mut self) { - self.portal_enabled = false; - self.portal_oauth = None; - self.portal_keys = None; - self.portal_eligibility = None; - } +/// The sources of the eligibility gate's two knobs (task 0189), each value +/// **probed once** so a mis-seeded parameter — `discord-guild-id` seeded +/// with a name instead of a snowflake, `min-account-age-minutes` holding +/// "five" — fails the load with the parameter named, rather than refusing +/// every visitor as "could not verify". +/// +/// What is kept is the **source**, not the probed value: every issuance +/// resolves it again, which is what makes an operator's `put-parameter` +/// take effect without a redeploy (bounded only by the Parameters and +/// Secrets extension's ~5 min cache). +/// +/// # Where the values come from +/// +/// `PORTAL_GUILD_ID_PARAM` and `PORTAL_MIN_ACCOUNT_AGE_PARAM` carry the +/// **names of SSM parameters** (`/prices/{env}/discord-guild-id`, +/// `/prices/{env}/min-account-age-minutes`), seeded by the operator at +/// deploy prep — never created by CDK, because a CloudFormation-managed +/// parameter is restored to the committed value by the next `cdk deploy`, +/// which would silently un-flip production back to the test guild after +/// [0179]. The direct-value overrides are local-only seams, compiled out +/// of the Lambda like `PORTAL_FREE_PLAN_ID`. +pub(crate) async fn portal_eligibility_from_env() +-> Result { + let settings = eligibility_settings()?; + tokio::try_join!(settings.guild_id(), settings.min_account_age_minutes()) + .map_err(PortalEligibilityError::Probe)?; + Ok(settings) +} + +/// The Lambda's portal load (`crate::portal::sources`): all five reads — the +/// OAuth secret, the free-plan id, the API id, the guild id and the minimum +/// account age — in flight at once, failing on the first error. +/// +/// Concurrent because the whole load sits inside +/// `crate::portal::sources::LOAD_BUDGET`, in front of the request that asked: +/// five reads one after another could each take the extension client's 2 s. +pub(crate) async fn load_portal_sources() -> Result +{ + let (oauth, gateway, settings) = tokio::try_join!( + async { portal_oauth_from_env().await.map_err(PortalLoadError::from) }, + async { portal_keys_from_env().await.map_err(PortalLoadError::from) }, + async { + portal_eligibility_from_env() + .await + .map_err(PortalLoadError::from) + }, + )?; + Ok(crate::portal::sources::Loaded { + oauth: Some(std::sync::Arc::new(oauth)), + gateway: Some(std::sync::Arc::new(gateway)), + settings: Some(std::sync::Arc::new(settings)), + }) } -/// Which portal source failed to load at cold start — the value -/// [`AppConfig::load_portal_or_close`] hands back, with the variable that -/// names the source, so the log line points at the runbook step. +/// Why the lazy portal load failed (`crate::portal::sources`): which source, +/// with the variable that names it, so the `portal sources failed to load` +/// line points at the runbook step. #[derive(Debug, thiserror::Error)] pub enum PortalLoadError { #[error("portal sign-in (PORTAL_OAUTH_SECRET_NAME): {0}")] Oauth(#[from] crate::portal::auth::secret::SecretError), - #[error("portal key issuance (PORTAL_FREE_PLAN_PARAM): {0}")] + #[error("portal key issuance ({var}): {0}", var = .0.variable())] Keys(#[from] PortalKeysError), #[error("portal eligibility gate (PORTAL_GUILD_ID_PARAM, PORTAL_MIN_ACCOUNT_AGE_PARAM): {0}")] Eligibility(#[from] PortalEligibilityError), + #[error( + "the portal's five reads (PORTAL_OAUTH_SECRET_NAME, PORTAL_FREE_PLAN_PARAM, \ + PORTAL_API_ID_PARAM, PORTAL_GUILD_ID_PARAM, PORTAL_MIN_ACCOUNT_AGE_PARAM) did not \ + finish within {0:?}" + )] + TimedOut(std::time::Duration), } -/// Why the eligibility gate could not be configured at cold start. +/// Why the eligibility gate could not be configured. #[derive(Debug, thiserror::Error)] pub enum PortalEligibilityError { #[error( @@ -462,7 +387,7 @@ fn eligibility_settings() }) } -/// Why key issuance could not be configured at cold start. +/// Why key issuance could not be configured. #[derive(Debug, thiserror::Error)] pub enum PortalKeysError { #[error( @@ -494,6 +419,23 @@ pub enum PortalKeysError { NoStage, } +impl PortalKeysError { + /// The variable naming the source that failed — the label on + /// [`PortalLoadError::Keys`]. Per variant, because the API id and the + /// stage are read from variables of their own (task 0311). + pub fn variable(&self) -> &'static str { + match self { + PortalKeysError::NoSource + | PortalKeysError::Fetch { .. } + | PortalKeysError::Empty { .. } => "PORTAL_FREE_PLAN_PARAM", + PortalKeysError::ApiIdNoSource + | PortalKeysError::ApiIdFetch { .. } + | PortalKeysError::ApiIdEmpty { .. } => "PORTAL_API_ID_PARAM", + PortalKeysError::NoStage => "PORTAL_API_STAGE", + } + } +} + /// Resolve the `pricing-api-free` usage-plan id. async fn free_plan_id() -> Result { // A direct id, for a local run against a real account. Checked first so a @@ -649,53 +591,83 @@ mod web_origin_tests { #[cfg(test)] mod portal_load_tests { - use super::AppConfig; - - /// The unit under test is the decision, not the loaders: with the portal - /// open and no source configured, a loader fails and the config must come - /// back CLOSED — flag and all three sources — rather than half-open. No - /// environment variable is set here on purpose (`set_var` races the other - /// test threads, see `AppConfig::portal_endpoints`): the loaders read - /// `PORTAL_OAUTH_SECRET_FILE`/`_NAME`, `PORTAL_FREE_PLAN_ID`/`_PARAM`, - /// `PORTAL_API_ID`/`_PARAM`, `PORTAL_API_STAGE` and - /// the eligibility seams, and a developer's shell exporting one of them - /// only moves which loader fails, not the outcome asserted. + use super::{AppConfig, PortalKeysError, PortalLoadError, load_portal_sources}; + + /// With no source configured, the lazy load fails, and says which + /// source. No environment variable is set here on purpose (`set_var` + /// races the other test threads, see `AppConfig::portal_endpoints`): the + /// loaders read `PORTAL_OAUTH_SECRET_FILE`/`_NAME`, + /// `PORTAL_FREE_PLAN_ID`/`_PARAM`, `PORTAL_API_ID`/`_PARAM`, + /// `PORTAL_API_STAGE` and the eligibility seams, and a developer's shell + /// exporting one of them only moves which source fails, not the outcome + /// asserted. #[tokio::test] - async fn a_failed_load_closes_the_portal_and_keeps_none_of_its_sources() { - let mut config = AppConfig { - portal_enabled: true, - ..AppConfig::from_env() + async fn a_load_with_no_source_fails_and_names_the_source() { + let err = match load_portal_sources().await { + Ok(_) => panic!("no portal source is configured in a unit test"), + Err(err) => err, }; - - let err = config - .load_portal_or_close() - .await - .expect_err("no portal source is configured in a unit test"); - - assert!( - !config.portal_enabled, - "the portal must be closed after a failed load ({err})" - ); - assert!(config.portal_oauth.is_none()); - assert!(config.portal_keys.is_none()); - assert!(config.portal_eligibility.is_none()); // The message names the variable the runbook step sets. assert!(err.to_string().starts_with("portal "), "{err}"); } + /// The label names the variable of the read that failed — not + /// `PORTAL_FREE_PLAN_PARAM` for all seven, which it used to. + #[test] + fn a_key_issuance_failure_names_its_own_variable() { + let name = || "/prices/production/x".to_string(); + let cases = [ + (PortalKeysError::NoSource, "PORTAL_FREE_PLAN_PARAM"), + ( + PortalKeysError::Fetch { + name: name(), + message: "m".into(), + }, + "PORTAL_FREE_PLAN_PARAM", + ), + ( + PortalKeysError::Empty { name: name() }, + "PORTAL_FREE_PLAN_PARAM", + ), + (PortalKeysError::ApiIdNoSource, "PORTAL_API_ID_PARAM"), + ( + PortalKeysError::ApiIdFetch { + name: name(), + message: "m".into(), + }, + "PORTAL_API_ID_PARAM", + ), + ( + PortalKeysError::ApiIdEmpty { name: name() }, + "PORTAL_API_ID_PARAM", + ), + (PortalKeysError::NoStage, "PORTAL_API_STAGE"), + ]; + for (err, var) in cases { + let line = PortalLoadError::Keys(err).to_string(); + assert!( + line.starts_with(&format!("portal key issuance ({var}): ")), + "{line}" + ); + } + } + + /// `serve.rs`'s eager loaders are no-ops on a closed portal: nothing is + /// read, nothing fails, nothing is filled. #[tokio::test] - async fn a_closed_portal_loads_nothing_and_stays_closed() { + async fn a_closed_portal_loads_nothing() { let mut config = AppConfig { portal_enabled: false, ..AppConfig::from_env() }; + config.load_portal_oauth().await.expect("nothing to load"); + config.load_portal_keys().await.expect("nothing to load"); config - .load_portal_or_close() + .load_portal_eligibility() .await - .expect("a closed portal has nothing to load and nothing to fail"); + .expect("nothing to load"); - assert!(!config.portal_enabled); assert!(config.portal_oauth.is_none()); assert!(config.portal_keys.is_none()); assert!(config.portal_eligibility.is_none()); diff --git a/packages/prices-api/src/lib.rs b/packages/prices-api/src/lib.rs index 000a62a1..aa66a55a 100644 --- a/packages/prices-api/src/lib.rs +++ b/packages/prices-api/src/lib.rs @@ -46,7 +46,35 @@ pub use state::AppState; /// 2. Stamp `servers` from `config.base_url`, expose the spec at /// `GET /api-docs-json`. /// 3. Layer the in-app API-key gate (armed only when `API_KEYS` is set). +/// +/// Building it reads nothing: with the portal open and no source supplied, +/// the portal's sources load on the first portal request that needs them +/// (`portal::sources`). pub fn app(config: &AppConfig, state: AppState) -> Router { + app_inner(config, state, portal::sources::sources_for(config)) +} + +/// [`app`], with the portal's sources supplied by the caller — already loaded +/// ([`portal::sources::PortalSources::ready`]) or behind a loader of the +/// test's choosing ([`portal::sources::PortalSources::lazy`]). +/// +/// Compiled out of the Lambda, like `Gateway::against` and +/// `IssueDeps::with_deadline`: the deployed build contains one loader, the one +/// that reads the environment. +#[cfg(not(feature = "lambda"))] +pub fn app_with_portal( + config: &AppConfig, + state: AppState, + sources: portal::sources::PortalSources, +) -> Router { + app_inner(config, state, sources) +} + +fn app_inner( + config: &AppConfig, + state: AppState, + sources: portal::sources::PortalSources, +) -> Router { let (router, mut spec) = openapi::register_routes() .with_state(state) .split_for_parts(); @@ -127,7 +155,7 @@ pub fn app(config: &AppConfig, state: AppState) -> Router { // Portal routes before the key gate, and exempt from it: a visitor signing // in has no API key by definition (task 0183). The gate inside `portal` // decides whether they are served at all. - let router = portal::apply(router, config); + let router = portal::apply_with(router, config, sources); auth::apply(router, config) } diff --git a/packages/prices-api/src/main.rs b/packages/prices-api/src/main.rs index f192c96c..3cef8d0c 100644 --- a/packages/prices-api/src/main.rs +++ b/packages/prices-api/src/main.rs @@ -24,43 +24,14 @@ async fn main() { .with_target(false) .init(); - let mut config = AppConfig::from_env(); + let config = AppConfig::from_env(); - // The portal's three sources, read through the same Parameters & Secrets - // extension as the mTLS bundle below — so no secret VALUE is ever an - // environment variable (ADR 0007, Tranche 3 AC 6): - // - // - sign-in credentials (task 0186): the Discord OAuth secret; - // - key issuance (task 0187): the `pricing-api-free` usage-plan id from - // SSM, plus the API Gateway control-plane client built from the - // execution role's credentials; - // - the eligibility gate (task 0189): the SSM parameter NAMES for the guild - // id and the minimum account age — resolved per issuance so operator - // changes need no redeploy — each probed once here so a mis-seeded - // parameter is found now and not at a visitor's click. - // - // A no-op while `PORTAL_ENABLED` is false; with it true (task 0194) all - // three are read at every cold start, and the reads are four: one secret, - // three parameters. See the deploy-gate note on `PORTAL_ENABLED` in - // `compute-stack.ts`. - // - // **Closed, not crashed.** A failed read closes the portal in this - // execution environment and is logged here; it does not panic init. This - // Lambda also serves `/v1`, and an init panic is a `502` to the next data - // API caller — for sources `/v1` never uses, read with no retry against a - // 40 TPS account-wide Parameter Store budget. The reasoning and the cost - // are on `AppConfig::load_portal_or_close`. The log line below is one - // signal a misconfigured or throttled deploy leaves; `/config` answering - // `enabled: false` is the other, and it is the probe the deploy runbook - // makes. - if let Err(err) = config.load_portal_or_close().await { - tracing::error!( - error = %err, - "portal closed at cold start: a portal source failed to load; /v1 is \ - unaffected, and the portal answers as closed in this execution \ - environment until it is recycled" - ); - } + // The cold start reads one source: the mTLS bundle below. The portal's + // five (the Discord OAuth secret and four SSM parameters) load on the + // first portal request that needs them (`prices_api::portal::sources`), + // so a burst of `/v1` cold starts makes no Parameter Store read, and a + // failed portal read costs that one request rather than the environment + // (task 0311). // Build the CH client eagerly at cold start; it is Arc-backed and shared via // AppState across warm invocations. `client_from_lambda_env` reads diff --git a/packages/prices-api/src/portal/auth/issue.rs b/packages/prices-api/src/portal/auth/issue.rs index 454b7ea7..c27de825 100644 --- a/packages/prices-api/src/portal/auth/issue.rs +++ b/packages/prices-api/src/portal/auth/issue.rs @@ -75,6 +75,7 @@ use super::session::{self, Session}; use super::state_token; use super::{AuthState, cookies, redirect}; use crate::portal::keys::gateway::Gateway; +use crate::portal::sources::{Loaded, PortalSources}; /// See the module table. Literals, like `?signin=…` — the dynamic one is /// [`too_young_query`], whose only variable part is a `u64` rendered in @@ -132,9 +133,10 @@ pub(super) fn capped_query(next_eligible_date: &str) -> String { /// service is" — which is exactly true here, and true *before* any check ran, /// which is why that state's copy does not claim eligibility passed. /// -/// Loud in CloudWatch, because this is a deployment fault and nothing else -/// reports it: the cold-start probes are supposed to make it unreachable, so -/// one of these lines means a container came up in a state they did not catch. +/// Loud in CloudWatch, because this is a deployment fault: the portal's load +/// yields all its sources or fails, so one of these lines follows either a +/// `portal sources failed to load` line for the same request (all three +/// flags `false`) or a state the load did not catch. pub(super) fn refuse_issue_start( home: &str, oauth: bool, @@ -213,6 +215,20 @@ pub(super) fn refuse_issue_discord( /// [`RECONCILE_FLOOR`] for what happens when that is not enough. const ISSUE_BUDGET: Duration = Duration::from_secs(12); +/// The callback's share of the invocation for loading the portal's sources +/// (task 0311), measured from arrival like [`ISSUE_BUDGET`]. +/// +/// The sources load on the first portal request an execution environment +/// sees, and the callback can be that request: login and callback often land +/// on different environments. The arithmetic on `discord::REQUEST_TIMEOUT` +/// already spends 14 s of the 15 s invocation on the exchange, the parameter +/// reads and two Discord reads, so the load gets what is left — under a +/// second, with room for the redirect. A callback that arrives here later +/// than this lands on a retryable failure BEFORE the token exchange, and the +/// next attempt finds the sources loaded. `budget_arithmetic_fits_the_lambda` +/// pins the sum. +pub(super) const SOURCES_ALLOWANCE: Duration = Duration::from_millis(500); + /// The least time worth starting a reconciliation with. /// /// A reconciliation needs at least a list, and then either an adoption (one @@ -240,16 +256,15 @@ const _: () = assert!( /// Everything the issue arm needs beyond what sign-in already carries. /// -/// All optional, like `AuthState::oauth` and `KeysState::gateway`, and for the -/// same reason: the api-handler boots with the portal closed and nothing -/// provisioned. `config::load_portal_or_close` closes the portal at cold start -/// when it is *open* with these missing, so a `None` here in production -/// means the portal is closed and the gate answers before any handler does. +/// The control-plane client and the eligibility settings live in `sources`, +/// the same lazily loaded cell `AuthState` holds (see `crate::portal::sources`). +/// Read only after the handler's own `AuthState` read succeeded, so in +/// production it is already loaded and answers at once; a fixture with +/// neither is "unwired", which every caller already refuses in its own words. #[derive(Clone)] pub struct IssueDeps { - pub(super) gateway: Option>, + sources: PortalSources, pub(super) usage_cache: Option, - pub(super) settings: Option>, /// The same wall-clock ceiling 0187's handler put on the reconciliation. pub(super) deadline: Duration, } @@ -257,9 +272,8 @@ pub struct IssueDeps { impl Default for IssueDeps { fn default() -> Self { Self { - gateway: None, + sources: PortalSources::ready(Loaded::default()), usage_cache: None, - settings: None, deadline: keys::RECONCILE_DEADLINE, } } @@ -272,13 +286,29 @@ impl IssueDeps { settings: Option, ) -> Self { Self { - gateway: gateway.map(Arc::new), + sources: PortalSources::ready(Loaded { + gateway: gateway.map(Arc::new), + settings: settings.map(Arc::new), + ..Loaded::default() + }), usage_cache, - settings: settings.map(Arc::new), deadline: keys::RECONCILE_DEADLINE, } } + /// Read the client and the settings from `sources` instead of the + /// constructor's arguments — how `portal::apply` shares one cell. + pub(crate) fn with_sources(mut self, sources: PortalSources) -> Self { + self.sources = sources; + self + } + + /// The loaded sources, or none. Called only after the caller's own + /// `AuthState` read succeeded, which in production is this same cell. + pub(super) async fn loaded(&self) -> Loaded { + self.sources.get().await.cloned().unwrap_or_default() + } + /// Shorten the deadline for tests — compiled out of the Lambda for the /// reason `KeysState::with_deadline` is: a deployed build must contain no /// way to set this to something that lets a slow control plane outlive @@ -288,11 +318,11 @@ impl IssueDeps { self.deadline = deadline; self } +} - /// Whether an issue round-trip could complete on this deployment. - pub(super) fn is_wired(&self) -> bool { - self.gateway.is_some() && self.settings.is_some() - } +/// Whether an issue round-trip could complete with these sources. +pub(super) fn is_wired(loaded: &Loaded) -> bool { + loaded.gateway.is_some() && loaded.settings.is_some() } /// Finish an `action=issue` callback: check, then issue, then land. @@ -318,7 +348,8 @@ pub(super) async fn complete_issue( // honoured without a redeploy. Failure is `unknown`, not a 5xx: the // visitor is mid-navigation, the fault is ours, and "could not verify" // is the honest refusal that does not accuse them of anything. - let verified = match state.issue.settings.as_deref() { + let wiring = state.issue.loaded().await; + let verified = match wiring.settings.as_deref() { None => { // `login` refuses `action=issue` on an unwired deployment, so // arriving here means the deployment changed under an in-flight @@ -467,7 +498,8 @@ use Landing::*; /// /// `started` is stamped when the **request** arrived — see [`ISSUE_BUDGET`]. async fn issue(state: &AuthState, user_id: &str, started: Instant) -> Landing { - let Some(gateway) = state.issue.gateway.as_deref() else { + let wiring = state.issue.loaded().await; + let Some(gateway) = wiring.gateway.as_deref() else { return Unwired; }; @@ -603,10 +635,18 @@ mod tests { /// Gateway's bare `502` in place of every screen this module lands on. /// The 15 is `apiHandler.timeoutSeconds` in `infra/envs/production.json`; /// raise either constant, or add a call, and this is what fails first. + /// + /// Since task 0311 a lazy load of the portal's sources can precede all of + /// it; past [`SOURCES_ALLOWANCE`] the callback lands before the exchange, + /// so the allowance is the term that load adds. A load that fails + /// outright is followed only by a redirect, so the whole + /// `sources::LOAD_BUDGET` has to fit too. #[test] fn budget_arithmetic_fits_the_lambda() { const LAMBDA_TIMEOUT: std::time::Duration = std::time::Duration::from_secs(15); - let worst = discord::REQUEST_TIMEOUT * 3 + crate::portal::eligibility::PARAMETER_TIMEOUT; + let worst = SOURCES_ALLOWANCE + + discord::REQUEST_TIMEOUT * 3 + + crate::portal::eligibility::PARAMETER_TIMEOUT; assert!( worst < LAMBDA_TIMEOUT, "worst case {worst:?} does not fit inside {LAMBDA_TIMEOUT:?}" @@ -614,6 +654,7 @@ mod tests { // And the reconciler's share is measured from arrival, so it cannot // extend the callback past the same line. assert!(ISSUE_BUDGET < LAMBDA_TIMEOUT); + assert!(crate::portal::sources::LOAD_BUDGET < LAMBDA_TIMEOUT); } /// Every landing state is a distinct literal under the portal home, @@ -661,10 +702,10 @@ mod tests { assert!(value.bytes().all(|b| b.is_ascii_digit())); } - #[test] - fn issue_deps_default_to_unwired_with_the_production_deadline() { + #[tokio::test] + async fn issue_deps_default_to_unwired_with_the_production_deadline() { let deps = IssueDeps::default(); - assert!(!deps.is_wired()); + assert!(!is_wired(&deps.loaded().await)); assert_eq!(deps.deadline, keys::RECONCILE_DEADLINE); } } diff --git a/packages/prices-api/src/portal/auth/mod.rs b/packages/prices-api/src/portal/auth/mod.rs index 7cac0b7c..3105cbb1 100644 --- a/packages/prices-api/src/portal/auth/mod.rs +++ b/packages/prices-api/src/portal/auth/mod.rs @@ -72,6 +72,7 @@ use crate::common::extract::ValidatedQuery; use crate::common::{cache_control, errors}; use super::eligibility; +use super::sources::{Loaded, PortalSources}; use secret::OauthSecret; use session::Session; use state_token::{Action, StateError}; @@ -175,15 +176,14 @@ const SIGN_IN_MISCONFIGURED: &str = "sign_in_misconfigured"; /// Everything the four handlers share, cloned per request. /// -/// `oauth` is `Option` because the api-handler must boot without it. Production -/// runs with `PORTAL_ENABLED=false` for the whole of the build, and a cold start -/// that insisted on reading a secret nobody has created yet would fail Lambda -/// init and take out `/v1` — the data API — to protect a portal that answers -/// `404` regardless. [`crate::AppConfig::load_portal_oauth`] therefore only -/// loads when the portal is open, and only then is a missing secret fatal. +/// The OAuth secret lives in `sources`, which the api-handler loads on the +/// first portal request that needs it rather than at cold start — so `/v1` +/// never waits on it, and a failed read costs one request (see +/// `crate::portal::sources`). Each handler asks `sources` FIRST and answers a +/// failed load in its own words. #[derive(Clone)] pub struct AuthState { - oauth: Option>, + sources: PortalSources, endpoints: std::sync::Arc, http: reqwest::Client, /// What the `action=issue` round-trip needs beyond sign-in (task 0189). @@ -199,7 +199,10 @@ pub struct AuthState { impl AuthState { pub fn new(oauth: Option, endpoints: discord::Endpoints) -> Self { Self { - oauth: oauth.map(std::sync::Arc::new), + sources: PortalSources::ready(Loaded { + oauth: oauth.map(std::sync::Arc::new), + ..Loaded::default() + }), endpoints: std::sync::Arc::new(endpoints), http: discord::build_client(), issue: issue::IssueDeps::default(), @@ -223,6 +226,14 @@ impl AuthState { self } + /// Read the OAuth secret from `sources` instead of the constructor's + /// argument — how [`super::apply`] hands every portal state one shared + /// cell. + pub(crate) fn with_sources(mut self, sources: PortalSources) -> Self { + self.sources = sources; + self + } + /// Wire the issue round-trip's dependencies in. A builder, like /// `KeysState::with_usage_cache`, so every existing constructor and test /// stays valid. @@ -289,22 +300,31 @@ async fn login( }, }; + // The sources could not be loaded for this request: the same landings an + // unprovisioned deployment gets, and the next press loads again. The + // failure itself was logged by `get`. + let Some(loaded) = state.sources.get().await else { + return match action { + Action::Issue => issue::refuse_issue_start(&state.home, false, false, false), + _ => unconfigured(&state.home), + }; + }; + // An issue round-trip on a deployment with no credentials, no control // plane or no eligibility parameters cannot end in a key — refuse before - // sending the visitor to Discord. Only reachable with the portal open and - // issuance unprovisioned: `load_portal_oauth` and `load_portal_eligibility` - // both fail the cold start on that combination, so this is the second - // line, not the first. - if action == Action::Issue && (state.oauth.is_none() || !state.issue.is_wired()) { + // sending the visitor to Discord. The production load yields all three or + // fails, so this is reachable only with a partial fixture. + let wiring = state.issue.loaded().await; + if action == Action::Issue && (loaded.oauth.is_none() || !issue::is_wired(&wiring)) { return issue::refuse_issue_start( &state.home, - state.oauth.is_some(), - state.issue.gateway.is_some(), - state.issue.settings.is_some(), + loaded.oauth.is_some(), + wiring.gateway.is_some(), + wiring.settings.is_some(), ); } - let Some(oauth) = state.oauth.as_ref() else { + let Some(oauth) = loaded.oauth.as_deref() else { return unconfigured(&state.home); }; @@ -429,7 +449,14 @@ async fn callback( let started = std::time::Instant::now(); let home = state.home.as_ref(); - let Some(oauth) = state.oauth.as_ref() else { + // A failed load lands on `?signin=failed`: which flow this was cannot be + // known before `state` is verified, and verifying it needs the very + // secret that failed to load. The pending cookie is left alone, as on + // every refusal before verification; the next attempt loads again. + let Some(loaded) = state.sources.get().await else { + return redirect(&format!("{home}{FAILED_QUERY}"), vec![]); + }; + let Some(oauth) = loaded.oauth.as_deref() else { return unconfigured(home); }; @@ -537,6 +564,26 @@ async fn callback( } } + // A lazy load in front of this callback (the first portal request in + // this environment) spent time the arithmetic on `discord::REQUEST_TIMEOUT` + // does not have. Past `issue::SOURCES_ALLOWANCE` the exchange and the + // three reads after it could outlive the invocation, so land a + // retryable failure now, before any Discord call: the next attempt finds + // the sources loaded. + if started.elapsed() > issue::SOURCES_ALLOWANCE { + let query = match accepted.action { + Action::Issue => issue::ISSUE_FAILED_QUERY, + _ => FAILED_QUERY, + }; + tracing::warn!( + elapsed_ms = started.elapsed().as_millis() as u64, + landing = query, + "sign-in callback spent its allowance loading the portal sources; \ + landing a retryable failure before the token exchange" + ); + return redirect(&format!("{home}{query}"), vec![drop_pending]); + } + let token = match discord::exchange_code( &state.http, &state.endpoints, @@ -601,8 +648,8 @@ async fn callback( // members included, over a value the sign-in itself never consults. The // age read failing now costs the visitor the key half only: they are // signed in, land plain, and the dashboard's issue control re-asks. - let (checked, min_age): (Option, Option) = match state - .issue + let wiring = state.issue.loaded().await; + let (checked, min_age): (Option, Option) = match wiring .settings .as_deref() { @@ -734,12 +781,21 @@ struct MeResponse { username: Option, } +/// Error code for `/auth/me` when the portal's sources could not be loaded +/// for this request. +const SESSION_UNAVAILABLE: &str = "session_unavailable"; + /// Report the caller's session. /// /// `200` with `authenticated: false` rather than `401`, deliberately. This is /// the question "am I signed in?", and refusing to answer it while signed out is /// as circular as [0183]'s `/config` refusing to say the portal is closed. The /// page asks it on every load and renders plain text either way. +/// +/// The one `503`: the sources failed to load for this request (task 0311). +/// Without the signing key no cookie can be checked, and "signed out" would +/// be a lie to a visitor who is signed in — the page renders its failure +/// state instead, and the next call loads again. async fn me(State(state): State, headers: HeaderMap) -> Response { let signed_out = MeResponse { authenticated: false, @@ -747,7 +803,13 @@ async fn me(State(state): State, headers: HeaderMap) -> Response { username: None, }; - let Some(oauth) = state.oauth.as_ref() else { + let Some(loaded) = state.sources.get().await else { + return no_store(errors::service_unavailable( + SESSION_UNAVAILABLE, + "portal sign-in is temporarily unavailable", + )); + }; + let Some(oauth) = loaded.oauth.as_deref() else { // Not `unconfigured()`: a deployment with no credentials has no // sessions, which is a truthful answer to this question and lets the // page render rather than showing an error it can do nothing about. @@ -848,12 +910,9 @@ fn no_store(mut response: Response) -> Response { response } -/// Land a deployment that reached these routes with no credentials. -/// -/// Only reachable if `PORTAL_ENABLED` is true and the secret is missing — -/// `AppConfig::load_portal_or_close` closes the portal on exactly that -/// combination, so the gate answers first and this is a second line rather -/// than the first. +/// Land a deployment that reached these routes with no credentials — the +/// portal open, and its sources either loaded without a secret (a test +/// fixture) or, on `login`, failed to load for this request. /// /// **A landing, not the `503 sign_in_unconfigured` envelope it used to be** /// (task 0194's review). Both call sites are reached by a browser following a diff --git a/packages/prices-api/src/portal/keys/mod.rs b/packages/prices-api/src/portal/keys/mod.rs index 191f836b..61670d26 100644 --- a/packages/prices-api/src/portal/keys/mod.rs +++ b/packages/prices-api/src/portal/keys/mod.rs @@ -109,6 +109,7 @@ use crate::common::{cache_control, errors}; use super::auth::secret::OauthSecret; use super::period::Period; +use super::sources::{Loaded, PortalSources}; use cap::Cap; use gateway::{Attachment, Disable, Gateway, GatewayError, KeyValue}; use naming::{ @@ -194,18 +195,19 @@ pub(crate) const RECONCILE_DEADLINE: Duration = Duration::from_secs(10); /// What both handlers need, cloned per request. /// -/// Both fields are `Option` for the same reason `AuthState::oauth` is: the -/// api-handler must boot with the portal closed and nothing provisioned. The -/// routes are mounted regardless, and answer `503` rather than not existing, so +/// The OAuth secret and the control-plane client live in `sources`, loaded on +/// the first portal request that needs them (`crate::portal::sources`). The +/// routes are mounted regardless, and answer `503` rather than not existing — +/// when the load failed for this request, and when it yielded no client — so /// that a deployment which opens the portal without wiring the usage plan says /// so instead of looking like a portal with no key issuance. #[derive(Clone)] pub struct KeysState { - /// Verifies the session cookie. The same secret sign-in issued it with — - /// there is one signing key and [`super::auth::crypto`]'s domain separation - /// is what keeps its three token kinds apart. - oauth: Option>, - gateway: Option>, + /// The OAuth secret — which verifies the session cookie, the same secret + /// sign-in issued it with; there is one signing key and + /// [`super::auth::crypto`]'s domain separation is what keeps its three + /// token kinds apart — and the control-plane client. + sources: PortalSources, /// [`RECONCILE_DEADLINE`], overridable only outside the Lambda build. deadline: Duration, /// Task 0188's usage cache, so a successful issue can evict a cached @@ -223,14 +225,24 @@ pub struct KeysState { impl KeysState { pub fn new(oauth: Option, gateway: Option) -> Self { Self { - oauth: oauth.map(Arc::new), - gateway: gateway.map(Arc::new), + sources: PortalSources::ready(Loaded { + oauth: oauth.map(Arc::new), + gateway: gateway.map(Arc::new), + ..Loaded::default() + }), deadline: RECONCILE_DEADLINE, usage_cache: None, web_origin: None, } } + /// Read the secret and the client from `sources` instead of the + /// constructor's arguments — how [`super::apply`] shares one cell. + pub(crate) fn with_sources(mut self, sources: PortalSources) -> Self { + self.sources = sources; + self + } + /// Name the bundle's origin, so a revoke from it — same-site, not /// same-origin, once the bundle lives on its own host (task 0194) — is /// accepted. Only that one origin; see [`is_same_origin_write`]. A @@ -330,10 +342,13 @@ async fn key(State(state): State, headers: HeaderMap) -> Response { /// The whole of the route: authenticate, look up, answer. Read-only — see the /// module docs for why that is [0189]'s invariant, not an optimisation. async fn reveal(state: &KeysState, headers: &HeaderMap) -> Response { - let Some(oauth) = state.oauth.as_ref() else { + let Some(loaded) = state.sources.get().await else { return unconfigured(); }; - let Some(gateway) = state.gateway.as_ref() else { + let Some(oauth) = loaded.oauth.as_ref() else { + return unconfigured(); + }; + let Some(gateway) = loaded.gateway.as_ref() else { return unconfigured(); }; @@ -579,10 +594,13 @@ struct RevokeResponse { /// | `404 no_key` | nothing to revoke | /// | `401` / `502` / `503` | as the reveal | async fn revoke(State(state): State, headers: HeaderMap) -> Response { - let Some(oauth) = state.oauth.as_ref() else { + let Some(loaded) = state.sources.get().await else { + return unconfigured(); + }; + let Some(oauth) = loaded.oauth.as_ref() else { return unconfigured(); }; - let Some(gateway) = state.gateway.as_ref() else { + let Some(gateway) = loaded.gateway.as_ref() else { return unconfigured(); }; // Before the session is even read: the one write a session can cause @@ -1421,7 +1439,8 @@ fn no_store(mut response: Response) -> Response { response } -/// `503` for a deployment that reached these routes with nothing wired. +/// `503` for a deployment that reached these routes with nothing wired, or +/// whose sources failed to load for this request. fn unconfigured() -> Response { no_store(errors::service_unavailable( KEYS_UNCONFIGURED, diff --git a/packages/prices-api/src/portal/mod.rs b/packages/prices-api/src/portal/mod.rs index 5ae8a551..e1a81f5b 100644 --- a/packages/prices-api/src/portal/mod.rs +++ b/packages/prices-api/src/portal/mod.rs @@ -46,6 +46,7 @@ pub mod auth; pub mod eligibility; pub mod keys; pub mod period; +pub mod sources; pub mod usage; use std::time::Duration; @@ -130,10 +131,15 @@ pub struct PortalConfig { } /// Cloneable gate state carried by the middleware. +/// +/// `enabled` is the static `PORTAL_ENABLED` flag and is all [`gate_portal`] +/// reads. `sources` is the portal's lazily loaded sources, the same cell every +/// portal state holds; only [`config_handler`] consults it here. #[derive(Clone)] pub struct PortalGate { enabled: bool, rate_limit: Option, + sources: sources::PortalSources, } impl PortalGate { @@ -147,11 +153,13 @@ impl PortalGate { /// 404 — and the whole suite stayed green with the gate deleted. /// Only [`gate_portal`] reads this state, and the gate turns on `enabled` /// alone — so the rate limit a test does not care about stays `None` here - /// rather than becoming a second argument at every call site. + /// rather than becoming a second argument at every call site, and the + /// sources are already loaded (with nothing in them). pub fn new(enabled: bool) -> Self { Self { enabled, rate_limit: None, + sources: sources::PortalSources::ready(sources::Loaded::default()), } } } @@ -164,10 +172,27 @@ impl PortalGate { /// half-built portal to every integrator reading the spec. [0195]'s API /// reference describes the public API; the portal describes itself to its own /// bundle. +/// +/// The sources come from [`sources::sources_for`]: lazily from the environment +/// in the Lambda, already loaded when the caller supplied them. pub fn apply(router: Router, config: &AppConfig) -> Router { + apply_with(router, config, sources::sources_for(config)) +} + +/// [`apply`], with the sources chosen by the caller (`crate::app_with_portal`). +/// +/// One [`sources::PortalSources`] is cloned into the gate and into all four +/// route states, so they share one cell: the first portal request that needs +/// the sources loads them for every route in this execution environment. +pub(crate) fn apply_with( + router: Router, + config: &AppConfig, + sources: sources::PortalSources, +) -> Router { let gate = PortalGate { enabled: config.portal_enabled, rate_limit: config.portal_rate_limit, + sources: sources.clone(), }; // Merged as its own `Router` rather than `.route()`d onto the caller's: // by this point the data routes have had `AppState` applied and the router @@ -179,22 +204,23 @@ pub fn apply(router: Router, config: &AppConfig) -> Router { // Usage against quota (task 0188), merged the same way and mounted under // the same conditions as everything below: unconditionally, answering - // `503` when nothing is provisioned rather than not existing. It shares + // `503` when nothing is provisioned — or when the sources failed to load + // for this request — rather than not existing. It shares // the key routes' control-plane client — usage is scoped to // `(usagePlanId, apiKeyId)` and the key id comes from the same lookup — // but carries a state of its own, because it also owns the in-process // cache that keeps dashboard refreshes off the account-wide control-plane // budget. Built first so sign-in and the key routes can hold the cache // handle below. - let usage_state = - usage::UsageState::new(config.portal_oauth.clone(), config.portal_keys.clone()); + let usage_state = usage::UsageState::new(None, None).with_sources(sources.clone()); let usage_cache = usage_state.cache_handle(); let usage = usage::routes(usage_state); // Sign-in (task 0186) and the eligibility-checked issue round-trip // (task 0189), merged the same way and for the same reason. Mounted // UNCONDITIONALLY, including when no OAuth credentials were loaded: the - // handlers answer `503` in that case rather than the routes silently not + // handlers answer with their unprovisioned landing (or, on a failed load, + // a failure landing and `/me`'s `503`) rather than the routes silently not // existing, so a deployment that opens the portal without provisioning the // secret says so instead of looking like a portal with no sign-in. While the // portal is closed the gate below makes the distinction moot — every path @@ -205,24 +231,26 @@ pub fn apply(router: Router, config: &AppConfig) -> Router { // (`keys::issue_for`) — the key ROUTE below is read-only, which is what // makes "issue is unreachable with a session cookie alone" structural. let sign_in = auth::routes( - auth::AuthState::new(config.portal_oauth.clone(), config.portal_endpoints.clone()) - .with_issue(auth::issue::IssueDeps::new( - config.portal_keys.clone(), - Some(usage_cache.clone()), - config.portal_eligibility.clone(), - )) + auth::AuthState::new(None, config.portal_endpoints.clone()) + .with_sources(sources.clone()) + .with_issue( + auth::issue::IssueDeps::new(None, Some(usage_cache.clone()), None) + .with_sources(sources.clone()), + ) .with_web_origin(config.portal_web_origin.as_deref()), ); // The key reveal (task 0187, read-only since task 0189), merged the same // way and mounted under the same conditions: unconditionally, answering - // `503` when nothing is provisioned rather than not existing. The state + // `503` when nothing is provisioned or the load failed, rather than not + // existing. The state // carries the OAuth secret because the session cookie is what authorizes a // reveal — showing the caller what already belongs to them, which is why a // session suffices here and does not for the issue above. The usage-cache // handle lets a successful reveal evict a cached "no key" (task 0188). let api_keys = keys::routes( - keys::KeysState::new(config.portal_oauth.clone(), config.portal_keys.clone()) + keys::KeysState::new(None, None) + .with_sources(sources) .with_usage_cache(usage_cache) .with_web_origin(config.portal_web_origin.as_deref()), ); @@ -301,15 +329,24 @@ pub(crate) fn cors_layer(web_origin: Option<&str>) -> CorsLayer { /// Report whether the portal is open. Always answers, in both states — it is /// the question "is the portal open?", so refusing to answer it while closed /// would be circular. +/// +/// Open means the flag is on AND the portal's sources are loaded. This is +/// usually the first portal request an execution environment sees (the page +/// asks it on every load), so it is what triggers the load. A failed load +/// answers `enabled: false` for this response only and the next call loads +/// again; the reason is in the log line and the alarm, not in the answer. +/// The flag is checked FIRST, so a closed portal never loads anything. async fn config_handler(State(gate): State) -> Response { + let enabled = gate.enabled && gate.sources.get().await.is_some(); let mut resp = Json(PortalConfig { - enabled: gate.enabled, + enabled, rate_limit_per_second: gate.rate_limit, }) .into_response(); - // Never cached: the flag changes on deploy, and a CDN or browser holding a - // stale `enabled: false` would keep the portal dark for its viewers long - // after it opened — with nothing on screen to suggest why. + // Never cached: the flag changes on deploy and a failed load changes on + // the next call, and a CDN or browser holding a stale `enabled: false` + // would keep the portal dark for its viewers long after it opened — with + // nothing on screen to suggest why. cache_control::attach(&mut resp, cache_control::NO_STORE); resp } diff --git a/packages/prices-api/src/portal/sources.rs b/packages/prices-api/src/portal/sources.rs new file mode 100644 index 00000000..fae9e7f6 --- /dev/null +++ b/packages/prices-api/src/portal/sources.rs @@ -0,0 +1,281 @@ +//! The portal's five sources, loaded on the first portal request that needs +//! them — never at cold start (task 0311). +//! +//! The five are the Discord OAuth secret (Secrets Manager) and four SSM +//! parameters — the free-plan id, the REST API id, the guild id and the +//! minimum account age — all read through the Parameters and Secrets +//! extension by [`crate::config::load_portal_sources`]. +//! +//! # Why lazily +//! +//! They used to be read in `main.rs` before the router existed, on every cold +//! start of a Lambda that also serves `/v1`. Measured 2026-09-24/25 (lore note +//! `R-five-plan-herd-and-capacity-test.md`): a burst of `/v1` cold starts +//! throttles SSM, Init grows from ~420 ms to 2.3 s for sources `/v1` never +//! uses, and one failed read closed the portal in that execution environment +//! for its whole life (7 of 68 environments on 09-24). Now a `/v1` cold start +//! reads only the mTLS bundle, and the portal pays for its own sources on its +//! own first request. +//! +//! # Why a failure is not cached +//! +//! The lifetime closure was the defect. A failed load answers **that request** +//! as unavailable (`/config` `enabled: false`, `503` on `/key`, `/usage` and +//! `/me`, a failure landing on sign-in) and the next portal request loads +//! again; a success is kept for the environment's life. Still closed, not +//! crashed: the load never panics, because a panic here would be a `502` on +//! the function that also serves `/v1`, over sources `/v1` does not use. +//! +//! # Single flight, for success only +//! +//! [`tokio::sync::OnceCell::get_or_try_init`] runs one init at a time and +//! hands a success to every waiter. A failure goes to its own caller only, and +//! the next waiter starts another attempt — so N waiters on a failing load run +//! N loads in turn. [`LOAD_BUDGET`] wraps the whole wait, which bounds each of +//! them. Standard Lambda runs one request per environment at a time, so in +//! production this bites only `serve` and the tests. +//! +//! The load is awaited inside the handler, never `tokio::spawn`ed: Lambda +//! freezes spawned work after the response. And the loader must never call +//! [`PortalSources::get`] on its own cell — that deadlocks. +//! +//! [`Loaded`] carries the OAuth secret, so it does not derive `Debug` and must +//! never be logged with `{:?}`. + +use std::future::Future; +use std::pin::Pin; +use std::sync::Arc; +use std::time::Duration; + +use tokio::sync::OnceCell; + +use crate::config::{AppConfig, PortalLoadError}; +use crate::portal::auth::secret::OauthSecret; +use crate::portal::eligibility::EligibilitySettings; +use crate::portal::keys::gateway::Gateway; + +/// The whole load's ceiling, the five reads concurrent inside it. +/// +/// Each read is bounded at 2 s by the extension client, so the budget leaves +/// room for a second attempt at a read that failed fast. Four seconds keeps a load plus the +/// slowest route after it inside the 15 s invocation — pinned below for +/// `/usage` and `/key`, and in `auth::issue` for the callback. +pub(crate) const LOAD_BUDGET: Duration = Duration::from_secs(4); + +/// What a load produced. +/// +/// The production loader returns all three `Some`, or fails. `Default` — all +/// `None` — is "loaded, nothing provisioned", which only test fixtures (and a +/// closed portal) produce; every handler already answers that shape with its +/// unprovisioned response. +#[derive(Clone, Default)] +pub struct Loaded { + pub oauth: Option>, + pub gateway: Option>, + pub settings: Option>, +} + +impl Loaded { + /// The sources a caller put on the config (`serve.rs`, tests). + pub(crate) fn from_config(config: &AppConfig) -> Loaded { + Loaded { + oauth: config.portal_oauth.clone().map(Arc::new), + gateway: config.portal_keys.clone().map(Arc::new), + settings: config.portal_eligibility.clone().map(Arc::new), + } + } +} + +/// One load attempt. +pub type LoadFuture = Pin> + Send>>; + +/// Starts a load attempt. Called once per attempt, never concurrently on one +/// cell. +pub type Loader = Arc LoadFuture + Send + Sync>; + +/// A shared handle to the portal's sources. Clones share one cell, so every +/// portal state built from one router loads once per environment. +#[derive(Clone)] +pub struct PortalSources { + cell: Arc>, + loader: Option, +} + +impl PortalSources { + /// Sources already in hand: the cell starts initialised and nothing is + /// ever loaded. + pub fn ready(loaded: Loaded) -> PortalSources { + PortalSources { + cell: Arc::new(OnceCell::new_with(Some(loaded))), + loader: None, + } + } + + /// The Lambda's loader: the five reads from the environment. + pub(crate) fn from_env() -> PortalSources { + PortalSources::with_loader(Arc::new(|| Box::pin(crate::config::load_portal_sources()))) + } + + /// A loader of the caller's choosing — a test seam, compiled out of the + /// Lambda so the deployed build has exactly one loader, the env one. + #[cfg(not(feature = "lambda"))] + pub fn lazy(loader: Loader) -> PortalSources { + PortalSources::with_loader(loader) + } + + fn with_loader(loader: Loader) -> PortalSources { + PortalSources { + cell: Arc::new(OnceCell::new()), + loader: Some(loader), + } + } + + /// The sources, loading them if this is the first ask (or every earlier + /// ask failed). `None` means this request answers as unavailable; the + /// failure is logged here, once per failed load, and the next call + /// retries. + pub async fn get(&self) -> Option<&Loaded> { + if let Some(loaded) = self.cell.get() { + return Some(loaded); + } + // A ready cell is always initialised, so only a lazy one gets here. + let loader = self.loader.as_ref()?; + let attempt = self.cell.get_or_try_init(|| loader()); + let err = match tokio::time::timeout(LOAD_BUDGET, attempt).await { + Ok(Ok(loaded)) => return Some(loaded), + Ok(Err(err)) => err, + Err(_) => PortalLoadError::TimedOut(LOAD_BUDGET), + }; + // The alarm's string: `prices-${env}-api-handler-portal-load-failed` + // matches this literal's prefix, pinned by + // `tools/scripts/portal-load-failed-filter-guard.test.mjs`. + tracing::error!( + error = %err, + "portal sources failed to load; this request answers as unavailable and the next \ + portal request retries; /v1 is unaffected" + ); + None + } +} + +/// Which handle `portal::apply` builds for `config`. +/// +/// A closed portal, or one whose caller already supplied a source +/// (`serve.rs`, which loads eagerly, and every test fixture), gets an +/// initialised cell from the config. Only an open portal with nothing supplied +/// — the Lambda — loads lazily from the environment. +pub(crate) fn sources_for(config: &AppConfig) -> PortalSources { + let supplied = config.portal_oauth.is_some() + || config.portal_keys.is_some() + || config.portal_eligibility.is_some(); + if !config.portal_enabled || supplied { + PortalSources::ready(Loaded::from_config(config)) + } else { + PortalSources::from_env() + } +} + +#[cfg(test)] +mod tests { + use std::collections::VecDeque; + use std::sync::Mutex; + use std::sync::atomic::{AtomicUsize, Ordering}; + + use super::*; + use crate::portal::auth::secret::SecretError; + + /// What one scripted attempt does. + #[derive(Clone)] + enum Step { + Fail, + Succeed, + Hang, + } + + /// A loader that plays `script` (repeating the last step) after `delay`, + /// counting its calls. + fn scripted(script: &[Step], delay: Duration) -> (Loader, Arc) { + let calls = Arc::new(AtomicUsize::new(0)); + let steps = Arc::new(Mutex::new(script.iter().cloned().collect::>())); + let last = Arc::new(Mutex::new(Step::Fail)); + let counter = calls.clone(); + let loader: Loader = Arc::new(move || { + counter.fetch_add(1, Ordering::SeqCst); + let step = { + let mut last = last.lock().unwrap(); + if let Some(step) = steps.lock().unwrap().pop_front() { + *last = step; + } + last.clone() + }; + Box::pin(async move { + tokio::time::sleep(delay).await; + match step { + Step::Fail => Err(PortalLoadError::Oauth(SecretError::NoSource)), + Step::Succeed => Ok(Loaded::default()), + Step::Hang => std::future::pending().await, + } + }) + }); + (loader, calls) + } + + #[tokio::test(start_paused = true)] + async fn concurrent_first_asks_share_one_successful_load() { + let (loader, calls) = scripted(&[Step::Succeed], Duration::from_millis(50)); + let sources = PortalSources::lazy(loader); + let mut set = tokio::task::JoinSet::new(); + for _ in 0..8 { + let sources = sources.clone(); + set.spawn(async move { sources.get().await.is_some() }); + } + while let Some(loaded) = set.join_next().await { + assert!(loaded.unwrap()); + } + assert_eq!(calls.load(Ordering::SeqCst), 1); + } + + #[tokio::test(start_paused = true)] + async fn a_failure_is_not_cached_and_a_success_is() { + let (loader, calls) = scripted(&[Step::Fail, Step::Succeed], Duration::ZERO); + let sources = PortalSources::lazy(loader); + assert!(sources.get().await.is_none()); + assert!(sources.get().await.is_some()); + assert!(sources.get().await.is_some()); + assert_eq!(calls.load(Ordering::SeqCst), 2); + } + + #[tokio::test(start_paused = true)] + async fn a_load_past_the_budget_gives_up_and_the_next_ask_loads_again() { + let (loader, calls) = scripted(&[Step::Hang, Step::Succeed], Duration::ZERO); + let sources = PortalSources::lazy(loader); + let started = tokio::time::Instant::now(); + assert!(sources.get().await.is_none()); + let waited = started.elapsed(); + assert!( + waited >= LOAD_BUDGET && waited < LOAD_BUDGET + Duration::from_millis(50), + "{waited:?}" + ); + assert!(sources.get().await.is_some()); + assert_eq!(calls.load(Ordering::SeqCst), 2); + } + + #[tokio::test] + async fn ready_sources_answer_without_a_loader() { + let sources = PortalSources::ready(Loaded::default()); + assert!(sources.loader.is_none()); + assert!(sources.get().await.is_some()); + } + + /// A load in front of the slowest routes must still leave the answer + /// inside the invocation. The 15 is `apiHandler.timeoutSeconds` in + /// `infra/envs/production.json`; the callback's share is pinned in + /// `auth::issue`. + #[test] + fn the_load_budget_fits_in_front_of_every_route() { + const LAMBDA_TIMEOUT: Duration = Duration::from_secs(15); + assert!(LOAD_BUDGET < LAMBDA_TIMEOUT); + assert!(LOAD_BUDGET + crate::portal::usage::USAGE_DEADLINE < LAMBDA_TIMEOUT); + assert!(LOAD_BUDGET + crate::portal::keys::RECONCILE_DEADLINE < LAMBDA_TIMEOUT); + } +} diff --git a/packages/prices-api/src/portal/usage/mod.rs b/packages/prices-api/src/portal/usage/mod.rs index 49c917c5..60d28745 100644 --- a/packages/prices-api/src/portal/usage/mod.rs +++ b/packages/prices-api/src/portal/usage/mod.rs @@ -111,6 +111,7 @@ use super::keys::cap::{self, Cap}; use super::keys::gateway::{Gateway, GatewayError, PlanInfo, Tier}; use super::keys::naming::{current_key, exact_matches, key_name, revocation_instant}; use super::period::Period; +use super::sources::{Loaded, PortalSources}; /// The one route. `GET` only — reading a counter must not share a path shape /// with anything that writes. @@ -155,17 +156,18 @@ const STALE_KEEP: Duration = Duration::from_secs(15 * 60); /// request-level bound (`list_named` and `usage_of` each page with a budget /// per call), and the alternative to answering `503` is Lambda killing the /// invocation with no response at all. -const USAGE_DEADLINE: Duration = Duration::from_secs(10); +pub(crate) const USAGE_DEADLINE: Duration = Duration::from_secs(10); /// What the usage route needs, cloned per request. The `Arc`s are shared /// across clones, which is what makes the cache one cache. #[derive(Clone)] pub struct UsageState { - /// Verifies the session cookie — the same secret sign-in issued it with. - oauth: Option>, - /// The control-plane client, carrying the free plan id and our API stage. `None` while the - /// portal is closed, exactly as `KeysState` holds it. - gateway: Option>, + /// The OAuth secret, which verifies the session cookie (the same secret + /// sign-in issued it with), and the control-plane client, carrying the + /// free plan id and our API stage — loaded on the first portal request + /// that needs them, exactly as `KeysState` holds them. The cache below is + /// process state, not a source, and stays outside. + sources: PortalSources, /// The last good answer per caller (session `sub`), plus the per-caller /// eviction epochs. See the module docs and [`CacheInner`]. cache: Arc>, @@ -276,8 +278,11 @@ impl UsageCache { impl UsageState { pub fn new(oauth: Option, gateway: Option) -> Self { Self { - oauth: oauth.map(Arc::new), - gateway: gateway.map(Arc::new), + sources: PortalSources::ready(Loaded { + oauth: oauth.map(Arc::new), + gateway: gateway.map(Arc::new), + ..Loaded::default() + }), cache: Arc::new(Mutex::new(CacheInner::default())), ttl: CACHE_TTL, deadline: USAGE_DEADLINE, @@ -304,6 +309,13 @@ impl UsageState { self } + /// Read the secret and the client from `sources` instead of the + /// constructor's arguments — how [`super::apply`] shares one cell. + pub(crate) fn with_sources(mut self, sources: PortalSources) -> Self { + self.sources = sources; + self + } + /// The handle the key routes hold — see [`UsageCache`]. pub fn cache_handle(&self) -> UsageCache { UsageCache(self.cache.clone()) @@ -503,10 +515,13 @@ struct CacheEntry { /// The whole route: authenticate, consult the cache, look the key up, ask AWS, /// answer. async fn usage(State(state): State, headers: HeaderMap) -> Response { - let Some(oauth) = state.oauth.as_ref() else { + let Some(loaded) = state.sources.get().await else { + return unconfigured(); + }; + let Some(oauth) = loaded.oauth.as_ref() else { return unconfigured(); }; - let Some(gateway) = state.gateway.as_ref() else { + let Some(gateway) = loaded.gateway.as_ref() else { return unconfigured(); }; @@ -889,7 +904,8 @@ fn no_store(mut response: Response) -> Response { response } -/// `503` for a deployment that reached this route with nothing wired. +/// `503` for a deployment that reached this route with nothing wired, or +/// whose sources failed to load for this request. fn unconfigured() -> Response { no_store(errors::service_unavailable( USAGE_UNCONFIGURED, diff --git a/packages/prices-api/tests/portal.rs b/packages/prices-api/tests/portal.rs index 24116aec..439a573f 100644 --- a/packages/prices-api/tests/portal.rs +++ b/packages/prices-api/tests/portal.rs @@ -10,10 +10,23 @@ use axum::Router; use axum::body::Body; use axum::http::{Request, StatusCode}; use axum::routing::get; +use prices_api::portal::sources::{Loaded, PortalSources}; use prices_api::portal::{CONFIG_PATH, OPENAPI_PATH, PortalGate, gate_portal}; -use prices_api::{AppConfig, AppState, app}; +use prices_api::{AppConfig, AppState}; use tower::ServiceExt; +/// The router under test, with the portal's sources already loaded (and +/// empty). +/// +/// Since task 0311 `/config` answers `enabled: true` only when the flag is on +/// AND the sources loaded, and `prices_api::app` would load them from the +/// environment — which a test process does not have. Handing it loaded +/// sources keeps every flag assertion below meaning "flag on, sources +/// loaded". What the LOAD decides is pinned in `tests/portal_lazy_load.rs`. +fn app(config: &AppConfig, state: AppState) -> Router { + prices_api::app_with_portal(config, state, PortalSources::ready(Loaded::default())) +} + fn config(portal_enabled: bool) -> AppConfig { config_with_keys(portal_enabled, vec![]) } diff --git a/packages/prices-api/tests/portal_auth.rs b/packages/prices-api/tests/portal_auth.rs index 0fc5ed2c..4db3d516 100644 --- a/packages/prices-api/tests/portal_auth.rs +++ b/packages/prices-api/tests/portal_auth.rs @@ -1682,9 +1682,10 @@ async fn logout_is_not_reachable_by_a_get() { // --------------------------------------------------------------------------- /// An open portal with no credentials must say so, not present a sign-in that -/// silently 404s. `AppConfig::load_portal_oauth` fails at cold start on this -/// combination, so reaching here means something bypassed it — the routes are -/// still mounted and still honest. +/// silently 404s. With nothing supplied on the config the router loads the +/// portal's sources from the environment on the first portal request (task +/// 0311), and a test process has none of them, so every request here is a +/// failed load — the state a throttled or unprovisioned deployment is in. /// /// Since task 0194's review it says so on the page: `/auth/login` is opened as /// a top-level navigation (a popup, in the bundle), so `503 JSON` was raw text @@ -1713,11 +1714,130 @@ async fn an_open_portal_with_no_credentials_lands_on_the_closed_card() { assert_eq!(login.status, StatusCode::SEE_OTHER); assert_eq!(login.location(), "/api/?signin=not_open"); - // `/auth/me` is the exception: "nobody is signed in" is true and lets the - // page render. + // `/auth/me` cannot check a cookie without the signing key, and "nobody + // is signed in" would be false for a visitor who is: a `503` the page + // renders as its failure state, never cached. let me = fetch(&router, ME_PATH, &[]).await; - assert_eq!(me.status, StatusCode::OK); - assert_eq!(me.json()["authenticated"], json!(false)); + assert_eq!(me.status, StatusCode::SERVICE_UNAVAILABLE); + assert_eq!(me.json()["code"], json!("session_unavailable")); + assert!( + me.headers + .get(header::CACHE_CONTROL) + .is_some_and(|v| v.to_str().unwrap().contains("no-store")) + ); +} + +// --------------------------------------------------------------------------- +// A slow lazy load in front of the callback (task 0311) +// --------------------------------------------------------------------------- + +/// Every source a round-trip needs, already loaded: the OAuth secret, a +/// control plane that is never reached, and the eligibility settings. +fn full_sources() -> prices_api::portal::sources::Loaded { + use std::sync::Arc; + prices_api::portal::sources::Loaded { + oauth: Some(Arc::new(oauth_secret())), + gateway: Some(Arc::new(test_gateway("http://127.0.0.1:9"))), + settings: Some(Arc::new(eligibility_settings())), + } +} + +/// A router against `mock` whose sources are `sources`. +fn router_with_sources( + mock: &MockDiscord, + sources: prices_api::portal::sources::PortalSources, +) -> Router { + let config = AppConfig { + ch_enabled: false, + base_url: None, + api_keys: vec![], + portal_enabled: true, + portal_oauth: None, + portal_endpoints: Endpoints { + api_base: mock.base.clone(), + ..Endpoints::default() + }, + portal_keys: None, + portal_eligibility: None, + portal_rate_limit: None, + portal_web_origin: None, + }; + prices_api::app_with_portal(&config, AppState::without_ch(), sources) +} + +/// Sources that take 1.5 s to load — well past `issue::SOURCES_ALLOWANCE`, +/// which the budget test keeps under a second — and then succeed. +fn slow_sources() -> prices_api::portal::sources::PortalSources { + use prices_api::portal::sources::{LoadFuture, PortalSources}; + PortalSources::lazy(std::sync::Arc::new(|| -> LoadFuture { + Box::pin(async { + tokio::time::sleep(std::time::Duration::from_millis(1500)).await; + Ok(full_sources()) + }) + })) +} + +/// Login on one environment (loaded), callback on another that has to load +/// first — the common shape once `/v1` warms most environments. The load +/// eats the callback's allowance, so it lands on a retryable failure before +/// the token exchange, and drops the pending cookie as a verified callback +/// does. +#[tokio::test] +async fn a_callback_that_spent_its_allowance_loading_lands_on_signin_failed() { + let mock = MockDiscord::start(GRANTED_SCOPE, None).await; + let loaded = router_with_sources( + &mock, + prices_api::portal::sources::PortalSources::ready(full_sources()), + ); + let cold = router_with_sources(&mock, slow_sources()); + + let started = start_login(&loaded).await; + let reply = fetch( + &cold, + &format!("{CALLBACK_PATH}?code=an-auth-code&state={}", started.state), + &[(cookies::PENDING_COOKIE, &started.pending)], + ) + .await; + + assert_eq!(reply.status, StatusCode::SEE_OTHER); + assert_eq!(reply.location(), "/api/?signin=failed"); + assert!(reply.clears(cookies::PENDING_COOKIE)); + assert!(reply.cookie(cookies::SESSION_COOKIE).is_none()); + assert_eq!(mock.exchanges(), 0, "the token exchange must not start"); +} + +/// The same on an issue round-trip: its own failure landing. +#[tokio::test] +async fn an_issue_callback_that_spent_its_allowance_loading_lands_on_issue_failed() { + let mock = MockDiscord::start(GRANTED_SCOPE, None).await; + let loaded = router_with_sources( + &mock, + prices_api::portal::sources::PortalSources::ready(full_sources()), + ); + let cold = router_with_sources(&mock, slow_sources()); + + let login = fetch(&loaded, &format!("{LOGIN_PATH}?action=issue"), &[]).await; + assert_eq!(login.status, StatusCode::SEE_OTHER); + let pending = login + .cookie(cookies::PENDING_COOKIE) + .expect("login must set the pending-login cookie"); + let query = login.location().split_once('?').unwrap().1.to_string(); + let state = form_urlencoded::parse(query.as_bytes()) + .find(|(k, _)| k == "state") + .map(|(_, v)| v.into_owned()) + .expect("the authorize URL must carry `state`"); + + let reply = fetch( + &cold, + &format!("{CALLBACK_PATH}?code=an-auth-code&state={state}"), + &[(cookies::PENDING_COOKIE, &pending)], + ) + .await; + + assert_eq!(reply.status, StatusCode::SEE_OTHER); + assert_eq!(reply.location(), "/api/?issue=failed"); + assert!(reply.clears(cookies::PENDING_COOKIE)); + assert_eq!(mock.exchanges(), 0, "the token exchange must not start"); } /// The routes are keyless — `crate::auth::is_exempt` exempts the whole prefix diff --git a/packages/prices-api/tests/portal_lazy_load.rs b/packages/prices-api/tests/portal_lazy_load.rs new file mode 100644 index 00000000..025dc5cd --- /dev/null +++ b/packages/prices-api/tests/portal_lazy_load.rs @@ -0,0 +1,366 @@ +//! The portal's sources load on the first portal request, not at cold start +//! (task 0311). +//! +//! Driven through the real router (`prices_api::app_with_portal`) with a +//! counting, scripted loader in place of the environment one. Pinned here: +//! +//! - building the router and serving `/health` and `/v1` read nothing; +//! - `/config` loads, answers `enabled` by the outcome, and a success is kept; +//! - a failure is not kept: the next request loads again; +//! - a portal route answers `503` on a failed load and works on the next; +//! - with `PORTAL_ENABLED=false` the portal is a `404` and nothing loads; +//! - concurrent first requests share one successful load; +//! - a sign-in callback on a failed or slow load lands on a retryable failure +//! (the slow case with a real login is in `tests/portal_auth.rs`); +//! - `main.rs` performs no portal load. +//! +//! The retry around each read is unit-tested in `portal::extension`. + +use std::collections::VecDeque; +use std::sync::atomic::{AtomicUsize, Ordering}; +use std::sync::{Arc, Mutex}; +use std::time::Duration; + +use axum::Router; +use axum::body::Body; +use axum::http::{HeaderMap, Request, StatusCode, header}; +use prices_api::config::PortalLoadError; +use prices_api::portal::auth::secret::{OauthSecret, SecretError}; +use prices_api::portal::keys::gateway::Gateway; +use prices_api::portal::sources::{Loaded, Loader, PortalSources}; +use prices_api::{AppConfig, AppState, app_with_portal}; +use serde_json::{Value, json}; +use tower::ServiceExt; + +// --------------------------------------------------------------------------- +// Fixtures +// --------------------------------------------------------------------------- + +fn oauth_secret() -> OauthSecret { + OauthSecret::parse( + &json!({ + "client_id": "a-client-id", + "client_secret": "the-client-secret", + "redirect_uri": "https://portal.example/api/auth/callback", + "session_signing_key": + "0123456789abcdef0123456789abcdef0123456789abcdef0123456789abcdef", + }) + .to_string(), + ) + .expect("the test bundle must be valid") +} + +/// Never contacted: every request in this file stops before the control +/// plane (no session cookie, or no load). +fn gateway() -> Gateway { + Gateway::against( + "http://127.0.0.1:9", + "plan-free".to_string(), + "api-id".to_string(), + "production".to_string(), + ) +} + +fn loaded_with_keys() -> Loaded { + Loaded { + oauth: Some(Arc::new(oauth_secret())), + gateway: Some(Arc::new(gateway())), + settings: None, + } +} + +fn config(portal_enabled: bool) -> AppConfig { + AppConfig { + ch_enabled: false, + base_url: None, + api_keys: vec![], + portal_enabled, + portal_oauth: None, + portal_endpoints: Default::default(), + portal_keys: None, + portal_eligibility: None, + portal_rate_limit: None, + portal_web_origin: None, + } +} + +/// What one scripted attempt does. +#[derive(Clone)] +enum Step { + Fail, + Succeed(Loaded), +} + +/// A loader that plays `script` (repeating the last step once it runs out), +/// each attempt after `delay`, counting its calls. +fn scripted(script: Vec, delay: Duration) -> (PortalSources, Arc) { + let calls = Arc::new(AtomicUsize::new(0)); + let steps = Arc::new(Mutex::new(script.into_iter().collect::>())); + let last = Arc::new(Mutex::new(Step::Fail)); + let counter = calls.clone(); + let loader: Loader = Arc::new(move || { + counter.fetch_add(1, Ordering::SeqCst); + let step = { + let mut last = last.lock().unwrap(); + if let Some(step) = steps.lock().unwrap().pop_front() { + *last = step; + } + last.clone() + }; + Box::pin(async move { + tokio::time::sleep(delay).await; + match step { + Step::Fail => Err(PortalLoadError::Oauth(SecretError::NoSource)), + Step::Succeed(loaded) => Ok(loaded), + } + }) + }); + (PortalSources::lazy(loader), calls) +} + +fn router(portal_enabled: bool, sources: PortalSources) -> Router { + app_with_portal(&config(portal_enabled), AppState::without_ch(), sources) +} + +struct Reply { + status: StatusCode, + headers: HeaderMap, + body: Vec, +} + +impl Reply { + fn json(&self) -> Value { + serde_json::from_slice(&self.body).expect("the body should be JSON") + } + + fn no_store(&self) -> bool { + self.headers + .get(header::CACHE_CONTROL) + .is_some_and(|v| v.to_str().unwrap().contains("no-store")) + } + + fn location(&self) -> String { + self.headers + .get(header::LOCATION) + .map(|v| v.to_str().unwrap().to_string()) + .unwrap_or_default() + } +} + +async fn get(router: &Router, uri: &str) -> Reply { + let response = router + .clone() + .oneshot(Request::builder().uri(uri).body(Body::empty()).unwrap()) + .await + .unwrap(); + let status = response.status(); + let headers = response.headers().clone(); + let body = axum::body::to_bytes(response.into_body(), usize::MAX) + .await + .unwrap() + .to_vec(); + Reply { + status, + headers, + body, + } +} + +// --------------------------------------------------------------------------- +// The cold-start path reads nothing +// --------------------------------------------------------------------------- + +#[tokio::test] +async fn the_router_health_and_v1_load_nothing_and_config_loads_once() { + let (sources, calls) = scripted(vec![Step::Succeed(Loaded::default())], Duration::ZERO); + let router = router(true, sources); + assert_eq!( + calls.load(Ordering::SeqCst), + 0, + "building the router loaded" + ); + + assert_eq!(get(&router, "/health").await.status, StatusCode::OK); + // A real `/v1` handler, reached without ClickHouse: `post_batch` refuses + // an empty list itself, before `state.ch()`. + let v1 = router + .clone() + .oneshot( + Request::builder() + .method("POST") + .uri("/v1/prices/batch") + .header(header::CONTENT_TYPE, "application/json") + .body(Body::from(r#"{"assets": []}"#)) + .unwrap(), + ) + .await + .unwrap(); + assert_eq!(v1.status(), StatusCode::BAD_REQUEST); + assert_eq!(calls.load(Ordering::SeqCst), 0, "/health or /v1 loaded"); + + let config = get(&router, "/api/config").await; + assert_eq!(config.status, StatusCode::OK); + assert_eq!(config.json()["enabled"], json!(true)); + assert_eq!(calls.load(Ordering::SeqCst), 1); + + let again = get(&router, "/api/config").await; + assert_eq!(again.json()["enabled"], json!(true)); + assert_eq!(calls.load(Ordering::SeqCst), 1, "a success must be kept"); +} + +/// `main.rs` cannot be run from a test (`required-features = ["lambda"]`), so +/// its source is the evidence: no portal load before the router is built. +/// `client_from_lambda_env` is asserted too, so this cannot pass on an empty +/// or moved file. +#[test] +fn main_rs_performs_no_portal_load() { + let main = include_str!("../src/main.rs"); + assert!(!main.contains("load_portal"), "main.rs loads the portal"); + assert!( + !main.contains("PortalSources"), + "main.rs builds portal sources" + ); + assert!(main.contains("client_from_lambda_env")); + assert!(main.contains("let config = AppConfig::from_env();")); +} + +// --------------------------------------------------------------------------- +// /config answers by the load's outcome, and a failure is not kept +// --------------------------------------------------------------------------- + +#[tokio::test] +async fn config_says_closed_on_a_failed_load_and_open_on_the_next() { + let (sources, calls) = scripted( + vec![Step::Fail, Step::Succeed(Loaded::default())], + Duration::ZERO, + ); + let router = router(true, sources); + + let failed = get(&router, "/api/config").await; + assert_eq!(failed.status, StatusCode::OK); + assert_eq!(failed.json()["enabled"], json!(false)); + assert!(failed.no_store(), "a closed answer must never be cached"); + + let recovered = get(&router, "/api/config").await; + assert_eq!(recovered.json()["enabled"], json!(true)); + assert_eq!(calls.load(Ordering::SeqCst), 2, "the failure was kept"); +} + +// --------------------------------------------------------------------------- +// Portal routes: 503 on a failed load, working on the next +// --------------------------------------------------------------------------- + +#[tokio::test] +async fn usage_is_503_on_a_failed_load_and_answers_on_the_next() { + let (sources, _) = scripted( + vec![Step::Fail, Step::Succeed(loaded_with_keys())], + Duration::ZERO, + ); + let router = router(true, sources); + + let failed = get(&router, "/api/usage").await; + assert_eq!(failed.status, StatusCode::SERVICE_UNAVAILABLE); + assert_eq!(failed.json()["code"], json!("usage_unconfigured")); + assert!(failed.no_store()); + + // Past the unprovisioned branch: no session cookie is now the answer. + let next = get(&router, "/api/usage").await; + assert_eq!(next.status, StatusCode::UNAUTHORIZED); +} + +#[tokio::test] +async fn key_is_503_on_a_failed_load_and_answers_on_the_next() { + let (sources, _) = scripted( + vec![Step::Fail, Step::Succeed(loaded_with_keys())], + Duration::ZERO, + ); + let router = router(true, sources); + + let failed = get(&router, "/api/key").await; + assert_eq!(failed.status, StatusCode::SERVICE_UNAVAILABLE); + assert_eq!(failed.json()["code"], json!("keys_unconfigured")); + assert!(failed.no_store()); + + let next = get(&router, "/api/key").await; + assert_eq!(next.status, StatusCode::UNAUTHORIZED); +} + +/// `/me` does not say "signed out" when it cannot check: that would be a lie +/// to a visitor who is signed in. +#[tokio::test] +async fn me_is_503_on_a_failed_load_and_answers_on_the_next() { + let (sources, _) = scripted( + vec![Step::Fail, Step::Succeed(loaded_with_keys())], + Duration::ZERO, + ); + let router = router(true, sources); + + let failed = get(&router, "/api/auth/me").await; + assert_eq!(failed.status, StatusCode::SERVICE_UNAVAILABLE); + assert_eq!(failed.json()["code"], json!("session_unavailable")); + assert!(failed.no_store()); + + let next = get(&router, "/api/auth/me").await; + assert_eq!(next.status, StatusCode::OK); + assert_eq!(next.json()["authenticated"], json!(false)); +} + +// --------------------------------------------------------------------------- +// PORTAL_ENABLED=false: a 404, and nothing ever loads +// --------------------------------------------------------------------------- + +#[tokio::test] +async fn a_closed_portal_is_a_404_and_never_loads() { + let (sources, calls) = scripted(vec![Step::Succeed(loaded_with_keys())], Duration::ZERO); + let router = router(false, sources); + + for path in ["/api/key", "/api/usage", "/api/auth/me"] { + let reply = get(&router, path).await; + assert_eq!(reply.status, StatusCode::NOT_FOUND, "{path}"); + assert!(reply.body.is_empty(), "{path} carried a body"); + } + let config = get(&router, "/api/config").await; + assert_eq!(config.json()["enabled"], json!(false)); + assert_eq!(calls.load(Ordering::SeqCst), 0, "a closed portal loaded"); +} + +// --------------------------------------------------------------------------- +// Single flight +// --------------------------------------------------------------------------- + +#[tokio::test] +async fn concurrent_first_requests_share_one_load() { + let (sources, calls) = scripted( + vec![Step::Succeed(Loaded::default())], + Duration::from_millis(50), + ); + let router = router(true, sources); + + let mut set = tokio::task::JoinSet::new(); + for _ in 0..8 { + let router = router.clone(); + set.spawn(async move { get(&router, "/api/config").await.json()["enabled"].clone() }); + } + while let Some(enabled) = set.join_next().await { + assert_eq!(enabled.unwrap(), json!(true)); + } + assert_eq!(calls.load(Ordering::SeqCst), 1); +} + +// --------------------------------------------------------------------------- +// The sign-in callback +// --------------------------------------------------------------------------- + +/// The action is not known before `state` is verified with the secret that +/// failed to load, so the landing is sign-in's failure, and the pending +/// cookie is left alone as on every refusal before verification. +#[tokio::test] +async fn a_callback_on_a_failed_load_lands_on_signin_failed() { + let (sources, _) = scripted(vec![Step::Fail], Duration::ZERO); + let router = router(true, sources); + + let reply = get(&router, "/api/auth/callback?code=c&state=s").await; + assert_eq!(reply.status, StatusCode::SEE_OTHER); + assert_eq!(reply.location(), "/api/?signin=failed"); + assert!(reply.headers.get(header::SET_COOKIE).is_none()); +} From c8b61979573eac0efdce4bab1881334ace191a68 Mon Sep 17 00:00:00 2001 From: Adam <65679285+adamkoot@users.noreply.github.com> Date: Fri, 25 Sep 2026 13:41:37 +0200 Subject: [PATCH 14/22] feat(lore-0311): alarm on a failed portal load instead of a cold-start closure The portal no longer closes at cold start; a failed load answers one request and the next retries. The portal-closed filter and alarm are replaced 1-for-1 by prices-${env}-api-handler-portal-load-failed on the new "portal sources failed to load" line (Prices/ApiHandler, PortalSourcesLoadFailed). New logical ids, so CloudFormation replaces both; DashboardAlarmCount is unchanged. - guard test renamed; it now reads sources.rs for the line, main.rs for the subscriber, and pins the alarm's metric wiring and the 1024-char description limit - ci.yml runs it when sources.rs or main.rs changes - infra comments and runbooks describe the lazy load and its retry --- .github/workflows/ci.yml | 12 +- docs/runbooks/manual-api-key-tier.md | 3 +- docs/runbooks/portal-oauth-deploy-prep.md | 51 +++-- infra/Makefile | 4 +- infra/src/lib/app.ts | 2 +- infra/src/lib/stacks/api-gateway-stack.ts | 2 +- infra/src/lib/stacks/compute-stack.ts | 89 ++++---- infra/src/lib/stacks/observability-stack.ts | 75 ++++--- .../portal-closed-filter-guard.test.mjs | 137 ------------ .../portal-load-failed-filter-guard.test.mjs | 196 ++++++++++++++++++ web/portal/src/api/portal.ts | 7 +- 11 files changed, 331 insertions(+), 247 deletions(-) delete mode 100644 tools/scripts/portal-closed-filter-guard.test.mjs create mode 100644 tools/scripts/portal-load-failed-filter-guard.test.mjs diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml index ce483b64..a9b68f47 100644 --- a/.github/workflows/ci.yml +++ b/.github/workflows/ci.yml @@ -82,10 +82,14 @@ jobs: - 'nx.json' - '.github/workflows/ci.yml' - 'tools/scripts/**' - # The portal-closed alarm's metric filter matches a log line this - # file writes (task 0249). The guard that ties the two together - # is `tools/scripts/portal-closed-filter-guard.test.mjs`, run by - # this job — so a PR that only rewords the line must run it. + # The portal-load-failed alarm's metric filter matches a log line + # `portal/sources.rs` writes, through the JSON subscriber + # `main.rs` sets up (tasks 0249, 0311). The guard that ties the + # three together is + # `tools/scripts/portal-load-failed-filter-guard.test.mjs`, run + # by this job — so a PR that only rewords the line, or only + # touches the subscriber, must run it. + - 'packages/prices-api/src/portal/sources.rs' - 'packages/prices-api/src/main.rs' typescript: diff --git a/docs/runbooks/manual-api-key-tier.md b/docs/runbooks/manual-api-key-tier.md index da39f6e7..1d656ecf 100644 --- a/docs/runbooks/manual-api-key-tier.md +++ b/docs/runbooks/manual-api-key-tier.md @@ -247,7 +247,8 @@ environment (`infra/envs/` holds only `production.json` and `cicd.json`). new plans and the removed `PortalAttachKeyToFreePlan` policy; in the Compute diff, the role policy gains exactly the three `/usageplans` statements (and the `api-gateway-id` read). -2. `/config` answers `enabled: true` (the portal did not close at cold start). +2. `/config` answers `enabled: true` (the portal's sources loaded; that call + triggers the load). 3. Move a test key free → Basic → Pro → free with the procedure above. After each step: - `get-usage-plans --key-id` shows exactly the target plan; diff --git a/docs/runbooks/portal-oauth-deploy-prep.md b/docs/runbooks/portal-oauth-deploy-prep.md index 7058ebf7..7a0b1e29 100644 --- a/docs/runbooks/portal-oauth-deploy-prep.md +++ b/docs/runbooks/portal-oauth-deploy-prep.md @@ -247,14 +247,17 @@ Both must print `prices/production/portal-discord-oauth`. `PORTAL_ENABLED` is still `false` at this point and the routes still answer an empty `404`. That is correct: **the api-handler does not read this secret while -the portal is closed** (see `AppConfig::load_portal_oauth`), so creating it does -not change any behaviour, and forgetting to create it before opening the portal -closes the portal again at the _next_ cold start — `/config` answers -`enabled: false` and the api-handler logs `portal closed at cold start` naming -`PORTAL_OAUTH_SECRET_NAME`, which also pages as -`prices-production-api-handler-portal-closed` (task 0249) — rather than -silently serving a broken sign-in (`AppConfig::load_portal_or_close`). `/v1` -is unaffected either way. +the portal is closed** (see `packages/prices-api/src/portal/sources.rs`), so +creating it does not change any behaviour. Forgetting to create it before +opening the portal fails the portal's load on the first portal request per +execution environment (the `/config` probe triggers it): that `/config` answers +`enabled: false` and the api-handler logs `portal sources failed to load` +naming `PORTAL_OAUTH_SECRET_NAME`, which pages as +`prices-production-api-handler-portal-load-failed` (tasks 0249, 0311) — rather +than silently serving a broken sign-in. The next request retries, so creating +the secret fixes it without a redeploy. A successful load is kept for the +execution environment's life, so changing the secret's VALUE later still needs +a recycle, as before. `/v1` is unaffected either way. ## 4. Verify locally before opening production @@ -429,21 +432,23 @@ openapi:verify-routes` asserts exactly this against the synthesized templates, so a drift fails CI rather than a deploy — but the _existence_ of the deployed parameter is not something CI can see. -**If the parameter is missing when `PORTAL_ENABLED` becomes `true`, the -api-handler closes the portal at cold start** — `/config` answers -`enabled: false` and the log carries `portal closed at cold start` naming -`PORTAL_FREE_PLAN_PARAM` — and `/v1` is unaffected. It used to fail init -instead, which took `/v1` down with it (one router serves every route group, -ADR 0008); task 0194's PR review is where that changed, and the reasoning is on -`AppConfig::load_portal_or_close`. The shape is still "found only at the moment -of opening", as with the OAuth secret in §3, and the alternative it avoids is -still a portal with a key button that answers `503` — a closed portal answers -before any button renders. What it costs: the closure pages as -`prices-production-api-handler-portal-closed` (task 0249), but only when a -cold start happens, so the `/config` probe after the deploy stays the check -that runs _now_, not an optional confirmation; the alarm is what catches a -closure in a LATER cold start (a throttled Parameter Store read in a -scale-out). +**If the parameter is missing when `PORTAL_ENABLED` becomes `true`, every load +of the portal's sources fails** — each `/config` answers `enabled: false` and +the log carries `portal sources failed to load` naming `PORTAL_FREE_PLAN_PARAM` +— and `/v1` is unaffected: since task 0311 the sources load on the first +portal request per execution environment, never at cold start. It used to fail +init, which took `/v1` down with it (one router serves every route group, ADR +0008), and then (task 0194) to close the portal in that environment for its +life. Now a failed load costs that one request, and the next retries, so +publishing the parameter fixes it without a redeploy or a recycle. The shape is +still "found only at the moment of opening", as with the OAuth secret in §3, +and the alternative it avoids is still a portal with a key button that answers +`503` — `/config` says the portal is not open before any button renders. Each +failed load pages as `prices-production-api-handler-portal-load-failed` (tasks +0249, 0311), so the alarm keeps firing while the parameter is missing; the +`/config` probe after the deploy stays the check that runs _now_ (it triggers +the load itself), not an optional confirmation, and the alarm is what catches a +LATER failed load (a throttled Parameter Store read under portal traffic). While the portal is closed the handler reads neither, so nothing here changes any behaviour until the flag moves. diff --git a/infra/Makefile b/infra/Makefile index 715dd001..69729302 100644 --- a/infra/Makefile +++ b/infra/Makefile @@ -202,8 +202,8 @@ deploy-production-eventbridge: build-production build-lambdas destroy-production-eventbridge: build-production npx cdk --app "$(PRODUCTION_APP)" destroy Prices-production-EventBridge --force -# Needs the api-handler log group to exist: the portal-closed metric filter -# (task 0249) is created on `/aws/lambda/prices-production-api-handler`, which +# Needs the api-handler log group to exist: the portal-load-failed metric +# filter (tasks 0249, 0311) is created on `/aws/lambda/prices-production-api-handler`, which # Compute owns and `destroy-production-compute` deletes (the log removal # policy is DESTROY). On a fresh environment, or after a Compute destroy, # deploy Compute first or this target fails with "log group does not exist". diff --git a/infra/src/lib/app.ts b/infra/src/lib/app.ts index f36bfeba..e3a88fb5 100644 --- a/infra/src/lib/app.ts +++ b/infra/src/lib/app.ts @@ -66,7 +66,7 @@ export function createApp({ config }: CreateAppOptions): void { // stack: alarms key on function names, queue names and log-group names // as plain strings, so it stays deployable on its own. One of its // resources still needs another stack's resource to EXIST: the - // portal-closed metric filter (task 0249) is created on the api-handler + // portal-load-failed metric filter (tasks 0249, 0311) is created on the api-handler // log group ComputeStack owns, and `fromLogGroupName` emits no // dependency for it. `addDependency` orders Compute first under // `deploy --all` without adding a reference; the `--exclusively` target diff --git a/infra/src/lib/stacks/api-gateway-stack.ts b/infra/src/lib/stacks/api-gateway-stack.ts index 9c1b0309..6c8b5acb 100644 --- a/infra/src/lib/stacks/api-gateway-stack.ts +++ b/infra/src/lib/stacks/api-gateway-stack.ts @@ -911,7 +911,7 @@ export class ApiGatewayStack extends cdk.Stack { }); usagePlan.addApiKey(apiKey); - // Also read by the portal backend at cold start (task 0311, + // Also read by the portal backend when its sources load (task 0311, // `PORTAL_API_ID_PARAM`): `plan_of` keeps only the plans whose apiStages // name this API + stage. Through SSM for the same cycle reason as the plan // id below. diff --git a/infra/src/lib/stacks/compute-stack.ts b/infra/src/lib/stacks/compute-stack.ts index 6b1fe52f..249fb391 100644 --- a/infra/src/lib/stacks/compute-stack.ts +++ b/infra/src/lib/stacks/compute-stack.ts @@ -203,8 +203,9 @@ export class ComputeStack extends cdk.Stack { * here would close a Compute -> Gateway -> Compute cycle — the same shape of * problem `apiBaseUrl` has. And it must not be hard-coded, because AWS * generates the id and it changes if the plan is ever replaced. So the - * handler reads it at cold start through the Parameters and Secrets extension - * already attached below, exactly as it reads secret VALUES by NAME. + * handler reads it on the first portal request per execution environment + * (task 0311) through the Parameters and Secrets extension already attached + * below, exactly as it reads secret VALUES by NAME. * * Two siblings since task 0311, set beside it on the Function env and for * the same reason: `PORTAL_API_ID_PARAM`, the NAME of the parameter holding @@ -497,12 +498,12 @@ export class ComputeStack extends cdk.Stack { // // The grant is on the by-name wildcard ARN, so it does not require the // secret to exist at synth time. That WAS harmless because a closed portal - // never asked; with `PORTAL_ENABLED` true (task 0194) the read happens at - // every cold start, and a missing or misnamed secret closes the portal in - // that execution environment with a `portal closed at cold start` error - // log — not an init panic, because the Lambda also serves `/v1`. See the - // deploy-gate note on `PORTAL_ENABLED` below and - // `AppConfig::load_portal_or_close`. + // never asked; with `PORTAL_ENABLED` true (task 0194) the read happens on + // the first portal request per execution environment (task 0311), and a + // missing or misnamed secret answers that request as unavailable with a + // `portal sources failed to load` error log; the next request retries, + // and nothing stays closed. See the deploy-gate note on `PORTAL_ENABLED` + // below and `packages/prices-api/src/portal/sources.rs`. this.apiHandlerRole.addToPrincipalPolicy( new iam.PolicyStatement({ sid: 'ReadPortalOauthSecret', @@ -764,14 +765,15 @@ export class ComputeStack extends cdk.Stack { }), ); - // The usage-plan id, read at cold start. + // The usage-plan id, read on the first portal request per execution + // environment (task 0311). // // **Currently redundant, and kept deliberately.** The baseline role already // carries `ReadSsmNamespaces`, which grants `ssm:GetParameter` across the // whole `/prices/${envName}/*` namespace — so this statement adds no access // today. It names the one parameter this feature depends on, so that // narrowing that baseline (which task 0194 may well want to) does not - // silently break key issuance at the next cold start. Stated rather than + // silently break key issuance at the next portal load. Stated rather than // left implicit, because an IAM statement that looks like the reason // something works, while something broader is the actual reason, is worse // than no statement at all. @@ -785,14 +787,14 @@ export class ComputeStack extends cdk.Stack { }), ); - // The REST API id, read at cold start (task 0311) — `plan_of` keeps only + // The REST API id, read with the plan id (task 0311) — `plan_of` keeps only // the usage plans whose apiStages name this API + stage. Currently // redundant and kept deliberately, for exactly the reason stated on // `PortalReadFreePlanIdParameter` above: the baseline's // `ReadSsmNamespaces` already covers `/prices/${envName}/*`, and this // statement names the parameter the portal depends on, stated rather than - // implied, so a narrowed baseline cannot silently close the portal at the - // next cold start. + // implied, so a narrowed baseline cannot silently fail every portal load + // from the next one on. this.apiHandlerRole.addToPrincipalPolicy( new iam.PolicyStatement({ sid: 'PortalReadApiGatewayIdParameter', @@ -804,7 +806,7 @@ export class ComputeStack extends cdk.Stack { ); // The eligibility gate's two knobs (task 0189), read at runtime — per - // issuance, not at cold start alone. The same currently-redundant-and-kept + // issuance, not at the portal's load alone. The same currently-redundant-and-kept // reasoning as `PortalReadFreePlanIdParameter` above: the baseline's // `ReadSsmNamespaces` already covers `/prices/${envName}/*`, and this // statement names the two parameters the gate depends on so a narrowed @@ -844,8 +846,8 @@ export class ComputeStack extends cdk.Stack { // // No `loggingFormat`: the function logs in Lambda's Text format, so each // JSON line the tracing subscriber writes reaches CloudWatch untouched, - // and ObservabilityStack's portal-closed metric filter (task 0249) - // matches it on `$.fields.message`. AWS documents that JSON format does + // and ObservabilityStack's portal-load-failed metric filter (tasks 0249, + // 0311) matches it on `$.fields.message`. AWS documents that JSON format does // not re-encode lines that are already JSON, so switching would likely // still match — but that is a claim, not a measurement: after ANY change // to how this function's logs reach CloudWatch, re-prove the filter with @@ -882,12 +884,15 @@ export class ComputeStack extends cdk.Stack { // opening creates has to be unwound to close it again. // // ⚠️ **This value is a deploy gate, not just a flag.** With it true the - // handler resolves the portal's configuration AT COLD START, from FIVE - // reads, and the portal opens only if every one of them succeeds: + // handler resolves the portal's configuration on the first portal + // request per execution environment (the `/config` probe triggers + // it; task 0311), from FIVE concurrent reads — `load_portal_sources` + // in `config.rs`, called by `portal/sources.rs` — and the portal + // answers as open only once every one of them has succeeded: // - // 1. `load_portal_oauth` (`config.rs`) on the Discord OAuth secret + // 1. `portal_oauth_from_env` (`config.rs`) on the Discord OAuth secret // named by `PORTAL_OAUTH_SECRET_NAME` — operator-created, runbook §2 - // 2. `load_portal_keys` (`config.rs`) on the SSM parameter named by + // 2. `portal_keys_from_env` (`config.rs`) on the SSM parameter named by // `PORTAL_FREE_PLAN_PARAM`, i.e. // `/prices/{env}/pricing-api-free-plan-id`. This one is NOT // operator-seeded and is easy to miss: it is published by @@ -896,32 +901,33 @@ export class ComputeStack extends cdk.Stack { // plan is replaced or renamed, this stack can be live with the flag // true while the parameter does not yet exist. Deploy order matters // here - // 3. + 4. the eligibility probe (`portal/eligibility.rs`) on + // 3. + 4. `portal_eligibility_from_env`'s probe (`portal/eligibility.rs`) on // `/prices/{env}/discord-guild-id` and // `/prices/{env}/min-account-age-minutes` — operator-seeded, // runbook §2a - // 5. `load_portal_keys` again (task 0311), on the SSM parameter named + // 5. `portal_keys_from_env` again (task 0311), on the SSM parameter named // by `PORTAL_API_ID_PARAM`, i.e. `/prices/{env}/api-gateway-id`. // The same deploy-order caveat as read 2: `ApiGatewayStack` // publishes it and deploys AFTER this stack. It has existed since // the gateway's first deploy, so this bites only a fresh - // environment — where read 2 already closes the portal until + // environment — where read 2 already fails the load until // `ApiGatewayStack` has deployed once, so this read adds no new // failure there. // - // A failed read CLOSES the portal in that execution environment and - // logs `portal closed at cold start` on the api-handler; it does NOT - // panic init, because this Lambda also serves `/v1` and an init panic - // is a `502` to the next data-API caller (task 0194's PR review, - // finding 1; the reasoning is on `AppConfig::load_portal_or_close`). - // So deploying this ahead of the operator steps ships a portal whose - // `/config` says `enabled: false`, not a data-API outage — and a - // closure pages as `prices-${env}-api-handler-portal-closed` - // (ObservabilityStack, task 0249), but only once a cold start - // happens, so the runbook's `/config` probe after the deploy is - // still the check that runs at deploy time. Runbook - // `portal-oauth-deploy-prep.md` §2, §2a and §5 are the steps; task - // 0194's audit is what verifies they were run. + // A failed read answers THAT request as unavailable (`/config` + // `enabled: false`, `503` on the portal's other routes) and logs + // `portal sources failed to load`, naming the variable, on the + // api-handler; the next portal request retries, so nothing stays + // closed and no recycle is needed. It never touches init: `/v1` cold + // starts read none of these (task 0311). So deploying this ahead of + // the operator steps ships a portal whose `/config` answers + // `enabled: false` on each call until the missing source exists, not + // a data-API outage — and each failed load pages as + // `prices-${env}-api-handler-portal-load-failed` (ObservabilityStack, + // tasks 0249, 0311). The runbook's `/config` probe after the deploy + // remains the deploy-time check, and it now itself triggers the + // load. Runbook `portal-oauth-deploy-prep.md` §2, §2a and §5 are the + // steps; task 0194's audit is what verifies they were run. PORTAL_ENABLED: 'true', // The NAME of the portal's Discord OAuth bundle, never its value // (task 0186; ADR 0007's precedent, audited by Tranche 3 AC 6). The @@ -932,15 +938,16 @@ export class ComputeStack extends cdk.Stack { // Set unconditionally, which is what kept opening the portal to the // one-word diff above rather than a two-line change made under time // pressure. With the flag now true the read is no longer conditional: - // this name resolving to a missing secret is read 1 of the five fatal - // cold-start reads listed on `PORTAL_ENABLED`. + // this name resolving to a missing secret fails read 1 of the five + // portal-load reads listed on `PORTAL_ENABLED`. PORTAL_OAUTH_SECRET_NAME: this.portalOauthSecretName, // The NAME of the SSM parameter holding the `pricing-api-free` usage // plan id (task 0187) — see `portalFreePlanParameterName` for why it is // a name, why it is not a cross-stack reference, and why it is not - // hard-coded. Read through the same extension layer, at cold start — - // with `PORTAL_ENABLED` now true the control-plane client IS built in - // every process, and this read is read 2 of the five listed on + // hard-coded. Read through the same extension layer, on the first + // portal request per execution environment — with `PORTAL_ENABLED` + // now true the control-plane client is built in every process that + // serves the portal, and this read is read 2 of the five listed on // `PORTAL_ENABLED`, the one whose parameter `ApiGatewayStack` publishes // after this stack deploys. // diff --git a/infra/src/lib/stacks/observability-stack.ts b/infra/src/lib/stacks/observability-stack.ts index 062b00a3..eec14927 100644 --- a/infra/src/lib/stacks/observability-stack.ts +++ b/infra/src/lib/stacks/observability-stack.ts @@ -321,10 +321,11 @@ export class ObservabilityStack extends cdk.Stack { */ public readonly apiHandlerErrorAlarm: cloudwatch.Alarm; /** - * api-handler `portal closed at cold start` alarm (task 0249): the portal - * is closed in that execution environment. + * api-handler `portal sources failed to load` alarm (tasks 0249, 0311): a + * portal source failed to load for one request, which answered as + * unavailable. Nothing stays closed; the next portal request retries. */ - public readonly apiHandlerPortalClosedAlarm: cloudwatch.Alarm; + public readonly apiHandlerPortalLoadFailedAlarm: cloudwatch.Alarm; /** * API Gateway 5xx alarm (task 0249): counts router-returned 5xx and Lambda * throttles, which AWS/Lambda `Errors` does not. @@ -1227,21 +1228,27 @@ export class ObservabilityStack extends cdk.Stack { description: `api-handler invocation-error alarm for ${config.envName}`, }); - // Task 0249 — the portal-closed signal. `main.rs` uses - // `tracing_subscriber::fmt().json()` without `flatten_event`, and the - // Lambda Text log format passes the line through raw, so the key is - // `$.fields.message` — confirmed against a real 2026-09-18 production + // Tasks 0249, 0311 — the portal-load-failed signal. The line is logged + // by `packages/prices-api/src/portal/sources.rs`, once per failed load of + // the portal's sources (the first portal request in an execution + // environment, or any request after a failure). `main.rs` owns the + // subscriber: `tracing_subscriber::fmt().json()` without `flatten_event`, + // and the Lambda Text log format passes the line through raw, so the key + // is `$.fields.message` — confirmed against a real 2026-09-18 production // line. The prefix wildcard keeps the match independent of the rest of - // the sentence (main.rs:58-63). Namespace follows this stack's - // `Prices/` convention. A metric filter publishes on behalf - // of CloudWatch Logs and needs no IAM grant. + // the sentence. Namespace follows this stack's `Prices/` + // convention. A metric filter publishes on behalf of CloudWatch Logs and + // needs no IAM grant. // - // The prefix below and the log line in main.rs are tied together by - // `tools/scripts/portal-closed-filter-guard.test.mjs` — reword one + // The prefix below and the log line in sources.rs are tied together by + // `tools/scripts/portal-load-failed-filter-guard.test.mjs` — reword one // without the other and that test fails, instead of the alarm going - // quiet. Expect this alarm during a load test: both times it would have - // fired so far (2026-09-18, 45 and 198 lines) were bursts of cold starts - // throttling the Parameter Store reads — real closures, not noise. + // quiet. `/v1` cold starts no longer read Parameter Store (task 0311), so + // a `/v1` load test should not fire this; portal traffic during SSM + // throttling can, and that is a real failure a visitor saw. + // + // Replaces task 0249's `portal-closed` filter and alarm 1-for-1 under + // new logical ids, so CloudFormation replaces both on deploy — expected. // // Log group imported BY NAME, not by ComputeStack construct reference — // ComputeStack creates it (compute-stack.ts:479); a construct reference @@ -1253,53 +1260,53 @@ export class ObservabilityStack extends cdk.Stack { 'ApiHandlerLogGroup', lambdaLogGroupName(config.envName, 'api-handler'), ); - const portalClosedFilter = new logs.MetricFilter( + const portalLoadFailedFilter = new logs.MetricFilter( this, - 'ApiHandlerPortalClosedFilter', + 'ApiHandlerPortalLoadFailedFilter', { logGroup: apiHandlerLogGroup, filterPattern: logs.FilterPattern.stringValue( '$.fields.message', '=', - 'portal closed at cold start*', + 'portal sources failed to load*', ), metricNamespace: 'Prices/ApiHandler', - metricName: 'PortalClosedAtColdStart', + metricName: 'PortalSourcesLoadFailed', metricValue: '1', // No defaultValue: missing data must stay missing, so the alarm // below reads OK on a quiet log group rather than a false zero. }, ); - this.apiHandlerPortalClosedAlarm = new cloudwatch.Alarm( + this.apiHandlerPortalLoadFailedAlarm = new cloudwatch.Alarm( this, - 'ApiHandlerPortalClosedAlarm', + 'ApiHandlerPortalLoadFailedAlarm', { - alarmName: `prices-${config.envName}-api-handler-portal-closed`, + alarmName: `prices-${config.envName}-api-handler-portal-load-failed`, // MetricFilter.metric() defaults to statistic 'avg'; pass Sum // explicitly (RESEARCH §2). - metric: portalClosedFilter.metric({ + metric: portalLoadFailedFilter.metric({ statistic: 'Sum', period: cdk.Duration.minutes(5), }), - alarmDescription: `The api-handler logged "portal closed at cold start": a portal source (the Discord OAuth secret, the free-plan id, or an eligibility parameter) failed to load at cold start, so the portal is CLOSED in that execution environment until it is recycled, while /v1 is unaffected. Closure is per execution environment, so /config may answer enabled: false from one environment and true from another. Fix: read the line's error field, which names the failing variable, fix that secret or parameter, then recycle the environments (redeploy, or bump the function configuration). The alarm returns to OK one period later, silently: OK does NOT mean the portal reopened. Runbook docs/runbooks/portal-oauth-deploy-prep.md; task 0249.`, + alarmDescription: `The api-handler logged "portal sources failed to load": a portal request (the first in an execution environment, or /config) could not load a portal source (the Discord OAuth secret, the free-plan id, the API id, the guild id or the min account age) after its retries. That one request answered as unavailable (/config enabled: false; /key, /usage and /me 503; sign-in lands on a failure page) and the next portal request retries, so no recycle is needed. /v1 is unaffected. Fix: read the line's error field, which names the failing variable. A persistent misconfiguration logs on every portal request, so the alarm keeps firing while it lasts. Runbook docs/runbooks/portal-oauth-deploy-prep.md; tasks 0249, 0311.`, threshold: 1, evaluationPeriods: 1, datapointsToAlarm: 1, comparisonOperator: cloudwatch.ComparisonOperator.GREATER_THAN_OR_EQUAL_TO_THRESHOLD, - // Missing = no closure logged = OK. + // Missing = no failed load logged = OK. treatMissingData: cloudwatch.TreatMissingData.NOT_BREACHING, }, ); - // Alarm action only, no OK action. The line is logged once, at the cold - // start that closed the portal; the next 5-min window is empty, so the - // alarm returns to OK while the environment is still closed. An OK - // notification would read as "recovered" when nothing was. - this.apiHandlerPortalClosedAlarm.addAlarmAction(snsAction); - - new cdk.CfnOutput(this, 'ApiHandlerPortalClosedAlarmName', { - value: this.apiHandlerPortalClosedAlarm.alarmName, - description: `api-handler portal-closed-at-cold-start alarm for ${config.envName}`, + // Alarm action only, no OK action. OK means no failed load in the last + // 5 minutes, and with no portal traffic in that window it proves + // nothing: an OK notification would read as "recovered" when nothing + // was tried. + this.apiHandlerPortalLoadFailedAlarm.addAlarmAction(snsAction); + + new cdk.CfnOutput(this, 'ApiHandlerPortalLoadFailedAlarmName', { + value: this.apiHandlerPortalLoadFailedAlarm.alarmName, + description: `api-handler portal-load-failed alarm for ${config.envName}`, }); // Task 0282 — the forced-progress escape hatch fired. The reconcile loop diff --git a/tools/scripts/portal-closed-filter-guard.test.mjs b/tools/scripts/portal-closed-filter-guard.test.mjs deleted file mode 100644 index ad9f7f72..00000000 --- a/tools/scripts/portal-closed-filter-guard.test.mjs +++ /dev/null @@ -1,137 +0,0 @@ -// The portal-closed alarm (task 0249) hangs on one string. ObservabilityStack's -// metric filter matches `{ $.fields.message = "portal closed at cold start*" }` -// on the api-handler log group; `packages/prices-api/src/main.rs` is what logs -// it. Nothing else ties the two together: reword the log line, or flatten the -// JSON subscriber, and the alarm goes quiet for good with every check green. -// Asserted here: -// -// - the filter reads `$.fields.message` and matches by prefix; -// - `main.rs` has a `tracing::error!` whose message starts with that prefix; -// - the subscriber is `fmt().json()` and not flattened, which is what puts -// the message under `fields`. -// -// The pattern itself was proven against AWS's evaluator with -// `aws logs test-metric-filter` (task 0249's notes); this guards the drift. -// -// Run: npx nx test @rumblefish/stellar-prices-api-aws-cdk - -import assert from 'node:assert/strict'; -import { readFileSync } from 'node:fs'; -import { dirname, join } from 'node:path'; -import { test } from 'node:test'; -import { fileURLToPath } from 'node:url'; - -const here = dirname(fileURLToPath(import.meta.url)); -const repoRoot = join(here, '..', '..'); -const stackSource = readFileSync( - join(repoRoot, 'infra', 'src', 'lib', 'stacks', 'observability-stack.ts'), - 'utf8', -); -const mainSource = readFileSync( - join(repoRoot, 'packages', 'prices-api', 'src', 'main.rs'), - 'utf8', -); - -// The portal-closed filter's own options object — from its construct id to -// the `);` that closes `new logs.MetricFilter(`. Searching inside that block -// only: an unbounded search would run on to the FIRST `stringValue(...)` -// anywhere later in the file, so a second filter added after this one could -// stand in for it and the guard would pass on the wrong filter. -const portalClosedFilterBlock = (source) => { - const block = source.match( - /new logs\.MetricFilter\(\s*this,\s*'ApiHandlerPortalClosedFilter',([\s\S]*?)\n\s*\);/, - ); - return block ? block[1] : undefined; -}; - -// `FilterPattern.stringValue('', '=', '*')` inside that block → -// { key, prefix }, or undefined when the filter is not a prefix match on a key. -const portalClosedFilter = (source) => { - const filter = portalClosedFilterBlock(source)?.match( - /FilterPattern\.stringValue\(\s*'([^']+)',\s*'='\s*,\s*'([^']+)\*',?\s*\)/, - ); - return filter ? { key: filter[1], prefix: filter[2] } : undefined; -}; - -// Does some `tracing::error!(…)` carry a message literal starting with `prefix`? -const logsErrorStartingWith = (source, prefix) => - [...source.matchAll(/tracing::error!\(([\s\S]*?)\);/g)].some(([, body]) => - body.includes(`"${prefix}`), - ); - -const subscriberPutsMessageUnderFields = (source) => { - const subscriber = source.match( - /tracing_subscriber::fmt\(\)([\s\S]*?)\.init\(\)/, - ); - return ( - subscriber !== null && - /\.json\(\)/.test(subscriber[1]) && - !/\.flatten_event\(\s*true\s*\)/.test(subscriber[1]) - ); -}; - -test('the portal-closed filter matches $.fields.message by prefix', () => { - const filter = portalClosedFilter(stackSource); - - assert.ok( - filter, - 'found no prefix FilterPattern on the portal-closed filter', - ); - assert.equal(filter.key, '$.fields.message'); -}); - -test('main.rs logs an error starting with the prefix the filter matches', () => { - const { prefix } = portalClosedFilter(stackSource); - - assert.ok( - logsErrorStartingWith(mainSource, prefix), - `no tracing::error! in main.rs starts with "${prefix}" — the portal-closed alarm would never fire`, - ); -}); - -test('the subscriber keeps the message under `fields`', () => { - assert.ok( - subscriberPutsMessageUnderFields(mainSource), - 'main.rs must log through fmt().json() without flatten_event(true), or $.fields.message matches nothing', - ); -}); - -test('a reworded log line is refused', () => { - const reworded = mainSource.replace( - 'portal closed at cold start', - 'the portal was closed at cold start', - ); - - assert.notEqual(reworded, mainSource); - assert.equal( - logsErrorStartingWith(reworded, 'portal closed at cold start'), - false, - ); -}); - -test('a flattened subscriber is refused', () => { - const flattened = mainSource.replace( - '.json()', - '.json().flatten_event(true)', - ); - - assert.notEqual(flattened, mainSource); - assert.equal(subscriberPutsMessageUnderFields(flattened), false); -}); - -test('a second filter later in the file cannot stand in for the portal-closed one', () => { - // The portal-closed filter moves off `stringValue`; another filter with the - // right pattern is added after it. The alarm now matches nothing. - const swapped = stackSource - .replace( - /filterPattern: logs\.FilterPattern\.stringValue\(\s*'\$\.fields\.message',\s*'=',\s*'portal closed at cold start\*',\s*\)/, - 'filterPattern: logs.FilterPattern.literal(\'{ $.message = "portal closed at cold start*" }\')', - ) - .replace( - 'this.apiHandlerPortalClosedAlarm = new cloudwatch.Alarm(', - "new logs.MetricFilter(this, 'OtherFilter', { logGroup: apiHandlerLogGroup, filterPattern: logs.FilterPattern.stringValue('$.fields.message', '=', 'portal closed at cold start*'), metricNamespace: 'X', metricName: 'Y' });\nthis.apiHandlerPortalClosedAlarm = new cloudwatch.Alarm(", - ); - - assert.notEqual(swapped, stackSource); - assert.equal(portalClosedFilter(swapped), undefined); -}); diff --git a/tools/scripts/portal-load-failed-filter-guard.test.mjs b/tools/scripts/portal-load-failed-filter-guard.test.mjs new file mode 100644 index 00000000..f625197b --- /dev/null +++ b/tools/scripts/portal-load-failed-filter-guard.test.mjs @@ -0,0 +1,196 @@ +// The portal-load-failed alarm (tasks 0249, 0311) hangs on one string. +// ObservabilityStack's metric filter matches +// `{ $.fields.message = "portal sources failed to load*" }` on the api-handler +// log group; `packages/prices-api/src/portal/sources.rs` is what logs it, and +// `packages/prices-api/src/main.rs` owns the JSON subscriber that puts the +// message under `fields`. Nothing else ties the three together: reword the log +// line, flatten the subscriber, or point the alarm at another metric, and the +// alarm goes quiet for good with every check green. Asserted here: +// +// - the filter reads `$.fields.message` and matches by the exact prefix; +// - `sources.rs` has a `tracing::error!` whose message starts with it; +// - the subscriber in `main.rs` is `fmt().json()` and not flattened; +// - the alarm takes its metric from that filter, under its own name; +// - the alarm's description fits CloudWatch's 1024-character limit. +// +// The pattern shape was proven against AWS's evaluator with +// `aws logs test-metric-filter` (task 0249's notes); this guards the drift. +// +// Run: npx nx test @rumblefish/stellar-prices-api-aws-cdk + +import assert from 'node:assert/strict'; +import { readFileSync } from 'node:fs'; +import { dirname, join } from 'node:path'; +import { test } from 'node:test'; +import { fileURLToPath } from 'node:url'; + +const PREFIX = 'portal sources failed to load'; + +const here = dirname(fileURLToPath(import.meta.url)); +const repoRoot = join(here, '..', '..'); +const stackSource = readFileSync( + join(repoRoot, 'infra', 'src', 'lib', 'stacks', 'observability-stack.ts'), + 'utf8', +); +const mainSource = readFileSync( + join(repoRoot, 'packages', 'prices-api', 'src', 'main.rs'), + 'utf8', +); +const sourcesSource = readFileSync( + join(repoRoot, 'packages', 'prices-api', 'src', 'portal', 'sources.rs'), + 'utf8', +); + +// The filter's own construct: `const = new logs.MetricFilter(` through +// the `);` that closes it. Searching inside that block only: an unbounded +// search would run on to the FIRST `stringValue(...)` anywhere later in the +// file, so a second filter added after this one could stand in for it and the +// guard would pass on the wrong filter. +const loadFailedFilterBlock = (source) => { + const block = source.match( + /const (\w+) = new logs\.MetricFilter\(\s*this,\s*'ApiHandlerPortalLoadFailedFilter',([\s\S]*?)\n\s*\);/, + ); + return block ? { variable: block[1], body: block[2] } : undefined; +}; + +// `FilterPattern.stringValue('', '=', '*')` inside that block → +// { key, prefix }, or undefined when the filter is not a prefix match on a key. +const loadFailedFilter = (source) => { + const filter = loadFailedFilterBlock(source)?.body.match( + /FilterPattern\.stringValue\(\s*'([^']+)',\s*'='\s*,\s*'([^']+)\*',?\s*\)/, + ); + return filter ? { key: filter[1], prefix: filter[2] } : undefined; +}; + +// The alarm's construct, from its id to the `);` that closes it. +const loadFailedAlarmBlock = (source) => { + const block = source.match( + /new cloudwatch\.Alarm\(\s*this,\s*'ApiHandlerPortalLoadFailedAlarm',([\s\S]*?)\n\s*\);/, + ); + return block ? block[1] : undefined; +}; + +// The alarm's `alarmDescription`, backtick or quoted. +const alarmDescription = (block) => { + const found = block?.match( + /alarmDescription:\s*(?:`([^`]*)`|'([^']*)'|"([^"]*)")/, + ); + return found ? (found[1] ?? found[2] ?? found[3]) : undefined; +}; + +// Does some `tracing::error!(…)` carry a message literal starting with `prefix`? +const logsErrorStartingWith = (source, prefix) => + [...source.matchAll(/tracing::error!\(([\s\S]*?)\);/g)].some(([, body]) => + body.includes(`"${prefix}`), + ); + +const subscriberPutsMessageUnderFields = (source) => { + const subscriber = source.match( + /tracing_subscriber::fmt\(\)([\s\S]*?)\.init\(\)/, + ); + return ( + subscriber !== null && + /\.json\(\)/.test(subscriber[1]) && + !/\.flatten_event\(\s*true\s*\)/.test(subscriber[1]) + ); +}; + +test('the portal-load-failed filter matches $.fields.message by the exact prefix', () => { + const filter = loadFailedFilter(stackSource); + + assert.ok( + filter, + 'found no prefix FilterPattern on the portal-load-failed filter', + ); + assert.equal(filter.key, '$.fields.message'); + assert.equal(filter.prefix, PREFIX); +}); + +test('sources.rs logs an error starting with the prefix the filter matches', () => { + const { prefix } = loadFailedFilter(stackSource); + + assert.ok( + logsErrorStartingWith(sourcesSource, prefix), + `no tracing::error! in sources.rs starts with "${prefix}" — the portal-load-failed alarm would never fire`, + ); +}); + +test('the subscriber in main.rs keeps the message under `fields`', () => { + assert.ok( + subscriberPutsMessageUnderFields(mainSource), + 'main.rs must log through fmt().json() without flatten_event(true), or $.fields.message matches nothing', + ); +}); + +test("the alarm takes its metric from the filter, under the alarm's own name", () => { + const { variable } = loadFailedFilterBlock(stackSource); + const alarm = loadFailedAlarmBlock(stackSource); + + assert.ok(alarm, 'found no ApiHandlerPortalLoadFailedAlarm'); + assert.ok( + alarm.includes(`${variable}.metric(`), + `the alarm must read ${variable}.metric(…), or it watches another metric`, + ); + assert.ok(alarm.includes('api-handler-portal-load-failed')); +}); + +test('the alarm description fits CloudWatch’s 1024-character limit', () => { + const description = alarmDescription(loadFailedAlarmBlock(stackSource)); + + assert.ok(description, 'found no alarmDescription on the alarm'); + assert.ok( + description.length <= 1024, + `alarmDescription is ${description.length} characters; CloudWatch refuses more than 1024`, + ); +}); + +test('a reworded log line is refused', () => { + const reworded = sourcesSource.replace( + PREFIX, + 'the portal sources failed to load', + ); + + assert.notEqual(reworded, sourcesSource); + assert.equal(logsErrorStartingWith(reworded, PREFIX), false); +}); + +test('a flattened subscriber is refused', () => { + const flattened = mainSource.replace( + '.json()', + '.json().flatten_event(true)', + ); + + assert.notEqual(flattened, mainSource); + assert.equal(subscriberPutsMessageUnderFields(flattened), false); +}); + +test('a second filter later in the file cannot stand in for the portal-load-failed one', () => { + // The filter moves off `stringValue`; another filter with the right + // pattern is added after it. The alarm now matches nothing. + const swapped = stackSource + .replace( + /filterPattern: logs\.FilterPattern\.stringValue\(\s*'\$\.fields\.message',\s*'=',\s*'portal sources failed to load\*',\s*\)/, + 'filterPattern: logs.FilterPattern.literal(\'{ $.message = "portal sources failed to load*" }\')', + ) + .replace( + 'this.apiHandlerPortalLoadFailedAlarm = new cloudwatch.Alarm(', + "new logs.MetricFilter(this, 'OtherFilter', { logGroup: apiHandlerLogGroup, filterPattern: logs.FilterPattern.stringValue('$.fields.message', '=', 'portal sources failed to load*'), metricNamespace: 'X', metricName: 'Y' });\nthis.apiHandlerPortalLoadFailedAlarm = new cloudwatch.Alarm(", + ); + + assert.notEqual(swapped, stackSource); + assert.equal(loadFailedFilter(swapped), undefined); +}); + +test('an alarm pointed at another metric is refused', () => { + const { variable } = loadFailedFilterBlock(stackSource); + const repointed = stackSource.replace( + `metric: ${variable}.metric(`, + 'metric: someOtherFilter.metric(', + ); + + assert.notEqual(repointed, stackSource); + assert.equal( + loadFailedAlarmBlock(repointed).includes(`${variable}.metric(`), + false, + ); +}); diff --git a/web/portal/src/api/portal.ts b/web/portal/src/api/portal.ts index 44a64036..9a0d706d 100644 --- a/web/portal/src/api/portal.ts +++ b/web/portal/src/api/portal.ts @@ -159,9 +159,10 @@ function failureMessage( * portal is open…" with no end, which is exactly the spinner that never resolves * the failure branch exists to avoid. * - * Ten seconds is well past a cold Lambda behind this route (the handler reads a - * cached SSM parameter and returns a single boolean) and well short of a - * visitor's patience. + * Ten seconds is well past a cold Lambda behind this route and well short of a + * visitor's patience. On the first call in an execution environment `/config` + * loads the portal's sources, bounded at 4 s by the backend; after that it + * returns a cached answer and a single boolean. */ const PROBE_TIMEOUT_MS = 10_000; From dfb3037b90c707bc551564029bbe9d22446d4bec Mon Sep 17 00:00:00 2001 From: Adam <65679285+adamkoot@users.noreply.github.com> Date: Fri, 25 Sep 2026 13:46:34 +0200 Subject: [PATCH 15/22] feat(lore-0311): retry every portal extension read with jittered backoff The extension's own SSM retry outlasts our 2 s client timeout, so a throttled read surfaced as "extension unreachable" with no retry on our side. Every portal read through the extension (the OAuth secret, the plan and API ids, the eligibility parameters) now goes through with_retry: 3 attempts, full-jitter backoff capped at 100 ms then 300 ms, from getrandom. Only MtlsError::Fetch is retried; missing env, parse and validation errors are not. - the per-issuance eligibility read keeps PARAMETER_TIMEOUT (2 s) as the bound around the whole retry, so the callback arithmetic is unchanged - prices_clickhouse::mtls, and the /v1 mTLS path, are untouched - stale cold-start wording in the portal modules describes the lazy load --- packages/prices-api/src/config.rs | 7 +- .../prices-api/src/portal/auth/discord.rs | 2 +- packages/prices-api/src/portal/auth/secret.rs | 6 +- packages/prices-api/src/portal/eligibility.rs | 29 ++- packages/prices-api/src/portal/extension.rs | 222 ++++++++++++++++++ .../prices-api/src/portal/keys/gateway.rs | 19 +- packages/prices-api/src/portal/mod.rs | 1 + packages/prices-api/src/portal/sources.rs | 5 +- 8 files changed, 263 insertions(+), 28 deletions(-) create mode 100644 packages/prices-api/src/portal/extension.rs diff --git a/packages/prices-api/src/config.rs b/packages/prices-api/src/config.rs index e0261cd2..9f11ff90 100644 --- a/packages/prices-api/src/config.rs +++ b/packages/prices-api/src/config.rs @@ -531,11 +531,12 @@ fn api_stage() -> Result { /// secret already use, so a warm container never calls Systems Manager on the /// path that issues a key. /// -/// The error is the message alone; the caller wraps it in the variant naming -/// which parameter it was reading. +/// Retried on a transient failure (`crate::portal::extension`). The error is +/// the message alone; the caller wraps it in the variant naming which +/// parameter it was reading. #[cfg(feature = "aws-mtls")] async fn fetch_parameter(name: &str) -> Result { - prices_clickhouse::mtls::fetch_parameter_string(name) + crate::portal::extension::parameter_string(name) .await .map_err(|e| e.to_string()) } diff --git a/packages/prices-api/src/portal/auth/discord.rs b/packages/prices-api/src/portal/auth/discord.rs index 27fa4f96..c1606dfa 100644 --- a/packages/prices-api/src/portal/auth/discord.rs +++ b/packages/prices-api/src/portal/auth/discord.rs @@ -379,7 +379,7 @@ const NOT_MEMBER_CODES: [u64; 2] = [10_007, 10_004]; /// which is what validates the operator's seed. Those two were allowed to /// disagree, and the disagreement had a cost: `guild_id` checked only for /// emptiness, so `stellar_test` — the value the task's own parameter table -/// named for the build period — passed the cold-start probe, deployed green, +/// named for the build period — passed the load-time probe, deployed green, /// and then answered "we could not verify your Discord membership" to every /// visitor forever, because the check ran here instead and produced /// [`MemberLookup::Unknown`] once per request. One predicate, so the seed diff --git a/packages/prices-api/src/portal/auth/secret.rs b/packages/prices-api/src/portal/auth/secret.rs index 836826ec..64596532 100644 --- a/packages/prices-api/src/portal/auth/secret.rs +++ b/packages/prices-api/src/portal/auth/secret.rs @@ -182,7 +182,7 @@ impl OauthSecret { /// doing so. #[cfg(feature = "aws-mtls")] async fn from_secrets_manager(name: &str) -> Result { - let json = prices_clickhouse::mtls::fetch_secret_string(name) + let json = crate::portal::extension::secret_string(name) .await .map_err(|e| SecretError::Fetch { name: name.to_string(), @@ -203,8 +203,8 @@ impl OauthSecret { /// Parse and validate. Split out from both loaders so the validation is /// testable without a file or an AWS runtime — it is the part that decides - /// whether a misconfiguration is caught at cold start or at a visitor's - /// callback. + /// whether a misconfiguration is caught when the portal's sources load or + /// at a visitor's callback. pub fn parse(json: &str) -> Result { let parsed: SecretJson = serde_json::from_str(json).map_err(|e| SecretError::Malformed(e.to_string()))?; diff --git a/packages/prices-api/src/portal/eligibility.rs b/packages/prices-api/src/portal/eligibility.rs index 0cf55491..7a23ebf6 100644 --- a/packages/prices-api/src/portal/eligibility.rs +++ b/packages/prices-api/src/portal/eligibility.rs @@ -72,6 +72,11 @@ const DISCORD_EPOCH_MS: u64 = 1_420_070_400_000; /// /// A read that exceeds it is [`EligibilityError::Fetch`], which every caller /// already renders as "could not verify" rather than as an accusation. +/// +/// It bounds the whole retried read (`crate::portal::extension`), not one +/// attempt, which is what keeps the callback's arithmetic unchanged (task +/// 0311). The retry sits inside the 2 s, so only fast failures — a quick +/// non-2xx, say — get a second attempt; a hung call spends the whole bound. /// Only the extension client can be slow — the `Direct` source is a value /// already in memory, and the build without the client fails immediately — so /// the constant lives with the code that can actually wait. @@ -86,7 +91,7 @@ pub(crate) const PARAMETER_TIMEOUT: std::time::Duration = std::time::Duration::f #[derive(Debug, Clone)] pub enum ParamSource { /// A literal value, for local runs and tests. Only constructed from the - /// environment in non-`lambda` builds — see `config::load_portal_eligibility`. + /// environment in non-`lambda` builds — see `config::portal_eligibility_from_env`. Direct(String), /// The **name** of an SSM parameter, fetched per action so an operator's /// change takes effect without a redeploy. @@ -127,15 +132,16 @@ impl EligibilitySettings { // caller uses to build the member URL, so the two cannot drift. // // Checking only for emptiness here is what let a guild *name* through: - // `stellar_test` passed the cold-start probe, so the deploy that + // `stellar_test` passed the load-time probe, so the deploy that // opened the portal came up green, and the refusal happened once per // visitor instead — as `Unknown`, which is the arm that deliberately // says nothing about anybody's membership. Every member would have // been told "we could not verify", indefinitely, with the actual fault // one `put-parameter` away. This is the failure the probe exists to - // turn into a cold-start error — a closed portal and a - // `portal closed at cold start` log line, per - // `AppConfig::load_portal_or_close`. + // turn into a failed portal load — a `portal sources failed to load` + // line naming the parameter, and `/config` answering + // `enabled: false` for that request and loading again on the next + // (`crate::portal::sources`). if !crate::portal::auth::discord::is_snowflake(&id) { return Err(EligibilityError::NotSnowflake { what: "discord-guild-id", @@ -155,8 +161,8 @@ impl EligibilitySettings { /// Why a parameter could not be resolved. /// -/// At cold start (the probe in `config::load_portal_eligibility`) any of these -/// is fatal; at action time they all land in [`Eligibility::Unknown`] — the +/// At load (the probe in `config::portal_eligibility_from_env`) any of these +/// fails the portal's load; at action time they all land in [`Eligibility::Unknown`] — the /// visitor is refused without accusation, and the log names the real fault. #[derive(Debug, thiserror::Error)] pub enum EligibilityError { @@ -178,10 +184,13 @@ pub enum EligibilityError { /// secret and the plan id already use. The extension's cache is what bounds /// how quickly an operator's change is honoured (~5 min), and is also why a /// per-action read does not call Systems Manager on a warm container. +/// +/// The retry (`crate::portal::extension`) runs INSIDE [`PARAMETER_TIMEOUT`], +/// so a per-issuance read still costs at most its 2 s. #[cfg(feature = "aws-mtls")] async fn fetch_parameter(name: &str) -> Result { - let fetch = prices_clickhouse::mtls::fetch_parameter_string(name); - match tokio::time::timeout(PARAMETER_TIMEOUT, fetch).await { + use crate::portal::extension; + match tokio::time::timeout(PARAMETER_TIMEOUT, extension::parameter_string(name)).await { Ok(result) => result.map_err(|e| EligibilityError::Fetch { name: name.to_string(), message: e.to_string(), @@ -513,7 +522,7 @@ mod tests { ); } - /// The cold-start probe must refuse a guild **name**. + /// The load-time probe must refuse a guild **name**. /// /// `stellar_test` is the value the task's own parameter table named for /// the build period, and before this check it passed: the deploy came up diff --git a/packages/prices-api/src/portal/extension.rs b/packages/prices-api/src/portal/extension.rs new file mode 100644 index 00000000..34d16dab --- /dev/null +++ b/packages/prices-api/src/portal/extension.rs @@ -0,0 +1,222 @@ +//! Retry around every portal read through the Parameters and Secrets extension +//! (task 0311). +//! +//! The extension retries SSM itself — three times, with backoff — and that is +//! slower than our 2 s client timeout, so a throttled read reached us as +//! "extension unreachable" with nothing retried on our side (lore note +//! `R-five-plan-herd-and-capacity-test.md`). One such read used to close the +//! portal. It now costs at most a retry. +//! +//! **Three attempts, two gaps: up to 100 ms, then up to 300 ms, full jitter.** +//! A fourth attempt never fits: at 2 s per call, three hung calls already +//! exceed `portal::sources::LOAD_BUDGET` (4 s), which is what bounds the whole +//! load. So a fast failure — an immediate non-2xx — gets all three attempts, +//! and a hung one about two. Jitter spreads a herd of environments retrying at +//! once; it comes from `getrandom`, already a dependency. +//! +//! **Only the raw fetch is retried**, and only its transient failure +//! (`MtlsError::Fetch`: unreachable, timed out, a non-2xx, an unreadable +//! envelope). A missing env var is permanent, and parsing and validation (a +//! malformed secret, an empty or non-snowflake parameter) happen outside this +//! wrapper, so they are never retried. +//! +//! `prices_clickhouse::mtls`, and with it the `/v1` mTLS path, is deliberately +//! untouched: this is a wrapper in prices-api around its two fetch functions. + +use std::fmt::Display; +use std::future::Future; +use std::time::Duration; + +/// How many times one read is tried, the first included. +#[cfg_attr(not(feature = "aws-mtls"), allow(dead_code))] +pub(crate) const ATTEMPTS: usize = 3; + +/// The ceiling of the jittered wait before the 2nd and the 3rd attempt. +#[cfg_attr(not(feature = "aws-mtls"), allow(dead_code))] +pub(crate) const BACKOFF_CAPS: [Duration; ATTEMPTS - 1] = + [Duration::from_millis(100), Duration::from_millis(300)]; + +/// Run `op` up to [`ATTEMPTS`] times, waiting a jittered backoff after each +/// failure `is_transient` accepts. Returns the first success, the first +/// permanent error at once, or the last error. +/// +/// Each retry logs one WARN naming `what`; the caller logs the final +/// failure, if it is one. +#[cfg_attr(not(feature = "aws-mtls"), allow(dead_code))] +pub(crate) async fn with_retry( + what: &str, + is_transient: impl Fn(&E) -> bool, + mut op: F, +) -> Result +where + E: Display, + F: FnMut() -> Fut, + Fut: Future>, +{ + let mut attempt = 0; + loop { + let error = match op().await { + Ok(value) => return Ok(value), + Err(error) => error, + }; + if attempt + 1 >= ATTEMPTS || !is_transient(&error) { + return Err(error); + } + let wait = full_jitter(BACKOFF_CAPS[attempt]); + attempt += 1; + tracing::warn!( + what, + attempt, + wait_ms = wait.as_millis() as u64, + error = %error, + "extension read failed; retrying" + ); + tokio::time::sleep(wait).await; + } +} + +/// A uniformly random wait in `[0, cap]` (AWS's "full jitter"), to the +/// millisecond. Half the cap if the OS cannot supply randomness — a backoff +/// is not a secret, and waiting is better than not retrying. +#[cfg_attr(not(feature = "aws-mtls"), allow(dead_code))] +fn full_jitter(cap: Duration) -> Duration { + let cap_ms = cap.as_millis() as u64; + match getrandom::u64() { + Ok(draw) => Duration::from_millis(draw % (cap_ms + 1)), + Err(_) => cap / 2, + } +} + +/// Read an SSM parameter through the extension, with the retry. +#[cfg(feature = "aws-mtls")] +pub(crate) async fn parameter_string( + name: &str, +) -> Result { + with_retry(name, is_transient, || { + prices_clickhouse::mtls::fetch_parameter_string(name) + }) + .await +} + +/// Read a Secrets Manager secret through the extension, with the retry. +#[cfg(feature = "aws-mtls")] +pub(crate) async fn secret_string( + name: &str, +) -> Result { + with_retry(name, is_transient, || { + prices_clickhouse::mtls::fetch_secret_string(name) + }) + .await +} + +/// Worth retrying: the fetch itself failed. Everything else — a missing env +/// var above all — fails the same way every time. +#[cfg(feature = "aws-mtls")] +fn is_transient(error: &prices_clickhouse::mtls::MtlsError) -> bool { + matches!(error, prices_clickhouse::mtls::MtlsError::Fetch(_)) +} + +#[cfg(test)] +mod tests { + use std::collections::VecDeque; + use std::sync::atomic::{AtomicUsize, Ordering}; + use std::sync::{Arc, Mutex}; + + use super::*; + + #[derive(Debug, PartialEq)] + enum Failure { + Transient(u32), + Permanent, + } + + impl Display for Failure { + fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result { + write!(f, "{self:?}") + } + } + + fn transient(error: &Failure) -> bool { + matches!(error, Failure::Transient(_)) + } + + /// Runs `script` through `with_retry`, returning the result and the call + /// count. + async fn run( + script: Vec>, + ) -> (Result<&'static str, Failure>, usize) { + let calls = Arc::new(AtomicUsize::new(0)); + let script = Arc::new(Mutex::new(script.into_iter().collect::>())); + let counter = calls.clone(); + let result = with_retry("test-read", transient, || { + counter.fetch_add(1, Ordering::SeqCst); + let next = script.lock().unwrap().pop_front().expect("script ran out"); + async move { next } + }) + .await; + (result, calls.load(Ordering::SeqCst)) + } + + #[tokio::test(start_paused = true)] + async fn two_transient_failures_then_a_success_takes_three_calls() { + let started = tokio::time::Instant::now(); + let (result, calls) = run(vec![ + Err(Failure::Transient(1)), + Err(Failure::Transient(2)), + Ok("value"), + ]) + .await; + assert_eq!(result, Ok("value")); + assert_eq!(calls, 3); + assert!(started.elapsed() <= BACKOFF_CAPS[0] + BACKOFF_CAPS[1]); + } + + #[tokio::test(start_paused = true)] + async fn three_transient_failures_return_the_last_and_stop() { + let (result, calls) = run(vec![ + Err(Failure::Transient(1)), + Err(Failure::Transient(2)), + Err(Failure::Transient(3)), + Ok("never reached"), + ]) + .await; + assert_eq!(result, Err(Failure::Transient(3))); + assert_eq!(calls, 3, "a fourth attempt was made"); + } + + #[tokio::test(start_paused = true)] + async fn a_permanent_failure_is_not_retried_and_does_not_wait() { + let started = tokio::time::Instant::now(); + let (result, calls) = run(vec![Err(Failure::Permanent), Ok("never reached")]).await; + assert_eq!(result, Err(Failure::Permanent)); + assert_eq!(calls, 1); + assert_eq!(started.elapsed(), Duration::ZERO); + } + + #[test] + fn full_jitter_stays_within_its_cap() { + for cap in BACKOFF_CAPS { + for _ in 0..1000 { + assert!(full_jitter(cap) <= cap); + } + } + assert_eq!(full_jitter(Duration::ZERO), Duration::ZERO); + } + + /// Two gaps, and the attempts fit the load budget only when failures are + /// fast: the arithmetic the module docs state. + #[test] + fn three_attempts_with_two_rising_caps() { + assert_eq!(ATTEMPTS, 3); + assert!(BACKOFF_CAPS[0] < BACKOFF_CAPS[1]); + } + + #[cfg(feature = "aws-mtls")] + #[test] + fn only_a_fetch_failure_is_transient() { + use prices_clickhouse::mtls::MtlsError; + assert!(is_transient(&MtlsError::Fetch("unreachable".into()))); + assert!(!is_transient(&MtlsError::MissingEnv("AWS_SESSION_TOKEN"))); + assert!(!is_transient(&MtlsError::BundleDecode("x".into()))); + } +} diff --git a/packages/prices-api/src/portal/keys/gateway.rs b/packages/prices-api/src/portal/keys/gateway.rs index dbb2ee86..e76394b8 100644 --- a/packages/prices-api/src/portal/keys/gateway.rs +++ b/packages/prices-api/src/portal/keys/gateway.rs @@ -398,12 +398,12 @@ fn select_plan(plans: &[UsagePlan], api_id: &str, stage: &str) -> Option Self { let shared = aws_config::load_defaults(aws_sdk_apigateway::config::BehaviorVersion::latest()).await; @@ -452,7 +453,7 @@ impl Gateway { if let Err(error) = provider.provide_credentials().await { tracing::warn!( error = %error, - "could not resolve AWS credentials at cold start; every control-plane \ + "could not resolve AWS credentials while loading the portal sources; every control-plane \ call will retry the resolution" ); } diff --git a/packages/prices-api/src/portal/mod.rs b/packages/prices-api/src/portal/mod.rs index e1a81f5b..bd7337ef 100644 --- a/packages/prices-api/src/portal/mod.rs +++ b/packages/prices-api/src/portal/mod.rs @@ -44,6 +44,7 @@ pub mod auth; pub mod eligibility; +pub(crate) mod extension; pub mod keys; pub mod period; pub mod sources; diff --git a/packages/prices-api/src/portal/sources.rs b/packages/prices-api/src/portal/sources.rs index fae9e7f6..34983721 100644 --- a/packages/prices-api/src/portal/sources.rs +++ b/packages/prices-api/src/portal/sources.rs @@ -56,8 +56,9 @@ use crate::portal::keys::gateway::Gateway; /// The whole load's ceiling, the five reads concurrent inside it. /// -/// Each read is bounded at 2 s by the extension client, so the budget leaves -/// room for a second attempt at a read that failed fast. Four seconds keeps a load plus the +/// Each read is bounded at 2 s by the extension client and tried up to three +/// times (`portal::extension`), so a fast failure gets its retries and a hung +/// read about two attempts. Four seconds keeps a load plus the /// slowest route after it inside the 15 s invocation — pinned below for /// `/usage` and `/key`, and in `auth::issue` for the callback. pub(crate) const LOAD_BUDGET: Duration = Duration::from_secs(4); From 0989f4e85697e3143fb2298ec4c514fcf2521258 Mon Sep 17 00:00:00 2001 From: Adam <65679285+adamkoot@users.noreply.github.com> Date: Fri, 25 Sep 2026 14:18:41 +0200 Subject: [PATCH 16/22] fix(lore-0311): land a failed portal load where each flow renders it - callback on a failed load follows the action the round-trip claims (pending cookie, then state, read unverified to pick between two failure literals only): an issue press lands on ?issue=failed, which the signed-in dashboard renders, instead of ?signin=failed, which it does not - login on a failed load lands on ?signin=failed / ?issue=failed, not the permanent "not yet available" card, and logs no second ERROR line - a lazy load before the callback is paid out of the token exchange's own 4 s timeout; the allowance rises to 2 s and the 1 s redirect margin is back, pinned in budget_arithmetic_fits_the_lambda - tests that relied on the environment loader now supply their sources --- .../prices-api/src/portal/auth/discord.rs | 11 + packages/prices-api/src/portal/auth/issue.rs | 67 +++-- packages/prices-api/src/portal/auth/mod.rs | 66 +++-- .../prices-api/src/portal/auth/state_token.rs | 62 +++++ packages/prices-api/tests/auth.rs | 19 +- packages/prices-api/tests/portal_auth.rs | 241 +++++++++++++----- 6 files changed, 356 insertions(+), 110 deletions(-) diff --git a/packages/prices-api/src/portal/auth/discord.rs b/packages/prices-api/src/portal/auth/discord.rs index c1606dfa..ee97ff8e 100644 --- a/packages/prices-api/src/portal/auth/discord.rs +++ b/packages/prices-api/src/portal/auth/discord.rs @@ -112,6 +112,10 @@ pub const DEFAULT_API_BASE: &str = "https://discord.com/api"; /// either constant, or adding a fourth call, needs this sum redone and /// `timeoutSeconds` in `infra/envs/production.json` checked against it — /// `issue::tests::budget_arithmetic_fits_the_lambda` does the sum. +/// +/// Since task 0311 the exchange's 4 s also pays for a lazy load of the +/// portal's sources in front of it: the callback passes [`exchange_code`] what +/// the load left of this timeout, so the sum above is unchanged. pub(super) const REQUEST_TIMEOUT: Duration = Duration::from_secs(4); /// Endpoints, separated from the credentials so tests can point them at a @@ -269,16 +273,23 @@ impl TokenResponse { /// endpoint documents it. `client_secret_post` rather than HTTP Basic is /// Discord's own example; both are RFC 6749-legal and the difference is not /// security-relevant over TLS. +/// +/// `timeout` replaces the client's [`REQUEST_TIMEOUT`] for this one call: the +/// callback gives the exchange what a lazy load of the portal's sources left +/// of it (task 0311), so the two together never exceed the one term the +/// callback's budget has for them. pub async fn exchange_code( client: &reqwest::Client, endpoints: &Endpoints, secret: &super::secret::OauthSecret, code: &str, code_verifier: &str, + timeout: Duration, ) -> Result { let url = endpoints.token_url(); let response = client .post(&url) + .timeout(timeout) .form(&[ ("grant_type", "authorization_code"), ("code", code), diff --git a/packages/prices-api/src/portal/auth/issue.rs b/packages/prices-api/src/portal/auth/issue.rs index c27de825..67ef42c0 100644 --- a/packages/prices-api/src/portal/auth/issue.rs +++ b/packages/prices-api/src/portal/auth/issue.rs @@ -133,10 +133,11 @@ pub(super) fn capped_query(next_eligible_date: &str) -> String { /// service is" — which is exactly true here, and true *before* any check ran, /// which is why that state's copy does not claim eligibility passed. /// -/// Loud in CloudWatch, because this is a deployment fault: the portal's load -/// yields all its sources or fails, so one of these lines follows either a -/// `portal sources failed to load` line for the same request (all three -/// flags `false`) or a state the load did not catch. +/// Loud in CloudWatch, because this is a deployment fault: the production +/// load yields all three sources or fails, and a failed load never reaches +/// here — `login` lands it on `?issue=failed` itself, after `get`'s one +/// `portal sources failed to load` line (review WR-04). So this line means a +/// partial fixture, or a state the load did not catch. pub(super) fn refuse_issue_start( home: &str, oauth: bool, @@ -215,19 +216,27 @@ pub(super) fn refuse_issue_discord( /// [`RECONCILE_FLOOR`] for what happens when that is not enough. const ISSUE_BUDGET: Duration = Duration::from_secs(12); -/// The callback's share of the invocation for loading the portal's sources -/// (task 0311), measured from arrival like [`ISSUE_BUDGET`]. +/// How much of the token exchange's [`discord::REQUEST_TIMEOUT`] a lazy load +/// of the portal's sources may spend (task 0311), measured from arrival like +/// [`ISSUE_BUDGET`]. /// /// The sources load on the first portal request an execution environment /// sees, and the callback can be that request: login and callback often land /// on different environments. The arithmetic on `discord::REQUEST_TIMEOUT` /// already spends 14 s of the 15 s invocation on the exchange, the parameter -/// reads and two Discord reads, so the load gets what is left — under a -/// second, with room for the redirect. A callback that arrives here later -/// than this lands on a retryable failure BEFORE the token exchange, and the -/// next attempt finds the sources loaded. `budget_arithmetic_fits_the_lambda` -/// pins the sum. -pub(super) const SOURCES_ALLOWANCE: Duration = Duration::from_millis(500); +/// reads and two Discord reads, leaving 1 s for the redirect and the runtime. +/// So the load does not get a term of its own: it comes OUT of the exchange, +/// which is given `REQUEST_TIMEOUT` minus the time already spent (review +/// WR-03). Load plus exchange stay 4 s, the sum stays 14 s, the margin 1 s. +/// +/// Two seconds, not the 500 ms it first was: under the SSM slowness this +/// task exists for, a successful load (five reads, credentials, a retry) +/// routinely passes half a second, and every callback on a fresh environment +/// then failed. Past two seconds the exchange would get less than two, and a +/// callback lands on a retryable failure BEFORE the exchange instead. The +/// next attempt finds the sources loaded only if it reaches this execution +/// environment. `budget_arithmetic_fits_the_lambda` pins all of it. +pub(super) const SOURCES_ALLOWANCE: Duration = Duration::from_secs(2); /// The least time worth starting a reconciliation with. /// @@ -637,24 +646,36 @@ mod tests { /// raise either constant, or add a call, and this is what fails first. /// /// Since task 0311 a lazy load of the portal's sources can precede all of - /// it; past [`SOURCES_ALLOWANCE`] the callback lands before the exchange, - /// so the allowance is the term that load adds. A load that fails - /// outright is followed only by a redirect, so the whole + /// it. It adds no term: it is paid out of the exchange's own + /// `REQUEST_TIMEOUT` (the callback passes the exchange what is left), and + /// past [`SOURCES_ALLOWANCE`] the callback lands before the exchange. A + /// load that fails outright is followed only by a redirect, so the whole /// `sources::LOAD_BUDGET` has to fit too. + /// + /// Pinned with the margin, not only `< 15 s` (review WR-03): the redirect + /// has to be serialised and sent, and the runtime has overhead of its + /// own, after the last timeout expires. #[test] fn budget_arithmetic_fits_the_lambda() { - const LAMBDA_TIMEOUT: std::time::Duration = std::time::Duration::from_secs(15); - let worst = SOURCES_ALLOWANCE - + discord::REQUEST_TIMEOUT * 3 - + crate::portal::eligibility::PARAMETER_TIMEOUT; + const LAMBDA_TIMEOUT: Duration = Duration::from_secs(15); + const REDIRECT_MARGIN: Duration = Duration::from_secs(1); + // Load + exchange, then the parameter reads, membership and identity. + let worst = discord::REQUEST_TIMEOUT + + crate::portal::eligibility::PARAMETER_TIMEOUT + + discord::REQUEST_TIMEOUT * 2; + assert!( + worst + REDIRECT_MARGIN <= LAMBDA_TIMEOUT, + "worst case {worst:?} leaves less than {REDIRECT_MARGIN:?} of {LAMBDA_TIMEOUT:?}" + ); + // The allowance leaves the exchange a real timeout: at least half. assert!( - worst < LAMBDA_TIMEOUT, - "worst case {worst:?} does not fit inside {LAMBDA_TIMEOUT:?}" + discord::REQUEST_TIMEOUT.saturating_sub(SOURCES_ALLOWANCE) + >= discord::REQUEST_TIMEOUT / 2 ); // And the reconciler's share is measured from arrival, so it cannot // extend the callback past the same line. - assert!(ISSUE_BUDGET < LAMBDA_TIMEOUT); - assert!(crate::portal::sources::LOAD_BUDGET < LAMBDA_TIMEOUT); + assert!(ISSUE_BUDGET + REDIRECT_MARGIN <= LAMBDA_TIMEOUT); + assert!(crate::portal::sources::LOAD_BUDGET + REDIRECT_MARGIN <= LAMBDA_TIMEOUT); } /// Every landing state is a distinct literal under the portal home, diff --git a/packages/prices-api/src/portal/auth/mod.rs b/packages/prices-api/src/portal/auth/mod.rs index 3105cbb1..4dd55cca 100644 --- a/packages/prices-api/src/portal/auth/mod.rs +++ b/packages/prices-api/src/portal/auth/mod.rs @@ -300,14 +300,17 @@ async fn login( }, }; - // The sources could not be loaded for this request: the same landings an - // unprovisioned deployment gets, and the next press loads again. The - // failure itself was logged by `get`. + // The sources could not be loaded for this request (review WR-04): a + // retryable failure, landed where the press came from — sign-in's card + // or the dashboard — and never `?signin=not_open`, whose closed-portal + // card states a permanent condition and offers nothing to press. Nothing + // is logged here: `get` already logged the one line this failure earns. let Some(loaded) = state.sources.get().await else { - return match action { - Action::Issue => issue::refuse_issue_start(&state.home, false, false, false), - _ => unconfigured(&state.home), + let query = match action { + Action::Issue => issue::ISSUE_FAILED_QUERY, + _ => FAILED_QUERY, }; + return redirect(&format!("{}{query}", state.home), vec![]); }; // An issue round-trip on a deployment with no credentials, no control @@ -449,12 +452,27 @@ async fn callback( let started = std::time::Instant::now(); let home = state.home.as_ref(); - // A failed load lands on `?signin=failed`: which flow this was cannot be - // known before `state` is verified, and verifying it needs the very - // secret that failed to load. The pending cookie is left alone, as on - // every refusal before verification; the next attempt loads again. + // A failed load cannot verify `state` — verifying it needs the very + // secret that failed to load — but it still has to land where the + // visitor's page renders a failure (review CR-01). An issue round-trip + // started from the signed-in dashboard, which renders `?issue=…` and + // deliberately not `?signin=failed`; a sign-in renders `?signin=…` on its + // card. So the landing follows the action the round-trip CLAIMS — this + // browser's pending cookie first, then `state` — read unverified, which + // is safe only because it picks between two failure literals and + // authorises nothing (`state_token::claimed_action`). No claim: sign-in's. + // The pending cookie is left alone, as on every refusal before + // verification; a later attempt loads again. let Some(loaded) = state.sources.get().await else { - return redirect(&format!("{home}{FAILED_QUERY}"), vec![]); + let claimed = cookies::read(&headers, cookies::PENDING_COOKIE) + .as_deref() + .and_then(state_token::claimed_action) + .or_else(|| query.state.as_deref().and_then(state_token::claimed_action)); + let landing = match claimed { + Some(Action::Issue) => issue::ISSUE_FAILED_QUERY, + _ => FAILED_QUERY, + }; + return redirect(&format!("{home}{landing}"), vec![]); }; let Some(oauth) = loaded.oauth.as_deref() else { return unconfigured(home); @@ -565,18 +583,22 @@ async fn callback( } // A lazy load in front of this callback (the first portal request in - // this environment) spent time the arithmetic on `discord::REQUEST_TIMEOUT` - // does not have. Past `issue::SOURCES_ALLOWANCE` the exchange and the - // three reads after it could outlive the invocation, so land a - // retryable failure now, before any Discord call: the next attempt finds - // the sources loaded. - if started.elapsed() > issue::SOURCES_ALLOWANCE { + // this environment) spends time the arithmetic on + // `discord::REQUEST_TIMEOUT` does not have, so it comes out of the token + // exchange's own 4 s (review WR-03): the exchange gets what the load left + // of it, and load plus exchange stay one `REQUEST_TIMEOUT`. Past + // `issue::SOURCES_ALLOWANCE` what is left is too little to exchange a + // code in, so land a retryable failure now, before any Discord call. The + // next attempt finds the sources loaded only if it reaches this execution + // environment; another fresh one loads again, under the same allowance. + let elapsed = started.elapsed(); + if elapsed > issue::SOURCES_ALLOWANCE { let query = match accepted.action { Action::Issue => issue::ISSUE_FAILED_QUERY, _ => FAILED_QUERY, }; tracing::warn!( - elapsed_ms = started.elapsed().as_millis() as u64, + elapsed_ms = elapsed.as_millis() as u64, landing = query, "sign-in callback spent its allowance loading the portal sources; \ landing a retryable failure before the token exchange" @@ -590,6 +612,7 @@ async fn callback( oauth, code, &accepted.verifier, + discord::REQUEST_TIMEOUT.saturating_sub(elapsed), ) .await { @@ -795,7 +818,7 @@ const SESSION_UNAVAILABLE: &str = "session_unavailable"; /// The one `503`: the sources failed to load for this request (task 0311). /// Without the signing key no cookie can be checked, and "signed out" would /// be a lie to a visitor who is signed in — the page renders its failure -/// state instead, and the next call loads again. +/// state instead, and a call after the load cooldown loads again. async fn me(State(state): State, headers: HeaderMap) -> Response { let signed_out = MeResponse { authenticated: false, @@ -911,8 +934,9 @@ fn no_store(mut response: Response) -> Response { } /// Land a deployment that reached these routes with no credentials — the -/// portal open, and its sources either loaded without a secret (a test -/// fixture) or, on `login`, failed to load for this request. +/// portal open, and its sources loaded without a secret (a test fixture). A +/// load that FAILED is not this: it is transient and lands on a failure the +/// visitor can retry (`?signin=failed` / `?issue=failed`, review WR-04). /// /// **A landing, not the `503 sign_in_unconfigured` envelope it used to be** /// (task 0194's review). Both call sites are reached by a browser following a diff --git a/packages/prices-api/src/portal/auth/state_token.rs b/packages/prices-api/src/portal/auth/state_token.rs index af88dba4..befe0956 100644 --- a/packages/prices-api/src/portal/auth/state_token.rs +++ b/packages/prices-api/src/portal/auth/state_token.rs @@ -264,6 +264,42 @@ pub fn accept( }) } +/// The action a `state` or pending-login token **claims**, read WITHOUT +/// verifying it. +/// +/// For one caller and one decision: which failure landing a callback takes +/// when the portal's sources failed to load (task 0311, review CR-01), so the +/// signing key that would verify the token is exactly what is missing. The +/// two landings render in different places — `?issue=failed` on the signed-in +/// dashboard an issue round-trip started from, `?signin=failed` on the sign-in +/// card — and a landing the visitor's page does not render is a press that +/// ends with nothing on screen. +/// +/// ⚠️ **This authorises nothing and must never be used to.** It chooses +/// between two fixed literals, both failures, both reachable by anyone who +/// types `/api/?issue=failed` into a link — so a forged claim buys an +/// attacker nothing they did not already have. Everything a claim could +/// matter for goes through [`accept`], which verifies both halves before +/// reading a byte of either. `None` on anything that is not a token of ours +/// in shape: the caller then falls back to the sign-in landing. +pub fn claimed_action(token: &str) -> Option { + /// Only the one field; the rest of either claims type is ignored. + #[derive(Deserialize)] + struct Claimed { + action: Action, + } + // Ours are ~200 bytes; a bound keeps an unauthenticated decode small. + const MAX_TOKEN_BYTES: usize = 1024; + if token.len() > MAX_TOKEN_BYTES { + return None; + } + let (encoded, _unverified_mac) = token.split_once('.')?; + let payload = crypto::b64_decode(encoded)?; + serde_json::from_slice::(&payload) + .ok() + .map(|claimed| claimed.action) +} + /// Seconds since the Unix epoch. /// /// Saturating rather than `expect`: a clock before 1970 is not a reason to panic @@ -511,6 +547,32 @@ mod tests { ); } + /// Review CR-01: both halves claim their action, readable without the + /// key; anything not shaped like our token claims nothing. + #[test] + fn a_claimed_action_is_read_from_either_half_and_nothing_else() { + for action in [Action::SignIn, Action::Issue] { + let started = start(KEY, action, NOW); + assert_eq!(claimed_action(&started.state_param), Some(action)); + assert_eq!(claimed_action(&started.pending_cookie), Some(action)); + } + let forged = format!("{}.not-a-mac", crypto::b64_encode(br#"{"action":"issue"}"#)); + // Read, unverified — which is why it may only pick a failure landing. + assert_eq!(claimed_action(&forged), Some(Action::Issue)); + for junk in [ + "", + "s", + "no-dot-here", + ".", + "%%%.mac", + &format!("{}.mac", crypto::b64_encode(br#"{"action":"rework"}"#)), + &format!("{}.mac", crypto::b64_encode(b"not json")), + &format!("{}.mac", "A".repeat(2000)), + ] { + assert_eq!(claimed_action(junk), None, "claimed from {junk:?}"); + } + } + #[test] fn an_unknown_action_query_value_is_rejected_rather_than_defaulted() { assert_eq!(Action::parse("signin"), Some(Action::SignIn)); diff --git a/packages/prices-api/tests/auth.rs b/packages/prices-api/tests/auth.rs index 69adfef1..ec8b70e7 100644 --- a/packages/prices-api/tests/auth.rs +++ b/packages/prices-api/tests/auth.rs @@ -41,11 +41,20 @@ async fn send(req: Request) -> StatusCode { } async fn send_with(config: &AppConfig, req: Request) -> StatusCode { - app(config, AppState::without_ch()) - .oneshot(req) - .await - .unwrap() - .status() + // An open portal with nothing supplied would load its sources from the + // process environment on the first portal request (task 0311), making + // these tests depend on the developer's shell. The gate is what is under + // test here, so an open portal gets sources already loaded, and empty. + let router = if config.portal_enabled { + prices_api::app_with_portal( + config, + AppState::without_ch(), + prices_api::portal::sources::PortalSources::ready(Default::default()), + ) + } else { + app(config, AppState::without_ch()) + }; + router.oneshot(req).await.unwrap().status() } #[tokio::test] diff --git a/packages/prices-api/tests/portal_auth.rs b/packages/prices-api/tests/portal_auth.rs index 4db3d516..8d5c6de4 100644 --- a/packages/prices-api/tests/portal_auth.rs +++ b/packages/prices-api/tests/portal_auth.rs @@ -477,21 +477,19 @@ async fn login_lands_an_unwired_issue_round_trip_on_failed() { /// both cases — so the sign-in arm now lands too, on `?signin=not_open` /// rather than the issue arm's `?issue=failed`. The two literals still differ, /// because the two flows start from different pages. +/// +/// Sources loaded with nothing in them, supplied rather than left to the +/// environment loader (review WR-06): that is the unprovisioned fixture this +/// test is about, and a developer's exported `serve` variables cannot turn +/// it into a successful load. A load that FAILS lands elsewhere — see +/// `an_open_portal_whose_load_fails_lands_on_a_retryable_failure`. #[tokio::test] async fn neither_arm_503s_on_a_deployment_with_no_credentials() { - let config = AppConfig { - ch_enabled: false, - base_url: None, - api_keys: vec![], - portal_enabled: true, - portal_oauth: None, - portal_endpoints: Endpoints::default(), - portal_keys: None, - portal_eligibility: None, - portal_rate_limit: None, - portal_web_origin: None, - }; - let router = app(&config, AppState::without_ch()); + let router = prices_api::app_with_portal( + &open_portal_config(Endpoints::default()), + AppState::without_ch(), + prices_api::portal::sources::PortalSources::ready(Default::default()), + ); let signin = fetch(&router, LOGIN_PATH, &[]).await; assert_eq!(signin.status, StatusCode::SEE_OTHER); @@ -1681,25 +1679,15 @@ async fn logout_is_not_reachable_by_a_get() { // Misconfiguration // --------------------------------------------------------------------------- -/// An open portal with no credentials must say so, not present a sign-in that -/// silently 404s. With nothing supplied on the config the router loads the -/// portal's sources from the environment on the first portal request (task -/// 0311), and a test process has none of them, so every request here is a -/// failed load — the state a throttled or unprovisioned deployment is in. -/// -/// Since task 0194's review it says so on the page: `/auth/login` is opened as -/// a top-level navigation (a popup, in the bundle), so `503 JSON` was raw text -/// in a window with no way back. `?signin=not_open` renders [0183]'s -/// closed-portal card, which is exactly what this deployment is. -#[tokio::test] -async fn an_open_portal_with_no_credentials_lands_on_the_closed_card() { - let config = AppConfig { +/// A config for a router whose portal sources are supplied separately. +fn open_portal_config(endpoints: Endpoints) -> AppConfig { + AppConfig { ch_enabled: false, base_url: None, api_keys: vec![], portal_enabled: true, portal_oauth: None, - portal_endpoints: Endpoints::default(), + portal_endpoints: endpoints, // Task 0187: the control-plane client for self-service keys. `None` // is what every non-portal test wants — with no client in the // config there is no code path here that can reach API Gateway. @@ -1707,12 +1695,42 @@ async fn an_open_portal_with_no_credentials_lands_on_the_closed_card() { portal_eligibility: None, portal_rate_limit: None, portal_web_origin: None, - }; - let router = app(&config, AppState::without_ch()); + } +} + +/// Sources whose every load fails — a throttled or unreadable extension. +/// +/// Scripted rather than left to the environment loader (review WR-06): with +/// nothing supplied, `prices_api::app` loads from the process environment, +/// and a developer who has exported `serve`'s variables would get a +/// successful load and a failing test. +fn failing_sources() -> prices_api::portal::sources::PortalSources { + use prices_api::config::PortalLoadError; + use prices_api::portal::auth::secret::SecretError; + use prices_api::portal::sources::{LoadFuture, PortalSources}; + PortalSources::lazy(std::sync::Arc::new(|| -> LoadFuture { + Box::pin(async { Err(PortalLoadError::Oauth(SecretError::NoSource)) }) + })) +} + +/// A failed load is transient, so it lands on a failure the visitor can +/// retry — never on the closed-portal card below, which states a permanent +/// condition (review WR-04). +/// +/// Since task 0194's review login answers with a landing, not `503 JSON`: +/// `/auth/login` is opened as a top-level navigation (a popup, in the +/// bundle), so JSON was raw text in a window with no way back. +#[tokio::test] +async fn an_open_portal_whose_load_fails_lands_on_a_retryable_failure() { + let router = prices_api::app_with_portal( + &open_portal_config(Endpoints::default()), + AppState::without_ch(), + failing_sources(), + ); let login = fetch(&router, LOGIN_PATH, &[]).await; assert_eq!(login.status, StatusCode::SEE_OTHER); - assert_eq!(login.location(), "/api/?signin=not_open"); + assert_eq!(login.location(), "/api/?signin=failed"); // `/auth/me` cannot check a cookie without the signing key, and "nobody // is signed in" would be false for a visitor who is: a `503` the page @@ -1727,6 +1745,28 @@ async fn an_open_portal_with_no_credentials_lands_on_the_closed_card() { ); } +/// An open portal whose sources loaded without credentials — only a fixture +/// can build one, since the production load yields all three or fails — must +/// say so, not present a sign-in that silently 404s. `?signin=not_open` +/// renders [0183]'s closed-portal card, which is exactly what it is. +#[tokio::test] +async fn an_open_portal_loaded_without_credentials_lands_on_the_closed_card() { + let router = prices_api::app_with_portal( + &open_portal_config(Endpoints::default()), + AppState::without_ch(), + prices_api::portal::sources::PortalSources::ready(Default::default()), + ); + + let login = fetch(&router, LOGIN_PATH, &[]).await; + assert_eq!(login.status, StatusCode::SEE_OTHER); + assert_eq!(login.location(), "/api/?signin=not_open"); + + // No credentials means no sessions: a truthful "signed out". + let me = fetch(&router, ME_PATH, &[]).await; + assert_eq!(me.status, StatusCode::OK); + assert_eq!(me.json()["authenticated"], json!(false)); +} + // --------------------------------------------------------------------------- // A slow lazy load in front of the callback (task 0311) // --------------------------------------------------------------------------- @@ -1747,36 +1787,124 @@ fn router_with_sources( mock: &MockDiscord, sources: prices_api::portal::sources::PortalSources, ) -> Router { - let config = AppConfig { - ch_enabled: false, - base_url: None, - api_keys: vec![], - portal_enabled: true, - portal_oauth: None, - portal_endpoints: Endpoints { - api_base: mock.base.clone(), - ..Endpoints::default() - }, - portal_keys: None, - portal_eligibility: None, - portal_rate_limit: None, - portal_web_origin: None, - }; + let config = open_portal_config(Endpoints { + api_base: mock.base.clone(), + ..Endpoints::default() + }); prices_api::app_with_portal(&config, AppState::without_ch(), sources) } -/// Sources that take 1.5 s to load — well past `issue::SOURCES_ALLOWANCE`, -/// which the budget test keeps under a second — and then succeed. -fn slow_sources() -> prices_api::portal::sources::PortalSources { +/// Sources that take `millis` to load and then succeed. +fn sources_loading_for(millis: u64) -> prices_api::portal::sources::PortalSources { use prices_api::portal::sources::{LoadFuture, PortalSources}; - PortalSources::lazy(std::sync::Arc::new(|| -> LoadFuture { - Box::pin(async { - tokio::time::sleep(std::time::Duration::from_millis(1500)).await; + PortalSources::lazy(std::sync::Arc::new(move || -> LoadFuture { + Box::pin(async move { + tokio::time::sleep(std::time::Duration::from_millis(millis)).await; Ok(full_sources()) }) })) } +/// Past `issue::SOURCES_ALLOWANCE` (2 s), inside `sources::LOAD_BUDGET` (4 s). +fn slow_sources() -> prices_api::portal::sources::PortalSources { + sources_loading_for(2_500) +} + +/// Start an `action=issue` round-trip on `router`: `(state, pending cookie)`. +async fn start_issue_login(router: &Router) -> (String, String) { + let login = fetch(router, &format!("{LOGIN_PATH}?action=issue"), &[]).await; + assert_eq!(login.status, StatusCode::SEE_OTHER); + let pending = login + .cookie(cookies::PENDING_COOKIE) + .expect("login must set the pending-login cookie"); + let query = login.location().split_once('?').unwrap().1.to_string(); + let state = form_urlencoded::parse(query.as_bytes()) + .find(|(k, _)| k == "state") + .map(|(_, v)| v.into_owned()) + .expect("the authorize URL must carry `state`"); + (state, pending) +} + +/// Review CR-01. An issue round-trip starts from the signed-in dashboard, +/// which renders `?issue=…` and deliberately not `?signin=failed`; a callback +/// on a failed load must land there, or the press of "Get my API key" ends +/// with nothing on screen. `state` cannot be verified (the key is what failed +/// to load), so the landing follows the action the round-trip claims. +#[tokio::test] +async fn an_issue_callback_on_a_failed_load_lands_on_issue_failed() { + let mock = MockDiscord::start(GRANTED_SCOPE, None).await; + let loaded = router_with_sources( + &mock, + prices_api::portal::sources::PortalSources::ready(full_sources()), + ); + let failing = router_with_sources(&mock, failing_sources()); + let (state, pending) = start_issue_login(&loaded).await; + let callback = format!("{CALLBACK_PATH}?code=an-auth-code&state={state}"); + + // With this browser's pending cookie, and with `state` alone. + let with_cookie: &[(&str, &str)] = &[(cookies::PENDING_COOKIE, pending.as_str())]; + let without: &[(&str, &str)] = &[]; + for sent in [with_cookie, without] { + let reply = fetch(&failing, &callback, sent).await; + assert_eq!(reply.status, StatusCode::SEE_OTHER); + assert_eq!(reply.location(), "/api/?issue=failed"); + assert!( + reply.headers.get(header::SET_COOKIE).is_none(), + "an unverified callback must leave the pending cookie alone" + ); + } + assert_eq!(mock.exchanges(), 0); +} + +/// The same failed load on a sign-in round-trip keeps sign-in's landing: its +/// card is the one that renders `?signin=failed`. +#[tokio::test] +async fn a_sign_in_callback_on_a_failed_load_lands_on_signin_failed() { + let mock = MockDiscord::start(GRANTED_SCOPE, None).await; + let loaded = router_with_sources( + &mock, + prices_api::portal::sources::PortalSources::ready(full_sources()), + ); + let failing = router_with_sources(&mock, failing_sources()); + let started = start_login(&loaded).await; + + let reply = fetch( + &failing, + &format!("{CALLBACK_PATH}?code=an-auth-code&state={}", started.state), + &[(cookies::PENDING_COOKIE, &started.pending)], + ) + .await; + assert_eq!(reply.status, StatusCode::SEE_OTHER); + assert_eq!(reply.location(), "/api/?signin=failed"); + assert!(reply.headers.get(header::SET_COOKIE).is_none()); + assert_eq!(mock.exchanges(), 0); +} + +/// Review WR-03. A load inside the allowance no longer fails the callback: +/// the exchange runs, on what the load left of its timeout. (The 500 ms +/// allowance this replaced refused any load slower than half a second.) +#[tokio::test] +async fn a_callback_whose_load_fits_the_allowance_still_exchanges_the_code() { + let mock = MockDiscord::start(GRANTED_SCOPE, None).await; + let loaded = router_with_sources( + &mock, + prices_api::portal::sources::PortalSources::ready(full_sources()), + ); + let cold = router_with_sources(&mock, sources_loading_for(1_000)); + + let started = start_login(&loaded).await; + let reply = fetch( + &cold, + &format!("{CALLBACK_PATH}?code=an-auth-code&state={}", started.state), + &[(cookies::PENDING_COOKIE, &started.pending)], + ) + .await; + + assert_eq!(reply.status, StatusCode::SEE_OTHER); + assert_ne!(reply.location(), "/api/?signin=failed"); + assert_eq!(mock.exchanges(), 1, "the token exchange must run"); +} + /// Login on one environment (loaded), callback on another that has to load /// first — the common shape once `/v1` warms most environments. The load /// eats the callback's allowance, so it lands on a retryable failure before @@ -1816,16 +1944,7 @@ async fn an_issue_callback_that_spent_its_allowance_loading_lands_on_issue_faile ); let cold = router_with_sources(&mock, slow_sources()); - let login = fetch(&loaded, &format!("{LOGIN_PATH}?action=issue"), &[]).await; - assert_eq!(login.status, StatusCode::SEE_OTHER); - let pending = login - .cookie(cookies::PENDING_COOKIE) - .expect("login must set the pending-login cookie"); - let query = login.location().split_once('?').unwrap().1.to_string(); - let state = form_urlencoded::parse(query.as_bytes()) - .find(|(k, _)| k == "state") - .map(|(_, v)| v.into_owned()) - .expect("the authorize URL must carry `state`"); + let (state, pending) = start_issue_login(&loaded).await; let reply = fetch( &cold, From 9b7369f8f2ca78fd414d0498f0b1c6008836d597 Mon Sep 17 00:00:00 2001 From: Adam <65679285+adamkoot@users.noreply.github.com> Date: Fri, 25 Sep 2026 14:18:51 +0200 Subject: [PATCH 17/22] fix(lore-0311): retry only transient reads and cool down after a failed load - retry an extension read only on unreachable/timeout, 5xx, 429 and a 400 that names throttling; a 403/404/other 400, a missing field, a parse or client-build failure returns at once; the classified messages are pinned against mtls.rs, which stays untouched - a failed load is remembered for LOAD_FAILURE_COOLDOWN (2 s): inside it a portal request answers unavailable without loading and without a second alarm line; queued waiters no longer load in turn - the log assertions live in a binary of their own (portal_load_logs.rs) - a malformed OAuth secret's error no longer echoes field values - pin that a closed portal gets no loader, that a hung read gets one retry inside the load budget, and the observed retry schedule - app_with_portal debug-asserts the config carries no portal sources - guard test requires the prefix in the message literal, not a field - alarm description, runbook and stack comments state the cooldown --- docs/runbooks/portal-oauth-deploy-prep.md | 6 +- infra/src/lib/stacks/compute-stack.ts | 11 +- infra/src/lib/stacks/observability-stack.ts | 5 +- packages/prices-api/src/lib.rs | 16 ++ packages/prices-api/src/portal/auth/secret.rs | 58 ++++- packages/prices-api/src/portal/extension.rs | 215 ++++++++++++++++-- packages/prices-api/src/portal/mod.rs | 10 +- packages/prices-api/src/portal/sources.rs | 211 +++++++++++++++-- packages/prices-api/tests/portal_lazy_load.rs | 92 ++++++-- packages/prices-api/tests/portal_load_logs.rs | 184 +++++++++++++++ .../portal-load-failed-filter-guard.test.mjs | 36 ++- 11 files changed, 764 insertions(+), 80 deletions(-) create mode 100644 packages/prices-api/tests/portal_load_logs.rs diff --git a/docs/runbooks/portal-oauth-deploy-prep.md b/docs/runbooks/portal-oauth-deploy-prep.md index 7a0b1e29..19fc4700 100644 --- a/docs/runbooks/portal-oauth-deploy-prep.md +++ b/docs/runbooks/portal-oauth-deploy-prep.md @@ -254,8 +254,10 @@ execution environment (the `/config` probe triggers it): that `/config` answers `enabled: false` and the api-handler logs `portal sources failed to load` naming `PORTAL_OAUTH_SECRET_NAME`, which pages as `prices-production-api-handler-portal-load-failed` (tasks 0249, 0311) — rather -than silently serving a broken sign-in. The next request retries, so creating -the secret fixes it without a redeploy. A successful load is kept for the +than silently serving a broken sign-in. A failed load is remembered for a 2 s +cooldown (requests inside it answer the same way without reading anything), and +the first request after it retries, so creating the secret fixes it without a +redeploy. A successful load is kept for the execution environment's life, so changing the secret's VALUE later still needs a recycle, as before. `/v1` is unaffected either way. diff --git a/infra/src/lib/stacks/compute-stack.ts b/infra/src/lib/stacks/compute-stack.ts index 249fb391..79b6bd8a 100644 --- a/infra/src/lib/stacks/compute-stack.ts +++ b/infra/src/lib/stacks/compute-stack.ts @@ -501,9 +501,10 @@ export class ComputeStack extends cdk.Stack { // never asked; with `PORTAL_ENABLED` true (task 0194) the read happens on // the first portal request per execution environment (task 0311), and a // missing or misnamed secret answers that request as unavailable with a - // `portal sources failed to load` error log; the next request retries, - // and nothing stays closed. See the deploy-gate note on `PORTAL_ENABLED` - // below and `packages/prices-api/src/portal/sources.rs`. + // `portal sources failed to load` error log; the first request after a + // 2 s cooldown retries, and nothing stays closed. See the deploy-gate + // note on `PORTAL_ENABLED` below and + // `packages/prices-api/src/portal/sources.rs`. this.apiHandlerRole.addToPrincipalPolicy( new iam.PolicyStatement({ sid: 'ReadPortalOauthSecret', @@ -917,8 +918,8 @@ export class ComputeStack extends cdk.Stack { // A failed read answers THAT request as unavailable (`/config` // `enabled: false`, `503` on the portal's other routes) and logs // `portal sources failed to load`, naming the variable, on the - // api-handler; the next portal request retries, so nothing stays - // closed and no recycle is needed. It never touches init: `/v1` cold + // api-handler; the first portal request after a 2 s cooldown + // retries, so nothing stays closed and no recycle is needed. It never touches init: `/v1` cold // starts read none of these (task 0311). So deploying this ahead of // the operator steps ships a portal whose `/config` answers // `enabled: false` on each call until the missing source exists, not diff --git a/infra/src/lib/stacks/observability-stack.ts b/infra/src/lib/stacks/observability-stack.ts index eec14927..2206b2f4 100644 --- a/infra/src/lib/stacks/observability-stack.ts +++ b/infra/src/lib/stacks/observability-stack.ts @@ -323,7 +323,8 @@ export class ObservabilityStack extends cdk.Stack { /** * api-handler `portal sources failed to load` alarm (tasks 0249, 0311): a * portal source failed to load for one request, which answered as - * unavailable. Nothing stays closed; the next portal request retries. + * unavailable. Nothing stays closed; the first portal request after a + * 2 s cooldown retries. */ public readonly apiHandlerPortalLoadFailedAlarm: cloudwatch.Alarm; /** @@ -1288,7 +1289,7 @@ export class ObservabilityStack extends cdk.Stack { statistic: 'Sum', period: cdk.Duration.minutes(5), }), - alarmDescription: `The api-handler logged "portal sources failed to load": a portal request (the first in an execution environment, or /config) could not load a portal source (the Discord OAuth secret, the free-plan id, the API id, the guild id or the min account age) after its retries. That one request answered as unavailable (/config enabled: false; /key, /usage and /me 503; sign-in lands on a failure page) and the next portal request retries, so no recycle is needed. /v1 is unaffected. Fix: read the line's error field, which names the failing variable. A persistent misconfiguration logs on every portal request, so the alarm keeps firing while it lasts. Runbook docs/runbooks/portal-oauth-deploy-prep.md; tasks 0249, 0311.`, + alarmDescription: `The api-handler logged "portal sources failed to load": a portal request (the first in an execution environment, or /config) could not load a portal source (the Discord OAuth secret, the free-plan id, the API id, the guild id or the min account age) after its retries. That one request answered as unavailable (/config enabled: false; /key, /usage and /me 503; sign-in lands on a failure page) and the first portal request after a 2 s cooldown retries, so no recycle is needed. /v1 is unaffected. Fix: read the line's error field, which names the failing variable. A persistent misconfiguration logs on every load (at most one per environment per cooldown), so the alarm keeps firing while it lasts. Runbook docs/runbooks/portal-oauth-deploy-prep.md; tasks 0249, 0311.`, threshold: 1, evaluationPeriods: 1, datapointsToAlarm: 1, diff --git a/packages/prices-api/src/lib.rs b/packages/prices-api/src/lib.rs index aa66a55a..151e8872 100644 --- a/packages/prices-api/src/lib.rs +++ b/packages/prices-api/src/lib.rs @@ -61,12 +61,28 @@ pub fn app(config: &AppConfig, state: AppState) -> Router { /// Compiled out of the Lambda, like `Gateway::against` and /// `IssueDeps::with_deadline`: the deployed build contains one loader, the one /// that reads the environment. +/// +/// **`sources` is the only source of the portal's sources here** (review +/// IN-04). `config.portal_oauth`, `portal_keys` and `portal_eligibility` are +/// what [`app`] builds a ready cell from, and this seam does not read them — +/// so a caller must leave them `None` (debug-asserted) rather than have them +/// silently dropped. `config.portal_enabled` still decides the gate: a closed +/// portal answers `404` whatever `sources` holds, but only [`app`] guarantees +/// it holds no loader (`sources::sources_for`), so a test that wants that +/// guarantee goes through [`app`]. #[cfg(not(feature = "lambda"))] pub fn app_with_portal( config: &AppConfig, state: AppState, sources: portal::sources::PortalSources, ) -> Router { + debug_assert!( + config.portal_oauth.is_none() + && config.portal_keys.is_none() + && config.portal_eligibility.is_none(), + "app_with_portal ignores the config's portal sources; pass them in `sources` \ + (PortalSources::ready) or call `app`" + ); app_inner(config, state, sources) } diff --git a/packages/prices-api/src/portal/auth/secret.rs b/packages/prices-api/src/portal/auth/secret.rs index 64596532..807a5f06 100644 --- a/packages/prices-api/src/portal/auth/secret.rs +++ b/packages/prices-api/src/portal/auth/secret.rs @@ -126,6 +126,30 @@ pub enum SecretError { RedirectUriMismatch { uri: String, expected: &'static str }, } +/// What a parse failure of the secret may say in a log line (review IN-02). +/// +/// serde_json's own message echoes the offending VALUE for a type or value +/// error — ``invalid type: integer `123…`, expected a string`` — and this +/// error reaches the `portal sources failed to load` ERROR line, logged on +/// every load while the secret stays malformed. So only what cannot carry +/// secret material survives: a missing or duplicated field is named (the names +/// are this struct's, not the secret's), and everything else is reduced to +/// its category and position. +fn describe_malformed(error: &serde_json::Error) -> String { + use serde_json::error::Category; + let message = error.to_string(); + if message.starts_with("missing field `") || message.starts_with("duplicate field `") { + return message; + } + let what = match error.classify() { + Category::Syntax => "not valid JSON", + Category::Eof => "truncated JSON", + Category::Data => "a field of the wrong type or shape", + Category::Io => "unreadable", + }; + format!("{what} at line {} column {}", error.line(), error.column()) +} + /// The JSON as it sits in Secrets Manager. #[derive(Deserialize)] struct SecretJson { @@ -206,8 +230,8 @@ impl OauthSecret { /// whether a misconfiguration is caught when the portal's sources load or /// at a visitor's callback. pub fn parse(json: &str) -> Result { - let parsed: SecretJson = - serde_json::from_str(json).map_err(|e| SecretError::Malformed(e.to_string()))?; + let parsed: SecretJson = serde_json::from_str(json) + .map_err(|e| SecretError::Malformed(describe_malformed(&e)))?; for (field, value) in [ ("client_id", &parsed.client_id), @@ -321,6 +345,36 @@ mod tests { )); } + /// Review IN-02: a malformed secret's error reaches an ERROR line, so it + /// must not echo a value — only the field names this struct owns. + #[test] + fn a_malformed_secret_never_echoes_a_value() { + for json in [ + r#"{"client_id": "c", "client_secret": 987654321987, "redirect_uri": "r", "session_signing_key": "k"}"#, + r#"{"client_id": "c", "client_secret": true, "redirect_uri": "r", "session_signing_key": "k"}"#, + r#"{"client_id": "c", "client_secret": ["sekrit-in-a-list"], "redirect_uri": "r", "session_signing_key": "k"}"#, + r#"["sekrit-in-a-list"]"#, + r#"{"client_secret": "sekrit-truncated"#, + ] { + let err = OauthSecret::parse(json).unwrap_err(); + assert!(matches!(err, SecretError::Malformed(_)), "{json}"); + let text = err.to_string(); + for leaked in ["987654321987", "true", "sekrit"] { + assert!(!text.contains(leaked), "{text:?} echoes {leaked}"); + } + assert!(text.contains("line 1 column"), "{text:?}"); + } + + // A missing field is still named: the name is ours, not the secret's. + let missing = OauthSecret::parse(r#"{"client_id": "c"}"#) + .unwrap_err() + .to_string(); + assert!( + missing.contains("missing field `client_secret`"), + "{missing}" + ); + } + /// Nothing sensitive may reach a log line, a panic message or a trace. #[test] fn debug_redacts_every_field() { diff --git a/packages/prices-api/src/portal/extension.rs b/packages/prices-api/src/portal/extension.rs index 34d16dab..4608dc65 100644 --- a/packages/prices-api/src/portal/extension.rs +++ b/packages/prices-api/src/portal/extension.rs @@ -8,17 +8,30 @@ //! portal. It now costs at most a retry. //! //! **Three attempts, two gaps: up to 100 ms, then up to 300 ms, full jitter.** -//! A fourth attempt never fits: at 2 s per call, three hung calls already -//! exceed `portal::sources::LOAD_BUDGET` (4 s), which is what bounds the whole -//! load. So a fast failure — an immediate non-2xx — gets all three attempts, -//! and a hung one about two. Jitter spreads a herd of environments retrying at -//! once; it comes from `getrandom`, already a dependency. +//! Jitter spreads a herd of environments retrying at once; it comes from +//! `getrandom`, already a dependency. //! -//! **Only the raw fetch is retried**, and only its transient failure -//! (`MtlsError::Fetch`: unreachable, timed out, a non-2xx, an unreadable -//! envelope). A missing env var is permanent, and parsing and validation (a -//! malformed secret, an empty or non-snowflake parameter) happen outside this -//! wrapper, so they are never retried. +//! **What a retry buys depends on how the read failed** (review IN-03): +//! +//! - A **fast** failure — a `5xx` or `429` answered at once — gets all three +//! attempts inside `portal::sources::LOAD_BUDGET` (4 s). +//! - A **hung** read — the measured throttling case, where the extension's own +//! SSM backoff outlives our 2 s client timeout — gets **one retry at most**, +//! and that retry is cut short: attempt 1 times out at 2 s, attempt 2 starts +//! by 2.1 s, and the 4 s budget cancels it after ~1.9–2.0 s, before its own +//! 2 s timeout. It helps only if SSM answers within that window. The +//! per-issuance eligibility reads run inside their own 2 s +//! `eligibility::PARAMETER_TIMEOUT`, so a hung one there gets no retry at +//! all. `a_hung_read_gets_one_retry_inside_the_load_budget` pins this. +//! +//! **Only a transient fetch failure is retried** (review WR-01), classified from +//! the `MtlsError::Fetch` text `prices_clickhouse::mtls` builds — see +//! [`is_transient`] for exactly what is matched. Everything that fails the same +//! way on every attempt returns at once: a `403`/`404`/other `4xx`, a response +//! that parsed but lacks its field, a client that could not be built, a missing +//! env var. Parsing and validation of the value (a malformed secret, an empty +//! or non-snowflake parameter) happen outside this wrapper and are never +//! retried either. //! //! `prices_clickhouse::mtls`, and with it the `/v1` mTLS path, is deliberately //! untouched: this is a wrapper in prices-api around its two fetch functions. @@ -109,13 +122,67 @@ pub(crate) async fn secret_string( .await } -/// Worth retrying: the fetch itself failed. Everything else — a missing env -/// var above all — fails the same way every time. +/// Worth retrying: the fetch failed in a way the next attempt may not. #[cfg(feature = "aws-mtls")] fn is_transient(error: &prices_clickhouse::mtls::MtlsError) -> bool { - matches!(error, prices_clickhouse::mtls::MtlsError::Fetch(_)) + match error { + prices_clickhouse::mtls::MtlsError::Fetch(message) => is_transient_fetch(message), + // A missing env var, above all, fails the same way every time. + _ => false, + } } +/// Classify one `MtlsError::Fetch` message. +/// +/// `prices_clickhouse::mtls` is frozen (the `/v1` path), so this reads the +/// text it builds rather than a typed status. The four shapes, pinned against +/// that file by `the_fetch_messages_this_classifies_are_the_ones_mtls_builds`: +/// +/// | message | retried | +/// | --- | --- | +/// | `extension at … unreachable …` — connect error or the 2 s timeout | yes | +/// | `extension returned HTTP {status} …` | `5xx` and `429`, and a `400` naming throttling (below) | +/// | `response body parse failed` / `response missing … field` | no | +/// | `reqwest client build failed` | no | +/// +/// ⚠️ **A throttled read the extension answers `400` is NOT retried today.** +/// The extension surfaces SSM/Secrets Manager throttling as HTTP `400` +/// (measured 2026-09-24/25), the same status as `ParameterNotFound`, and +/// `mtls` does not put the response body in the message — so the two cannot +/// be told apart, and retrying every `400` would retry a missing parameter on +/// every load. A `400` is retried only when the message itself names +/// throttling (`ThrottlingException`, `Rate exceeded`), which it will if `mtls` +/// ever carries the body. The measured throttling failure is the hung read, +/// which arrives as "unreachable" and IS retried. +/// +/// Anything unrecognised is permanent: a retry is a cost paid on every load +/// while a fault lasts, so it has to be earned. +#[cfg_attr(not(feature = "aws-mtls"), allow(dead_code))] +fn is_transient_fetch(message: &str) -> bool { + if message.starts_with("extension at ") && message.contains(" unreachable ") { + return true; + } + let Some(status) = message + .strip_prefix("extension returned HTTP ") + .and_then(|rest| rest.get(..3)) + .and_then(|code| code.parse::().ok()) + else { + return false; + }; + match status { + 429 | 500..=599 => true, + 400 => THROTTLING_MARKERS + .iter() + .any(|marker| message.contains(marker)), + _ => false, + } +} + +/// What AWS calls throttling in an error body: the SSM / Secrets Manager +/// exception name and the API Gateway-style message. +#[cfg_attr(not(feature = "aws-mtls"), allow(dead_code))] +const THROTTLING_MARKERS: [&str; 2] = ["ThrottlingException", "Rate exceeded"]; + #[cfg(test)] mod tests { use std::collections::VecDeque; @@ -203,19 +270,127 @@ mod tests { assert_eq!(full_jitter(Duration::ZERO), Duration::ZERO); } - /// Two gaps, and the attempts fit the load budget only when failures are - /// fast: the arithmetic the module docs state. + /// The schedule as it runs (review IN-05): a read that keeps failing fast + /// is called exactly [`ATTEMPTS`] times, and the gap before the n-th retry + /// is never longer than its cap — observed on the paused clock, not read + /// back from the constants. + #[tokio::test(start_paused = true)] + async fn a_read_that_keeps_failing_is_called_three_times_within_the_caps() { + let calls = Arc::new(Mutex::new(Vec::new())); + let seen = calls.clone(); + let result: Result<(), Failure> = with_retry("test-read", transient, || { + seen.lock().unwrap().push(tokio::time::Instant::now()); + async { Err(Failure::Transient(0)) } + }) + .await; + assert!(result.is_err()); + let calls = calls.lock().unwrap(); + assert_eq!(calls.len(), ATTEMPTS); + for (gap, cap) in calls.windows(2).map(|w| w[1] - w[0]).zip(BACKOFF_CAPS) { + assert!( + gap <= cap, + "waited {gap:?} before a retry capped at {cap:?}" + ); + } + } + + /// What the module doc claims for the measured failure (review IN-03): a + /// read that hangs until the client's 2 s timeout gets ONE retry inside + /// `sources::LOAD_BUDGET`, and that retry is cancelled by the budget + /// before its own timeout. The 2 s is `EXTENSION_REQUEST_TIMEOUT` in + /// `prices_clickhouse::mtls`, private there and pinned below. + #[tokio::test(start_paused = true)] + async fn a_hung_read_gets_one_retry_inside_the_load_budget() { + const CLIENT_TIMEOUT: Duration = Duration::from_secs(2); + let calls = Arc::new(AtomicUsize::new(0)); + let counter = calls.clone(); + let budget = crate::portal::sources::LOAD_BUDGET; + let outcome = tokio::time::timeout( + budget, + with_retry("test-read", transient, || { + counter.fetch_add(1, Ordering::SeqCst); + async { + tokio::time::sleep(CLIENT_TIMEOUT).await; + Err::<(), _>(Failure::Transient(0)) + } + }), + ) + .await; + assert!(outcome.is_err(), "the budget, not the retry, ended it"); + assert_eq!( + calls.load(Ordering::SeqCst), + 2, + "a hung read got more than one retry" + ); + // And a third attempt could not have started even with no wait at all. + assert!(CLIENT_TIMEOUT * 2 >= budget); + assert!( + include_str!("../../../prices-clickhouse/src/mtls.rs") + .contains("const EXTENSION_REQUEST_TIMEOUT: Duration = Duration::from_secs(2);"), + "mtls.rs's client timeout moved; redo this arithmetic and the module doc" + ); + } + + /// Review WR-01: only a failure the next attempt may not repeat. + #[test] + fn only_a_transient_fetch_message_is_retried() { + // Built exactly as `prices_clickhouse::mtls` builds them. + let unreachable = "extension at http://localhost:2773/systemsmanager/parameters/get \ + unreachable (verify Parameters and Secrets layer ARN is attached and layer is \ + initialised): error sending request"; + let status = |s: &str| format!("extension returned HTTP {s} (parameter=/prices/x)"); + + assert!(is_transient_fetch(unreachable)); + assert!(is_transient_fetch(&status("500 Internal Server Error"))); + assert!(is_transient_fetch(&status("503 Service Unavailable"))); + assert!(is_transient_fetch(&status("429 Too Many Requests"))); + assert!(is_transient_fetch(&format!( + "{}: ThrottlingException: Rate exceeded", + status("400 Bad Request") + ))); + + assert!(!is_transient_fetch(&status("400 Bad Request"))); + assert!(!is_transient_fetch(&status("403 Forbidden"))); + assert!(!is_transient_fetch(&status("404 Not Found"))); + assert!(!is_transient_fetch( + "response missing `Parameter.Value` field" + )); + assert!(!is_transient_fetch("response missing `SecretString` field")); + assert!(!is_transient_fetch("response body parse failed: EOF")); + assert!(!is_transient_fetch("reqwest client build failed: tls")); + assert!(!is_transient_fetch("something mtls has never said")); + } + + /// The shapes [`is_transient_fetch`] reads are the ones `mtls.rs` writes: + /// a reworded message there must fail here, not silently turn a transient + /// failure permanent (or the reverse). #[test] - fn three_attempts_with_two_rising_caps() { - assert_eq!(ATTEMPTS, 3); - assert!(BACKOFF_CAPS[0] < BACKOFF_CAPS[1]); + fn the_fetch_messages_this_classifies_are_the_ones_mtls_builds() { + let mtls = include_str!("../../../prices-clickhouse/src/mtls.rs"); + for shape in [ + "\"extension at {EXTENSION_URL} unreachable (verify", + "\"extension at {EXTENSION_PARAMETER_URL} unreachable (verify", + "\"extension returned HTTP {status} (secret={secret_name})\"", + "\"extension returned HTTP {status} (parameter={name})\"", + "\"reqwest client build failed: {e}\"", + "\"response body parse failed: {e}\"", + "\"response missing `SecretString` field\"", + "\"response missing `Parameter.Value` field\"", + ] { + assert!(mtls.contains(shape), "mtls.rs no longer builds {shape}"); + } } #[cfg(feature = "aws-mtls")] #[test] - fn only_a_fetch_failure_is_transient() { + fn only_a_transient_fetch_failure_is_transient() { use prices_clickhouse::mtls::MtlsError; - assert!(is_transient(&MtlsError::Fetch("unreachable".into()))); + assert!(is_transient(&MtlsError::Fetch( + "extension returned HTTP 503 Service Unavailable (parameter=x)".into() + ))); + assert!(!is_transient(&MtlsError::Fetch( + "extension returned HTTP 404 Not Found (parameter=x)".into() + ))); assert!(!is_transient(&MtlsError::MissingEnv("AWS_SESSION_TOKEN"))); assert!(!is_transient(&MtlsError::BundleDecode("x".into()))); } diff --git a/packages/prices-api/src/portal/mod.rs b/packages/prices-api/src/portal/mod.rs index bd7337ef..5d12a282 100644 --- a/packages/prices-api/src/portal/mod.rs +++ b/packages/prices-api/src/portal/mod.rs @@ -334,8 +334,10 @@ pub(crate) fn cors_layer(web_origin: Option<&str>) -> CorsLayer { /// Open means the flag is on AND the portal's sources are loaded. This is /// usually the first portal request an execution environment sees (the page /// asks it on every load), so it is what triggers the load. A failed load -/// answers `enabled: false` for this response only and the next call loads -/// again; the reason is in the log line and the alarm, not in the answer. +/// answers `enabled: false`, as does every call inside the short cooldown +/// after it (`sources::LOAD_FAILURE_COOLDOWN`, which starts no load); the +/// first call after that loads again. The reason is in the log line and the +/// alarm, not in the answer. /// The flag is checked FIRST, so a closed portal never loads anything. async fn config_handler(State(gate): State) -> Response { let enabled = gate.enabled && gate.sources.get().await.is_some(); @@ -344,8 +346,8 @@ async fn config_handler(State(gate): State) -> Response { rate_limit_per_second: gate.rate_limit, }) .into_response(); - // Never cached: the flag changes on deploy and a failed load changes on - // the next call, and a CDN or browser holding a stale `enabled: false` + // Never cached: the flag changes on deploy and a failed load changes + // within seconds, and a CDN or browser holding a stale `enabled: false` // would keep the portal dark for its viewers long after it opened — with // nothing on screen to suggest why. cache_control::attach(&mut resp, cache_control::NO_STORE); diff --git a/packages/prices-api/src/portal/sources.rs b/packages/prices-api/src/portal/sources.rs index 34983721..eda74f11 100644 --- a/packages/prices-api/src/portal/sources.rs +++ b/packages/prices-api/src/portal/sources.rs @@ -17,23 +17,33 @@ //! reads only the mTLS bundle, and the portal pays for its own sources on its //! own first request. //! -//! # Why a failure is not cached +//! # Why a failure is kept for at most a cooldown //! //! The lifetime closure was the defect. A failed load answers **that request** //! as unavailable (`/config` `enabled: false`, `503` on `/key`, `/usage` and -//! `/me`, a failure landing on sign-in) and the next portal request loads -//! again; a success is kept for the environment's life. Still closed, not -//! crashed: the load never panics, because a panic here would be a `502` on -//! the function that also serves `/v1`, over sources `/v1` does not use. +//! `/me`, a failure landing on sign-in); a success is kept for the +//! environment's life. Still closed, not crashed: the load never panics, +//! because a panic here would be a `502` on the function that also serves +//! `/v1`, over sources `/v1` does not use. //! -//! # Single flight, for success only +//! A failure is remembered for [`LOAD_FAILURE_COOLDOWN`] and no longer (review +//! WR-02). Inside it, a portal request answers as unavailable **without** +//! loading and without a second alarm-prefixed ERROR line; the first request +//! after it loads again. Without the cooldown, `/config` — anonymous, and +//! called on every page view — re-ran all five reads with their retries on +//! every request while SSM was throttling, prolonging the very throttle this +//! task escapes. Two seconds bounds that to one load per environment per +//! cooldown and still leaves nothing broken for longer than a reload. +//! +//! # Single flight //! //! [`tokio::sync::OnceCell::get_or_try_init`] runs one init at a time and -//! hands a success to every waiter. A failure goes to its own caller only, and -//! the next waiter starts another attempt — so N waiters on a failing load run -//! N loads in turn. [`LOAD_BUDGET`] wraps the whole wait, which bounds each of -//! them. Standard Lambda runs one request per environment at a time, so in -//! production this bites only `serve` and the tests. +//! hands a success to every waiter. A failure goes to its own caller only; +//! the waiters queued behind it then find the cooldown recorded and answer +//! unavailable without a load of their own — the failure is recorded inside +//! the init, before the next waiter can start one. [`LOAD_BUDGET`] wraps the +//! whole wait. Standard Lambda runs one request per environment at a time, +//! so in production queued waiters exist only in `serve` and the tests. //! //! The load is awaited inside the handler, never `tokio::spawn`ed: Lambda //! freezes spawned work after the response. And the loader must never call @@ -44,10 +54,11 @@ use std::future::Future; use std::pin::Pin; -use std::sync::Arc; +use std::sync::{Arc, Mutex}; use std::time::Duration; use tokio::sync::OnceCell; +use tokio::time::Instant; use crate::config::{AppConfig, PortalLoadError}; use crate::portal::auth::secret::OauthSecret; @@ -63,6 +74,14 @@ use crate::portal::keys::gateway::Gateway; /// `/usage` and `/key`, and in `auth::issue` for the callback. pub(crate) const LOAD_BUDGET: Duration = Duration::from_secs(4); +/// How long a failed load is remembered: inside it, [`PortalSources::get`] +/// answers unavailable without loading (see the module doc). +/// +/// Two seconds: long enough that a burst of page views — `/config` fires on +/// every one — costs one load per environment rather than one each, short +/// enough that a visitor who reloads after a transient fault finds it gone. +pub const LOAD_FAILURE_COOLDOWN: Duration = Duration::from_secs(2); + /// What a load produced. /// /// The production loader returns all three `Some`, or fails. `Default` — all @@ -100,6 +119,43 @@ pub type Loader = Arc LoadFuture + Send + Sync>; pub struct PortalSources { cell: Arc>, loader: Option, + /// When the last load failed, for [`LOAD_FAILURE_COOLDOWN`]. Shared by + /// every clone, like the cell. + last_failure: Arc>>, +} + +/// Why one ask came back without the sources. +enum Unavailable { + /// A load ran and failed. + Failed(PortalLoadError), + /// A load failed this long ago, inside the cooldown; none was started. + CoolingDown(Duration), +} + +/// Records a failure when dropped, unless disarmed by a success. +/// +/// A drop rather than a line after the `await`, so a load cancelled by +/// [`LOAD_BUDGET`] is recorded too — and recorded while the cell's init +/// permit is still held, before the next waiter can start a load of its own. +/// A load cancelled because its request went away counts as well: it starts +/// a cooldown without an ERROR line, which errs toward loading less. +struct FailureRecorder<'a> { + last_failure: &'a Mutex>, + armed: bool, +} + +impl Drop for FailureRecorder<'_> { + fn drop(&mut self) { + if self.armed { + *lock(self.last_failure) = Some(Instant::now()); + } + } +} + +/// The failure clock's lock. A poisoned one still holds a valid instant: the +/// only writer stores one value and cannot panic half-way. +fn lock(m: &Mutex>) -> std::sync::MutexGuard<'_, Option> { + m.lock().unwrap_or_else(std::sync::PoisonError::into_inner) } impl PortalSources { @@ -109,6 +165,7 @@ impl PortalSources { PortalSources { cell: Arc::new(OnceCell::new_with(Some(loaded))), loader: None, + last_failure: Arc::default(), } } @@ -128,23 +185,49 @@ impl PortalSources { PortalSources { cell: Arc::new(OnceCell::new()), loader: Some(loader), + last_failure: Arc::default(), } } /// The sources, loading them if this is the first ask (or every earlier - /// ask failed). `None` means this request answers as unavailable; the - /// failure is logged here, once per failed load, and the next call - /// retries. + /// ask failed). `None` means this request answers as unavailable. + /// + /// A failed load logs one alarm-prefixed ERROR here. An ask inside the + /// [`LOAD_FAILURE_COOLDOWN`] after it starts no load and logs one WARN + /// without the prefix, so the alarm counts loads, not requests. pub async fn get(&self) -> Option<&Loaded> { if let Some(loaded) = self.cell.get() { return Some(loaded); } // A ready cell is always initialised, so only a lazy one gets here. let loader = self.loader.as_ref()?; - let attempt = self.cell.get_or_try_init(|| loader()); + let attempt = self.cell.get_or_try_init(|| async { + // Checked inside the init, not before it: a request queued behind + // a failing load gets here after that failure was recorded. + if let Some(ago) = self.failed_within_cooldown() { + return Err(Unavailable::CoolingDown(ago)); + } + let mut recorder = FailureRecorder { + last_failure: &self.last_failure, + armed: true, + }; + let loaded = loader().await.map_err(Unavailable::Failed)?; + recorder.armed = false; + Ok(loaded) + }); let err = match tokio::time::timeout(LOAD_BUDGET, attempt).await { Ok(Ok(loaded)) => return Some(loaded), - Ok(Err(err)) => err, + Ok(Err(Unavailable::Failed(err))) => err, + Ok(Err(Unavailable::CoolingDown(ago))) => { + // Not the alarm's prefix: this request started no load. + tracing::warn!( + failed_ms_ago = ago.as_millis() as u64, + cooldown_ms = LOAD_FAILURE_COOLDOWN.as_millis() as u64, + "portal request answered as unavailable without a load: the last load \ + failed inside the cooldown" + ); + return None; + } Err(_) => PortalLoadError::TimedOut(LOAD_BUDGET), }; // The alarm's string: `prices-${env}-api-handler-portal-load-failed` @@ -152,11 +235,18 @@ impl PortalSources { // `tools/scripts/portal-load-failed-filter-guard.test.mjs`. tracing::error!( error = %err, - "portal sources failed to load; this request answers as unavailable and the next \ - portal request retries; /v1 is unaffected" + "portal sources failed to load; this request answers as unavailable and the first \ + portal request after a short cooldown retries; /v1 is unaffected" ); None } + + /// How long ago the last load failed, if that is inside the cooldown. + fn failed_within_cooldown(&self) -> Option { + let failed_at = (*lock(&self.last_failure))?; + let ago = failed_at.elapsed(); + (ago < LOAD_FAILURE_COOLDOWN).then_some(ago) + } } /// Which handle `portal::apply` builds for `config`. @@ -237,17 +327,18 @@ mod tests { } #[tokio::test(start_paused = true)] - async fn a_failure_is_not_cached_and_a_success_is() { + async fn a_failure_is_kept_only_for_the_cooldown_and_a_success_for_good() { let (loader, calls) = scripted(&[Step::Fail, Step::Succeed], Duration::ZERO); let sources = PortalSources::lazy(loader); assert!(sources.get().await.is_none()); + tokio::time::advance(LOAD_FAILURE_COOLDOWN).await; assert!(sources.get().await.is_some()); assert!(sources.get().await.is_some()); assert_eq!(calls.load(Ordering::SeqCst), 2); } #[tokio::test(start_paused = true)] - async fn a_load_past_the_budget_gives_up_and_the_next_ask_loads_again() { + async fn a_load_past_the_budget_gives_up_and_the_first_ask_after_the_cooldown_loads_again() { let (loader, calls) = scripted(&[Step::Hang, Step::Succeed], Duration::ZERO); let sources = PortalSources::lazy(loader); let started = tokio::time::Instant::now(); @@ -257,10 +348,58 @@ mod tests { waited >= LOAD_BUDGET && waited < LOAD_BUDGET + Duration::from_millis(50), "{waited:?}" ); + // A timed-out load is a failed one: it starts the cooldown too. + assert!(sources.get().await.is_none()); + assert_eq!(calls.load(Ordering::SeqCst), 1); + tokio::time::advance(LOAD_FAILURE_COOLDOWN).await; assert!(sources.get().await.is_some()); assert_eq!(calls.load(Ordering::SeqCst), 2); } + /// Review WR-02. Inside the cooldown an ask starts no load — the retry + /// storm against a throttled SSM is bounded to one load per cooldown — + /// and the first ask after it loads. That it adds no alarm-prefixed + /// ERROR line is asserted in `tests/portal_load_logs.rs`, a binary of its + /// own: `tracing`'s callsite cache makes a log capture beside parallel + /// tests unreliable (see that file). + #[tokio::test(start_paused = true)] + async fn inside_the_cooldown_an_ask_does_not_load() { + let (loader, calls) = scripted(&[Step::Fail, Step::Succeed], Duration::ZERO); + let sources = PortalSources::lazy(loader); + + assert!(sources.get().await.is_none()); + for _ in 0..5 { + tokio::time::advance(LOAD_FAILURE_COOLDOWN / 10).await; + assert!(sources.get().await.is_none()); + } + assert_eq!( + calls.load(Ordering::SeqCst), + 1, + "a load ran inside the cooldown" + ); + + tokio::time::advance(LOAD_FAILURE_COOLDOWN).await; + assert!(sources.get().await.is_some()); + assert_eq!(calls.load(Ordering::SeqCst), 2); + } + + /// Requests queued behind a failing load do not each run one of their + /// own: the failure is recorded before the next waiter's init starts. + #[tokio::test(start_paused = true)] + async fn waiters_behind_a_failing_load_do_not_load_again() { + let (loader, calls) = scripted(&[Step::Fail], Duration::from_millis(50)); + let sources = PortalSources::lazy(loader); + let mut set = tokio::task::JoinSet::new(); + for _ in 0..8 { + let sources = sources.clone(); + set.spawn(async move { sources.get().await.is_some() }); + } + while let Some(loaded) = set.join_next().await { + assert!(!loaded.unwrap()); + } + assert_eq!(calls.load(Ordering::SeqCst), 1); + } + #[tokio::test] async fn ready_sources_answer_without_a_loader() { let sources = PortalSources::ready(Loaded::default()); @@ -268,6 +407,36 @@ mod tests { assert!(sources.get().await.is_some()); } + fn config(portal_enabled: bool) -> AppConfig { + AppConfig { + ch_enabled: false, + base_url: None, + api_keys: vec![], + portal_enabled, + portal_oauth: None, + portal_endpoints: Default::default(), + portal_keys: None, + portal_eligibility: None, + portal_rate_limit: None, + portal_web_origin: None, + } + } + + /// Review WR-05. What keeps a closed Lambda from ever building a + /// control-plane client (`Gateway::from_ambient_config`) is that + /// `sources_for` gives it no loader at all — the gate's `404` is the + /// second line, not the first. Only an open portal with nothing supplied + /// gets the environment loader. + #[tokio::test] + async fn a_closed_portal_never_gets_a_loader() { + let closed = sources_for(&config(false)); + assert!(closed.loader.is_none(), "a closed portal holds a loader"); + let loaded = closed.get().await.expect("a closed portal's cell is ready"); + assert!(loaded.oauth.is_none() && loaded.gateway.is_none() && loaded.settings.is_none()); + + assert!(sources_for(&config(true)).loader.is_some()); + } + /// A load in front of the slowest routes must still leave the answer /// inside the invocation. The 15 is `apiHandler.timeoutSeconds` in /// `infra/envs/production.json`; the callback's share is pinned in diff --git a/packages/prices-api/tests/portal_lazy_load.rs b/packages/prices-api/tests/portal_lazy_load.rs index 025dc5cd..e5a0fbd3 100644 --- a/packages/prices-api/tests/portal_lazy_load.rs +++ b/packages/prices-api/tests/portal_lazy_load.rs @@ -6,15 +6,21 @@ //! //! - building the router and serving `/health` and `/v1` read nothing; //! - `/config` loads, answers `enabled` by the outcome, and a success is kept; -//! - a failure is not kept: the next request loads again; -//! - a portal route answers `503` on a failed load and works on the next; +//! - a failure is kept only for `LOAD_FAILURE_COOLDOWN`: inside it a request +//! answers unavailable without loading, and the first after it loads again; +//! - a portal route answers `503` on a failed load and works after the cooldown; //! - with `PORTAL_ENABLED=false` the portal is a `404` and nothing loads; //! - concurrent first requests share one successful load; -//! - a sign-in callback on a failed or slow load lands on a retryable failure -//! (the slow case with a real login is in `tests/portal_auth.rs`); +//! - login on a failed load lands on a retryable failure; +//! - a callback on a failed load lands where its flow renders a failure (the +//! cases with a real login, and the slow load, are in `tests/portal_auth.rs`); //! - `main.rs` performs no portal load. //! -//! The retry around each read is unit-tested in `portal::extension`. +//! The tests that load twice run on tokio's paused clock, so the cooldown is +//! skipped with `advance` rather than slept through. +//! +//! The retry around each read is unit-tested in `portal::extension`; the log +//! lines a failed load and its cooldown write are in `tests/portal_load_logs.rs`. use std::collections::VecDeque; use std::sync::atomic::{AtomicUsize, Ordering}; @@ -27,7 +33,7 @@ use axum::http::{HeaderMap, Request, StatusCode, header}; use prices_api::config::PortalLoadError; use prices_api::portal::auth::secret::{OauthSecret, SecretError}; use prices_api::portal::keys::gateway::Gateway; -use prices_api::portal::sources::{Loaded, Loader, PortalSources}; +use prices_api::portal::sources::{LOAD_FAILURE_COOLDOWN, Loaded, Loader, PortalSources}; use prices_api::{AppConfig, AppState, app_with_portal}; use serde_json::{Value, json}; use tower::ServiceExt; @@ -225,11 +231,11 @@ fn main_rs_performs_no_portal_load() { } // --------------------------------------------------------------------------- -// /config answers by the load's outcome, and a failure is not kept +// /config answers by the load's outcome; a failure is kept for the cooldown // --------------------------------------------------------------------------- -#[tokio::test] -async fn config_says_closed_on_a_failed_load_and_open_on_the_next() { +#[tokio::test(start_paused = true)] +async fn config_says_closed_on_a_failed_load_and_open_after_the_cooldown() { let (sources, calls) = scripted( vec![Step::Fail, Step::Succeed(Loaded::default())], Duration::ZERO, @@ -241,17 +247,33 @@ async fn config_says_closed_on_a_failed_load_and_open_on_the_next() { assert_eq!(failed.json()["enabled"], json!(false)); assert!(failed.no_store(), "a closed answer must never be cached"); + // Inside the cooldown (review WR-02): the same answer, and no load — an + // anonymous page view must not re-run five retried reads against a + // throttled SSM. + let cooling = get(&router, "/api/config").await; + assert_eq!(cooling.json()["enabled"], json!(false)); + assert_eq!( + calls.load(Ordering::SeqCst), + 1, + "a load ran inside the cooldown" + ); + + tokio::time::advance(LOAD_FAILURE_COOLDOWN).await; let recovered = get(&router, "/api/config").await; assert_eq!(recovered.json()["enabled"], json!(true)); - assert_eq!(calls.load(Ordering::SeqCst), 2, "the failure was kept"); + assert_eq!( + calls.load(Ordering::SeqCst), + 2, + "the failure outlived the cooldown" + ); } // --------------------------------------------------------------------------- // Portal routes: 503 on a failed load, working on the next // --------------------------------------------------------------------------- -#[tokio::test] -async fn usage_is_503_on_a_failed_load_and_answers_on_the_next() { +#[tokio::test(start_paused = true)] +async fn usage_is_503_on_a_failed_load_and_answers_after_the_cooldown() { let (sources, _) = scripted( vec![Step::Fail, Step::Succeed(loaded_with_keys())], Duration::ZERO, @@ -264,12 +286,13 @@ async fn usage_is_503_on_a_failed_load_and_answers_on_the_next() { assert!(failed.no_store()); // Past the unprovisioned branch: no session cookie is now the answer. + tokio::time::advance(LOAD_FAILURE_COOLDOWN).await; let next = get(&router, "/api/usage").await; assert_eq!(next.status, StatusCode::UNAUTHORIZED); } -#[tokio::test] -async fn key_is_503_on_a_failed_load_and_answers_on_the_next() { +#[tokio::test(start_paused = true)] +async fn key_is_503_on_a_failed_load_and_answers_after_the_cooldown() { let (sources, _) = scripted( vec![Step::Fail, Step::Succeed(loaded_with_keys())], Duration::ZERO, @@ -281,14 +304,15 @@ async fn key_is_503_on_a_failed_load_and_answers_on_the_next() { assert_eq!(failed.json()["code"], json!("keys_unconfigured")); assert!(failed.no_store()); + tokio::time::advance(LOAD_FAILURE_COOLDOWN).await; let next = get(&router, "/api/key").await; assert_eq!(next.status, StatusCode::UNAUTHORIZED); } /// `/me` does not say "signed out" when it cannot check: that would be a lie /// to a visitor who is signed in. -#[tokio::test] -async fn me_is_503_on_a_failed_load_and_answers_on_the_next() { +#[tokio::test(start_paused = true)] +async fn me_is_503_on_a_failed_load_and_answers_after_the_cooldown() { let (sources, _) = scripted( vec![Step::Fail, Step::Succeed(loaded_with_keys())], Duration::ZERO, @@ -300,6 +324,7 @@ async fn me_is_503_on_a_failed_load_and_answers_on_the_next() { assert_eq!(failed.json()["code"], json!("session_unavailable")); assert!(failed.no_store()); + tokio::time::advance(LOAD_FAILURE_COOLDOWN).await; let next = get(&router, "/api/auth/me").await; assert_eq!(next.status, StatusCode::OK); assert_eq!(next.json()["authenticated"], json!(false)); @@ -351,11 +376,13 @@ async fn concurrent_first_requests_share_one_load() { // The sign-in callback // --------------------------------------------------------------------------- -/// The action is not known before `state` is verified with the secret that -/// failed to load, so the landing is sign-in's failure, and the pending -/// cookie is left alone as on every refusal before verification. +/// A callback that claims no action — no pending cookie, a `state` that is +/// not one of ours — lands on sign-in's failure, and the pending cookie is +/// left alone as on every refusal before verification. The issue and +/// sign-in round-trips on a failed load, with real tokens, are in +/// `tests/portal_auth.rs`. #[tokio::test] -async fn a_callback_on_a_failed_load_lands_on_signin_failed() { +async fn a_callback_on_a_failed_load_that_claims_no_action_lands_on_signin_failed() { let (sources, _) = scripted(vec![Step::Fail], Duration::ZERO); let router = router(true, sources); @@ -364,3 +391,28 @@ async fn a_callback_on_a_failed_load_lands_on_signin_failed() { assert_eq!(reply.location(), "/api/?signin=failed"); assert!(reply.headers.get(header::SET_COOKIE).is_none()); } + +// --------------------------------------------------------------------------- +// Login on a failed load (review WR-04) +// --------------------------------------------------------------------------- + +/// A failed load is transient, so login lands it on a failure the visitor +/// can retry — sign-in's card for a sign-in, the dashboard for an issue +/// press — never on `?signin=not_open`'s "not yet available", which states a +/// permanent condition. That it earns exactly ONE ERROR line (the alarm's, +/// with no "deployment that cannot complete one" line beside it) is asserted +/// in `tests/portal_load_logs.rs`, a binary of its own. +#[tokio::test] +async fn login_on_a_failed_load_lands_on_a_retryable_failure() { + for (uri, landing) in [ + ("/api/auth/login", "/api/?signin=failed"), + ("/api/auth/login?action=issue", "/api/?issue=failed"), + ] { + let (sources, _) = scripted(vec![Step::Fail], Duration::ZERO); + let router = router(true, sources); + let reply = get(&router, uri).await; + assert_eq!(reply.status, StatusCode::SEE_OTHER, "{uri}"); + assert_eq!(reply.location(), landing, "{uri}"); + assert!(reply.headers.get(header::SET_COOKIE).is_none(), "{uri}"); + } +} diff --git a/packages/prices-api/tests/portal_load_logs.rs b/packages/prices-api/tests/portal_load_logs.rs new file mode 100644 index 00000000..d2846f9f --- /dev/null +++ b/packages/prices-api/tests/portal_load_logs.rs @@ -0,0 +1,184 @@ +//! **A failed portal load writes one alarm line, and its cooldown writes none** +//! — one test, one test binary (task 0311, review WR-02 / WR-04). +//! +//! # Why a binary of its own +//! +//! The same reason as `portal_keys_logs.rs`: `tracing` caches each callsite's +//! `Interest` globally, set by whichever thread reaches it first, while +//! `tracing::subscriber::set_default` is thread-local. Beside parallel tests +//! that also fail a load, the `portal sources failed to load` callsite can be +//! cached as `never` before this test's subscriber exists, and the capture +//! comes back empty — measured: it did, in `portal_lazy_load.rs`, about one +//! run in seven. Alone in a process there is no other thread to lose to. +//! +//! # What it proves +//! +//! The alarm `prices-${env}-api-handler-portal-load-failed` counts log lines +//! by their message prefix, so the prefix has to mean "a load failed": +//! +//! - a failed load logs it exactly once, and login adds no ERROR of its own +//! (the old "deployment that cannot complete one" line, review WR-04); +//! - an ask inside `LOAD_FAILURE_COOLDOWN` starts no load and logs a WARN +//! WITHOUT the prefix, so throttling cannot multiply alarm datapoints by +//! page views (review WR-02). + +use std::sync::atomic::{AtomicUsize, Ordering}; +use std::sync::{Arc, Mutex}; + +use axum::body::Body; +use axum::http::{Request, StatusCode}; +use prices_api::config::PortalLoadError; +use prices_api::portal::auth::secret::SecretError; +use prices_api::portal::sources::{LOAD_FAILURE_COOLDOWN, LoadFuture, PortalSources}; +use prices_api::{AppConfig, AppState, app_with_portal}; +use tower::ServiceExt; + +/// The alarm's prefix, as the metric filter and the guard test hold it. +const ALARM_PREFIX: &str = "portal sources failed to load"; + +/// A `MakeWriter` that keeps every byte the subscriber emits. +#[derive(Clone, Default)] +struct CapturedLogs(Arc>>); + +impl std::io::Write for CapturedLogs { + fn write(&mut self, buf: &[u8]) -> std::io::Result { + self.0.lock().unwrap().extend_from_slice(buf); + Ok(buf.len()) + } + fn flush(&mut self) -> std::io::Result<()> { + Ok(()) + } +} + +impl<'a> tracing_subscriber::fmt::MakeWriter<'a> for CapturedLogs { + type Writer = Self; + fn make_writer(&'a self) -> Self::Writer { + self.clone() + } +} + +impl CapturedLogs { + /// Takes the lines written so far, leaving the buffer empty. + fn take(&self) -> Vec { + let bytes = std::mem::take(&mut *self.0.lock().unwrap()); + String::from_utf8(bytes) + .unwrap() + .lines() + .map(str::to_string) + .collect() + } +} + +/// Sources whose every load fails, counting the loads. +fn failing() -> (PortalSources, Arc) { + let calls = Arc::new(AtomicUsize::new(0)); + let counter = calls.clone(); + let sources = PortalSources::lazy(Arc::new(move || -> LoadFuture { + counter.fetch_add(1, Ordering::SeqCst); + Box::pin(async { Err(PortalLoadError::Oauth(SecretError::NoSource)) }) + })); + (sources, calls) +} + +fn config() -> AppConfig { + AppConfig { + ch_enabled: false, + base_url: None, + api_keys: vec![], + portal_enabled: true, + portal_oauth: None, + portal_endpoints: Default::default(), + portal_keys: None, + portal_eligibility: None, + portal_rate_limit: None, + portal_web_origin: None, + } +} + +async fn get_status(router: &axum::Router, uri: &str) -> StatusCode { + router + .clone() + .oneshot(Request::builder().uri(uri).body(Body::empty()).unwrap()) + .await + .unwrap() + .status() +} + +#[tokio::test(start_paused = true)] +async fn a_failed_load_alarms_once_and_its_cooldown_never() { + let logs = CapturedLogs::default(); + let subscriber = tracing_subscriber::fmt() + .with_writer(logs.clone()) + .with_ansi(false) + .finish(); + // A guard rather than `with_default`'s closure: `#[tokio::test]` drives + // this future on this thread, so the subscriber covers every `await`. + let _guard = tracing::subscriber::set_default(subscriber); + tracing::error!("probe: the captured writer sees ERROR"); + assert_eq!(logs.take().len(), 1, "the capture is not live"); + + // Login, both arms: one ERROR each, and it is the alarm's. + for uri in ["/api/auth/login", "/api/auth/login?action=issue"] { + let (sources, _) = failing(); + let router = app_with_portal(&config(), AppState::without_ch(), sources); + assert_eq!(get_status(&router, uri).await, StatusCode::SEE_OTHER); + let errors: Vec<_> = logs + .take() + .into_iter() + .filter(|line| line.contains("ERROR")) + .collect(); + assert_eq!(errors.len(), 1, "{uri}: {errors:#?}"); + assert!(errors[0].contains(ALARM_PREFIX), "{uri}: {errors:#?}"); + } + + // A burst of page views on one environment while loads fail. + let (sources, calls) = failing(); + let router = app_with_portal(&config(), AppState::without_ch(), sources); + assert_eq!(get_status(&router, "/api/config").await, StatusCode::OK); + let first = logs.take(); + assert_eq!( + first.iter().filter(|l| l.contains(ALARM_PREFIX)).count(), + 1, + "{first:#?}" + ); + + for _ in 0..5 { + tokio::time::advance(LOAD_FAILURE_COOLDOWN / 10).await; + assert_eq!(get_status(&router, "/api/config").await, StatusCode::OK); + assert_eq!( + get_status(&router, "/api/usage").await, + StatusCode::SERVICE_UNAVAILABLE + ); + } + let cooling = logs.take(); + assert_eq!( + calls.load(Ordering::SeqCst), + 1, + "a load ran inside the cooldown" + ); + assert!( + cooling + .iter() + .all(|l| !l.contains(ALARM_PREFIX) && !l.contains("ERROR")), + "the cooldown alarmed: {cooling:#?}" + ); + assert_eq!( + cooling + .iter() + .filter(|l| l.contains("WARN") && l.contains("without a load")) + .count(), + 10, + "{cooling:#?}" + ); + + // After it, a load again — and its failure alarms again, once. + tokio::time::advance(LOAD_FAILURE_COOLDOWN).await; + assert_eq!(get_status(&router, "/api/config").await, StatusCode::OK); + assert_eq!(calls.load(Ordering::SeqCst), 2); + let after = logs.take(); + assert_eq!( + after.iter().filter(|l| l.contains(ALARM_PREFIX)).count(), + 1, + "{after:#?}" + ); +} diff --git a/tools/scripts/portal-load-failed-filter-guard.test.mjs b/tools/scripts/portal-load-failed-filter-guard.test.mjs index f625197b..c0583165 100644 --- a/tools/scripts/portal-load-failed-filter-guard.test.mjs +++ b/tools/scripts/portal-load-failed-filter-guard.test.mjs @@ -8,7 +8,8 @@ // alarm goes quiet for good with every check green. Asserted here: // // - the filter reads `$.fields.message` and matches by the exact prefix; -// - `sources.rs` has a `tracing::error!` whose message starts with it; +// - `sources.rs` has a `tracing::error!` whose message (its last literal, +// not a field value) starts with it; // - the subscriber in `main.rs` is `fmt().json()` and not flattened; // - the alarm takes its metric from that filter, under its own name; // - the alarm's description fits CloudWatch's 1024-character limit. @@ -78,10 +79,25 @@ const alarmDescription = (block) => { return found ? (found[1] ?? found[2] ?? found[3]) : undefined; }; -// Does some `tracing::error!(…)` carry a message literal starting with `prefix`? +// The message of one `tracing::error!(…)` body: its LAST string literal, and +// only when nothing but a trailing comma follows it. Fields come first in +// `tracing`'s macro syntax, so a prefix sitting in a field value +// (`reason = "portal sources failed to load"`) is not the message, and the +// filter on `$.fields.message` would never see it. +const messageLiteral = (body) => { + // `\\[\s\S]`, not `\\.`: a Rust line continuation is a backslash and a + // newline, which `.` does not match. + const literals = [...body.matchAll(/"((?:[^"\\]|\\[\s\S])*)"/g)]; + const last = literals.at(-1); + if (!last) return undefined; + const after = body.slice(last.index + last[0].length); + return /^\s*,?\s*$/.test(after) ? last[1] : undefined; +}; + +// Does some `tracing::error!(…)` carry a MESSAGE starting with `prefix`? const logsErrorStartingWith = (source, prefix) => - [...source.matchAll(/tracing::error!\(([\s\S]*?)\);/g)].some(([, body]) => - body.includes(`"${prefix}`), + [...source.matchAll(/tracing::error!\(([\s\S]*?)\);/g)].some( + ([, body]) => messageLiteral(body)?.startsWith(prefix) ?? false, ); const subscriberPutsMessageUnderFields = (source) => { @@ -154,6 +170,18 @@ test('a reworded log line is refused', () => { assert.equal(logsErrorStartingWith(reworded, PREFIX), false); }); +test('the prefix in a field value, beside a reworded message, is refused', () => { + // The message is reworded and the old words survive as a field: the + // metric filter reads `$.fields.message`, so this alarms on nothing. + const moved = sourcesSource.replace( + `error = %err,\n "${PREFIX}`, + `error = %err,\n reason = "${PREFIX}",\n "the portal could not load`, + ); + + assert.notEqual(moved, sourcesSource); + assert.equal(logsErrorStartingWith(moved, PREFIX), false); +}); + test('a flattened subscriber is refused', () => { const flattened = mainSource.replace( '.json()', From a3ae084cf37da445acba355e58bbcb2db4d094f1 Mon Sep 17 00:00:00 2001 From: Adam <65679285+adamkoot@users.noreply.github.com> Date: Fri, 25 Sep 2026 14:18:52 +0200 Subject: [PATCH 18/22] fix(lore-0311): wait out the backend's slowest portal answer on usage - USAGE_TIMEOUT_MS 15 s -> 20 s: /usage can load the portal's sources first (4 s) before its 10 s deadline, so its 503 can arrive at 14 s - KEY_TIMEOUT_MS keeps 20 s; its comment now cites the same arithmetic --- web/portal/src/api/portal.ts | 32 ++++++++++++++++++++------------ 1 file changed, 20 insertions(+), 12 deletions(-) diff --git a/web/portal/src/api/portal.ts b/web/portal/src/api/portal.ts index 9a0d706d..612237f9 100644 --- a/web/portal/src/api/portal.ts +++ b/web/portal/src/api/portal.ts @@ -169,25 +169,33 @@ const PROBE_TIMEOUT_MS = 10_000; /** * How long the page waits on the key reveal. * - * Longer than {@link PROBE_TIMEOUT_MS}, because this call is not a probe: a cold - * Lambda resolves credentials, reads an SSM parameter and then walks a - * paginated listing plus a read. The api-handler's own Lambda timeout is 15s - * and API Gateway cuts everything off at 29s, so 20s is inside the window - * where an answer — including a `502` — is still possible. + * Longer than {@link PROBE_TIMEOUT_MS}, because this call is not a probe: it + * can be the first portal request in an execution environment, so it may load + * the portal's sources first (up to 4s, `LOAD_BUDGET` in + * `portal/sources.rs`), and then it walks a paginated listing plus a read + * under the reconciliation's 10s (`RECONCILE_DEADLINE`) — 14s at worst. The + * api-handler's own Lambda timeout is 15s and API Gateway cuts everything off + * at 29s, so 20s is past any answer the backend can still give — including a + * `502` from a killed invocation — and inside the gateway's cap. */ const KEY_TIMEOUT_MS = 20_000; /** * How long the page waits on the usage read. * - * Between the other two, because the call is: the backend's own wall-clock - * deadline on the lookup is 10s (`USAGE_DEADLINE` in `portal/usage/mod.rs`), - * after which it answers a `503` that names the condition. Waiting only - * {@link PROBE_TIMEOUT_MS} would tie with that deadline and the page would - * report its own timeout instead of the backend's more useful answer; 15s - * leaves the answer time to arrive while staying inside the gateway's 29s cap. + * The same as {@link KEY_TIMEOUT_MS}, for the same arithmetic. The backend's + * own wall-clock deadline on the lookup is 10s (`USAGE_DEADLINE` in + * `portal/usage/mod.rs`), after which it answers a `503` that names the + * condition — but `/usage` can be the first portal request in an execution + * environment, and then the sources load first (up to 4s, `LOAD_BUDGET`), so + * that `503` can arrive at 14s (task 0311; pinned by + * `the_load_budget_fits_in_front_of_every_route`). It was 15s, which left the + * page one second over a 14s answer before network and gateway latency — a tie + * in which the page reports its own timeout instead of the backend's more + * useful answer. 20s clears the 15s Lambda timeout, past which nothing can + * answer, and stays inside the gateway's 29s cap. */ -const USAGE_TIMEOUT_MS = 15_000; +const USAGE_TIMEOUT_MS = 20_000; /** Whether a rejection from `fetch` is this timeout firing. See the call site. */ const isTimeout = (error: unknown): boolean => From 5945b7a153940a24d085d9d9e57e285e5dd7cc02 Mon Sep 17 00:00:00 2001 From: Adam <65679285+adamkoot@users.noreply.github.com> Date: Fri, 25 Sep 2026 15:25:08 +0200 Subject: [PATCH 19/22] docs(lore-0311): record the five-plan herd and capacity test Measure why the portal closed on 2026-09-24 while /v1 kept serving, how much the shared ClickHouse holds per endpoint mix, and the cold-start curve against SSM. Record the lazy portal load deployed the same day: a replay of the 09-24 herd (69 cold starts) now makes zero SSM reads and closes no portal. --- .../R-five-plan-herd-and-capacity-test.md | 188 ++++++++++++++++++ 1 file changed, 188 insertions(+) create mode 100644 lore/1-tasks/active/0311_FEATURE_paid-plans-and-dashboard-shows-the-keys-own-plan/notes/R-five-plan-herd-and-capacity-test.md diff --git a/lore/1-tasks/active/0311_FEATURE_paid-plans-and-dashboard-shows-the-keys-own-plan/notes/R-five-plan-herd-and-capacity-test.md b/lore/1-tasks/active/0311_FEATURE_paid-plans-and-dashboard-shows-the-keys-own-plan/notes/R-five-plan-herd-and-capacity-test.md new file mode 100644 index 00000000..1863ad0e --- /dev/null +++ b/lore/1-tasks/active/0311_FEATURE_paid-plans-and-dashboard-shows-the-keys-own-plan/notes/R-five-plan-herd-and-capacity-test.md @@ -0,0 +1,188 @@ +# R: five-plan herd and capacity test (production, 2026-09-25) + +Follow-up to the 2026-09-24 portal-closed alarm (a 60 s five-plan limit check that opened +69 cold starts at once). Run on production 2026-09-25 09:32–10:30 UTC with Adam's go-ahead, +in parallel with the 0286 phase-3 backfill (Oskar's machine). + +Tags: **[M]** measured, **[E]** derived, **[H]** hypothesis. + +- **Artefacts:** branch `test/0311-five-plan-loadtest` (never merged), which holds + `packages/prices-api/loadtest/five-plan/` and `docs/loadtest-results/2026-09-25-0311-*`. +- **Keys:** 10 temporary keys, two per plan, created and deleted; deletion was verified. +- **Generator:** k6 v2.2.0 on the laptop. +- **Observer:** samples ClickHouse every 5 s and AWS every 60 s, and aborts automatically. + +## Answers + +**Why the portal closed and `/v1` kept working.** +- **Closure by design [M].** Each cold start makes five portal reads through the Parameters and + Secrets extension (`main.rs:56`, `config.rs:324-329`): one secret plus four SSM parameters. + Any failed read closes the portal in that execution environment for its lifetime + (`config.rs:367-393`). +- **`/v1` is not gated by the portal [M].** `gate_portal` 404s only portal paths + (`portal/mod.rs:333-338`). +- **What makes a read fail [M].** SSM throttles under a burst of cold starts; the extension + retries 3× with backoff, and our 2 s client timeout (`prices-clickhouse/src/mtls.rs:110`) + expires first. + - A closure logged as "extension unreachable" is that timeout. The extension logs the + `ThrottlingException` seconds later (10:15:22 closure → 10:15:27 throttle line). +- **The portal still sits on `/v1`'s path [M].** The five reads run sequentially before the + mTLS read, so they delay every cold `/v1` request. Init max was 1.7–2.3 s when SSM + throttled, against ~420 ms when it did not. + +**How much ClickHouse holds, and at which mix.** +- **Realistic mix [M].** The mix was 60 % `/price`, 10 % batch, 10 % detail, 10 % ohlcv 1h/7d, + 5 % ohlcv 1d/1y and 5 % list, drawn from a 5,480-asset pool. It costs ~70 ms of host CPU per + request and scales linearly: 44 rps → 3.1 cores, 88 rps → 6.1 cores. p99 stayed ≤ 176 ms and + there was no knee. +- **ohlcv carries the cost [M].** ohlcv is 15 % of the requests but 63 % of the CPU (1d grain + 517 ms, 1h 206 ms CPU per query). +- **Budget [E].** With a typical 9-core API budget (12 cores minus ~3 of background), about + **130 rps of this mix** can be sustained, i.e. ~3 full five-plan sets. Per plan alone at full + rate: ~5 Pro, ~13 Lite, ~26 Analyst, ~43 Basic or ~130 free keys. +- **List is bounded by the quota, not CPU [M/E].** A cache-busting list miss costs ~385 ms CPU + and ~0.93 M read rows, i.e. ~0.43 cores per req/s, with no knee up to 20 req/s. The + `prices_read` quota (50e9 rows/h, API-wide) caps it at **~15 req/s sustained**. One Pro key + paging `/v1/assets` exhausts the hour in ~36 min, after which **every** API query fails until + the hour turns. +- **The backfill saturates the box on its own [M].** At 10:17:26–10:20 the 0286 backfill held + the host at 19–24 cores with no API load: 842 `price_ohlcv_1m` inserts took 897 CPU-s, + `ingest_cursor FINAL` selects 178 CPU-s, and the 15m MV cascade 121 CPU-s. During such steps + the API has no headroom. + +## Results per phase + +| Phase | What | Result | +|---|---|---| +| F0 | background, 09:32–09:38 | host ~0.3 cores (RMV spikes at :00–:07 of even minutes); writer insert p95 ~65 ms; `assets` 7 parts, 2 copies [M] | +| F1 | `/backfill/status` at 1.5× each plan's rate, 120 s | 2xx 124 / 375 / 623 / 1,250 / 3,129 vs expected rate·t + burst 125 / 375 / 625 / 1,250 / 3,125 (±0.3 %); the rest 429; 0 × 403 / 5xx [M] | +| F2 | N simultaneous uncached `/price` requests on fresh environments (env bump) | see the curve below | +| F3 | `/v1/assets` at 1.5× plan rate, 180 s (replay of 09-24) | limits again within ±0.5 %; handler concurrency peaked at 25, not 69; CH ≤ 9 cores [M]. Herd ≈ arrival rate × miss latency: 66 rps × ~0.4 s ≈ 25 [E]. 09-24 misses took ~2.4 s (cold start + 4-copy `assets`), hence 69 | +| F4 | cache-busting list, 5 → 10 → 15 → 20 req/s | CPU/query p50 ~385 ms, wall p95 60–80 ms flat, host +2.5 / +4.5 / +6.5 / +8.8 cores; no knee; 0 × 5xx [M] | +| F5 | realistic mix, 5 keys (44 rps) then 10 keys (88 rps) | host avg 3.6 / 6.6 cores (max 8.2), in-flight ≤ 11, control `/price` p99 158 / 176 ms, 0 × 5xx / 429 [M] | +| F6 | F2 at N=200 with SSM high-throughput on (then reset) | 181 and 195 cold starts, **0 closures, 0 throttles**, Init max 423–429 ms [M] | + +### F2: simultaneous cold starts → closed portals [M] + +| Cold starts at once | Portals closed | SSM throttles | Init max | +|---|---|---|---| +| 10, 20, 28, 28, 54, 70, 96 | 0 | 0 (CloudTrail: 384 GetParameter/s at N=100 accepted) | ≤ 600 ms | +| 147 | 1 (0.7 %) | 1 | 2,326 ms | +| 200 | 3 (1.5 %) | 3 | 1,730 ms | +| 200 + SSM high-throughput (×2) | 0 | 0 | 429 ms | +| *2026-09-24: 68* | *7 (10 %)* | *5 + 2 "unreachable"* | *2,405 ms* | + +- **The threshold is not stable [M].** Today SSM absorbed 384 calls/s. On 09-24 it threw + throttles at 18 calls in the second after a 252-call second. "~10 cold starts close the + portal" (the pre-test estimate) is refuted. +- **Why [H].** SSM's standard-throughput limit behaves like a shared, stateful bucket that we + cannot observe. +- **Client artefact [M].** Opening 50+ TCP connections at once from the laptop stalls (connect + p50 ~1 s). N=50 and the first N=70 therefore produced only 28 cold starts. Pre-connecting + each VU on `/health` (a MOCK integration) fixed it. +- **09-24 was not a load problem for ClickHouse [M].** Its 69 list queries were 4× more + expensive because `assets` held 4 unmerged copies under `FINAL`. `prices_writer` reseeds all + 209.9 k rows every 75–105 min and merges only every 4–6 h; this is not caused by 0216. + +## What breaks first (in order of how easily it happens) + +1. **The portal, on a cold-start burst.** + - A plan burst can do it on its own: Pro's burst of 125 plus a slow miss. + - It is stochastic: 0–10 % of the environments in a burst. + - Closures persist until the environments are recycled. +2. **The `prices_read` quota, on list- or ohlcv-heavy cache misses.** + - Takes the whole API down until the hour turns. + - List-only traffic reaches it at ~15 req/s sustained. +3. **ClickHouse CPU.** + - ~130 rps of the realistic mix fit in the typical budget. + - During heavy backfill steps there is ~0 headroom. + - At the 2026-09-18 collapse (~800 rps `/price`), the per-query thread fan-out is what + amplifies the load. +4. **Lambda concurrency: not reached.** + - Peak 200, out of an account pool of 1,000 shared with the explorer. + - No throttles. + +The gateway method throttles (10000/400, the API Gateway defaults) never bind. + +## Mitigations, ranked by effect per cost + +1. **Lazy portal load plus retry with jitter** (Stanisław's note of 2026-09-25, and ours). + - A `/v1` cold start then does zero SSM reads. + - A failed read fails one request instead of closing the environment. + - Removes failure 1 entirely and trims ~100–2,000 ms from every cold `/v1`. + - A code change only. +2. **SSM high-throughput.** + - Measured: 3 → 0 closures at N=200 (small sample) and no retry tail on Init. + - One account setting, $0.05 per 10 k calls. + - The handler makes ~100 calls a day organically, so the cost is negligible. + - It still closes the portal if a throttle ever happens, so it is a stopgap next to (1). + - Reset to default after the test. +3. **Remove the `assets` sawtooth.** + - Reseed only the changed rows, or merge after the reseed. + - List and detail become 3–6× cheaper and the quota ceiling rises proportionally. +4. **An alarm on `prices_read` quota usage, Lambda throttles and concurrency.** + - None exists today. +5. **Cheaper ohlcv.** + - `usd_rate FINAL` (134 parts) and the 1h/1d part counts; the 0286 backfill currently + inflates them. + - ohlcv is 63 % of the mix's CPU. +6. **`prices_reader` `max_threads` (2–4) and `max_concurrent_queries_for_user`.** + - Damps the 09-18 fan-out collapse. + - Needs the explorer team. +7. **Reserved concurrency (50–100) for the handler.** + - Protects the explorer's share of the pool. + - Not needed at the measured peaks. + +## After the fix (deployed 2026-09-25 12:47 UTC, feat/0311 @ a3ae084c) + +Mitigations 1 and 2 are live: SSM high-throughput has been on since 10:41 UTC, and the +lazy portal load with retry shipped in Compute + Observability, with the alarm renamed +`api-handler-portal-load-failed`. The same burst was then rerun: F2 at N=200, fresh +environments, five plans. + +| | before (10:21) | after (12:52) | +|---|---|---| +| simultaneous cold starts | 200 | 193 | +| SSM `GetParameter` in the burst minute | ~800 | **0** (AWS/Usage CallCount, no datapoint for 12:52) | +| Init p50 / max | 354 / 1,730 ms | **213 / 436 ms** | +| portal lines in the log | 3 closures | none | +| `/api/config` afterwards | — | `enabled: true`; the first call on a fresh environment loads in ~0.4–1.0 s, later calls take 0.1 s | + +The only SSM reads after the deploy are single portal loads of 4 calls each, from +`/api/config` probes. The 10 temporary keys were deleted and the deletion verified; +`LOADTEST_EPOCH` was removed. + +### Replay of the 2026-09-24 herd itself (13:20:45 UTC, after the fix) [M] + +k6 phase `HERD`, set up like the 09-24 incident: +- five temporary keys, one per plan, carrying 2 / 5 / 8 / 16 / 38 virtual users (69 in total); +- every user pre-connected, then all fired one default `GET /v1/assets` at the same instant; +- the gateway cache was empty and every environment cold (env bump). + +| | 2026-09-24 | 2026-09-25 after the fix | +|---|---|---| +| simultaneous cold starts | 69 | 69 (11 + 58 across 13:20:45–46) | +| SSM `GetParameter` in that minute | ~278, 19 throttled | **0** (no AWS/Usage datapoint for 13:20; the 4 at 13:22 are one `/api/config` probe) | +| Init p50 / max | 375 / 2,405 ms | 215 / 274 ms | +| portals closed / `portal-load-failed` or `portal-closed` alarm | 7 / ALARM | 0 / OK | +| 5xx | 0 | 0 | +| ClickHouse | 23.5 cores for ~3 s | ≤ 1.5 cores (69 list queries) | + +The `curl` replay by five Haiku sub-agents, run earlier the same afternoon, did not form a herd. +The agents started seconds apart, so the first miss filled the 60 s cache and only 4 cold starts +happened. It still confirmed every key and limit; the steady-rate overshoot was again a +laptop-curl artefact. + +## Side effects of the test + +- `api-handler-portal-closed` went to ALARM at 10:16:51 (N=150/200, intended). + - The environments were recycled when `LOADTEST_EPOCH` was removed (~10:26). +- `prices-production-asset-discovery-no-invocations` was in ALARM 10:18:25 → 10:21:25. + - It coincides with the backfill's saturation minute; its link to the test is not established [H]. +- No explorer alarm fired. Writer insert p95 stayed ≤ 85 ms during our phases. +- The observer aborted one phase itself. + - The first N=200 attempt, at 10:17, stopped before it fired, due to the backfill saturation above. + - It was rerun at 10:21 once the host was back at 0.3 cores. + - Its summary file was overwritten by the rerun. +- The 0296 rule "capacity tests only with the loadtest key and never as a cold-start burst" was + knowingly broken for this test, with Adam's approval and an announced window. From 0e0deb8e51cd9347ed4f392d17dced6607d09e18 Mon Sep 17 00:00:00 2001 From: Adam <65679285+adamkoot@users.noreply.github.com> Date: Fri, 25 Sep 2026 15:39:47 +0200 Subject: [PATCH 20/22] ci(lore-0311): pin the three usage-plan grants The route check still required 0188's single GetUsage grant on the free plan, which 0311 replaced with three grants on the api-handler role: list plans by key, usage on the key's plan, attach a key to a plan. The old check fails on the current templates. Pin exactly that set in ComputeStack and none in ApiGatewayStack. Count any allowed apigateway resource whose glob reaches a usage plan (`::/*`, `::/usage*`), not only those spelling `/usageplans`, and refuse NotAction/NotResource statements, which cannot be enumerated. --- tools/scripts/verify-openapi-routes.mjs | 166 +++++++++++++++--------- 1 file changed, 102 insertions(+), 64 deletions(-) diff --git a/tools/scripts/verify-openapi-routes.mjs b/tools/scripts/verify-openapi-routes.mjs index ce7a4fc2..8e62817e 100755 --- a/tools/scripts/verify-openapi-routes.mjs +++ b/tools/scripts/verify-openapi-routes.mjs @@ -448,7 +448,7 @@ if (planParamWritten.length !== 1) { `error: expected exactly one SSM parameter ending in ` + `\`/pricing-api-free-plan-id\` in the ApiGateway template, found ` + `${planParamWritten.length}.`, - ' → task 0187 reads the usage-plan id from it at cold start. Without it ' + + ' → task 0187 reads the usage-plan id from it on the first portal request. Without it ' + 'the portal cannot attach a key to a plan, and a key on no plan ' + 'authenticates and is then refused.', ); @@ -460,8 +460,8 @@ if (planParamRead !== planParamWritten[0]) { `the plan id at ${JSON.stringify(planParamWritten[0])}.`, ' → the two stacks cannot reference each other (Compute -> Gateway ' + 'would be a cycle), so this pair is two hand-typed strings. A mismatch ' + - 'fails Lambda INIT on the deploy that opens the portal, which takes ' + - '/v1 down with it. Fix both in infra/src/lib/stacks/compute-stack.ts ' + + 'fails every portal load on the deploy that opens the portal (the ' + + 'portal-load-failed alarm). Fix both in infra/src/lib/stacks/compute-stack.ts ' + 'and api-gateway-stack.ts.', ); } @@ -517,73 +517,111 @@ for (const st of apigatewayStatements) { } } -// --- 5b. Task 0188: the GetUsage grant exists, and is the narrow form. --- -// The dashboard's `GetUsage` needs `apigateway:GET` on the free plan's -// `/usage` sub-resource — a statement in ApiGatewayStack's standalone policy, -// because only that stack knows the plan id. A missing statement fails at -// runtime with AccessDenied, only once the portal opens, and reads as a -// backend bug; a broadened one (`/usageplans/*`, or the plan root) hands the -// api-handler reads this feature never makes. The `apigateway:*` and -// `Resource: "*"` refusals above already bound the worst case; this pins the -// intended shape. -const usageGrants = apigatewayStatements.filter((st) => { - const resources = [st.Resource ?? []].flat(); - // The resource is an Fn::Join carrying the plan id ref, so it is matched as - // serialized JSON rather than as a string. `/usage"` — with the closing - // quote — is the sub-resource as a path SUFFIX; a bare `/usage` would also - // match every `/usageplans/…` ARN, 0187's `/keys` attach included. - return resources.some((r) => JSON.stringify(r).includes('/usage"')); -}); -if (usageGrants.length !== 1) { - fail( - `error: expected exactly one IAM statement on the usage plan's /usage ` + - `sub-resource, found ${usageGrants.length}.`, - ' → task 0188 reads per-key usage with GetUsage, granted as ' + - '`apigateway:GET` on `/usageplans/{planId}/usage` in ' + - 'api-gateway-stack.ts (the standalone portal policy — the plan id ' + - 'lives in that stack). Without it every dashboard load fails with ' + - 'AccessDenied once the portal opens.', +// --- 5b. Tasks 0188 + 0311: the `/usageplans` grants are exactly three. --- +// Task 0188 granted `GetUsage` on the free plan's `/usage` alone. Task 0311 +// widens that deliberately — the dashboard reads the key's OWN plan, and a +// rework re-attaches a paid key to its paid plan — to three statements on the +// api-handler role in ComputeStack (see `PortalListUsagePlansByKey` in +// compute-stack.ts): +// +// - `apigateway:GET` on `/usageplans` (`GetUsagePlans?keyId=`) +// - `apigateway:GET` on `/usageplans/*/usage` (`GetUsage` on the key's plan) +// - `apigateway:POST` on `/usageplans/*/keys` (attach a key to a plan) +// +// A missing one fails at runtime with AccessDenied, only once the portal +// serves a key, and reads as a backend bug. Anything beyond them — a plan +// root (`/usageplans/*`), `PATCH`/`DELETE`, a member listing — lets the +// api-handler change the limits the whole service is metered by. So this pins +// the set, and that none of it sits in the Gateway template (the grants name +// no plan id since 0311, and ComputeStack deploys first, so the code never +// runs ahead of its grants). +const EXPECTED_USAGEPLAN_GRANTS = [ + 'apigateway:GET /usageplans', + 'apigateway:GET /usageplans/*/usage', + 'apigateway:POST /usageplans/*/keys', +]; +// Paths a usage-plan grant can reach. A resource whose IAM glob matches any of +// them counts as a `/usageplans` grant, whether or not it spells the prefix out +// — `::/*` or `::/usage*` reaches every plan exactly as `/usageplans/*` does. +const USAGEPLAN_PROBES = [ + '/usageplans', + '/usageplans/p', + '/usageplans/p/usage', + '/usageplans/p/keys', + '/usageplans/p/keys/k', +]; +const APIGATEWAY_ARN = /^arn:aws[a-z-]*:apigateway:[^:]*::/; +/** Does an IAM resource glob (`*`, `?`) match any usage-plan path? */ +const reachesUsagePlans = (glob) => { + const re = new RegExp( + '^' + + glob + .replace(/[.+^${}()|[\]\\]/g, '\\$&') + .replace(/\*/g, '.*') + .replace(/\?/g, '.') + + '$', ); -} -{ - const actions = [usageGrants[0].Action ?? []].flat().map(String); - if (actions.length !== 1 || actions[0] !== 'apigateway:GET') { - fail( - `error: the /usage grant carries actions ${JSON.stringify(actions)}.`, - ' → GetUsage needs `apigateway:GET` and nothing else on this ' + - 'resource. Anything more (PATCH is UpdateUsage — moving the quota ' + - 'counter) is a different feature and a different decision.', - ); + return USAGEPLAN_PROBES.some((p) => re.test(p)); +}; +/** `ACTION /path` for every allowed (action, resource) pair that reaches `/usageplans`. */ +const usagePlanGrantsIn = (tpl) => { + const grants = []; + for (const [, policy] of resourcesOfType(tpl, 'AWS::IAM::Policy')) { + for (const st of policy.Properties?.PolicyDocument?.Statement ?? []) { + // A Deny grants nothing; NotAction/NotResource cannot be enumerated at + // all, so they are refused on sight rather than silently skipped. + if (st.Effect === 'Deny') continue; + if (st.NotAction !== undefined || st.NotResource !== undefined) { + fail( + `error: an IAM statement uses NotAction/NotResource: ` + + `${JSON.stringify(st)}.`, + ' → an allow-by-exclusion grant cannot be checked against the ' + + 'usage-plan set below — it can reach every plan in the account. ' + + 'List the actions and resources explicitly.', + ); + } + const actions = [st.Action ?? []].flat().map(String); + if (!actions.some((a) => a.startsWith('apigateway:'))) continue; + for (const resource of [st.Resource ?? []].flat()) { + // The ARN is a plain string when the region is concrete; anything + // else (an Fn::Join) is reported as it serializes whenever it could + // name a plan, so it can never match the expected set by accident. + let path; + if (typeof resource === 'string') { + if (!APIGATEWAY_ARN.test(resource)) continue; + path = resource.replace(APIGATEWAY_ARN, ''); + if (!reachesUsagePlans(path)) continue; + } else { + path = JSON.stringify(resource); + if (!/\/usageplans|::\/?\*|::\/usage/.test(path)) continue; + } + for (const action of actions) grants.push(`${action} ${path}`); + } + } } - // The narrow form has two more properties the count and action cannot see: - // the resource names THIS plan (a wildcard `/usageplans/*/usage` would read - // every plan's usage and still count as one statement), and the statement - // lives in the GATEWAY template — only that stack knows the plan id, so a - // copy in ComputeStack would necessarily be hard-coded or wildcarded. - const serialized = JSON.stringify([usageGrants[0].Resource ?? []].flat()); - if (serialized.includes('*')) { + return grants; +}; +{ + const inGateway = usagePlanGrantsIn(template); + if (inGateway.length !== 0) { fail( - `error: the /usage grant's resource contains a wildcard: ${serialized}.`, - ' → the grant is meant to name the one pricing-api-free plan by ' + - 'reference (api-gateway-stack.ts). A wildcard reads usage for every ' + - 'plan in the account.', + `error: \`/usageplans\` grants in the ApiGateway template: ` + + `${JSON.stringify(inGateway)}.`, + " → since task 0311 the portal's usage-plan grants live on the " + + 'api-handler role in compute-stack.ts, which deploys first. A copy in ' + + 'ApiGatewayStack reaches IAM after the code that needs it.', ); } - const inCompute = resourcesOfType(computeTemplate, 'AWS::IAM::Policy').some( - ([, policy]) => - (policy.Properties?.PolicyDocument?.Statement ?? []).some((st) => - [st.Resource ?? []] - .flat() - .some((r) => JSON.stringify(r).includes('/usage"')), - ), - ); - if (inCompute) { + const inCompute = usagePlanGrantsIn(computeTemplate).sort(); + const expected = [...EXPECTED_USAGEPLAN_GRANTS].sort(); + if (JSON.stringify(inCompute) !== JSON.stringify(expected)) { fail( - 'error: a /usage grant appears in the Compute template.', - ' → the GetUsage statement belongs in ApiGatewayStack’s standalone ' + - 'portal policy, where the plan id is a reference rather than a ' + - 'hand-typed string. See the cycle argument on `apiHandlerRole` in ' + - 'api-gateway-stack.ts.', + `error: the api-handler's \`/usageplans\` grants are ` + + `${JSON.stringify(inCompute)}, expected ${JSON.stringify(expected)}.`, + ' → task 0311 grants exactly these three (compute-stack.ts, ' + + '`PortalListUsagePlansByKey` and the two after it). A missing one ' + + 'fails the dashboard or the rework with AccessDenied; an extra one ' + + 'lets the api-handler read or change plans it never needs to.', ); } } From 380291ce025973d9d3a59574e9b92a307c42bf1d Mon Sep 17 00:00:00 2001 From: Adam <65679285+adamkoot@users.noreply.github.com> Date: Fri, 25 Sep 2026 15:59:42 +0200 Subject: [PATCH 21/22] docs(lore-0311): record the 09-25 fix, deploy and live checks History for the 09-24 deploy and alarm, and for 09-25: the herd test, the lazy portal load, SSM high-throughput, the deploy, the herd replay and the live portal check. Eight of nine criteria met; the rework criterion stays open with unit tests only, its live half skipped by Adam's decision. Decisions 8-10 added. --- .../README.md | 100 +++++++++++++++--- 1 file changed, 87 insertions(+), 13 deletions(-) diff --git a/lore/1-tasks/active/0311_FEATURE_paid-plans-and-dashboard-shows-the-keys-own-plan/README.md b/lore/1-tasks/active/0311_FEATURE_paid-plans-and-dashboard-shows-the-keys-own-plan/README.md index fa60c461..44a5d520 100644 --- a/lore/1-tasks/active/0311_FEATURE_paid-plans-and-dashboard-shows-the-keys-own-plan/README.md +++ b/lore/1-tasks/active/0311_FEATURE_paid-plans-and-dashboard-shows-the-keys-own-plan/README.md @@ -43,6 +43,30 @@ history: who: akot note: > Activated; implementation starting. + - date: "2026-09-24" + status: active + who: akot + note: > + Compute + ApiGateway deployed from feat/0311 @ cfe3fce4; the four paid + plans are live (basic o9zex6, analyst 7azpdw, lite d3l658, pro f1e8cb). + The 60 s limit check on five temporary keys matched every plan within + ±4 %, but its cache-miss herd (69 cold starts) closed the portal in 7 + execution environments and fired the portal-closed alarm. + - date: "2026-09-25" + status: active + who: akot + note: > + Five-plan herd and capacity test on production, run beside the 0286 + backfill (notes/R-five-plan-herd-and-capacity-test.md). Root cause: SSM + throttling of the portal reads made at every cold start. SSM + high-throughput enabled. The portal sources now load lazily on the + first portal request, with retry, and the alarm is renamed + portal-load-failed. Six commits, then a /code-review pass: 1 blocker, + 7 warnings and 6 info items, all fixed. Compute + Observability + deployed at a3ae084c. A replay of the 09-24 herd made 69 cold starts, + zero SSM reads and no alarm. Sign-in and dashboard tested live with + the key moved free → Basic → Pro → free (restored to Lite). PR #351 + pushed at 0e0deb8e. --- # Five usage plans, and the dashboard states the key's own plan @@ -192,24 +216,31 @@ reachable; burst is 5× the rate. ## Acceptance Criteria -- [ ] Four paid plans exist in CDK with per-env config; the free plan is - untouched (no replacement in the CFN diff). -- [ ] `/api/usage` reports the key's actual plan (tier, rate, burst, quota, +- [x] Four paid plans exist in CDK with per-env config; the free plan is + untouched (no replacement in the CFN diff). Live since 2026-09-24; + free is still `71t9im`. +- [x] `/api/usage` reports the key's actual plan (tier, rate, burst, quota, period) and the usage counted on that plan. -- [ ] A key moved to any tier: both cards show that tier's figures within - 60 s. Free keys look as they do today. -- [ ] No-plan and unlimited/custom-plan keys render stated, distinct states. -- [ ] IAM widened only to `GET /usageplans`, `GET /usageplans/*/usage` and - `POST /usageplans/*/keys`. +- [x] A key moved to any tier: both cards show that tier's figures within + 60 s. Free keys look as they do today. Seen live on 2026-09-25 for + free, Basic, Pro and Lite on Adam's key. +- [x] No-plan and unlimited/custom-plan keys render stated, distinct states + (asserted in the portal specs; not observed live). +- [x] IAM widened only to `GET /usageplans`, `GET /usageplans/*/usage` and + `POST /usageplans/*/keys`. The set is pinned by + `tools/scripts/verify-openapi-routes.mjs` §5b (0e0deb8e). - [ ] A rework on a paid (and on a custom) plan issues the new key on the **same** plan; tested in unit tests (plan read before delete) and on dev (Basic key → rework → next period → new key on Basic). A free key's - rework stays on free. -- [ ] Runbook for upgrading a user (Step 4) in the wiki or ops docs. -- [ ] Rate Limit card header shows the plan pill (`Free`…`Pro`, `Custom`) + rework stays on free. **Unit tests only.** The live half was skipped + by Adam's decision (2026-09-25): there is no dev environment, and a + rework would revoke Adam's only key until 2026-10-01. +- [x] Runbook for upgrading a user (Step 4) in the wiki or ops docs + (`docs/runbooks/manual-api-key-tier.md`). +- [x] Rate Limit card header shows the plan pill (`Free`…`Pro`, `Custom`) beside "Active", asserted per tier in `app.spec.tsx`. -- [ ] The contact line links to `RUMBLEFISH_CONTACT`, with the copy for each - tier from decision 6. +- [x] The contact line links to `RUMBLEFISH_CONTACT`, with the copy for each + tier from decision 6. Seen live for free, Basic and Pro. ## Decisions (Adam) @@ -252,6 +283,49 @@ reachable; burst is 5× the rate. 7. **A rework keeps the key on the same plan** (2026-09-24). This supersedes the 2026-09-23 "not now": with paid keys being issued, a rework must not be a silent downgrade to free. See Step 2b. +8. **SSM high-throughput on** (2026-09-25). A stopgap for Parameter Store + throttling. It costs about $0.05 per 10 k calls, well under $1 a month. +9. **Lazy portal load, and the alarm renamed** (2026-09-25). + - A cold start reads nothing for the portal. + - `/api/config` triggers the load (decision A). A failed load answers + `enabled:false` for that one response only. + - Other portal routes answer 503 while the load has failed. + - The alarm is `api-handler-portal-load-failed`; `portal-closed` is gone. + - `PORTAL_ENABLED` stays as the operator's kill switch. +10. **No follow-up tasks** (2026-09-25) for the capacity risks found by the + test: the `prices_read` quota alarm, the Lambda throttle and concurrency + alarms, and a soak test after the 0286 backfill. + +## Portal load after 2026-09-25 (quick 260925-i0u) + +- **Commits.** Six commits, e8c1fc30..a3ae084c, on top of 34c2fae4: + - `portal/sources.rs` holds one shared, single-flight lazy handle. The + five reads run concurrently under a 4 s budget. A failed load is cached + for at most 2 s (`LOAD_FAILURE_COOLDOWN`). + - `portal/extension.rs` retries only transient reads: 3 attempts, with + backoff capped at 100 and 300 ms and full jitter. + - `mtls.rs` and the `/v1` cold path are unchanged. +- **What the user sees on a failed load.** + - Sign-in callback: the load has a 2 s allowance, taken out of the token + exchange, so the worst case stays at 14 s against the 15 s Lambda + timeout. An issue flow lands on `?issue=failed`, anything else on + `?signin=failed`. + - `/me`, `/key` and `/usage` answer 503 (the frontend already renders + these). `USAGE_TIMEOUT_MS` was raised to 20 s. +- **Tests.** 550 prices-api tests pass, with new `portal_lazy_load` and + `portal_load_logs` binaries. Infra 41/41, web portal 265. The guard test + was renamed `portal-load-failed-filter-guard.test.mjs`. +- **Known limit.** `mtls.rs` errors carry only the HTTP status, not the + body. So a throttle that the extension reports as HTTP 400 is not + retried; hung reads are. SSM high-throughput covers this. +- **Measured after the deploy.** + + | Burst | Cold starts | SSM reads | Init p50 / max | + |---|---|---|---| + | `/price` burst | 193 | 0 | 213 / 436 ms | + | 09-24 herd replay | 69 | 0 | 215 / 274 ms | + + Before the change, Init was 354–375 ms, with a tail up to 2.4 s. ## Out of scope From 1a8aac9643500cc17ae2194103693b5aa1cf64fd Mon Sep 17 00:00:00 2001 From: Adam <65679285+adamkoot@users.noreply.github.com> Date: Fri, 25 Sep 2026 16:14:37 +0200 Subject: [PATCH 22/22] fix(lore-0311): settle a 409 attach without reading the plan back CreateUsagePlanKey answers 409 when the key is already on the plan it was asked for, so the refusal names the plan. Reading it back with GetUsagePlans, which can lag the attach, let a double-submit re-run until the attempts ran out and land on ?issue=failed for a key that works. Before 0311 a 409 settled at once, and it does again. Only the 400 "same API Stage" refusal, which does not say which plan, still asks plan_of. The 409 race test now lags GetUsagePlans for good and must still end in ?issue=ok; it fails on the previous mapping. Refs: PR #351 review --- .../prices-api/src/portal/keys/gateway.rs | 43 ++++++++++--------- packages/prices-api/src/portal/keys/mod.rs | 16 ++++--- packages/prices-api/tests/portal_issue.rs | 17 +++++--- packages/prices-api/tests/portal_rework.rs | 8 ++-- 4 files changed, 48 insertions(+), 36 deletions(-) diff --git a/packages/prices-api/src/portal/keys/gateway.rs b/packages/prices-api/src/portal/keys/gateway.rs index e76394b8..490732eb 100644 --- a/packages/prices-api/src/portal/keys/gateway.rs +++ b/packages/prices-api/src/portal/keys/gateway.rs @@ -132,20 +132,19 @@ impl std::fmt::Debug for KeyValue { /// to find out rather than assume (task 0311). #[derive(Debug, Clone, Copy, PartialEq, Eq)] pub enum Attachment { - /// This call put the key on the usage plan it was given. + /// The key is on the usage plan it was given: this call put it there, or + /// AWS answered `409 ConflictException` — by its own wording the key + /// "already exists in the usage plan", i.e. THIS plan. A `409` names the + /// plan, so it is not read back: `GetUsagePlans` can lag the attach, and + /// waiting for it to catch up turned a double-submit into `?issue=failed` + /// for a key that works (PR #351 review). OnPlan, - /// The key was already on a usage plan for this API stage, so AWS refused - /// the attach (task 0311). Two refusals land here: - /// - /// - `409 ConflictException` — by AWS's own wording "already exists in the - /// usage plan", i.e. THIS plan; - /// - `400 BadRequestException` "… cannot reference multiple Usage Plans - /// with the same API Stage" — ANOTHER plan on the same stage, since a - /// key belongs to one plan per stage. - /// - /// Neither status code is trusted to say which plan: the caller asks - /// [`Gateway::plan_of`]. A key a concurrent issue, or an operator, put on - /// a plan a moment ago is a working key, not a failure. + /// The key is on ANOTHER usage plan for this API stage: AWS refused the + /// attach with `400 BadRequestException` "… cannot reference multiple + /// Usage Plans with the same API Stage", since a key belongs to one plan + /// per stage (task 0311). The refusal does not say which plan, so the + /// caller asks [`Gateway::plan_of`]. A key an operator, or a concurrent + /// issue, put on a plan a moment ago is a working key, not a failure. AlreadyOnAPlan, /// The key no longer exists, so there was nothing to attach. KeyGone, @@ -790,11 +789,12 @@ impl Gateway { /// **Idempotent**, and that is what lets the caller run it on every key it /// is about to hand out rather than only on keys it just created. API /// Gateway answers `409 ConflictException` when the key is already on the - /// plan, and `400 BadRequestException` ("cannot reference multiple Usage - /// Plans with the same API Stage") when it is already on ANOTHER plan for - /// the same stage. Neither is a failure: both are - /// [`Attachment::AlreadyOnAPlan`], and the caller asks [`Self::plan_of`] - /// which plan that is (task 0311). Any other `400` is still an error. + /// plan — [`Attachment::OnPlan`], as for a fresh attach — and + /// `400 BadRequestException` ("cannot reference multiple Usage Plans with + /// the same API Stage") when it is already on ANOTHER plan for the same + /// stage — [`Attachment::AlreadyOnAPlan`], and the caller asks + /// [`Self::plan_of`] which plan that is (task 0311). Any other `400` is + /// still an error. /// /// A `404` is **ambiguous** and is resolved before it is acted on. API /// Gateway answers `NotFoundException` both when the key is gone and when @@ -823,9 +823,10 @@ impl Gateway { Err(e) => { let message = sdk_message(&e); let service_error = e.into_service_error(); - if service_error.is_conflict_exception() - || (service_error.is_bad_request_exception() - && is_same_stage_refusal(service_error.message())) + if service_error.is_conflict_exception() { + Ok(Attachment::OnPlan) + } else if service_error.is_bad_request_exception() + && is_same_stage_refusal(service_error.message()) { Ok(Attachment::AlreadyOnAPlan) } else if service_error.is_not_found_exception() { diff --git a/packages/prices-api/src/portal/keys/mod.rs b/packages/prices-api/src/portal/keys/mod.rs index 61670d26..36ba8432 100644 --- a/packages/prices-api/src/portal/keys/mod.rs +++ b/packages/prices-api/src/portal/keys/mod.rs @@ -1319,12 +1319,16 @@ enum Settled { /// Attach `key_id` to `plan_id`, and settle what AWS answered (task 0311). /// -/// [`Attachment::AlreadyOnAPlan`] is the interesting case. A `409` (this -/// plan) or the `400` "cannot reference multiple Usage Plans with the same -/// API Stage" (another plan) means the key is ALREADY usable — some writer got -/// there first: the other half of a double-submit, a sign-in that ran inside -/// an operator's delete→create gap, or the operator. Which plan it is, is -/// asked of [`Gateway::plan_of`] rather than inferred from the status code: +/// A `409` (already on THIS plan) arrives as [`Attachment::OnPlan`]: the +/// refusal names the plan, so it is settled without a read-back that +/// `GetUsagePlans` could lag (PR #351 review). +/// +/// [`Attachment::AlreadyOnAPlan`] is the interesting case. The `400` "cannot +/// reference multiple Usage Plans with the same API Stage" (another plan) +/// means the key is ALREADY usable — some writer got there first: a sign-in +/// that ran inside an operator's delete→create gap, the operator, or a +/// concurrent issue that resolved a different plan. Which plan it is, is +/// asked of [`Gateway::plan_of`], since the refusal does not name it: /// /// - on a plan for our stage → [`Settled::Ready`]; if that is not the plan /// this attempt wanted, the key is left where it is (this code has no grant diff --git a/packages/prices-api/tests/portal_issue.rs b/packages/prices-api/tests/portal_issue.rs index e200bf07..a2165796 100644 --- a/packages/prices-api/tests/portal_issue.rs +++ b/packages/prices-api/tests/portal_issue.rs @@ -1072,10 +1072,12 @@ async fn attaching_to_a_paid_plan_puts_the_key_on_it() { ); } -/// AWS's two "already on a plan" refusals are both `AlreadyOnAPlan` (task -/// 0311, review WR-05): `409` for the same plan, and `400` "cannot reference -/// multiple Usage Plans with the same API Stage" for another plan on the -/// stage. Neither moves the key, and neither is an error. +/// AWS's two "already on a plan" refusals (task 0311, review WR-05; PR #351 +/// review): `409` means the key is already on THE plan asked for, so it is +/// `OnPlan` — the refusal names the plan and needs no read-back — while `400` +/// "cannot reference multiple Usage Plans with the same API Stage" means +/// ANOTHER plan on the stage, `AlreadyOnAPlan`. Neither moves the key, and +/// neither is an error. #[tokio::test] async fn both_already_on_a_plan_refusals_are_reported_as_such() { use prices_api::portal::keys::gateway::Attachment; @@ -1087,10 +1089,13 @@ async fn both_already_on_a_plan_refusals_are_reported_as_such() { }); let client = test_gateway(&gateway.base); - for plan in [BASIC_PLAN_ID, PLAN_ID] { + for (plan, expected) in [ + (BASIC_PLAN_ID, Attachment::OnPlan), + (PLAN_ID, Attachment::AlreadyOnAPlan), + ] { assert_eq!( client.attach_to_plan(&key, plan).await.expect(plan), - Attachment::AlreadyOnAPlan, + expected, "{plan}" ); } diff --git a/packages/prices-api/tests/portal_rework.rs b/packages/prices-api/tests/portal_rework.rs index 100d9b1e..8ed04fe0 100644 --- a/packages/prices-api/tests/portal_rework.rs +++ b/packages/prices-api/tests/portal_rework.rs @@ -1516,8 +1516,10 @@ async fn an_attach_refused_for_another_plan_on_the_stage_keeps_that_plan() { /// The double-submit rework: the other invocation already put the new key on /// Basic (the previous key's plan) but this one's `GetUsagePlans` for it -/// lags. It resolves Basic from the revoked record, attaches, is refused with -/// `409` — already on THIS plan — confirms Basic and carries on to the sweep. +/// lags — and keeps lagging. It resolves Basic from the revoked record, +/// attaches, is refused with `409` — already on THIS plan — and settles on +/// that alone: no read-back that the lag could stall into `?issue=failed` +/// for a key that works (PR #351 review), and the sweep still runs. #[tokio::test] async fn a_conflict_on_the_same_plan_is_settled_and_the_sweep_still_runs() { let discord = MockDiscord::start(GRANTED_SCOPE, None).await; @@ -1530,7 +1532,7 @@ async fn a_conflict_on_the_same_plan_is_settled_and_the_sweep_still_runs() { ); let new = gateway.with(|s| { let new = s.seed_on_plan(&key_name(), 2_000, BASIC_PLAN_ID); - s.plans_hidden_for.insert(new.clone(), 1); + s.plans_hidden_for.insert(new.clone(), usize::MAX); new }); let app = app_with_discord(&discord, &gateway);