diff --git a/.air.toml b/.air.toml new file mode 100644 index 000000000..a9a550d79 --- /dev/null +++ b/.air.toml @@ -0,0 +1,49 @@ +# Air configuration for hot reload in development +# https://github.com/air-verse/air + +root = "." +testdata_dir = "testdata" +tmp_dir = "tmp" + +[build] + args_bin = [] + bin = "./tmp/main" + cmd = "go build -o ./tmp/main ./cmd/server" + delay = 1000 + exclude_dir = ["assets", "tmp", "vendor", "testdata", "frontend", "node_modules"] + exclude_file = [] + exclude_regex = ["_test.go"] + exclude_unchanged = false + follow_symlink = false + full_bin = "" + include_dir = [] + include_ext = ["go", "tpl", "tmpl", "html"] + include_file = [] + kill_delay = "0s" + log = "build-errors.log" + poll = false + poll_interval = 0 + post_cmd = [] + pre_cmd = [] + rerun = false + rerun_delay = 500 + send_interrupt = false + stop_on_error = false + +[color] + app = "" + build = "yellow" + main = "magenta" + runner = "green" + watcher = "cyan" + +[log] + main_only = false + time = false + +[misc] + clean_on_exit = false + +[screen] + clear_on_rebuild = false + keep_scroll = true diff --git a/.coderabbit.yaml b/.coderabbit.yaml new file mode 100644 index 000000000..d00510690 --- /dev/null +++ b/.coderabbit.yaml @@ -0,0 +1,3 @@ +reviews: + path_filters: + - '**/package-lock.json' diff --git a/.dockerignore b/.dockerignore new file mode 100644 index 000000000..1eed8dfe6 --- /dev/null +++ b/.dockerignore @@ -0,0 +1,39 @@ +# Terraform state and providers +**/.terraform +*.tfstate +*.tfstate.backup +*.tfbackend + +# Frontend build artifacts and dependencies +frontend/node_modules +frontend/coverage +frontend/dist + +# IDE and OS +.idea +.vscode +.DS_Store + +# Git +.git + +# Go vendor (dependencies downloaded at build time) +vendor/ + +# Temporary files +tmp/ + +# Docker compose dev files +docker-compose.yml +Dockerfile.dev + +# Secrets and credentials +.env +.env.* +*.pem +*.key +*.p12 +credentials/ +*.tfstate +*.tfstate.backup +.terraform/ diff --git a/.env.example b/.env.example new file mode 100644 index 000000000..196f90174 --- /dev/null +++ b/.env.example @@ -0,0 +1,123 @@ +# CUDly local development env template +# +# Copy this file to `.env.local` (already in .gitignore) and fill in +# the placeholders. Loaded by: load-env.sh / your IDE / `direnv` — +# CUDly itself reads these via os.Getenv at runtime. +# +# All values here are PLACEHOLDERS. Never commit real secrets to .env* +# files; the .gitignore at the repo root already excludes everything +# matching `.env*` except this template. + +# --------------------------------------------------------------------- +# Required: secrets resolver +# --------------------------------------------------------------------- +# SECRET_PROVIDER selects which secret store the resolver fetches from. +# aws | gcp | azure — production: real Secrets Manager / Key Vault +# env — local dev: resolve secret names to env vars +# In `env` mode, every secret-ref var (ADMIN_PASSWORD_SECRET, +# API_KEY_SECRET_ARN, etc.) holds the NAME of another env var whose +# value is the actual secret. See `internal/secrets/env_resolver.go`. +# Pairs with EMAIL_ENABLED=false so the email factory's no-op sender +# kicks in (see PR #333) — otherwise `env` is not a recognised email +# backend and the factory would fail dispatch. +SECRET_PROVIDER=env + +# --------------------------------------------------------------------- +# Required: scheduled-task auth mode (no default — must be explicit) +# --------------------------------------------------------------------- +# Selects how the internal /api/scheduled/* endpoints authenticate. +# oidc — production: verify Google-issued OIDC ID tokens +# bearer — shared-secret token in Authorization header +# disabled — local dev: no auth check +# Required by `internal/server/scheduledauth/config.go`; app refuses +# to start when unset. +SCHEDULED_TASK_AUTH_MODE=disabled + +# --------------------------------------------------------------------- +# Required: email gate (factory short-circuit, see PR #333) +# --------------------------------------------------------------------- +# When `false`, internal/email/factory.go returns a no-op sender that +# logs each invocation at debug level instead of dispatching to a +# cloud-specific backend. Pair with SECRET_PROVIDER=env for local dev. +EMAIL_ENABLED=false + +# --------------------------------------------------------------------- +# Required: credential encryption key +# --------------------------------------------------------------------- +# In production exactly ONE of the per-cloud secret refs is set; the +# Go side reads them in priority order (ARN → NAME → ID → raw KEY). +# For local dev set CREDENTIAL_ENCRYPTION_ALLOW_DEV_KEY=1 to use the +# all-zero dev key without touching a Secrets Manager / Key Vault. +CREDENTIAL_ENCRYPTION_ALLOW_DEV_KEY=1 + +# CREDENTIAL_ENCRYPTION_KEY_SECRET_ARN=arn:aws:secretsmanager:us-east-1:000000000000:secret:cudly-cred-enc-key-PLACEHOLDER +# CREDENTIAL_ENCRYPTION_KEY_SECRET_NAME=cudly-credential-encryption-key +# CREDENTIAL_ENCRYPTION_KEY_SECRET_ID=cudly-credential-encryption-key +# CREDENTIAL_ENCRYPTION_KEY=<64-hex-chars> + +# --------------------------------------------------------------------- +# Required: admin auth + API +# --------------------------------------------------------------------- +# With SECRET_PROVIDER=env, the *_SECRET / *_SECRET_ARN vars hold the +# NAME of another env var whose value is the actual secret. The matched +# env var must then be defined below. Two reasons for the indirection: +# (1) production uses the same var name pointing at a real ARN/name; +# (2) the dev value is co-located with its lookup key so future readers +# can trace the chain in one file. +ADMIN_EMAIL=admin@cudly.local +ADMIN_PASSWORD_SECRET=ADMIN_PASSWORD_DEV +ADMIN_PASSWORD_DEV=LocalDev!Pass123 +API_KEY_SECRET_ARN=ADMIN_API_KEY_DEV +ADMIN_API_KEY_DEV=cudly-local-dev-api-key-not-for-prod +# Required for signed one-click notification unsubscribe links. Generate a +# unique value for every environment (for example: openssl rand -hex 32). +NOTIFICATION_MUTE_SECRET= +# Production examples (override SECRET_PROVIDER and these): +# ADMIN_PASSWORD_SECRET=arn:aws:secretsmanager:us-east-1:000000000000:secret:cudly-admin-password-PLACEHOLDER +# API_KEY_SECRET_ARN=arn:aws:secretsmanager:us-east-1:000000000000:secret:cudly-api-key-PLACEHOLDER + +# --------------------------------------------------------------------- +# Optional: web frontend / CORS / dashboard +# --------------------------------------------------------------------- +CORS_ALLOWED_ORIGIN=http://localhost:3000 +DASHBOARD_URL=http://localhost:3000 +# ENABLE_DASHBOARD=true +# DASHBOARD_BUCKET=cudly-dashboard-PLACEHOLDER + +# --------------------------------------------------------------------- +# Optional: PostgreSQL (lazy — unset to skip) +# --------------------------------------------------------------------- +# DB_HOST=localhost +# DB_PORT=5432 +# DB_USER=cudly +# DB_NAME=cudly +# DB_PASSWORD_SECRET=arn:aws:secretsmanager:us-east-1:000000000000:secret:cudly-db-PLACEHOLDER +# CUDLY_MIGRATION_TIMEOUT=2m + +# --------------------------------------------------------------------- +# Optional: cloud-provider profiles for multi-cloud onboarding +# --------------------------------------------------------------------- +# AWS_CONFIG_FILE=$HOME/.aws/config +# AZURE_TENANT_ID=00000000-0000-0000-0000-000000000000 +# AZURE_CLIENT_ID=00000000-0000-0000-0000-000000000000 +# AZURE_SUBSCRIPTION_ID=00000000-0000-0000-0000-000000000000 +# AZURE_KEY_VAULT_URL=https://cudly-vault-placeholder.vault.azure.net/ +# GCP_PROJECT_ID=cudly-placeholder +# AWS_REGION=us-east-1 + +# --------------------------------------------------------------------- +# Optional: SES / Azure ACS / SendGrid email config +# --------------------------------------------------------------------- +# EMAIL_ADDRESS=noreply@cudly.example +# AZURE_SMTP_HOST=smtp.azurecomm.net +# AZURE_SMTP_USERNAME_SECRET=arn:aws:secretsmanager:us-east-1:000000000000:secret:cudly-smtp-user-PLACEHOLDER +# AZURE_SMTP_PASSWORD_SECRET=arn:aws:secretsmanager:us-east-1:000000000000:secret:cudly-smtp-pass-PLACEHOLDER + +# --------------------------------------------------------------------- +# Optional: tunables +# --------------------------------------------------------------------- +# CUDLY_RECOMMENDATION_CACHE_TTL=15m +# CUDLY_MAX_ACCOUNT_PARALLELISM=8 +# DEFAULT_PAYMENT_OPTION=no-upfront +# DEFAULT_RAMP_SCHEDULE=quarterly +# ENVIRONMENT=local diff --git a/.gitallowed b/.gitallowed new file mode 100644 index 000000000..6ff918dde --- /dev/null +++ b/.gitallowed @@ -0,0 +1,48 @@ +# git-secrets allowlist: regexes that should NOT trigger the AWS-secret scanner. +# +# The scanner's `--register-aws` adds a 12-digit account-ID pattern that +# matches our test fixtures. The values below are obvious test placeholders +# (sequential 123456789012, all-same-digit blocks like 111111111111, and +# countdown patterns like 999888777666) — chosen specifically because they +# cannot be real customer accounts. Adding these is safer than disabling the +# AWS-account check entirely. +# +# DO NOT add a real account ID here. If a real account ID lands in the repo, +# rotate it and treat the leak seriously instead of silencing the scanner. +# +# Format: one regex per line, matched against each line of file content +# (path-scoping is not supported by .gitallowed). + +# Sequential test placeholders (ascending and descending). +123456789012 +210987654321 + +# All-same-digit blocks of 12 — used as table-row distinctions in fixtures. +# Listed explicitly rather than via a regex back-reference because not every +# git-secrets build supports back-refs in its allowlist regexes. +000000000000 +111111111111 +222222222222 +333333333333 +444444444444 +555555555555 +666666666666 +777777777777 +888888888888 +999999999999 + +# Group-block placeholders used in cmd/helpers_test.go variants and +# cmd/multi_service_helpers_test.go. +111222333444 +555666777888 +999888777666 +# Descending-step fixture added in #956 tests (store_postgres_pgxmock_test.go, +# handler_analytics_test.go, handler_history_test.go, handler_dashboard_test.go). +999988887777 +# Repeating-pair-block fixture used in handler_analytics_test.go, +# handler_inventory_test.go, store_postgres_pgxmock_test.go. +111122223333 + +# UUID-shaped account ID used in handler_accounts_test.go (synthetic, not a +# real subscription/account). +11111111-1111-1111-1111-111111111111 diff --git a/.github/runbooks/compromised-dependency.md b/.github/runbooks/compromised-dependency.md new file mode 100644 index 000000000..579eb3503 --- /dev/null +++ b/.github/runbooks/compromised-dependency.md @@ -0,0 +1,138 @@ +# Runbook: Compromised Dependency + +**Trigger**: A Go module, npm package, Docker base image, or GitHub Action used by CUDly has been found to be malicious or critically vulnerable. + +**Owner**: On-call engineer +**Severity**: P1 (active exploitation) / P2 (known critical CVE, no evidence of exploitation) + +--- + +## Step 1: Assess Impact + +```bash +# For Go dependency CVE +govulncheck ./... + +# For npm dependency CVE +cd frontend && npm audit + +# For Docker image CVE +trivy image + +# For GitHub Actions compromise +# Check: https://github.com/advisories (filter by Actions category) +``` + +Questions to answer: + +- Is the vulnerable code path reachable in production? +- Does exploitation require authentication? +- Is there evidence of active exploitation in logs? + +--- + +## Step 2: Pin to Safe Version Immediately + +### Go Dependency + +```bash +# Pin to last known-good version +go get github.com/affected/package@v1.2.3-safe + +# If no safe version exists, pin to a known commit hash +go get github.com/affected/package@ + +go mod tidy +go mod verify + +# Run tests +go test ./... +``` + +### npm Dependency + +```bash +cd frontend + +# Force a specific version +npm install affected-package@ + +# Or use npm audit fix +npm audit fix + +# Verify +npm audit +npm test +``` + +### Docker Base Image + +Update the `FROM` line in Dockerfile(s) to a patched version: + +```dockerfile +# Before +FROM golang:1.25.4-alpine3.21 + +# After (example: patch to new minor that fixes CVE) +FROM golang:1.25.5-alpine3.21 +``` + +Rebuild and push the container image. + +### GitHub Action + +Pin the compromised action to the last known-good commit SHA: + +```yaml +# Before (vulnerable) +- uses: some-org/some-action@v1.2.3 + +# After (pinned to safe commit) +- uses: some-org/some-action@ +``` + +Or remove the action entirely and replace with equivalent logic using trusted actions or shell commands. + +--- + +## Step 3: Deploy the Fix + +```bash +# Tag and push +git add go.mod go.sum Dockerfile +git commit -m "security: pin to safe version (CVE-YYYY-XXXXX)" +git push + +# CI/CD will build and deploy +# If CI is also affected by the compromised dependency, run manually: +make build && make push +``` + +--- + +## Step 4: Investigate for Active Exploitation + +```bash +# Check application logs for exploitation patterns +aws logs filter-log-events \ + --log-group-name /aws/lambda/cudly \ + --filter-pattern "" \ + --start-time + +# Check CloudTrail for unexpected API calls from application role +aws cloudtrail lookup-events \ + --lookup-attributes AttributeKey=Username,AttributeValue= \ + --start-time +``` + +--- + +## Step 5: Verify & Post-Mortem + +- [ ] Confirm `govulncheck` / `npm audit` / `trivy` no longer report the CVE +- [ ] Run full test suite +- [ ] Deploy to staging, then production +- [ ] Document the incident timeline +- [ ] Add the CVE to monitoring rules (detect if patched version is reverted) +- [ ] Review dependency update cadence — consider automating with Dependabot/Renovate +- [ ] Review GitHub Actions pinning strategy → see SC-003 recommendation diff --git a/.github/runbooks/credential-compromise.md b/.github/runbooks/credential-compromise.md new file mode 100644 index 000000000..0f3163547 --- /dev/null +++ b/.github/runbooks/credential-compromise.md @@ -0,0 +1,129 @@ +# Runbook: Credential Compromise + +**Trigger**: A secret, API key, password, or access token has been exposed (committed to git, found in logs, leaked via third-party breach, etc.) + +**Owner**: On-call engineer +**Severity**: P1 (if production credentials) / P2 (if dev/staging credentials) + +--- + +## Step 1: Identify Scope (5 min) + +Determine which credentials were compromised: + +- [ ] AWS IAM credentials (access key ID + secret)? +- [ ] Azure Service Principal client secret? +- [ ] GCP Service Account JSON key? +- [ ] Database passwords (PostgreSQL)? +- [ ] Application secrets (session secret, JWT secret)? +- [ ] SMTP/email credentials? +- [ ] Terraform state credentials? + +--- + +## Step 2: Rotate Immediately + +Do NOT wait until you understand the full impact. Rotate first, investigate after. + +### AWS Credentials + +```bash +# Disable the old key immediately +aws iam update-access-key --access-key-id --status Inactive + +# Create new key +aws iam create-access-key --user-name + +# Update the secret in Secrets Manager +aws secretsmanager update-secret --secret-id --secret-string '{"access_key_id":"NEW","secret_access_key":"NEW"}' + +# Delete old key after confirming new one works +aws iam delete-access-key --access-key-id +``` + +### Database Password (PostgreSQL on RDS) + +```bash +# Generate a new strong password +NEW_PASS=$(openssl rand -base64 32) + +# Rotate via AWS Secrets Manager (if using auto-rotation) +aws secretsmanager rotate-secret --secret-id + +# Or manually via RDS +aws rds modify-db-instance --db-instance-identifier \ + --master-user-password "$NEW_PASS" --apply-immediately + +# Update the Secrets Manager value +aws secretsmanager put-secret-value --secret-id \ + --secret-string "{\"password\":\"$NEW_PASS\"}" +``` + +### Azure Client Secret + +1. Go to Azure Portal → Azure Active Directory → App Registrations → CUDly app +2. Certificates & Secrets → Delete the old secret → Add new secret +3. Update in AWS Secrets Manager: `aws secretsmanager put-secret-value --secret-id --secret-string '{"tenant_id":"...","client_id":"...","client_secret":"NEW_SECRET","subscription_id":"..."}'` + +### Application Secrets (session-secret, jwt-secret) + +```bash +# Generate new secret +NEW_SECRET=$(openssl rand -hex 32) + +# Update in AWS Secrets Manager +aws secretsmanager put-secret-value --secret-id \ + --secret-string "{\"value\":\"$NEW_SECRET\"}" + +# IMPORTANT: This invalidates all existing sessions. Users will be logged out. +``` + +--- + +## Step 3: Invalidate Active Sessions + +If application secrets were compromised, all existing user sessions must be invalidated: + +```sql +-- Connect to the production database +DELETE FROM sessions WHERE created_at < NOW(); + +-- Or to be safe, invalidate all sessions +TRUNCATE TABLE sessions; +``` + +--- + +## Step 4: Investigate + +After credentials are rotated: + +- [ ] Determine when the credentials were first exposed (git blame, CloudTrail, log search) +- [ ] Check CloudTrail for unauthorized API calls using the compromised credentials: + + ```bash + aws cloudtrail lookup-events \ + --lookup-attributes AttributeKey=AccessKeyId,AttributeValue= \ + --start-time --end-time + ``` + +- [ ] Check AWS Config for resources created/modified during the exposure window +- [ ] Review CloudWatch Logs for unusual application activity + +--- + +## Step 5: Notify + +- [ ] Notify incident channel immediately with: "Credential X rotated. Investigating exposure window." +- [ ] If AWS credentials were used maliciously: contact AWS Support to assist with forensics +- [ ] If user data was accessed: initiate data breach response → see [data-breach-response.md](data-breach-response.md) +- [ ] If git commit: remove from history using BFG Repo Cleaner, force-push, notify all contributors to re-clone + +--- + +## Step 6: Verify & Harden + +- [ ] Confirm new credentials are working in production +- [ ] Add the exposed credential type to `.gitignore` / `.dockerignore` (if applicable) +- [ ] Enable gitleaks pre-commit hook to prevent future commits of secrets +- [ ] Add detection alert so this exposure type triggers an alarm in future diff --git a/.github/runbooks/data-breach-response.md b/.github/runbooks/data-breach-response.md new file mode 100644 index 000000000..a7ea44207 --- /dev/null +++ b/.github/runbooks/data-breach-response.md @@ -0,0 +1,126 @@ +# Runbook: Data Breach Response + +**Trigger**: Personal data (email addresses, passwords, MFA secrets, purchase history) may have been accessed or exfiltrated without authorization. + +**Owner**: On-call engineer + management +**Severity**: P1 + +--- + +## Immediate Actions (first 30 min) + +- [ ] **Do not panic or act hastily** — preservation of evidence is critical +- [ ] Assign Incident Commander immediately +- [ ] Open private incident channel; do not discuss in public channels or on public issue trackers +- [ ] Snapshot affected system logs **before making any changes** + +--- + +## Step 1: Assess Scope + +Determine what data may have been exposed: + +| Data Type | Location | Sensitivity | +| --------- | -------- | ----------- | +| Email addresses | `users` table | High | +| Hashed passwords | `users` table | High | +| MFA secrets (TOTP) | `mfa_secrets` table | Critical | +| Session tokens | `sessions` table | Critical | +| API keys | `api_keys` table | High | +| Cloud account data | `cloud_accounts` table | High | +| Purchase history | `purchases` table | High | + +Questions to answer: + +- Which database tables were accessed? +- How many rows / which user records? +- Was the access read-only or did data leave the system? +- What was the attack vector? + +--- + +## Step 2: Contain + +- [ ] If breach is ongoing: take the service offline or restrict to maintenance mode +- [ ] Invalidate all active sessions immediately: + + ```sql + TRUNCATE TABLE sessions; + ``` + +- [ ] Rotate application secrets (session-secret) → see [credential-compromise.md](credential-compromise.md) +- [ ] If MFA secrets were exposed: disable TOTP for affected users and force re-enrolment +- [ ] If API keys were exposed: revoke all affected API keys: + + ```sql + UPDATE api_keys SET revoked = true WHERE user_id IN (); + ``` + +- [ ] Block the attacker's IP/user-agent at WAF or security group level + +--- + +## Step 3: Preserve Evidence + +```bash +# Export relevant CloudWatch Logs before they expire +aws logs create-export-task \ + --log-group-name /aws/lambda/cudly \ + --from \ + --to \ + --destination \ + --destination-prefix incident-YYYY-MM-DD + +# Capture RDS audit logs +aws rds download-db-log-file-portion \ + --db-instance-identifier \ + --log-file-name +``` + +--- + +## Step 4: Eradicate + +- [ ] Patch the vulnerability that allowed the breach +- [ ] Deploy the fix to staging and verify +- [ ] Deploy to production + +--- + +## Step 5: Notify (GDPR Art. 33 / 34) + +**The 72-hour clock starts when you become aware of the breach.** + +### Data Protection Authority (DPA) + +If the breach involves EU residents' personal data and is likely to result in risk: + +- [ ] File notification with the relevant DPA within **72 hours** +- Required information: + - Nature of the breach (categories of data, approximate number of records) + - Contact details of the DPO/responsible person + - Likely consequences of the breach + - Measures taken or proposed to address it + +DPA contacts: + +### Affected Users + +If the breach is likely to result in **high risk** to users' rights: + +- Notify affected users **without undue delay** (aim for 24-48 hours after DPA notification) +- Include: + - What happened (in plain language) + - What data was involved + - Recommended actions (change password, enable MFA, watch for phishing) + - Contact for questions + +--- + +## Step 6: Post-Breach Hardening + +- [ ] Enable field-level encryption for sensitive columns (MFA secrets) +- [ ] Add database query audit logging +- [ ] Implement anomaly detection on query volume (detect bulk exports) +- [ ] Review and tighten IAM/database permissions +- [ ] Conduct a full security review of authentication flows diff --git a/.github/runbooks/ddos-mitigation.md b/.github/runbooks/ddos-mitigation.md new file mode 100644 index 000000000..1c997a928 --- /dev/null +++ b/.github/runbooks/ddos-mitigation.md @@ -0,0 +1,113 @@ +# Runbook: DDoS Mitigation + +**Trigger**: Service is experiencing unusually high traffic volumes causing degraded performance or unavailability. + +**Owner**: On-call engineer +**Severity**: P1 (service down) / P2 (degraded) + +--- + +## Step 1: Confirm It's DDoS (5 min) + +Check CloudWatch metrics: + +```bash +# Lambda: request count spike +aws cloudwatch get-metric-statistics \ + --namespace AWS/Lambda \ + --metric-name Invocations \ + --period 60 --statistics Sum \ + --start-time $(date -u -d '30 minutes ago' +%Y-%m-%dT%H:%M:%SZ) \ + --end-time $(date -u +%Y-%m-%dT%H:%M:%SZ) \ + --dimensions Name=FunctionName,Value= + +# CloudFront: request rate +aws cloudwatch get-metric-statistics \ + --namespace AWS/CloudFront \ + --metric-name Requests \ + --period 60 --statistics Sum \ + --start-time $(date -u -d '30 minutes ago' +%Y-%m-%dT%H:%M:%SZ) \ + --end-time $(date -u +%Y-%m-%dT%H:%M:%SZ) +``` + +Differentiate DDoS from legitimate traffic spike: + +- DDoS: requests from single/few IPs, unusual user-agents, no session cookies, hitting non-existent endpoints +- Legitimate spike: distributed IPs, real user agents, normal endpoint distribution + +--- + +## Step 2: Immediate Mitigation + +### Option A: WAF Rate Limiting (if WAF is enabled) + +```bash +# Add a rate-based rule blocking IPs with >1000 req/5min +aws wafv2 create-rule-group --scope CLOUDFRONT --name ddos-emergency \ + --capacity 100 --visibility-config ... +``` + +### Option B: Block IPs at Security Group Level + +```bash +# Block specific attacker IP ranges +aws ec2 authorize-security-group-ingress \ + --group-id \ + --protocol tcp --port 443 \ + --cidr \ + --description "DDoS block $(date +%Y-%m-%d)" +``` + +Wait — security groups are ALLOW lists, not deny lists. Use NACLs to block: + +```bash +# Block at NACL level (evaluated before security groups) +aws ec2 create-network-acl-entry \ + --network-acl-id \ + --rule-number 100 \ + --protocol -1 \ + --rule-action deny \ + --ingress \ + --cidr-block +``` + +### Option C: Lambda Throttling (reduce blast radius) + +```bash +# Set reserved concurrency to limit Lambda scale +aws lambda put-function-concurrency \ + --function-name \ + --reserved-concurrent-executions 50 +``` + +### Option D: AWS Shield Advanced + +If attacks are sustained and large-scale, engage AWS Shield Advanced: + +1. Go to AWS Shield console +2. Enable Shield Advanced (if not already enabled) +3. Contact AWS DDoS Response Team (DRT): available 24/7 to Shield Advanced customers + +--- + +## Step 3: Monitor and Adjust + +```bash +# Watch CloudFront 5xx error rate +watch -n 10 aws cloudwatch get-metric-statistics \ + --namespace AWS/CloudFront --metric-name 5xxErrorRate \ + --period 60 --statistics Average \ + --start-time $(date -u -d '5 minutes ago' +%Y-%m-%dT%H:%M:%SZ) \ + --end-time $(date -u +%Y-%m-%dT%H:%M:%SZ) +``` + +--- + +## Step 4: Post-Mitigation + +- [ ] Document attacker characteristics (IPs, ASNs, user-agents, request patterns) +- [ ] Enable WAF permanently with rate-based rules (see NET-001 recommendation) +- [ ] Enable AWS Shield Standard (free, always-on) at minimum for CloudFront distributions +- [ ] Consider Shield Advanced for production workloads +- [ ] Set up CloudWatch alarm for request rate spikes (>10x baseline) +- [ ] Remove temporary NACL/SG blocks once attack subsides diff --git a/.github/scripts/server-json-validator/package-lock.json b/.github/scripts/server-json-validator/package-lock.json new file mode 100644 index 000000000..3b345d852 --- /dev/null +++ b/.github/scripts/server-json-validator/package-lock.json @@ -0,0 +1,84 @@ +{ + "name": "cudly-mcp-server-json-validator", + "lockfileVersion": 3, + "requires": true, + "packages": { + "": { + "name": "cudly-mcp-server-json-validator", + "dependencies": { + "ajv": "8.20.0", + "ajv-formats": "3.0.1" + } + }, + "node_modules/ajv": { + "version": "8.20.0", + "resolved": "https://registry.npmjs.org/ajv/-/ajv-8.20.0.tgz", + "integrity": "sha512-Thbli+OlOj+iMPYFBVBfJ3OmCAnaSyNn4M1vz9T6Gka5Jt9ba/HIR56joy65tY6kx/FCF5VXNB819Y7/GUrBGA==", + "license": "MIT", + "dependencies": { + "fast-deep-equal": "^3.1.3", + "fast-uri": "^3.0.1", + "json-schema-traverse": "^1.0.0", + "require-from-string": "^2.0.2" + }, + "funding": { + "type": "github", + "url": "https://github.com/sponsors/epoberezkin" + } + }, + "node_modules/ajv-formats": { + "version": "3.0.1", + "resolved": "https://registry.npmjs.org/ajv-formats/-/ajv-formats-3.0.1.tgz", + "integrity": "sha512-8iUql50EUR+uUcdRQ3HDqa6EVyo3docL8g5WJ3FNcWmu62IbkGUue/pEyLBW8VGKKucTPgqeks4fIU1DA4yowQ==", + "license": "MIT", + "dependencies": { + "ajv": "^8.0.0" + }, + "peerDependencies": { + "ajv": "^8.0.0" + }, + "peerDependenciesMeta": { + "ajv": { + "optional": true + } + } + }, + "node_modules/fast-deep-equal": { + "version": "3.1.3", + "resolved": "https://registry.npmjs.org/fast-deep-equal/-/fast-deep-equal-3.1.3.tgz", + "integrity": "sha512-f3qQ9oQy9j2AhBe/H9VC91wLmKBCCU/gDOnKNAYG5hswO7BLKj09Hc5HYNz9cGI++xlpDCIgDaitVs03ATR84Q==", + "license": "MIT" + }, + "node_modules/fast-uri": { + "version": "3.1.6", + "resolved": "https://registry.npmjs.org/fast-uri/-/fast-uri-3.1.6.tgz", + "integrity": "sha512-7Ical1vFEMr0onbVzEDIreM22I4khW+fzyQPwvAFWBp1iwdshSZRsL4jjRvPG9JP1uiqMHRto+YU6R2/CzDz5Q==", + "funding": [ + { + "type": "github", + "url": "https://github.com/sponsors/fastify" + }, + { + "type": "opencollective", + "url": "https://opencollective.com/fastify" + } + ], + "license": "BSD-3-Clause" + }, + "node_modules/json-schema-traverse": { + "version": "1.0.0", + "resolved": "https://registry.npmjs.org/json-schema-traverse/-/json-schema-traverse-1.0.0.tgz", + "integrity": "sha512-NM8/P9n3XjXhIZn1lLhkFaACTOURQXjWhV4BA/RnOv8xvgqtqpAX9IO4mRQxSx1Rlo4tqzeqb0sOlruaOy3dug==", + "license": "MIT" + }, + "node_modules/require-from-string": { + "version": "2.0.2", + "resolved": "https://registry.npmjs.org/require-from-string/-/require-from-string-2.0.2.tgz", + "integrity": "sha512-Xf0nWe6RseziFMu+Ap9biiUbmplq6S9/p+7w7YXP/JBHhrUDDUhwa+vANyubuqfZWTveU//DYVGsDG7RKL/vEw==", + "license": "MIT", + "engines": { + "node": ">=0.10.0" + } + } + } +} diff --git a/.github/scripts/server-json-validator/package.json b/.github/scripts/server-json-validator/package.json new file mode 100644 index 000000000..5e759c0fa --- /dev/null +++ b/.github/scripts/server-json-validator/package.json @@ -0,0 +1,8 @@ +{ + "name": "cudly-mcp-server-json-validator", + "private": true, + "dependencies": { + "ajv": "8.20.0", + "ajv-formats": "3.0.1" + } +} diff --git a/.github/scripts/server-json-validator/validate.cjs b/.github/scripts/server-json-validator/validate.cjs new file mode 100644 index 000000000..3a182e6ce --- /dev/null +++ b/.github/scripts/server-json-validator/validate.cjs @@ -0,0 +1,22 @@ +const fs = require("fs"); +const Ajv = require("ajv"); +const addFormats = require("ajv-formats"); + +const [schemaPath, dataPath] = process.argv.slice(2); +if (!schemaPath || !dataPath) { + console.error("usage: node .github/scripts/server-json-validator/validate.cjs "); + process.exit(2); +} + +const schema = JSON.parse(fs.readFileSync(schemaPath, "utf8")); +const data = JSON.parse(fs.readFileSync(dataPath, "utf8")); +const ajv = new Ajv({ strict: false, allErrors: true }); +addFormats(ajv); + +const valid = ajv.validate(schema, data); +if (!valid) { + console.error(JSON.stringify(ajv.errors, null, 2)); + process.exit(1); +} + +console.log(`${dataPath} valid`); diff --git a/.github/workflows/README.md b/.github/workflows/README.md new file mode 100644 index 000000000..bf781e357 --- /dev/null +++ b/.github/workflows/README.md @@ -0,0 +1,686 @@ +# GitHub Actions Workflows + +This directory contains CI/CD workflows for the CUDly project, providing automated testing, deployment, and operations across AWS, GCP, and Azure. + +## 📋 Workflows Overview + +| Workflow | Purpose | Trigger | Duration | +|----------|---------|---------|----------| +| [ci.yml](#ci-workflow) | Continuous Integration | PR, Push to main | ~10 min | +| [deploy-aws-lambda.yml](#aws-lambda-deployment) | Deploy to AWS Lambda | Push to main, Manual | ~8 min | +| [deploy-aws-fargate.yml](#aws-fargate-deployment) | Deploy to AWS Fargate | Manual | ~10 min | +| [deploy-gcp.yml](#gcp-deployment) | Deploy to GCP Cloud Run | Manual | ~8 min | +| [deploy-azure.yml](#azure-deployment) | Deploy to Azure Container Apps | Manual | ~10 min | +| [deploy-all.yml](#multi-cloud-deployment) | Deploy to all clouds | Manual, Release | ~15 min | +| [database-migration.yml](#database-migrations) | Run DB migrations | Manual | ~5 min | +| [rollback.yml](#rollback) | Rollback deployment | Manual | ~5 min | + +> **Note:** Frontend deployment is handled automatically via Terraform as part of the backend deployment workflows. + +--- + +## CI Workflow + +**File:** `ci.yml` + +### Purpose + +Runs comprehensive quality checks on every pull request and push to main branch. + +### Jobs + +1. **Lint** - golangci-lint, go vet +2. **Unit Tests** - Go tests with race detection, coverage reporting +3. **Integration Tests** - Tests with real PostgreSQL +4. **Docker Build** - Build and test Docker image +5. **Terraform Validate** - Validate all Terraform configs (AWS, GCP, Azure) +6. **Security Scan** - gosec, trivy, tfsec +7. **Snyk Scan** - Dependency vulnerability scanning +8. **E2E Tests** - Docker Compose end-to-end tests +9. **Cost Estimate** - Infracost cost estimation (PR only) + +### Triggers + +- Pull requests to `main` or `develop` +- Pushes to `main` or `develop` +- Manual dispatch + +### Required Secrets + +- `SNYK_TOKEN` (optional - for Snyk scanning) +- `INFRACOST_API_KEY` (optional - for cost estimation) + +### Required Variables + +- `GO_VERSION` (default: 1.26.6) + +### Example + +```bash +# Automatically runs on PR +git push origin feature-branch + +# Or trigger manually +gh workflow run ci.yml +``` + +--- + +## AWS Lambda Deployment + +**File:** `deploy-aws-lambda.yml` + +### Purpose + +Deploy CUDly to AWS Lambda with Function URL. Serverless, event-driven platform. + +### Jobs + +1. **Prepare** - Determine environment and image tag +2. **Build & Push** - Build Docker image, push to ECR +3. **Deploy** - Deploy with Terraform +4. **Test** - Health check and smoke tests + +### Triggers + +- Push to `main` (deploys to dev) +- Release creation (deploys to prod) +- Manual dispatch with environment selection + +### Required Secrets + +- `AWS_ACCESS_KEY_ID` +- `AWS_SECRET_ACCESS_KEY` + +### Required Variables + +- `AWS_REGION` (default: us-east-1) +- `AWS_ACCOUNT_ID` +- `ECR_REPOSITORY` (default: cudly) + +### Example + +```bash +# Deploy to dev +gh workflow run deploy-aws-lambda.yml -f environment=dev + +# Push to main also deploys to dev +git push origin main + +# Deploy to prod +gh release create v1.0.0 +``` + +### Output + +- Function URL: `https://.lambda-url.us-east-1.on.aws` +- Deployment info artifact + +--- + +## AWS Fargate Deployment + +**File:** `deploy-aws-fargate.yml` + +### Purpose + +Deploy CUDly to AWS ECS Fargate with ALB. Always-on containerized platform. + +### Jobs + +1. **Build & Push** - Build Docker image, push to ECR +2. **Deploy** - Deploy with Terraform (Fargate mode) +3. **Test** - Health check verification + +### Triggers + +- Manual dispatch only + +### Required Secrets + +- Same as AWS Lambda + +### Example + +```bash +# Deploy to staging with Fargate +gh workflow run deploy-aws-fargate.yml -f environment=staging +``` + +--- + +## GCP Deployment + +**File:** `deploy-gcp.yml` + +### Purpose + +Deploy CUDly to GCP Cloud Run. Serverless container platform. + +### Jobs + +1. **Build & Deploy** - Build, push to Artifact Registry, deploy with Terraform +2. **Test** - Health check and smoke tests + +### Triggers + +- Manual dispatch +- Called by deploy-all.yml + +### Required Secrets + +- `GCP_SA_KEY` (Service Account JSON with permissions) +- `GCP_PROJECT_ID` + +### Required Variables + +- `GCP_REGION` (default: us-central1) +- `ARTIFACT_REGISTRY_REPO` (default: cudly) + +### Example + +```bash +# Deploy to GCP dev +gh workflow run deploy-gcp.yml -f environment=dev +``` + +### Output + +- Service URL: `https://cudly--uc.a.run.app` + +--- + +## Azure Deployment + +**File:** `deploy-azure.yml` + +### Purpose + +Deploy CUDly to Azure Container Apps. Serverless container platform with built-in HTTPS. + +### Jobs + +1. **Build & Deploy** - Build, push to ACR, deploy with Terraform +2. **Test** - Health check and smoke tests + +### Triggers + +- Manual dispatch +- Called by deploy-all.yml + +### Required Secrets + +- `AZURE_CREDENTIALS` (Service Principal JSON) +- `AZURE_SUBSCRIPTION_ID` + +### Required Variables + +- `AZURE_LOCATION` (default: eastus) +- `ACR_NAME` (default: cudlyacr) +- `RESOURCE_GROUP` (default: cudly-rg) + +### Example + +```bash +# Deploy to Azure staging +gh workflow run deploy-azure.yml -f environment=staging +``` + +### Output + +- App URL: `https://..azurecontainerapps.io` + +--- + +## Multi-Cloud Deployment + +**File:** `deploy-all.yml` + +### Purpose + +Orchestrate deployment to multiple cloud providers in parallel. + +### Jobs + +1. **Determine Strategy** - Choose which clouds to deploy to +2. **Deploy AWS Lambda** - Parallel deployment +3. **Deploy AWS Fargate** - Parallel deployment (optional) +4. **Deploy GCP** - Parallel deployment +5. **Deploy Azure** - Parallel deployment +6. **Notify** - Aggregate results + +### Triggers + +- Manual dispatch with provider selection +- Release creation (deploys to all clouds in prod) + +### Required Secrets + +- All secrets from individual deployment workflows + +### Deployment Options + +- `all` - Deploy to AWS, GCP, and Azure +- `aws-only` - AWS Lambda only +- `gcp-only` - GCP Cloud Run only +- `azure-only` - Azure Container Apps only +- `aws-gcp` - AWS and GCP +- `aws-azure` - AWS and Azure +- `gcp-azure` - GCP and Azure + +### Example + +```bash +# Deploy to all clouds (staging) +gh workflow run deploy-all.yml -f environment=staging -f deploy_to=all + +# Deploy to AWS and GCP (prod) +gh workflow run deploy-all.yml -f environment=prod -f deploy_to=aws-gcp + +# Automatic on release +gh release create v1.0.0 +``` + +### Benefits + +- **Disaster Recovery** - Multi-cloud redundancy +- **Cost Optimization** - Compare costs across providers +- **Testing** - Validate across all platforms +- **Global Reach** - Deploy to optimal regions per cloud + +--- + +## Database Migrations + +**File:** `database-migration.yml` + +### Purpose + +Apply or rollback database schema migrations across cloud providers. + +### Jobs + +1. **Validate** - Safety checks +2. **Migrate AWS** - Run golang-migrate on Aurora +3. **Migrate GCP** - Run golang-migrate on Cloud SQL +4. **Migrate Azure** - Run golang-migrate on Flexible Server + +### Triggers + +- Manual dispatch only (safety measure) +- Can be called by deployment workflows + +### Required Secrets + +- `DB_PASSWORD_AWS` +- `DB_PASSWORD_GCP` +- `DB_PASSWORD_AZURE` +- Cloud credentials (same as deployment workflows) + +### Required Variables + +- Database endpoints per environment + +### Migration Directions + +- `up` - Apply migrations (default) +- `down` - Rollback migrations (DANGEROUS) + +### Example + +```bash +# Apply all migrations to AWS dev +gh workflow run database-migration.yml \ + -f cloud=aws \ + -f environment=dev \ + -f direction=up + +# Rollback last 2 migrations on GCP staging +gh workflow run database-migration.yml \ + -f cloud=gcp \ + -f environment=staging \ + -f direction=down \ + -f steps=2 + +# Rollback 1 migration on AWS prod (requires typed confirmation) +gh workflow run database-migration.yml \ + -f cloud=aws \ + -f environment=prod \ + -f direction=down \ + -f steps=1 \ + -f confirm=rollback-prod + +# Apply to all clouds +gh workflow run database-migration.yml \ + -f cloud=all \ + -f environment=prod \ + -f direction=up +``` + +### Safety Features + +- **Validation** - Checks migration files exist before running +- **Explicit steps required** - `direction=down` requires an explicit positive `steps` value; `steps=0` (the default, which would run `down -all` and drop the entire schema) is rejected +- **Production confirmation** - `direction=down` on `environment=prod` additionally requires typing `rollback-prod` in the `confirm` input; omitting or mistyping it blocks the run +- **Defense in depth** - each migrate job independently re-validates the positive-steps constraint, so a future validate regression cannot reach `down -all` +- **Audit Trail** - Records all migrations in the step summary + +--- + +## Rollback + +**File:** `rollback.yml` + +### Purpose + +Quickly rollback to a previous deployment version by redeploying a known-good Docker image. + +### Jobs + +1. **Validate** - Validate image tag and construct image URI +2. **Rollback** - Confirm the image exists in the registry, then deploy it with Terraform +3. **Summary** - Create audit record + +Image existence is verified *inside* each rollback job rather than in a +standalone job. A separate verify job would have to assume the same cloud +deploy role while carrying no `environment:` binding, which is exactly the +ungated-but-credentialed shape that made the workflow exploitable. The +tradeoff is that a rollback to a nonexistent tag now fails after the +environment approval rather than before it. + +### Triggers + +- Manual dispatch only (safety measure) + +### Required Secrets + +- Cloud credentials (same as deployment workflows) + +### Example + +```bash +# Rollback AWS Lambda production to previous version +gh workflow run rollback.yml \ + -f cloud=aws-lambda \ + -f environment=prod \ + -f image_tag=sha-abc123 \ + -f reason="Critical bug in v1.2.3" + +# Rollback GCP staging +gh workflow run rollback.yml \ + -f cloud=gcp \ + -f environment=staging \ + -f image_tag=v1.2.2 +``` + +### Safety Features + +- **Image Verification** - Confirms image exists before deploying +- **Audit Trail** - Records all rollbacks (365 day retention) +- **Reason Tracking** - Requires reason for accountability +- **Manual Only** - Cannot be triggered automatically + +### Finding Image Tags + +```bash +# AWS ECR +aws ecr list-images --repository-name cudly + +# GCP Artifact Registry +gcloud artifacts docker images list -docker.pkg.dev///cudly + +# Azure ACR +az acr repository show-tags --name cudlyacr --repository cudly +``` + +--- + +## Setup Guide + +### 1. Configure GitHub Secrets + +**AWS:** + +```bash +# Create secrets +gh secret set AWS_ACCESS_KEY_ID +gh secret set AWS_SECRET_ACCESS_KEY +gh secret set DB_PASSWORD_AWS +``` + +**GCP:** + +```bash +# Create service account and download JSON +gcloud iam service-accounts create cudly-cicd --project= + +# Grant permissions +gcloud projects add-iam-policy-binding \ + --member="serviceAccount:cudly-cicd@.iam.gserviceaccount.com" \ + --role="roles/run.admin" + +# Create and download key +gcloud iam service-accounts keys create key.json \ + --iam-account=cudly-cicd@.iam.gserviceaccount.com + +# Set secrets +gh secret set GCP_SA_KEY < key.json +gh secret set GCP_PROJECT_ID -b"" +gh secret set DB_PASSWORD_GCP +``` + +**Azure:** + +```bash +# Create service principal +az ad sp create-for-rbac --name cudly-cicd --sdk-auth > azure-credentials.json + +# Set secrets +gh secret set AZURE_CREDENTIALS < azure-credentials.json +gh secret set AZURE_SUBSCRIPTION_ID -b"" +gh secret set DB_PASSWORD_AZURE +``` + +**Optional:** + +```bash +gh secret set SNYK_TOKEN +gh secret set INFRACOST_API_KEY +``` + +### 2. Configure GitHub Variables + +```bash +# AWS +gh variable set AWS_REGION -b"us-east-1" +gh variable set AWS_ACCOUNT_ID -b"123456789012" +gh variable set ECR_REPOSITORY -b"cudly" + +# GCP +gh variable set GCP_REGION -b"us-central1" +gh variable set ARTIFACT_REGISTRY_REPO -b"cudly" + +# Azure +gh variable set AZURE_LOCATION -b"eastus" +gh variable set ACR_NAME -b"cudlyacr" +gh variable set RESOURCE_GROUP -b"cudly-rg" + +# Frontend +gh variable set CLOUD_PROVIDER -b"aws" +gh variable set FRONTEND_BUCKET -b"cudly-frontend-prod" +gh variable set CLOUDFRONT_DISTRIBUTION_ID -b"E1234567890" +gh variable set API_URL -b"https://api.cudly.example.com" +``` + +### 3. Set Up Environments + +GitHub Environments provide deployment protection and environment-specific secrets: + +1. Go to **Settings** → **Environments** +2. Create environments: + - `aws-lambda-dev`, `aws-lambda-staging`, `aws-lambda-prod` + - `aws-fargate-dev`, `aws-fargate-staging`, `aws-fargate-prod` + - `gcp-dev`, `gcp-staging`, `gcp-prod` + - `azure-dev`, `azure-staging`, `azure-prod` + - `frontend-aws-dev`, etc. + +3. Configure protection rules: + - **Production**: Require approvals, restrict to main branch + - **Staging**: Optional approvals + - **Dev**: No restrictions + +--- + +## Troubleshooting + +### CI Workflow Fails + +**Unit tests fail:** + +```bash +# Run locally +make test-unit +``` + +**Integration tests fail:** + +```bash +# Run with testcontainers +make test-integration +``` + +**Security scan fails:** + +```bash +# Run locally +make security-scan-all +``` + +### Deployment Fails + +**AWS - Image not found:** + +```bash +# Check ECR +aws ecr describe-images --repository-name cudly --region us-east-1 + +# Re-push image +docker push .dkr.ecr.us-east-1.amazonaws.com/cudly:latest +``` + +**GCP - Permission denied:** + +```bash +# Check service account permissions +gcloud projects get-iam-policy + +# Grant missing roles +gcloud projects add-iam-policy-binding \ + --member="serviceAccount:@.iam.gserviceaccount.com" \ + --role="roles/run.admin" +``` + +**Azure - Resource not found:** + +```bash +# Verify resource group exists +az group show --name cudly-rg + +# Create if missing +az group create --name cudly-rg --location eastus +``` + +### Database Migration Fails + +**Connection timeout:** + +- Check database security groups/firewall rules +- Verify VPN/bastion access if required +- Check database is running + +**Migration already applied:** + +```bash +# Check current version +migrate -path migrations -database version + +# Force version (use with caution) +migrate -path migrations -database force +``` + +--- + +## Best Practices + +### 1. Branch Protection + +- Require CI to pass before merging +- Require code reviews +- Restrict direct pushes to main + +### 2. Environment Strategy + +- **Dev**: Auto-deploy on push to develop branch +- **Staging**: Auto-deploy on push to main +- **Prod**: Manual approval required, deploy on release + +### 3. Rollback Strategy + +- Keep last 10 images in each registry +- Document rollback procedures +- Test rollback in staging first + +### 4. Monitoring + +- Set up CloudWatch/Cloud Logging alerts +- Monitor deployment success rates +- Track deployment frequency + +### 5. Security + +- Rotate secrets regularly +- Use environment protection rules +- Enable secret scanning +- Review security scan results + +--- + +## Metrics & Monitoring + +### Workflow Success Rate + +```bash +# View recent workflow runs +gh run list --limit 50 + +# View specific workflow +gh run list --workflow=ci.yml --limit 20 +``` + +### Deployment Frequency + +- Target: Multiple deployments per day +- Track via GitHub Actions insights + +### Mean Time to Recovery (MTTR) + +- Use rollback workflow for quick recovery +- Target: < 15 minutes + +### CI Duration + +- Unit tests: ~5 min +- Integration tests: ~3 min +- Security scans: ~2 min +- Total: ~10 min target + +--- + +## Additional Resources + +- [GitHub Actions Documentation](https://docs.github.com/en/actions) +- [AWS ECR Documentation](https://docs.aws.amazon.com/ecr/) +- [GCP Artifact Registry](https://cloud.google.com/artifact-registry/docs) +- [Azure Container Registry](https://docs.microsoft.com/en-us/azure/container-registry/) +- [golang-migrate](https://github.com/golang-migrate/migrate) +- [Terraform Cloud](https://www.terraform.io/cloud) diff --git a/.github/workflows/aws_sanity.yml b/.github/workflows/aws_sanity.yml new file mode 100644 index 000000000..c9618918e --- /dev/null +++ b/.github/workflows/aws_sanity.yml @@ -0,0 +1,82 @@ +name: AWS Sanity (Read-only Dry Run) + +on: + pull_request: + push: + branches: ["main"] + workflow_dispatch: + +permissions: + contents: read + +jobs: + sanity: + runs-on: ubuntu-latest + permissions: + id-token: write + contents: read + env: + AWS_REGION: us-east-1 + REPORT_PATH: sanity_report.json + EXPECTED_ACCOUNT: ${{ secrets.AWS_EXPECTED_ACCOUNT_ID }} + + steps: + - name: Precheck secrets (skip if not configured) + id: precheck + shell: bash + env: + AWS_CICD_READONLY_ROLE_ARN: ${{ secrets.AWS_CICD_READONLY_ROLE_ARN }} + AWS_EXPECTED_ACCOUNT_ID: ${{ secrets.AWS_EXPECTED_ACCOUNT_ID }} + run: | + if [[ -z "${AWS_CICD_READONLY_ROLE_ARN}" || -z "${AWS_EXPECTED_ACCOUNT_ID}" ]]; then + echo "should_run=false" >> "$GITHUB_OUTPUT" + echo "Missing AWS secrets. Skipping AWS sanity." + else + echo "should_run=true" >> "$GITHUB_OUTPUT" + fi + + - name: Checkout + if: steps.precheck.outputs.should_run == 'true' + uses: actions/checkout@93cb6efe18208431cddfb8368fd83d5badbf9bfd # v5.0.1 + with: + persist-credentials: false + + - name: Setup Go + if: steps.precheck.outputs.should_run == 'true' + uses: actions/setup-go@4a3601121dd01d1626a1e23e37211e3254c1c06c # v6.4.0 + with: + go-version-file: go.mod + + - name: Configure AWS credentials via OIDC (read-only role) + if: steps.precheck.outputs.should_run == 'true' + uses: aws-actions/configure-aws-credentials@d979d5b3a71173a29b74b5b88418bfda9437d885 # v6.1.1 + with: + role-to-assume: ${{ secrets.AWS_CICD_READONLY_ROLE_ARN }} + aws-region: ${{ env.AWS_REGION }} + role-session-name: cudly-sanity + + - name: Build + if: steps.precheck.outputs.should_run == 'true' + run: | + go test ./ci_cd_sanity_tests/... -count=1 + go build -o sanity ./ci_cd_sanity_tests/cmd/sanity + + - name: Run sanity (read-only) + if: steps.precheck.outputs.should_run == 'true' + run: | + ./sanity \ + --region "${AWS_REGION}" \ + --expected-account "${EXPECTED_ACCOUNT}" \ + --out "${REPORT_PATH}" + + - name: Upload report artifact + if: always() && steps.precheck.outputs.should_run == 'true' + uses: actions/upload-artifact@b7c566a772e6b6bfb58ed0dc250532a479d7789f # v6.0.0 + with: + name: aws-sanity-report + path: ${{ env.REPORT_PATH }} + if-no-files-found: ignore + + - name: Skipped summary + if: steps.precheck.outputs.should_run != 'true' + run: echo "AWS sanity skipped because required secrets are not configured." diff --git a/.github/workflows/azure_sanity.yml b/.github/workflows/azure_sanity.yml new file mode 100644 index 000000000..8c91c1802 --- /dev/null +++ b/.github/workflows/azure_sanity.yml @@ -0,0 +1,81 @@ +name: Azure Sanity (Read-only Dry Run) + +on: + pull_request: + push: + branches: ["main"] + workflow_dispatch: + +permissions: + contents: read + +jobs: + sanity: + runs-on: ubuntu-latest + permissions: + id-token: write + contents: read + env: + REPORT_PATH: azure_sanity_report.json + + steps: + - name: Precheck secrets (skip if not configured) + id: precheck + shell: bash + env: + AZURE_CLIENT_ID: ${{ secrets.AZURE_CLIENT_ID }} + AZURE_TENANT_ID: ${{ secrets.AZURE_TENANT_ID }} + AZURE_SUBSCRIPTION_ID: ${{ secrets.AZURE_SUBSCRIPTION_ID }} + run: | + if [[ -z "${AZURE_CLIENT_ID}" || -z "${AZURE_TENANT_ID}" || -z "${AZURE_SUBSCRIPTION_ID}" ]]; then + echo "should_run=false" >> "$GITHUB_OUTPUT" + echo "Missing Azure secrets. Skipping Azure sanity." + else + echo "should_run=true" >> "$GITHUB_OUTPUT" + fi + + - name: Checkout + if: steps.precheck.outputs.should_run == 'true' + uses: actions/checkout@93cb6efe18208431cddfb8368fd83d5badbf9bfd # v5.0.1 + with: + persist-credentials: false + + - name: Setup Go + if: steps.precheck.outputs.should_run == 'true' + uses: actions/setup-go@4a3601121dd01d1626a1e23e37211e3254c1c06c # v6.4.0 + with: + go-version-file: go.mod + + - name: Azure Login (OIDC) + if: steps.precheck.outputs.should_run == 'true' + uses: azure/login@532459ea530d8321f2fb9bb10d1e0bcf23869a43 # v3.0.0 + with: + client-id: ${{ secrets.AZURE_CLIENT_ID }} + tenant-id: ${{ secrets.AZURE_TENANT_ID }} + subscription-id: ${{ secrets.AZURE_SUBSCRIPTION_ID }} + + - name: Build + Run Azure sanity + if: steps.precheck.outputs.should_run == 'true' + env: + AZURE_SUBSCRIPTION_ID: ${{ secrets.AZURE_SUBSCRIPTION_ID }} + AZURE_TENANT_ID: ${{ secrets.AZURE_TENANT_ID }} + run: | + go test ./ci_cd_sanity_tests/... -count=1 + go build -o azure-sanity ./ci_cd_sanity_tests/cmd/azure_sanity + ./azure-sanity \ + --subscription-id "${AZURE_SUBSCRIPTION_ID}" \ + --expected-subscription "${AZURE_SUBSCRIPTION_ID}" \ + --expected-tenant "${AZURE_TENANT_ID}" \ + --out "${REPORT_PATH}" + + - name: Upload report artifact + if: always() && steps.precheck.outputs.should_run == 'true' + uses: actions/upload-artifact@b7c566a772e6b6bfb58ed0dc250532a479d7789f # v6.0.0 + with: + name: azure-sanity-report + path: ${{ env.REPORT_PATH }} + if-no-files-found: ignore + + - name: Skipped summary + if: steps.precheck.outputs.should_run != 'true' + run: echo "Azure sanity skipped because required secrets are not configured." diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml new file mode 100644 index 000000000..ccfb79127 --- /dev/null +++ b/.github/workflows/ci.yml @@ -0,0 +1,1154 @@ +# CI Workflow - Build, Test, Lint, Security +# +# This workflow runs on every pull request and push to main/develop branches. +# It performs comprehensive quality checks before allowing code to be merged. +# +# Required GitHub Secrets: None (all checks run without cloud credentials) +# +# Required GitHub Variables: +# - GO_VERSION: Go version to use (default: 1.26.6) +# +# Triggered by: +# - Pull requests to main/develop +# - Pushes to main/develop +# - Manual workflow dispatch + +name: CI - Build & Test + +permissions: + contents: read + +on: + pull_request: + branches: [main, develop] + push: + branches: [main, develop] + workflow_dispatch: + +env: + GO_VERSION: '1.26.6' + DOCKER_BUILDKIT: 1 + COMPOSE_DOCKER_CLI_BUILD: 1 + +jobs: + # Go code linting + lint: + name: Lint Code + runs-on: ubuntu-latest + permissions: + contents: read + steps: + - name: Checkout code + uses: actions/checkout@93cb6efe18208431cddfb8368fd83d5badbf9bfd # v5.0.1 + with: + persist-credentials: false + + - name: Set up Go + uses: actions/setup-go@4a3601121dd01d1626a1e23e37211e3254c1c06c # v6.4.0 + with: + go-version: ${{ env.GO_VERSION }} + cache: true + + - name: Run golangci-lint + uses: golangci/golangci-lint-action@82606bf257cbaff209d206a39f5134f0cfbfd2ee # v9.2.1 + with: + version: v2.10.1 + args: --timeout=10m + + - name: Run go vet + run: go vet ./... + + - name: Install gocyclo + run: go install github.com/fzipp/gocyclo/cmd/gocyclo@v0.6.0 + + - name: Check cyclomatic complexity + run: | + echo "Checking for functions with cyclomatic complexity over 10..." + # Exclude _test.go files to stay consistent with .golangci.yml, which + # excludes gocyclo on test files (test helpers/table-driven tests are + # allowed higher complexity). Production code is still gated at >10. + COMPLEXITY_ISSUES=$(gocyclo -over 10 -ignore "_test\.go" . 2>&1 || true) + if [ -n "$COMPLEXITY_ISSUES" ]; then + echo "❌ Found functions with cyclomatic complexity over 10:" + echo "$COMPLEXITY_ISSUES" + echo "" + echo "⚠️ Please refactor these functions to reduce complexity." + echo "📖 Tip: Extract helper functions, use early returns, or simplify logic." + exit 1 + fi + echo "✅ All functions have acceptable cyclomatic complexity (≤10)" + + # GitHub Actions workflow linting + # + # Nothing else in this repo reads .github/workflows/ for defects. govulncheck + # and gosec are Go source scanners, trivy-config targets Terraform/Dockerfile/ + # Kubernetes, and check-yaml only proves the YAML parses. That gap is why the + # rollback.yml and deploy-aws-lambda.yml expression injections (#1542, #1649), + # both reaching production cloud credentials, passed every CI run. + # + # Both linters are needed and neither substitutes for the other: + # - actionlint catches workflow-level defects and, via shellcheck, shell + # bugs inside run: blocks. + # - zizmor has a template-injection audit that names the injection itself. + # Measured against the pre-fix rollback.yml, actionlint exited 1 only on + # unrelated SC2086 noise and never flagged the injected heredoc at all; + # zizmor flagged that exact line high severity, high confidence. actionlint + # alone would not have caught the bug this job exists to prevent. + workflow-lint: + name: Lint Workflows + runs-on: ubuntu-latest + permissions: + contents: read + env: + # Pinned by digest, not only by tag. A tag is mutable, and a linter whose + # ruleset changes without a change in this repo turns main red on its own + # schedule -- the hadolint :latest failure in #1695. + ACTIONLINT_IMAGE: 'rhysd/actionlint:1.7.12@sha256:b1934ee5f1c509618f2508e6eb47ee0d3520686341fec936f3b79331f9315667' + ZIZMOR_VERSION: '1.29.0' + steps: + - name: Checkout code + uses: actions/checkout@93cb6efe18208431cddfb8368fd83d5badbf9bfd # v5.0.1 + with: + persist-credentials: false + + - name: Assert there are workflows to lint + run: | + set -euo pipefail + count=$(find .github/workflows -maxdepth 1 -type f \( -name '*.yml' -o -name '*.yaml' \) | wc -l | tr -d ' ') + # Defence in depth, not the only guard: both linters do exit 3 on an + # empty input set. This catches the case they cannot, where the path + # still resolves but the set silently shrinks, and it prints the count + # so a drop is visible in the log rather than inferred from silence. + if [ "$count" -eq 0 ]; then + echo "::error::no workflow files found under .github/workflows" + exit 1 + fi + echo "Linting $count workflow files" + + - name: Assert shellcheck is available to actionlint + # actionlint does not fail when shellcheck is missing from PATH: it + # silently skips every run: block and still exits 0. That silent skip is + # the failure mode this job exists to close, so assert the binary is + # present rather than trusting the image to keep bundling it. + run: | + set -euo pipefail + docker run --rm --entrypoint sh "$ACTIONLINT_IMAGE" -c ' + command -v shellcheck >/dev/null || { + echo "::error::shellcheck is not present in the actionlint image; shell linting would be silently skipped" + exit 1 + } + shellcheck --version | sed -n "1,3p"' + + - name: Run actionlint + # No file arguments: actionlint discovers .github/workflows itself, so + # it also covers .yaml files and any workflow added later. Passing an + # explicit *.yml glob would silently skip a .yaml workflow that the + # count step above still counts. + run: | + set -euo pipefail + docker run --rm -v "$PWD:/repo" -w /repo "$ACTIONLINT_IMAGE" -color + + - name: Run zizmor + # --offline on purpose: the online audits query the GitHub API for action + # metadata, so findings could change without a change in this repo and + # redden main, the same class of failure the hadolint digest pin fixed. + # + # Coverage note: zizmor reads the directory non-recursively, while + # actionlint walks it. A workflow under .github/workflows/sub/ would + # therefore reach actionlint but not zizmor. GitHub itself ignores + # workflows in subdirectories, so this is not a live hole, and the + # -maxdepth 1 count above fails loud if the set ever moves down a level. + # + # Two independent filters, and it matters which does what. + # + # --persona=pedantic rather than the default regular: regular hides + # three high-severity findings this repo actually had (workflow-level + # id-token: write in both sanity workflows, and an unpinned postgres + # service image). Those are fixed rather than filtered, so the stricter + # persona costs nothing today and gates more. auditor is the one level + # up and is documented as accepting false positives, so it is not used. + # + # --min-severity=medium is a threshold, not a suppression: no baseline + # file, no per-finding ignore, no only-new-issues. Every medium and + # high finding the pedantic persona surfaces fails this job, and the + # injection class this job exists to catch scores high. Below the line + # sit 106 findings, none above low: 66 template-injection on values + # #1649 already assessed as non-injectable (github.actor, github.sha + # and similar), plus undocumented-permissions, concurrency-limits and + # anonymous-definition. Clearing those means rewriting the deploy + # workflows, so they are left to a follow-up rather than silenced here. + run: | + set -euo pipefail + # `pipx run --spec` rather than `pipx install`: it pins the version in + # the same statement that invokes it and does not assume pipx's bin + # directory is on PATH. + pipx run --spec "zizmor==${ZIZMOR_VERSION}" zizmor \ + --offline --persona=pedantic --min-severity=medium \ + --color=always .github/workflows/ + + # Unit tests with race detection + unit-tests: + name: Unit Tests + runs-on: ubuntu-latest + permissions: + contents: read + steps: + - name: Checkout code + uses: actions/checkout@93cb6efe18208431cddfb8368fd83d5badbf9bfd # v5.0.1 + with: + persist-credentials: false + + - name: Set up Go + uses: actions/setup-go@4a3601121dd01d1626a1e23e37211e3254c1c06c # v6.4.0 + with: + go-version: ${{ env.GO_VERSION }} + cache: true + + - name: Download dependencies + run: | + # Multi-module repo: `go mod download` resolves the current module + # only, so running it at the root left pkg/ and providers/* unverified + # (issue #1751). Mirror the govulncheck per-module loop below. + set -e + for mod in . pkg providers/aws providers/azure providers/gcp tests/e2e; do + echo "==> go mod download/verify in $mod" + (cd "$mod" && go mod download && go mod verify) + done + + - name: Run unit tests (all modules) + run: | + # Multi-module repo: each ./... only walks the current module, so + # running from the root alone never compiled or asserted pkg/, + # providers/* or tests/e2e -- roughly 1600 test functions that had + # never gated a merge (issue #1751). Mirror the govulncheck and gosec + # per-module loops, collect one coverage profile per module, then + # merge for the upload steps below. + # + # A failing module must not abort the loop. Aborting would hide every + # later module's result behind the first failure, which is the same + # shape as the gosec bug in issue #1717. Guard each run, record the + # status, and fail at the end so every module is reported. + set -uo pipefail + status=0 + for mod in . pkg providers/aws providers/azure providers/gcp tests/e2e; do + tag=$(echo "$mod" | tr './' '--' | sed 's/^-/root/') + log="$RUNNER_TEMP/unit-${tag}.log" + # tests/e2e holds only //go:build e2e files, so `go test ./...` + # there matches no packages and exits 1 -- it cannot be run like the + # others. Its tests need the running stack and are executed by the + # e2e-tests job over docker compose; what this job can add is a + # type-check under that tag, which nothing else here does. Handled + # by name, not by a pattern, so the difference is visible in review. + if [ "$mod" = "tests/e2e" ]; then + echo "==> type-check $mod under -tags=e2e (tests run in the e2e-tests job)" + if ! (cd "$mod" && go vet -tags=e2e ./...) 2>&1 | tee "$log"; then + echo "::error::$mod failed to type-check under -tags=e2e" + status=1 + fi + continue + fi + echo "==> unit tests in $mod" + # Profiles go to $RUNNER_TEMP, not the checkout: never leave working + # files in the repository root (repo coding guideline). + if ! (cd "$mod" && go test -v -race -short \ + -coverprofile="$RUNNER_TEMP/coverage-${tag}.out" \ + -covermode=atomic ./...) 2>&1 | tee "$log"; then + echo "::error::unit tests failed in $mod" + status=1 + fi + # A module that runs zero tests is issue #1751 wearing a new + # costume: the loop would "cover" it while asserting nothing. Count + # top-level results (subtest lines are indented) and fail loudly. + ran=$(grep -cE '^--- (PASS|FAIL|SKIP)' "$log" || true) + echo "==> $mod ran $ran top-level test(s)" + if [ "$ran" -eq 0 ]; then + echo "::error::$mod ran zero tests -- it is in the loop but asserting nothing" + status=1 + fi + done + + # Merge the per-module profiles into the single file the steps below + # expect. A Go profile is one "mode:" header followed by block lines, + # so keep one header and concatenate the bodies. + shopt -s nullglob + profiles=("$RUNNER_TEMP"/coverage-*.out) + if [ "${#profiles[@]}" -eq 0 ]; then + echo "::error::no module produced a coverage profile" >&2 + exit 1 + fi + # Take the mode header from the first profile rather than hardcoding + # it: go test picks the default covermode itself (atomic whenever + # -race is on), so a literal here would silently mislabel the merge if + # the flags change. + merged="$RUNNER_TEMP/coverage.out" + head -n 1 "${profiles[0]}" > "$merged" + for p in "${profiles[@]}"; do + tail -n +2 "$p" >> "$merged" + done + echo "Merged ${#profiles[@]} coverage profile(s), $(( $(wc -l < "$merged") - 1 )) block(s)" + exit "$status" + + - name: Upload coverage to Codecov + uses: codecov/codecov-action@e79a6962e0d4c0c17b229090214935d2e33f8354 # v6.0.1 + with: + files: ${{ runner.temp }}/coverage.out + flags: unittests + name: codecov-umbrella + fail_ci_if_error: false + + - name: Generate coverage report + run: | + go tool cover -html="$RUNNER_TEMP/coverage.out" -o "$RUNNER_TEMP/coverage.html" + + - name: Upload coverage artifacts + uses: actions/upload-artifact@b7c566a772e6b6bfb58ed0dc250532a479d7789f # v6.0.0 + with: + name: coverage-report + path: | + ${{ runner.temp }}/coverage.out + ${{ runner.temp }}/coverage.html + retention-days: 30 + + - name: Check coverage threshold + run: | + coverage=$(go tool cover -func="$RUNNER_TEMP/coverage.out" | grep total | awk '{print $3}' | sed 's/%//') + echo "Total coverage: ${coverage}%" + if (( $(echo "$coverage < 80" | bc -l) )); then + echo "::warning::Coverage is below 80% (current: ${coverage}%)" + fi + + mcp-build: + name: Build MCP (${{ matrix.os }}) + runs-on: ${{ matrix.os }} + permissions: + contents: read + strategy: + fail-fast: false + matrix: + include: + - os: ubuntu-latest + binary_format: ELF + - os: macos-latest + binary_format: Mach-O + steps: + - name: Checkout code + uses: actions/checkout@93cb6efe18208431cddfb8368fd83d5badbf9bfd # v5.0.1 + with: + persist-credentials: false + + - name: Set up Go + uses: actions/setup-go@4a3601121dd01d1626a1e23e37211e3254c1c06c # v6.4.0 + with: + go-version: ${{ env.GO_VERSION }} + cache: true + + - name: Assert MCP artifact is absent + run: test ! -e bin/cudly-mcp + + - name: Build MCP + run: make build-mcp VERSION=v0.0.0-version-test + + - name: Assert host-native binary format + run: file bin/cudly-mcp | grep -F '${{ matrix.binary_format }}' + + - name: Verify Make-built MCP version + env: + CUDLY_MCP_TEST_BINARY: ${{ github.workspace }}/bin/cudly-mcp + run: go test ./cmd/cudly-mcp -run '^TestBuiltBinaryReportsInjectedVersion$' -count=1 + + - name: Assert clean target includes MCP artifact + run: make -n clean | grep -Fx 'rm -f cudly bootstrap bin/cudly-server bin/cudly-mcp' + + # Integration tests with real PostgreSQL + integration-tests: + name: Integration Tests + runs-on: ubuntu-latest + permissions: + contents: read + + services: + postgres: + # Pinned by digest for the same reason as the linter images below: a + # floating tag lets the service container change under an unchanged + # repo, which is how #1695 turned main red. + image: postgres:16-alpine@sha256:cf78e76683b9ca8c5733cbbdce6c9262b45b6767934dd0a95e671f9a0fc20685 + env: + POSTGRES_DB: cudly_test + POSTGRES_USER: cudly_test + POSTGRES_PASSWORD: test_password # CI-only throwaway password — not used in any real environment + options: >- + --health-cmd pg_isready + --health-interval 10s + --health-timeout 5s + --health-retries 5 + ports: + - 5432:5432 + + steps: + - name: Checkout code + uses: actions/checkout@93cb6efe18208431cddfb8368fd83d5badbf9bfd # v5.0.1 + with: + persist-credentials: false + + - name: Set up Go + uses: actions/setup-go@4a3601121dd01d1626a1e23e37211e3254c1c06c # v6.4.0 + with: + go-version: ${{ env.GO_VERSION }} + cache: true + + - name: Install golang-migrate + run: | + # -tags 'pgx5', matching the Dockerfile and the library import in + # internal/database/postgres/migrations: the postgres tag links + # lib/pq, which carries unfixable advisories (issue #1849). No + # @version suffix: installed as a package of this module, so the + # version and dependency set come from go.mod. + go install -tags 'pgx5' github.com/golang-migrate/migrate/v4/cmd/migrate + + - name: Run database migrations + env: + DB_HOST: localhost + DB_PORT: 5432 + DB_NAME: cudly_test + DB_USER: cudly_test + DB_PASSWORD: test_password # CI-only throwaway password — not used in any real environment + run: | + if [ -d "internal/database/postgres/migrations" ]; then + migrate -path internal/database/postgres/migrations \ + -database "pgx5://${DB_USER}:${DB_PASSWORD}@${DB_HOST}:${DB_PORT}/${DB_NAME}?sslmode=disable" \ + up + fi + + - name: Run integration tests + env: + DB_HOST: localhost + DB_PORT: 5432 + DB_NAME: cudly_test + DB_USER: cudly_test + DB_PASSWORD: test_password # CI-only throwaway password — not used in any real environment + DB_SSL_MODE: disable + run: | + # Same multi-module gap as the unit job (issue #1751): a bare ./... + # here only ever walked the root module. Mirror the same loop. + # + # Note that -tags=integration ADDS a build tag rather than selecting + # only tagged files, so each module runs its untagged tests too. Today + # every //go:build integration file lives in the root module, so for + # pkg/ and providers/* this run is a compile-under-tag check plus a + # re-run of their unit tests. That is the point: it is what makes an + # integration test added to those modules later actually gate. + set -uo pipefail + status=0 + for mod in . pkg providers/aws providers/azure providers/gcp tests/e2e; do + tag=$(echo "$mod" | tr './' '--' | sed 's/^-/root/') + log="$RUNNER_TEMP/integration-${tag}.log" + # See the unit job: tests/e2e has no package to test without its own + # tag, so it gets a type-check here too rather than a run. + if [ "$mod" = "tests/e2e" ]; then + echo "==> type-check $mod under -tags=e2e (tests run in the e2e-tests job)" + if ! (cd "$mod" && go vet -tags=e2e ./...) 2>&1 | tee "$log"; then + echo "::error::$mod failed to type-check under -tags=e2e" + status=1 + fi + continue + fi + echo "==> integration tests in $mod" + if ! (cd "$mod" && go test -v -race -tags=integration \ + -coverprofile="$RUNNER_TEMP/coverage-integration-${tag}.out" \ + ./...) 2>&1 | tee "$log"; then + echo "::error::integration tests failed in $mod" + status=1 + fi + ran=$(grep -cE '^--- (PASS|FAIL|SKIP)' "$log" || true) + echo "==> $mod ran $ran top-level test(s)" + if [ "$ran" -eq 0 ]; then + echo "::error::$mod ran zero tests -- it is in the loop but asserting nothing" + status=1 + fi + done + + # Merge per-module profiles for the upload step (see the unit job). + shopt -s nullglob + profiles=("$RUNNER_TEMP"/coverage-integration-*.out) + if [ "${#profiles[@]}" -eq 0 ]; then + echo "::error::no module produced an integration coverage profile" >&2 + exit 1 + fi + merged="$RUNNER_TEMP/coverage-integration.out" + head -n 1 "${profiles[0]}" > "$merged" + for p in "${profiles[@]}"; do + tail -n +2 "$p" >> "$merged" + done + echo "Merged ${#profiles[@]} integration coverage profile(s)" + exit "$status" + + - name: Upload integration coverage + uses: actions/upload-artifact@b7c566a772e6b6bfb58ed0dc250532a479d7789f # v6.0.0 + with: + name: integration-coverage + path: ${{ runner.temp }}/coverage-integration.out + retention-days: 30 + + # Docker image build test + docker-build: + name: Build Docker Image + runs-on: ubuntu-latest + permissions: + contents: read + steps: + - name: Checkout code + uses: actions/checkout@93cb6efe18208431cddfb8368fd83d5badbf9bfd # v5.0.1 + with: + persist-credentials: false + + - name: Set up Docker Buildx + uses: docker/setup-buildx-action@4d04d5d9486b7bd6fa91e7baf45bbb4f8b9deedd # v4.0.0 + + - name: Build Docker image + uses: docker/build-push-action@f9f3042f7e2789586610d6e8b85c8f03e5195baf # v7.2.0 + with: + context: . + push: false + # Export the built image into the local docker image store so the + # scan step below inspects the artifact this job just produced, + # rather than rebuilding one and hoping it is the same. + load: true + tags: cudly:${{ github.sha }} + cache-from: type=gha + cache-to: type=gha,mode=max + build-args: | + VERSION=${{ github.sha }} + + - name: Test Docker image + run: | + docker build -t cudly:test . + docker run --rm cudly:test /app/cudly --version || true + docker run --rm cudly:test /app/cudly --help || true + + # Nothing else in the pipeline looks at the artifact: govulncheck runs in + # source mode, and Trivy runs with scan-type fs and config, all against + # the repository. A vulnerable binary baked into the image was invisible + # by construction, which is how #1833 shipped twice (issue #1836). These + # steps live in this job rather than a new one so the image is scanned + # without being built a third time; docker-build is already in + # ci-success's needs, so the gate is wired. + - name: Set up Go + uses: actions/setup-go@4a3601121dd01d1626a1e23e37211e3254c1c06c # v6.4.0 + with: + go-version: ${{ env.GO_VERSION }} + + - name: Install govulncheck + run: | + # Same pin as the security-scan job: a govulncheck release with new + # detection logic must not silently change this gate's verdict + # between PRs. + go install golang.org/x/vuln/cmd/govulncheck@v1.1.4 + + - name: Run image scan self-tests + # Asserts both directions of the verdict against recorded govulncheck + # output, so a scanner that can no longer fail cannot ship as coverage. + run: bash scripts/test-scan-shipped-image.sh + + - name: Scan the shipped image for Go advisories + run: bash scripts/scan-shipped-image.sh cudly:${{ github.sha }} + + # Terraform validation for all environments + terraform-validate: + name: Validate Terraform (${{ matrix.cloud }}) + runs-on: ubuntu-latest + permissions: + contents: read + strategy: + matrix: + cloud: [aws, gcp, azure] + fail-fast: false + + steps: + - name: Checkout code + uses: actions/checkout@93cb6efe18208431cddfb8368fd83d5badbf9bfd # v5.0.1 + with: + persist-credentials: false + + - name: Setup Terraform + uses: hashicorp/setup-terraform@dfe3c3f87815947d99a8997f908cb6525fc44e9e # v4.0.1 + with: + # Must satisfy `required_version = ">= 1.10.0"` declared by every + # terraform/environments/*/main.tf -- pinning below 1.10 makes + # `terraform init`/`validate` abort with "Unsupported Terraform + # Core version". Matches the pin in pre-commit.yml so both + # workflows resolve to the same binary. + terraform_version: 1.10.5 + + - name: Terraform Format Check + run: terraform fmt -check -recursive terraform/ + + - name: Terraform Init + run: | + cd terraform/environments/${{ matrix.cloud }} + terraform init -backend=false + + - name: Terraform Validate + run: | + cd terraform/environments/${{ matrix.cloud }} + terraform validate + + - name: Validate environment tfvars files + run: | + cd terraform/environments/${{ matrix.cloud }} + for env in dev staging prod; do + # Check local tfvars if present + if [ -f "${env}.tfvars" ]; then + echo "Checking ${env}.tfvars syntax..." + terraform fmt -check "${env}.tfvars" || echo "Note: ${env}.tfvars may need formatting" + fi + # Check GitHub CI/CD tfvars + if [ -f "github-${env}.tfvars" ]; then + echo "Checking github-${env}.tfvars syntax..." + terraform fmt -check "github-${env}.tfvars" || echo "Note: github-${env}.tfvars may need formatting" + fi + done + + # Security scanning + security-scan: + name: Security Scanning + runs-on: ubuntu-latest + permissions: + security-events: write + contents: read + + steps: + - name: Checkout code + id: checkout + uses: actions/checkout@93cb6efe18208431cddfb8368fd83d5badbf9bfd # v5.0.1 + with: + persist-credentials: false + + - name: Set up Go + id: setup_go + uses: actions/setup-go@4a3601121dd01d1626a1e23e37211e3254c1c06c # v6.4.0 + with: + go-version: ${{ env.GO_VERSION }} + + - name: Run govulncheck CVE scanner (all modules) + # always(): each scanner step below is independent evidence for the + # Security tab. Without this, one scanner failing (e.g. a live npm + # advisory) skips every scanner after it in the same job, silently + # disabling Go SAST and Terraform IaC coverage repo-wide. The job + # still fails overall if any scanner step here fails. Still requires + # checkout and Go setup to have actually succeeded -- an infra + # failure there must not be papered over as "scanner found nothing". + if: always() && steps.checkout.outcome == 'success' && steps.setup_go.outcome == 'success' + run: | + # Pinned (not @latest): a govulncheck release with new + # detection logic could silently change the gate's verdict + # between PRs without an intentional bump in this repo. + # Bumping is a deliberate review item, not a drift. + go install golang.org/x/vuln/cmd/govulncheck@v1.1.4 + # Multi-module repo: each ./... only walks the current module, + # so scanning the root would silently miss pkg/ and providers/*. + # Walk every module independently and fail on any HIGH/CRITICAL. + set -e + for mod in . pkg providers/aws providers/azure providers/gcp tests/e2e; do + echo "==> govulncheck in $mod" + (cd "$mod" && govulncheck ./...) + done + + - name: Run npm audit (frontend) + # always(): must not skip the scanners below it just because + # govulncheck failed, and its own failure must not skip gosec/Trivy. + # Still requires checkout to have succeeded (see govulncheck above). + if: always() && steps.checkout.outcome == 'success' && steps.setup_go.outcome == 'success' + run: | + if [ -f frontend/package.json ]; then + cd frontend && npm audit --audit-level=high + fi + + - name: Run gosec Security Scanner + id: gosec + # always(): don't let an earlier scanner's failure (e.g. npm audit) + # skip Go SAST coverage. The job still fails if gosec itself fails. + # Still requires checkout and Go setup to have succeeded. + if: always() && steps.checkout.outcome == 'success' && steps.setup_go.outcome == 'success' + run: | + # Install pinned gosec using the job's existing setup-go. + # The securego/gosec Docker action bundles its own Go toolchain which + # cannot satisfy the module's go directive, causing a toolchain mismatch. + go install github.com/securego/gosec/v2/cmd/gosec@v2.28.0 + # Multi-module repo: each ./... only walks the current module so scanning root + # alone silently misses pkg/ and providers/*. Mirror the govulncheck per-module + # loop, collect per-module SARIF, then merge for the upload step. + # + # gosec exits non-zero both for real findings and for a processing + # error. A bare `set -e` loop aborts at the first non-zero module, + # skipping the merge below entirely -- the upload step then fails + # on a missing file instead of surfacing the actual finding + # (issue #1717). Guarding the gosec call in `if ! ( ... )` exempts + # it from errexit so every module still gets scanned and merged; + # `status` records whether the job should still fail at the end. + set -e + status=0 + for mod in . pkg providers/aws providers/azure providers/gcp tests/e2e; do + tag=$(echo "$mod" | tr './' '--' | sed 's/^-/root/') + out="$RUNNER_TEMP/gosec-${tag}.sarif" + echo "==> gosec in $mod" + if ! (cd "$mod" && gosec -fmt sarif -out "$out" ./...); then + echo "::warning::gosec exited non-zero in $mod (findings or a scan error)" + status=1 + fi + if [ ! -f "$out" ]; then + echo "::error::gosec produced no SARIF output for $mod" + status=1 + continue + fi + # Code scanning rejects a SARIF file whose runs share a category + # (github.blog changelog 2025-07-21), so give each module's run a + # unique automationDetails.id before merging. + jq --arg id "gosec-${tag}/" '.runs |= map(.automationDetails = {id: $id})' \ + "$out" > "$out.tmp" && mv "$out.tmp" "$out" + done + # Merge per-module SARIF runs into one file for the upload step. + # Only modules that actually produced a file are included; if + # gosec crashed before writing any of them, fail loud here rather + # than uploading an empty result silently. + shopt -s nullglob + sarif_files=("$RUNNER_TEMP"/gosec-*.sarif) + if [ "${#sarif_files[@]}" -eq 0 ]; then + echo "::error::no gosec SARIF output was produced by any module" >&2 + exit 1 + fi + # jq is preinstalled on the GitHub Ubuntu runner image (no new deps). + # Written to $RUNNER_TEMP, not the checkout root: never save working + # files in the repository root (repo coding guideline), and this + # file is purely a hand-off to the upload step below. + merged="$RUNNER_TEMP/gosec-results.sarif" + jq -s '{version: "2.1.0", "$schema": "https://json.schemastore.org/sarif-2.1.0.json", runs: [.[].runs[]]}' \ + "${sarif_files[@]}" > "$merged" + echo "Merged $(jq '.runs | length' "$merged") SARIF runs" + exit "$status" + + - name: Upload gosec results to GitHub Security + # Tolerate a missing SARIF only when the gosec step itself never ran + # (e.g. checkout/setup-go failed, so gosec's own `if:` above skipped + # it). If gosec ran and the file is still missing, fail loud instead + # of masking it as a skip. + if: always() && steps.checkout.outcome == 'success' && steps.setup_go.outcome == 'success' && steps.gosec.outcome != 'skipped' + uses: github/codeql-action/upload-sarif@7211b7c8077ea37d8641b6271f6a365a22a5fbfa # v4.36.0 + with: + sarif_file: ${{ runner.temp }}/gosec-results.sarif + + - name: Run Trivy vulnerability scanner (filesystem) + id: trivy_fs + # always(): don't let an earlier scanner's failure skip Trivy fs + # coverage. This scan uses the default exit-code 0 (see the IaC + # scan comment below), so it does not gate the job on its own. + # Still requires checkout and Go setup to have succeeded. + if: always() && steps.checkout.outcome == 'success' && steps.setup_go.outcome == 'success' + uses: aquasecurity/trivy-action@ed142fd0673e97e23eac54620cfb913e5ce36c25 # v0.36.0 + with: + scan-type: 'fs' + scan-ref: '.' + format: 'sarif' + # $RUNNER_TEMP, not the checkout root (repo coding guideline: never + # save working files in the repository root). + output: '${{ runner.temp }}/trivy-results.sarif' + severity: 'CRITICAL,HIGH' + # Pin the Trivy binary independently of the action SHA. v0.36.0 is + # the latest trivy-action release but bundles Trivy v0.70.0, which + # panics in adaptDefaultTags on terraform/environments/aws/main.tf + # (null default_tags vars). Fixed in Trivy >= v0.72.0. The action + # forwards this input to aquasecurity/setup-trivy, so both scan + # steps run the same pinned binary. Keep both steps on this version. + version: 'v0.72.0' + + - name: Upload Trivy results to GitHub Security + # Tolerate a missing SARIF only when the Trivy fs step never ran. + if: always() && steps.checkout.outcome == 'success' && steps.setup_go.outcome == 'success' && steps.trivy_fs.outcome != 'skipped' + uses: github/codeql-action/upload-sarif@7211b7c8077ea37d8641b6271f6a365a22a5fbfa # v4.36.0 + with: + sarif_file: '${{ runner.temp }}/trivy-results.sarif' + + # Terraform IaC misconfiguration scanning. Replaces the deprecated + # aquasecurity/tfsec-action, whose bundled HCL parser rejects Terraform + # 1.5+ `check {}` blocks (e.g. terraform/modules/deployment-checks/main.tf) + # with a hard "scan failed" parse error that soft_fail does not suppress + # (soft_fail only downgrades findings, not scan errors). Trivy is tfsec's + # official successor, parses `check {}` blocks, and (like the fs scan + # above) uses the default exit-code 0 so misconfig findings are reported + # to the Security tab without gating the job -- matching tfsec's prior + # soft_fail: true behaviour while preserving Terraform IaC coverage. + - name: Run Trivy IaC misconfiguration scanner (Terraform) + id: trivy_iac + # always(): don't let an earlier scanner's failure skip Terraform + # IaC coverage. Still requires checkout and Go setup to have + # succeeded. + if: always() && steps.checkout.outcome == 'success' && steps.setup_go.outcome == 'success' + uses: aquasecurity/trivy-action@ed142fd0673e97e23eac54620cfb913e5ce36c25 # v0.36.0 + with: + scan-type: 'config' + scan-ref: 'terraform/' + format: 'sarif' + # $RUNNER_TEMP, not the checkout root (repo coding guideline: never + # save working files in the repository root). + output: '${{ runner.temp }}/trivy-config-results.sarif' + severity: 'CRITICAL,HIGH' + # Same pinned Trivy binary as the filesystem scan above (>= v0.72.0 + # avoids the adaptDefaultTags panic on null default_tags vars). + version: 'v0.72.0' + + - name: Upload Trivy IaC results to GitHub Security + # Tolerate a missing SARIF only when the Trivy IaC step never ran. + if: always() && steps.checkout.outcome == 'success' && steps.setup_go.outcome == 'success' && steps.trivy_iac.outcome != 'skipped' + uses: github/codeql-action/upload-sarif@7211b7c8077ea37d8641b6271f6a365a22a5fbfa # v4.36.0 + with: + sarif_file: '${{ runner.temp }}/trivy-config-results.sarif' + # Distinct category so this IaC analysis does not overwrite the + # filesystem Trivy analysis uploaded above (both report as "Trivy"). + category: 'trivy-iac' + + # Snyk security scanning + snyk-scan: + name: Snyk Security Scan + runs-on: ubuntu-latest + if: github.event_name != 'pull_request' || github.event.pull_request.head.repo.full_name == github.repository + permissions: + contents: read + + steps: + - name: Checkout code + uses: actions/checkout@93cb6efe18208431cddfb8368fd83d5badbf9bfd # v5.0.1 + with: + persist-credentials: false + + - name: Set up Go + uses: actions/setup-go@4a3601121dd01d1626a1e23e37211e3254c1c06c # v6.4.0 + with: + go-version: ${{ env.GO_VERSION }} + + - name: Run Snyk to check for vulnerabilities + uses: snyk/actions/golang@b98d498629f1c368650224d6d212bf7dfa89e4bf # 0.4.0 + continue-on-error: true + env: + SNYK_TOKEN: ${{ secrets.SNYK_TOKEN }} + with: + args: --severity-threshold=high + + # Docker Compose E2E tests: build the app + test-runner images, bring up + # postgres + cudly-app, then run the black-box suite in tests/e2e against + # the app over HTTP. The test-runner service is profile-gated, so every + # compose invocation needs --profile test (issue #1180). + e2e-tests: + name: E2E Tests + runs-on: ubuntu-latest + permissions: + contents: read + + steps: + - name: Checkout code + uses: actions/checkout@93cb6efe18208431cddfb8368fd83d5badbf9bfd # v5.0.1 + with: + persist-credentials: false + + - name: Set up Docker Buildx + uses: docker/setup-buildx-action@4d04d5d9486b7bd6fa91e7baf45bbb4f8b9deedd # v4.0.0 + + - name: Build images + run: | + docker compose -f docker-compose.test.yml --profile test build + + - name: Run E2E tests with docker compose + run: | + docker compose -f docker-compose.test.yml --profile test up --abort-on-container-exit --exit-code-from test-runner + env: + COMPOSE_INTERACTIVE_NO_CLI: 1 + + - name: Cleanup + if: always() + run: | + docker compose -f docker-compose.test.yml --profile test down -v + + # Assert that the Azure custom-role actions list is identical in the TF module + # and the ARM onboarding template. Fast (shell + jq only), so it always runs. + # Path changes that trigger drift will be caught regardless of PR context. + azure-role-parity: + name: Azure role actions parity (ARM vs TF) + runs-on: ubuntu-latest + permissions: + contents: read + + steps: + - name: Checkout code + uses: actions/checkout@93cb6efe18208431cddfb8368fd83d5badbf9bfd # v5.0.1 + with: + persist-credentials: false + + - name: Assert ARM/TF actions parity + run: bash scripts/check-azure-role-parity.sh + + - name: Run parity script self-tests + run: bash scripts/test-azure-role-parity.sh + + # Assert that the AWS IAM action lists are identical across the CFN stack, + # the TF lambda/fargate modules, and the federation CFN/TF/CLI templates. + # Fast (shell only), so it always runs. + aws-iam-parity: + name: AWS IAM actions parity (CFN vs TF) + runs-on: ubuntu-latest + permissions: + contents: read + + steps: + - name: Checkout code + uses: actions/checkout@93cb6efe18208431cddfb8368fd83d5badbf9bfd # v5.0.1 + with: + persist-credentials: false + + - name: Assert CFN/TF actions parity + run: bash scripts/check-aws-iam-parity.sh + + - name: Run parity script self-tests + run: bash scripts/test-aws-iam-parity.sh + + # Assert that no Terraform file grants a Secret Manager role at project, + # folder or organization scope; those scopes hand the member every secret in + # the scope. Fast (shell only), so it always runs. + gcp-secret-scope: + name: GCP Secret Manager grant scope + runs-on: ubuntu-latest + permissions: + contents: read + + steps: + - name: Checkout code + uses: actions/checkout@93cb6efe18208431cddfb8368fd83d5badbf9bfd # v5.0.1 + with: + persist-credentials: false + + - name: Assert no scope-wide Secret Manager grants + run: bash scripts/check-gcp-secret-scope.sh + + - name: Run guard script self-tests + run: bash scripts/test-gcp-secret-scope.sh + + # Assert that the ECR repository selector used by destroy-fargate-dev.yml and + # cleanup-staging.yml picks the repository each state owns and nothing else. + # The consumers force-delete what the selector prints, so both directions are + # asserted: over-matching deletes images the workflow does not own, and + # matching nothing leaves that state's repository behind. The suite also + # asserts the wiring, which is what #1592 and #1820 each escaped: all three + # destroy steps still call scripts/force-delete-owned-ecr-repo.sh, that script + # still deletes only what the selector yields, and nothing else under + # .github/workflows or scripts/ runs `aws ecr delete-repository` unguarded. + # That last claim is checked over a GLOB of both directories, not a list of + # known files, so a script added later is covered without anyone remembering + # to name it; the sweep is asserted to have opened a non-zero number of files + # first, since an empty swept set has no violations either. + # Fast (shell only), so it always runs. + ecr-delete-selection: + name: ECR delete selection scope + runs-on: ubuntu-latest + # This job checks out the tree and runs a shell script against it, so + # `contents: read` is all it needs. Same shape as security-scan above, + # which adds `security-events: write` only because it uploads SARIF. + # ci.yml now also declares `contents: read` at workflow level, so this + # block narrows nothing on its own; it is kept explicit so the job states + # its own requirement rather than inheriting silently. + permissions: + contents: read + + steps: + - name: Checkout code + uses: actions/checkout@93cb6efe18208431cddfb8368fd83d5badbf9bfd # v5.0.1 + with: + persist-credentials: false + + - name: Run selector self-tests + run: bash scripts/test-ecr-delete-selection.sh + + # Assert that the RDS instance selector used by destroy-fargate-dev.yml and + # cleanup-staging.yml unprotects the instance each state owns and nothing + # else. Deletion protection is the last line of defence on a database, so + # stripping it from an instance a state does not own leaves that database + # exposed to the next destroy that does match it. Both directions are + # asserted: a prefix near-miss is refused, and the owned instance is still + # selected, since a selector matching nothing passes every refusal assertion + # while leaving the destroy broken. The suite also asserts the wiring, which + # is what #1592, #1820 and #1821 each escaped: all three destroy steps call + # scripts/disable-owned-rds-deletion-protection.sh, that script unprotects + # only what the selector yields and swallows nothing, and nothing else under + # .github/workflows or scripts/ runs `aws rds modify-db-instance` unguarded. + # That last claim is checked over a GLOB of both directories, not a list of + # known files, so a script added later is covered without anyone remembering + # to name it; the sweep is asserted to have opened a non-zero number of files + # first, since an empty swept set has no violations either. + # It additionally runs the script end to end against stubbed terraform and aws, + # so "a failed strip is not swallowed" is asserted as behaviour, not as text. + # Fast (shell only), so it always runs. + rds-deletion-protection-scope: + name: RDS deletion protection scope + runs-on: ubuntu-latest + # Same shape as ecr-delete-selection above: this job checks out the tree and + # runs a shell script against it, so `contents: read` is all it needs. + permissions: + contents: read + + steps: + - name: Checkout code + uses: actions/checkout@93cb6efe18208431cddfb8368fd83d5badbf9bfd # v5.0.1 + with: + persist-credentials: false + + - name: Run RDS scope self-tests + run: bash scripts/test-rds-deletion-protection-scope.sh + + # Assert that the Cloud SQL instance selector used by cleanup-staging.yml's + # destroy-gcp job deletes the instance the staging state owns and nothing + # else. deletion_protection = true on that instance means a selector that + # matches nothing leaves it behind and the destroy fails on it; a selector + # that over-matches deletes a database this state never owned. Both + # directions are asserted: every near-miss name is refused, and the owned + # instance is still selected out of a hostile listing first, since a + # selector matching nothing passes every refusal assertion while leaving + # the destroy broken. The suite also asserts the wiring, which is what + # #1592, #1820 and #1821 each escaped on other resources: the destroy step + # calls scripts/delete-owned-cloud-sql-instance.sh, that script lists + # instances with no `--filter` and deletes only what the selector yields, + # the three `terraform state rm` calls are module-qualified and run only + # after a successful delete, and nothing else under .github/workflows or + # scripts/ runs `gcloud sql instances delete` unguarded. That last claim is + # checked over a GLOB of both directories, not a list of known files, so a + # script added later is covered without anyone remembering to name it; the + # sweep is asserted to have opened a non-zero number of files first, since + # an empty swept set has no violations either. + # It additionally runs the script end to end against stubbed terraform and + # gcloud, so "a failed delete is not swallowed" and "state is removed only + # after a successful delete" are asserted as behaviour, not as text. + # Fast (shell only), so it always runs. + cloud-sql-delete-scope: + name: Cloud SQL delete selection scope + runs-on: ubuntu-latest + # Same shape as ecr-delete-selection above: this job checks out the tree and + # runs a shell script against it, so `contents: read` is all it needs. + permissions: + contents: read + + steps: + - name: Checkout code + uses: actions/checkout@93cb6efe18208431cddfb8368fd83d5badbf9bfd # v5.0.1 + with: + persist-credentials: false + + - name: Run Cloud SQL scope self-tests + run: bash scripts/test-cloud-sql-delete-scope.sh + + # Assert that every job applying a `compute_platform` writes the Terraform + # state namespace that platform owns: `compute_platform=lambda` into + # github-/, `compute_platform=fargate` into github-fargate-/. The + # AWS environment is one Terraform root applied twice into two state objects, + # nothing in Terraform ties the backend key to the platform, and a job that + # pairs them wrongly initialises, plans and applies cleanly while rewriting + # the other platform's stack and recording it in the wrong state file. That + # was #1811, where rollback.yml's Fargate rollback keyed on the Lambda + # namespace, so both rollback jobs wrote the same object. A workflow run + # proves nothing about this, so the pairing is asserted as text. + # Both directions are asserted, and the positive one first: the seven real + # pairings are named and checked before any absence, since a scan that + # recognizes no state-writing job has no violations either. The negative half + # is checked over a GLOB of .github/workflows and scripts/, not a list of + # known files, so a job added later is covered without anyone naming it. + # Fast (shell only), so it always runs. + aws-tfstate-platform-key: + name: AWS Terraform state namespace per platform + runs-on: ubuntu-latest + # Same shape as ecr-delete-selection above: this job checks out the tree and + # runs a shell script against it, so `contents: read` is all it needs. + permissions: + contents: read + + steps: + - name: Checkout code + uses: actions/checkout@93cb6efe18208431cddfb8368fd83d5badbf9bfd # v5.0.1 + with: + persist-credentials: false + + - name: Run state namespace self-tests + run: bash scripts/test-aws-tfstate-platform-key.sh + + # Assert that no Terraform file declares an azurerm_key_vault_access_policy + # resource. This project's only Key Vault sets enable_rbac_authorization = + # true, and an RBAC-enabled vault ignores access policies entirely, so such a + # grant applies cleanly and then does nothing at runtime (#1621). The nested + # `access_policy` block form is deliberately not covered; #1839 tracks it. + # Fast (shell only), so it always runs. + azure-kv-access-policy: + name: Azure Key Vault grant model + runs-on: ubuntu-latest + permissions: + contents: read + + steps: + - name: Checkout code + uses: actions/checkout@93cb6efe18208431cddfb8368fd83d5badbf9bfd # v5.0.1 + with: + persist-credentials: false + + - name: Assert no Key Vault access policies + run: bash scripts/check-azure-kv-access-policy.sh + + - name: Run guard script self-tests + run: bash scripts/test-azure-kv-access-policy.sh + + # Summary job - all checks must pass + ci-success: + name: CI Success + runs-on: ubuntu-latest + needs: + - lint + - workflow-lint + - unit-tests + - mcp-build + - integration-tests + - docker-build + - terraform-validate + - security-scan + - e2e-tests + - azure-role-parity + - aws-iam-parity + - gcp-secret-scope + - ecr-delete-selection + - rds-deletion-protection-scope + - cloud-sql-delete-scope + - aws-tfstate-platform-key + - azure-kv-access-policy + if: always() + permissions: + contents: read + + steps: + - name: Check all jobs + # Allowlist on success instead of denylisting 'failure' and + # 'cancelled'. A job that never dispatched reports 'skipped', and the + # denylist form reported that as a pass -- so a gate could satisfy this + # summary by not running at all, which is the same shape as the scanner + # gap in #1836. Anything that is not exactly 'success' now fails, and + # names itself in the log. + env: + NEEDS_JSON: ${{ toJSON(needs) }} + run: | + # jq is preinstalled on the GitHub Ubuntu runner image (no new deps). + not_success="$(jq -r 'to_entries[] + | select(.value.result != "success") + | " \(.key): \(.value.result)"' <<<"$NEEDS_JSON")" + if [[ -n "$not_success" ]]; then + echo "::error::not every required CI job succeeded" + echo "$not_success" + exit 1 + fi + echo "All CI checks passed!" + + - name: Post status + run: | + { + echo "## CI Status" + echo "" + echo "✅ All CI checks completed successfully!" + } >> "$GITHUB_STEP_SUMMARY" diff --git a/.github/workflows/cleanup-staging.yml b/.github/workflows/cleanup-staging.yml new file mode 100644 index 000000000..8d8a57cfa --- /dev/null +++ b/.github/workflows/cleanup-staging.yml @@ -0,0 +1,424 @@ +# Cleanup Staging Resources (one-off) +# +# Destroys all resources tracked in the staging Terraform state files. +# Run this once after switching the default pipeline environment to 'dev'. +# +# Required GitHub Secrets (same as deploy workflows): +# - TF_BACKEND_AWS, TF_BACKEND_AZURE, TF_BACKEND_GCP +# - ADMIN_EMAIL +# - AZURE_CLIENT_ID, AZURE_TENANT_ID, AZURE_SUBSCRIPTION_ID +# +# Required GitHub Variables: +# - AWS_REGION, AWS_ROLE_TO_ASSUME +# - GCP_PROJECT_ID, GCP_WORKLOAD_IDENTITY_PROVIDER, GCP_SERVICE_ACCOUNT + +name: Cleanup Staging Resources + +on: + workflow_dispatch: + inputs: + confirm: + description: 'Type "destroy" to confirm destruction of ALL staging resources' + required: true + +# Least privilege: `id-token: write` is granted per job, only to the four jobs +# that assume a cloud deploy role, and each of those is bound to the `staging` +# deployment environment. `guard` authenticates to nothing and must not hold it. +# +# IMPORTANT — what the `environment:` binding does and does not buy. It scopes +# secrets/vars and sets the OIDC subject to `repo::environment:staging`. +# What each cloud does with that subject differs; see the per-job comments. +# +# It is NOT a reviewer gate. GitHub only blocks a job once required-reviewer +# protection rules are configured on the environment, and at the time of writing +# no environment in this repo has any. Note GitHub AUTO-CREATES a referenced +# environment bare on first use — no reviewers, no branch policy — so a binding +# to an environment nobody has configured runs straight through. Until that is +# configured out-of-band these destroy jobs still run unapproved. See #1660 for +# the live state; deliberately not restated here so this comment cannot rot into +# false reassurance. +# +# The environment subject is ref-agnostic, so the binding does not by itself keep +# this workflow on `main`. The `guard` job below checks the ref, but that check is +# defense-in-depth against ACCIDENTS ONLY and is NOT a security boundary: +# `workflow_dispatch` runs the workflow file as it exists on the dispatched ref, so +# anyone able to push a branch can delete the check and still present the +# `environment:staging` subject. The only control that survives that is a +# deployment branch policy, which GitHub evaluates before the job starts and +# before the token is minted. See #1660. +permissions: + contents: read + +env: + TF_VERSION: '1.10.0' + FORCE_JAVASCRIPT_ACTIONS_TO_NODE24: true + +jobs: + guard: + name: Confirm Destruction + runs-on: ubuntu-latest + permissions: {} + steps: + # An OIDC `sub` is EITHER `…:ref:refs/heads/` OR + # `…:environment:` -- never both (absent a custom sub-claim + # template, and this repo sets none). So binding the destroy jobs to an + # environment (below) REPLACES the ref-scoped subject with a ref-agnostic + # one, and neither `dev` nor `staging` carries a deployment_branch_policy. + # Without a ref check that trades a main-only restriction for any-branch + # access -- a widening, on the workflow this change exists to narrow. + # + # This step catches the accidental case only. It is NOT a security + # boundary: `workflow_dispatch` runs the workflow file as it exists on the + # dispatched ref, so anyone who can push a branch can delete this step and + # still present the `environment:staging` subject. The control that + # survives that is a deployment branch policy on the environment + # (custom_branch_policies + a `main` pattern), which GitHub evaluates + # before the job starts and before the token is minted, and which does NOT + # require `main` to be a protected branch. Tracked in #1660. + - name: Restrict to main + env: + REF: ${{ github.ref }} + run: | + if [ "$REF" != "refs/heads/main" ]; then + echo "::error::Refusing to destroy from '$REF'; this workflow may only be dispatched from refs/heads/main" + exit 1 + fi + + - name: Check confirmation + env: + CONFIRM: ${{ inputs.confirm }} + run: | + if [ "$CONFIRM" != "destroy" ]; then + echo "You must type 'destroy' exactly to confirm. Got: '$CONFIRM'" + exit 1 + fi + echo "Confirmation accepted. Proceeding with staging resource destruction." + + destroy-aws-lambda: + name: Destroy AWS Lambda (staging) + runs-on: ubuntu-24.04-arm + needs: guard + # Shares s3:///github-staging/terraform.tfstate with + # deploy-aws-lambda.yml and rollback.yml, so it shares their concurrency + # group: one writer per state file, whatever workflow or ref it came from + # (#1806). The suffix is a literal because this job's state key is too. + concurrency: + group: aws-tfstate-staging + cancel-in-progress: false + permissions: + id-token: write + contents: read + # AWS: role.tf's sub allowlist includes `environment:staging`, so this is + # the subject the trust policy matches. Not a reviewer gate until + # protection rules exist -- see the note at the top of this file. + environment: staging + env: + AWS_REGION: ${{ vars.AWS_REGION || 'us-east-1' }} + + steps: + - name: Checkout code + uses: actions/checkout@93cb6efe18208431cddfb8368fd83d5badbf9bfd # v5.0.1 + with: + persist-credentials: false + + - name: Configure AWS credentials + uses: aws-actions/configure-aws-credentials@d979d5b3a71173a29b74b5b88418bfda9437d885 # v6.1.1 + with: + role-to-assume: ${{ vars.AWS_ROLE_TO_ASSUME }} + aws-region: ${{ env.AWS_REGION }} + + - name: Setup Terraform + uses: hashicorp/setup-terraform@dfe3c3f87815947d99a8997f908cb6525fc44e9e # v4.0.1 + with: + terraform_version: ${{ env.TF_VERSION }} + + - name: Terraform Init (staging state) + env: + TF_BACKEND: ${{ secrets.TF_BACKEND_AWS }} + run: | + printf '%s\nkey = "github-staging/terraform.tfstate"\n' "$TF_BACKEND" > /tmp/backend.tfbackend + cd terraform/environments/aws + terraform init -backend-config=/tmp/backend.tfbackend + + # Runs before `terraform destroy`: the repository is created with + # force_delete = false, so the destroy fails while images remain. Deletes + # only the repository THIS state owns, by exact name. The `cudly-staging*` + # prefix this used to select by also matched `cudly-staging-prod-mirror`, + # `cudly-staging--backup` and the sibling staging state's repository, + # and force-deleted every image in them (#1820). Rationale, the + # `output -json` handling and why nothing here is swallowed: the script's + # header. Both staging jobs and destroy-fargate-dev.yml call the same + # script, so the guard cannot land in one workflow and not its sibling -- + # which is how #1592 became #1820. + - name: Force-delete the ECR repo this state owns + run: ./scripts/force-delete-owned-ecr-repo.sh terraform/environments/aws + + # Runs before `terraform destroy`: an instance applied with + # deletion_protection = true blocks the destroy. Unprotects only the + # instance THIS state owns, by exact identifier. The + # `starts_with(DBInstanceIdentifier,'cudly-staging')` filter this step used + # to run also matched `cudly-staging-prod-mirror`, + # `cudly-staging--postgres-replica` and the sibling staging state's + # instance, and stripped the last line of defence from them, leaving them + # exposed to the next destroy that did match (#1821). Rationale and why + # nothing here is swallowed: the script's header. Both staging jobs and + # destroy-fargate-dev.yml call the same script, so the guard cannot land in + # one workflow and not its sibling -- which is how #1592 became #1820. + - name: Disable RDS deletion protection on the instance this state owns + run: ./scripts/disable-owned-rds-deletion-protection.sh terraform/environments/aws + + - name: Terraform Destroy + env: + TF_VAR_admin_email: ${{ secrets.ADMIN_EMAIL }} + run: | + cd terraform/environments/aws + terraform destroy \ + -var-file="github-staging.tfvars" \ + -var="compute_platform=lambda" \ + -auto-approve + + destroy-aws-fargate: + name: Destroy AWS Fargate (staging) + runs-on: ubuntu-24.04-arm + needs: guard + # Shares s3:///github-fargate-staging/terraform.tfstate with + # deploy-aws-fargate.yml, so it shares that workflow's concurrency group + # (#1806). This is a different state object from destroy-aws-lambda's + # above, so the two jobs are deliberately in different groups and may run + # concurrently. The suffix is a literal because this job's state key is too. + concurrency: + group: aws-fargate-tfstate-staging + cancel-in-progress: false + permissions: + id-token: write + contents: read + # AWS: role.tf's sub allowlist includes `environment:staging`, so this is + # the subject the trust policy matches. Not a reviewer gate until + # protection rules exist -- see the note at the top of this file. + environment: staging + env: + AWS_REGION: ${{ vars.AWS_REGION || 'us-east-1' }} + + steps: + - name: Checkout code + uses: actions/checkout@93cb6efe18208431cddfb8368fd83d5badbf9bfd # v5.0.1 + with: + persist-credentials: false + + - name: Configure AWS credentials + uses: aws-actions/configure-aws-credentials@d979d5b3a71173a29b74b5b88418bfda9437d885 # v6.1.1 + with: + role-to-assume: ${{ vars.AWS_ROLE_TO_ASSUME }} + aws-region: ${{ env.AWS_REGION }} + + - name: Setup Terraform + uses: hashicorp/setup-terraform@dfe3c3f87815947d99a8997f908cb6525fc44e9e # v4.0.1 + with: + terraform_version: ${{ env.TF_VERSION }} + + - name: Terraform Init (staging state) + env: + TF_BACKEND: ${{ secrets.TF_BACKEND_AWS }} + run: | + printf '%s\nkey = "github-fargate-staging/terraform.tfstate"\n' "$TF_BACKEND" > /tmp/backend.tfbackend + cd terraform/environments/aws + terraform init -backend-config=/tmp/backend.tfbackend + + # Runs before `terraform destroy`: the repository is created with + # force_delete = false, so the destroy fails while images remain. Deletes + # only the repository THIS state owns, by exact name. The `cudly-staging*` + # prefix this used to select by also matched `cudly-staging-prod-mirror`, + # `cudly-staging--backup` and the sibling staging state's repository, + # and force-deleted every image in them (#1820). Rationale, the + # `output -json` handling and why nothing here is swallowed: the script's + # header. Both staging jobs and destroy-fargate-dev.yml call the same + # script, so the guard cannot land in one workflow and not its sibling -- + # which is how #1592 became #1820. + - name: Force-delete the ECR repo this state owns + run: ./scripts/force-delete-owned-ecr-repo.sh terraform/environments/aws + + # Runs before `terraform destroy`: an instance applied with + # deletion_protection = true blocks the destroy. Unprotects only the + # instance THIS state owns, by exact identifier. The + # `starts_with(DBInstanceIdentifier,'cudly-staging')` filter this step used + # to run also matched `cudly-staging-prod-mirror`, + # `cudly-staging--postgres-replica` and the sibling staging state's + # instance, and stripped the last line of defence from them, leaving them + # exposed to the next destroy that did match (#1821). Rationale and why + # nothing here is swallowed: the script's header. Both staging jobs and + # destroy-fargate-dev.yml call the same script, so the guard cannot land in + # one workflow and not its sibling -- which is how #1592 became #1820. + - name: Disable RDS deletion protection on the instance this state owns + run: ./scripts/disable-owned-rds-deletion-protection.sh terraform/environments/aws + + - name: Terraform Destroy + env: + TF_VAR_admin_email: ${{ secrets.ADMIN_EMAIL }} + run: | + cd terraform/environments/aws + terraform destroy \ + -var-file="github-staging.tfvars" \ + -var="compute_platform=fargate" \ + -auto-approve + + destroy-azure: + name: Destroy Azure (staging) + runs-on: ubuntu-latest + needs: guard + # Shares the Azure Terraform state blob github-staging.terraform.tfstate + # with deploy-azure.yml and rollback.yml, so it shares their concurrency + # group: one writer per state file, whatever workflow or ref it came from + # (#1801). The suffix is a literal because this job's state key is too. + concurrency: + group: azure-tfstate-staging + cancel-in-progress: false + permissions: + id-token: write + contents: read + # Azure: REQUIRES the bootstrap module + # terraform/environments/azure/ci-cd-permissions to have been re-applied + # with `github_environments` including "staging". Until that manual apply + # happens this job fails with AADSTS70021, because the only federated + # credentials that exist are ref:refs/heads/main and pull_request (#1648). + # Not a reviewer gate either -- see the note at the top of this file. + environment: staging + env: + ARM_USE_OIDC: "true" + ARM_CLIENT_ID: ${{ secrets.AZURE_CLIENT_ID }} + ARM_TENANT_ID: ${{ secrets.AZURE_TENANT_ID }} + ARM_SUBSCRIPTION_ID: ${{ secrets.AZURE_SUBSCRIPTION_ID }} + + steps: + - name: Checkout code + uses: actions/checkout@93cb6efe18208431cddfb8368fd83d5badbf9bfd # v5.0.1 + with: + persist-credentials: false + + - name: Azure Login + uses: azure/login@532459ea530d8321f2fb9bb10d1e0bcf23869a43 # v3.0.0 + with: + client-id: ${{ secrets.AZURE_CLIENT_ID }} + tenant-id: ${{ secrets.AZURE_TENANT_ID }} + subscription-id: ${{ secrets.AZURE_SUBSCRIPTION_ID }} + + - name: Setup Terraform + uses: hashicorp/setup-terraform@dfe3c3f87815947d99a8997f908cb6525fc44e9e # v4.0.1 + with: + terraform_version: ${{ env.TF_VERSION }} + + # This step used to also break the blob lease on the state file, with no + # age check and no --lease-break-period, so it broke a live lease held by + # a concurrent writer. Removed with #1801; the job-level `concurrency` + # group above is what keeps writers apart now, and a loud "Error acquiring + # the state lock" is the correct outcome if one ever slips through. + - name: Write Terraform backend config (staging state) + env: + TF_BACKEND: ${{ secrets.TF_BACKEND_AZURE }} + run: | + printf '%s\nkey = "github-staging.terraform.tfstate"\n' "$TF_BACKEND" > /tmp/backend.tfbackend + + - name: Terraform Init (staging state) + run: | + cd terraform/environments/azure + terraform init -backend-config=/tmp/backend.tfbackend + + - name: Terraform Destroy + env: + TF_VAR_admin_email: ${{ secrets.ADMIN_EMAIL }} + TF_VAR_subscription_id: ${{ secrets.AZURE_SUBSCRIPTION_ID }} + run: | + cd terraform/environments/azure + terraform destroy \ + -var-file="github-staging.tfvars" \ + -var="location=eastus" \ + -auto-approve + + destroy-gcp: + name: Destroy GCP (staging) + runs-on: ubuntu-latest + needs: guard + # Shares gs:///github-staging/default.tfstate with deploy-gcp.yml + # and rollback.yml, so it shares their concurrency group (#1806). This job + # writes state twice: the `terraform state rm` calls below as well as the + # destroy. The suffix is a literal because this job's backend prefix is too. + concurrency: + group: gcp-tfstate-staging + cancel-in-progress: false + permissions: + id-token: write + contents: read + # GCP: the WIF provider keys on assertion.repository and assertion.ref + # only (gcp/ci-cd-permissions/github_oidc.tf:37), and the impersonation + # binding is on attribute.repository -- nothing reads the subject. So this + # binding does NOT affect whether the token mints; it is here for parity + # and so protection rules can apply once they exist. GCP therefore stays + # main-only via assertion.ref regardless. + environment: staging + env: + GCP_REGION: ${{ vars.GCP_REGION || 'us-central1' }} + + steps: + - name: Checkout code + uses: actions/checkout@93cb6efe18208431cddfb8368fd83d5badbf9bfd # v5.0.1 + with: + persist-credentials: false + + - name: Authenticate to GCP + uses: google-github-actions/auth@7c6bc770dae815cd3e89ee6cdf493a5fab2cc093 # v3.0.0 + with: + workload_identity_provider: ${{ vars.GCP_WORKLOAD_IDENTITY_PROVIDER }} + service_account: ${{ vars.GCP_SERVICE_ACCOUNT }} + + - name: Setup Terraform + uses: hashicorp/setup-terraform@dfe3c3f87815947d99a8997f908cb6525fc44e9e # v4.0.1 + with: + terraform_version: ${{ env.TF_VERSION }} + + - name: Terraform Init (staging state) + env: + TF_BACKEND: ${{ secrets.TF_BACKEND_GCP }} + run: | + printf '%s\nprefix = "github-staging"\n' "$TF_BACKEND" > /tmp/backend.tfbackend + cd terraform/environments/gcp + terraform init -backend-config=/tmp/backend.tfbackend + + # Runs before `terraform destroy`: destroying google_sql_user and + # google_sql_database through Terraform deadlocks on PostgreSQL object + # ownership, and github-staging.tfvars applies the instance with + # deletion_protection = true. Deletes only the instance THIS state owns, + # by exact name, then removes it from state. The + # `--filter="name:cudly-staging" | head -1` selection this step used to run + # also matched `cudly-staging-postgres-replica`, `backup-cudly-staging` and + # any operator-named `cudly-staging-*` instance, deleted with `|| true`, + # and then ran `terraform state rm` regardless (#1971). Rationale, the + # measured gcloud filter semantics and why nothing here is swallowed: the + # script's header. Same shape as the ECR and RDS steps in the AWS jobs + # above, which is how the guard reaches every sibling site (#1592 -> #1820 + # -> #1821 was a guard landing on one site and not the next). + - name: Delete the Cloud SQL instance this state owns + env: + TF_VAR_project_id: ${{ vars.GCP_PROJECT_ID }} + GCP_PROJECT_ID: ${{ vars.GCP_PROJECT_ID }} + run: | + # Routed via env: rather than interpolated into this script's source, + # per the zero-expression-interpolation-in-run-blocks rule from #1641. + PROJECT="$GCP_PROJECT_ID" + ./scripts/delete-owned-cloud-sql-instance.sh terraform/environments/gcp "$PROJECT" + # Remove the Service Networking Connection from state and delete directly + # GCP needs extra time after Cloud SQL deletion to release VPC peering + cd terraform/environments/gcp + terraform state rm module.networking.google_service_networking_connection.private_vpc_connection 2>/dev/null || true + gcloud services vpc-peerings delete \ + --service=servicenetworking.googleapis.com \ + --network=cudly-staging-vpc \ + --project="$PROJECT" --quiet 2>/dev/null || echo "VPC peering already gone or not found" + + - name: Terraform Destroy + env: + TF_VAR_admin_email: ${{ secrets.ADMIN_EMAIL }} + TF_VAR_project_id: ${{ vars.GCP_PROJECT_ID }} + run: | + cd terraform/environments/gcp + terraform destroy \ + -var-file="github-staging.tfvars" \ + -auto-approve diff --git a/.github/workflows/database-migration.yml b/.github/workflows/database-migration.yml new file mode 100644 index 000000000..973f68772 --- /dev/null +++ b/.github/workflows/database-migration.yml @@ -0,0 +1,583 @@ +# Run Database Migrations +# +# This workflow manages database schema migrations across all cloud providers. +# It can run migrations up (apply) or down (rollback) with safety checks. +# +# Required GitHub Secrets: +# - DB_PASSWORD_AWS: Database password for AWS Aurora +# - DB_PASSWORD_GCP: Database password for GCP Cloud SQL +# - DB_PASSWORD_AZURE: Database password for Azure Flexible Server +# - (AWS auth via OIDC: vars.AWS_ROLE_TO_ASSUME) +# - (GCP auth via WIF: vars.GCP_WORKLOAD_IDENTITY_PROVIDER + vars.GCP_SERVICE_ACCOUNT) +# - AZURE_CLIENT_ID, AZURE_TENANT_ID, AZURE_SUBSCRIPTION_ID (OIDC federation) +# +# Required GitHub Variables: +# - DB_HOST_AWS_DEV: Aurora endpoint (dev) +# - DB_HOST_AWS_STAGING: Aurora endpoint (staging) +# - DB_HOST_AWS_PROD: Aurora endpoint (prod) +# - (Similar for GCP and Azure) +# +# Triggered by: +# - Manual workflow dispatch +# - Before deployment workflows (optional) + +name: Database Migration + +# Least privilege by default: only the jobs that actually authenticate to a +# cloud provider get `id-token: write`, and every one of those jobs is bound to +# a deployment environment (`aws-db-`, `gcp-db-`, `azure-db-`). +# `validate` and `summary` never authenticate, so they stay on the `contents: +# read` floor below. +permissions: + contents: read + +on: + workflow_dispatch: + inputs: + cloud: + description: 'Cloud provider' + required: true + type: choice + options: [aws, gcp, azure, all] + environment: + description: 'Environment' + required: true + type: choice + options: [dev, staging, prod] + direction: + description: 'Migration direction' + required: true + type: choice + options: [up, down] + default: up + steps: + description: 'Number of migrations to apply/rollback (0 = all; for direction=down an explicit positive value is required)' + required: false + type: number + default: 0 + confirm: + description: 'Type "rollback-prod" to confirm a down migration on prod (ignored otherwise)' + required: false + type: string + default: '' + workflow_call: + inputs: + cloud: + required: true + type: string + environment: + required: true + type: string + direction: + required: false + type: string + default: up + +env: + MIGRATIONS_PATH: internal/database/postgres/migrations + +jobs: + # Validate migration request + validate: + name: Validate Migration Request + runs-on: ubuntu-latest + permissions: + contents: read + outputs: + is_safe: ${{ steps.check.outputs.is_safe }} + + steps: + - name: Checkout code + uses: actions/checkout@93cb6efe18208431cddfb8368fd83d5badbf9bfd # v5.0.1 + with: + persist-credentials: false + + - name: Safety checks + id: check + env: + CLOUD: ${{ inputs.cloud }} + ENVIRONMENT: ${{ inputs.environment }} + DIRECTION: ${{ inputs.direction }} + STEPS: ${{ inputs.steps }} + CONFIRM: ${{ inputs.confirm }} + run: | + set -uo pipefail + IS_SAFE=true + + # `workflow_call` never declares a `steps` input, so on that path this + # renders empty. Treat that the same as the explicit default of "0" + # (apply/rollback everything) rather than rejecting it outright. + STEPS="${STEPS:-0}" + + # workflow_dispatch's cloud/environment/direction are `type: choice`, + # enforced server-side before the run starts. workflow_call's are plain + # `type: string` with no such enforcement, so re-validate them here + # against the same allowlist regardless of which trigger fired. + case "$CLOUD" in + aws|gcp|azure|all) ;; + *) + echo "❌ Invalid cloud provider: '$CLOUD' (expected aws, gcp, azure, or all)" + IS_SAFE=false + ;; + esac + + case "$ENVIRONMENT" in + dev|staging|prod) ;; + *) + echo "❌ Invalid environment: '$ENVIRONMENT' (expected dev, staging, or prod)" + IS_SAFE=false + ;; + esac + + case "$DIRECTION" in + up|down) ;; + *) + echo "❌ Invalid direction: '$DIRECTION' (expected up or down)" + IS_SAFE=false + ;; + esac + + # Check if migrations directory exists + if [ ! -d "$MIGRATIONS_PATH" ]; then + echo "❌ Migrations directory not found: $MIGRATIONS_PATH" + IS_SAFE=false + fi + + # `steps` must be a non-negative integer regardless of direction: 0 + # means "everything" for direction=up, and direction=down requires an + # explicit positive value (checked separately below). Reject anything + # else before it can reach a migrate invocation in either direction. + if ! [[ "$STEPS" =~ ^(0|[1-9][0-9]*)$ ]]; then + echo "❌ 'steps' must be a non-negative integer (got '$STEPS')." + IS_SAFE=false + fi + + # Down migrations must specify an explicit positive step count. + # steps=0 (the default) would run 'migrate down -all' and drop the + # entire schema, so it is rejected here for direction=down. + if [[ "$DIRECTION" == "down" ]]; then + if ! [[ "$STEPS" =~ ^[1-9][0-9]*$ ]]; then + echo "❌ direction=down requires an explicit positive 'steps' value (got '$STEPS')." + echo "Rolling back ALL migrations at once is not supported by this workflow." + IS_SAFE=false + fi + + # Production down migrations additionally require typed confirmation. + if [[ "$ENVIRONMENT" == "prod" ]]; then + echo "⚠️ WARNING: Attempting to rollback migrations on PRODUCTION" + echo "This operation is destructive and may cause data loss!" + if [[ "$CONFIRM" != "rollback-prod" ]]; then + echo "❌ Production rollback requires typing 'rollback-prod' in the 'confirm' input." + IS_SAFE=false + fi + fi + fi + + # Check migration files + if [ -d "$MIGRATIONS_PATH" ]; then + UP_COUNT=$(find "$MIGRATIONS_PATH" -maxdepth 1 -type f -name '*.up.sql' 2>/dev/null | wc -l) + DOWN_COUNT=$(find "$MIGRATIONS_PATH" -maxdepth 1 -type f -name '*.down.sql' 2>/dev/null | wc -l) + + echo "Migration files found:" + echo " Up migrations: $UP_COUNT" + echo " Down migrations: $DOWN_COUNT" + + if [ "$UP_COUNT" -eq 0 ]; then + echo "❌ No migration files found" + IS_SAFE=false + fi + fi + + echo "is_safe=$IS_SAFE" >> "$GITHUB_OUTPUT" + + if [[ "$IS_SAFE" != "true" ]]; then + echo "Validation failed; refusing to run migrations." + exit 1 + fi + + - name: Display migration plan + env: + CLOUD: ${{ inputs.cloud }} + ENVIRONMENT: ${{ inputs.environment }} + DIRECTION: ${{ inputs.direction }} + STEPS: ${{ inputs.steps || 'all' }} + run: | + set -uo pipefail + { + echo "## Database Migration Plan" + echo "" + echo "**Cloud:** $CLOUD" + echo "**Environment:** $ENVIRONMENT" + echo "**Direction:** $DIRECTION" + echo "**Steps:** $STEPS" + echo "" + + if [[ "$ENVIRONMENT" == "prod" ]] && [[ "$DIRECTION" == "down" ]]; then + echo "⚠️ **WARNING:** This will rollback migrations on PRODUCTION!" + fi + } >> "$GITHUB_STEP_SUMMARY" + + # Run AWS migrations + migrate-aws: + name: Migrate AWS Database + runs-on: ubuntu-latest + needs: validate + if: | + needs.validate.outputs.is_safe == 'true' && + (inputs.cloud == 'aws' || inputs.cloud == 'all') + # Serializes migrations against one database: two dispatches for the same + # cloud and environment must not run `migrate` concurrently against the same + # schema_migrations table (#1593). Deliberately NOT the `aws-tfstate-*` group + # deploy-aws-lambda.yml and rollback.yml use: this job only runs `terraform + # init` + `terraform output -raw` to read the DB endpoint, takes no state + # lock, and the resource needing protection is the database, not the state + # file. Different clouds, and the same cloud in different environments, are + # different databases and stay concurrent. + # + # `inputs.environment` is a required `choice` on `workflow_dispatch`, so that + # path is constrained to dev|staging|prod server-side. On the `workflow_call` + # path it is a plain `type: string`; `required: true` makes the caller pass + # the key but GitHub does not enforce that the value is non-empty, and an + # empty suffix would collapse every environment into one group. The fallback + # below keeps such a call in its own group instead. `validate` rejects an + # environment outside the allowlist before this job runs, so the sentinel + # should be unreachable in practice; it is here so the group can never be + # silently ambiguous if that ordering ever changes. + # + # `cancel-in-progress: false` because cancelling mid-`migrate` leaves + # schema_migrations dirty, which then needs the CUDLY_FORCE_MIGRATION_VERSION + # recovery path in internal/database/postgres/migrations/migrate.go. Two + # consequences, both accepted as better than concurrent writers to one + # database: + # - GitHub keeps one pending entry per group, so a queued migration can be + # evicted by a later dispatch. Unlike rollback.yml, the `summary` job + # below does not exit 1 on a non-success result, so an evicted migration + # leaves the run green; read the per-cloud results it prints. + # - if the `environment:` binding below carries required reviewers, it is + # undocumented whether a job parked awaiting approval holds its group. + concurrency: + group: db-migrate-aws-${{ inputs.environment || 'unset' }} + cancel-in-progress: false + permissions: + id-token: write + contents: read + environment: + name: aws-db-${{ inputs.environment }} + + steps: + - name: Checkout code + uses: actions/checkout@93cb6efe18208431cddfb8368fd83d5badbf9bfd # v5.0.1 + with: + persist-credentials: false + + - name: Set up Go + uses: actions/setup-go@4a3601121dd01d1626a1e23e37211e3254c1c06c # v6.4.0 + with: + go-version-file: go.mod + + - name: Install golang-migrate + # No @version suffix: installed as a package of this module, so the + # migrate version and its dependency set come from go.mod, the same way + # the Dockerfile and `make install-tools` build it. -tags 'pgx5' keeps + # lib/pq out of the binary that talks to the production database. + # See issue #1849. + run: go install -tags 'pgx5' github.com/golang-migrate/migrate/v4/cmd/migrate + + - name: Configure AWS credentials + uses: aws-actions/configure-aws-credentials@d979d5b3a71173a29b74b5b88418bfda9437d885 # v6.1.1 + with: + role-to-assume: ${{ vars.AWS_ROLE_TO_ASSUME }} + aws-region: ${{ vars.AWS_REGION || 'us-east-1' }} + + - name: Get database endpoint from Terraform + id: get-endpoint + run: | + cd terraform/environments/aws + terraform init + DB_ENDPOINT=$(terraform output -raw database_proxy_endpoint 2>/dev/null || echo "") + if [ -z "$DB_ENDPOINT" ]; then + echo "Failed to get database endpoint" + exit 1 + fi + echo "endpoint=$DB_ENDPOINT" >> "$GITHUB_OUTPUT" + + - name: Run migrations + env: + DB_PASSWORD: ${{ secrets.DB_PASSWORD_AWS }} + DB_ENDPOINT: ${{ steps.get-endpoint.outputs.endpoint }} + DIRECTION: ${{ inputs.direction }} + STEPS: ${{ inputs.steps }} + run: | + set -uo pipefail + # pgx5://, not postgresql://: migrate is installed with -tags 'pgx5' + # above and golang-migrate dispatches on the URL scheme (issue #1849). + DB_URL="pgx5://cudly:${DB_PASSWORD}@${DB_ENDPOINT}:5432/cudly?sslmode=require" + + # `validate` already rejects a malformed `steps`/`direction` before this + # job is ever reached, but this job sits behind an environment gate and + # is the thing that actually runs `migrate`, so the checks are repeated + # here against the re-parsed shell variable as defense in depth. + STEPS="${STEPS:-0}" + + if [[ "$DIRECTION" == "up" ]]; then + if ! [[ "$STEPS" =~ ^(0|[1-9][0-9]*)$ ]]; then + echo "❌ 'steps' must be a non-negative integer (got '$STEPS')." + exit 1 + fi + if [[ "$STEPS" == "0" ]]; then + echo "Applying all pending migrations..." + migrate -path "$MIGRATIONS_PATH" -database "$DB_URL" up + else + echo "Applying $STEPS migration(s)..." + migrate -path "$MIGRATIONS_PATH" -database "$DB_URL" up "$STEPS" + fi + else + if ! [[ "$STEPS" =~ ^[1-9][0-9]*$ ]]; then + echo "❌ Refusing to roll back without an explicit positive 'steps' value." + exit 1 + fi + echo "Rolling back $STEPS migration(s)..." + migrate -path "$MIGRATIONS_PATH" -database "$DB_URL" down "$STEPS" + fi + + - name: Get migration version + env: + DB_PASSWORD: ${{ secrets.DB_PASSWORD_AWS }} + DB_ENDPOINT: ${{ steps.get-endpoint.outputs.endpoint }} + run: | + set -uo pipefail + # pgx5://, not postgresql://: migrate is installed with -tags 'pgx5' + # above and golang-migrate dispatches on the URL scheme (issue #1849). + DB_URL="pgx5://cudly:${DB_PASSWORD}@${DB_ENDPOINT}:5432/cudly?sslmode=require" + VERSION=$(migrate -path "$MIGRATIONS_PATH" -database "$DB_URL" version 2>&1 || echo "unknown") + echo "Current migration version: $VERSION" + echo "MIGRATION_VERSION=$VERSION" >> "$GITHUB_ENV" + + # Run GCP migrations + migrate-gcp: + name: Migrate GCP Database + runs-on: ubuntu-latest + needs: validate + if: | + needs.validate.outputs.is_safe == 'true' && + (inputs.cloud == 'gcp' || inputs.cloud == 'all') + # Serializes migrations against the GCP database for one environment + # (#1593). A different database from `migrate-aws`'s, so a different group: + # `cloud=all` still migrates the three clouds concurrently. The group-key, + # empty-suffix and `cancel-in-progress: false` reasoning documented on + # `migrate-aws` applies here identically. + concurrency: + group: db-migrate-gcp-${{ inputs.environment || 'unset' }} + cancel-in-progress: false + permissions: + id-token: write + contents: read + environment: + name: gcp-db-${{ inputs.environment }} + + steps: + - name: Checkout code + uses: actions/checkout@93cb6efe18208431cddfb8368fd83d5badbf9bfd # v5.0.1 + with: + persist-credentials: false + + - name: Set up Go + uses: actions/setup-go@4a3601121dd01d1626a1e23e37211e3254c1c06c # v6.4.0 + with: + go-version-file: go.mod + + - name: Install golang-migrate + # No @version suffix: installed as a package of this module, so the + # migrate version and its dependency set come from go.mod, the same way + # the Dockerfile and `make install-tools` build it. -tags 'pgx5' keeps + # lib/pq out of the binary that talks to the production database. + # See issue #1849. + run: go install -tags 'pgx5' github.com/golang-migrate/migrate/v4/cmd/migrate + + - name: Authenticate to Google Cloud + uses: google-github-actions/auth@7c6bc770dae815cd3e89ee6cdf493a5fab2cc093 # v3.0.0 + with: + workload_identity_provider: ${{ vars.GCP_WORKLOAD_IDENTITY_PROVIDER }} + service_account: ${{ vars.GCP_SERVICE_ACCOUNT }} + + - name: Set up Cloud SDK + uses: google-github-actions/setup-gcloud@aa5489c8933f4cc7a4f7d45035b3b1440c9c10db # v3.0.1 + + - name: Get database endpoint from Terraform + id: get-endpoint + run: | + cd terraform/environments/gcp + terraform init + DB_ENDPOINT=$(terraform output -raw database_private_ip 2>/dev/null || echo "") + if [ -z "$DB_ENDPOINT" ]; then + echo "Failed to get database endpoint" + exit 1 + fi + echo "endpoint=$DB_ENDPOINT" >> "$GITHUB_OUTPUT" + + - name: Run migrations + env: + DB_PASSWORD: ${{ secrets.DB_PASSWORD_GCP }} + DB_ENDPOINT: ${{ steps.get-endpoint.outputs.endpoint }} + DIRECTION: ${{ inputs.direction }} + STEPS: ${{ inputs.steps }} + run: | + set -uo pipefail + # pgx5://, not postgresql://: migrate is installed with -tags 'pgx5' + # above and golang-migrate dispatches on the URL scheme (issue #1849). + DB_URL="pgx5://cudly:${DB_PASSWORD}@${DB_ENDPOINT}:5432/cudly?sslmode=require" + + STEPS="${STEPS:-0}" + + if [[ "$DIRECTION" == "up" ]]; then + if ! [[ "$STEPS" =~ ^(0|[1-9][0-9]*)$ ]]; then + echo "❌ 'steps' must be a non-negative integer (got '$STEPS')." + exit 1 + fi + if [[ "$STEPS" == "0" ]]; then + migrate -path "$MIGRATIONS_PATH" -database "$DB_URL" up + else + migrate -path "$MIGRATIONS_PATH" -database "$DB_URL" up "$STEPS" + fi + else + if ! [[ "$STEPS" =~ ^[1-9][0-9]*$ ]]; then + echo "❌ Refusing to roll back without an explicit positive 'steps' value." + exit 1 + fi + migrate -path "$MIGRATIONS_PATH" -database "$DB_URL" down "$STEPS" + fi + + # Run Azure migrations + migrate-azure: + name: Migrate Azure Database + runs-on: ubuntu-latest + needs: validate + if: | + needs.validate.outputs.is_safe == 'true' && + (inputs.cloud == 'azure' || inputs.cloud == 'all') + # Serializes migrations against the Azure database for one environment + # (#1593). A different database from `migrate-aws`'s and `migrate-gcp`'s, so + # a different group: `cloud=all` still migrates the three clouds + # concurrently. The group-key, empty-suffix and `cancel-in-progress: false` + # reasoning documented on `migrate-aws` applies here identically. + concurrency: + group: db-migrate-azure-${{ inputs.environment || 'unset' }} + cancel-in-progress: false + permissions: + id-token: write + contents: read + environment: + name: azure-db-${{ inputs.environment }} + + steps: + - name: Checkout code + uses: actions/checkout@93cb6efe18208431cddfb8368fd83d5badbf9bfd # v5.0.1 + with: + persist-credentials: false + + - name: Set up Go + uses: actions/setup-go@4a3601121dd01d1626a1e23e37211e3254c1c06c # v6.4.0 + with: + go-version-file: go.mod + + - name: Install golang-migrate + # No @version suffix: installed as a package of this module, so the + # migrate version and its dependency set come from go.mod, the same way + # the Dockerfile and `make install-tools` build it. -tags 'pgx5' keeps + # lib/pq out of the binary that talks to the production database. + # See issue #1849. + run: go install -tags 'pgx5' github.com/golang-migrate/migrate/v4/cmd/migrate + + - name: Azure Login + uses: azure/login@532459ea530d8321f2fb9bb10d1e0bcf23869a43 # v3.0.0 + with: + client-id: ${{ secrets.AZURE_CLIENT_ID }} + tenant-id: ${{ secrets.AZURE_TENANT_ID }} + subscription-id: ${{ secrets.AZURE_SUBSCRIPTION_ID }} + + - name: Get database endpoint from Terraform + id: get-endpoint + run: | + cd terraform/environments/azure + terraform init + DB_ENDPOINT=$(terraform output -raw database_fqdn 2>/dev/null || echo "") + if [ -z "$DB_ENDPOINT" ]; then + echo "Failed to get database endpoint" + exit 1 + fi + echo "endpoint=$DB_ENDPOINT" >> "$GITHUB_OUTPUT" + + - name: Run migrations + env: + DB_PASSWORD: ${{ secrets.DB_PASSWORD_AZURE }} + DB_ENDPOINT: ${{ steps.get-endpoint.outputs.endpoint }} + DIRECTION: ${{ inputs.direction }} + STEPS: ${{ inputs.steps }} + run: | + set -uo pipefail + # pgx5://, not postgresql://: migrate is installed with -tags 'pgx5' + # above and golang-migrate dispatches on the URL scheme (issue #1849). + DB_URL="pgx5://cudly:${DB_PASSWORD}@${DB_ENDPOINT}:5432/cudly?sslmode=require" + + STEPS="${STEPS:-0}" + + if [[ "$DIRECTION" == "up" ]]; then + if ! [[ "$STEPS" =~ ^(0|[1-9][0-9]*)$ ]]; then + echo "❌ 'steps' must be a non-negative integer (got '$STEPS')." + exit 1 + fi + if [[ "$STEPS" == "0" ]]; then + migrate -path "$MIGRATIONS_PATH" -database "$DB_URL" up + else + migrate -path "$MIGRATIONS_PATH" -database "$DB_URL" up "$STEPS" + fi + else + if ! [[ "$STEPS" =~ ^[1-9][0-9]*$ ]]; then + echo "❌ Refusing to roll back without an explicit positive 'steps' value." + exit 1 + fi + migrate -path "$MIGRATIONS_PATH" -database "$DB_URL" down "$STEPS" + fi + + # Summary + summary: + name: Migration Summary + runs-on: ubuntu-latest + permissions: + contents: read + needs: [validate, migrate-aws, migrate-gcp, migrate-azure] + if: always() + + steps: + - name: Post summary + env: + CLOUD: ${{ inputs.cloud }} + ENVIRONMENT: ${{ inputs.environment }} + DIRECTION: ${{ inputs.direction }} + RESULT_AWS: ${{ needs.migrate-aws.result }} + RESULT_GCP: ${{ needs.migrate-gcp.result }} + RESULT_AZURE: ${{ needs.migrate-azure.result }} + run: | + set -uo pipefail + { + echo "## Database Migration Results" + echo "" + echo "**Cloud:** $CLOUD" + echo "**Environment:** $ENVIRONMENT" + echo "**Direction:** $DIRECTION" + echo "" + + echo "### Results" + + if [[ "$CLOUD" == "aws" ]] || [[ "$CLOUD" == "all" ]]; then + echo "- AWS: $RESULT_AWS" + fi + + if [[ "$CLOUD" == "gcp" ]] || [[ "$CLOUD" == "all" ]]; then + echo "- GCP: $RESULT_GCP" + fi + + if [[ "$CLOUD" == "azure" ]] || [[ "$CLOUD" == "all" ]]; then + echo "- Azure: $RESULT_AZURE" + fi + } >> "$GITHUB_STEP_SUMMARY" diff --git a/.github/workflows/deploy-all.yml b/.github/workflows/deploy-all.yml new file mode 100644 index 000000000..c89d51d69 --- /dev/null +++ b/.github/workflows/deploy-all.yml @@ -0,0 +1,358 @@ +# Deploy to All Cloud Providers in Parallel +# +# This workflow orchestrates deployment to multiple cloud providers simultaneously. +# Useful for disaster recovery, multi-cloud redundancy, or testing across platforms. +# +# Required GitHub Secrets: +# - All secrets from individual deployment workflows (AWS, GCP, Azure) +# +# Required GitHub Variables: +# - All variables from individual deployment workflows +# +# Triggered by: +# - Manual workflow dispatch +# - Release creation (deploys to prod across all clouds) + +name: Deploy to All Clouds + +permissions: + contents: read + +# Adding a trigger here, or a second caller of the deploy-* workflows, means +# revisiting the `prepare` comments in deploy-azure.yml, deploy-gcp.yml and +# deploy-aws-lambda.yml. Since #1805 those three resolve their environment from +# `inputs.environment` and fall back on `github.event_name`, which inside a +# reusable workflow is whatever event started THIS run. They are correct today +# only because this workflow is their only caller, has no `push` trigger, and +# allowlists the environment it passes. A `push:` trigger here would make +# Azure and GCP silently resolve dev for a caller that asked for something +# else, which is #1805 restored. +on: + workflow_dispatch: + inputs: + environment: + description: 'Environment to deploy to' + required: true + type: choice + options: + - dev + - staging + - prod + deploy_to: + description: 'Cloud providers to deploy to' + required: true + type: choice + options: + - all + - aws-only + - gcp-only + - azure-only + - aws-gcp + - aws-azure + - gcp-azure + release: + types: [created] + +# None of the four `uses:` caller jobs below carries a `concurrency` group, and +# that is deliberate (#1801, #1806). The Terraform state guard is the group on +# each called workflow's own deploying job, which runs as a real job of THIS run +# and is serialized there. Putting the same group on a caller job would deadlock: +# the caller would hold the group while waiting on the inner job queued behind +# it. GitHub documents the sibling hazard for `cancel-in-progress: true` (sharing +# a group between caller and called cancels the already-running caller); at +# `false` it stalls instead. The per-job note on `deploy-azure` spells this out. +jobs: + # Determine deployment strategy + determine-deployment: + name: Determine Deployment Strategy + runs-on: ubuntu-latest + permissions: + contents: read + outputs: + environment: ${{ steps.set-env.outputs.environment }} + deploy-aws-lambda: ${{ steps.set-clouds.outputs.deploy-aws-lambda }} + deploy-aws-fargate: ${{ steps.set-clouds.outputs.deploy-aws-fargate }} + deploy-gcp: ${{ steps.set-clouds.outputs.deploy-gcp }} + deploy-azure: ${{ steps.set-clouds.outputs.deploy-azure }} + + steps: + - name: Set environment + id: set-env + env: + # This workflow is always the entry point of its own run, never a + # reusable one, so `github.event_name` here really is the trigger that + # started it. The called workflows cannot rely on that and read + # `inputs.environment` instead (issue #1805). + EVENT_NAME: ${{ github.event_name }} + INPUT_ENVIRONMENT: ${{ inputs.environment }} + run: | + set -euo pipefail + INPUT_ENVIRONMENT="${INPUT_ENVIRONMENT:-}" + + if [ "$EVENT_NAME" = release ]; then + ENVIRONMENT=prod + else + ENVIRONMENT="$INPUT_ENVIRONMENT" + fi + + # Every called workflow takes this value verbatim, so an empty or + # unknown value has to stop the fan-out here rather than reach four + # deployments. + case "$ENVIRONMENT" in + dev|staging|prod) ;; + *) + echo "::error::Refusing unknown environment: '$ENVIRONMENT'" + exit 1 + ;; + esac + + echo "environment=$ENVIRONMENT" >> "$GITHUB_OUTPUT" + + - name: Set cloud providers + id: set-clouds + run: | + DEPLOY_TO="${{ inputs.deploy_to || 'all' }}" + + # AWS Lambda + if [[ "$DEPLOY_TO" == "all" ]] || \ + [[ "$DEPLOY_TO" == "aws-only" ]] || \ + [[ "$DEPLOY_TO" == "aws-gcp" ]] || \ + [[ "$DEPLOY_TO" == "aws-azure" ]]; then + echo "deploy-aws-lambda=true" >> "$GITHUB_OUTPUT" + else + echo "deploy-aws-lambda=false" >> "$GITHUB_OUTPUT" + fi + + # AWS Fargate (optional, can be enabled separately) + echo "deploy-aws-fargate=false" >> "$GITHUB_OUTPUT" + + # GCP + if [[ "$DEPLOY_TO" == "all" ]] || \ + [[ "$DEPLOY_TO" == "gcp-only" ]] || \ + [[ "$DEPLOY_TO" == "aws-gcp" ]] || \ + [[ "$DEPLOY_TO" == "gcp-azure" ]]; then + echo "deploy-gcp=true" >> "$GITHUB_OUTPUT" + else + echo "deploy-gcp=false" >> "$GITHUB_OUTPUT" + fi + + # Azure + if [[ "$DEPLOY_TO" == "all" ]] || \ + [[ "$DEPLOY_TO" == "azure-only" ]] || \ + [[ "$DEPLOY_TO" == "aws-azure" ]] || \ + [[ "$DEPLOY_TO" == "gcp-azure" ]]; then + echo "deploy-azure=true" >> "$GITHUB_OUTPUT" + else + echo "deploy-azure=false" >> "$GITHUB_OUTPUT" + fi + + - name: Display deployment plan + run: | + { + echo "## Deployment Plan" + echo "" + echo "**Environment:** ${{ steps.set-env.outputs.environment }}" + echo "**Deploy Strategy:** ${{ inputs.deploy_to || 'all' }}" + echo "" + echo "### Clouds" + echo "- AWS Lambda: ${{ steps.set-clouds.outputs.deploy-aws-lambda }}" + echo "- AWS Fargate: ${{ steps.set-clouds.outputs.deploy-aws-fargate }}" + echo "- GCP Cloud Run: ${{ steps.set-clouds.outputs.deploy-gcp }}" + echo "- Azure Container Apps: ${{ steps.set-clouds.outputs.deploy-azure }}" + } >> "$GITHUB_STEP_SUMMARY" + + # Deploy to AWS Lambda + deploy-aws-lambda: + name: Deploy AWS Lambda + needs: determine-deployment + if: needs.determine-deployment.outputs.deploy-aws-lambda == 'true' + permissions: + id-token: write + contents: read + uses: ./.github/workflows/deploy-aws-lambda.yml + with: + environment: ${{ needs.determine-deployment.outputs.environment }} + secrets: + ADMIN_EMAIL: ${{ secrets.ADMIN_EMAIL }} + DASHBOARD_URL: ${{ secrets.DASHBOARD_URL }} + FROM_EMAIL: ${{ secrets.FROM_EMAIL }} + TF_BACKEND_AWS: ${{ secrets.TF_BACKEND_AWS }} + + # Deploy to AWS Fargate + deploy-aws-fargate: + name: Deploy AWS Fargate + needs: determine-deployment + if: needs.determine-deployment.outputs.deploy-aws-fargate == 'true' + permissions: + id-token: write + contents: read + uses: ./.github/workflows/deploy-aws-fargate.yml + with: + environment: ${{ needs.determine-deployment.outputs.environment }} + secrets: + ADMIN_EMAIL: ${{ secrets.ADMIN_EMAIL }} + TF_BACKEND_AWS: ${{ secrets.TF_BACKEND_AWS }} + + # Deploy to GCP Cloud Run + deploy-gcp: + name: Deploy GCP Cloud Run + needs: determine-deployment + if: needs.determine-deployment.outputs.deploy-gcp == 'true' + permissions: + id-token: write + contents: read + uses: ./.github/workflows/deploy-gcp.yml + with: + environment: ${{ needs.determine-deployment.outputs.environment }} + secrets: + ADMIN_EMAIL: ${{ secrets.ADMIN_EMAIL }} + TF_BACKEND_GCP: ${{ secrets.TF_BACKEND_GCP }} + + # Deploy to Azure Container Apps + # + # Deliberately carries no `concurrency` (#1801). The Terraform state guard is + # the group on the called workflow's own build-and-deploy job, which runs as a + # real job of this run and is serialized there. That group is derived from the + # called workflow's own `prepare` output, which since #1805 is the + # `environment` computed above verbatim: `prepare` takes the non-empty input + # it is passed and fails the run rather than substituting a default. + # + # Putting that same group on this caller job would deadlock: this job would + # hold the group while waiting on the inner job queued behind it. GitHub + # documents the sibling hazard for `cancel-in-progress: true`: sharing a group + # between caller and called cancels the already-running caller. At `false` it + # stalls instead. + deploy-azure: + name: Deploy Azure Container Apps + needs: determine-deployment + if: needs.determine-deployment.outputs.deploy-azure == 'true' + permissions: + id-token: write + contents: read + uses: ./.github/workflows/deploy-azure.yml + with: + environment: ${{ needs.determine-deployment.outputs.environment }} + secrets: + ADMIN_EMAIL: ${{ secrets.ADMIN_EMAIL }} + AZURE_CLIENT_ID: ${{ secrets.AZURE_CLIENT_ID }} + AZURE_SUBSCRIPTION_ID: ${{ secrets.AZURE_SUBSCRIPTION_ID }} + AZURE_TENANT_ID: ${{ secrets.AZURE_TENANT_ID }} + TF_BACKEND_AZURE: ${{ secrets.TF_BACKEND_AZURE }} + + # Aggregate results and notify + notify: + name: Deployment Results + needs: + - determine-deployment + - deploy-aws-lambda + - deploy-aws-fargate + - deploy-gcp + - deploy-azure + if: always() + runs-on: ubuntu-latest + permissions: + contents: read + + steps: + - name: Aggregate results + run: | + { + echo "## Multi-Cloud Deployment Results" + echo "" + echo "**Environment:** ${{ needs.determine-deployment.outputs.environment }}" + echo "**Timestamp:** $(date -u +%Y-%m-%dT%H:%M:%SZ)" + echo "" + + echo "### Deployment Status" + echo "" + + # AWS Lambda + if [[ "${{ needs.determine-deployment.outputs.deploy-aws-lambda }}" == "true" ]]; then + if [[ "${{ needs.deploy-aws-lambda.result }}" == "success" ]]; then + echo "- ✅ AWS Lambda: Success" + else + echo "- ❌ AWS Lambda: ${{ needs.deploy-aws-lambda.result }}" + fi + else + echo "- ⏭️ AWS Lambda: Skipped" + fi + + # AWS Fargate + if [[ "${{ needs.determine-deployment.outputs.deploy-aws-fargate }}" == "true" ]]; then + if [[ "${{ needs.deploy-aws-fargate.result }}" == "success" ]]; then + echo "- ✅ AWS Fargate: Success" + else + echo "- ❌ AWS Fargate: ${{ needs.deploy-aws-fargate.result }}" + fi + else + echo "- ⏭️ AWS Fargate: Skipped" + fi + + # GCP + if [[ "${{ needs.determine-deployment.outputs.deploy-gcp }}" == "true" ]]; then + if [[ "${{ needs.deploy-gcp.result }}" == "success" ]]; then + echo "- ✅ GCP Cloud Run: Success" + else + echo "- ❌ GCP Cloud Run: ${{ needs.deploy-gcp.result }}" + fi + else + echo "- ⏭️ GCP Cloud Run: Skipped" + fi + + # Azure + if [[ "${{ needs.determine-deployment.outputs.deploy-azure }}" == "true" ]]; then + if [[ "${{ needs.deploy-azure.result }}" == "success" ]]; then + echo "- ✅ Azure Container Apps: Success" + else + echo "- ❌ Azure Container Apps: ${{ needs.deploy-azure.result }}" + fi + else + echo "- ⏭️ Azure Container Apps: Skipped" + fi + } >> "$GITHUB_STEP_SUMMARY" + + - name: Check for failures + run: | + FAILED=false + + # determine-deployment is checked for anything other than success, not + # just "failure". When it stops the run the four deploy jobs are + # SKIPPED rather than failed, so testing only those would report "all + # deployments completed successfully" for a run that deployed nothing. + # + # The four deploy jobs are deliberately still tested for "failure" + # only. A CANCELLED run leaves them `cancelled` and reaches the + # success line, which is the same false report. That predates this + # change, so it is tracked in #1810 rather than fixed here. + # + # Note this step is the only success-by-default check in the repo. All + # four per-cloud `summary` jobs allowlist on `== "success"` instead, + # so an unenumerated result falls to their failure branch and errs + # safe. This denylist is the outlier, not the pattern. + if [[ "${{ needs.determine-deployment.result }}" != "success" ]] || \ + [[ "${{ needs.deploy-aws-lambda.result }}" == "failure" ]] || \ + [[ "${{ needs.deploy-aws-fargate.result }}" == "failure" ]] || \ + [[ "${{ needs.deploy-gcp.result }}" == "failure" ]] || \ + [[ "${{ needs.deploy-azure.result }}" == "failure" ]]; then + FAILED=true + fi + + if [ "$FAILED" = true ]; then + echo "" >> "$GITHUB_STEP_SUMMARY" + echo "❌ **Deployment did not complete successfully. Check individual job logs.**" >> "$GITHUB_STEP_SUMMARY" + exit 1 + else + echo "" >> "$GITHUB_STEP_SUMMARY" + echo "✅ **All deployments completed successfully!**" >> "$GITHUB_STEP_SUMMARY" + fi + + # Optional: Send notification to Slack, Discord, email, etc. + # - name: Send notification + # if: always() + # uses: slackapi/slack-github-action@91efab103c0de0a537f72a35f6b8cda0ee76bf0a # v2.1.1 + # with: + # method: chat.postMessage + # token: ${{ secrets.SLACK_BOT_TOKEN }} + # payload: | + # channel: ${{ secrets.SLACK_CHANNEL_ID }} + # text: "Multi-cloud deployment to ${{ needs.determine-deployment.outputs.environment }}: ${{ job.status }}" diff --git a/.github/workflows/deploy-aws-fargate.yml b/.github/workflows/deploy-aws-fargate.yml new file mode 100644 index 000000000..3de317f97 --- /dev/null +++ b/.github/workflows/deploy-aws-fargate.yml @@ -0,0 +1,365 @@ +# Deploy to AWS ECS Fargate +# +# This workflow deploys the CUDly application to AWS ECS Fargate with ALB. +# Alternative to Lambda for always-on containerized workloads. +# +# Required GitHub Secrets: +# - ADMIN_EMAIL: Admin email for notifications +# +# Required GitHub Variables: +# - AWS_REGION: AWS region +# - AWS_ROLE_TO_ASSUME: IAM role ARN for OIDC keyless auth +# - ECR_REPOSITORY: ECR repository name +# +# Triggered by: +# - Manual workflow dispatch only (push trigger disabled; Fargate is not the primary compute platform) +# - Workflow call from deploy-all.yml +# +# Lock cleanup is intentionally manual. If a prior run crashed and left a stuck +# lock, re-run this workflow with clear_stale_lock=true (workflow_dispatch only). +# See runbooks/terraform-stuck-lock.md for diagnosis and manual recovery steps. + +name: Deploy to AWS Fargate + +permissions: + contents: read + +# No workflow-level `concurrency` on purpose. What needs protecting is the +# Terraform state object, which is keyed on the environment, and only the +# job-level key can see `needs.prepare.outputs.environment` -- this level is +# limited to the `github`, `inputs` and `vars` contexts. Keying it on +# `github.ref` put two runs on different refs that both resolve to the same +# environment into different groups against one state file (#1806). The group +# lives on `deploy` below. +on: + workflow_dispatch: + inputs: + environment: + description: 'Environment' + required: true + type: choice + options: [dev, staging, prod] + clear_stale_lock: + description: 'Force-clear a stale S3 state lock from a prior crashed run (use only when you know the prior run is dead)' + required: false + type: boolean + default: false + workflow_call: + inputs: + environment: + required: true + type: string + image_uri: + required: false + type: string + secrets: + ADMIN_EMAIL: + required: true + TF_BACKEND_AWS: + required: true + +env: + AWS_REGION: ${{ vars.AWS_REGION || 'us-east-1' }} + TF_VERSION: '1.10.0' + FORCE_JAVASCRIPT_ACTIONS_TO_NODE24: true + +jobs: + # Determine deployment environment + prepare: + name: Prepare Deployment + runs-on: ubuntu-latest + permissions: + contents: read + outputs: + environment: ${{ steps.set-env.outputs.environment }} + image_tag: ${{ steps.set-tag.outputs.tag }} + + steps: + - name: Determine environment + id: set-env + env: + INPUT_ENVIRONMENT: ${{ inputs.environment }} + run: | + set -euo pipefail + INPUT_ENVIRONMENT="${INPUT_ENVIRONMENT:-}" + + # workflow_dispatch and workflow_call are this workflow's only + # triggers and both declare `environment` as `required: true`, so the + # input is the single source of truth and there is nothing to fall + # back to. Branching on `github.event_name` here was issue #1805: + # inside a reusable workflow it reports the CALLER's event and is + # never "workflow_call", so a release-triggered caller passing + # environment=prod would have mis-resolved to a `dev` default arm. + # deploy-all.yml hardcodes this workflow off, so the case was latent + # here and live only for Azure and GCP. + ENVIRONMENT="$INPUT_ENVIRONMENT" + + # workflow_dispatch constrains this to the declared `choice` options, + # but the workflow_call input is typed as a free-form string, so the + # allowlist is enforced here rather than assumed. + case "$ENVIRONMENT" in + dev|staging|prod) ;; + *) + echo "::error::Refusing unknown environment: '$ENVIRONMENT'" + exit 1 + ;; + esac + + echo "environment=$ENVIRONMENT" >> "$GITHUB_OUTPUT" + + - name: Set image tag + id: set-tag + run: | + echo "tag=${{ github.sha }}" >> "$GITHUB_OUTPUT" + + # Deploy with Terraform + deploy: + name: Deploy to Fargate + runs-on: ubuntu-24.04-arm + needs: prepare + permissions: + id-token: write + contents: read + # Every run that writes + # s3:///github-fargate-/terraform.tfstate serializes + # here, whatever ref or workflow it came from (#1806). This is a DIFFERENT + # state object from the Lambda one (github-/), so it takes its + # own group rather than sharing `aws-tfstate-*`; sharing would serialize two + # independent state files against each other for no gain. The suffix is the + # exact value the backend key below is built from, and `prepare` fails the + # run on anything outside dev|staging|prod so it can never be empty. + # + # `cancel-in-progress: false` is stated rather than left to the default: + # cancelling mid-`terraform apply` is how you get a half-applied stack and a + # lock nobody releases. + concurrency: + group: aws-fargate-tfstate-${{ needs.prepare.outputs.environment }} + cancel-in-progress: false + outputs: + alb_url: ${{ steps.outputs.outputs.alb_url }} + service_name: ${{ steps.outputs.outputs.service_name }} + environment: + name: aws-fargate-${{ needs.prepare.outputs.environment }} + url: ${{ steps.outputs.outputs.alb_url }} + + steps: + - name: Checkout code + uses: actions/checkout@93cb6efe18208431cddfb8368fd83d5badbf9bfd # v5.0.1 + with: + persist-credentials: false + + - name: Configure AWS credentials + uses: aws-actions/configure-aws-credentials@d979d5b3a71173a29b74b5b88418bfda9437d885 # v6.1.1 + with: + role-to-assume: ${{ vars.AWS_ROLE_TO_ASSUME }} + aws-region: ${{ env.AWS_REGION }} + + - name: Setup Terraform + uses: hashicorp/setup-terraform@dfe3c3f87815947d99a8997f908cb6525fc44e9e # v4.0.1 + with: + terraform_version: ${{ env.TF_VERSION }} + + - name: Clear stale state lock (operator-triggered only) + if: ${{ inputs.clear_stale_lock == true }} + env: + TF_BACKEND: ${{ secrets.TF_BACKEND_AWS }} + ENVIRONMENT: ${{ needs.prepare.outputs.environment }} + run: | + printf '%s\nkey = "github-fargate-%s/terraform.tfstate"\n' "$TF_BACKEND" "$ENVIRONMENT" > /tmp/backend.tfbackend + BUCKET=$(grep -E '^\s*bucket\s*=' /tmp/backend.tfbackend 2>/dev/null | tr -d ' "' | cut -d= -f2) + if [ -n "$BUCKET" ]; then + LOCK_KEY="github-fargate-${ENVIRONMENT}/terraform.tfstate.tflock" + echo "Removing stale lock: s3://${BUCKET}/${LOCK_KEY}" + aws s3 rm "s3://${BUCKET}/${LOCK_KEY}" 2>/dev/null || echo "No lock file found (already clean)" + else + echo "Could not parse bucket from backend config; skipping lock clear" + fi + + - name: Terraform Init + env: + TF_BACKEND: ${{ secrets.TF_BACKEND_AWS }} + ENVIRONMENT: ${{ needs.prepare.outputs.environment }} + run: | + printf '%s\nkey = "github-fargate-%s/terraform.tfstate"\n' "$TF_BACKEND" "$ENVIRONMENT" > /tmp/backend.tfbackend + cd terraform/environments/aws + terraform init -backend-config=/tmp/backend.tfbackend + + - name: Terraform Plan + env: + TF_VAR_admin_email: ${{ secrets.ADMIN_EMAIL }} + ENVIRONMENT: ${{ needs.prepare.outputs.environment }} + run: | + cd terraform/environments/aws + terraform plan \ + -var-file="github-${ENVIRONMENT}.tfvars" \ + -var="compute_platform=fargate" \ + -out=tfplan + + - name: Terraform Apply + env: + TF_VAR_admin_email: ${{ secrets.ADMIN_EMAIL }} + run: | + cd terraform/environments/aws + terraform apply -auto-approve tfplan + + # No automatic state-lock release on failure. A blanket failure-path + # delete is itself a state-corruption risk: a run that fails *because* + # it could not acquire the lock would delete the lock held by another + # run that is still actively applying. Terraform already releases its + # own lock on a clean apply error; a lock that survives a run means the + # run died abnormally, which requires operator confirmation that the + # owning run is dead before clearing it via the clear_stale_lock input. + # See runbooks/terraform-stuck-lock.md. + + - name: Get outputs + id: outputs + run: | + cd terraform/environments/aws + echo "alb_url=$(terraform output -raw fargate_alb_dns_name 2>/dev/null | grep -v '::' || echo "")" >> "$GITHUB_OUTPUT" + echo "service_name=$(terraform output -raw fargate_service_name 2>/dev/null | grep -v '::' || echo "")" >> "$GITHUB_OUTPUT" + + - name: Save deployment info + env: + ENVIRONMENT: ${{ needs.prepare.outputs.environment }} + IMAGE_TAG: ${{ needs.prepare.outputs.image_tag }} + ALB_URL: ${{ steps.outputs.outputs.alb_url }} + SERVICE_NAME: ${{ steps.outputs.outputs.service_name }} + DEPLOYED_BY: ${{ github.actor }} + COMMIT: ${{ github.sha }} + run: | + set -euo pipefail + jq -n \ + --arg environment "$ENVIRONMENT" \ + --arg image_tag "$IMAGE_TAG" \ + --arg alb_url "$ALB_URL" \ + --arg service_name "$SERVICE_NAME" \ + --arg deployed_at "$(date -u +%Y-%m-%dT%H:%M:%SZ)" \ + --arg deployed_by "$DEPLOYED_BY" \ + --arg commit "$COMMIT" \ + '{environment: $environment, image_tag: $image_tag, alb_url: $alb_url, service_name: $service_name, deployed_at: $deployed_at, deployed_by: $deployed_by, commit: $commit}' \ + > deployment-info.json + cat deployment-info.json + + - name: Upload deployment info + uses: actions/upload-artifact@b7c566a772e6b6bfb58ed0dc250532a479d7789f # v6.0.0 + with: + name: deployment-info-fargate-${{ needs.prepare.outputs.environment }} + path: deployment-info.json + retention-days: 90 + + # Test deployment + test-deployment: + name: Test Deployment + runs-on: ubuntu-latest + needs: [prepare, deploy] + if: always() && needs.deploy.result == 'success' + permissions: + contents: read + + steps: + - name: Get ALB URL + id: get-url + run: | + ALB_URL="${{ needs.deploy.outputs.alb_url }}" + if [ -z "$ALB_URL" ]; then + echo "Failed to get ALB URL from deploy outputs" + exit 1 + fi + # Ensure URL has scheme and no trailing slash + case "$ALB_URL" in + http://*|https://*) ;; # already has scheme + *) ALB_URL="http://$ALB_URL" ;; + esac + ALB_URL="${ALB_URL%/}" + echo "url=$ALB_URL" >> "$GITHUB_OUTPUT" + + - name: Wait for deployment + run: | + echo "Waiting 60 seconds for ECS service to stabilize..." + sleep 60 + + - name: Test health endpoint + run: | + URL="${{ steps.get-url.outputs.url }}" + echo "Testing health endpoint: $URL/health" + + for i in {1..10}; do + if curl -f -s "$URL/health" > /dev/null; then + RESPONSE=$(curl -s "$URL/health") + echo "Response: $RESPONSE" + if echo "$RESPONSE" | grep -q '"status"'; then + echo "Health check passed!" + exit 0 + fi + fi + echo "Attempt $i failed, retrying in 10 seconds..." + sleep 10 + done + + echo "Health check failed after 10 attempts" + exit 1 + + - name: Run smoke tests + run: | + URL="${{ steps.get-url.outputs.url }}" + + echo "Running basic smoke tests..." + + for i in {1..3}; do + STATUS=$(curl -s -o /dev/null -w "%{http_code}" "$URL/health") + if [ "$STATUS" -eq 200 ]; then + echo "Health check $i: passed (HTTP $STATUS)" + else + echo "Health check $i: failed (HTTP $STATUS)" + exit 1 + fi + sleep 2 + done + + echo "All smoke tests passed!" + + # Summary + summary: + name: Deployment Summary + runs-on: ubuntu-latest + needs: [prepare, deploy, test-deployment] + if: always() + permissions: + contents: read + + steps: + - name: Download deployment info + uses: actions/download-artifact@37930b1c2abaa49bbe596cd826c3c89aef350131 # v7.0.0 + with: + name: deployment-info-fargate-${{ needs.prepare.outputs.environment }} + continue-on-error: true + + - name: Post summary + run: | + { + echo "## AWS Fargate Deployment Summary" + echo "" + echo "**Environment:** ${{ needs.prepare.outputs.environment }}" + echo "**Deploy Status:** ${{ needs.deploy.result }}" + echo "" + + if [ -f deployment-info.json ]; then + echo "### Deployment Details" + echo "\`\`\`json" + cat deployment-info.json + echo "\`\`\`" + fi + + echo "" + echo "### Job Results" + echo "- Deploy: ${{ needs.deploy.result }}" + echo "- Test: ${{ needs.test-deployment.result }}" + + if [ "${{ needs.deploy.result }}" == "success" ] && [ "${{ needs.test-deployment.result }}" == "success" ]; then + echo "" + echo "**Deployment successful!**" + else + echo "" + echo "**Deployment failed. Check logs for details.**" + fi + } >> "$GITHUB_STEP_SUMMARY" diff --git a/.github/workflows/deploy-aws-lambda.yml b/.github/workflows/deploy-aws-lambda.yml new file mode 100644 index 000000000..2932a72d0 --- /dev/null +++ b/.github/workflows/deploy-aws-lambda.yml @@ -0,0 +1,510 @@ +# Deploy to AWS Lambda +# +# This workflow deploys the CUDly application to AWS Lambda with Function URL. +# It builds a Docker image, pushes to ECR, and deploys via Terraform. +# +# Required GitHub Secrets: +# - ADMIN_EMAIL: Admin email for notifications +# - FROM_EMAIL: SES-verified FROM address used for outbound approval emails. +# Required for purchase-approval emails on deployments that +# don't set subdomain_zone_name (bare Lambda Function URL). +# Must be verified in the target AWS account's SES identities. +# +# Required GitHub Variables: +# - AWS_REGION: AWS region (e.g., us-east-1) +# - AWS_ROLE_TO_ASSUME: IAM role ARN for OIDC keyless auth +# +# Triggered by: +# - Pushes to main branch +# - Manual workflow dispatch with environment selection +# - Release creation (auto-deploy to prod) +# - Workflow call from other workflows + +name: Deploy to AWS Lambda + +# Least privilege by default: `id-token: write` is granted per job, only to the +# jobs that actually assume the AWS deploy role, and both of those are bound to +# a deployment environment. Declaring it here would hand it to `prepare` and +# `summary` too, neither of which authenticates and neither of which is bound +# to an environment. +permissions: + contents: read + +# No workflow-level `concurrency` on purpose. What needs protecting is the +# Terraform state object, which is keyed on the environment, and only the +# job-level key can see `needs.prepare.outputs.target_environment` -- this level +# is limited to the `github`, `inputs` and `vars` contexts. Keying it on +# `github.ref` put two runs on different refs that both resolve to the same +# environment into different groups against one state file (#1806). The group +# lives on `build-and-deploy` below. +on: + push: + branches: [main] + paths: + - 'cmd/**' + - 'internal/**' + - 'providers/**' + - 'frontend/**' + - 'Dockerfile' + - 'terraform/modules/compute/aws/lambda/**' + - 'terraform/modules/database/aws/**' + - 'terraform/modules/secrets/aws/**' + - 'terraform/modules/networking/aws/**' + - 'terraform/environments/aws/**' + - '.github/workflows/deploy-aws-lambda.yml' + workflow_dispatch: + inputs: + environment: + description: 'Deployment environment' + required: true + type: choice + options: + - dev + - staging + - prod + release: + types: [created] + workflow_call: + inputs: + environment: + required: true + type: string + secrets: + ADMIN_EMAIL: + required: true + DASHBOARD_URL: + required: false + FROM_EMAIL: + required: false + TF_BACKEND_AWS: + required: true + +env: + AWS_REGION: ${{ vars.AWS_REGION || 'us-east-1' }} + TF_VERSION: '1.10.0' + FORCE_JAVASCRIPT_ACTIONS_TO_NODE24: true + +jobs: + # Determine deployment environment + prepare: + name: Prepare Deployment + runs-on: ubuntu-latest + permissions: + contents: read + # NOTE: `target_environment` below is an OUTPUT, not an `environment:` + # binding — this job is deliberately ungated and therefore must never hold + # `id-token: write`. It was previously named `environment`, which made the + # `outputs:` block read like a gate at a glance while gating nothing. Do not + # rename it back. + outputs: + target_environment: ${{ steps.set-env.outputs.target_environment }} + image_tag: ${{ steps.set-tag.outputs.tag }} + + steps: + - name: Determine environment + id: set-env + env: + EVENT_NAME: ${{ github.event_name }} + INPUT_ENVIRONMENT: ${{ inputs.environment }} + run: | + set -euo pipefail + INPUT_ENVIRONMENT="${INPUT_ENVIRONMENT:-}" + + # Branch on whether an environment was REQUESTED, not on the event + # name. Inside a reusable workflow `github.event_name` is the CALLER's + # event and is never "workflow_call", so no event name can identify a + # workflow_call invocation (issue #1805). A non-empty input is always + # an explicit request and wins. + # + # An empty input means none was supplied, which this workflow's own + # `release` and `push` triggers are the legitimate causes of; every + # other event stops the run. Note that `release` is inherited by a + # workflow_call from a release-triggered caller, so a caller passing + # an EMPTY environment would resolve to prod here rather than + # failing. deploy-all.yml is the only caller and allowlists the value + # it passes, so that cannot happen. Telling the two apart would need + # `github.job_workflow_ref`, which is not worth the machinery for an + # unreachable path. + if [ -n "$INPUT_ENVIRONMENT" ]; then + TARGET_ENVIRONMENT="$INPUT_ENVIRONMENT" + elif [ "$EVENT_NAME" = release ]; then + TARGET_ENVIRONMENT=prod + elif [ "$EVENT_NAME" = push ]; then + TARGET_ENVIRONMENT=dev + else + echo "::error::No environment supplied on a '$EVENT_NAME'-triggered run; refusing to guess a deployment target." + exit 1 + fi + + # workflow_dispatch constrains this to the declared `choice` options, + # but the workflow_call input is typed as a free-form string, so the + # allowlist is enforced here rather than assumed. + case "$TARGET_ENVIRONMENT" in + dev|staging|prod) ;; + *) + echo "::error::Refusing unknown environment: '$TARGET_ENVIRONMENT'" + exit 1 + ;; + esac + + echo "target_environment=$TARGET_ENVIRONMENT" >> "$GITHUB_OUTPUT" + + - name: Set image tag + id: set-tag + env: + EVENT_NAME: ${{ github.event_name }} + RELEASE_TAG: ${{ github.event.release.tag_name }} + COMMIT_SHA: ${{ github.sha }} + run: | + set -euo pipefail + + if [ "$EVENT_NAME" = "release" ]; then + TAG="${RELEASE_TAG:-}" + else + TAG="$COMMIT_SHA" + fi + + # Shell metacharacters in a tag (';', '$', '`', '|') are already inert: + # the value arrives via `env:` and is only ever referenced quoted, so + # it is data, not code. What a charset filter must actually stop is a + # NEWLINE, because this value is written to $GITHUB_OUTPUT as a single + # `tag=` line — a newline would let a crafted tag append extra + # attacker-chosen output keys, and those outputs are consumed by the + # two credentialed downstream jobs. + # + # So reject control characters and reject empty, and allow the rest of + # the git ref charset. A stricter OCI-grammar filter would reject + # `release/1.0` and `v1.2.3+build.5` — both legitimate git tags, and + # this repo already uses slash-namespaced ones — hard-failing a + # production deploy for no security gain. + # + # NOTE: this deliberately does NOT enforce the OCI tag grammar, + # because the value is cosmetic today: it reaches only + # deployment-info.json (via `jq --arg`) and the step summary. + # Terraform derives the real image tag from the git commit in + # modules/build. If this is ever wired to `custom_image_tag`, add an + # OCI-grammar check HERE at the same time. + if [ -z "$TAG" ]; then + echo "::error::Refusing empty image tag" + exit 1 + fi + if [[ "$TAG" =~ [[:cntrl:]] ]]; then + echo "::error::Refusing image tag containing control characters or newlines" + exit 1 + fi + + echo "tag=$TAG" >> "$GITHUB_OUTPUT" + + # Deploy infrastructure with Terraform (includes Docker build via build module) + build-and-deploy: + name: Build & Deploy + runs-on: ubuntu-24.04-arm # ARM runner: Docker build targets linux/arm64 (Lambda/Fargate Graviton2) + needs: prepare + permissions: + # Assumes the AWS deploy role. The `environment:` binding below is what + # scopes this job's OIDC `sub` to `repo::environment:`, + # which is the subject the trust policy matches. NOTE it is not by itself + # a reviewer gate: an Environment only blocks a job once required-reviewer + # protection rules are configured on it in repo settings, and as of this + # change none of this repo's Environments have any. See #1648. + id-token: write + contents: read + # Every run that writes s3:///github-/terraform.tfstate + # serializes here, whatever ref or workflow it came from (#1806). The suffix + # is the exact value the backend key below is built from, so the group and + # the state object cannot drift apart, and `prepare` fails the run on + # anything outside dev|staging|prod so it can never be empty. + # + # `cancel-in-progress: false` is stated rather than left to the default: + # cancelling mid-`terraform apply` is how you get a half-applied stack and a + # lock nobody releases. + concurrency: + group: aws-tfstate-${{ needs.prepare.outputs.target_environment }} + cancel-in-progress: false + # Bind to the named GitHub Environment matching the target so + # secrets.* resolve to environment-scoped values when defined, + # falling back to repo-scoped secrets otherwise. Without this, + # DASHBOARD_URL (and any other per-env secret like FROM_EMAIL, + # ADMIN_EMAIL) would be repo-wide and dev/staging/prod would + # all share one value — producing wrong email links in customer + # mailboxes. Per CR review on PR #368. + environment: ${{ needs.prepare.outputs.target_environment }} + outputs: + function_url: ${{ steps.outputs.outputs.function_url }} + function_name: ${{ steps.outputs.outputs.function_name }} + log_group: ${{ steps.outputs.outputs.log_group }} + + steps: + - name: Checkout code + uses: actions/checkout@93cb6efe18208431cddfb8368fd83d5badbf9bfd # v5.0.1 + with: + persist-credentials: false + + - name: Configure AWS credentials + uses: aws-actions/configure-aws-credentials@d979d5b3a71173a29b74b5b88418bfda9437d885 # v6.1.1 + with: + role-to-assume: ${{ vars.AWS_ROLE_TO_ASSUME }} + aws-region: ${{ env.AWS_REGION }} + + - name: Setup Terraform + uses: hashicorp/setup-terraform@dfe3c3f87815947d99a8997f908cb6525fc44e9e # v4.0.1 + with: + terraform_version: ${{ env.TF_VERSION }} + + - name: Terraform Init + env: + TF_BACKEND: ${{ secrets.TF_BACKEND_AWS }} + ENVIRONMENT: ${{ needs.prepare.outputs.target_environment }} + run: | + printf '%s\nkey = "github-%s/terraform.tfstate"\n' "$TF_BACKEND" "$ENVIRONMENT" > /tmp/backend.tfbackend + cd terraform/environments/aws + terraform init -backend-config=/tmp/backend.tfbackend + + - name: Terraform Plan + id: plan + env: + TF_VAR_admin_email: ${{ secrets.ADMIN_EMAIL }} + TF_VAR_from_email: ${{ secrets.FROM_EMAIL }} + # DASHBOARD_URL is the customer-facing dashboard origin used in + # email links (invite, password reset, welcome). For Lambda- + # Function-URL deployments (no custom domain), set the repo + # secret DASHBOARD_URL to the function URL — see #363 + + # github-dev.tfvars for the bootstrap pattern. Empty string is + # accepted by the var (the Terraform local then falls back to + # frontend_domain_names[0]). + TF_VAR_dashboard_url: ${{ secrets.DASHBOARD_URL }} + ENVIRONMENT: ${{ needs.prepare.outputs.target_environment }} + run: | + cd terraform/environments/aws + terraform plan \ + -var-file="github-${ENVIRONMENT}.tfvars" \ + -var="compute_platform=lambda" \ + -out=tfplan + + - name: Terraform Apply + env: + TF_VAR_admin_email: ${{ secrets.ADMIN_EMAIL }} + TF_VAR_from_email: ${{ secrets.FROM_EMAIL }} + # DASHBOARD_URL is the customer-facing dashboard origin used in + # email links (invite, password reset, welcome). For Lambda- + # Function-URL deployments (no custom domain), set the repo + # secret DASHBOARD_URL to the function URL — see #363 + + # github-dev.tfvars for the bootstrap pattern. Empty string is + # accepted by the var (the Terraform local then falls back to + # frontend_domain_names[0]). + TF_VAR_dashboard_url: ${{ secrets.DASHBOARD_URL }} + run: | + cd terraform/environments/aws + terraform apply -auto-approve tfplan + + # No automatic state-lock release on failure. The step that used to live + # here ran on `failure() || cancelled()` with no age check and no check + # that the lock was this run's, so it deleted whatever lock object was + # there. The `cancelled()` half is the decisive one: GitHub runs those + # steps while `terraform apply` is still shutting down and may still be + # writing state, so it destroyed a lock out from under an active writer. + # A run that failed *because* it could not acquire the lock would also + # have deleted the lock held by the run still applying. Removed with + # #1806, matching the shape deploy-aws-fargate.yml already uses. + # Terraform releases its own lock on a clean apply error; a lock that + # survives a run means the run died abnormally, which needs operator + # confirmation that the owning run is dead. Recovery is + # `terraform force-unlock ` -- see runbooks/terraform-stuck-lock.md. + + - name: Get Terraform outputs + id: outputs + run: | + cd terraform/environments/aws + { + echo "function_url=$(terraform output -raw lambda_function_url 2>/dev/null | grep -v '::' || echo "")" + echo "function_name=$(terraform output -raw lambda_function_name 2>/dev/null | grep -v '::' || echo "")" + echo "log_group=$(terraform output -raw lambda_log_group_name 2>/dev/null | grep -v '::' || echo "")" + } >> "$GITHUB_OUTPUT" + + - name: Save deployment info + env: + ENVIRONMENT: ${{ needs.prepare.outputs.target_environment }} + IMAGE_TAG: ${{ needs.prepare.outputs.image_tag }} + FUNCTION_URL: ${{ steps.outputs.outputs.function_url }} + FUNCTION_NAME: ${{ steps.outputs.outputs.function_name }} + DEPLOYED_BY: ${{ github.actor }} + COMMIT: ${{ github.sha }} + run: | + set -euo pipefail + + # Built with jq rather than an unquoted `cat < deployment-info.json + + cat deployment-info.json + + - name: Upload deployment info + uses: actions/upload-artifact@b7c566a772e6b6bfb58ed0dc250532a479d7789f # v6.0.0 + with: + name: deployment-info-lambda-${{ needs.prepare.outputs.target_environment }} + path: deployment-info.json + retention-days: 90 + + # Test the deployment + test-deployment: + name: Test Deployment + runs-on: ubuntu-latest + needs: [prepare, build-and-deploy] + if: always() && needs.build-and-deploy.result == 'success' + permissions: + # Assumes the AWS deploy role. The `environment:` binding below is what + # scopes this job's OIDC `sub` to `repo::environment:`, + # which is the subject the trust policy matches. NOTE it is not by itself + # a reviewer gate: an Environment only blocks a job once required-reviewer + # protection rules are configured on it in repo settings, and as of this + # change none of this repo's Environments have any. See #1648. + id-token: write + contents: read + # Bind to the same named environment as build-and-deploy so that + # vars.AWS_ROLE_TO_ASSUME resolves to the per-environment scoped value. + environment: ${{ needs.prepare.outputs.target_environment }} + + steps: + - name: Configure AWS credentials + uses: aws-actions/configure-aws-credentials@d979d5b3a71173a29b74b5b88418bfda9437d885 # v6.1.1 — pinned per SC review + with: + role-to-assume: ${{ vars.AWS_ROLE_TO_ASSUME }} + aws-region: ${{ env.AWS_REGION }} + + - name: Get Function URL + id: get-url + env: + FUNCTION_NAME: ${{ needs.build-and-deploy.outputs.function_name }} + run: | + if [ -z "$FUNCTION_NAME" ]; then + echo "Failed to get function name from deploy outputs" + exit 1 + fi + FUNCTION_URL=$(aws lambda get-function-url-config \ + --function-name "$FUNCTION_NAME" \ + --query 'FunctionUrl' \ + --output text) + if [ -z "$FUNCTION_URL" ] || [ "$FUNCTION_URL" = "None" ]; then + echo "Failed to get function URL from AWS Lambda API for $FUNCTION_NAME" + exit 1 + fi + echo "url=$FUNCTION_URL" >> "$GITHUB_OUTPUT" + + - name: Wait for Lambda to be ready + run: | + echo "Waiting 30 seconds for Lambda to be fully ready..." + sleep 30 + + - name: Test health endpoint + env: + FUNCTION_URL_RAW: ${{ steps.get-url.outputs.url }} + run: | + URL="${FUNCTION_URL_RAW%/}" + echo "Testing health endpoint: $URL/health" + + for i in {1..5}; do + if curl -f -s "$URL/health" > /dev/null; then + RESPONSE=$(curl -s "$URL/health") + echo "Response: $RESPONSE" + if echo "$RESPONSE" | grep -q '"status"'; then + echo "Health check passed!" + exit 0 + fi + fi + echo "Attempt $i failed, retrying in 10 seconds..." + sleep 10 + done + + echo "Health check failed after 5 attempts" + exit 1 + + - name: Run smoke tests + env: + FUNCTION_URL_RAW: ${{ steps.get-url.outputs.url }} + run: | + URL="${FUNCTION_URL_RAW%/}" + + echo "Running basic smoke tests..." + + for i in {1..3}; do + STATUS=$(curl -s -o /dev/null -w "%{http_code}" "$URL/health") + if [ "$STATUS" -eq 200 ]; then + echo "Health check $i: passed (HTTP $STATUS)" + else + echo "Health check $i: failed (HTTP $STATUS)" + exit 1 + fi + sleep 2 + done + + echo "All smoke tests passed!" + + # Post deployment summary + summary: + name: Deployment Summary + runs-on: ubuntu-latest + needs: [prepare, build-and-deploy, test-deployment] + if: always() + permissions: + contents: read + + steps: + - name: Download deployment info + uses: actions/download-artifact@37930b1c2abaa49bbe596cd826c3c89aef350131 # v7.0.0 + with: + name: deployment-info-lambda-${{ needs.prepare.outputs.target_environment }} + continue-on-error: true + + # Every value arrives via `env:`, including the ones whose safety is + # currently guaranteed by a check in a DIFFERENT job. Relying on "an + # upstream job validated this" makes the rule "raw interpolation is fine + # when someone else checked" — an implicit invariant that breaks silently + # the moment the upstream guard moves. One uniform rule instead. + - name: Post summary + env: + TARGET_ENVIRONMENT: ${{ needs.prepare.outputs.target_environment }} + IMAGE_TAG: ${{ needs.prepare.outputs.image_tag }} + DEPLOY_RESULT: ${{ needs.build-and-deploy.result }} + TEST_RESULT: ${{ needs.test-deployment.result }} + run: | + set -euo pipefail + + { + echo "## AWS Lambda Deployment Summary" + echo "" + echo "**Environment:** $TARGET_ENVIRONMENT" + echo "**Image Tag:** $IMAGE_TAG" + echo "**Status:** $DEPLOY_RESULT" + echo "" + + if [ -f deployment-info.json ]; then + echo "### Deployment Details" + echo '```json' + cat deployment-info.json + echo '```' + fi + + echo "" + echo "### Job Results" + echo "- Deploy: $DEPLOY_RESULT" + echo "- Test: $TEST_RESULT" + echo "" + + if [ "$DEPLOY_RESULT" = "success" ] && [ "$TEST_RESULT" = "success" ]; then + echo "**Deployment successful!**" + else + echo "**Deployment failed. Check logs for details.**" + fi + } >> "$GITHUB_STEP_SUMMARY" diff --git a/.github/workflows/deploy-azure.yml b/.github/workflows/deploy-azure.yml new file mode 100644 index 000000000..92f1348c5 --- /dev/null +++ b/.github/workflows/deploy-azure.yml @@ -0,0 +1,507 @@ +# Deploy to Azure Container Apps +# +# This workflow deploys the CUDly application to Azure Container Apps. +# Serverless container platform with automatic scaling and built-in HTTPS. +# +# Required GitHub Secrets: +# - AZURE_CLIENT_ID: App registration client ID for OIDC keyless auth +# - AZURE_TENANT_ID: Azure AD tenant ID for OIDC keyless auth +# - AZURE_SUBSCRIPTION_ID: Azure subscription ID +# - ADMIN_EMAIL: Admin email for notifications +# +# Required GitHub Variables: +# - AZURE_LOCATION: Azure region (e.g., eastus) +# - ACR_NAME: Azure Container Registry name +# - RESOURCE_GROUP: Azure resource group name +# - KEY_VAULT_NAME: Azure Key Vault name +# +# Triggered by: +# - Pushes to main branch +# - Manual workflow dispatch +# - Workflow call from deploy-all.yml + +name: Deploy to Azure Container Apps + +permissions: + contents: read + +# No workflow-level `concurrency` on purpose. What needs protecting is the +# Terraform state blob, which is keyed on the environment, and only the +# job-level key can see `needs.prepare.outputs.environment` -- this level is +# limited to the `github`, `inputs` and `vars` contexts. Keying it on +# `github.ref` here was issue #1801: two refs resolving to the same environment +# landed in different groups and applied against one state file. See +# `build-and-deploy` below. + +on: + push: + branches: [main] + workflow_dispatch: + inputs: + environment: + description: 'Environment' + required: true + type: choice + options: [dev, staging, prod] + workflow_call: + inputs: + environment: + required: true + type: string + secrets: + ADMIN_EMAIL: + required: true + AZURE_CLIENT_ID: + required: true + AZURE_SUBSCRIPTION_ID: + required: true + AZURE_TENANT_ID: + required: true + TF_BACKEND_AZURE: + required: true + +env: + AZURE_LOCATION: ${{ vars.AZURE_LOCATION || 'westus2' }} + ACR_NAME: ${{ vars.ACR_NAME || 'cudlyacr' }} + RESOURCE_GROUP: ${{ vars.RESOURCE_GROUP || 'cudly-rg' }} + TF_VERSION: '1.10.0' + FORCE_JAVASCRIPT_ACTIONS_TO_NODE24: true + +jobs: + # Determine deployment environment + prepare: + name: Prepare Deployment + runs-on: ubuntu-latest + permissions: + contents: read + outputs: + environment: ${{ steps.set-env.outputs.environment }} + steps: + - name: Determine environment + id: set-env + env: + EVENT_NAME: ${{ github.event_name }} + INPUT_ENVIRONMENT: ${{ inputs.environment }} + run: | + set -euo pipefail + INPUT_ENVIRONMENT="${INPUT_ENVIRONMENT:-}" + + # Branch on whether an environment was REQUESTED, not on the event + # name. Inside a reusable workflow `github.event_name` is the CALLER's + # event and is never "workflow_call", so no event name can identify a + # workflow_call invocation: a release-triggered deploy-all.yml passing + # environment=prod would have landed on a `dev` default arm and + # resolved to dev while the summary said prod (issue #1805). + # + # `inputs` is populated on workflow_dispatch and workflow_call, the + # only two triggers that carry an environment, so a non-empty input is + # always an explicit request and wins. An empty input means none was + # supplied: under `push` that is this workflow's own main-branch + # deploy and means dev, and every other event stops the run rather + # than guessing. A workflow_call from a push-triggered caller would + # read as `push` here too, since `github.event_name` is inherited, but + # deploy-all.yml is the only caller, has no push trigger, and + # allowlists the value it passes. + if [ -n "$INPUT_ENVIRONMENT" ]; then + ENVIRONMENT="$INPUT_ENVIRONMENT" + elif [ "$EVENT_NAME" = push ]; then + ENVIRONMENT=dev + else + echo "::error::No environment supplied on a '$EVENT_NAME'-triggered run; refusing to guess a deployment target." + exit 1 + fi + + # workflow_dispatch constrains this to the declared `choice` options, + # but the workflow_call input is typed as a free-form string, so the + # allowlist is enforced here rather than assumed. + case "$ENVIRONMENT" in + dev|staging|prod) ;; + *) + echo "::error::Refusing unknown environment: '$ENVIRONMENT'" + exit 1 + ;; + esac + + echo "environment=$ENVIRONMENT" >> "$GITHUB_OUTPUT" + + # Build Docker image (via Terraform build module) and deploy + build-and-deploy: + name: Build & Deploy + runs-on: ubuntu-latest + needs: prepare + permissions: + id-token: write + contents: read + # Every run that writes github-.terraform.tfstate serializes + # here, whatever its ref and whatever workflow it came from. The group name + # is a 1:1 function of the state key, and the same literal prefix is used by + # cleanup-staging.yml and rollback.yml, which write the same blob. + # + # `prepare` rejects anything outside dev|staging|prod and fails the run, so + # this expression cannot evaluate to an empty suffix while this job runs -- + # that would collapse every environment into one group. + # + # cancel-in-progress stays false, stated explicitly rather than left to the + # default: cancelling mid-`terraform apply` is how you get a half-applied + # stack and a lease nobody releases. + concurrency: + group: azure-tfstate-${{ needs.prepare.outputs.environment }} + cancel-in-progress: false + outputs: + app_url: ${{ steps.deploy.outputs.app_url }} + app_name: ${{ steps.deploy.outputs.app_name }} + env: + ARM_USE_OIDC: "true" + ARM_CLIENT_ID: ${{ secrets.AZURE_CLIENT_ID }} + ARM_TENANT_ID: ${{ secrets.AZURE_TENANT_ID }} + ARM_SUBSCRIPTION_ID: ${{ secrets.AZURE_SUBSCRIPTION_ID }} + + steps: + - name: Checkout code + uses: actions/checkout@93cb6efe18208431cddfb8368fd83d5badbf9bfd # v5.0.1 + with: + persist-credentials: false + + - name: Azure Login + uses: azure/login@532459ea530d8321f2fb9bb10d1e0bcf23869a43 # v3.0.0 + with: + client-id: ${{ secrets.AZURE_CLIENT_ID }} + tenant-id: ${{ secrets.AZURE_TENANT_ID }} + subscription-id: ${{ secrets.AZURE_SUBSCRIPTION_ID }} + + - name: Setup Terraform + uses: hashicorp/setup-terraform@dfe3c3f87815947d99a8997f908cb6525fc44e9e # v4.0.1 + with: + terraform_version: ${{ env.TF_VERSION }} + + # This step used to also run `az storage blob lease break` on the state + # blob, with no lease-age check and no --lease-break-period, so it broke a + # *live* lease immediately. Removed with #1801: the lease is the last line + # of defence against two writers, and a loud "Error acquiring the state + # lock" is the correct outcome of a real collision. A genuinely stranded + # lease (runner killed mid-apply) is recovered with `terraform + # force-unlock `, using the lock ID Terraform prints in that error. + - name: Write Terraform backend config + env: + TF_BACKEND: ${{ secrets.TF_BACKEND_AZURE }} + ENVIRONMENT: ${{ needs.prepare.outputs.environment }} + run: | + printf '%s\nkey = "github-%s.terraform.tfstate"\n' "$TF_BACKEND" "$ENVIRONMENT" > /tmp/backend.tfbackend + + - name: Terraform Init + run: | + cd terraform/environments/azure + terraform init -backend-config=/tmp/backend.tfbackend + + - name: Terraform Plan + env: + TF_VAR_admin_email: ${{ secrets.ADMIN_EMAIL }} + TF_VAR_subscription_id: ${{ secrets.AZURE_SUBSCRIPTION_ID }} + run: | + # pipefail for the same reason as the apply step below: the default + # shell is `bash -e`, where a pipeline takes tee's exit status and a + # failed plan would be reported as success. + set -o pipefail + cd terraform/environments/azure + terraform plan \ + -var-file="github-${{ needs.prepare.outputs.environment }}.tfvars" \ + -var="location=${{ env.AZURE_LOCATION }}" \ + -out=tfplan 2>&1 | tee "${RUNNER_TEMP}/tf-plan.log" + + + - name: Wait for resource group deletion to complete + run: | + ENVIRONMENT="${{ needs.prepare.outputs.environment }}" + RG_NAME="cudly-${ENVIRONMENT}-rg" + echo "Checking resource group ${RG_NAME} state..." + for i in $(seq 1 30); do + STATE=$(az group show --name "${RG_NAME}" --query "properties.provisioningState" -o tsv 2>/dev/null || echo "NotFound") + if [ "$STATE" = "Deleting" ]; then + echo " Resource group is being deleted ($i/30), waiting 30s..." + sleep 30 + else + echo " Resource group state: ${STATE} — proceeding" + break + fi + if [ "$i" -eq 30 ]; then + echo "Timed out waiting for resource group deletion" + exit 1 + fi + done + + - name: Remove orphaned or soft-deleted Key Vault + run: | + ENVIRONMENT="${{ needs.prepare.outputs.environment }}" + KV_NAME="cudly-${ENVIRONMENT}-kv" + # If the KV exists in Azure but not in Terraform state (orphaned from a failed run), + # delete it so Terraform can re-create it cleanly + KV_RG=$(az keyvault show --name "${KV_NAME}" --query "resourceGroup" -o tsv 2>/dev/null || echo "") + if [ -n "$KV_RG" ]; then + echo "Key Vault ${KV_NAME} exists in Azure (rg: ${KV_RG}) — deleting so Terraform can recreate it" + az keyvault delete --name "${KV_NAME}" --resource-group "${KV_RG}" 2>/dev/null || true + fi + # Purge any soft-deleted Key Vault (including the one we just deleted) + # so the name can be immediately reused + DELETED=$(az keyvault list-deleted --query "[?name=='${KV_NAME}'].name" -o tsv 2>/dev/null || echo "") + if [ -n "$DELETED" ]; then + echo "Purging soft-deleted Key Vault ${KV_NAME}..." + az keyvault purge --name "${KV_NAME}" --location "${{ env.AZURE_LOCATION }}" 2>/dev/null \ + && echo "Purged" || echo "Purge failed (continuing)" + fi + + - name: Terraform Apply + id: deploy + env: + TF_VAR_admin_email: ${{ secrets.ADMIN_EMAIL }} + TF_VAR_subscription_id: ${{ secrets.AZURE_SUBSCRIPTION_ID }} + run: | + # pipefail is required: the default shell is `bash -e`, where a pipeline + # takes tee's exit status and a failed apply would be reported as success. + set -o pipefail + cd terraform/environments/azure + terraform apply -auto-approve tfplan 2>&1 | tee "${RUNNER_TEMP}/tf-apply.log" + + # Get app URL + APP_URL=$(terraform output -raw container_app_url 2>/dev/null | grep -v '::' || echo "") + echo "app_url=$APP_URL" >> "$GITHUB_OUTPUT" + + # Get app name + APP_NAME=$(terraform output -raw container_app_name 2>/dev/null | grep -v '::' || echo "") + echo "app_name=$APP_NAME" >> "$GITHUB_OUTPUT" + + # The provider fails the data read outright when the custom role is absent, + # so a Terraform precondition can never run for this case. Issue #1794: the + # Azure deploy failed on every main run since 2026-07-19 because the + # bootstrap stack's role definition carried an action that does not exist + # in Azure's operation catalog, so Azure rejected the whole role definition + # with InvalidActionOrNotAction and the role was never created. That action + # is removed in this PR. The message below still covers the general "role + # missing" case for any future recurrence. + - name: Explain a missing bootstrap role + if: failure() + run: | + # Both logs are checked. Today the role lookup runs at plan: it lives in + # the root module (terraform/environments/azure/compute.tf), outside the + # module-level depends_on that used to defer it to apply and churn the + # assignment on every deploy (#1802). Keeping the apply log in the loop + # costs nothing and means the diagnostic does not have to move if the + # read ever shifts back. + # + # Each file is checked separately so that "this log does not exist" is an + # explicit, expected case at the point of reading -- either log is absent + # when an earlier step failed first. Passing both to one grep would also + # work (POSIX: -q exits 0 if a line is selected, and the only combinations + # returning 2 are those with no match anywhere, where silence is wanted), + # but it leaves that reasoning implicit. + found=0 + for log in "${RUNNER_TEMP}/tf-plan.log" "${RUNNER_TEMP}/tf-apply.log"; do + [ -f "$log" ] || continue + if grep -qiE 'could not find role|Role Definition .* was not found' "$log"; then + found=1 + break + fi + done + [ "$found" -eq 1 ] || exit 0 + cat >&2 <<'EOT' + ::error title=Missing bootstrap role::The custom reservation-purchaser role does not exist in this subscription. + + Apply the bootstrap stack that creates it: + terraform/environments/azure/ci-cd-permissions + + Do NOT grant Microsoft.Authorization/roleDefinitions/write to the deploy + service principal to work around this. The split is deliberate: this + pipeline holds roleAssignments/write only. + + If the bootstrap HAS been applied, check in this order: + 1. the bootstrap stack's own apply log for InvalidActionOrNotAction. + That means the role definition itself was rejected by Azure and + never created, not merely unassigned (issue #1794's root cause: + a nonexistent action in the role's actions list). Make sure the + bootstrap module is on a commit that includes that fix. + 2. the deploy SP has Microsoft.Authorization/roleDefinitions/read. + Without it the lookup returns empty, which is indistinguishable + from the role being absent. + 3. the role was not renamed or deleted out of band. + 4. the name suffix still matches. The runtime module looks up + "CUDly Reservation Purchaser (custom) - "; + the bootstrap builds the name from its own var.name_suffix. + EOT + + # A "Release state lock on failure" step used to break the lease here on + # `failure() || cancelled()`. Removed with #1801. On failure Terraform + # ordinarily unlocks on its way out, so the step usually did nothing; on + # cancelled() a SIGINT'd apply is still shutting down and may still be + # writing state, so it broke the lease out from under it. That is the + # decisive half: a live lease must never be broken. + # + # The residue is real and is accepted, not denied. A lease IS stranded + # when the unlock call itself fails ("Error releasing the state lock", + # e.g. OIDC expiry mid-apply), when Terraform crashes, or when Actions + # SIGKILLs after the cancellation grace period. The azurerm backend takes + # an INFINITE lease, so none of those self-expire. Recovery is deliberate + # and manual: `terraform force-unlock ` with the ID Terraform prints. + # A step that clears those cases can only do so by also breaking live + # leases, which is the bug this issue is about. + + - name: Save deployment info + env: + ENVIRONMENT: ${{ needs.prepare.outputs.environment }} + APP_URL: ${{ steps.deploy.outputs.app_url }} + APP_NAME: ${{ steps.deploy.outputs.app_name }} + DEPLOYED_BY: ${{ github.actor }} + COMMIT: ${{ github.sha }} + run: | + set -euo pipefail + jq -n \ + --arg environment "$ENVIRONMENT" \ + --arg app_url "$APP_URL" \ + --arg app_name "$APP_NAME" \ + --arg deployed_at "$(date -u +%Y-%m-%dT%H:%M:%SZ)" \ + --arg deployed_by "$DEPLOYED_BY" \ + --arg commit "$COMMIT" \ + '{environment: $environment, app_url: $app_url, app_name: $app_name, deployed_at: $deployed_at, deployed_by: $deployed_by, commit: $commit}' \ + > deployment-info.json + cat deployment-info.json + + - name: Upload deployment info + uses: actions/upload-artifact@b7c566a772e6b6bfb58ed0dc250532a479d7789f # v6.0.0 + with: + name: deployment-info-azure-${{ needs.prepare.outputs.environment }} + path: deployment-info.json + retention-days: 90 + + # Test the deployment + test-deployment: + name: Test Deployment + runs-on: ubuntu-latest + needs: [prepare, build-and-deploy] + if: always() && needs.build-and-deploy.result == 'success' + permissions: + id-token: write + contents: read + + steps: + - name: Azure Login + uses: azure/login@532459ea530d8321f2fb9bb10d1e0bcf23869a43 # v3.0.0 + with: + client-id: ${{ secrets.AZURE_CLIENT_ID }} + tenant-id: ${{ secrets.AZURE_TENANT_ID }} + subscription-id: ${{ secrets.AZURE_SUBSCRIPTION_ID }} + + - name: Get Container App URL + id: get-url + run: | + APP_URL="${{ needs.build-and-deploy.outputs.app_url }}" + if [ -z "$APP_URL" ]; then + echo "Failed to get app URL from deploy outputs" + exit 1 + fi + # Ensure URL has scheme and no trailing slash + case "$APP_URL" in + http://*|https://*) ;; # already has scheme + *) APP_URL="https://$APP_URL" ;; + esac + APP_URL="${APP_URL%/}" + echo "url=$APP_URL" >> "$GITHUB_OUTPUT" + + - name: Wait for Container App to be ready + run: | + echo "Waiting 30 seconds for Container App to be ready..." + sleep 30 + + - name: Test health endpoint + run: | + URL="${{ steps.get-url.outputs.url }}" + echo "Testing health endpoint: $URL/health" + + for i in {1..5}; do + if curl -f -s "$URL/health" > /dev/null; then + RESPONSE=$(curl -s "$URL/health") + echo "Response: $RESPONSE" + if echo "$RESPONSE" | grep -q '"status"'; then + echo "Health check passed!" + exit 0 + fi + fi + echo "Attempt $i failed, retrying in 10 seconds..." + sleep 10 + done + + echo "Health check failed after 5 attempts" + exit 1 + + - name: Run smoke tests + run: | + URL="${{ steps.get-url.outputs.url }}" + + echo "Running basic smoke tests..." + + for i in {1..3}; do + STATUS=$(curl -s -o /dev/null -w "%{http_code}" "$URL/health") + if [ "$STATUS" -eq 200 ]; then + echo "Health check $i: passed (HTTP $STATUS)" + else + echo "Health check $i: failed (HTTP $STATUS)" + exit 1 + fi + sleep 2 + done + + echo "All smoke tests passed!" + + - name: Check Container App logs + run: | + APP_NAME="${{ needs.build-and-deploy.outputs.app_name }}" + if [ -n "$APP_NAME" ]; then + echo "Recent Container App logs:" + az containerapp logs show \ + --name $APP_NAME \ + --resource-group ${{ env.RESOURCE_GROUP }} \ + --tail 50 || true + fi + + # Post deployment summary + summary: + name: Deployment Summary + runs-on: ubuntu-latest + needs: [prepare, build-and-deploy, test-deployment] + if: always() + permissions: + contents: read + + steps: + - name: Download deployment info + uses: actions/download-artifact@37930b1c2abaa49bbe596cd826c3c89aef350131 # v7.0.0 + with: + name: deployment-info-azure-${{ needs.prepare.outputs.environment }} + continue-on-error: true + + - name: Post summary + run: | + { + echo "## Azure Container Apps Deployment Summary" + echo "" + echo "**Environment:** ${{ needs.prepare.outputs.environment }}" + echo "**Location:** ${{ env.AZURE_LOCATION }}" + echo "" + + if [ -f deployment-info.json ]; then + echo "### Deployment Details" + echo "\`\`\`json" + cat deployment-info.json + echo "\`\`\`" + fi + + echo "" + echo "### Job Results" + echo "- Deploy: ${{ needs.build-and-deploy.result }}" + echo "- Test: ${{ needs.test-deployment.result }}" + + if [ "${{ needs.build-and-deploy.result }}" == "success" ] && [ "${{ needs.test-deployment.result }}" == "success" ]; then + echo "" + echo "✅ **Deployment successful!**" + else + echo "" + echo "❌ **Deployment failed. Check logs for details.**" + fi + } >> "$GITHUB_STEP_SUMMARY" diff --git a/.github/workflows/deploy-gcp.yml b/.github/workflows/deploy-gcp.yml new file mode 100644 index 000000000..46f93ed45 --- /dev/null +++ b/.github/workflows/deploy-gcp.yml @@ -0,0 +1,346 @@ +# Deploy to GCP Cloud Run +# +# This workflow deploys the CUDly application to GCP Cloud Run. +# Serverless container platform with automatic scaling. +# +# Required GitHub Secrets: +# - ADMIN_EMAIL: Admin email for notifications +# +# Required GitHub Variables: +# - GCP_REGION: GCP region (e.g., us-central1) +# - GCP_PROJECT_ID: GCP project ID +# - GCP_WORKLOAD_IDENTITY_PROVIDER: WIF provider resource name for OIDC keyless auth +# - GCP_SERVICE_ACCOUNT: SA email for OIDC keyless auth +# - ARTIFACT_REGISTRY_REPO: Artifact Registry repository name (default: cudly) +# +# Triggered by: +# - Pushes to main branch +# - Manual workflow dispatch +# - Workflow call from deploy-all.yml + +name: Deploy to GCP Cloud Run + +permissions: + contents: read + +# No workflow-level `concurrency` on purpose. What needs protecting is the +# Terraform state object, which is keyed on the environment, and only the +# job-level key can see `needs.prepare.outputs.environment` -- this level is +# limited to the `github`, `inputs` and `vars` contexts. Keying it on +# `github.ref` put two runs on different refs that both resolve to `dev` into +# different groups against one state file (#1806). The group lives on +# `build-and-deploy` below. +on: + push: + branches: [main] + workflow_dispatch: + inputs: + environment: + description: 'Environment' + required: true + type: choice + options: [dev, staging, prod] + workflow_call: + inputs: + environment: + required: true + type: string + secrets: + ADMIN_EMAIL: + required: true + TF_BACKEND_GCP: + required: true + +env: + GCP_REGION: ${{ vars.GCP_REGION || 'us-central1' }} + ARTIFACT_REGISTRY_REPO: ${{ vars.ARTIFACT_REGISTRY_REPO || 'cudly' }} + TF_VERSION: '1.10.0' + FORCE_JAVASCRIPT_ACTIONS_TO_NODE24: true + +jobs: + # Determine deployment environment + prepare: + name: Prepare Deployment + runs-on: ubuntu-latest + permissions: + contents: read + outputs: + environment: ${{ steps.set-env.outputs.environment }} + steps: + - name: Determine environment + id: set-env + env: + EVENT_NAME: ${{ github.event_name }} + INPUT_ENVIRONMENT: ${{ inputs.environment }} + run: | + set -euo pipefail + INPUT_ENVIRONMENT="${INPUT_ENVIRONMENT:-}" + + # Branch on whether an environment was REQUESTED, not on the event + # name. Inside a reusable workflow `github.event_name` is the CALLER's + # event and is never "workflow_call", so no event name can identify a + # workflow_call invocation: a release-triggered deploy-all.yml passing + # environment=prod would have landed on a `dev` default arm and + # resolved to dev while the summary said prod (issue #1805). + # + # `inputs` is populated on workflow_dispatch and workflow_call, the + # only two triggers that carry an environment, so a non-empty input is + # always an explicit request and wins. An empty input means none was + # supplied: under `push` that is this workflow's own main-branch + # deploy and means dev, and every other event stops the run rather + # than guessing. A workflow_call from a push-triggered caller would + # read as `push` here too, since `github.event_name` is inherited, but + # deploy-all.yml is the only caller, has no push trigger, and + # allowlists the value it passes. + if [ -n "$INPUT_ENVIRONMENT" ]; then + ENVIRONMENT="$INPUT_ENVIRONMENT" + elif [ "$EVENT_NAME" = push ]; then + ENVIRONMENT=dev + else + echo "::error::No environment supplied on a '$EVENT_NAME'-triggered run; refusing to guess a deployment target." + exit 1 + fi + + # workflow_dispatch constrains this to the declared `choice` options, + # but the workflow_call input is typed as a free-form string, so the + # allowlist is enforced here rather than assumed. + case "$ENVIRONMENT" in + dev|staging|prod) ;; + *) + echo "::error::Refusing unknown environment: '$ENVIRONMENT'" + exit 1 + ;; + esac + + echo "environment=$ENVIRONMENT" >> "$GITHUB_OUTPUT" + + # Build Docker image (via Terraform build module) and deploy + build-and-deploy: + name: Build & Deploy + runs-on: ubuntu-latest + needs: prepare + permissions: + id-token: write + contents: read + # Every run that writes gs:///github-/default.tfstate + # serializes here, whatever ref or workflow it came from (#1806). The suffix + # is the exact value the backend prefix below is built from, so the group and + # the state object cannot drift apart, and `prepare` fails the run on + # anything outside dev|staging|prod so it can never be empty. + # + # `cancel-in-progress: false` is stated rather than left to the default: + # cancelling mid-`terraform apply` is how you get a half-applied stack and a + # lock nobody releases. + concurrency: + group: gcp-tfstate-${{ needs.prepare.outputs.environment }} + cancel-in-progress: false + outputs: + service_url: ${{ steps.deploy.outputs.service_url }} + + steps: + - name: Checkout code + uses: actions/checkout@93cb6efe18208431cddfb8368fd83d5badbf9bfd # v5.0.1 + with: + persist-credentials: false + + - name: Authenticate to Google Cloud + uses: google-github-actions/auth@7c6bc770dae815cd3e89ee6cdf493a5fab2cc093 # v3.0.0 + with: + workload_identity_provider: ${{ vars.GCP_WORKLOAD_IDENTITY_PROVIDER }} + service_account: ${{ vars.GCP_SERVICE_ACCOUNT }} + + - name: Set up Cloud SDK + uses: google-github-actions/setup-gcloud@aa5489c8933f4cc7a4f7d45035b3b1440c9c10db # v3.0.1 + + - name: Setup Terraform + uses: hashicorp/setup-terraform@dfe3c3f87815947d99a8997f908cb6525fc44e9e # v4.0.1 + with: + terraform_version: ${{ env.TF_VERSION }} + + # This step used to `gsutil rm` the state lock object before every init, + # unconditionally and with no check that the lock was stale or anyone + # else's, so it deleted a live lock held by a concurrent writer. Removed + # with #1806; the job-level `concurrency` group above is what keeps + # writers apart now, and a loud "Error acquiring the state lock" is the + # correct outcome if one ever slips through. Recovery from a genuinely + # stranded lock is `terraform force-unlock ` -- see + # runbooks/terraform-stuck-lock.md. + - name: Terraform Init + env: + TF_BACKEND: ${{ secrets.TF_BACKEND_GCP }} + ENVIRONMENT: ${{ needs.prepare.outputs.environment }} + run: | + printf '%s\nprefix = "github-%s"\n' "$TF_BACKEND" "$ENVIRONMENT" > /tmp/backend.tfbackend + cd terraform/environments/gcp + terraform init -backend-config=/tmp/backend.tfbackend + + - name: Terraform Plan + env: + TF_VAR_admin_email: ${{ secrets.ADMIN_EMAIL }} + run: | + cd terraform/environments/gcp + terraform plan \ + -var-file="github-${{ needs.prepare.outputs.environment }}.tfvars" \ + -var="project_id=${{ vars.GCP_PROJECT_ID }}" \ + -out=tfplan + + - name: Terraform Apply + id: deploy + env: + TF_VAR_admin_email: ${{ secrets.ADMIN_EMAIL }} + run: | + cd terraform/environments/gcp + terraform apply -auto-approve tfplan + + # Get service URL + SERVICE_URL=$(terraform output -raw cloud_run_service_url 2>/dev/null | grep -v '::' || echo "") + echo "service_url=$SERVICE_URL" >> "$GITHUB_OUTPUT" + + # No automatic state-lock release on failure. The step that used to live + # here ran on `failure() || cancelled()` with no age check and no check + # that the lock was this run's, so it deleted whatever lock object was + # there. The `cancelled()` half is the decisive one: GitHub runs those + # steps while `terraform apply` is still shutting down and may still be + # writing state, so it destroyed a lock out from under an active writer. + # Terraform releases its own lock on a clean apply error; a lock that + # survives a run means the run died abnormally, which needs operator + # confirmation that the owning run is dead. Recovery is + # `terraform force-unlock ` -- see runbooks/terraform-stuck-lock.md. + + - name: Save deployment info + env: + ENVIRONMENT: ${{ needs.prepare.outputs.environment }} + SERVICE_URL: ${{ steps.deploy.outputs.service_url }} + DEPLOYED_BY: ${{ github.actor }} + COMMIT: ${{ github.sha }} + run: | + set -euo pipefail + jq -n \ + --arg environment "$ENVIRONMENT" \ + --arg service_url "$SERVICE_URL" \ + --arg deployed_at "$(date -u +%Y-%m-%dT%H:%M:%SZ)" \ + --arg deployed_by "$DEPLOYED_BY" \ + --arg commit "$COMMIT" \ + '{environment: $environment, service_url: $service_url, deployed_at: $deployed_at, deployed_by: $deployed_by, commit: $commit}' \ + > deployment-info.json + cat deployment-info.json + + - name: Upload deployment info + uses: actions/upload-artifact@b7c566a772e6b6bfb58ed0dc250532a479d7789f # v6.0.0 + with: + name: deployment-info-gcp-${{ needs.prepare.outputs.environment }} + path: deployment-info.json + retention-days: 90 + + # Test the deployment + test-deployment: + name: Test Deployment + runs-on: ubuntu-latest + needs: [prepare, build-and-deploy] + if: always() && needs.build-and-deploy.result == 'success' + permissions: + contents: read + + steps: + - name: Get Service URL + id: get-url + run: | + SERVICE_URL="${{ needs.build-and-deploy.outputs.service_url }}" + if [ -z "$SERVICE_URL" ]; then + echo "Failed to get service URL from deploy outputs" + exit 1 + fi + SERVICE_URL="${SERVICE_URL%/}" + echo "url=$SERVICE_URL" >> "$GITHUB_OUTPUT" + + - name: Wait for Cloud Run to be ready + run: | + echo "Waiting 30 seconds for Cloud Run to be ready..." + sleep 30 + + - name: Test health endpoint + run: | + URL="${{ steps.get-url.outputs.url }}" + echo "Testing health endpoint: $URL/health" + + for i in {1..5}; do + if curl -f -s "$URL/health" > /dev/null; then + RESPONSE=$(curl -s "$URL/health") + echo "Response: $RESPONSE" + if echo "$RESPONSE" | grep -q '"status"'; then + echo "Health check passed!" + exit 0 + fi + fi + echo "Attempt $i failed, retrying in 10 seconds..." + sleep 10 + done + + echo "Health check failed after 5 attempts" + exit 1 + + - name: Run smoke tests + run: | + URL="${{ steps.get-url.outputs.url }}" + + echo "Running basic smoke tests..." + + # Test health endpoint multiple times + for i in {1..3}; do + STATUS=$(curl -s -o /dev/null -w "%{http_code}" "$URL/health") + if [ "$STATUS" -eq 200 ]; then + echo "Health check $i: passed (HTTP $STATUS)" + else + echo "Health check $i: failed (HTTP $STATUS)" + exit 1 + fi + sleep 2 + done + + echo "All smoke tests passed!" + + # Post deployment summary + summary: + name: Deployment Summary + runs-on: ubuntu-latest + needs: [prepare, build-and-deploy, test-deployment] + if: always() + permissions: + contents: read + + steps: + - name: Download deployment info + uses: actions/download-artifact@37930b1c2abaa49bbe596cd826c3c89aef350131 # v7.0.0 + with: + name: deployment-info-gcp-${{ needs.prepare.outputs.environment }} + continue-on-error: true + + - name: Post summary + run: | + { + echo "## GCP Cloud Run Deployment Summary" + echo "" + echo "**Environment:** ${{ needs.prepare.outputs.environment }}" + echo "**Region:** ${{ env.GCP_REGION }}" + echo "" + + if [ -f deployment-info.json ]; then + echo "### Deployment Details" + echo "\`\`\`json" + cat deployment-info.json + echo "\`\`\`" + fi + + echo "" + echo "### Job Results" + echo "- Deploy: ${{ needs.build-and-deploy.result }}" + echo "- Test: ${{ needs.test-deployment.result }}" + + if [ "${{ needs.build-and-deploy.result }}" == "success" ] && [ "${{ needs.test-deployment.result }}" == "success" ]; then + echo "" + echo "✅ **Deployment successful!**" + else + echo "" + echo "❌ **Deployment failed. Check logs for details.**" + fi + } >> "$GITHUB_STEP_SUMMARY" diff --git a/.github/workflows/destroy-fargate-dev.yml b/.github/workflows/destroy-fargate-dev.yml new file mode 100644 index 000000000..d01fc2bb1 --- /dev/null +++ b/.github/workflows/destroy-fargate-dev.yml @@ -0,0 +1,168 @@ +# Destroy AWS Fargate Dev Resources (one-off) +# +# Destroys all Fargate dev resources tracked in the Terraform state. +# Run once to clean up the dev Fargate environment. + +name: Destroy Fargate Dev + +on: + workflow_dispatch: + inputs: + confirm: + description: 'Type "destroy" to confirm' + required: true + +# Least privilege: `id-token: write` is granted per job, only to the job that +# assumes the AWS deploy role, and that job is bound to the `dev` deployment +# environment. `guard` authenticates to nothing and must not hold it. +# +# IMPORTANT — what the `environment:` binding does and does not buy. It scopes +# secrets/vars and sets the OIDC subject to `repo::environment:dev`, +# which the AWS trust policy's sub allowlist accepts (role.tf:30). +# +# It is NOT a reviewer gate. GitHub only blocks a job once required-reviewer +# protection rules are configured on the environment, and at the time of writing +# no environment in this repo has any. Until that is configured out-of-band this +# destroy job still runs unapproved. See #1660 for the live state — deliberately +# not restated here, so this comment cannot rot into false reassurance. +# +# The environment subject is ref-agnostic, so the binding does not by itself keep +# this workflow on `main`. The `guard` job below checks the ref, but that check is +# defense-in-depth against ACCIDENTS ONLY and is NOT a security boundary: +# `workflow_dispatch` runs the workflow file as it exists on the dispatched ref, so +# anyone able to push a branch can delete the check and still present the +# `environment:dev` subject. The only control that survives that is a deployment +# branch policy, which GitHub evaluates before the job starts and before the token +# is minted. See #1660. +permissions: + contents: read + +env: + TF_VERSION: '1.10.0' + FORCE_JAVASCRIPT_ACTIONS_TO_NODE24: true + +jobs: + guard: + name: Confirm Destruction + runs-on: ubuntu-latest + permissions: {} + steps: + # Catches the accidental "dispatched from the wrong branch" case. It does + # NOT stop a deliberate one: this file is attacker-controlled on the ref + # being dispatched, so the step can simply be deleted on that branch. The + # server-side equivalent is a deployment branch policy on `dev` + # (custom_branch_policies + a `main` pattern) — which does NOT require + # `main` to be a protected branch — tracked in #1660. + - name: Restrict to main + env: + REF: ${{ github.ref }} + run: | + if [ "$REF" != "refs/heads/main" ]; then + echo "::error::Refusing to destroy from '$REF'; this workflow may only be dispatched from refs/heads/main" + exit 1 + fi + + - name: Check confirmation + env: + CONFIRM: ${{ inputs.confirm }} + run: | + if [ "$CONFIRM" != "destroy" ]; then + echo "You must type 'destroy' exactly to confirm. Got: '$CONFIRM'" + exit 1 + fi + + destroy: + name: Terraform Destroy (Fargate dev) + runs-on: ubuntu-24.04-arm + needs: guard + # Destroys s3:///github-fargate-dev/terraform.tfstate, the same + # state object deploy-aws-fargate.yml and cleanup-staging.yml write, so it + # takes the same concurrency group: one writer per state file, whatever + # workflow or ref it came from (#1806). The suffix is a literal because this + # job's state key is too. `cancel-in-progress: false` because cancelling + # mid-`terraform destroy` leaves a half-destroyed stack and a stuck lock. + concurrency: + group: aws-fargate-tfstate-dev + cancel-in-progress: false + permissions: + id-token: write + contents: read + # Binds the OIDC subject to repo::environment:dev, which the AWS + # trust policy matches on. Not a reviewer gate until protection rules exist + # on this environment -- see the note at the top of this file. + environment: dev + # Set at job level, not workflow level: workflow-level `env:` is resolved + # before this job's `dev` environment is in scope, so an environment-scoped + # AWS_REGION variable would be invisible there. Matches cleanup-staging.yml. + env: + AWS_REGION: ${{ vars.AWS_REGION || 'us-east-1' }} + + steps: + - name: Checkout code + uses: actions/checkout@93cb6efe18208431cddfb8368fd83d5badbf9bfd # v5.0.1 + with: + persist-credentials: false + + - name: Configure AWS credentials + uses: aws-actions/configure-aws-credentials@d979d5b3a71173a29b74b5b88418bfda9437d885 # v6.1.1 + with: + role-to-assume: ${{ vars.AWS_ROLE_TO_ASSUME }} + aws-region: ${{ env.AWS_REGION }} + + - name: Setup Terraform + uses: hashicorp/setup-terraform@dfe3c3f87815947d99a8997f908cb6525fc44e9e # v4.0.1 + with: + terraform_version: ${{ env.TF_VERSION }} + + # This step used to `aws s3 rm` the state lock object before every init, + # unconditionally and with no check that the lock was stale or anyone + # else's, so it deleted a live lock held by a concurrent writer. Removed + # with #1806; the job-level `concurrency` group above is what keeps + # writers apart now, and a loud "Error acquiring the state lock" is the + # correct outcome if one ever slips through. Recovery from a genuinely + # stranded lock is `terraform force-unlock ` -- see + # runbooks/terraform-stuck-lock.md. + - name: Terraform Init + env: + TF_BACKEND: ${{ secrets.TF_BACKEND_AWS }} + run: | + printf '%s\nkey = "github-fargate-dev/terraform.tfstate"\n' "$TF_BACKEND" > /tmp/backend.tfbackend + cd terraform/environments/aws + terraform init -backend-config=/tmp/backend.tfbackend + + # Runs before `terraform destroy`: the repository is created with + # force_delete = false, so the destroy fails while images remain. Deletes + # only the repository THIS state owns, by exact name. The + # `contains(repositoryName,'cudly-dev')` filter this step used to run also + # force-deleted `backup-cudly-dev` and `cudly-dev-prod-mirror`, with every + # image in them (#1592). Rationale, the `output -json` handling and why + # nothing here is swallowed: the script's header. This job and both + # cleanup-staging.yml staging jobs call the same script, so the guard + # cannot land in one workflow and not its sibling -- which is how #1592 + # became #1820. + - name: Force-delete ECR repo + run: ./scripts/force-delete-owned-ecr-repo.sh terraform/environments/aws + + # Runs before `terraform destroy`: an instance applied with + # deletion_protection = true blocks the destroy. Unprotects only the + # instance THIS state owns, by exact identifier. The + # `starts_with(DBInstanceIdentifier,'cudly-dev')` filter this step used to + # run also matched `cudly-dev-prod-mirror` and + # `cudly-dev--postgres-replica` and stripped the last line of defence + # from them, leaving them exposed to the next destroy that did match + # (#1821). Rationale and why nothing here is swallowed: the script's + # header. This job and both cleanup-staging.yml staging jobs call the same + # script, so the guard cannot land in one workflow and not its sibling -- + # which is how #1592 became #1820. + - name: Disable RDS deletion protection on the instance this state owns + run: ./scripts/disable-owned-rds-deletion-protection.sh terraform/environments/aws + + - name: Terraform Destroy + env: + TF_VAR_admin_email: ${{ secrets.ADMIN_EMAIL }} + run: | + cd terraform/environments/aws + terraform destroy \ + -var-file="github-dev.tfvars" \ + -var="compute_platform=fargate" \ + -auto-approve diff --git a/.github/workflows/frontend-build-sentinel.yml b/.github/workflows/frontend-build-sentinel.yml new file mode 100644 index 000000000..3e7a1af95 --- /dev/null +++ b/.github/workflows/frontend-build-sentinel.yml @@ -0,0 +1,63 @@ +name: frontend-build-sentinel + +# Fast-fail guard against frontend builds that break on the protected +# branch but slipped past pre-commit + PR CI. Catches: +# - Rebases (pre-commit hooks don't run on rebase). +# - Merge commits authored via the GitHub UI (no pre-commit there). +# - Push races where two commits interleave on the protected branch +# in an unintended order. +# +# Designed to fire within ~1 minute of landing so the team gets paged +# before the per-cloud deploys (which run the same build inside their +# Docker frontend-builder stage) hit the failure 30+ minutes later. +# +# See #177 for the post-mortem of the PR #160 / PR #172 incident that +# motivated this. + +on: + push: + branches: + - main + - "feat/**" + +permissions: + contents: read + +concurrency: + # Successive pushes to the same ref only need the latest tip built. + group: frontend-build-${{ github.ref }} + cancel-in-progress: true + +jobs: + build: + name: Build frontend + runs-on: ubuntu-latest + timeout-minutes: 5 + defaults: + run: + working-directory: frontend + + steps: + - name: Checkout + uses: actions/checkout@93cb6efe18208431cddfb8368fd83d5badbf9bfd # v5.0.1 + with: + persist-credentials: false + + - name: Set up Node.js + uses: actions/setup-node@48b55a011bda9f5d6aeb4c2d9c7362e8dae4041e # v6.4.0 + with: + node-version: "24" + cache: "npm" + cache-dependency-path: frontend/package-lock.json + + - name: Install dependencies + run: npm ci + + - name: TypeScript typecheck + run: npx tsc --noEmit + + - name: Build + run: npm run build + + - name: Run frontend tests + run: npx jest --no-coverage --silent diff --git a/.github/workflows/frontend-build.yml b/.github/workflows/frontend-build.yml new file mode 100644 index 000000000..38ae527e2 --- /dev/null +++ b/.github/workflows/frontend-build.yml @@ -0,0 +1,68 @@ +name: Frontend build (PR) + +# Pre-merge gate: catches TypeScript errors and webpack build failures +# BEFORE a PR is merged, not after. Motivated by the post-mortem in #191: +# a TS6133 error rode through to the protected branch and caused all three +# post-merge deploy jobs to fail after spending minutes on infrastructure +# only to hit the same tsc error inside the Dockerfile frontend-builder stage. +# +# Companion to frontend-build-sentinel.yml, which guards the protected +# branch post-merge against rebases and GitHub-UI merge commits. +# This job guards the PR gate itself — the earlier the catch, the cheaper. +# +# Uses pull_request (not pull_request_target) so this job runs in the +# PR-head context with no access to repository secrets. pull_request_target +# executes in the base-branch context WITH secrets, which is a known +# fork-exfiltration vector and must not be used for untrusted code builds. +# +# See also: #177 (sentinel), #191 (this PR gate). + +on: + pull_request: + branches: + - main + paths: + - "frontend/**" + - ".github/workflows/frontend-build.yml" + +permissions: + contents: read + +concurrency: + # Cancel redundant runs when new commits are pushed to the same PR. + group: frontend-build-pr-${{ github.ref }} + cancel-in-progress: true + +jobs: + build: + name: Build frontend + runs-on: ubuntu-latest + timeout-minutes: 5 + defaults: + run: + working-directory: frontend + + steps: + - name: Checkout + uses: actions/checkout@93cb6efe18208431cddfb8368fd83d5badbf9bfd # v5.0.1 + with: + persist-credentials: false + + - name: Set up Node.js + uses: actions/setup-node@48b55a011bda9f5d6aeb4c2d9c7362e8dae4041e # v6.4.0 + with: + node-version: "24" + cache: "npm" + cache-dependency-path: frontend/package-lock.json + + - name: Install dependencies + run: npm ci + + - name: Lint + run: npm run lint + + - name: TypeScript typecheck + run: npm run typecheck + + - name: Build + run: npm run build diff --git a/.github/workflows/frontend-e2e.yml b/.github/workflows/frontend-e2e.yml new file mode 100644 index 000000000..50000e07e --- /dev/null +++ b/.github/workflows/frontend-e2e.yml @@ -0,0 +1,129 @@ +name: Frontend E2E + +on: + pull_request: + branches: + - main + paths: + - "frontend/**" + - ".github/workflows/frontend-e2e.yml" + # On main only to populate the browser cache in the default-branch scope: a + # cache written by a pull_request run is readable only within that PR, so + # without this the first run of every PR misses. + push: + branches: + - main + paths: + - "frontend/**" + - ".github/workflows/frontend-e2e.yml" + +permissions: + contents: read + +concurrency: + group: frontend-e2e-${{ github.ref }} + # PRs only. Superseding a push on main would cancel the run that populates + # the cache and the one carrying the post-merge signal, which are the two + # reasons the push trigger exists. + cancel-in-progress: ${{ github.event_name == 'pull_request' }} + +jobs: + playwright: + name: Playwright Chromium + runs-on: ubuntu-latest + # Raised from 10 so the install step's own 8-minute cap is always what + # fires. Every other step at its worst across all attempts sums to 78s + # (75s if you look only at green runs, which understates it, and no sample + # exercised the spec-failure path at all), plus roughly 30s for the cache + # restore and save. So a full 480s install lands at ~600s: exactly the old + # cap, leaving nothing for the job timeout to be a backstop with. And a + # job timeout cancels, which is the conclusion #1869 is about. A cap is + # not a cost: the median successful job takes 69s, and only pathological + # runs ever approach either limit. + timeout-minutes: 12 + defaults: + run: + working-directory: frontend + env: + CI: "true" + + steps: + - name: Checkout + uses: actions/checkout@93cb6efe18208431cddfb8368fd83d5badbf9bfd # v5.0.1 + with: + persist-credentials: false + + - name: Set up Node.js + uses: actions/setup-node@48b55a011bda9f5d6aeb4c2d9c7362e8dae4041e # v6.4.0 + with: + node-version: "24" + cache: "npm" + cache-dependency-path: frontend/package-lock.json + + - name: Install dependencies + run: npm ci + + - name: Build + run: npm run build + + # The resolved version, not the `^1.60.0` range in package.json, so a + # floating range cannot silently reuse the previous browser build. + - name: Resolve Playwright version + id: playwright + run: | + set -euo pipefail + version="$(jq -re '.packages["node_modules/@playwright/test"].version' package-lock.json)" + echo "version=${version}" >> "$GITHUB_OUTPUT" + echo "Playwright ${version}" + + # Split restore/save rather than the combined actions/cache, whose save is + # a post step gated on `post-if: success()` and so skips the run being + # iterated on: the one whose specs are still failing. + - name: Restore Chromium + id: chromium-cache + uses: actions/cache/restore@0057852bfaa89a56745cba8c7296529d2fc39830 # v4.3.0 + with: + path: ~/.cache/ms-playwright + key: ${{ runner.os }}-${{ runner.arch }}-playwright-${{ steps.playwright.outputs.version }}-chromium + + - name: Install Chromium + # This step timeout is the fix for #1869, where this install wedged + # five times on 2026-08-19 (#1864, #1868) and ran until the job cap, + # ending each run `cancelled` with every spec skipped. Cancelled is not + # failed, so the missing e2e signal showed up as nothing at all. A step + # that busts its own timeout FAILS, so a future wedge is a red X naming + # this step. + # + # 8 minutes sits in the gap measured over the last 40 runs and all + # their attempts: 39 successful installs, 36 at or under 39s with + # outliers at 101s, 140s and 381s, against 5 wedges none under 585s. + # Above the healthy maximum, below every wedge, under the job cap. + # That gap is also why there is no retry: an attempt capped under 381s + # fails healthy runs, and two capped above it do not fit in the job. + timeout-minutes: 8 + # Idempotent on a cache hit: skips the download when the restored + # browser is complete and re-fetches when it is not. + run: npx playwright install --with-deps chromium + + - name: Save Chromium + # Skipped on a hit, and on an install failure (an untaken `if:` still + # implies success()), so only a complete browser reaches the key. + if: steps.chromium-cache.outputs.cache-hit != 'true' + uses: actions/cache/save@0057852bfaa89a56745cba8c7296529d2fc39830 # v4.3.0 + with: + path: ~/.cache/ms-playwright + key: ${{ steps.chromium-cache.outputs.cache-primary-key }} + + - name: Run Playwright tests + run: npm run test:e2e + + - name: Upload Playwright failure artifacts + if: failure() + uses: actions/upload-artifact@b7c566a772e6b6bfb58ed0dc250532a479d7789f # v6.0.0 + with: + name: playwright-report-${{ github.run_id }} + path: | + frontend/playwright-report/ + frontend/test-results/ + if-no-files-found: ignore + retention-days: 7 diff --git a/.github/workflows/mcp-server-json.yml b/.github/workflows/mcp-server-json.yml new file mode 100644 index 000000000..efc464a15 --- /dev/null +++ b/.github/workflows/mcp-server-json.yml @@ -0,0 +1,80 @@ +name: MCP server.json + +# Validates server.json (the MCP Registry listing manifest) on every PR that +# touches it, and gates tag pushes on server.json's version matching the tag. +# The registry rejects republishing a version, so a mismatch here must fail +# loud rather than let release automation publish the wrong metadata. + +on: + pull_request: + paths: + - "server.json" + - ".github/workflows/mcp-server-json.yml" + - ".github/scripts/server-json-validator/**" + push: + tags: ["v*"] + workflow_dispatch: + +permissions: + contents: read + +jobs: + validate-schema: + runs-on: ubuntu-latest + permissions: + contents: read + steps: + - name: Checkout + uses: actions/checkout@93cb6efe18208431cddfb8368fd83d5badbf9bfd # v5.0.1 + with: + persist-credentials: false + + - name: Fetch the MCP Registry server.json schema + run: | + set -euo pipefail + expected_schema_url="https://static.modelcontextprotocol.io/schemas/2025-12-11/server.schema.json" + expected_registry_name="io.github.LeanerCloud/cudly-mcp" + if ! jq -e --arg expected "$expected_schema_url" \ + '.["$schema"] | type == "string" and . == $expected' server.json >/dev/null; then + echo "::error::server.json must declare the official MCP Registry \$schema URL" + exit 1 + fi + if ! jq -e --arg expected "$expected_registry_name" \ + '.name | type == "string" and . == $expected' server.json >/dev/null; then + echo "::error::server.json name must be $expected_registry_name" + exit 1 + fi + curl -sSfL --connect-timeout 10 --max-time 30 \ + "$expected_schema_url" -o /tmp/server.schema.json + + - name: Install server.json validator dependencies + working-directory: .github/scripts/server-json-validator + run: npm ci --ignore-scripts --no-audit --no-fund + + - name: Validate server.json against the schema + run: node .github/scripts/server-json-validator/validate.cjs /tmp/server.schema.json server.json + + assert-version-matches-tag: + if: startsWith(github.ref, 'refs/tags/v') + needs: validate-schema + runs-on: ubuntu-latest + permissions: + contents: read + steps: + - name: Checkout + uses: actions/checkout@93cb6efe18208431cddfb8368fd83d5badbf9bfd # v5.0.1 + with: + persist-credentials: false + + - name: Assert server.json version matches the pushed tag + run: | + set -euo pipefail + tag="${GITHUB_REF#refs/tags/v}" + server_version=$(jq -r '.version' server.json) + if [[ "$tag" != "$server_version" ]]; then + echo "::error::server.json version ($server_version) does not match tag v$tag." \ + "Correct server.json.version in the release commit before creating or recreating the tag;" \ + "this workflow never rewrites the file." + exit 1 + fi + echo "server.json version ($server_version) matches tag v$tag" diff --git a/.github/workflows/pre-commit.yml b/.github/workflows/pre-commit.yml new file mode 100644 index 000000000..8e91b95e5 --- /dev/null +++ b/.github/workflows/pre-commit.yml @@ -0,0 +1,284 @@ +name: pre-commit + +on: + pull_request: + branches: [main] + push: + branches: [main] + +permissions: + contents: read + +jobs: + pre-commit: + name: Run pre-commit hooks + runs-on: ubuntu-latest + # 35 minutes accommodates the 3-attempt retry wrapper on the + # `Run pre-commit` step below (3 attempts * 10 min per-attempt + # timeout + 2 * 90 s retry waits = 33 min worst case) plus a + # small margin for setup/install steps. Without the bump, the + # job-level cap killed any retry attempt before it could start, + # making the retry policy ineffective (CR finding on #697). + timeout-minutes: 35 + steps: + - name: Checkout code + uses: actions/checkout@93cb6efe18208431cddfb8368fd83d5badbf9bfd # v5.0.1 + with: + persist-credentials: false + + - name: Set up Python + uses: actions/setup-python@a309ff8b426b58ec0e2a45f0f869d46889d02405 # v6.2.0 + with: + python-version: "3.13" + + - name: Set up Go + uses: actions/setup-go@4a3601121dd01d1626a1e23e37211e3254c1c06c # v6.4.0 + with: + # Read from go.mod rather than a hardcoded string, matching + # aws_sanity / azure_sanity / database-migration. One fewer place the + # Go version has to be bumped by hand (issue #1833). + go-version-file: go.mod + + - name: Set up Node.js + uses: actions/setup-node@48b55a011bda9f5d6aeb4c2d9c7362e8dae4041e # v6.4.0 + with: + node-version: "24" + # Cache the npm download cache keyed on the frontend lockfile. + # Saves ~30-40s per run vs an uncached `npm ci`. Same pattern + # already used in frontend-build-sentinel.yml. + cache: "npm" + cache-dependency-path: frontend/package-lock.json + + - name: Set up Terraform + # Required by the terraform_fmt + terraform_validate pre-commit hooks. + # terraform_validate calls `terraform init` per module, which the + # action wraps with HTTP-cached provider downloads. + # + # Pin must satisfy `required_version = ">= 1.10.0"` declared by every + # `terraform/environments/*/main.tf` — pinning to a sub-1.10 version + # makes init abort before validate even runs. Action major matches + # `.github/workflows/ci.yml` so both workflows resolve to the same + # Terraform binary; otherwise a behavioural drift between the two + # could pass one and fail the other. + uses: hashicorp/setup-terraform@dfe3c3f87815947d99a8997f908cb6525fc44e9e # v4.0.1 + with: + terraform_version: "1.10.5" + terraform_wrapper: false + + - name: Install tflint + # Pinned to a release tag (not master) so a malicious or accidental + # change to install_linux.sh on master can't silently land on this + # CI runner. `curl -fsSL` makes transport errors fail loudly + # instead of writing an HTML error page to stdin and feeding it + # to bash. + env: + TFLINT_VERSION: v0.55.0 + run: | + set -euo pipefail + curl -fsSL -o /tmp/tflint-install.sh \ + "https://raw.githubusercontent.com/terraform-linters/tflint/${TFLINT_VERSION}/install_linux.sh" + bash /tmp/tflint-install.sh + + # Cache the tflint ruleset plugins (aws/azurerm/google) that + # `tflint --init` downloads from the GitHub Releases API. Without + # this cache EVERY run re-downloads all three plugins and is exposed + # to transient GitHub release-API 503s — a sustained 503 outrode the + # GITHUB_TOKEN auth + 3-attempt pre-commit retry below and reddened + # this job repo-wide (blocking every PR). Keyed on .tflint.hcl so a + # plugin-version bump re-downloads; restore-keys seeds a warm start. + - name: Cache tflint plugins + uses: actions/cache@0057852bfaa89a56745cba8c7296529d2fc39830 # v4.3.0 + with: + path: ~/.tflint.d/plugins + key: tflint-plugins-${{ runner.os }}-${{ hashFiles('.tflint.hcl') }} + restore-keys: | + tflint-plugins-${{ runner.os }}- + + # Pre-populate the plugin cache with a dedicated, authenticated, + # retried `tflint --init` BEFORE pre-commit runs. On a cache hit this + # is a fast no-op (tflint skips download when the pinned plugin + # versions are already present — no API call, so immune to the 503). + # On a cache miss (version bump / cold cache) the retry loop rides + # out transient release-API 503s at the init level instead of + # re-running every hook via the coarse outer retry. GITHUB_TOKEN + # raises the release-API limit above the 60/hr anonymous ceiling. + - name: Initialize tflint plugins + env: + GITHUB_TOKEN: ${{ secrets.GITHUB_TOKEN }} + run: | + set -uo pipefail + for attempt in 1 2 3 4 5 6; do + if tflint --init --config="${GITHUB_WORKSPACE}/.tflint.hcl"; then + echo "tflint --init succeeded (attempt ${attempt})" + exit 0 + fi + wait=$((attempt * 20)) + echo "tflint --init failed (attempt ${attempt}/6); retrying in ${wait}s..." >&2 + sleep "${wait}" + done + echo "tflint --init failed after 6 attempts — GitHub release API likely unavailable" >&2 + exit 1 + + # Cache the installed tool binaries (gosec, gocyclo). Keyed on the + # pinned version strings so a tool-version bump still triggers a + # fresh install. Both binaries land in ~/go/bin which setup-go@v6 + # already adds to PATH. Restored BEFORE the install steps so the + # `if: cache-hit != 'true'` guards below can short-circuit them on + # cache-hit runs (the `go install` invocations cost ~3-5s each + # even when the module cache is warm; skipping them on cache-hit + # is worth the extra `if`). + - name: Cache Go-installed tools (gosec, gocyclo) + id: cache-go-tools + uses: actions/cache@0057852bfaa89a56745cba8c7296529d2fc39830 # v4.3.0 + with: + path: ~/go/bin + key: go-tools-${{ runner.os }}-gosec-v2.28.0-gocyclo-v0.6.0 + + - name: Install gosec + if: steps.cache-go-tools.outputs.cache-hit != 'true' + # Pinned to the same version ci.yml's `securego/gosec` Action uses, + # so an upstream gosec release with rule changes can't silently + # downgrade the gate between the two workflows. + run: go install github.com/securego/gosec/v2/cmd/gosec@v2.28.0 + + - name: Install gocyclo + if: steps.cache-go-tools.outputs.cache-hit != 'true' + # Pinned to match ci.yml — security tool installs must not use + # @latest; that's exactly the supply-chain weakness this PR is + # closing for Dockerfile FROMs. + run: go install github.com/fzipp/gocyclo/cmd/gocyclo@v0.6.0 + + - name: Install Trivy + # Pinned to v0.69.3 (the latest release with published GitHub-release + # tarballs as of writing). Tags exist for v0.58 onwards but several + # mid-range releases skipped publishing assets to the Releases page; + # the install.sh script fetches via GitHub Releases, so picking one + # of those tags makes install bail silently after detecting the + # version. v0.69.3 ships the standard `trivy__Linux-64bit.tar.gz` + # asset. + # + # The installer itself is fetched from the same pinned release tag + # (not the mutable `main` branch) and downloaded to a file before + # execution, matching the tflint step above: a malicious or + # accidental change to install.sh on main can't silently execute + # on this CI runner, and `curl -fsSL` makes transport errors fail + # loudly instead of piping an HTML error page into sh. + env: + TRIVY_VERSION: v0.69.3 + run: | + set -euo pipefail + curl -fsSL -o /tmp/trivy-install.sh \ + "https://raw.githubusercontent.com/aquasecurity/trivy/${TRIVY_VERSION}/contrib/install.sh" + sh /tmp/trivy-install.sh -b /usr/local/bin "${TRIVY_VERSION}" + + - name: Install git-secrets + # Pinned to a release tag rather than master HEAD. After install + # we register the AWS pattern set and ASSERT at least one pattern + # was registered — without the assert, a registration failure + # produces a patternless scanner that exits 0 unconditionally, + # leaving the gate silently downgraded. + run: | + set -euo pipefail + git clone --depth 1 --branch 1.3.0 https://github.com/awslabs/git-secrets.git /tmp/git-secrets + sudo make -C /tmp/git-secrets install + git secrets --register-aws --global + git secrets --list --global | grep -q '.' || { + echo "git-secrets registration produced no patterns — gate would be silently disabled" + exit 1 + } + + # Note: the local `hadolint` hook in .pre-commit-config.yaml (search for + # `id: hadolint`; no line number, because this comment has already gone + # stale twice as that file shifted) runs ghcr.io/hadolint/hadolint pinned + # by digest. The version lives in that entry and is the single source of + # truth; do not restate it here, or this comment rots again. We do NOT + # install a host binary here — it would be dead code (never invoked by + # the hook) AND a supply-chain hole (latest tag, no checksum). Note: + # an earlier version of this comment claimed the hook already pinned + # the image to v2.14.0 via the hook repo's `rev:` -- it did not; that + # `rev:` only pins hadolint's *hook definition*, whose upstream entry + # (`ghcr.io/hadolint/hadolint hadolint`) has no image tag and floats to + # `:latest`. See the digest pin in .pre-commit-config.yaml for the fix. + # If a future change switches the hook to a host binary, install a + # pinned + sha256-verified binary here. + + - name: Install pre-commit + run: pip install 'pre-commit==4.0.1' + + # Cache pre-commit's per-hook environments (Go, Python, Node, etc. + # virtualenvs it builds on first run). Keyed on the hook config + # because pre-commit will rebuild any env whose pinned rev changes. + # Saves ~30-60s per cache-hit run; safe because pre-commit verifies + # env integrity on use. + - name: Cache pre-commit environments + uses: actions/cache@0057852bfaa89a56745cba8c7296529d2fc39830 # v4.3.0 + with: + path: ~/.cache/pre-commit + key: pre-commit-${{ runner.os }}-${{ hashFiles('.pre-commit-config.yaml') }} + restore-keys: | + pre-commit-${{ runner.os }}- + + # Cache the Go build cache so `go vet`, `gosec`, and any other + # Go-compiling hooks reuse compiled object files instead of + # rebuilding from source. setup-go@v6 caches ~/go/pkg/mod + # (modules) but NOT ~/.cache/go-build (compiled output) — this + # step covers the latter. Keyed on go.sum so a dep upgrade still + # invalidates the cache and gets clean builds. + - name: Cache Go build cache + uses: actions/cache@0057852bfaa89a56745cba8c7296529d2fc39830 # v4.3.0 + with: + path: ~/.cache/go-build + key: go-build-${{ runner.os }}-${{ hashFiles('**/go.sum') }} + restore-keys: | + go-build-${{ runner.os }}- + + - name: Install frontend deps + run: | + if [ -f frontend/package-lock.json ]; then + cd frontend && npm ci + fi + + - name: Run pre-commit + # SKIP=terraform_validate: that hook calls `terraform init` per + # module, which creates `.terraform.lock.hcl` files. Those are + # gitignored, so on a fresh CI checkout they don't exist and the + # init step "modifies files", which pre-commit reports as a + # failure. Local pre-commit runs work because lock files persist + # between invocations. terraform_fmt and terraform_tflint still + # run and catch the syntax/style issues that terraform_validate + # would catch; the deeper schema validation runs in + # `terraform plan` during deploy workflows. + # + # GITHUB_TOKEN is passed so terraform_tflint's `tflint --init` + # step authenticates against the GitHub API (5000/hr per-token) + # when it downloads ruleset plugin releases. Without the token, + # tflint goes anonymous and hits the 60/hr per-IP limit shared + # across every workflow on the runner's NAT IP, which trips + # intermittently when PRs land in the same hour (issue #564). + # + # nick-fields/retry wraps the run with up to 3 attempts and a + # 90-second wait so transient flakes (GitHub Releases blips, + # tflint plugin download timeouts, etc.) do not require a + # manual rerun. The GITHUB_TOKEN fix above is the primary fix; + # the retry wrapper is the cheap defense-in-depth for the + # residual flakes that token alone cannot eliminate. + uses: nick-fields/retry@ce71cc2ab81d554ebbe88c79ab5975992d79ba08 # v3.0.2 + env: + # SKIP the `gosec` pre-commit hook in CI: it is designed to scan + # only the changed packages of a local commit, but `pre-commit run + # --all-files` (this CI job) feeds it EVERY .go file across all six + # modules at once, and gosec's whole-repo analysis exhausts the + # runner's memory — the job dies with "The runner has received a + # shutdown signal" at this step on every run. gosec is NOT dropped: + # the dedicated `Security Scanning` job in ci.yml runs gosec v2.28.0 + # per-module (SARIF) as the authoritative gate, so this only removes + # the duplicate that OOMs CI — the same dedup rationale as the + # already-skipped `terraform_validate` (covered by Validate Terraform). + # Local developers still get the fast per-changed-package gosec hook. + SKIP: terraform_validate,gosec + GITHUB_TOKEN: ${{ secrets.GITHUB_TOKEN }} + with: + timeout_minutes: 10 + max_attempts: 3 + retry_wait_seconds: 90 + command: pre-commit run --all-files diff --git a/.github/workflows/rollback.yml b/.github/workflows/rollback.yml new file mode 100644 index 000000000..6da5d4e27 --- /dev/null +++ b/.github/workflows/rollback.yml @@ -0,0 +1,636 @@ +# Rollback Deployment +# +# This workflow allows quick rollback to a previous deployment version. +# It redeploys a previously deployed Docker image without rebuilding. +# +# Required GitHub Secrets: +# - Same as deployment workflows (AWS, GCP, Azure credentials) +# +# Triggered by: +# - Manual workflow dispatch only (safety measure) + +name: Rollback Deployment + +# Least privilege by default: only the jobs that actually authenticate to a +# cloud provider get `id-token: write`, and every one of those jobs is bound to +# a deployment environment, so no OIDC token can be minted by a job outside the +# environment's protection rules. +# +# The binding is necessary but not sufficient: GitHub auto-creates a referenced +# environment on first use with NO protection rules. The `--rollback` +# environments must therefore be configured with required reviewers in repo +# settings for the approval gate to actually stop anything. +permissions: + contents: read + +on: + workflow_dispatch: + inputs: + cloud: + description: 'Cloud provider' + required: true + type: choice + options: [aws-lambda, aws-fargate, gcp, azure] + environment: + description: 'Environment' + required: true + type: choice + options: [dev, staging, prod] + image_tag: + description: 'Image tag to rollback to (e.g., sha-abc123, v1.2.3)' + required: true + type: string + reason: + description: 'Reason for rollback' + required: false + type: string + +env: + TF_VERSION: '1.6.0' + +jobs: + # Validate rollback request + validate: + name: Validate Rollback + runs-on: ubuntu-latest + timeout-minutes: 5 + permissions: + contents: read + outputs: + image_uri: ${{ steps.check.outputs.image_uri }} + + steps: + - name: Checkout code + uses: actions/checkout@93cb6efe18208431cddfb8368fd83d5badbf9bfd # v5.0.1 + with: + persist-credentials: false + + - name: Validate inputs + id: check + env: + CLOUD: ${{ inputs.cloud }} + IMAGE_TAG: ${{ inputs.image_tag }} + ECR_REPOSITORY: ${{ vars.ECR_REPOSITORY }} + AWS_ACCOUNT_ID: ${{ vars.AWS_ACCOUNT_ID }} + AWS_REGION: ${{ vars.AWS_REGION || 'us-east-1' }} + GCP_REGION: ${{ vars.GCP_REGION || 'us-central1' }} + GCP_PROJECT_ID: ${{ secrets.GCP_PROJECT_ID }} + ARTIFACT_REGISTRY_REPO: ${{ vars.ARTIFACT_REGISTRY_REPO || 'cudly' }} + ACR_NAME: ${{ vars.ACR_NAME || 'cudlyacr' }} + run: | + set -euo pipefail + + # `env:` entries whose expression renders empty are exported empty, + # but normalise anyway so `set -u` can never abort a rollback on an + # unset optional repository variable. + ECR_REPOSITORY="${ECR_REPOSITORY:-}" + AWS_ACCOUNT_ID="${AWS_ACCOUNT_ID:-}" + GCP_PROJECT_ID="${GCP_PROJECT_ID:-}" + + # The tag is embedded in image URIs that later steps hand to the cloud + # CLIs, so reject anything outside the registry-safe character set here + # rather than downstream. Constraining the tag also keeps the operator- + # supplied half of the $GITHUB_OUTPUT write below free of newlines; the + # rest of the URI comes from admin-controlled repository variables. + if [[ ! "$IMAGE_TAG" =~ ^[a-zA-Z0-9._-]+$ ]]; then + echo "::error::Invalid image tag format: only [a-zA-Z0-9._-] are allowed" + exit 1 + fi + + # Construct image URI based on cloud provider + case "$CLOUD" in + aws-lambda|aws-fargate) + if [ -z "$ECR_REPOSITORY" ]; then + echo "::error::ECR_REPOSITORY repository variable is not set; refusing to guess the repo name" + exit 1 + fi + if [ -z "$AWS_ACCOUNT_ID" ]; then + echo "::error::AWS_ACCOUNT_ID repository variable is not set; refusing to guess the registry" + exit 1 + fi + IMAGE_URI="${AWS_ACCOUNT_ID}.dkr.ecr.${AWS_REGION}.amazonaws.com/${ECR_REPOSITORY}:${IMAGE_TAG}" + ;; + gcp) + if [ -z "$GCP_PROJECT_ID" ]; then + echo "::error::GCP_PROJECT_ID secret is not set; refusing to guess the project" + exit 1 + fi + IMAGE_URI="${GCP_REGION}-docker.pkg.dev/${GCP_PROJECT_ID}/${ARTIFACT_REGISTRY_REPO}/cudly:${IMAGE_TAG}" + ;; + azure) + IMAGE_URI="${ACR_NAME}.azurecr.io/cudly:${IMAGE_TAG}" + ;; + *) + echo "::error::Unknown cloud provider: $CLOUD" + exit 1 + ;; + esac + + echo "image_uri=$IMAGE_URI" >> "$GITHUB_OUTPUT" + + - name: Display rollback plan + env: + CLOUD: ${{ inputs.cloud }} + ENVIRONMENT: ${{ inputs.environment }} + IMAGE_TAG: ${{ inputs.image_tag }} + IMAGE_URI: ${{ steps.check.outputs.image_uri }} + REASON: ${{ inputs.reason }} + run: | + set -euo pipefail + # The step summary renders as markdown, so fold the free-text reason + # onto one line: a dispatcher could otherwise embed newlines and forge + # headings or a fake result line in the rendered summary. + REASON="${REASON:-}" + REASON="${REASON//$'\n'/ }" + REASON="${REASON//$'\r'/ }" + { + echo "## Rollback Plan" + echo "" + echo "**Cloud:** $CLOUD" + echo "**Environment:** $ENVIRONMENT" + echo "**Image Tag:** $IMAGE_TAG" + echo "**Image URI:** $IMAGE_URI" + echo "" + + if [ -n "$REASON" ]; then + echo "**Reason:** $REASON" + echo "" + fi + + if [ "$ENVIRONMENT" = "prod" ]; then + echo "⚠️ **WARNING:** Rolling back PRODUCTION environment!" + fi + } >> "$GITHUB_STEP_SUMMARY" + + # NOTE: image existence is verified inside each rollback job rather than in a + # standalone job. A separate verify job would have to assume the same + # cloud deploy role, and it could not be covered by the rollback + # environment's reviewer gate without prompting the operator for a second + # approval on every rollback. + + # Rollback AWS Lambda + rollback-aws-lambda: + name: Rollback AWS Lambda + runs-on: ubuntu-latest + timeout-minutes: 30 + needs: validate + if: inputs.cloud == 'aws-lambda' + # `terraform apply` against s3:///github-/terraform.tfstate, + # the same object deploy-aws-lambda.yml and cleanup-staging.yml write, so it + # takes the same concurrency group (#1806). `inputs.environment` is a + # required `choice` constrained to dev|staging|prod, so the suffix is never + # empty; it is the same value this job interpolates into the state key below. + # The eviction and approval-gate caveats documented on `rollback-azure` apply + # here identically. + concurrency: + group: aws-tfstate-${{ inputs.environment }} + cancel-in-progress: false + permissions: + id-token: write + contents: read + environment: + name: aws-lambda-${{ inputs.environment }}-rollback + + steps: + - name: Checkout code + uses: actions/checkout@93cb6efe18208431cddfb8368fd83d5badbf9bfd # v5.0.1 + with: + persist-credentials: false + + - name: Configure AWS credentials + uses: aws-actions/configure-aws-credentials@d979d5b3a71173a29b74b5b88418bfda9437d885 # v6.1.1 + with: + role-to-assume: ${{ vars.AWS_ROLE_TO_ASSUME }} + aws-region: ${{ vars.AWS_REGION || 'us-east-1' }} + + - name: Verify image exists + env: + IMAGE_TAG: ${{ inputs.image_tag }} + ECR_REPOSITORY: ${{ vars.ECR_REPOSITORY }} + AWS_REGION: ${{ vars.AWS_REGION || 'us-east-1' }} + run: | + set -euo pipefail + + if [ -z "${ECR_REPOSITORY:-}" ]; then + echo "::error::ECR_REPOSITORY repository variable is not set" + exit 1 + fi + + echo "Checking if image exists: $ECR_REPOSITORY:$IMAGE_TAG" + + if aws ecr describe-images \ + --repository-name "$ECR_REPOSITORY" \ + --image-ids "imageTag=$IMAGE_TAG" \ + --region "$AWS_REGION" > /dev/null 2>&1; then + echo "✅ Image exists in ECR" + else + echo "::error::Image not found in ECR" + exit 1 + fi + + - name: Setup Terraform + uses: hashicorp/setup-terraform@dfe3c3f87815947d99a8997f908cb6525fc44e9e # v4.0.1 + with: + terraform_version: ${{ env.TF_VERSION }} + + - name: Rollback with Terraform + env: + TF_VAR_admin_email: ${{ secrets.ADMIN_EMAIL }} + TF_BACKEND: ${{ secrets.TF_BACKEND_AWS }} + ENVIRONMENT: ${{ inputs.environment }} + IMAGE_URI: ${{ needs.validate.outputs.image_uri }} + run: | + set -euo pipefail + printf '%s\nkey = "github-%s/terraform.tfstate"\n' "$TF_BACKEND" "$ENVIRONMENT" > /tmp/backend.tfbackend + cd terraform/environments/aws + terraform init -backend-config=/tmp/backend.tfbackend + + terraform apply -auto-approve \ + -var-file="github-${ENVIRONMENT}.tfvars" \ + -var="image_uri=${IMAGE_URI}" \ + -var="compute_platform=lambda" + + - name: Verify rollback + run: | + set -euo pipefail + sleep 30 # Wait for Lambda to update + cd terraform/environments/aws + # Not silenced: if the output is missing, the real terraform error is + # what tells the operator why the rollback cannot be verified. + FUNCTION_URL=$(terraform output -raw lambda_function_url) + + if curl -f -s "$FUNCTION_URL/health" > /dev/null; then + echo "✅ Rollback successful - health check passed" + else + echo "❌ Rollback verification failed - health check failed" + exit 1 + fi + + # Rollback AWS Fargate + rollback-aws-fargate: + name: Rollback AWS Fargate + runs-on: ubuntu-latest + timeout-minutes: 30 + needs: validate + if: inputs.cloud == 'aws-fargate' + # `terraform apply` against + # s3:///github-fargate-/terraform.tfstate, the same + # object deploy-aws-fargate.yml, cleanup-staging.yml and + # destroy-fargate-dev.yml write, so it takes the same concurrency group + # (#1806). Until #1811 this job built the key from the LAMBDA namespace + # (`github-/`) and so carried `aws-tfstate-*` to match; the + # group and the key move together, because a group that does not name the + # object the job locks serializes against a state file it never touches + # while writing one unguarded. `inputs.environment` is a required `choice` + # constrained to dev|staging|prod, so the suffix is never empty; it is the + # same value this job interpolates into the state key below. + # scripts/test-aws-tfstate-platform-key.sh is what keeps the key and the + # `compute_platform` this job applies from drifting apart again. + concurrency: + group: aws-fargate-tfstate-${{ inputs.environment }} + cancel-in-progress: false + permissions: + id-token: write + contents: read + environment: + name: aws-fargate-${{ inputs.environment }}-rollback + + steps: + - name: Checkout code + uses: actions/checkout@93cb6efe18208431cddfb8368fd83d5badbf9bfd # v5.0.1 + with: + persist-credentials: false + + - name: Configure AWS credentials + uses: aws-actions/configure-aws-credentials@d979d5b3a71173a29b74b5b88418bfda9437d885 # v6.1.1 + with: + role-to-assume: ${{ vars.AWS_ROLE_TO_ASSUME }} + aws-region: ${{ vars.AWS_REGION || 'us-east-1' }} + + - name: Verify image exists + env: + IMAGE_TAG: ${{ inputs.image_tag }} + ECR_REPOSITORY: ${{ vars.ECR_REPOSITORY }} + AWS_REGION: ${{ vars.AWS_REGION || 'us-east-1' }} + run: | + set -euo pipefail + + if [ -z "${ECR_REPOSITORY:-}" ]; then + echo "::error::ECR_REPOSITORY repository variable is not set" + exit 1 + fi + + echo "Checking if image exists: $ECR_REPOSITORY:$IMAGE_TAG" + + if aws ecr describe-images \ + --repository-name "$ECR_REPOSITORY" \ + --image-ids "imageTag=$IMAGE_TAG" \ + --region "$AWS_REGION" > /dev/null 2>&1; then + echo "✅ Image exists in ECR" + else + echo "::error::Image not found in ECR" + exit 1 + fi + + - name: Setup Terraform + uses: hashicorp/setup-terraform@dfe3c3f87815947d99a8997f908cb6525fc44e9e # v4.0.1 + with: + terraform_version: ${{ env.TF_VERSION }} + + - name: Rollback with Terraform + env: + TF_VAR_admin_email: ${{ secrets.ADMIN_EMAIL }} + TF_BACKEND: ${{ secrets.TF_BACKEND_AWS }} + ENVIRONMENT: ${{ inputs.environment }} + IMAGE_URI: ${{ needs.validate.outputs.image_uri }} + run: | + set -euo pipefail + printf '%s\nkey = "github-fargate-%s/terraform.tfstate"\n' "$TF_BACKEND" "$ENVIRONMENT" > /tmp/backend.tfbackend + cd terraform/environments/aws + terraform init -backend-config=/tmp/backend.tfbackend + + terraform apply -auto-approve \ + -var-file="github-${ENVIRONMENT}.tfvars" \ + -var="image_uri=${IMAGE_URI}" \ + -var="compute_platform=fargate" + + # Rollback GCP + rollback-gcp: + name: Rollback GCP Cloud Run + runs-on: ubuntu-latest + timeout-minutes: 30 + needs: validate + if: inputs.cloud == 'gcp' + # `terraform apply` against gs:///github-/default.tfstate, + # the same object deploy-gcp.yml and cleanup-staging.yml write, so it takes + # the same concurrency group (#1806). `inputs.environment` is a required + # `choice` constrained to dev|staging|prod, so the suffix is never empty; it + # is the same value this job interpolates into the backend prefix below. + concurrency: + group: gcp-tfstate-${{ inputs.environment }} + cancel-in-progress: false + permissions: + id-token: write + contents: read + environment: + name: gcp-${{ inputs.environment }}-rollback + + steps: + - name: Checkout code + uses: actions/checkout@93cb6efe18208431cddfb8368fd83d5badbf9bfd # v5.0.1 + with: + persist-credentials: false + + - name: Authenticate to Google Cloud + uses: google-github-actions/auth@7c6bc770dae815cd3e89ee6cdf493a5fab2cc093 # v3.0.0 + with: + workload_identity_provider: ${{ vars.GCP_WORKLOAD_IDENTITY_PROVIDER }} + service_account: ${{ vars.GCP_SERVICE_ACCOUNT }} + + - name: Verify image exists + env: + IMAGE_URI: ${{ needs.validate.outputs.image_uri }} + run: | + set -euo pipefail + + echo "Checking if image exists: $IMAGE_URI" + + if gcloud artifacts docker images describe "$IMAGE_URI" > /dev/null 2>&1; then + echo "✅ Image exists in Artifact Registry" + else + echo "::error::Image not found in Artifact Registry" + exit 1 + fi + + - name: Setup Terraform + uses: hashicorp/setup-terraform@dfe3c3f87815947d99a8997f908cb6525fc44e9e # v4.0.1 + with: + terraform_version: ${{ env.TF_VERSION }} + + - name: Rollback with Terraform + env: + TF_VAR_admin_email: ${{ secrets.ADMIN_EMAIL }} + TF_BACKEND: ${{ secrets.TF_BACKEND_GCP }} + ENVIRONMENT: ${{ inputs.environment }} + IMAGE_URI: ${{ needs.validate.outputs.image_uri }} + GCP_PROJECT_ID: ${{ secrets.GCP_PROJECT_ID }} + run: | + set -euo pipefail + printf '%s\nprefix = "github-%s"\n' "$TF_BACKEND" "$ENVIRONMENT" > /tmp/backend.tfbackend + cd terraform/environments/gcp + terraform init -backend-config=/tmp/backend.tfbackend + + terraform apply -auto-approve \ + -var-file="github-${ENVIRONMENT}.tfvars" \ + -var="image_uri=${IMAGE_URI}" \ + -var="project_id=${GCP_PROJECT_ID}" + + # Rollback Azure + rollback-azure: + name: Rollback Azure Container Apps + runs-on: ubuntu-latest + timeout-minutes: 30 + needs: validate + if: inputs.cloud == 'azure' + # `terraform apply` against github-.terraform.tfstate, the same + # blob deploy-azure.yml and cleanup-staging.yml write, so it takes the same + # concurrency group (#1801). `inputs.environment` is a required `choice` + # constrained to dev|staging|prod, so the suffix is never empty; it is the + # same value this job interpolates into the state key below. + # + # Two consequences of sharing the group across workflows, both accepted as + # better than concurrent writers: + # - GitHub keeps one pending entry per group, so a queued rollback can be + # evicted by a later arrival. That surfaces: the summary job below exits + # 1 on any non-success result, so an evicted rollback reddens the run. + # - if the `environment:` binding below carries required reviewers (see + # #1660 for the live state, deliberately not restated here), it is + # undocumented whether a job parked awaiting approval holds its group. + # If it does, an unapproved rollback blocks deploys to that environment + # for the whole approval window. + concurrency: + group: azure-tfstate-${{ inputs.environment }} + cancel-in-progress: false + permissions: + id-token: write + contents: read + environment: + name: azure-${{ inputs.environment }}-rollback + + steps: + - name: Checkout code + uses: actions/checkout@93cb6efe18208431cddfb8368fd83d5badbf9bfd # v5.0.1 + with: + persist-credentials: false + + - name: Azure Login + uses: azure/login@532459ea530d8321f2fb9bb10d1e0bcf23869a43 # v3.0.0 + with: + client-id: ${{ secrets.AZURE_CLIENT_ID }} + tenant-id: ${{ secrets.AZURE_TENANT_ID }} + subscription-id: ${{ secrets.AZURE_SUBSCRIPTION_ID }} + + - name: Verify image exists + env: + IMAGE_TAG: ${{ inputs.image_tag }} + ACR_NAME: ${{ vars.ACR_NAME || 'cudlyacr' }} + run: | + set -euo pipefail + + echo "Checking if image exists: $ACR_NAME/cudly:$IMAGE_TAG" + + # Look the tag up directly instead of listing tags and grepping. + # `show-tags` is paginated, so a valid but older tag can fall off the + # first page and be reported missing; and under `set -o pipefail` the + # early exit of `grep -q` can SIGPIPE the `az` process, failing the + # pipeline even on a match. + if az acr repository show \ + --name "$ACR_NAME" \ + --image "cudly:$IMAGE_TAG" > /dev/null 2>&1; then + echo "✅ Image exists in ACR" + else + echo "::error::Image not found in ACR" + exit 1 + fi + + - name: Setup Terraform + uses: hashicorp/setup-terraform@dfe3c3f87815947d99a8997f908cb6525fc44e9e # v4.0.1 + with: + terraform_version: ${{ env.TF_VERSION }} + + - name: Rollback with Terraform + env: + TF_VAR_admin_email: ${{ secrets.ADMIN_EMAIL }} + TF_VAR_subscription_id: ${{ secrets.AZURE_SUBSCRIPTION_ID }} + TF_VAR_key_vault_name: ${{ vars.KEY_VAULT_NAME }} + TF_BACKEND: ${{ secrets.TF_BACKEND_AZURE }} + ENVIRONMENT: ${{ inputs.environment }} + IMAGE_URI: ${{ needs.validate.outputs.image_uri }} + run: | + set -euo pipefail + printf '%s\nkey = "github-%s.terraform.tfstate"\n' "$TF_BACKEND" "$ENVIRONMENT" > /tmp/backend.tfbackend + cd terraform/environments/azure + terraform init -backend-config=/tmp/backend.tfbackend + + terraform apply -auto-approve \ + -var-file="github-${ENVIRONMENT}.tfvars" \ + -var="image_uri=${IMAGE_URI}" + + # Summary + summary: + name: Rollback Summary + runs-on: ubuntu-latest + timeout-minutes: 10 + permissions: + contents: read + needs: + - validate + - rollback-aws-lambda + - rollback-aws-fargate + - rollback-gcp + - rollback-azure + if: always() + + steps: + - name: Determine rollback result + id: result + env: + CLOUD: ${{ inputs.cloud }} + RESULT_AWS_LAMBDA: ${{ needs.rollback-aws-lambda.result }} + RESULT_AWS_FARGATE: ${{ needs.rollback-aws-fargate.result }} + RESULT_GCP: ${{ needs.rollback-gcp.result }} + RESULT_AZURE: ${{ needs.rollback-azure.result }} + run: | + set -euo pipefail + + case "$CLOUD" in + aws-lambda) RESULT="$RESULT_AWS_LAMBDA" ;; + aws-fargate) RESULT="$RESULT_AWS_FARGATE" ;; + gcp) RESULT="$RESULT_GCP" ;; + azure) RESULT="$RESULT_AZURE" ;; + *) RESULT="unknown" ;; + esac + + echo "result=$RESULT" >> "$GITHUB_OUTPUT" + + - name: Record rollback + env: + CLOUD: ${{ inputs.cloud }} + ENVIRONMENT: ${{ inputs.environment }} + IMAGE_TAG: ${{ inputs.image_tag }} + IMAGE_URI: ${{ needs.validate.outputs.image_uri }} + REASON: ${{ inputs.reason }} + RESULT: ${{ steps.result.outputs.result }} + PERFORMED_BY: ${{ github.actor }} + WORKFLOW_RUN: ${{ github.server_url }}/${{ github.repository }}/actions/runs/${{ github.run_id }} + run: | + set -euo pipefail + REASON="${REASON:-}" + + # Build the audit record with jq so every value is JSON-escaped by the + # tool. Interpolating them into a here-document would both break the + # JSON on a quote and let $(...) in an input execute as shell. + jq -n \ + --arg cloud "$CLOUD" \ + --arg environment "$ENVIRONMENT" \ + --arg image_tag "$IMAGE_TAG" \ + --arg image_uri "$IMAGE_URI" \ + --arg reason "$REASON" \ + --arg result "$RESULT" \ + --arg performed_by "$PERFORMED_BY" \ + --arg performed_at "$(date -u +%Y-%m-%dT%H:%M:%SZ)" \ + --arg workflow_run "$WORKFLOW_RUN" \ + '$ARGS.named' > rollback-record.json + + echo "Rollback record:" + cat rollback-record.json + + - name: Upload rollback record + uses: actions/upload-artifact@b7c566a772e6b6bfb58ed0dc250532a479d7789f # v6.0.0 + with: + name: rollback-record-${{ github.run_id }} + path: rollback-record.json + retention-days: 365 # Keep rollback records for 1 year + + # Runs last so that a failed rollback still leaves the audit record above + # uploaded before this step fails the job. + - name: Post summary + env: + CLOUD: ${{ inputs.cloud }} + ENVIRONMENT: ${{ inputs.environment }} + IMAGE_TAG: ${{ inputs.image_tag }} + REASON: ${{ inputs.reason }} + RESULT: ${{ steps.result.outputs.result }} + run: | + set -euo pipefail + # Folded onto one line for the same reason as in the rollback plan: + # the summary is markdown and must not be forgeable from an input. + REASON="${REASON:-}" + REASON="${REASON//$'\n'/ }" + REASON="${REASON//$'\r'/ }" + + { + echo "## Rollback Results" + echo "" + echo "**Cloud:** $CLOUD" + echo "**Environment:** $ENVIRONMENT" + echo "**Image Tag:** $IMAGE_TAG" + echo "**Timestamp:** $(date -u +%Y-%m-%dT%H:%M:%SZ)" + echo "" + + if [ -n "$REASON" ]; then + echo "**Reason:** $REASON" + echo "" + fi + + if [ "$RESULT" = "success" ]; then + echo "✅ **Rollback completed successfully!**" + else + echo "❌ **Rollback failed. Status: $RESULT**" + echo "Check the job logs for details." + fi + } >> "$GITHUB_STEP_SUMMARY" + + if [ "$RESULT" != "success" ]; then + exit 1 + fi diff --git a/.gitignore b/.gitignore index ff4c87eac..8465ec5af 100644 --- a/.gitignore +++ b/.gitignore @@ -4,6 +4,14 @@ # Go binaries rds-ri-tool ri-helper +cudly +cmd/cmd +cmd.test + +# Local build output directory at repo root (e.g. `bin/cudly-server` from +# `go build -o bin/...` in Makefile:45). Anchored with a leading slash so +# nested bin/ directories elsewhere in the tree are not affected. +/bin/ # Go build artifacts *.exe @@ -14,13 +22,27 @@ ri-helper # Test binary, built with `go test -c` *.test +# ... but not the E2E test-runner image definition +!Dockerfile.test # Output of the go coverage tool *.out +# Log files +*.log + +# Go workspace files -- go.work is committed (canonical module list). +# go.work.local is gitignored for per-developer worktree additions. +go.work.local +go.work.local.sum + # Dependency directories vendor/ +# Legacy/dev-notes (local exploration, not for commit) +.legacy/ +.dev-notes/ + # IDE files .idea/ .vscode/ @@ -28,6 +50,103 @@ vendor/ *.swo *~ +# Environment files +.env* +!.env.example + # OS files .DS_Store -Thumbs.db \ No newline at end of file +Thumbs.db + +# Backup files +*.tar.gz +*.bak +*.backup + +# Temporary/exploration files +ARCHITECTURE*.md +FINDINGS*.md +IMPLEMENTATION*.md +MULTI_CLOUD*.md +MYSQL*.md +*_SUMMARY.md +*_INDEX.md + +#build artifacts (binaries) +/sanity +/azure-sanity +/ri-exchange +ci_cd_sanity_tests/ri-exchange + +#sanity reports / artifacts +*sanity_report.json +azure_sanity_report.json +ri-exchange_*.json +sanity_report*.json + +# Compiled binaries +cudly-final +cudly-test +cleanup-lambda +/server +/lambda + +# Frontend build artifacts +frontend/dist/ +frontend/coverage/ +frontend/node_modules/ + +# Frontend Playwright artefacts — generated per run, not committed +frontend/playwright-report/ +frontend/test-results/ +frontend/tests-e2e/.cache/ + +# Temporary files +tmp/ + +# Terraform local config (contain environment-specific settings) +*.tfvars +!*.tfvars.example +*.tfbackend +backend.hcl + +# Terraform plan files +tfplan +tfplan.* +*.tfplan +*.tfplan.* + +# Terraform state and outputs +terraform-apply-*.txt +terraform-plan-*.txt +*.tfstate +*.tfstate.* +.terraform/ +.terraform.lock.hcl + +# PR review swarm output +PR_review/ + +# Claude Code / claude-flow framework runtime files (not project source) +.claude-flow/ +.swarm/ +.mcp.json +clear-rate-limit +security-audit-swarm.zip +security-audit-swarm/ + +# Per-session Claude Code artefacts under .claude/: commit-message +# drafts, one-off tmp scripts, project-local memory, known-issues +# scratch space. Session tooling, not committed code. See +# ~/.claude/CLAUDE.md §2a (Tool Selection) for the full rationale. +.claude/ + +# Generated docs (security runbooks etc.). Only the generated output +# directory is ignored; hand-written documentation under docs/ is +# tracked normally (e.g. docs/DEPLOYMENT.md, docs/DEVELOPMENT.md, and the +# docs/cli/ CLI reference). Do not re-add a blanket docs/* ignore: it +# silently hid hand-written docs, which is the regression this narrows away. +docs/generated/ + +# Graphify knowledge graph output — regenerated locally, not committed +graphify-out/ diff --git a/.gitleaksignore b/.gitleaksignore new file mode 100644 index 000000000..671857722 --- /dev/null +++ b/.gitleaksignore @@ -0,0 +1,3 @@ +# AWS official example credentials used in tests +# See: https://docs.aws.amazon.com/IAM/latest/UserGuide/security-creds.html#sec-access-keys-and-secret-access-keys +internal/secrets/aws_resolver_httptest_test.go diff --git a/.golangci.yml b/.golangci.yml new file mode 100644 index 000000000..6afe5382f --- /dev/null +++ b/.golangci.yml @@ -0,0 +1,167 @@ +version: "2" +run: + build-tags: + - integration + tests: true +linters: + enable: + - bodyclose + - errorlint + - gocritic + - gocyclo + - godot + - gosec + - misspell + - noctx + - prealloc + - revive + - unconvert + - unparam + - whitespace + settings: + errcheck: + check-type-assertions: true + check-blank: true + gocritic: + disabled-checks: + - dupImport + - ifElseChain + - octalLiteral + - whyNoLint + enabled-tags: + - diagnostic + - performance + - style + settings: + # aws.Config and similar SDK/config structs are designed to be passed + # by value (the AWS SDK v2 copies Config intentionally). 1024 bytes covers + # these idiomatic large-but-value-typed params without suppressing + # genuinely oversized plain domain structs. + hugeParam: + sizeThreshold: 1024 + gocyclo: + min-complexity: 15 + gosec: + excludes: + - G104 + severity: medium + confidence: medium + config: + G101: + ignore_entropy: false + pattern: (?i)passwd|pass|password|pwd|secret|token + govet: + enable-all: true + disable: + # fieldalignment: many production structs have alignment violations that + # pre-date this PR; fixing them is a separate concern. Suppress globally + # rather than per-struct since golangci-lint v2.10.x (CI) doesn't treat + # severity:warning as a non-fatal exit, unlike v2.11+. + - fieldalignment + misspell: + locale: US + revive: + rules: + - name: var-naming + disabled: false + - name: package-comments + disabled: true + - name: exported + disabled: false + - name: indent-error-flow + disabled: false + - name: error-return + disabled: false + - name: error-naming + disabled: false + - name: error-strings + disabled: false + exclusions: + generated: lax + presets: + - comments + - common-false-positives + - legacy + - std-error-handling + rules: + - linters: + - errcheck + - gocritic + - gocyclo + - gosec + path: _test\.go + # unusedwrite: test fixture struct fields may be set for documentation + # completeness even when only a subset of fields is read by a given subtest. + # Scoped to test files so production unusedwrite bugs remain visible. + # Note: fieldalignment is disabled globally in govet.disable (many + # pre-existing production violations; separate cleanup PR needed). + - linters: + - govet + text: "unusedwrite" + path: _test\.go + - linters: + - errcheck + - gosec + path: internal/testutil/ + - linters: + - all + path: .*_gen\.go + # revive var-naming: "api" is an intentional package name for the HTTP API + # layer; it is not a utility package and "avoid meaningless package names" + # is a false positive here. All 88+ files in internal/api use this name. + - linters: + - revive + path: internal/api/ + text: "avoid meaningless package names" + # revive var-naming: "common" is the established public package name in + # the nested pkg module; renaming it would break unaliased consumers. + - linters: + - revive + path: ^pkg/common/ + text: "avoid meaningless package names" + # noctx: exec.Command calls in configure CLI helpers intentionally + # omit context (CommandRunner interface contract; operator-controlled inputs). + # Suppressed per-file rather than per-line so #nosec G204 can lead the comment. + - linters: + - noctx + path: (cmd/configure_gcp|cmd/configure_azure)\.go + # noctx: http.Get and net.Listen in test helpers appropriately use the + # background context; threading a test context through every helper would + # add noise without improving test correctness. + - linters: + - noctx + path: _test\.go + # misspell: "cancelled" is the canonical DB column name (migration 000035, + # UK spelling throughout; migration #1277 tracks the rename). "initialised" + # appears in established comments in internal/config/types.go matching the + # existing UK convention. Allow both spellings in the files listed below. + - linters: + - misspell + path: (internal/config/(store_postgres|store_postgres_pgxmock_test|store_postgres_ladder|types|interfaces)|pkg/ladder/(store|types_test)|internal/api/(handler_history|handler_ri_exchange(_test)?|router_handlers_test|coverage_extras_test)|internal/mocks/stores|internal/purchase/scheduled_fire_test)\.go + text: "cancelled|initialised" + paths: + - third_party$ + - builtin$ + - examples$ + - node_modules +issues: + max-issues-per-linter: 0 + max-same-issues: 0 + new: false +severity: + default: warning + rules: + - linters: + - gosec + severity: error +formatters: + enable: + - gofmt + - goimports + exclusions: + generated: lax + paths: + - third_party$ + - builtin$ + - examples$ + - node_modules diff --git a/.hadolint.yaml b/.hadolint.yaml new file mode 100644 index 000000000..166310311 --- /dev/null +++ b/.hadolint.yaml @@ -0,0 +1,9 @@ +# Hadolint configuration +# https://github.com/hadolint/hadolint + +# Ignore rules that are impractical for alpine-based images +ignored: + # DL3018: Pin versions in apk add + # Alpine package versions change with each release and vary by architecture. + # Using --no-cache ensures fresh packages; pinning creates maintenance burden. + - DL3018 diff --git a/.markdownlint.yaml b/.markdownlint.yaml new file mode 100644 index 000000000..5eb615eba --- /dev/null +++ b/.markdownlint.yaml @@ -0,0 +1,15 @@ +# Markdownlint configuration +# Line length - disabled for technical docs with URLs, commands, and tables +MD013: false + +# Allow duplicate headings in different sections (e.g., multiple "Infrastructure Created") +MD024: + siblings_only: true + +# Tables: pin to GitHub-flavored convention (`| cell |` with leading and +# trailing whitespace inside each cell). Default `consistent` infers the +# style from the first table; if a separator row happens to look "compact", +# every aligned table downstream cascades-fails. Pinning the style avoids +# that and matches the convention every README in the repo already uses. +MD060: + style: leading_and_trailing diff --git a/.markdownlintignore b/.markdownlintignore new file mode 100644 index 000000000..6b2f94c6f --- /dev/null +++ b/.markdownlintignore @@ -0,0 +1,5 @@ +# Generated audit reports are verbatim records: reviewer text, quoted code +# excerpts and verifier verdicts. Autoformatting them rewrites evidence +# (for example stripping a meaningful trailing space inside a code span), +# so they are linted by the process that writes them, not by markdownlint. +docs/audits/ diff --git a/.pre-commit-config.yaml b/.pre-commit-config.yaml new file mode 100644 index 000000000..af6a3f9ec --- /dev/null +++ b/.pre-commit-config.yaml @@ -0,0 +1,277 @@ +# Pre-commit hooks configuration +# Install: pip install pre-commit +# Setup: pre-commit install + +repos: + # Go formatting and linting + - repo: https://github.com/dnephin/pre-commit-golang + rev: v0.5.1 + hooks: + - id: go-fmt + name: Run gofmt + - id: go-mod-tidy + name: Run go mod tidy + + - repo: local + hooks: + - id: go-vet + name: Run go vet + entry: bash -c 'go vet ./...' + language: system + pass_filenames: false + files: \.go$ + + # Terraform formatting + - repo: https://github.com/antonbabenko/pre-commit-terraform + rev: v1.105.0 + hooks: + - id: terraform_fmt + name: Terraform format + - id: terraform_validate + name: Terraform validate + - id: terraform_tflint + name: Terraform lint + args: + - --args=--config=__GIT_WORKING_DIR__/.tflint.hcl + + # General file checks + - repo: https://github.com/pre-commit/pre-commit-hooks + rev: v6.0.0 + hooks: + - id: trailing-whitespace + name: Trim trailing whitespace + - id: end-of-file-fixer + name: Fix end of files + - id: check-yaml + name: Check YAML syntax + exclude: '^(docker-compose.*\.yml|\.github/workflows/.*|cloudformation/.*|iac/.*/cloudformation/.*)$' + - id: check-json + name: Check JSON syntax + - id: check-added-large-files + name: Check for large files + args: ['--maxkb=1024'] + - id: check-merge-conflict + name: Check for merge conflicts + - id: detect-private-key + name: Detect private keys + # Audit (PR5): internal/credentials/resolver.go contains zero + # private-key-shaped patterns (verified by grep). The exclusion + # was a historical artifact; removing it tightens the gate + # without breaking any legitimate code. + exclude: '(_test\.go|frontend/src/index\.html)$' + - id: check-case-conflict + name: Check for case conflicts + + # Dockerfile linting + # + # The upstream hadolint-docker hook (hadolint/hadolint's own + # .pre-commit-hooks.yaml) declares `entry: ghcr.io/hadolint/hadolint + # hadolint` with no image tag, so Docker resolves it to `:latest` on every + # run -- independent of any `rev:` pin, which only pins which commit of the + # *hook definition* is used, not the Docker image it runs. When upstream + # published hadolint 2.15.1 as `latest`, CI silently started linting with a + # newer ruleset (new DL3066/DL3025 findings on unchanged Dockerfiles) with no + # corresponding change in this repo, and `main` went red (issue #1695). + # + # Defined as a local hook instead, pinned to an immutable digest, so the + # linter version can only change through an explicit edit to this line. + # + # Pinned to 2.15.1 -- the version that surfaced the findings -- rather than + # back to 2.14.0. Pinning to the older image would also have made CI green, + # but only by un-seeing findings that are real; the four it reported are + # fixed in the Dockerfiles rather than silenced. Bumping this digest is + # expected to require fixing whatever the new version finds. + - repo: local + hooks: + - id: hadolint + name: Lint Dockerfiles + language: docker_image + entry: ghcr.io/hadolint/hadolint:v2.15.1@sha256:32dac94127fd60b7b7e3fbfc65e1383b9b5e25c9bfd7b8536de7a539fe68a12d hadolint + types: [dockerfile] + + # GitHub Actions workflow linting + # + # Runs the same two linters as ci.yml's workflow-lint job, at the same pinned + # versions, so an expression injection is caught before the push rather than + # after. Keep the versions here and in ci.yml in lockstep. + # + # actionlint is a local docker_image hook rather than the upstream + # rhysd/actionlint hook for the reason spelled out above for hadolint: the + # upstream actionlint-docker hook's entry pins the image by mutable tag, and + # `rev:` only pins the hook definition, not the image it runs. This entry + # pins the digest, so the ruleset can only change by editing this line. + # + # The image bundles shellcheck. actionlint silently skips every run: block + # and still exits 0 when shellcheck is absent from PATH, which is how the + # rollback.yml injection (#1542) went unflagged, so a system hook that + # depends on each developer having shellcheck installed would reintroduce + # exactly that gap. + - repo: local + hooks: + - id: actionlint + name: Lint GitHub Actions workflows + language: docker_image + entry: rhysd/actionlint:1.7.12@sha256:b1934ee5f1c509618f2508e6eb47ee0d3520686341fec936f3b79331f9315667 + types: [yaml] + files: ^\.github/workflows/ + + # zizmor is what actually closes the injection gap: actionlint only + # catches this class indirectly, through shellcheck on the expanded + # script, and against the pre-fix rollback.yml it never flagged the + # injected heredoc at all. zizmor's template-injection audit named that + # exact line. Upstream publishes no .pre-commit-hooks.yaml, so this is a + # local hook with the version pinned in additional_dependencies. + # + # Flags mirror ci.yml exactly; keep them in lockstep. --persona=pedantic + # because the default regular persona hides high-severity findings this + # repo had, and --min-severity=medium is a threshold rather than a + # suppression: no baseline file and no per-finding ignores, so every + # medium and high finding fails. The pre-fix injection scored high. + - id: zizmor + name: Audit GitHub Actions workflows for injection + language: python + additional_dependencies: ['zizmor==1.29.0'] + entry: zizmor --offline --persona=pedantic --min-severity=medium + types: [yaml] + files: ^\.github/workflows/ + + # Markdown linting + - repo: https://github.com/igorshubovych/markdownlint-cli + rev: v0.47.0 + hooks: + - id: markdownlint + name: Lint Markdown files + args: ['--fix'] + + # Code quality checks + - repo: local + hooks: + - id: gocyclo + name: Check cyclomatic complexity + entry: bash -c 'gocyclo -over 10 $(git ls-files "*.go" | grep -v _test.go | grep -v vendor/) || (echo "⚠️ Functions with cyclomatic complexity over 10 detected. Please refactor." && exit 1)' + language: system + pass_filenames: false + files: \.go$ + + # Security scanning + - repo: local + hooks: + - id: git-secrets + name: Scan for AWS secrets + entry: git-secrets --scan + language: system + types: [file] + + - id: gosec + name: Go security scanner (per-module, per-changed-package) + # Scans only the Go packages that contain staged files, resolved to + # their owning module (root, pkg/, providers/aws, providers/azure, + # providers/gcp). Fast: never whole-repo, always per-changed-package. + # + # gosec v2.28.0 is auto-installed to + # ~/.cache/pre-commit-gosec/v2.28.0/gosec on first use. + # + # Exclusion rationale and flag list live in scripts/gosec-hook.sh. + # Keep in sync with the exclude= flags there. + entry: bash scripts/gosec-hook.sh + language: system + pass_filenames: true + files: \.go$ + + - id: trivy-config + name: Trivy config scanner + # Trivy is a required tool: the previous fallback `|| echo + # "skipping"` masked an absent gate, which is worse than no gate. + # Install via `brew install trivy` or `apt install trivy`. CI + # installs trivy v0.69.3 via the workflow step in + # .github/workflows/pre-commit.yml, so PRs are always scanned + # regardless of a developer's local setup. + # + # `**/.terraform` is skipped because it is a gitignored, generated + # local cache (downloaded provider schemas + cached plan snapshots + # from `terraform init`) — never committed and never present in CI + # (which checks out from git), so findings there are local-only + # false positives on third-party provider cache, not on our source. + # The actual terraform module source files are still fully scanned. + entry: bash -c 'trivy config --severity HIGH,CRITICAL --exit-code 1 --skip-dirs terraform/environments/aws --skip-dirs "**/.terraform" --skip-dirs .claude .' + language: system + pass_filenames: false + + # Migration checks + - repo: local + hooks: + - id: check-migration-conflicts + name: Check for conflicting migration numbers + entry: bash -c 'dups=$(ls internal/database/postgres/migrations/*.up.sql 2>/dev/null | sed "s/.*\///" | cut -c1-6 | sort | uniq -d); if [ -n "$dups" ]; then echo "Duplicate migration number(s) found:"; echo "$dups"; exit 1; fi' + language: system + pass_filenames: false + files: ^internal/database/postgres/migrations/ + + # Permissions codegen: regenerate frontend/src/permissions.generated.ts + # from internal/auth/types.go and fail if the committed copy is stale. + # Triggers on changes to the backend defaults or the generator itself, + # plus the generated file (in case a dev hand-edits it). + - repo: local + hooks: + - id: permissions-codegen + name: Regenerate frontend permissions from Go defaults + entry: bash -c 'go run ./cmd/gen-permissions && git diff --exit-code -- frontend/src/permissions.generated.ts || { echo "permissions.generated.ts is stale. Run go run ./cmd/gen-permissions and commit the result."; exit 1; }' + language: system + pass_filenames: false + files: ^(internal/auth/types\.go|cmd/gen-permissions/.*\.go|frontend/src/permissions\.generated\.ts)$ + + # Heavy test execution: pre-push stage only. + # + # These three hooks rebuild + run the full Go and frontend test suites, + # which is ~6-7 min of work and the bulk of the CI pre-commit job's + # runtime. They are *redundant in CI* — the same suites are run by + # dedicated workflows that PRs and pushes already trigger: + # + # - go-test (-short -race ./...) : ci.yml `unit-tests` runs the same + # suite with -race AND an integration + # pass with -tags=integration. + # - frontend-build (npm run build): frontend-build.yml runs npm run + # typecheck + npm run build on PRs; + # frontend-build-sentinel.yml runs + # the build on every push to main / + # feat/**. + # - frontend-test (jest) : frontend-build-sentinel.yml runs + # `npx jest --no-coverage --silent` + # on every push to feat/** (which + # fires on every PR-branch update). + # + # Moving them to the pre-push stage keeps the local safety net (devs + # who run `pre-commit install --hook-type pre-push` still get these + # tests on `git push`) while letting the CI pre-commit workflow stay + # focused on style/security/syntax. Pre-commit's default stage filter + # is `pre-commit`, so the CI workflow's `pre-commit run --all-files` + # skips these hooks automatically. + - repo: local + hooks: + - id: go-test + name: Run Go tests (pre-push only; CI covers via ci.yml) + entry: bash -c 'go test -short -race ./...' + language: system + pass_filenames: false + files: \.go$ + stages: [pre-push] + + - id: frontend-build + name: Build frontend (pre-push only; CI covers via frontend-build.yml) + entry: bash -c 'cd frontend && npm run build' + language: system + pass_filenames: false + files: ^frontend/src/ + stages: [pre-push] + + - id: frontend-test + name: Run frontend tests (pre-push only; CI covers via frontend-build-sentinel.yml) + entry: bash -c 'cd frontend && npx jest --no-coverage --silent' + language: system + pass_filenames: false + files: ^frontend/src/ + stages: [pre-push] + +# Global configuration +default_stages: [pre-commit, pre-push] +fail_fast: false diff --git a/.snyk b/.snyk new file mode 100644 index 000000000..1ee37bf2a --- /dev/null +++ b/.snyk @@ -0,0 +1,43 @@ +# Snyk (https://snyk.io) policy file +# Used to ignore specific vulnerabilities or set severity thresholds + +# Ignore specific vulnerabilities +ignore: + # Example: Ignore specific CVE + # 'SNYK-GOLANG-GITHUBCOMAWSAWSSDKGOSERVICE-1234567': + # - '*': + # reason: False positive or accepted risk + # expires: 2024-12-31T00:00:00.000Z + +# Language-specific settings +patch: {} + +# Exclude paths from scanning +exclude: + global: + - '**/*_test.go' + - '**/testdata/**' + - '**/vendor/**' + - '**/node_modules/**' + - '**/.terraform/**' + - '**/migrations/**' + +# Severity thresholds +failOnSeverity: high + +# License policy +license: + # Allow these licenses + allow: + - MIT + - Apache-2.0 + - BSD-2-Clause + - BSD-3-Clause + - ISC + - MPL-2.0 + + # Deny these licenses + deny: + - GPL-2.0 + - GPL-3.0 + - AGPL-3.0 diff --git a/.tflint.hcl b/.tflint.hcl new file mode 100644 index 000000000..84ef5fcd6 --- /dev/null +++ b/.tflint.hcl @@ -0,0 +1,95 @@ +# TFLint configuration for CUDly +# https://github.com/terraform-linters/tflint + +config { + # Enable module inspection (call_module_type replaces deprecated 'module' in v0.54+) + call_module_type = "all" + + # Force to return an error when issues are found + force = false +} + +# AWS Plugin +plugin "aws" { + enabled = true + version = "0.32.0" + source = "github.com/terraform-linters/tflint-ruleset-aws" +} + +# Azure Plugin +plugin "azurerm" { + enabled = true + version = "0.27.0" + source = "github.com/terraform-linters/tflint-ruleset-azurerm" +} + +# Google Cloud Plugin +plugin "google" { + enabled = true + version = "0.30.0" + source = "github.com/terraform-linters/tflint-ruleset-google" +} + +# Terraform language rules +rule "terraform_deprecated_interpolation" { + enabled = true +} + +rule "terraform_deprecated_index" { + enabled = true +} + +rule "terraform_unused_declarations" { + enabled = true +} + +rule "terraform_comment_syntax" { + enabled = true +} + +rule "terraform_documented_outputs" { + enabled = true +} + +rule "terraform_documented_variables" { + enabled = true +} + +rule "terraform_typed_variables" { + enabled = true +} + +rule "terraform_module_pinned_source" { + enabled = true +} + +rule "terraform_naming_convention" { + enabled = true + format = "snake_case" +} + +rule "terraform_required_version" { + enabled = true +} + +rule "terraform_required_providers" { + enabled = true +} + +rule "terraform_standard_module_structure" { + enabled = true +} + +rule "terraform_workspace_remote" { + enabled = true +} + +# Disable azurerm rules that crash on sensitive for_each values +# https://github.com/terraform-linters/tflint-ruleset-azurerm/issues/563 +rule "azurerm_key_vault_invalid_name" { + enabled = false +} + +rule "azurerm_key_vault_invalid_sku_name" { + enabled = false +} diff --git a/.trivyignore b/.trivyignore new file mode 100644 index 000000000..48f88492f --- /dev/null +++ b/.trivyignore @@ -0,0 +1,111 @@ +# Trivy ignore file — config-scan suppressions +# +# Each entry below documents an accepted finding from `trivy config`. +# The findings predate the PR5 supply-chain hardening and are +# intentional design choices for the current threat model. When any +# underlying terraform changes (especially the networking module), +# re-evaluate these — they are accepted on the current shape, not +# forever. + +# CloudFront distribution without WAF. +# Justification: CUDly's dashboard distribution serves a small number +# of authenticated users; a WAF would add operational cost and latency +# disproportionate to the threat model. Revisit if the distribution +# starts serving anonymous traffic. +AVD-AWS-0011 + +# CloudFront minimum TLS protocol version. +# Justification: pre-existing default (TLS 1.0) was tightened in a +# prior PR; the trivy ID still trips when the value is inherited +# rather than set explicitly. Tracked as a follow-up to pin +# TLSv1.2_2021 explicitly in modules/frontend/aws. +AVD-AWS-0013 + +# ALB drop_invalid_header_fields. +# Justification: CUDly does not pass arbitrary client headers to +# upstream services in a security-sensitive way; the Lambda backend +# parses the request body, not arbitrary headers. Tracked as a +# hardening follow-up. +AVD-AWS-0052 + +# Public-facing ALB. +# Justification: the ALB is intentionally public — it's the entry +# point for the dashboard. Internal-only would defeat the product. +AVD-AWS-0053 + +# Egress security group rule unrestricted. +# Justification: outbound to AWS service endpoints (Cost Explorer, +# Pricing API, multi-region cloud APIs) requires broad egress; AWS +# does not publish a stable IP range for "all AWS APIs the SDK might +# use". Mitigations are at the IAM layer (see PR #103) which +# restricts what the runtime can DO, not where it can reach. +AVD-AWS-0104 + +# S3 bucket encryption with default keys. +# Justification: dashboard assets bucket — no PII stored. Customer- +# managed key would add KMS cost without proportionate benefit. +AVD-AWS-0132 + +# SNS topic encryption with default keys. +# Justification: same as S3. SNS messages contain only execution +# notification metadata, no credentials. +AVD-AWS-0136 + +# Public subnet has map_public_ip_on_launch=true. +# Justification: the public subnets are deliberately public — they +# host the NAT gateway and the public ALB. The runtime Lambda lives +# in the private subnets and routes egress through the NAT. +AVD-AWS-0164 + +# Azure Function app HTTPS enforcement. +# Justification: the cleanup Function in +# terraform/modules/compute/azure/cleanup-function/ is invoked by +# Azure-internal triggers (timers + queue), never directly by external +# HTTP. https_only=true would still be correct hygiene; tracked as a +# follow-up to set explicitly. +AVD-AZU-0004 + +# Azure storage account network rules default allow. +# Justification: the cleanup function's storage account is in the +# same vnet as the function and is firewall-restricted at the Azure +# resource-group level. The trivy ID still trips because the network +# rules block isn't declared in the resource itself. +AVD-AZU-0012 + +# Azure Key Vault network ACL default action. +# ID: AZU-0013 (surfaced by trivy v0.70.0; v0.69.3 did not report this ID). +# Justification: the vault is in terraform/modules/secrets/azure/main.tf and +# exposes default_network_acl_action as a variable with a safe default. The +# network ACL is set at instantiation time; the module itself cannot enforce +# a default_action value without knowing the caller's network topology. +# Tracked as a hardening follow-up to set default_action = "Deny" in the +# module default. +AZU-0013 + +# Azure NSG rule allows unrestricted ingress. +# ID: AZU-0047 (surfaced by trivy v0.70.0; v0.69.3 did not report this ID). +# Justification: the networking module (terraform/modules/networking/azure/) +# uses a permissive NSG rule to allow HTTPS ingress from the internet to the +# Application Gateway. Restricting source IPs would break public access. +# Mitigations are at the Azure Front Door / Application Gateway WAF layer. +AZU-0047 + +# GCP Cloud SQL instance does not require TLS. +# ID: GCP-0015 (surfaced by trivy v0.70.0; v0.69.3 did not report this ID). +# Justification: the database module (terraform/modules/database/gcp/main.tf) +# exposes require_ssl as a variable. TLS is enforced at the application layer +# via Go's pgx driver TLS config. Tracked as a hardening follow-up to set +# require_ssl = true in the module default. +GCP-0015 + +# GCP Cloud Storage bucket: cleanup Cloud Function source bucket. +# ID: GCP-0001 (surfaced by trivy v0.70.0; v0.69.3 did not report this ID). +# Justification: the only GCS bucket in the GCP deployment is the cleanup +# Cloud Function source bucket (terraform/modules/compute/gcp/cleanup-function/main.tf). +# This bucket is NOT public: it has uniform_bucket_level_access = true and +# public_access_prevention = "enforced". Trivy may still flag the resource if +# it does not recognise the public_access_prevention attribute. +# The GCP frontend is served from Cloud Run (not a GCS bucket), so no frontend +# build artifacts -- including source maps (*.map, produced by hidden-source-map +# in webpack.config.js) -- are uploaded to any GCS bucket. +GCP-0001 diff --git a/CHANGELOG.md b/CHANGELOG.md new file mode 100644 index 000000000..643c1df69 --- /dev/null +++ b/CHANGELOG.md @@ -0,0 +1,173 @@ +# Changelog + +All notable changes to CUDly are documented in this file. +The format is based on [Keep a Changelog](https://keepachangelog.com/). + +## [Unreleased] + +### Notices + +- **Federation IaC bundles downloaded before 2026-04-22 need to be + re-downloaded** to get zero-touch registration. Older bundles silently + skip auto-registration unless manually edited (Terraform `registration.tf` + gated `do_register` on `cudly_api_url`; CLI shell scripts included the + registration call only when `CUDlyAPIURL` was present at render time; + CloudFormation deploy scripts had no registration call at all). + Re-download the bundle from the CUDly UI and the new copy will register + your account automatically with no manual edits required. +- **Federation IaC bundles deployed before #1219 need to be re-applied** to + pick up reconciled IAM action grants. Older bundles silently degrade on + federated accounts: the Cost Explorer `Get*Coverage` actions are missing + (coverage-targeted sizing assumes zero existing coverage), and the + cross-account CloudFormation flavor lacks the optional `EnableOrgDiscovery` + parameter and the legacy CE statement that the original `CUDly-CrossAccount` + template carried. Re-download the federation bundle from the CUDly UI and + re-apply (`terraform apply` / `aws cloudformation update-stack`) -- no + manual edits required, no changes to existing CUDly resources. Customers + on the runtime CloudFormation stack or Terraform lambda/fargate modules + also need an update to pick up the new `ec2:*ReservedInstancesExchangeQuote` + actions, `ec2:DescribeRegions`, `rds:DescribeDBInstances`, and the + new-style `es:*ReservedInstance*` OpenSearch actions (replacing the legacy + `es:*ReservedElasticsearch*` names). + +### Fixed + +- Remove debug console.log from frontend recommendation handler +- Align pre-commit gocyclo threshold (10) with CI pipeline +- Pin tool versions in GitHub Actions for reproducible builds +- Update README Go version badge to match go.mod (1.25+) + +## [0.9.0] - 2026-03-06 + +### Added + +- RI Exchange feature: reshape analysis with normalization factors, API + endpoints, and frontend page for managing convertible Reserved Instances +- RI utilization tracking from Cost Explorer with pagination support +- Convertible RI listing in EC2 client +- Security headers on all Lambda responses +- Admin password resolution from cloud secret managers + +### Fixed + +- Harden RI exchange handlers with validation and error sanitization +- Fix async race conditions and input validation in RI exchange frontend +- Fix base64 encoding for saveProfile and resetPassword +- Remove duplicate logout event handler +- Guard DNS zone outputs against missing resources (GCP, Azure) +- Fix Azure CDN redirect type and SPA routing +- Add network policies and resource quotas to AKS module +- Add security headers to Azure Front Door and GCP load balancer +- Wire admin password secrets through all cloud environment root modules + +## [0.8.0] - 2026-02-01 + +### Added + +- Deployment health check blocks for AWS, Azure, and GCP Terraform modules +- GCP self-signed cert for dev HTTPS +- Azure Front Door API routing and custom domain support +- Cross-provider deployment test harness script +- Azure ACR resource and registry authentication + +### Fixed + +- Enforce SSL-only connections on GCP Cloud SQL +- Migrate GCP load balancer to EXTERNAL_MANAGED with SPA routing +- Fix Azure Container Apps config and CDN delivery rule names +- Fix GCP frontend build trigger and database password generation +- Expand frontend CSP connect-src for Azure and GCP API origins +- Fix Fargate EventBridge container name +- Capture migration exit code correctly in entrypoint.sh + +### Changed + +- Convert AWS database from Aurora Serverless v2 to standalone RDS +- Move GCP Secret Manager out of database module +- Replace Azure Container App Jobs with Logic Apps scheduled tasks +- Simplify Azure database module + +## [0.7.0] - 2026-01-15 + +### Added + +- Full Terraform infrastructure for AWS (Fargate, Lambda, CloudFront, RDS), + Azure (Container Apps, AKS, Front Door, PostgreSQL), and GCP (Cloud Run, + GKE, Cloud SQL) with CI-specific tfvars +- PostgreSQL database with connection pool, migrations, and secret resolvers +- Authentication service with RBAC and API key support +- REST API with rate limiting, CORS, and middleware stack +- Email service with SMTP sender and cloud credential resolution +- Analytics collector, purchase execution, and scheduled task runner +- Docker containerization with multi-stage builds and compose configs +- GitHub Actions CI/CD pipeline (lint, test, security scan, Docker build, + Terraform validate, E2E tests, Infracost) +- Frontend web dashboard with TypeScript, webpack, Chart.js + +### Fixed + +- Sanitize user input in dashboard and recommendations (XSS prevention) +- Add connection pool limits and graceful shutdown to server +- Add nil checks across Azure service clients +- Enforce 12-char minimum password with complexity requirements +- Use hidden-source-map for production frontend builds +- Use rightmost X-Forwarded-For IP for client identification +- Add SHA256 checksum verification for migrate binary in Docker +- Tighten git-secrets patterns to reduce false positives + +## [0.6.0] - 2025-11-01 + +### Added + +- Database Savings Plans support and SP type filtering +- OSL-3.0 license and contributing guidelines + +### Fixed + +- RDS RI purchase failing on details assertion and invalid reservation ID +- OpenSearch RI resource type and offering lookup +- Deduplicate reservation ID sanitization into pkg/common + +### Changed + +- Refactor internal packages to providers (aws, azure, gcp) +- Add provider-specific mocking infrastructure and tests + +## [0.5.0] - 2025-09-01 + +### Added + +- Multi-cloud support (Azure experimental, GCP experimental) +- API-based RDS extended support detection +- Instance type validation system +- CSV reader for recommendation import +- Duplicate RI purchase prevention +- Account alias lookup +- Confirmation prompt and instance limit features + +### Changed + +- Replace global variables with Config struct pattern +- Improve rate limiting and test performance +- Refactor all purchase clients with enhanced error handling + +## [0.4.0] - 2025-07-01 + +### Added + +- Multi-service RI support: EC2, ElastiCache, MemoryDB, OpenSearch, Redshift +- Multi-service orchestration and CLI +- Comprehensive test coverage (80%+ across packages) + +### Fixed + +- CSV pricing calculations to use AWS-provided cost data + +## [0.3.0] - 2025-05-01 + +### Added + +- Initial CLI tool for RDS Reserved Instance purchasing +- Recommendations fetching from AWS Cost Explorer +- CSV output for analysis results +- Go module setup with AWS SDK v2 diff --git a/CLAUDE.md b/CLAUDE.md new file mode 100644 index 000000000..8cd91dc75 --- /dev/null +++ b/CLAUDE.md @@ -0,0 +1,427 @@ +# Claude Code Configuration - RuFlo V3 + +## Behavioral Rules (Always Enforced) + +- Do what has been asked; nothing more, nothing less +- NEVER create files unless they're absolutely necessary for achieving your goal +- ALWAYS prefer editing an existing file to creating a new one +- NEVER proactively create documentation files (*.md) or README files unless explicitly requested +- NEVER save working files, text/mds, or tests to the root folder +- Never continuously check status after spawning a swarm — wait for results +- ALWAYS read a file before editing it +- NEVER commit secrets, credentials, or .env files + +## Planning (ALWAYS follow this process) + +- **EVERY TIME** you create or update a plan, you MUST enter a review loop: thoroughly review the plan and fix any issues found. Repeat until 3 consecutive review passes find no issues. Do NOT skip this step — it is mandatory for all plans, no exceptions. +- In each loop iteration, print a summary of found issues before and after fixing them. + +## File Organization + +- NEVER save working files to the root folder; use the directories below +- `cmd/`: CLI and server entry points (main packages) +- `internal/`: backend application code (API, auth, purchase, scheduler, ...) +- `pkg/`: shared library code (separate Go module, see Go Module Notes) +- `providers/`: cloud provider integrations (AWS, Azure, GCP) +- `frontend/`: TypeScript web frontend (webpack + jest) +- `terraform/`, `cloudformation/`, `arm/`, `iac/`: infrastructure as code +- `docs/`: documentation and markdown files +- `scripts/`: utility scripts +- `tests/`: end-to-end tests (Go unit tests live next to the code they test) + +## Project Architecture + +- Follow Domain-Driven Design with bounded contexts +- Keep files under 500 lines +- Use typed interfaces for all public APIs +- Prefer TDD London School (mock-first) for new code +- Use event sourcing for state changes +- Ensure input validation at system boundaries + +### Project Config + +- **Topology**: hierarchical-mesh +- **Max Agents**: 15 +- **Memory**: hybrid +- **HNSW**: Enabled +- **Neural**: Enabled + +## Go Module Notes + +- This project does NOT use a vendor directory. Do not use `go mod vendor`. +- The `pkg/` directory is a separate Go module (`github.com/LeanerCloud/CUDly/pkg`) with a `replace` directive in the root `go.mod`. +- Build and test normally with `go build ./...` and `go test ./...` from the root. +- Run `go test ./pkg/...` from the `pkg/` directory when working on the submodule. + +## Build & Test + +The root of the repo is a Go project; the npm scripts live in `frontend/`. + +```bash +# Build (backend, from the repo root) +go build ./... # or: make build + +# Test (backend) +go test ./... # or: make test-unit + +# Lint (backend) +make lint # golangci-lint; also: make vet, make fmt + +# Frontend (run from frontend/) +cd frontend && npm ci +npm run build # webpack production build +npm test # jest --coverage +npm run lint # eslint src/**/*.ts +``` + +- ALWAYS run tests after making code changes +- ALWAYS verify build succeeds before committing + +## Known Issues + +The `known_issues/` directory tracks deferred tech debt and surfaced bugs. +When a referenced GitHub issue is closed, move the corresponding doc to +`known_issues/resolved/` (do not delete it) so the rationale is preserved. +Do this in the same PR that closes the issue. A full sweep of the directory +should happen at the start of each sprint. Full convention and entry format +are in `CONTRIBUTING.md` under "Known Issues Sweep". + +## Post-push CI watcher (MANDATORY — even for one-line fix commits) + +After **every** `git push` that publishes new commits to a PR branch on +this repo (including follow-up CodeRabbit-nitpick fixes), launch a +background watcher per workflow run **before** ending the turn. This is +the project-level reinforcement of the global rule in +`~/.claude/git-workflow.md` §Post-push CI watcher — CUDly's pre-commit +job alone takes 9–15 min and is silently broken by routine changes +(missing CI tools, pre-existing markdownlint debt, git-secrets +allowlist drift), so leaving a push unwatched routinely lets CI +failures sit overnight. + +Mechanics: + +1. After `git push`, list the runs the push triggered: + + ```bash + gh run list --repo LeanerCloud/CUDly --commit "$(git rev-parse HEAD)" \ + --limit 10 --json databaseId,workflowName,status + ``` + +2. For each run still in `queued`/`in_progress`, launch one background + watcher script via `Bash` with `run_in_background: true` (do NOT use + foreground polling — the main session must stay unblocked). The + minimum viable watcher polls `gh run view --json status,conclusion` + every 30s and on completion either reports success or dumps the failed + step digest. A reusable template lives at + `.claude/scripts/watch-ci-run.sh` if present, or write a one-shot to + `/tmp/claude/watch--.sh`. + +3. If a watcher reports failure, the same session investigates and + pushes a fix on the same branch (which fires a fresh watcher round). + Decisions that need a human go to PR comments; everything else is + autonomous per the global rule. + +Forgetting this rule has been a recurring failure mode in this +project. Before declaring a "pushed and done" turn complete, confirm +at least one `ci-watch-*` background task is armed. + +## CodeRabbit loop — iterate to silence (MANDATORY) + +CodeRabbit reviews this repo on every push to a PR branch. The full +rules live in `~/.claude/git-workflow.md` §"Post-PR review loop" +(§§3, 3a) — read them. The minimum-viable loop for this project: + +1. After every push, ping `@coderabbitai review` on the PR (CR doesn't + always re-review automatically; the explicit ping makes it + deterministic). +2. Wait for the review (60–120s polling, soft-handle 429s). +3. Triage every Actionable / Outside-diff / Nitpick finding into: + actionable-fix-now, dismiss-with-justification-on-thread, or + genuine-nitpick-batch-into-one-fix-commit. +4. Push the fix(es), comment on the PR summarising what was addressed + vs. dismissed (and why), end the comment with a fresh + `@coderabbitai review` ping. +5. **Repeat until CR's most recent review has zero Actionable items + AND every Nitpick is either fixed or has a justification reply.** + "I fixed pass 1" is not loop-exit. CUDly PRs commonly run 3–6 CR + passes before settling. +6. If a fix push triggers conflicts (`mergeStateStatus: DIRTY`), resolve + them per `~/.claude/git-workflow.md` §3a — `git rebase + origin/`, atomic conflict resolution, `git push + --force-with-lease`, post a rebase note on the PR, then continue + the CR loop. + +Forgetting this rule leaves CR threads silently unresolved and pushes +the triage burden onto the human reviewer. + +**When delegating PR work to a subagent**: the prompt MUST include the +full CR loop, not stop at the first `@coderabbitai review` ping. A fork +that pushes the PR, pings CR, then exits leaves the CR threads +unresolved — same failure mode as forgetting the post-push CI watcher. +The subagent's exit criteria must be: "CR's most recent review has zero +Actionable items AND every Nitpick is either fixed or has a +justification reply on the thread", not "PR opened and CR pinged". When +in doubt, copy the iteration loop above (steps 2–6) into the fork +prompt verbatim. + +## PR labeling — mirror closing-issue labels (MANDATORY) + +Every PR opened in this repo must carry the **same** triage labels as +the issue it closes — `priority/*`, `severity/*`, `urgency/*`, +`impact/*`, `effort/*`, `type/*`, plus `triaged` if the issue carries +it. Skipping this leaves PRs invisible to the same priority queries +that surface the issues, so an unlabeled PR is effectively +unreviewable in priority order. + +Mechanics — fold into the **same `gh pr create` round**, before pinging +CodeRabbit: + +```bash +# Right after `gh pr create ...` returns the PR URL: +# Derive PR_NUM from the current branch context (avoids brittle hand-copying). +PR_NUM=$(gh pr view "$(git rev-parse --abbrev-ref HEAD)" --repo LeanerCloud/CUDly --json number --jq '.number') +ISSUE_NUM= + +LABELS=$(gh issue view "$ISSUE_NUM" --repo LeanerCloud/CUDly --json labels \ + --jq '[.labels[].name | select(test("^(priority|severity|urgency|impact|effort|type)/")) ] + + (if [.labels[].name] | any(. == "triaged") then ["triaged"] else [] end) + | join(",")') + +# Guard against empty $LABELS — gh pr edit --add-label "" fails, which would +# silently break this MANDATORY flow. If the closing issue has no triage +# labels in the selected classes, surface the gap deterministically instead. +if [ -n "$LABELS" ]; then + gh pr edit "$PR_NUM" --repo LeanerCloud/CUDly --add-label "$LABELS" +else + echo "WARN: issue #$ISSUE_NUM has no priority/severity/urgency/impact/effort/type labels" + echo " Triage the issue first, then re-run the label-mirror step." + echo " Surface this gap in the PR body or as a comment on issue #$ISSUE_NUM." +fi + +# Verify +gh pr view "$PR_NUM" --repo LeanerCloud/CUDly --json labels \ + --jq '[.labels[].name] | sort | join(",")' +``` + +If the closing issue lacks the `triaged` label, do NOT apply +`triaged` to the PR — that would lie. Surface the gap in the PR body +(or as a comment on the issue) so the human can triage. + +If a label doesn't yet exist in the repo (rare — the label set is +populated from the existing issue queue), `gh label create` it with +the same color/description as a sibling label BEFORE applying. + +For PRs that close more than one issue (e.g., a PR that closes +`#A` and `#B`): take the **highest** `priority/*` and `severity/*` +across the issues; `union` the rest (`type/*`, `effort/*`, `impact/*`, +etc.). The PR represents the work for both, so it should be +discoverable under either filter. + +Forgetting this rule has the same shape as forgetting the post-push +CI watcher: it silently breaks priority-ordered review and triage. +Before declaring a "PR opened and done" turn complete, confirm the +label set on the PR matches the closing issue's set (or the merged +set for multi-close PRs). + +## Security Rules + +- NEVER hardcode API keys, secrets, or credentials in source files +- NEVER commit .env files or any file containing secrets +- Always validate user input at system boundaries +- Always sanitize file paths to prevent directory traversal +- Run `npx @claude-flow/cli@latest security scan` after security-related changes + +## CI/CD IAM — bootstrap vs runtime split + +The per-cloud `terraform/environments/*/ci-cd-permissions/` modules provision +the CI/CD deploy identities and are **applied once, manually, by a privileged +human** — not by the CI workflow itself. The main deploy workflow assumes a +deploy SA already exists and only has permission to manage workloads. Keep +this split when adding new IAM: + +- **Bootstrap-only permissions** (AWS `iam:*`, Azure RBAC role assignments, + GCP `roles/iam.roleAdmin`, `roles/resourcemanager.projectIamAdmin`, + `roles/cloudkms.admin`) live in `ci-cd-permissions/`. They let the deploy + SA manage its own downstream grants but are not granted to anything + ephemeral. +- **Runtime permissions** for the Lambda / Cloud Run / Container App service + accounts are defined in the per-cloud compute module (`modules/compute/ + {aws,gcp,azure}/...`) with the **narrowest possible scope**. Prefer custom + roles (GCP `google_project_iam_custom_role`) or prefixed resource ARNs + (AWS `arn:aws:iam::*:role/{prefix}*`) over broad predefined roles like + `roles/compute.admin` or `Resource = "*"`. +- **No silent fallbacks to over-privileged roles.** If a runtime grant + requires a bootstrap permission the deploy SA doesn't have, the apply + SHOULD 403 — that's the signal to re-run the bootstrap, not to paper over + with a wider grant. Fallback flags are allowed only as short-term + workarounds and must be removed once the bootstrap has been re-applied. +- **GCP WIF attribute_condition** in `ci-cd-permissions/github_oidc.tf` + restricts which branch can impersonate the deploy SA. Re-applying the + module with a different `deploy_ref` (or the default) resets the + condition. Pin `deploy_ref` in `terraform.tfvars` (gitignored, per-env) + to avoid silently locking out the current feature branch. + +## Concurrency: 1 MESSAGE = ALL RELATED OPERATIONS + +- All operations MUST be concurrent/parallel in a single message +- Use Claude Code's Task tool for spawning agents, not just MCP +- ALWAYS batch ALL todos in ONE TodoWrite call (5-10+ minimum) +- ALWAYS spawn ALL agents in ONE message with full instructions via Task tool +- ALWAYS batch ALL file reads/writes/edits in ONE message +- ALWAYS batch ALL Bash commands in ONE message + +## Swarm Orchestration + +- MUST initialize the swarm using CLI tools when starting complex tasks +- MUST spawn concurrent agents using Claude Code's Task tool +- Never use CLI tools alone for execution — Task tool agents do the actual work +- MUST call CLI tools AND Task tool in ONE message for complex work + +### 3-Tier Model Routing (ADR-026) + +| Tier | Handler | Latency | Cost | Use Cases | +| ------ | --------- | --------- | ------ | ----------- | +| **1** | Agent Booster (WASM) | <1ms | $0 | Simple transforms (var→const, add types) — Skip LLM | +| **2** | Haiku | ~500ms | $0.0002 | Simple tasks, low complexity (<30%) | +| **3** | Sonnet/Opus | 2-5s | $0.003-0.015 | Complex reasoning, architecture, security (>30%) | + +- Always check for `[AGENT_BOOSTER_AVAILABLE]` or `[TASK_MODEL_RECOMMENDATION]` before spawning agents +- Use Edit tool directly when `[AGENT_BOOSTER_AVAILABLE]` + +## Swarm Configuration & Anti-Drift + +- ALWAYS use hierarchical topology for coding swarms +- Keep maxAgents at 6-8 for tight coordination +- Use specialized strategy for clear role boundaries +- Use `raft` consensus for hive-mind (leader maintains authoritative state) +- Run frequent checkpoints via `post-task` hooks +- Keep shared memory namespace for all agents + +```bash +npx @claude-flow/cli@latest swarm init --topology hierarchical --max-agents 8 --strategy specialized +``` + +## Swarm Execution Rules + +- ALWAYS use `run_in_background: true` for all agent Task calls +- ALWAYS put ALL agent Task calls in ONE message for parallel execution +- After spawning, STOP — do NOT add more tool calls or check status +- Never poll TaskOutput or check swarm status — trust agents to return +- When agent results arrive, review ALL results before proceeding + +## V3 CLI Commands + +### Core Commands + +| Command | Subcommands | Description | +| --------- | ------------- | ------------- | +| `init` | 4 | Project initialization | +| `agent` | 8 | Agent lifecycle management | +| `swarm` | 6 | Multi-agent swarm coordination | +| `memory` | 11 | AgentDB memory with HNSW search | +| `task` | 6 | Task creation and lifecycle | +| `session` | 7 | Session state management | +| `hooks` | 17 | Self-learning hooks + 12 workers | +| `hive-mind` | 6 | Byzantine fault-tolerant consensus | + +### Quick CLI Examples + +```bash +npx @claude-flow/cli@latest init --wizard +npx @claude-flow/cli@latest agent spawn -t coder --name my-coder +npx @claude-flow/cli@latest swarm init --v3-mode +npx @claude-flow/cli@latest memory search --query "authentication patterns" +npx @claude-flow/cli@latest doctor --fix +``` + +## Available Agents (60+ Types) + +### Core Development + +`coder`, `reviewer`, `tester`, `planner`, `researcher` + +### Specialized + +`security-architect`, `security-auditor`, `memory-specialist`, `performance-engineer` + +### Swarm Coordination + +`hierarchical-coordinator`, `mesh-coordinator`, `adaptive-coordinator` + +### GitHub & Repository + +`pr-manager`, `code-review-swarm`, `issue-tracker`, `release-manager` + +### SPARC Methodology + +`sparc-coord`, `sparc-coder`, `specification`, `pseudocode`, `architecture` + +## Memory Commands Reference + +```bash +# Store (REQUIRED: --key, --value; OPTIONAL: --namespace, --ttl, --tags) +npx @claude-flow/cli@latest memory store --key "pattern-auth" --value "JWT with refresh" --namespace patterns + +# Search (REQUIRED: --query; OPTIONAL: --namespace, --limit, --threshold) +npx @claude-flow/cli@latest memory search --query "authentication patterns" + +# List (OPTIONAL: --namespace, --limit) +npx @claude-flow/cli@latest memory list --namespace patterns --limit 10 + +# Retrieve (REQUIRED: --key; OPTIONAL: --namespace) +npx @claude-flow/cli@latest memory retrieve --key "pattern-auth" --namespace patterns +``` + +## Quick Setup + +```bash +claude mcp add claude-flow -- npx -y @claude-flow/cli@latest +npx @claude-flow/cli@latest daemon start +npx @claude-flow/cli@latest doctor --fix +``` + +## Claude Code vs CLI Tools + +- Claude Code's Task tool handles ALL execution: agents, file ops, code generation, git +- CLI tools handle coordination via Bash: swarm init, memory, hooks, routing +- NEVER use CLI tools as a substitute for Task tool agents + +## Multi-Agent Communication + +When multiple Claude instances or agents work on this project concurrently, they coordinate through a shared filesystem bus at `~/.claude/agent-comms/`. **Read `~/.claude/multi-agent-comms.md`** for the full protocol. + +Key rules: + +- Post a `sync` message at session start, after completing work, and before ending +- Post an `intent` message **before committing** and wait ~5s for conflicts +- `claim` the test runner lock before running the full test suite +- Post a `result` after commits and test runs so other agents stay informed +- Check for recent messages when resuming work to avoid conflicts +- Lock `git-commit` and `git-push` resources for the duration of those operations + +Directory structure: `~/.claude/agent-comms/{messages,locks,status}` + +## Knowledge graph (graphify) + +**This project uses the graphify knowledge graph at `graphify-out/` — always consult it for architecture/codebase questions, and (re)build it when missing or stale.** + +Rules, in priority order: + +1. **Before answering architecture or codebase questions**: read `graphify-out/GRAPH_REPORT.md` for god nodes + community structure, and `graphify-out/wiki/index.md` for a navigable summary. The graph often surfaces helpers/utilities that grep misses because names don't overlap. +2. **If `graphify-out/` is missing**: build it FIRST before doing any non-trivial exploration. Command (from the repo root): + + ```bash + "${GRAPHIFY_PYTHON:-python3}" \ + -c "from graphify.watch import _rebuild_code; from pathlib import Path; _rebuild_code(Path('.'))" + ``` + + Set `GRAPHIFY_PYTHON` to your graphify venv's Python interpreter (`/.venv/bin/python3`); falls back to `python3` if the package is installed system-wide. Runs for ~1–3 minutes on this repo; run it in the background (`run_in_background: true`) so you can start reading other things while it finishes. +3. **After modifying code files in a session**: re-run the same command to keep the graph current. The installed `PreToolUse` hook in `.claude/settings.json` handles this automatically for Write/Edit/MultiEdit, but the hook has a 5-second timeout — on a large edit batch the rebuild may be skipped; run the command manually in that case. +4. **Never edit code you haven't mapped** when the graph is available. Prefer graph-assisted navigation over raw grep for cross-cutting questions ("who calls X", "where is Y implemented", "what would break if I rename Z"). + +If `graphify` is not on PATH, locate your local install (typically `~/bin/graphify` or your venv's `bin/`). The `graphify claude install` subcommand re-registers the PreToolUse hook if it's been removed. + +## Support + +- Documentation: +- Issues: diff --git a/CONTRIBUTING.md b/CONTRIBUTING.md new file mode 100644 index 000000000..491072bbd --- /dev/null +++ b/CONTRIBUTING.md @@ -0,0 +1,429 @@ +# Contributing to CUDly + +Thank you for your interest in contributing to CUDly! This document provides guidelines and instructions for contributing. + +## Code of Conduct + +By participating in this project, you agree to maintain a respectful and inclusive environment. Be kind, constructive, and professional in all interactions. + +## How to Contribute + +### Reporting Bugs + +1. **Search existing issues** - Check if the bug has already been reported +2. **Create a detailed report** including: + - CUDly version (`./cudly --version`) + - Go version (`go version`) + - Operating system and architecture + - Cloud provider and service affected + - Steps to reproduce + - Expected vs actual behavior + - Relevant logs (with sensitive data removed) + +### Suggesting Features + +1. **Search existing issues** - Your idea may already be proposed +2. **Open a feature request** with: + - Clear description of the feature + - Use case and benefits + - Proposed implementation (if applicable) + - Any potential drawbacks + +### Submitting Code + +1. **Fork the repository** +2. **Create a feature branch** from `main`: + + ```bash + git checkout -b feature/your-feature-name + ``` + +3. **Make your changes** following our coding standards +4. **Write or update tests** for your changes +5. **Run the test suite** to ensure everything passes +6. **Commit with clear messages** following our commit conventions +7. **Push to your fork** and submit a Pull Request + +## Development Setup + +### Prerequisites + +- Go 1.26.6 or later (the floor set by the `go` directive in `go.mod`) +- AWS/Azure/GCP credentials for integration testing +- Git + +### Getting Started + +```bash +# Clone your fork +git clone https://github.com/YOUR_USERNAME/CUDly.git +cd CUDly + +# Add upstream remote +git remote add upstream https://github.com/LeanerCloud/CUDly.git + +# Install dependencies +go mod download + +# Build the project +go build -o cudly cmd/*.go + +# Run tests +go test ./... +``` + +### Go workspace and worktrees (gopls setup) + +The repo ships a `go.work` that lists every module in this repository (the +root module, `pkg`, the three provider modules, and `tests/e2e`). This is +enough for standard clones. When you are working across multiple git worktrees +simultaneously, gopls needs each worktree's module added to the workspace or it +flags every file in the sibling trees with `BrokenImport` / `undefined: `. + +**Do not edit the committed `go.work`** for local paths -- they vary per +developer and per session. + +Instead, create a `go.work.local` next to `go.work` (it is gitignored): start +from a copy of the committed `go.work` and append your active worktrees: + +```go +// go.work.local -- gitignored, developer-local +// Keep this `go` line at or above the modules' own directive, otherwise the +// workspace is rejected. Copy it from the committed go.work. +go 1.26.6 + +use ( + . + ./pkg + ./providers/aws + ./providers/azure + ./providers/gcp + ./tests/e2e + ../.worktrees/CUDly/fix-516 + ../.worktrees/CUDly/feat-something +) +``` + +Then point gopls at it by setting `GOWORK` before launching your editor, or by +symlinking it over `go.work` temporarily: + +```bash +# Option A: set GOWORK in your shell profile or editor launcher +export GOWORK="$PWD/go.work.local" + +# Option B: create go.work.local and let gopls auto-discover it +# (gopls respects GOWORK when set; otherwise it walks up for go.work) +``` + +After adding or removing a worktree, update `go.work.local` to match: + +```bash +# Quick regeneration from git worktree list (space-safe: keeps full paths, +# adds one -use entry per worktree; skips the main checkout on line 1) +git worktree list --porcelain | sed -n 's/^worktree //p' | tail -n +2 | + while IFS= read -r wt; do go work edit -use "$wt"; done +``` + +The committed `go.work` (listing only this repository's own modules) keeps +`go build ./...` and CI clean for everyone without requiring any local setup. + +### Running Tests + +```bash +# Run all tests +go test ./... + +# Run tests with coverage +go test -cover ./... + +# Run tests for a specific package +go test ./providers/aws/... + +# Run tests with verbose output +go test -v ./... + +# Run a specific test +go test -run TestFunctionName ./path/to/package +``` + +### Test Coverage Goals + +We aim to maintain the following minimum test coverage: + +| Package | Minimum Coverage | +|---------|-----------------| +| Service clients | 80% | +| Provider implementations | 70% | +| Common/shared packages | 80% | +| CLI/cmd | 60% | + +## Coding Standards + +### Go Style + +- Follow the [Effective Go](https://golang.org/doc/effective_go) guidelines +- Use `gofmt` to format code +- Use `golint` and `go vet` to catch issues +- Keep functions focused and reasonably sized +- Write clear, self-documenting code + +### Naming Conventions + +- Use CamelCase for exported names, camelCase for unexported +- Use meaningful, descriptive names +- Interfaces describing behavior should end in `-er` (e.g., `Reader`, `Writer`) +- Test files: `*_test.go` +- Mock implementations: prefix with `mock` + +### Documentation + +- All exported functions, types, and packages must have doc comments +- Use complete sentences starting with the name being documented +- Include usage examples for complex functionality +- Keep comments up to date with code changes + +### Error Handling + +- Always handle errors explicitly +- Wrap errors with context using `fmt.Errorf("context: %w", err)` +- Use custom error types for domain-specific errors +- Never ignore errors silently + +### Testing + +- Write table-driven tests where appropriate +- Use interfaces and dependency injection for testability +- Mock external dependencies (AWS/Azure/GCP SDKs) +- Test both success and error paths +- Include edge cases in test coverage + +## Project Structure + +```text +CUDly/ +├── cmd/ # CLI entry point +├── pkg/ # Shared packages +│ ├── common/ # Cloud-agnostic types +│ └── provider/ # Provider abstraction +├── providers/ # Cloud implementations +│ ├── aws/ # AWS provider +│ │ ├── services/ # Service clients +│ │ └── internal/ # Internal packages +│ ├── azure/ # Azure provider +│ └── gcp/ # GCP provider +└── internal/ # Private packages +``` + +### Adding a New Service + +1. Create the service client in `providers//services/` +2. Implement the `ServiceClient` interface from `pkg/provider` +3. Register the service in the provider's `GetServiceClient` method +4. Add recommendations support if applicable +5. Write comprehensive tests +6. Update documentation + +### Adding a New Cloud Provider + +1. Create a new directory under `providers/` +2. Implement the `Provider` interface from `pkg/provider` +3. Implement required service clients +4. Register the provider using `provider.RegisterProvider()` in `init()` +5. Add authentication documentation +6. Write comprehensive tests +7. Update README with new provider information + +## Commit Guidelines + +### Commit Message Format + +```text +type(scope): brief description + +Longer description if needed. Explain what and why, +not how (the code shows how). + +Fixes #123 +``` + +### Types + +- `feat`: New feature +- `fix`: Bug fix +- `docs`: Documentation only +- `style`: Formatting, missing semicolons, etc. +- `refactor`: Code change that neither fixes a bug nor adds a feature +- `perf`: Performance improvement +- `test`: Adding or updating tests +- `chore`: Build process, dependencies, etc. + +### Examples + +```text +feat(aws): add MemoryDB reserved node support + +Implements purchase and recommendation fetching for +Amazon MemoryDB reserved nodes. + +Fixes #42 +``` + +```text +fix(azure): handle subscription pagination correctly + +The previous implementation missed subscriptions after +the first page. Now properly iterates all pages. +``` + +## Pull Request Process + +1. **Update documentation** for any user-facing changes +2. **Add or update tests** for your changes +3. **Ensure all tests pass** before submitting +4. **Fill out the PR template** completely +5. **Request review** from maintainers +6. **Address feedback** promptly and constructively + +### PR Checklist + +- [ ] Code follows project style guidelines +- [ ] Tests added/updated and passing +- [ ] Documentation updated +- [ ] Commit messages follow conventions +- [ ] No sensitive data in code or commits +- [ ] Changes are backwards compatible (or breaking changes documented) + +## Known Issues Sweep + +The `known_issues/` directory tracks open tech debt, deferred fixes, and +surfaced bugs that are out of scope for the current PR. To stay useful, it +needs periodic housekeeping. + +### Entry format + +Each file must begin with a `# Known Issues: ` heading followed by an +audit-status line: + +```text +> **Audit status ():** ` needs triage · resolved` +``` + +Subsequent sections use `## SEVERITY: Short title` (e.g. `## MEDIUM: ...`) and +include at minimum: **Files** (affected paths), **Description**, **Why +deferred**, and **Status**. + +### When to ADD an entry + +- Tech debt or a follow-up bug is discovered during a PR review but is + explicitly out of scope for that PR. +- A test is marked flaky and a root-cause fix is deferred. +- A deliberate deferral is made (e.g. "fix after the current refactor lands"). + +Create a new file `known_issues/_.md` (sequential number, lowercase +slug) and open a corresponding GitHub issue so it can be tracked and closed. + +### When to REMOVE (archive) an entry + +An entry is stale when its corresponding GitHub issue is **closed** OR when the +entry has had **no recurrence for more than 6 months** and no open issue +references it. Do not delete stale files; move them to `known_issues/resolved/` +so the rationale is preserved for future readers. + +### Who runs the sweep and when + +Any contributor working in a file covered by a `known_issues/` doc should +check whether that doc's referenced issue is still open; archive it if not. + +A dedicated sweep over the whole directory should happen: + +- At the start of each sprint (or monthly if sprints are not used). +- After any PR that explicitly closes multiple issues. +- When a new contributor is onboarding and doing a codebase walkthrough. + +To perform a sweep: + +```bash +# List docs referencing a specific issue number +grep -rl "#" known_issues/ + +# Cross-check all referenced issues in bulk +grep -h "closes #\|Fixes #\|#[0-9]\+" known_issues/*.md \ + | grep -oE '#[0-9]+' | sort -u \ + | xargs -I{} gh issue view {} --json state,number,title --jq '[.number,.state,.title]' +``` + +Move resolved docs to `known_issues/resolved/`: + +```bash +git mv known_issues/.md known_issues/resolved/ +``` + +Include the archive in the same PR that closes the underlying issue, or in a +dedicated `chore(docs): archive resolved known_issues` commit. + +## Security + +### Reporting Vulnerabilities + +**Do not report security vulnerabilities through public issues.** + +Instead, please email security concerns to the maintainers directly. Include: + +- Description of the vulnerability +- Steps to reproduce +- Potential impact +- Suggested fix (if any) + +### Security Best Practices + +- Never commit credentials or secrets +- Use environment variables for sensitive configuration +- Validate all external input +- Follow least-privilege principles +- Keep dependencies updated + +## Areas for Contribution + +We welcome contributions in these areas: + +### High Priority + +- Additional AWS services (Lambda, DynamoDB, etc.) +- Azure service implementations +- GCP service implementations +- Improved error messages and user experience + +### Medium Priority + +- Enhanced reporting and analytics +- Terraform/CloudFormation integration +- Web UI dashboard +- Performance optimizations + +### Documentation + +- Usage tutorials and guides +- Architecture documentation +- API documentation +- Translation to other languages + +## Getting Help + +- **Issues**: Open a GitHub issue for bugs or features +- **Discussions**: Use GitHub Discussions for questions +- **Documentation**: Check the README and code comments + +## License + +By contributing to CUDly, you agree that your contributions will be licensed under the Open Software License 3.0 (OSL-3.0). + +This means: + +- Your contributions can be used commercially +- Derivative works must also be OSL-3.0 licensed +- You grant a patent license for your contributions +- Attribution must be maintained + +## Acknowledgments + +Thank you to all contributors who help make CUDly better! Your time and expertise are greatly appreciated. diff --git a/Dockerfile b/Dockerfile new file mode 100644 index 000000000..990e64bc5 --- /dev/null +++ b/Dockerfile @@ -0,0 +1,218 @@ +# ============================================== +# Multi-stage build for cloud-agnostic deployment +# Works on: AWS Lambda, AWS Fargate, GCP Cloud Run, Azure Container Apps +# Supports: ARM64 (default) and AMD64 architectures +# ============================================== + +# Build arguments for multi-architecture support +# TARGETARCH and TARGETOS are set automatically by docker buildx +ARG TARGETARCH +ARG TARGETOS=linux + +# Build stage +# Image pinned to a SHA256 digest for reproducible builds — a registry +# tag mutation (Docker Hub allows re-tagging) cannot poison this build. +# To refresh: `docker buildx imagetools inspect golang:1.26.6-alpine3.24` +# (or use the Docker Hub API tags endpoint) and update the digest below. +# A Renovate / Dependabot config can automate this if desired. +FROM --platform=$BUILDPLATFORM golang:1.26.6-alpine3.24@sha256:3889b425f035be855a72fb4755265311293b6d414521f0a519d819df32222d83 AS builder + +# Re-declare args for use in this stage +ARG TARGETARCH +ARG TARGETOS + +# Build metadata stamped into the binary via ldflags and surfaced by the +# public GET /version endpoint. GIT_COMMIT and BUILD_DATE are supplied by the +# terraform build module (modules/build); they default to "unknown" so a bare +# `docker build .` still succeeds without git context. +ARG VERSION=dev +ARG GIT_COMMIT=unknown +ARG BUILD_DATE + +# Install build dependencies +RUN apk add --no-cache \ + git \ + ca-certificates \ + postgresql-client + +# Set shell with pipefail for safer pipe operations +SHELL ["/bin/ash", "-eo", "pipefail", "-c"] + +WORKDIR /app + +# Copy go module files +COPY go.mod go.sum ./ + +# Copy provider modules (multi-module setup) +COPY pkg/go.mod pkg/go.sum ./pkg/ +COPY providers/aws/go.mod providers/aws/go.sum providers/aws/ +COPY providers/azure/go.mod providers/azure/go.sum providers/azure/ +COPY providers/gcp/go.mod providers/gcp/go.sum providers/gcp/ + +# Download dependencies +RUN go mod download + +# Build golang-migrate from source on this stage's pinned Go toolchain, the same +# way `make install-tools` does. Upstream's prebuilt release tarballs carry +# whatever toolchain upstream built them with (v4.19.1 ships go1.25.4), which is +# how issue #1833's stdlib CVEs reached the runtime image, where entrypoint.sh +# runs `migrate up` against the database on every container start. +# +# Built as a package of THIS module (no @version suffix) rather than +# `go install ...@v4.19.1`, and therefore placed after `go mod download`. +# The @version form resolves dependencies from golang-migrate's own go.mod, +# which pins jackc/pgx v5.5.4, x/crypto v0.45.0 and x/text v0.31.0 - all +# superseded, and together 17 advisories with published fixes that +# scripts/scan-shipped-image.sh correctly fails on. Building from the main +# module instead resolves them at this repo's own versions, which are already +# above every one of those fixes. The migrate version therefore comes from +# go.mod, and is read back out of the build list rather than restated. +# +# -tags=pgx5, not postgres: the postgres tag links the lib/pq driver, which +# carries advisories with no fixed version in any release (GO-2026-6166, 6168, +# 6170, 6171, 6172) reached through Driver.Open and conn.Exec on every +# container start. The pgx5 tag builds the same driver on jackc/pgx v5, which +# the application already uses. It registers the "pgx5" URL scheme only, so +# scripts/entrypoint.sh and .github/workflows/database-migration.yml must build +# pgx5:// URLs - a postgres:// URL against this binary fails at runtime with +# "unknown driver". See issue #1849. +RUN MIGRATE_VERSION="$(go list -m -f '{{.Version}}' github.com/golang-migrate/migrate/v4)" && \ + echo "Building golang-migrate ${MIGRATE_VERSION} for ${TARGETOS}/${TARGETARCH}" && \ + CGO_ENABLED=0 GOOS=${TARGETOS} GOARCH=${TARGETARCH} go build \ + -tags=pgx5 \ + -ldflags="-s -w -X main.Version=${MIGRATE_VERSION}" \ + -o /usr/local/bin/migrate \ + github.com/golang-migrate/migrate/v4/cmd/migrate + +# Copy source code +COPY . . + +# Build unified server binary (cloud-agnostic) +# Supports both ARM64 and AMD64 via build args +# Default: ARM64 for cost optimization (20% savings on AWS Fargate) +RUN echo "Building for ${TARGETOS}/${TARGETARCH}" && \ + BUILD_TIME="${BUILD_DATE:-$(date -u +%Y-%m-%dT%H:%M:%SZ)}" && \ + CGO_ENABLED=0 GOOS=${TARGETOS} GOARCH=${TARGETARCH} go build \ + -p 6 \ + -ldflags="-s -w -X main.Version=${VERSION:-dev} -X main.BuildTime=${BUILD_TIME} -X main.GitSHA=${GIT_COMMIT:-unknown}" \ + -o /app/cudly \ + ./cmd/server + +# Binary built successfully + +# ============================================== +# Frontend build stage +# ============================================== +# Image pinned to a SHA256 digest for reproducible builds. Refresh via +# the Docker Hub API tags endpoint and update the digest below when the +# `node:24-alpine` tag is bumped. +FROM --platform=$BUILDPLATFORM node:24-alpine@sha256:d1b3b4da11eefd5941e7f0b9cf17783fc99d9c6fc34884a665f40a06dbdfc94f AS frontend-builder + +WORKDIR /frontend +COPY frontend/package*.json ./ +# `--no-progress`: disables the npm progress reporter, whose worker has a +# long-standing race condition in npm 10/11 that surfaces as +# `npm error Exit handler never called!` on memory-constrained hosts +# (Linux VMs, low-RAM CI runners). See npm/cli issues for the bug; the +# fix is "don't run that worker". +# `--maxsockets 1`: serialise registry fetches so peak memory during the +# install stays low. With 810 lockfile entries, parallel fetch + the +# gzip-decode workers race the OOM killer on hosts with <2 GB free. +# `--no-audit --no-fund`: skip post-install network calls that aren't +# relevant to a build context. +# `test -x`: existing guard against silent zero-exit npm failures +# (kept from #044dc583c — addresses a different failure mode where +# npm exits 0 but leaves node_modules empty). +RUN npm ci --no-progress --maxsockets 1 --no-audit --no-fund && \ + test -x node_modules/.bin/webpack +COPY frontend/ ./ +RUN npm run build + +# ============================================== +# Runtime stage - multi-arch base image +# ============================================== +# Image pinned to a SHA256 digest for reproducible builds. +# To refresh: `docker buildx imagetools inspect alpine:3.24.1` and update the +# digest below. This is the multi-arch index digest, not a per-platform one, so +# it stays correct for every TARGETARCH this image is built for. +FROM alpine:3.24.1@sha256:28bd5fe8b56d1bd048e5babf5b10710ebe0bae67db86916198a6eec434943f8b + +# Re-declare args for use in this stage +ARG TARGETARCH +ARG TARGETOS + +# Install runtime dependencies +RUN apk add --no-cache \ + ca-certificates \ + postgresql-client \ + curl \ + tzdata + +# Create non-root user for security +RUN addgroup -g 1000 cudly && \ + adduser -D -u 1000 -G cudly cudly + +# Create app directory +WORKDIR /app + +# Copy binary, migrations, and frontend from build stages +COPY --from=builder --chown=cudly:cudly /app/cudly /app/cudly +COPY --from=builder --chown=cudly:cudly /usr/local/bin/migrate /usr/local/bin/migrate +COPY --chown=cudly:cudly internal/database/postgres/migrations /app/migrations +COPY --from=frontend-builder --chown=cudly:cudly /frontend/dist /app/static + +# Copy unified entrypoint script and set permissions +COPY --chown=cudly:cudly scripts/entrypoint.sh /entrypoint.sh +RUN chmod +x /entrypoint.sh + +# Switch to non-root user. Numeric form so the identity is resolvable without +# the image's /etc/passwd: Kubernetes `runAsNonRoot` and similar admission +# checks cannot verify a name-form USER and will refuse to start the pod. +# 1000:1000 is exactly the uid:gid created above, so this is a rename only. +USER 1000:1000 + +# Environment defaults +ENV DB_MIGRATIONS_PATH=/app/migrations \ + DB_AUTO_MIGRATE=true \ + RUNTIME_MODE=auto \ + PORT=8080 \ + STATIC_DIR=/app/static \ + GOARCH=${TARGETARCH} \ + GOOS=${TARGETOS} + +# Expose HTTP port (used by Fargate, Cloud Run, Container Apps) +# Lambda ignores this +EXPOSE 8080 + +# Health check (works for HTTP mode, ignored in Lambda mode) +# +# JSON (exec) form, but invoking /bin/sh explicitly: the `||` is a shell +# operator, so a bare exec-form list would hand `||` and `exit` to curl as +# literal arguments and the healthcheck would never report unhealthy correctly. +# This is the same process tree the shell form produced, written so the +# dependency on a shell is declared rather than implied. +HEALTHCHECK --interval=30s --timeout=3s --start-period=10s --retries=3 \ + CMD ["/bin/sh", "-c", "curl -f http://localhost:8080/health || exit 1"] + +# Unified entrypoint handles both Lambda and HTTP modes +ENTRYPOINT ["/entrypoint.sh"] +CMD ["/app/cudly"] + +# ============================================== +# Build Instructions: +# ============================================== +# +# Build for ARM64 (AWS Lambda/Fargate with Graviton): +# docker buildx build --platform linux/arm64 -t cudly:arm64 . +# +# Build for AMD64 (GCP Cloud Run, Azure Container Apps): +# docker buildx build --platform linux/amd64 -t cudly:amd64 . +# +# CI/CD builds (GitHub Actions): +# AWS Lambda/Fargate: --platform linux/arm64 (Graviton2, 20% cost savings) +# GCP Cloud Run: --platform linux/amd64 (ARM64 not supported) +# Azure Container Apps: --platform linux/amd64 (ARM64 not supported) +# +# Build and load for local testing: +# docker buildx build --platform linux/arm64 -t cudly:arm64 --load . +# ============================================== diff --git a/Dockerfile.dev b/Dockerfile.dev new file mode 100644 index 000000000..5537bd61a --- /dev/null +++ b/Dockerfile.dev @@ -0,0 +1,70 @@ +# Development Dockerfile with hot reload using Air +# Image pinned to a SHA256 digest for reproducible builds; a registry +# tag mutation (Docker Hub allows re-tagging) cannot poison this build. +# To refresh: `docker buildx imagetools inspect golang:1.26.6-alpine3.24` +# (or use the Docker Hub API tags endpoint) and update the digest below. +# A Renovate / Dependabot config can automate this if desired. +# Keep this digest in sync with the builder stage in Dockerfile. +FROM golang:1.26.6-alpine3.24@sha256:3889b425f035be855a72fb4755265311293b6d414521f0a519d819df32222d83 AS development + +# Install development tools and Air for hot reload +RUN apk add --no-cache \ + git \ + postgresql-client \ + curl \ + build-base && \ + go install github.com/air-verse/air@v1.61.7 + +# Set shell with pipefail for safer pipe operations +SHELL ["/bin/ash", "-eo", "pipefail", "-c"] + +# Install golang-migrate for database migrations (architecture-aware, checksum-verified) +ARG TARGETARCH +RUN MIGRATE_ARCH=$([ "$TARGETARCH" = "arm64" ] && echo "arm64" || echo "amd64") && \ + if [ "$MIGRATE_ARCH" = "arm64" ]; then \ + MIGRATE_SHA256="9c95441cc430ffdac89276d14de5e2f18bfafca00796c2895490d62e3776d104"; \ + else \ + MIGRATE_SHA256="26c53c9162c9c4aaa84c47cd12455d4a9ac725befbe82850a5937b5ec1e7b8e6"; \ + fi && \ + curl -Lo migrate.tar.gz "https://github.com/golang-migrate/migrate/releases/download/v4.17.0/migrate.linux-${MIGRATE_ARCH}.tar.gz" && \ + echo "${MIGRATE_SHA256} migrate.tar.gz" | sha256sum -c - && \ + tar xvzf migrate.tar.gz && \ + mv migrate /usr/local/bin/migrate && \ + chmod +x /usr/local/bin/migrate && \ + rm migrate.tar.gz + +WORKDIR /app + +# Copy go mod files +COPY go.mod go.sum ./ + +# Copy provider and pkg go.mod files (multi-module setup) +COPY pkg/go.mod pkg/go.sum ./pkg/ +COPY providers/aws/go.mod providers/aws/go.sum providers/aws/ +COPY providers/azure/go.mod providers/azure/go.sum providers/azure/ +COPY providers/gcp/go.mod providers/gcp/go.sum providers/gcp/ + +# Download dependencies +RUN go mod download + +# Copy source code +COPY . . + +# Create Air configuration if it doesn't exist +RUN if [ ! -f .air.toml ]; then air init; fi + +EXPOSE 8080 + +# Create non-root user for security; grant ownership of app dir and Go paths. +# uid/gid are pinned explicitly rather than left to `adduser -S`, which picks +# the first free system id and so varies with the base image's existing users. +# A numeric USER below needs a value that is known here, not discovered later. +RUN addgroup -g 1001 devuser && \ + adduser -D -u 1001 -G devuser devuser && \ + chown -R devuser:devuser /app /root/go /root/.cache 2>/dev/null || true + +# Numeric form so the identity is resolvable without the image's /etc/passwd. +USER 1001:1001 + +# Start with Air for hot reload +CMD ["air", "-c", ".air.toml"] diff --git a/Dockerfile.test b/Dockerfile.test new file mode 100644 index 000000000..d7560d01f --- /dev/null +++ b/Dockerfile.test @@ -0,0 +1,31 @@ +# Test-runner image for the docker-compose E2E suite (docker-compose.test.yml, +# profile "test"). It only needs the Go toolchain and the stdlib-only module +# under tests/e2e, so the build is fast and independent of the app's +# dependency tree. +# +# Base image pinned by digest for the same supply-chain reasons as Dockerfile; +# keep the digest in sync with the builder stage there. +FROM golang:1.26.6-alpine3.24@sha256:3889b425f035be855a72fb4755265311293b6d414521f0a519d819df32222d83 + +# Run the suite as a non-root user (trivy DS-0002). tests/e2e is a stdlib-only +# module so no module downloads are needed; GOCACHE lives in the user's home, +# which it owns and can write to during both the pre-compile below and the +# runtime CMD. +RUN adduser -D -u 10001 e2e +ENV GOCACHE=/home/e2e/.cache/go-build + +WORKDIR /e2e + +COPY --chown=e2e:e2e tests/e2e/ ./ + +# Numeric form so the identity is resolvable without the image's /etc/passwd. +# 10001 is exactly the uid created above, so this is a rename only. +USER 10001 + +# Pre-compile the test binary so `docker compose up` failures are runtime +# failures, not compile errors discovered after the stack is already up. +RUN go vet -tags=e2e ./... && go test -tags=e2e -run NONE ./... + +# docker-compose.test.yml overrides the command; this default keeps the image +# usable standalone. +CMD ["go", "test", "-v", "-tags=e2e", "./..."] diff --git a/LICENSE b/LICENSE new file mode 100644 index 000000000..17f7d0629 --- /dev/null +++ b/LICENSE @@ -0,0 +1,172 @@ +Open Software License ("OSL") v. 3.0 + +This Open Software License (the "License") applies to any original work of +authorship (the "Original Work") whose owner (the "Licensor") has placed the +following licensing notice adjacent to the copyright notice for the Original +Work: + +Licensed under the Open Software License version 3.0 + +1) Grant of Copyright License. Licensor grants You a worldwide, royalty-free, +non-exclusive, sublicensable license, for the duration of the copyright, to do +the following: + + a) to reproduce the Original Work in copies, either alone or as part of a + collective work; + + b) to translate, adapt, alter, transform, modify, or arrange the Original + Work, thereby creating derivative works ("Derivative Works") based upon the + Original Work; + + c) to distribute or communicate copies of the Original Work and Derivative + Works to the public, with the proviso that copies of Original Work or + Derivative Works that You distribute or communicate shall be licensed under + this Open Software License; + + d) to perform the Original Work publicly; and + + e) to display the Original Work publicly. + +2) Grant of Patent License. Licensor grants You a worldwide, royalty-free, +non-exclusive, sublicensable license, under patent claims owned or controlled +by the Licensor that are embodied in the Original Work as furnished by the +Licensor, for the duration of the patents, to make, use, sell, offer for sale, +have made, and import the Original Work and Derivative Works. + +3) Grant of Source Code License. The term "Source Code" means the preferred +form of the Original Work for making modifications to it and all available +documentation describing how to modify the Original Work. Licensor agrees to +provide a machine-readable copy of the Source Code of the Original Work along +with each copy of the Original Work that Licensor distributes. Licensor +reserves the right to satisfy this obligation by placing a machine-readable +copy of the Source Code in an information repository reasonably calculated to +permit inexpensive and convenient access by You for as long as Licensor +continues to distribute the Original Work. + +4) Exclusions From License Grant. Neither the names of Licensor, nor the names +of any contributors to the Original Work, nor any of their trademarks or +service marks, may be used to endorse or promote products derived from this +Original Work without express prior permission of the Licensor. Except as +expressly stated herein, nothing in this License grants any license to +Licensor's trademarks, copyrights, patents, trade secrets or any other +intellectual property. No patent license is granted to make, use, sell, offer +for sale, have made, or import embodiments of any patent claims other than the +licensed claims defined in Section 2. No license is granted to the trademarks +of Licensor even if such marks are included in the Original Work. Nothing in +this License shall be interpreted to prohibit Licensor from licensing under +terms different from this License any Original Work that Licensor otherwise +would have a right to license. + +5) External Deployment. The term "External Deployment" means the use, +distribution, or communication of the Original Work or Derivative Works in any +way such that the Original Work or Derivative Works may be used by anyone +other than You, whether those works are distributed or communicated to those +persons or made available as an application intended for use over a network. +As an express condition for the grants of license hereunder, You must treat +any External Deployment by You of the Original Work or a Derivative Work as a +distribution under section 1(c). + +6) Attribution Rights. You must retain, in the Source Code of any Derivative +Works that You create, all copyright, patent, or trademark notices from the +Source Code of the Original Work, as well as any notices of licensing and any +descriptive text identified therein as an "Attribution Notice." You must cause +the Source Code for any Derivative Works that You create to carry a prominent +Attribution Notice reasonably calculated to inform recipients that You have +modified the Original Work. + +7) Warranty of Provenance and Disclaimer of Warranty. Licensor warrants that +the copyright in and to the Original Work and the patent rights granted herein +by Licensor are owned by the Licensor or are sublicensed to You under the +terms of this License with the permission of the contributor(s) of those +copyrights and patent rights. Except as expressly stated in the immediately +preceding sentence, the Original Work is provided under this License on an "AS +IS" BASIS and WITHOUT WARRANTY, either express or implied, including, without +limitation, the warranties of non-infringement, merchantability or fitness for +a particular purpose. THE ENTIRE RISK AS TO THE QUALITY OF THE ORIGINAL WORK +IS WITH YOU. This DISCLAIMER OF WARRANTY constitutes an essential part of this +License. No license to the Original Work is granted by this License except +under this disclaimer. + +8) Limitation of Liability. Under no circumstances and under no legal theory, +whether in tort (including negligence), contract, or otherwise, shall the +Licensor be liable to anyone for any indirect, special, incidental, or +consequential damages of any character arising as a result of this License or +the use of the Original Work including, without limitation, damages for loss +of goodwill, work stoppage, computer failure or malfunction, or any and all +other commercial damages or losses. This limitation of liability shall not +apply to the extent applicable law prohibits such limitation. + +9) Acceptance and Termination. If, at any time, You expressly assented to this +License, that assent indicates your clear and irrevocable acceptance of this +License and all of its terms and conditions. If You distribute or communicate +copies of the Original Work or a Derivative Work, You must make a reasonable +effort under the circumstances to obtain the express assent of recipients to +the terms of this License. This License conditions your rights to undertake +the activities listed in Section 1, including your right to create Derivative +Works based upon the Original Work, and doing so without honoring these terms +and conditions is prohibited by copyright law and international treaty. +Nothing in this License is intended to affect copyright exceptions and +limitations (including "fair use" or "fair dealing"). This License shall +terminate immediately and You may no longer exercise any of the rights granted +to You by this License upon your failure to honor the conditions in Section +1(c). + +10) Termination for Patent Action. This License shall terminate automatically +and You may no longer exercise any of the rights granted to You by this +License as of the date You commence an action, including a cross-claim or +counterclaim, against Licensor or any licensee alleging that the Original Work +infringes a patent. This termination provision shall not apply for an action +alleging patent infringement by combinations of the Original Work with other +software or hardware. + +11) Jurisdiction, Venue and Governing Law. Any action or suit relating to this +License may be brought only in the courts of a jurisdiction wherein the +Licensor resides or in which Licensor conducts its primary business, and under +the laws of that jurisdiction excluding its conflict-of-law provisions. The +application of the United Nations Convention on Contracts for the +International Sale of Goods is expressly excluded. Any use of the Original +Work outside the scope of this License or after its termination shall be +subject to the requirements and penalties of copyright or patent law in the +appropriate jurisdiction. This section shall survive the termination of this +License. + +12) Attorneys' Fees. In any action to enforce the terms of this License or +seeking damages relating thereto, the prevailing party shall be entitled to +recover its costs and expenses, including, without limitation, reasonable +attorneys' fees and costs incurred in connection with such action, including +any appeal of such action. This section shall survive the termination of this +License. + +13) Miscellaneous. If any provision of this License is held to be +unenforceable, such provision shall be reformed only to the extent necessary +to make it enforceable. + +14) Definition of "You" in This License. "You" throughout this License, +whether in upper or lower case, means an individual or a legal entity +exercising rights under, and complying with all of the terms of, this License. +For legal entities, "You" includes any entity that controls, is controlled by, +or is under common control with you. For purposes of this definition, +"control" means (i) the power, direct or indirect, to cause the direction or +management of such entity, whether by contract or otherwise, or (ii) ownership +of fifty percent (50%) or more of the outstanding shares, or (iii) beneficial +ownership of such entity. + +15) Right to Use. You may use the Original Work in all ways not otherwise +restricted or conditioned by this License or by law, and Licensor promises not +to interfere with or be responsible for such uses by You. + +16) Modification of This License. This License is Copyright (C) 2005 Lawrence +Rosen. Permission is granted to copy, distribute, or communicate this License +without modification. Nothing in this License permits You to modify this +License as applied to the Original Work or to Derivative Works. However, You +may modify the text of this License and copy, distribute or communicate your +modified version (the "Modified License") and apply it to other original works +of authorship subject to the following conditions: (i) You may not indicate in +any way that your Modified License is the "Open Software License" or "OSL" and +you may not use those names in the name of your Modified License; (ii) You +must replace the notice specified in the first paragraph above with the notice +"Licensed under " or with a notice of your own +that is not confusingly similar to the notice in this License; and (iii) You +may not claim that your original works are open source software unless your +Modified License has been approved by Open Source Initiative (OSI) and You +comply with its license review and certification process. diff --git a/Makefile b/Makefile new file mode 100644 index 000000000..fbc0d39c4 --- /dev/null +++ b/Makefile @@ -0,0 +1,268 @@ +.PHONY: build clean test deploy help all build-server build-lambda build-mcp test-unit test-integration \ + test-coverage full-test security-scan terraform-validate docker-build \ + fmt vet lint complexity complexity-report security-scan-go security-scan-docker \ + security-scan-terraform terraform-fmt terraform-fmt-check iac-arm docker-test pre-commit \ + setup-git-secrets security-scan-snyk security-scan-all ci docker-compose-test \ + install-dev-tools + +# Variables +VERSION?=dev +BUILD_TIME?=$(shell date -u '+%Y-%m-%dT%H:%M:%SZ') +GIT_SHA?=$(shell git rev-parse --short HEAD 2>/dev/null || echo unknown) + +# Dev tool versions - keep in sync with the CI pins in +# .github/workflows/ci.yml, pre-commit.yml and database-migration.yml +GOLANGCI_LINT_VERSION?=v2.10.1 +GOSEC_VERSION?=v2.28.0 +GOCYCLO_VERSION?=v0.6.0 +# golang-migrate deliberately has no version variable: it is installed as a +# package of this module (see install-tools), so its version and its whole +# dependency set come from go.mod. See issue #1849. +# staticcheck has no CI pin; it is used by scripts/security-scan.sh +STATICCHECK_VERSION?=v0.7.0 +LDFLAGS=-ldflags "-s -w -X main.Version=$(VERSION) -X main.BuildTime=$(BUILD_TIME) -X main.GitSHA=$(GIT_SHA)" + +# Default target +all: build + +help: ## Display available targets + @echo "Available targets:" + @echo " build - Build the CLI" + @echo " build-server - Build the unified server" + @echo " build-lambda - Build for AWS Lambda" + @echo " build-mcp - Build the MCP server (cmd/cudly-mcp)" + @echo " test - Run all unit tests" + @echo " test-unit - Run unit tests only" + @echo " test-integration - Run integration tests with testcontainers" + @echo " test-coverage - Run tests with coverage report" + @echo " clean - Remove build artifacts" + @echo " fmt - Format Go code" + @echo " lint - Run golangci-lint" + @echo " complexity - Check cyclomatic complexity" + @echo " complexity-report - Generate detailed complexity report" + @echo " security-scan - Run security scanners (gosec, trivy, tfsec)" + @echo " security-scan-all - Run all security scanners including Snyk" + @echo " setup-git-secrets - Set up git-secrets for preventing credential leaks" + @echo " terraform-validate - Validate Terraform configurations" + @echo " docker-build - Build Docker image" + @echo " docker-compose-test - Run E2E tests with docker-compose" + @echo " ci - Run CI pipeline locally" + +# Build the CLI +build: + go build -o cudly ./cmd + +# Build the unified server +build-server: + CGO_ENABLED=0 go build $(LDFLAGS) -o bin/cudly-server ./cmd/server + +# Build for Lambda (backward compatible) +build-lambda: + CGO_ENABLED=0 GOOS=linux GOARCH=arm64 go build -ldflags="-s -w" -o bootstrap ./cmd/lambda + +# Build the MCP server (see mcp/README.md). Uses the same $(LDFLAGS)/$(VERSION) +# as build-server so a tagged release reports its version in the MCP +# initialize response instead of "dev" (see cmd/cudly-mcp/main.go). +build-mcp: + mkdir -p bin + CGO_ENABLED=0 go build $(LDFLAGS) -o bin/cudly-mcp ./cmd/cudly-mcp + +# Run unit tests +test: test-unit + +test-unit: + @echo "Running unit tests..." + go test -v -race -short ./... + +# Run integration tests (requires testcontainers) +test-integration: + @echo "Running integration tests..." + go test -v -race -tags=integration ./... + +# Run tests with coverage +test-coverage: + @echo "Generating coverage report..." + go test -v -race -coverprofile=coverage.out -covermode=atomic ./... + go tool cover -html=coverage.out -o coverage.html + @echo "Coverage report: coverage.html" + @go tool cover -func=coverage.out | grep total + +# Run full test suite +full-test: test-unit test-integration test-coverage + +# Clean build artifacts +clean: + rm -f cudly bootstrap bin/cudly-server bin/cudly-mcp + rm -f coverage.out coverage.html + rm -f gosec-report.json trivy-report.json tfsec-report.json + go clean + +# Deploy (requires AWS credentials and terraform profiles) +deploy: + ./scripts/tf-deploy.sh aws dev + +# Format code +fmt: + go fmt ./... + terraform fmt -recursive terraform/ + +# Lint code +lint: + @echo "Running golangci-lint..." + @if command -v golangci-lint > /dev/null; then \ + golangci-lint run --timeout=5m; \ + else \ + echo "golangci-lint not installed. Install: make install-dev-tools"; \ + fi + +# Go vet +vet: + go vet ./... + +# Check cyclomatic complexity +complexity: + @echo "Checking cyclomatic complexity (threshold: 10)..." + @if command -v gocyclo > /dev/null; then \ + COMPLEXITY_ISSUES=$$(gocyclo -over 10 . 2>&1 || true); \ + if [ -n "$$COMPLEXITY_ISSUES" ]; then \ + echo "❌ Found functions with cyclomatic complexity over 10:"; \ + echo "$$COMPLEXITY_ISSUES"; \ + echo ""; \ + echo "⚠️ Please refactor these functions to reduce complexity."; \ + echo "📖 Tip: Extract helper functions, use early returns, or simplify logic."; \ + exit 1; \ + else \ + echo "✅ All functions have acceptable cyclomatic complexity (≤10)"; \ + fi \ + else \ + echo "gocyclo not installed. Install: make install-dev-tools"; \ + exit 1; \ + fi + +# Generate detailed complexity report +complexity-report: + @echo "Generating cyclomatic complexity report..." + @if command -v gocyclo > /dev/null; then \ + gocyclo -top 20 . | tee complexity-report.txt; \ + echo ""; \ + echo "📊 Top 20 most complex functions saved to: complexity-report.txt"; \ + else \ + echo "gocyclo not installed. Install: make install-dev-tools"; \ + fi + +# Security scanning +security-scan: security-scan-go security-scan-docker security-scan-terraform + +security-scan-go: + @echo "Running gosec..." + @if command -v gosec > /dev/null; then \ + gosec -fmt=json -out=gosec-report.json -exclude=G101,G104,G115,G204,G301,G304,G402,G505 ./...; \ + echo "✓ Go security scan complete: gosec-report.json"; \ + else \ + echo "gosec not installed. Install: make install-dev-tools"; \ + fi + +security-scan-docker: + @echo "Running trivy..." + @if command -v trivy > /dev/null; then \ + trivy fs --security-checks vuln,config . --format json --output trivy-report.json; \ + echo "✓ Container security scan complete: trivy-report.json"; \ + else \ + echo "trivy not installed. Install: https://aquasecurity.github.io/trivy/"; \ + fi + +security-scan-terraform: + @echo "Running tfsec..." + @if command -v tfsec > /dev/null; then \ + tfsec terraform/ --format json --out tfsec-report.json; \ + echo "✓ Terraform security scan complete: tfsec-report.json"; \ + else \ + echo "tfsec not installed. Install: https://aquasecurity.github.io/tfsec/"; \ + fi + +# Terraform validation +terraform-validate: + @echo "Validating Terraform configurations..." + @for dir in terraform/environments/*/dev; do \ + echo "Validating $$dir..."; \ + (cd $$dir && terraform init -backend=false && terraform validate) || exit 1; \ + done + @echo "✓ Terraform validation complete" + +terraform-fmt: + terraform fmt -recursive terraform/ + +terraform-fmt-check: + terraform fmt -check -recursive terraform/ + +# Regenerate the committed ARM JSON from the Bicep source. CI verifies sync via +# `make iac-arm && git diff --exit-code`. +iac-arm: + az bicep build \ + --file iac/federation/azure-target/bicep/azure-wif.bicep \ + --outfile iac/federation/azure-target/bicep/azure-wif.arm.json + +# Docker +docker-build: + @echo "Building Docker image..." + docker build -t cudly:$(VERSION) -t cudly:latest --build-arg VERSION=$(VERSION) . + @echo "✓ Docker image built: cudly:$(VERSION)" + +docker-test: docker-build + @echo "Testing Docker image..." + docker run --rm cudly:$(VERSION) /app/cudly --help || true + +# CI pipeline +ci: fmt vet complexity test-unit security-scan terraform-validate + @echo "✓ CI pipeline complete" + +# Pre-commit checks +pre-commit: fmt vet complexity test-unit + @echo "✓ Pre-commit checks complete" + +# Git secrets setup +setup-git-secrets: + @echo "Setting up git-secrets..." + @bash scripts/setup-git-secrets.sh + +# Snyk security scanning +security-scan-snyk: + @echo "Running Snyk security scan..." + @if command -v snyk > /dev/null; then \ + snyk test --severity-threshold=high; \ + echo "✓ Snyk scan complete"; \ + else \ + echo "snyk not installed. Install: npm install -g snyk"; \ + fi + +# Run all security scanners including Snyk +security-scan-all: security-scan security-scan-snyk + @echo "✓ All security scans complete" + +# Docker Compose E2E tests +docker-compose-test: + @echo "Running E2E tests with docker-compose..." + docker compose -f docker-compose.test.yml up --abort-on-container-exit --exit-code-from test-runner + docker compose -f docker-compose.test.yml down -v + +# Install development dependencies +install-dev-tools: + @echo "Installing development tools..." + @echo "Installing golangci-lint $(GOLANGCI_LINT_VERSION)..." + @go install github.com/golangci/golangci-lint/v2/cmd/golangci-lint@$(GOLANGCI_LINT_VERSION) + @echo "Installing gosec $(GOSEC_VERSION)..." + @go install github.com/securego/gosec/v2/cmd/gosec@$(GOSEC_VERSION) + @echo "Installing staticcheck $(STATICCHECK_VERSION)..." + @go install honnef.co/go/tools/cmd/staticcheck@$(STATICCHECK_VERSION) + @echo "Installing gocyclo $(GOCYCLO_VERSION)..." + @go install github.com/fzipp/gocyclo/cmd/gocyclo@$(GOCYCLO_VERSION) + @echo "Installing golang-migrate $$(go list -m -f '{{.Version}}' github.com/golang-migrate/migrate/v4)..." + @go install -tags 'pgx5' github.com/golang-migrate/migrate/v4/cmd/migrate + @echo "✓ Development tools installed" + @echo "" + @echo "Additional tools to install manually:" + @echo " - trivy: https://aquasecurity.github.io/trivy/" + @echo " - tfsec: https://aquasecurity.github.io/tfsec/" + @echo " - git-secrets: https://github.com/awslabs/git-secrets" + @echo " - snyk: npm install -g snyk" + @echo " - pre-commit: pip install pre-commit" diff --git a/Makefile.terraform b/Makefile.terraform new file mode 100644 index 000000000..b00196a17 --- /dev/null +++ b/Makefile.terraform @@ -0,0 +1,150 @@ +# Terraform Deployment Makefile +# Simplified commands for common Terraform operations + +.PHONY: help deploy plan destroy profile-list profile-show clean clean-locks \ + output aws-dev aws-prod azure-dev gcp-dev quick-plan aws-dev-plan quick-deploy \ + validate fmt state-list state-show docker-skip frontend-only + +# Default profile settings +PROVIDER ?= aws +PROFILE ?= dev +ACTION ?= apply + +help: ## Show this help message + @echo "CUDly Terraform Deployment Commands" + @echo "━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━" + @echo "" + @echo "Quick Deploy:" + @echo " make deploy # Deploy to AWS dev (default)" + @echo " make deploy PROFILE=prod # Deploy to AWS prod" + @echo " make deploy PROVIDER=azure PROFILE=dev" + @echo "" + @echo "Planning:" + @echo " make plan # Plan AWS dev deployment" + @echo " make plan PROFILE=prod # Plan AWS prod deployment" + @echo "" + @echo "Profile Management:" + @echo " make profile-list # List all available profiles" + @echo " make profile-show # Show current profile contents" + @echo "" + @echo "Docker Operations:" + @echo " make docker-skip # Deploy without building the Docker image" + @echo "" + @echo "Cleanup:" + @echo " make destroy # Destroy infrastructure (asks for confirmation)" + @echo " make clean # Clean Terraform cache files" + @echo "" + @echo "Examples:" + @echo " make deploy PROVIDER=aws PROFILE=staging" + @echo " make plan PROVIDER=gcp PROFILE=prod" + @echo " make destroy PROVIDER=azure PROFILE=dev" + @echo "" + @echo "━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━" + +deploy: ## Deploy infrastructure (default: AWS dev) + @echo "🚀 Deploying to $(PROVIDER) $(PROFILE)..." + @./scripts/tf-deploy.sh $(PROVIDER) $(PROFILE) apply + +plan: ## Show deployment plan (default: AWS dev) + @echo "📋 Planning deployment to $(PROVIDER) $(PROFILE)..." + @./scripts/tf-deploy.sh $(PROVIDER) $(PROFILE) plan + +destroy: ## Destroy infrastructure (asks for confirmation) + @echo "⚠️ Destroying $(PROVIDER) $(PROFILE) infrastructure..." + @./scripts/tf-deploy.sh $(PROVIDER) $(PROFILE) destroy + +output: ## Show Terraform outputs + @./scripts/tf-deploy.sh $(PROVIDER) $(PROFILE) output + +profile-list: ## List all available profiles + @echo "Available Profiles:" + @echo "━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━" + @echo "" + @echo "AWS:" + @ls -1 terraform/profiles/aws/*.tfvars 2>/dev/null | xargs -n1 basename | sed 's/.tfvars$$/ /' | sed 's/^/ /' || echo " (none)" + @echo "" + @echo "Azure:" + @ls -1 terraform/profiles/azure/*.tfvars 2>/dev/null | xargs -n1 basename | sed 's/.tfvars$$/ /' | sed 's/^/ /' || echo " (none)" + @echo "" + @echo "GCP:" + @ls -1 terraform/profiles/gcp/*.tfvars 2>/dev/null | xargs -n1 basename | sed 's/.tfvars$$/ /' | sed 's/^/ /' || echo " (none)" + @echo "" + +profile-show: ## Show current profile contents + @echo "Profile: $(PROVIDER)/$(PROFILE)" + @echo "━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━" + @cat terraform/profiles/$(PROVIDER)/$(PROFILE).tfvars 2>/dev/null || echo "Profile not found" + +clean: ## Clean Terraform cache and state files (preserves .terraform.lock.hcl) + @echo "🧹 Cleaning Terraform cache files..." + @find terraform/environments -name ".terraform" -type d -exec rm -rf {} + 2>/dev/null || true + @find terraform/environments -name "*.tfstate.backup" -type f -delete 2>/dev/null || true + @echo "✅ Clean complete" + +clean-locks: ## Delete .terraform.lock.hcl files (causes provider re-resolution on next init) + @echo "⚠️ Deleting lock files — run terraform init to re-resolve providers" + @find terraform/environments -name ".terraform.lock.hcl" -type f -delete 2>/dev/null || true + @echo "✅ Lock files deleted" + +# Convenience shortcuts +aws-dev: ## Deploy to AWS dev + @$(MAKE) deploy PROVIDER=aws PROFILE=dev + +aws-prod: ## Deploy to AWS prod + @$(MAKE) deploy PROVIDER=aws PROFILE=prod + +azure-dev: ## Deploy to Azure dev + @$(MAKE) deploy PROVIDER=azure PROFILE=dev + +gcp-dev: ## Deploy to GCP dev + @$(MAKE) deploy PROVIDER=gcp PROFILE=dev + +# Quick actions +quick-plan: aws-dev-plan ## Quick plan for AWS dev +aws-dev-plan: + @$(MAKE) plan PROVIDER=aws PROFILE=dev + +quick-deploy: aws-dev ## Quick deploy to AWS dev + +# Validation +validate: ## Validate Terraform configuration + @echo "🔍 Validating Terraform configuration..." + @cd terraform/environments/$(PROVIDER)/$(PROFILE) && terraform validate + +fmt: ## Format Terraform files + @echo "✨ Formatting Terraform files..." + @terraform fmt -recursive terraform/ + +# State management +state-list: ## List resources in Terraform state + @cd terraform/environments/$(PROVIDER)/$(PROFILE) && terraform state list + +state-show: ## Show detailed state for a resource + @cd terraform/environments/$(PROVIDER)/$(PROFILE) && terraform state show $(RESOURCE) + +# Docker operations +# +# Note: an earlier draft of this file also had a "docker-build" target +# (build the image but skip the ECR push, via a "skip_docker_push" var) +# and a "frontend-skip" target (skip frontend build, via an +# "enable_frontend_build" var). Neither variable exists anywhere in +# terraform/environments/*/build.tf or variables.tf any more: +# - the build module has no build-without-push option, so "docker-build" +# has no current equivalent; +# - frontend assets are now bundled into the Docker/app build rather than +# built as a separate Terraform-controlled step, and the closest +# surviving flag (enable_cdn) only toggles the CloudFront distribution +# in front of the frontend (defaulting to false already), not whether +# the frontend gets built - so a "frontend-skip" wrapper would be +# redundant with plain `make deploy`. +# Only the "skip the whole Docker build" case still maps to a real +# variable (enable_docker_build, replacing the old skip_docker_build). +docker-skip: ## Deploy without building the Docker image (uses pre-built image_uri) + @echo "⏭️ Deploying without Docker build..." + @./scripts/tf-deploy.sh $(PROVIDER) $(PROFILE) apply -var="enable_docker_build=false" + +# Frontend operations +frontend-only: ## Deploy frontend only + @echo "🎨 Deploying frontend..." + @cd terraform/environments/$(PROVIDER)/$(PROFILE) && \ + terraform apply -var-file="../../../profiles/$(PROVIDER)/$(PROFILE).tfvars" -target=module.frontend diff --git a/NOTICE b/NOTICE new file mode 100644 index 000000000..32ee9608b --- /dev/null +++ b/NOTICE @@ -0,0 +1,93 @@ +CUDly - Multi-Cloud Commitment & Usage Discount Manager +Copyright (c) LeanerCloud + +This product includes software developed by third parties. + +======================================================================== +Go Dependencies (from go.mod) +======================================================================== + +AWS SDK for Go v2 Apache-2.0 + github.com/aws/aws-sdk-go-v2 + https://github.com/aws/aws-sdk-go-v2/blob/main/LICENSE.txt + +AWS Lambda Go Apache-2.0 + github.com/aws/aws-lambda-go + https://github.com/aws/aws-lambda-go/blob/main/LICENSE + +Azure SDK for Go MIT + github.com/Azure/azure-sdk-for-go + https://github.com/Azure/azure-sdk-for-go/blob/main/LICENSE.txt + +Google Cloud Go SDK Apache-2.0 + cloud.google.com/go + https://github.com/googleapis/google-cloud-go/blob/main/LICENSE + +Google API Go Client BSD-3-Clause + google.golang.org/api + https://github.com/googleapis/google-api-go-client/blob/main/LICENSE + +gRPC-Go Apache-2.0 + google.golang.org/grpc + https://github.com/grpc/grpc-go/blob/master/LICENSE + +Cobra Apache-2.0 + github.com/spf13/cobra + https://github.com/spf13/cobra/blob/main/LICENSE.txt + +Testify MIT + github.com/stretchr/testify + https://github.com/stretchr/testify/blob/master/LICENSE + +pgx (PostgreSQL driver) MIT + github.com/jackc/pgx/v5 + https://github.com/jackc/pgx/blob/master/LICENSE + +pgxmock BSD-2-Clause + github.com/pashagolub/pgxmock/v4 + https://github.com/pashagolub/pgxmock/blob/master/LICENSE + +golang-migrate MIT + github.com/golang-migrate/migrate/v4 + https://github.com/golang-migrate/migrate/blob/master/LICENSE + +google/uuid BSD-3-Clause + github.com/google/uuid + https://github.com/google/uuid/blob/master/LICENSE + +testcontainers-go MIT + github.com/testcontainers/testcontainers-go + https://github.com/testcontainers/testcontainers-go/blob/main/LICENSE + +Go standard library extensions BSD-3-Clause + golang.org/x/crypto, golang.org/x/term + https://github.com/golang/crypto/blob/master/LICENSE + +YAML v3 MIT / Apache-2.0 + gopkg.in/yaml.v3 + https://github.com/go-yaml/yaml/blob/v3/LICENSE + +======================================================================== +Frontend Dependencies (from frontend/package.json) +======================================================================== + +Chart.js MIT + https://github.com/chartjs/Chart.js/blob/master/LICENSE.md + +webpack MIT + https://github.com/webpack/webpack/blob/main/LICENSE + +TypeScript Apache-2.0 + https://github.com/microsoft/TypeScript/blob/main/LICENSE.txt + +Jest MIT + https://github.com/jestjs/jest/blob/main/LICENSE + +Testing Library MIT + https://github.com/testing-library/dom-testing-library/blob/main/LICENSE + +ESLint MIT + https://github.com/eslint/eslint/blob/main/LICENSE + +Babel MIT + https://github.com/babel/babel/blob/main/LICENSE diff --git a/README.md b/README.md new file mode 100644 index 000000000..187689000 --- /dev/null +++ b/README.md @@ -0,0 +1,756 @@ +# CUDly - Multi-Cloud Commitment & Usage Discount Manager + +[![License: OSL-3.0](https://img.shields.io/badge/License-OSL--3.0-blue.svg)](https://opensource.org/licenses/OSL-3.0) +[![Go Version](https://img.shields.io/badge/Go-1.25+-00ADD8.svg)](https://go.dev/) + +CUDly is a comprehensive CLI tool for managing cloud cost commitments across AWS, Azure, and GCP. It helps organizations optimize cloud spending by automating the discovery, analysis, and purchase of multiple Reserved Instances, Savings Plans, and Committed Use Discounts by running a single command. + +## CLI Reference + +Full flag documentation, examples, and subcommand reference: [docs/cli/README.md](docs/cli/README.md) + +Topic pages: + +- [Filtering](docs/cli/filtering.md) - account, region, engine, instance-type, SP-type, and threshold filters +- [Purchase Safety](docs/cli/purchase-safety.md) - dry-run, audit log, idempotency window, and guardrails +- [Cloud Setup](docs/cli/cloud-setup.md) - `configure-azure` and `configure-gcp` self-hosted credential bootstrap + +## MCP Server + +CUDly also ships an MCP server (`cudly-mcp`) that lets Claude and other MCP clients search recommendations and drive RI, Savings Plan, and CUD purchases across AWS, Azure, and GCP, with the same dry-run-by-default safety as the CLI. + +Setup and usage: [mcp/README.md](mcp/README.md) + +## Key Features + +- **Multi-Cloud Support** - Unified interface for AWS (production), Azure (experimental), and GCP (experimental) +- **Intelligent Recommendations** - Fetches and analyzes commitment recommendations from cloud provider APIs +- **Safe Purchase Automation** - Execute purchases with built-in safety controls (dry-run by default) +- **Flexible Coverage Control** - Purchase only a percentage of recommendations for gradual adoption +- **CSV Workflow** - Generate recommendations, review offline, then execute purchases +- **Advanced Filtering** - Filter by region, instance type, engine, and account +- **Comprehensive Reporting** - Detailed cost estimates, savings calculations, and audit trails + +## Supported Cloud Providers & Services + +### AWS Services + +| Service | Commitment Type | Description | +|---------|----------------|-------------| +| Amazon RDS | Reserved Instances | MySQL, PostgreSQL, MariaDB, Oracle, SQL Server, Aurora | +| Amazon ElastiCache | Reserved Nodes | Redis, Memcached | +| Amazon EC2 | Reserved Instances | All instance families | +| Amazon OpenSearch | Reserved Instances | Search domain instances | +| Amazon Redshift | Reserved Nodes | DC2 and RA3 node types | +| Amazon MemoryDB | Reserved Nodes | Memory-optimized nodes | +| Savings Plans | Hourly Commitments | Compute, EC2 Instance, SageMaker, Database | + +### Azure Services (Experimental) + +| Service | Commitment Type | +|---------|----------------| +| Azure SQL Database | Reserved Capacity | +| Azure Virtual Machines | Reserved Instances | +| Azure Cache for Redis | Reserved Capacity | +| Azure Cosmos DB | Reserved Capacity | +| Azure Cognitive Search | Reserved Capacity | +| Azure Managed Redis | Reserved Capacity | +| Azure Savings Plans | Hourly Commitments | +| Azure Synapse Analytics | Reserved Capacity | + +### GCP Services (Experimental) + +| Service | Commitment Type | +|---------|----------------| +| Compute Engine | Committed Use Discounts | +| Cloud SQL | Committed Use Discounts | +| Memorystore | Committed Use Discounts | + +### AWS CLI Support Matrix + +**Tested** means the service has been exercised end-to-end with real AWS +accounts and validated in production workloads. **Experimental** means the +implementation exists and is functional, but needs real-world validation -- +contributions and testers are very welcome. + +| AWS Service | CLI Flag | Status | +| ----------- | -------- | ------ | +| Amazon RDS | `rds` | **Tested** | +| Amazon ElastiCache | `elasticache` | **Tested** | +| Amazon EC2 (Reserved Instances) | `ec2` | Experimental (seeking testers) | +| Amazon OpenSearch | `opensearch` | Experimental (seeking testers) | +| Amazon Redshift | `redshift` | Experimental (seeking testers) | +| Amazon MemoryDB | `memorydb` | Experimental (seeking testers) | +| Savings Plans (Compute, EC2 Instance, SageMaker, Database) | `savingsplans` | Experimental (seeking testers) | + +## Installation + +### From Source + +```bash +git clone https://github.com/LeanerCloud/CUDly.git +cd CUDly +go build -o cudly cmd/*.go +``` + +### Using Go Install + +```bash +go install github.com/LeanerCloud/CUDly/cmd@latest +``` + +## Quick Start + +### 1. Get Recommendations (Dry Run) + +```bash +# Get RDS recommendations with default settings (3-year, no-upfront, 80% coverage) +./cudly --services rds + +# Get recommendations for multiple services +./cudly --services rds,elasticache,ec2 + +# Get recommendations for all supported services +./cudly --all-services +``` + +### 2. Review and Refine + +```bash +# Apply filters to narrow down recommendations +./cudly --services rds \ + --include-regions us-east-1,eu-west-1 \ + --exclude-instance-types db.t2.micro \ + --coverage 50 +``` + +### 3. Execute Purchases + +```bash +# Purchase from generated CSV (requires explicit --purchase flag) +./cudly --input-csv cudly-dryrun-*.csv --purchase + +# Skip confirmation prompt +./cudly --input-csv cudly-dryrun-*.csv --purchase --yes +``` + +## Command Reference + +### Service Selection + +| Flag | Description | Default | +|------|-------------|---------| +| `-s, --services` | Comma-separated service list. Per-RI services: `rds`, `elasticache`, `ec2`, `opensearch`, `redshift`, `memorydb`. Per-plan-type Savings Plans: `savings-plans-compute`, `savings-plans-ec2instance`, `savings-plans-sagemaker`, `savings-plans-database`. Fan-out aliases: `savingsplans`, `savings-plans`, and `sp` expand to all four SP plan types. | rds | +| `--all-services` | Process all supported services | false | + +### Purchase Configuration + +| Flag | Description | Default | +|------|-------------|---------| +| `-p, --payment` | Payment option: `all-upfront`, `partial-upfront`, `no-upfront` | no-upfront | +| `-t, --term` | Term in years: `1` or `3` | 3 | +| `-c, --coverage` | Coverage percentage (0-100) — % of each recommendation's instance count to purchase | 80 | +| `-u, --target-coverage` | Target % (0-100) of historical demand to cover with commitments; the rest spills to on-demand. Sizes counts so projected coverage approximates target, projected utilization stays near 100%. Overrides `--coverage`. | 0 (disabled) | +| `--max-instances` | Maximum instances to purchase (0 = unlimited) | 0 | +| `--override-count` | Override recommended count with specific value | 0 | + +> **`--coverage` vs `--target-coverage`**: two related but distinct +> sizing levers. `--coverage` scales each AWS recommendation's instance +> count by a fixed fraction (`rec.Count * coverage/100`). +> `--target-coverage` sizes against historical average hourly usage +> instead (`floor(avg * target/100)`), so the resulting count reflects +> real demand rather than AWS's recommended count. Both lean the same +> direction (higher value = more RIs, lower value = fewer), but +> `--target-coverage` is the right lever when the historical-usage +> signal is what you want to size by and you're explicitly leaving +> on-demand headroom for growth or bursts. + +### Execution Control + +| Flag | Description | Default | +| ----------------- | --------------------------------------------- | -------------- | +| `--purchase` | Execute actual purchases (dry-run by default) | false | +| `--yes` | Skip confirmation prompts | false | +| `-i, --input-csv` | Input CSV file with recommendations | - | +| `-o, --output` | Output CSV file path | auto-generated | + +### Filtering + +| Flag | Description | +| ---------------------------- | --------------------------------------------------------------------------------- | +| `--include-regions` | Only include these regions | +| `--exclude-regions` | Exclude these regions | +| `--include-instance-types` | Only include these instance types | +| `--exclude-instance-types` | Exclude these instance types | +| `--include-engines` | Only include these database engines | +| `--exclude-engines` | Exclude these database engines | +| `--include-accounts` | Only include these account names | +| `--exclude-accounts` | Exclude these account names | +| `--include-extended-support` | Include instances on extended support engine versions (see below) | +| `--include-sp-types` | Only include these Savings Plan types (Compute, EC2Instance, SageMaker, Database) | +| `--exclude-sp-types` | Exclude these Savings Plan types | + +### Extended Support Filtering + +By default, CUDly excludes instances running on database engine versions that are in AWS Extended Support. This is because Extended Support incurs additional per-vCPU-hour charges that may offset RI savings. + +For example, MySQL 5.7 and PostgreSQL 11 are in Extended Support. Instances running these versions are automatically excluded from RI recommendations. + +**Note:** This feature requires the `--validation-profile` flag to specify an AWS profile with permissions to describe RDS instances across all member accounts in your organization. + +```bash +# Extended support filtering with validation profile +./cudly --services rds --validation-profile my-org-reader-profile + +# Include extended support instances (skip filtering) +./cudly --services rds --include-extended-support +``` + +This is useful if you plan to upgrade the database version before the RI term ends, or if the Extended Support charges are acceptable for your use case. + +### Duplicate Purchase Prevention + +CUDly automatically checks for Reserved Instances purchased within the last 24 hours and adjusts recommendations to avoid duplicate purchases. This is useful when running the tool multiple times in quick succession or when recovering from partial purchase failures. + +For example, if you purchase 5 db.r6g.large RIs and run CUDly again within 24 hours, those 5 instances will be subtracted from the recommendation count to prevent double-purchasing. + +### Authentication + +| Flag | Description | +| ---------------------- | ---------------------------------------- | +| `--profile` | AWS profile to use | +| `--validation-profile` | AWS profile for instance type validation | + +## Usage Examples + +### Example 1: Conservative RDS Adoption + +Purchase 50% of 1-year partial-upfront RDS recommendations: + +```bash +./cudly --services rds \ + --payment partial-upfront \ + --term 1 \ + --coverage 50 +``` + +### Example 2: Multi-Service with Different Coverage + +Apply different coverage percentages per service: + +```bash +./cudly \ + --services rds,elasticache,ec2 \ + --rds-coverage 50 \ + --elasticache-coverage 80 \ + --ec2-coverage 100 \ + --payment no-upfront \ + --term 3 +``` + +### Example 3: Regional Focus + +Only process specific regions with instance limits: + +```bash +./cudly --services ec2 \ + --include-regions us-east-1,us-west-2 \ + --max-instances 50 \ + --payment all-upfront \ + --term 3 +``` + +### Example 4: CSV-Based Workflow + +```bash +# Step 1: Generate recommendations +./cudly --all-services --output recommendations.csv + +# Step 2: Review CSV file externally + +# Step 3: Purchase with filters +./cudly \ + --input-csv recommendations.csv \ + --include-regions us-east-1 \ + --exclude-instance-types db.t2.micro,cache.t2.micro \ + --coverage 75 \ + --purchase +``` + +### Example 5: Exclude Small Instances + +```bash +./cudly --services rds,elasticache \ + --exclude-instance-types db.t2.micro,db.t2.small,db.t3.micro,cache.t2.micro \ + --payment partial-upfront \ + --term 3 +``` + +### Example 6: Database Savings Plans Only + +```bash +# Get only Database Savings Plans recommendations using the per-plan-type slug +./cudly --services savings-plans-database \ + --term 1 \ + --coverage 80 +``` + +### Example 7: Compute + EC2 Instance Savings Plans + +```bash +# Pick exactly the SP plan-types you want by listing per-plan-type slugs +./cudly --services savings-plans-compute,savings-plans-ec2instance \ + --term 3 \ + --coverage 80 +``` + +### Example 8: All Savings Plans via Fan-out Alias + +```bash +# `savingsplans` (and `savings-plans`, `sp`) is shorthand that fans out to every SP +# plan type -- equivalent to listing all four per-plan-type slugs. +./cudly --services savingsplans \ + --term 3 \ + --coverage 80 +``` + +> **Per-plan-type vs alias**: prefer the explicit per-plan-type slugs +> (`savings-plans-compute`, `savings-plans-ec2instance`, +> `savings-plans-sagemaker`, `savings-plans-database`) when you want +> precise scope. Use the `savingsplans` / `savings-plans` / `sp` alias only when you +> intentionally want all four SP plan types together. + +## Coverage Percentage + +The coverage percentage controls what portion of recommendations to act on: + +| Coverage | Description | Use Case | +| -------- | ----------- | -------- | +| 100% | All recommended instances | Maximum savings, stable workloads | +| 75% | Three-quarters of recommendations | Balanced approach | +| 50% | Half of recommendations | Conservative adoption | +| 25% | Quarter of recommendations | Testing/validation | +| 0% | Skip service entirely | Exclude from processing | + +## Per-Account Service Overrides + +Per-account service overrides let you tweak the global Settings → Purchasing +defaults (term, payment, coverage) on a per-account, per-service basis. + +### When to use them + +Use overrides when an account's purchasing policy differs from the rest of +your fleet: + +- **Dev/staging accounts**: prefer 1-year, no-upfront for RDS to keep + flexibility cheap, while production uses the global 3-year all-upfront. +- **Workload-shape outliers**: an account that runs a steady ElastiCache + cluster can override to 3-year all-upfront for ElastiCache only, while + the same account inherits the global default for everything else. +- **Pilot rollouts**: enable a new SP plan-type on one account first, + leave it disabled globally, then promote when proven. + +If every account should get the same change, edit the **global** Settings -> +Purchasing card instead; that propagates without per-account work. + +### How to create an override (web UI) + +1. Open the dashboard → **Settings → Accounts**. +2. Find the account row → click the row to expand it → click + **Service overrides**. +3. The override modal opens. Pick the (provider, service) pair you want + to override and fill in any of `term`, `payment`, `coverage`, + `enabled`. Fields you leave blank inherit the global default (see + "What 'Inherit' means" below). +4. Click **Save**. The override row appears under the account; the + recommendation engine reads it on the next refresh. + +> **All providers supported**: the override modal lists services for +> the account's provider (AWS, Azure, GCP). The UI and the backend +> both accept overrides for any provider. + +### How to edit an existing override + +- All override rows support **inline edit** of Term, Payment, Coverage, and + Enabled directly on the row. Each field persists immediately on change. +- **Delete** removes the override entirely; the account falls back to + the global default for that (provider, service) pair. + +### What "Inherit" means + +A blank field on an override is **not stored** as a sentinel value. The +PUT request omits the field, the row stays sparse, and the recommendation +engine reads the global default at evaluation time. So if you set the +global default from `3yr no-upfront` to `1yr no-upfront`, every override +that left `term` blank starts producing 1-year recommendations +automatically. Overrides that explicitly set `term: 3yr` keep that. + +### API parity + +The override modal targets the same endpoint as scripted setups: +`PUT /api/accounts/{id}/service-overrides/{provider}/{service}`. Existing +automation continues to work without change; the UI and the API write to +the same `account_service_overrides` row. + +## Safety Features + +CUDly includes multiple safety mechanisms to prevent unintended purchases: + +1. **Dry-run by default** - No purchases without explicit `--purchase` flag +2. **Interactive confirmation** - Prompts before actual purchases (unless `--yes`) +3. **CSV workflow** - Review recommendations before purchasing +4. **Coverage control** - Purchase only what you need +5. **Instance limits** - Cap total purchases with `--max-instances` +6. **Duplicate prevention** - Checks for existing commitments +7. **Instance type validation** - Validates against known types +8. **Detailed logging** - Full audit trail of operations +9. **CSV exports** - Permanent record of all recommendations and purchases + +## Cloud Provider Authentication + +### AWS + +CUDly uses the standard AWS SDK credential chain: + +1. Environment variables (`AWS_ACCESS_KEY_ID`, `AWS_SECRET_ACCESS_KEY`) +2. Shared credentials file (`~/.aws/credentials`) +3. AWS config file (`~/.aws/config`) +4. IAM instance role (EC2/ECS) + +#### Required IAM Permissions + +```json +{ + "Version": "2012-10-17", + "Statement": [ + { + "Sid": "CostExplorer", + "Effect": "Allow", + "Action": [ + "ce:GetReservationPurchaseRecommendation", + "ce:GetReservationUtilization", + "ce:GetReservationCoverage", + "ce:GetSavingsPlansPurchaseRecommendation" + ], + "Resource": "*" + }, + { + "Sid": "ReservedInstanceOperations", + "Effect": "Allow", + "Action": [ + "rds:DescribeReservedDBInstancesOfferings", + "rds:DescribeReservedDBInstances", + "rds:PurchaseReservedDBInstancesOffering", + "elasticache:DescribeReservedCacheNodesOfferings", + "elasticache:DescribeReservedCacheNodes", + "elasticache:PurchaseReservedCacheNodesOffering", + "ec2:DescribeReservedInstancesOfferings", + "ec2:DescribeReservedInstances", + "ec2:PurchaseReservedInstancesOffering", + "es:DescribeReservedInstanceOfferings", + "es:DescribeReservedInstances", + "es:PurchaseReservedInstanceOffering", + "redshift:DescribeReservedNodeOfferings", + "redshift:DescribeReservedNodes", + "redshift:PurchaseReservedNodeOffering", + "memorydb:DescribeReservedNodesOfferings", + "memorydb:DescribeReservedNodes", + "memorydb:PurchaseReservedNodesOffering", + "savingsplans:DescribeSavingsPlans", + "savingsplans:CreateSavingsPlan" + ], + "Resource": "*" + }, + { + "Sid": "RegionDiscovery", + "Effect": "Allow", + "Action": [ + "ec2:DescribeRegions", + "ec2:DescribeInstanceTypeOfferings" + ], + "Resource": "*" + }, + { + "Sid": "AccountDiscovery", + "Effect": "Allow", + "Action": [ + "sts:GetCallerIdentity", + "organizations:ListAccounts" + ], + "Resource": "*" + } + ] +} +``` + +### Azure (Experimental) + +Uses Azure SDK DefaultAzureCredential: + +1. Azure CLI (`az login`) +2. Environment variables (`AZURE_TENANT_ID`, `AZURE_CLIENT_ID`, `AZURE_CLIENT_SECRET`) +3. Managed Identity (Azure VM) + +### GCP (Experimental) + +Uses Google Cloud SDK credential chain: + +1. Service account JSON (`GOOGLE_APPLICATION_CREDENTIALS`) +2. Application Default Credentials +3. gcloud CLI authentication + +## Output Format + +CUDly generates CSV files with comprehensive details: + +```csv +Timestamp,Status,Service,Provider,Account,Region,ResourceType,Count,Term,PaymentOption,UpfrontCost,RecurringCost,TotalCost,EstimatedSavings,PurchaseID +``` + +### File Naming Convention + +- Dry run: `cudly-dryrun-YYYYMMDD-HHMMSS.csv` +- Purchase: `cudly-purchase-YYYYMMDD-HHMMSS.csv` + +## Architecture + +```text +CUDly/ +├── cmd/ # CLI entry point and orchestration +├── pkg/ # Shared multi-cloud packages +│ ├── common/ # Cloud-agnostic types and interfaces +│ └── provider/ # Provider abstraction layer +└── providers/ # Cloud-specific implementations + ├── aws/ # AWS provider (production) + │ ├── services/ # Service clients (RDS, EC2, etc.) + │ └── recommendations/ # Cost Explorer integration + ├── azure/ # Azure provider (experimental) + │ └── services/ # Azure service clients + └── gcp/ # GCP provider (experimental) + └── services/ # GCP service clients +``` + +### Design Principles + +- **Interface-driven** - All implementations follow defined interfaces for testability +- **Multi-cloud abstraction** - Unified types and behaviors across providers +- **Plugin architecture** - Services registered and discovered at runtime +- **Safety-first** - Multiple layers of protection against unintended purchases + +## Web Interface (Experimental) + +> **Note: The web GUI is experimental.** It is under active development and +> has not been validated at scale. Use the CLI for production workloads. + +In addition to the CLI, this branch ships a browser-based dashboard. The same +Go binary that runs the CLI also acts as the application server: it serves the +pre-built TypeScript/Webpack frontend as static files (controlled by the +`STATIC_DIR` environment variable) and exposes a REST API at `/api/`. There is +no separate web server process. + +### What the dashboard provides + +| Area | What you can do | +| ---- | --------------- | +| **Dashboard** | Summary of active commitments, upcoming expirations, and savings trends | +| **Recommendations** | Browse and refresh commitment recommendations; trigger purchases from the UI | +| **Purchase plans** | Create, approve, pause, resume, and delete planned-purchase workflows; view execution history | +| **History** | Full purchase history with analytics and cost-breakdown views | +| **Inventory & Coverage** | List active commitments across accounts; view per-provider, per-service coverage breakdown | +| **RI Exchange** | AWS Convertible RI exchange: reshape recommendations, quote, and execute exchanges | +| **Settings** | Application configuration, cloud account credentials, user/group management, API keys | + +### Capabilities and limitations + +The web interface is included in `main`. The dashboard is operational for AWS workloads; Azure +and GCP support in the web UI follows the same maturity as the CLI providers +(both are experimental). Specifically: + +- **AWS**: recommendations, purchases, RI exchange, inventory, and coverage + views are all wired and backed by real AWS APIs (Cost Explorer, EC2, RDS, + etc.). This is the primary tested path. +- **Azure**: reservation recommendations and purchases are implemented in the + API handlers (see `internal/api/handler_recommendations.go`, + `providers/azure/`), but Azure support is experimental. The RI Exchange + feature covers Azure Convertible RIs as a distinct code path. +- **GCP**: GCP commitment recommendations and purchases are experimental. The + handler routing exists, but end-to-end coverage is limited compared to AWS. +- The RI Exchange feature currently targets AWS Convertible EC2 Reserved + Instances only. +- Multi-account support (AWS Organizations) is implemented; Azure/GCP + multi-account federation is in progress. + +### Deployment (self-hosted via Terraform) + +CUDly is **self-hosted only**. You deploy it into your own cloud account using +the Terraform configurations under `terraform/environments/`. The Terraform +modules build and push a Docker container image, provision the database, +secrets, and networking, and deploy the application to one of the supported +runtimes. + +| Cloud | Runtime | Terraform environment | +| ----- | ------- | --------------------- | +| AWS | Lambda (default) or Fargate (ECS) | `terraform/environments/aws/` | +| GCP | Cloud Run | `terraform/environments/gcp/` | +| Azure | Container Apps | `terraform/environments/azure/` | + +#### Prerequisites + +- Terraform >= 1.6.0 +- Docker with buildx +- Go 1.26.6+ +- Cloud CLI authenticated: `aws`, `gcloud`, or `az` + +#### Quick deploy (using the helper script) + +```bash +# AWS dev +./scripts/tf-deploy.sh aws dev + +# GCP dev +./scripts/tf-deploy.sh gcp dev + +# Azure dev +./scripts/tf-deploy.sh azure dev +``` + +#### Manual Terraform (AWS example) + +```bash +cd terraform/environments/aws +cp dev.tfvars.example dev.tfvars # edit with your values +terraform init -backend-config=backends/dev.tfbackend +terraform plan -var-file=dev.tfvars +terraform apply -var-file=dev.tfvars +``` + +See [`docs/DEPLOYMENT.md`](docs/DEPLOYMENT.md) for the full deployment guide, +including Azure and GCP details, CDN/CloudFront configuration, remote state +backends, and CI/CD integration. + +**Key `tfvars` fields** + +| Variable | Purpose | +| -------- | ------- | +| `admin_email` | Email address for the initial administrator account | +| `admin_password` | Initial admin password (leave unset to auto-generate and store in Secrets Manager) | +| `compute_platform` | AWS only: `"lambda"` (default, scale-to-zero) or `"fargate"` (always-warm ECS) | + +The Terraform apply also handles Docker image build/push and database +migrations automatically on each apply. + +### Accessing the dashboard + +After `terraform apply` completes, retrieve the application URL from the +Terraform outputs: + +```bash +# AWS Lambda +terraform -chdir=terraform/environments/aws output lambda_function_url + +# AWS Fargate (ALB) +terraform -chdir=terraform/environments/aws output fargate_api_url + +# GCP Cloud Run +terraform -chdir=terraform/environments/gcp output cloud_run_service_url + +# Azure Container Apps +terraform -chdir=terraform/environments/azure output container_app_url +``` + +Open that URL in your browser. On a fresh deployment the login page includes a +one-time **"Set up admin"** step. Provide the `admin_email` you configured in +`tfvars` and either the password you set or the one retrieved from Secrets +Manager: + +```bash +# Retrieve the auto-generated admin password (AWS) +aws secretsmanager get-secret-value \ + --secret-id "$(terraform -chdir=terraform/environments/aws output -raw admin_password_secret_name)" \ + --query SecretString --output text +``` + +After the admin account is created, log in with that email and password. You +can then add more users, configure cloud account credentials, and begin using +the dashboard. + +## Development + +See [`docs/DEVELOPMENT.md`](docs/DEVELOPMENT.md) for the full development +guide, including the local Docker environment, database migrations, hot +reload, and debugging workflows. + +### Prerequisites + +- Go 1.26.6 or later +- AWS/Azure/GCP credentials for integration testing + +### Building + +```bash +# Build binary +go build -o cudly cmd/*.go + +# Run tests +go test ./... + +# Run tests with coverage +go test -cover ./... + +# Run specific package tests +go test ./providers/aws/... +``` + +### Project Structure + +| Directory | Purpose | +| --------- | ------- | +| `cmd/` | CLI implementation, flag parsing, orchestration | +| `pkg/common/` | Cloud-agnostic types (Provider, Service, Commitment) | +| `pkg/provider/` | Provider interface, registry, factory | +| `providers/aws/` | AWS implementation with 8 service clients | +| `providers/azure/` | Azure implementation (experimental) | +| `providers/gcp/` | GCP implementation (experimental) | + +## Contributing + +Contributions are welcome! Please see [CONTRIBUTING.md](CONTRIBUTING.md) for guidelines. + +### Areas for Contribution + +- Additional commitment-eligible services (any provider service that offers reserved capacity, savings plans, or committed-use discounts) +- Azure and GCP service implementations +- Enhanced reporting and analytics +- Web UI dashboard + +## License + +This project is licensed under the Open Software License 3.0 (OSL-3.0). See the [LICENSE](LICENSE) file for details. + +The OSL-3.0 is an OSI-approved open source license that: + +- Allows commercial use, modification, and distribution +- Requires attribution and license preservation +- Includes a patent grant +- Requires derivative works to be licensed under OSL-3.0 + +## Disclaimer + +**This tool can make actual cloud commitment purchases when used with the `--purchase` flag.** + +- Always verify recommendations before purchasing +- Test thoroughly in dry-run mode first +- Start with low coverage percentages +- Use instance limits for safety +- The authors are not responsible for unintended purchases or financial commitments + +## Support + +- **Issues**: [GitHub Issues](https://github.com/LeanerCloud/CUDly/issues) +- **Discussions**: [GitHub Discussions](https://github.com/LeanerCloud/CUDly/discussions) + +## Shameless Plug + +This tool is brought to you by [LeanerCloud](https://github.com/LeanerCloud). We help companies reduce their cloud costs using a mix of services and tools such as [AutoSpotting](https://github.com/LeanerCloud/AutoSpotting). + +Running at significant scale on AWS and looking for cost optimization help? We can help you avoid committing to suboptimal resources by rightsizing and other optimizations before purchasing commitments. [Contact us](https://leanercloud.com). diff --git a/SECURITY.md b/SECURITY.md new file mode 100644 index 000000000..43ceb16ad --- /dev/null +++ b/SECURITY.md @@ -0,0 +1,125 @@ +# Security Policy & Incident Response Plan + +## Reporting a Vulnerability + +If you discover a security vulnerability in CUDly, **do not open a public GitHub issue**. + +Contact the maintainers directly via email (see repository settings for contact). Provide: + +- A description of the vulnerability and its potential impact +- Steps to reproduce +- Any suggested mitigations + +We commit to acknowledging reports within 48 hours and providing an initial assessment within 7 days. + +--- + +## Incident Response Plan + +### Severity Definitions + +| Level | Criteria | Example | +| ----- | -------- | ------- | +| **P1 – Critical** | Active exploitation, data breach in progress, or credential compromise | Leaked API keys found in logs; active data exfiltration | +| **P2 – High** | Vulnerability exploitable without authentication, or significant data exposure | Unauthenticated endpoint bypassed; PII query result in error log | +| **P3 – Medium** | Exploitable with authentication, or limited impact | Authenticated user can access another user's read-only data | +| **P4 – Low** | Theoretical risk, no active exploitation | Verbose error message reveals stack trace | + +### Response Timeline + +| Severity | Detection → Triage | Triage → Containment | Resolution | +| -------- | ----------------- | -------------------- | ---------- | +| P1 | 15 min | 1 hour | 4 hours | +| P2 | 1 hour | 4 hours | 24 hours | +| P3 | 4 hours | 24 hours | 7 days | +| P4 | 24 hours | 7 days | 30 days | + +### Response Phases + +#### 1. Detection & Triage + +- Confirm the incident is real (not a false alarm) +- Determine severity using the table above +- Assign an Incident Commander (IC) +- Open a private incident channel (Slack/Discord/email thread) + +#### 2. Containment + +- Isolate affected systems (disable credentials, block IPs, take service offline if necessary) +- Preserve logs and evidence **before** making changes +- Notify stakeholders per the communication plan below + +#### 3. Eradication + +- Remove the root cause (patch code, rotate credentials, fix misconfiguration) +- Verify the fix does not introduce new issues +- Deploy to staging and validate + +#### 4. Recovery + +- Deploy fix to production +- Gradually restore service (canary if possible) +- Monitor closely for 24 hours post-recovery + +#### 5. Post-Mortem + +- Within 5 business days of resolution +- Document: timeline, root cause, impact, fix, action items +- Update runbooks and monitoring based on learnings +- Share a sanitised summary with affected customers if applicable + +--- + +## Emergency Contacts + +| Resource | Contact / URL | +| -------- | ------------- | +| AWS Support | (create case) | +| Azure Support | → Support + troubleshooting | +| GCP Support | | +| GitHub Security Advisories | | +| Domain Registrar | (update with your registrar's emergency contact) | + +--- + +## Communication Plan + +### Internal + +- Notify all engineers with production access immediately for P1/P2 +- Use a dedicated private channel; do not discuss in public channels + +### External (GDPR Art. 33/34) + +- For personal data breaches: notify the relevant Data Protection Authority **within 72 hours** of becoming aware +- Notify affected data subjects without undue delay if the breach is likely to result in high risk to their rights + +### Customer Notification + +- For P1/P2 incidents affecting customer data: notify affected customers within 24 hours of confirmed impact +- Include: what happened, what data was affected, what actions customers should take + +--- + +## Post-Incident Checklist + +- [ ] Incident timeline documented +- [ ] Root cause identified and fixed +- [ ] All affected credentials rotated +- [ ] All affected sessions invalidated +- [ ] Monitoring/alerting updated to detect recurrence +- [ ] Post-mortem written and shared +- [ ] GDPR notification filed (if applicable) +- [ ] Runbooks updated +- [ ] Action items tracked in issue tracker + +--- + +## Runbooks + +See [.github/runbooks/](.github/runbooks/) for step-by-step procedures: + +- [Credential Compromise](.github/runbooks/credential-compromise.md) +- [Data Breach Response](.github/runbooks/data-breach-response.md) +- [DDoS Mitigation](.github/runbooks/ddos-mitigation.md) +- [Compromised Dependency](.github/runbooks/compromised-dependency.md) diff --git a/arm/CUDly-CrossSubscription/setup-gcp-wif.sh b/arm/CUDly-CrossSubscription/setup-gcp-wif.sh new file mode 100755 index 000000000..2eb122b67 --- /dev/null +++ b/arm/CUDly-CrossSubscription/setup-gcp-wif.sh @@ -0,0 +1,512 @@ +#!/usr/bin/env bash +# setup-gcp-wif.sh — Configure a GCP Workload Identity Pool and Provider so that +# CUDly (running on AWS or with an OIDC token) can access GCP without a service +# account key file. For an AWS provider it outputs the external-account +# credential config JSON to store as gcp_workload_identity_config in CUDly. For +# an OIDC provider that config can only be built once you say where the CUDly +# deployment reads its token from (--oidc-credential-source); without it the +# script prints the WIF audience to register as gcp_wif_audience instead, which +# is all CUDly needs when it signs its own subject token. +# +# SECURITY: the provider is always created with an --attribute-condition and the +# service account impersonation grant is always scoped to a single principal. +# Both require an explicit identity to pin (--aws-role-name / --oidc-subject); +# there is no permissive default, and the script exits nonzero if one is missing. +# +# Prerequisites: +# gcloud auth login (with roles/iam.workloadIdentityPoolAdmin + roles/iam.serviceAccountAdmin +# on the target project, and roles/iam.serviceAccountTokenCreator on the SA) +# +# Usage (AWS provider): +# ./setup-gcp-wif.sh \ +# --project my-gcp-project \ +# --pool-id cudly-pool \ +# --provider-id cudly-aws \ +# --provider-type aws \ +# --aws-account-id 123456789012 \ +# --aws-role-name CUDly-Execution \ +# --sa-email cudly@my-gcp-project.iam.gserviceaccount.com +# +# Usage (OIDC provider): +# ./setup-gcp-wif.sh \ +# --project my-gcp-project \ +# --pool-id cudly-pool \ +# --provider-id cudly-oidc \ +# --provider-type oidc \ +# --issuer-uri https://token.actions.githubusercontent.com \ +# --oidc-subject 'repo:my-org/my-repo:ref:refs/heads/main' \ +# --sa-email cudly@my-gcp-project.iam.gserviceaccount.com \ +# --oidc-credential-source /var/run/secrets/cudly/token +# +# --oidc-credential-source is where the CUDly deployment reads its OIDC token +# from: an absolute path (file-sourced) or an https:// URL (URL-sourced). It is +# only needed to emit the credential config JSON; omit it when CUDly signs its +# own subject token, and register the printed gcp_wif_audience instead. Add +# --oidc-credential-source-field when the URL returns the token inside a +# JSON object rather than as the raw response body. +# +# Upgrading from an earlier run: versions of this script before the +# --aws-role-name / --oidc-subject requirement granted +# roles/iam.workloadIdentityUser to every identity in the pool +# (principalSet://.../workloadIdentityPools//*). Re-running this script +# detects that binding and refuses to report success while it exists; pass +# --remove-legacy-pool-binding to delete it once the narrow grant is in place. + +set -euo pipefail + +PROJECT="" +POOL_ID="cudly-pool" +PROVIDER_ID="cudly-provider" +PROVIDER_TYPE="" # "aws" or "oidc" +SA_EMAIL="" +AWS_ACCOUNT_ID="" +AWS_ROLE_NAME="" +ISSUER_URI="" +OIDC_SUBJECT="" +OIDC_CREDENTIAL_SOURCE="" +OIDC_CREDENTIAL_SOURCE_FIELD="" +REMOVE_LEGACY_POOL_BINDING="false" + +die() { echo "Error: $*" >&2; exit 1; } + +# ── Argument parsing ─────────────────────────────────────────────────────────── +while [[ $# -gt 0 ]]; do + case $1 in + --project) PROJECT="$2"; shift 2 ;; + --pool-id) POOL_ID="$2"; shift 2 ;; + --provider-id) PROVIDER_ID="$2"; shift 2 ;; + --provider-type) PROVIDER_TYPE="$2"; shift 2 ;; + --sa-email) SA_EMAIL="$2"; shift 2 ;; + --aws-account-id) AWS_ACCOUNT_ID="$2"; shift 2 ;; + --aws-role-name) AWS_ROLE_NAME="$2"; shift 2 ;; + --issuer-uri) ISSUER_URI="$2"; shift 2 ;; + --oidc-subject) OIDC_SUBJECT="$2"; shift 2 ;; + --oidc-credential-source) OIDC_CREDENTIAL_SOURCE="$2"; shift 2 ;; + --oidc-credential-source-field) OIDC_CREDENTIAL_SOURCE_FIELD="$2"; shift 2 ;; + --remove-legacy-pool-binding) REMOVE_LEGACY_POOL_BINDING="true"; shift ;; + # Removed on purpose: --subject-condition took a raw CEL expression, so a + # value such as "true" or "assertion.sub != ''" looked like a restriction + # while admitting every subject from the issuer. --oidc-subject takes the + # subject itself and this script builds the equality condition around it. + --subject-condition) + die "--subject-condition has been removed; pass --oidc-subject '' instead (this script builds the attribute condition, so a hand-written CEL expression can no longer silently admit every subject)" ;; + *) die "Unknown argument: $1" ;; + esac +done + +# ── Value validation ─────────────────────────────────────────────────────────── +# Every value below is interpolated into either the provider's CEL +# --attribute-condition or the IAM principal identifier of the impersonation +# grant. Reject the forms that fail OPEN there rather than merely looking wrong: +# '*' would widen an IAM principal identifier to match every identity; +# '$' catches pasted ${...} placeholders, which IAM does not expand here +# (they yield a binding that matches nothing, or in a context that +# does expand them, one that matches everything); +# ' " \ ` would terminate the CEL string literal in --attribute-condition and +# let the rest of the value rewrite the condition (e.g. "|| true"). +validate_principal_value() { + local flag="$1" value="$2" pattern="$3" + [[ -n "$value" ]] || die "$flag must not be empty" + case "$value" in + *'*'*) die "$flag must not contain '*' (got: $value): a wildcard would match every identity and defeat the restriction this flag exists to apply" ;; + *'$'*) die "$flag must not contain '\$' (got: $value): it looks like an unexpanded \${...} placeholder, which is not substituted here" ;; + *\'*|*\"*|*\\*|*'`'*) die "$flag must not contain quotes or backslashes (got: $value): they would break out of the CEL string literal in the provider's attribute condition" ;; + esac + [[ "$value" =~ $pattern ]] || die "$flag has an unexpected format (got: $value)" +} + +if [[ -z "$PROJECT" || -z "$SA_EMAIL" || -z "$PROVIDER_TYPE" ]]; then + die "--project, --sa-email, and --provider-type are required" +fi +if [[ "$PROVIDER_TYPE" != "aws" && "$PROVIDER_TYPE" != "oidc" ]]; then + die "--provider-type must be 'aws' or 'oidc'" +fi + +validate_principal_value "--project" "$PROJECT" '^[a-z][-a-z0-9]{4,28}[a-z0-9]$' +validate_principal_value "--pool-id" "$POOL_ID" '^[a-z0-9][-a-z0-9]{2,30}[a-z0-9]$' +validate_principal_value "--provider-id" "$PROVIDER_ID" '^[a-z0-9][-a-z0-9]{2,30}[a-z0-9]$' +validate_principal_value "--sa-email" "$SA_EMAIL" '^[a-zA-Z0-9][-a-zA-Z0-9._]*@[a-z0-9.-]+\.gserviceaccount\.com$' + +if [[ "$PROVIDER_TYPE" == "aws" ]]; then + # An AWS provider's credential source is AWS IMDS (--aws below). Silently + # ignoring these would emit an IMDS-sourced config while the operator believes + # they pinned a token file, and would skip the shape validation the OIDC + # branch applies to them. + [[ -z "$OIDC_CREDENTIAL_SOURCE" && -z "$OIDC_CREDENTIAL_SOURCE_FIELD" ]] \ + || die "--oidc-credential-source/--oidc-credential-source-field do not apply to --provider-type aws, whose credential source is AWS IMDS" + [[ -n "$AWS_ACCOUNT_ID" ]] || die "--aws-account-id is required for --provider-type aws" + # Mandatory: without a role to pin, the provider has no attribute condition and + # every IAM principal in the AWS account can federate as the service account. + [[ -n "$AWS_ROLE_NAME" ]] || die "--aws-role-name is required for --provider-type aws. Without it the provider has no attribute condition and any IAM role in account ${AWS_ACCOUNT_ID} can federate as ${SA_EMAIL}." + validate_principal_value "--aws-account-id" "$AWS_ACCOUNT_ID" '^[0-9]{12}$' + # Matches AWS's own role-name charset [\w+=,.@-]{1,64}. The comma is safe here: + # the role name reaches --attribute-condition (a plain string flag), never the + # comma-delimited --attribute-mapping dict. + validate_principal_value "--aws-role-name" "$AWS_ROLE_NAME" '^[A-Za-z0-9_+=,.@-]{1,64}$' + # Standard AWS partition. A GovCloud or China-partition caller presents a + # different ARN prefix, so the condition below simply would not match: the + # token exchange is refused rather than admitted on a partition mismatch. + AWS_ROLE_ARN="arn:aws:sts::${AWS_ACCOUNT_ID}:assumed-role/${AWS_ROLE_NAME}" + EXPECTED_CONDITION="attribute.aws_role == '${AWS_ROLE_ARN}'" + # attribute.aws_role normalises the session ARN + # (arn:aws:sts:::assumed-role//) down to the role ARN, + # so both EXPECTED_CONDITION above and the IAM grant below can match it + # exactly instead of relying on a substring test. Kept in a variable (not + # inlined at the create-provider call below) so the reuse path can validate + # an existing provider against the exact same expression instead of a + # second hand-copied literal that could silently drift from this one. + EXPECTED_MAPPING="google.subject=assertion.arn,attribute.aws_role=assertion.arn.contains('assumed-role') ? assertion.arn.extract('{account_arn}assumed-role/') + 'assumed-role/' + assertion.arn.extract('assumed-role/{role_name}/') : assertion.arn" + # google.subject is the full session ARN here and GCP caps it at 127 bytes, + # not characters (127 chars is used below only because both charset + # regexes constraining this value -- --aws-role-name above and + # --oidc-subject below -- are ASCII-only, so chars == bytes; this budget + # would silently under/over-count if that charset were ever widened). + # "arn:aws:sts::<12 digits>:assumed-role/" is 39 characters, and the session + # name is preceded by a "/", so the role name leaves 127 - 39 - len(role) - 1 + # for a session name that AWS allows to reach 64. + AWS_SESSION_BUDGET=$((127 - 39 - ${#AWS_ROLE_NAME} - 1)) + if [[ $AWS_SESSION_BUDGET -lt 64 ]]; then + echo "Note: role name '${AWS_ROLE_NAME}' leaves only ${AWS_SESSION_BUDGET} characters" >&2 + echo " for the session name before google.subject exceeds GCP's 127-character" >&2 + echo " limit. Longer session names will be rejected at token-exchange time." >&2 + fi +else + [[ -n "$ISSUER_URI" ]] || die "--issuer-uri is required for --provider-type oidc" + # Mandatory: public issuers such as token.actions.githubusercontent.com will + # mint a token for any workload on the platform, so without a pinned subject + # any repository in the world can federate as the service account. + [[ -n "$OIDC_SUBJECT" ]] || die "--oidc-subject is required for --provider-type oidc. Without it the provider has no attribute condition and any subject issued by ${ISSUER_URI} can federate as ${SA_EMAIL}." + validate_principal_value "--issuer-uri" "$ISSUER_URI" '^https://[a-zA-Z0-9.-]+(:[0-9]+)?(/[-a-zA-Z0-9._~/]*)?$' + # '|' is permitted because Auth0/Okta-style subjects use it (google-oauth2|123). + # It is inert in both destinations: quotes are rejected above, so it cannot + # escape the CEL string literal, and it carries no meaning in a principal path. + validate_principal_value "--oidc-subject" "$OIDC_SUBJECT" '^[A-Za-z0-9][-A-Za-z0-9._:/@=+~|]*$' + # google.subject is capped at 127 characters by GCP; a longer value would be + # rejected at token-exchange time, long after this script reported success. + [[ ${#OIDC_SUBJECT} -le 127 ]] || die "--oidc-subject must be at most 127 characters (google.subject limit); got ${#OIDC_SUBJECT}" + EXPECTED_CONDITION="google.subject == '${OIDC_SUBJECT}'" + EXPECTED_MAPPING="google.subject=assertion.sub" + # --oidc-credential-source says where the CUDly deployment reads its subject + # token from, which is what create-cred-config below needs and what an OIDC + # provider does not imply. The two forms cannot be confused (a URL cannot + # start with '/' and a path cannot start with 'https://'), so the gcloud flag + # is derived from the value rather than from a second flag that could + # contradict it. Plain http:// is refused: the token would cross the network + # in cleartext. + CRED_SOURCE_FLAG="" + if [[ -n "$OIDC_CREDENTIAL_SOURCE" ]]; then + case "$OIDC_CREDENTIAL_SOURCE" in + /*) CRED_SOURCE_FLAG="--credential-source-file" ;; + https://*) CRED_SOURCE_FLAG="--credential-source-url" ;; + *) die "--oidc-credential-source must be an absolute path (file-sourced) or an https:// URL (URL-sourced); got: $OIDC_CREDENTIAL_SOURCE" ;; + esac + elif [[ -n "$OIDC_CREDENTIAL_SOURCE_FIELD" ]]; then + die "--oidc-credential-source-field only applies to a credential source; pass --oidc-credential-source as well" + fi +fi + +PROJECT_NUMBER=$(gcloud projects describe "$PROJECT" --format='value(projectNumber)') +[[ "$PROJECT_NUMBER" =~ ^[0-9]+$ ]] || die "could not resolve a numeric project number for '$PROJECT' (got: '$PROJECT_NUMBER')" + +echo "Project : $PROJECT (number: $PROJECT_NUMBER)" +echo "Pool : $POOL_ID" +echo "Provider : $PROVIDER_ID ($PROVIDER_TYPE)" +echo "Service Acct : $SA_EMAIL" +if [[ "$PROVIDER_TYPE" == "aws" ]]; then + echo "Trusted role : $AWS_ROLE_ARN" +else + echo "Trusted subj : $OIDC_SUBJECT (issuer: $ISSUER_URI)" +fi +echo "" + +# ── Idempotent pool creation ─────────────────────────────────────────────────── +if ! gcloud iam workload-identity-pools describe "$POOL_ID" \ + --project="$PROJECT" --location=global &>/dev/null; then + echo "Creating workload identity pool '${POOL_ID}'..." + gcloud iam workload-identity-pools create "$POOL_ID" \ + --project="$PROJECT" --location=global \ + --display-name="CUDly WIF pool" --quiet +else + echo "Reusing existing pool '${POOL_ID}'" +fi + +# GCP principal identifiers are pool-scoped, not provider-scoped: the grant below +# names an attribute value, and any provider in this pool that can mint that +# attribute satisfies it. A pool dedicated to this provider keeps the trust +# boundary equal to the attribute condition set below. +# +# Printed unconditionally, not only when reusing a pool: a pool this run creates +# fresh is equally exposed the moment a second provider is added to it later, and +# that operator would otherwise never have seen the warning. No script can close +# this - GCP has no provider-scoped principal form - so a notice is the ceiling. +echo "Note: principal identifiers are pool-scoped, not provider-scoped. Another" >&2 +echo " provider in pool '${POOL_ID}' that maps the same attribute value would" >&2 +echo " also satisfy the grant below. Keep this pool dedicated to CUDly" >&2 +echo " (--pool-id) unless you intend to share it." >&2 + +# gcloud's `value(...)` printer renders a map field (attributeMapping) as its +# entries sorted by key and delimiter-joined with ';' by CsvPrinter._AddRecord +# (the same code path ValuePrinter inherits from), not the comma-joined +# "key=value,key=value" form the --attribute-mapping flag takes as input. +# Comparing the raw describe output against EXPECTED_MAPPING as strings would +# therefore report every correctly-configured provider as mismatched. This +# splits each side on its own pair separator, re-sorts, and compares parsed +# key/value pairs so neither the input-vs-output shape difference nor a +# hypothetical future change to gcloud's own key ordering can produce a false +# positive; a mapping that actually differs still compares unequal. +normalize_mapping() { + local mapping="$1" pair_sep="$2" + local -a pairs + IFS="$pair_sep" read -ra pairs <<<"$mapping" + printf '%s\n' "${pairs[@]}" | sort +} + +# ── Idempotent provider creation ─────────────────────────────────────────────── +if ! gcloud iam workload-identity-pools providers describe "$PROVIDER_ID" \ + --project="$PROJECT" --location=global \ + --workload-identity-pool="$POOL_ID" &>/dev/null; then + echo "Creating ${PROVIDER_TYPE} provider '${PROVIDER_ID}'..." + if [[ "$PROVIDER_TYPE" == "aws" ]]; then + gcloud iam workload-identity-pools providers create-aws "$PROVIDER_ID" \ + --project="$PROJECT" --location=global \ + --workload-identity-pool="$POOL_ID" \ + --account-id="$AWS_ACCOUNT_ID" \ + --attribute-mapping="$EXPECTED_MAPPING" \ + --attribute-condition="$EXPECTED_CONDITION" \ + --quiet + else + # --attribute-mapping is required; without it no subject claims are mapped and + # all token exchanges are rejected at runtime despite a successful setup. + # --attribute-condition is built here from --oidc-subject so that only that + # exact subject is admitted to the pool. + gcloud iam workload-identity-pools providers create-oidc "$PROVIDER_ID" \ + --project="$PROJECT" --location=global \ + --workload-identity-pool="$POOL_ID" \ + --issuer-uri="$ISSUER_URI" \ + --attribute-mapping="$EXPECTED_MAPPING" \ + --attribute-condition="$EXPECTED_CONDITION" \ + --quiet + fi +else + # An existing provider keeps whatever attribute condition it was created with. + # Checking only that the condition is non-empty would grandfather exactly the + # values --subject-condition was removed to prevent: "true", or + # "assertion.sub != ''", both non-empty and both admitting every identity. The + # condition is therefore compared to the one this script would have written. + DELETE_HINT=" gcloud iam workload-identity-pools providers delete ${PROVIDER_ID} --project=${PROJECT} --location=global --workload-identity-pool=${POOL_ID}" + EXISTING_CONDITION=$(gcloud iam workload-identity-pools providers describe "$PROVIDER_ID" \ + --project="$PROJECT" --location=global \ + --workload-identity-pool="$POOL_ID" \ + --format='value(attributeCondition)') + # A provider of the other type would emit a credential config that does not + # match it and mint an attribute the grant below never names. Compared as an + # exact match on the single relevant field, not a prefix/suffix test against + # the account-id/issuer-uri pair gcloud tab-joins under `value(a,b)`: a + # suffix test here would accept any issuer URI merely ENDING in the expected + # one (e.g. an attacker-hosted https://evil.example.com/, + # whose /.well-known/openid-configuration GCP would then fetch from the + # attacker, who can mint a token with any subject), and a prefix test would + # accept any account ID merely STARTING with the expected one. + if [[ "$PROVIDER_TYPE" == "aws" ]]; then + EXISTING_ACCOUNT_ID=$(gcloud iam workload-identity-pools providers describe "$PROVIDER_ID" \ + --project="$PROJECT" --location=global \ + --workload-identity-pool="$POOL_ID" \ + --format='value(aws.accountId)') + if [[ "$EXISTING_ACCOUNT_ID" != "$AWS_ACCOUNT_ID" ]]; then + die "provider '${PROVIDER_ID}' already exists but is not an AWS provider for account ${AWS_ACCOUNT_ID} (describe reports account: ${EXISTING_ACCOUNT_ID:-none}). Delete it and re-run: +${DELETE_HINT}" + fi + else + EXISTING_ISSUER=$(gcloud iam workload-identity-pools providers describe "$PROVIDER_ID" \ + --project="$PROJECT" --location=global \ + --workload-identity-pool="$POOL_ID" \ + --format='value(oidc.issuerUri)') + if [[ "$EXISTING_ISSUER" != "$ISSUER_URI" ]]; then + die "provider '${PROVIDER_ID}' already exists but is not an OIDC provider for issuer ${ISSUER_URI} (describe reports issuer: ${EXISTING_ISSUER:-none}). Delete it and re-run: +${DELETE_HINT}" + fi + fi + if [[ "$EXISTING_CONDITION" != "$EXPECTED_CONDITION" ]]; then + die "provider '${PROVIDER_ID}' already exists with a different attribute condition, so this script cannot vouch for which identities enter pool '${POOL_ID}'. + found : ${EXISTING_CONDITION:-(none)} + expected: ${EXPECTED_CONDITION} +A condition that is merely non-empty is not a restriction: 'true' and +\"assertion.sub != ''\" both admit every identity the issuer will vouch for. +Delete the provider and re-run: +${DELETE_HINT}" + fi + # The condition alone does not decide who gets in: it is evaluated against + # whatever value the mapping produces, so a provider can carry the exact + # expected condition while a different mapping feeds it a normalised value + # derived from an identity this script never authorised (e.g. one where + # attribute.aws_role is built from a caller-controlled ARN path segment + # rather than the actual assumed-role name). Validating the condition + # without also validating the mapping only checks half of that property. + EXISTING_MAPPING=$(gcloud iam workload-identity-pools providers describe "$PROVIDER_ID" \ + --project="$PROJECT" --location=global \ + --workload-identity-pool="$POOL_ID" \ + --format='value(attributeMapping)') + if [[ "$(normalize_mapping "$EXISTING_MAPPING" ';')" != "$(normalize_mapping "$EXPECTED_MAPPING" ',')" ]]; then + die "provider '${PROVIDER_ID}' already exists with a different attribute mapping, so this script cannot vouch for which identities enter pool '${POOL_ID}'. + found : ${EXISTING_MAPPING:-(none)} + expected: ${EXPECTED_MAPPING} +A matching attribute condition is not sufficient on its own: the mapping +decides what value the condition is evaluated against, so a mismatched +mapping can satisfy the condition while admitting an identity this script +never authorised. +Delete the provider and re-run: +${DELETE_HINT}" + fi + echo "Reusing existing provider '${PROVIDER_ID}' (attribute condition and mapping match expected values)" +fi + +# ── Grant service account impersonation to one principal ────────────────────── +POOL_RESOURCE="projects/${PROJECT_NUMBER}/locations/global/workloadIdentityPools/${POOL_ID}" +if [[ "$PROVIDER_TYPE" == "aws" ]]; then + # Session ARNs carry a per-session suffix, so the grant names the normalised + # role attribute rather than an exact subject. + MEMBER="principalSet://iam.googleapis.com/${POOL_RESOURCE}/attribute.aws_role/${AWS_ROLE_ARN}" +else + MEMBER="principal://iam.googleapis.com/${POOL_RESOURCE}/subject/${OIDC_SUBJECT}" +fi +echo "Granting roles/iam.workloadIdentityUser on ${SA_EMAIL} to:" +echo " ${MEMBER}" +gcloud iam service-accounts add-iam-policy-binding "$SA_EMAIL" \ + --role=roles/iam.workloadIdentityUser \ + --member="$MEMBER" \ + --project="$PROJECT" \ + --quiet + +# ── Retire pool-wide grants left by earlier versions of this script ─────────── +# Scanned across every pool and every role, not just this run's --pool-id and +# roles/iam.workloadIdentityUser. Scoping the scan to the current pool would let +# a customer hide the original grant simply by re-running with a different +# --pool-id (which the pool-reuse note above actively recommends), and scoping it +# to one role would miss the same wildcard under roles/iam.serviceAccountTokenCreator, +# which confers the same impersonation power via generateAccessToken. +# +# Emitted one "rolemember" line per member so the role owning each wildcard +# is known; a bare policy grep cannot tell which role to remove it from. +# +# The fetch is deliberately kept out of the grep pipeline. Piping gcloud straight +# into `grep ... || true` would let a failed get-iam-policy produce empty output +# and read as "no wildcard grants found", reporting success over the very grant +# this check exists to catch. As a plain assignment it aborts under `set -e`. +sa_bindings() { + gcloud iam service-accounts get-iam-policy "$SA_EMAIL" \ + --project="$PROJECT" --flatten="bindings[].members" \ + --format="value(bindings.role,bindings.members)" +} +pool_wide_grants() { + grep -E $'\t''principalSet://iam\.googleapis\.com/projects/[0-9]+/locations/[^/]+/workloadIdentityPools/[^/]+/\*$' <<<"$1" || true +} + +SA_BINDINGS=$(sa_bindings) +LEGACY_GRANTS=$(pool_wide_grants "$SA_BINDINGS") +if [[ -n "$LEGACY_GRANTS" ]]; then + if [[ "$REMOVE_LEGACY_POOL_BINDING" == "true" ]]; then + while IFS=$'\t' read -r legacy_role legacy_member; do + [[ -n "$legacy_member" ]] || continue + echo "Removing pool-wide grant: ${legacy_role} -> ${legacy_member}" + gcloud iam service-accounts remove-iam-policy-binding "$SA_EMAIL" \ + --role="$legacy_role" \ + --member="$legacy_member" \ + --project="$PROJECT" \ + --quiet + done <<<"$LEGACY_GRANTS" + # Re-read the policy: removing one role's wildcard says nothing about the + # others, and reporting success here is the whole point of the check. + SA_BINDINGS=$(sa_bindings) + REMAINING=$(pool_wide_grants "$SA_BINDINGS") + [[ -z "$REMAINING" ]] || die "pool-wide grants survived removal on ${SA_EMAIL}: +${REMAINING}" + else + die "${SA_EMAIL} still grants impersonation to every identity in a workload identity pool: +${LEGACY_GRANTS} +Grants of this shape were created by earlier versions of this script and make the +narrow grant added above irrelevant: anything admitted to that pool can still +impersonate the service account. Re-run with --remove-legacy-pool-binding, or +remove each one yourself with: + gcloud iam service-accounts remove-iam-policy-binding ${SA_EMAIL} \\ + --role='' --member='' --project=${PROJECT}" + fi +fi + +# Narrow grants for identities other than this run's are not removed (they may +# belong to a second legitimate CUDly account), but they are surfaced: a re-run +# with a corrected role or subject otherwise leaves the previous one impersonating. +# The member is compared as a whole field, not with `grep -vF`. A substring +# exclusion drops every line CONTAINING this run's member, so a sibling grant +# whose member merely starts with it is suppressed: pinning CUDly-Execution would +# hide an existing grant to CUDly-Execution-Admin, in exactly the "re-ran with a +# corrected role" case this notice exists to surface. +OTHER_GRANTS=$(awk -F'\t' -v pool="iam.googleapis.com/${POOL_RESOURCE}/" -v m="$MEMBER" \ + 'index($2, pool) && $2 != m' <<<"$SA_BINDINGS" || true) +if [[ -n "$OTHER_GRANTS" ]]; then + echo "Note: ${SA_EMAIL} is also impersonable by other identities in pool '${POOL_ID}':" >&2 + while IFS= read -r grant_line; do + echo " ${grant_line}" >&2 + done <<<"$OTHER_GRANTS" + echo " Remove any that are no longer expected." >&2 +fi + +# ── Report the values CUDly needs (no secrets) ──────────────────────────────── +PROVIDER_RESOURCE="${POOL_RESOURCE}/providers/${PROVIDER_ID}" +WIF_AUDIENCE="//iam.googleapis.com/${PROVIDER_RESOURCE}" +echo "" +if [[ "$PROVIDER_TYPE" == "aws" ]]; then + CRED_CONFIG_ARGS=(--aws) +elif [[ -n "$OIDC_CREDENTIAL_SOURCE" ]]; then + CRED_CONFIG_ARGS=("${CRED_SOURCE_FLAG}=${OIDC_CREDENTIAL_SOURCE}") + if [[ -n "$OIDC_CREDENTIAL_SOURCE_FIELD" ]]; then + # A JSON response needs both the format and the key holding the token; + # gcloud's default (text) would treat the whole response body as the token. + CRED_CONFIG_ARGS+=(--credential-source-type=json "--credential-source-field-name=${OIDC_CREDENTIAL_SOURCE_FIELD}") + fi +else + CRED_CONFIG_ARGS=() +fi + +if [[ ${#CRED_CONFIG_ARGS[@]} -gt 0 ]]; then + echo "══════════════════════════════════════════════════════════════" + echo " External account credential config (store in CUDly as" + echo " gcp_workload_identity_config — contains no secrets)" + echo "══════════════════════════════════════════════════════════════" + gcloud iam workload-identity-pools create-cred-config \ + "$PROVIDER_RESOURCE" \ + --service-account="$SA_EMAIL" \ + "${CRED_CONFIG_ARGS[@]}" \ + --output-file=/dev/stdout +else + # No credential source, so no credential config. create-cred-config requires + # exactly one of (--aws | --azure | --credential-source-file | + # --credential-source-url | --executable-command); this branch used to call it + # with none, which gcloud rejects, and under `set -e` that aborted the run + # here, after the pool, the provider and the impersonation grant had all been + # created. Nothing is lost by skipping it: CUDly needs the credential config + # only when something other than CUDly holds the token. When CUDly signs its + # own subject token it needs the audience below, which is what the served + # onboarding script (internal/iacfiles/templates/gcp-wif-cli.sh.tmpl) emits. + echo "══════════════════════════════════════════════════════════════" + echo " No credential config generated: none was requested." + echo " Pass --oidc-credential-source to emit" + echo " one, or register the gcp_wif_audience below, which is all CUDly" + echo " needs when it signs its own subject token. In that case the" + echo " trusted subject '${OIDC_SUBJECT}' must be the one your CUDly" + echo " deployment signs, and ${ISSUER_URI} must be its OIDC issuer." + echo "══════════════════════════════════════════════════════════════" +fi +echo "" +echo "══════════════════════════════════════════════════════════════" +echo " CUDly account registration values:" +echo " provider : gcp" +echo " gcp_auth_mode : workload_identity_federation" +echo " gcp_project_id : ${PROJECT}" +echo " gcp_client_email (optional) : ${SA_EMAIL}" +if [[ "$PROVIDER_TYPE" != "aws" ]]; then + # OIDC only: an AWS-federated account registers the credential config printed + # above. Registering the audience instead would have CUDly present a + # self-signed assertion to a provider that only trusts AWS role ARNs. + echo " gcp_wif_audience : ${WIF_AUDIENCE}" +fi +echo "══════════════════════════════════════════════════════════════" diff --git a/arm/CUDly-CrossSubscription/setup.sh b/arm/CUDly-CrossSubscription/setup.sh new file mode 100755 index 000000000..42410c4c8 --- /dev/null +++ b/arm/CUDly-CrossSubscription/setup.sh @@ -0,0 +1,144 @@ +#!/usr/bin/env bash +# setup.sh — Create the CUDly Azure AD service principal, deploy role assignments, +# and print the values needed to register the subscription in CUDly. +# +# Prerequisites: +# az login (with an account that has Application Administrator + Owner/User Access +# Administrator on the target subscription) +# +# Usage: +# ./setup.sh [--subscription ] [--app-name ] +# +# The script is idempotent: re-running it reuses the existing App Registration +# if one with the same display name already exists. +# +# Output: prints all values needed to register the account in CUDly. + +set -euo pipefail + +APP_NAME="CUDly" +SUBSCRIPTION_ID="" +MODE="client_secret" # or "wif" for certificate-based workload identity federation + +# ── Argument parsing ─────────────────────────────────────────────────────────── +while [[ $# -gt 0 ]]; do + case $1 in + --subscription) SUBSCRIPTION_ID="$2"; shift 2 ;; + --app-name) APP_NAME="$2"; shift 2 ;; + --mode) MODE="$2"; shift 2 ;; + *) echo "Unknown argument: $1"; exit 1 ;; + esac +done + +if [[ "$MODE" != "client_secret" && "$MODE" != "wif" ]]; then + echo "Error: --mode must be 'client_secret' or 'wif'" >&2 + exit 1 +fi + +# ── Resolve subscription ─────────────────────────────────────────────────────── +if [[ -z "$SUBSCRIPTION_ID" ]]; then + SUBSCRIPTION_ID=$(az account show --query id -o tsv) +fi +az account set --subscription "$SUBSCRIPTION_ID" +TENANT_ID=$(az account show --query tenantId -o tsv) +echo "Subscription : $SUBSCRIPTION_ID" +echo "Tenant : $TENANT_ID" +echo "" + +# ── Create or reuse App Registration ────────────────────────────────────────── +echo "Looking for existing App Registration '${APP_NAME}'..." +APP_ID=$(az ad app list --display-name "$APP_NAME" --query "[0].appId" -o tsv 2>/dev/null || true) + +if [[ -z "$APP_ID" || "$APP_ID" == "None" ]]; then + echo "Creating App Registration '${APP_NAME}'..." + APP_ID=$(az ad app create \ + --display-name "$APP_NAME" \ + --sign-in-audience AzureADMyOrg \ + --query appId -o tsv) + echo "Created App: $APP_ID" +else + echo "Reusing existing App: $APP_ID" +fi + +# ── Create or reuse Service Principal ───────────────────────────────────────── +SP_OBJECT_ID=$(az ad sp show --id "$APP_ID" --query id -o tsv 2>/dev/null || true) +if [[ -z "$SP_OBJECT_ID" || "$SP_OBJECT_ID" == "None" ]]; then + echo "Creating Service Principal..." + SP_OBJECT_ID=$(az ad sp create --id "$APP_ID" --query id -o tsv) +fi +echo "SP Object ID : $SP_OBJECT_ID" +echo "" + +# ── Credential setup ────────────────────────────────────────────────────────── +CLIENT_SECRET="" +if [[ "$MODE" == "wif" ]]; then + # Use a per-run temp directory to avoid predictable /tmp paths. + WORK_DIR=$(mktemp -d) + trap 'command -v shred >/dev/null && shred -u "${WORK_DIR}/cudly-wif.key" 2>/dev/null; rm -rf "$WORK_DIR"' EXIT + + echo "Generating self-signed certificate for workload identity federation..." + openssl genrsa -out "${WORK_DIR}/cudly-wif.key" 2048 2>/dev/null + openssl req -new -x509 -key "${WORK_DIR}/cudly-wif.key" -out "${WORK_DIR}/cudly-wif.crt" \ + -days 730 -subj "/CN=CUDly-WIF" 2>/dev/null + CERT_B64=$(base64 < "${WORK_DIR}/cudly-wif.crt" | tr -d '\n') + az ad app credential reset --id "$APP_ID" --cert "$CERT_B64" --append --output none + + # WARNING: The key+cert below will appear in terminal scrollback and any session + # recording. Do NOT run this script in CI/CD or any environment that captures stdout. + # Both blocks are written to stderr so stdout redirects do not capture them. + # Store the ENTIRE output (key PEM + certificate PEM) as azure_wif_private_key in CUDly. + # The certificate block provides the x5t thumbprint required by Azure AD client assertions. + printf '\n=== Key + Certificate PEM (store BOTH blocks as azure_wif_private_key in CUDly) ===\n' >&2 + cat "${WORK_DIR}/cudly-wif.key" >&2 + cat "${WORK_DIR}/cudly-wif.crt" >&2 + printf '=== end — copy both blocks now, they will be deleted from disk ===\n\n' >&2 + # shred (Linux) preferred; dd overwrite as fallback for macOS + if command -v shred >/dev/null; then + shred -u "${WORK_DIR}/cudly-wif.key" 2>/dev/null + else + KEY_SIZE=$(wc -c < "${WORK_DIR}/cudly-wif.key") + dd if=/dev/zero of="${WORK_DIR}/cudly-wif.key" bs=1 count="$KEY_SIZE" conv=notrunc 2>/dev/null + rm -f "${WORK_DIR}/cudly-wif.key" + fi + rm -f "${WORK_DIR}/cudly-wif.crt" + trap - EXIT # key already deleted; cancel trap +else + echo "Creating client secret (valid 2 years)..." + CLIENT_SECRET=$(az ad app credential reset \ + --id "$APP_ID" \ + --display-name "CUDly-$(date +%Y%m%d)" \ + --years 2 \ + --query password -o tsv) +fi + +# ── Deploy ARM role assignments ──────────────────────────────────────────────── +SCRIPT_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)" +echo "Deploying role assignments to subscription ${SUBSCRIPTION_ID}..." +az deployment sub create \ + --location eastus \ + --template-file "${SCRIPT_DIR}/template.json" \ + --parameters servicePrincipalObjectId="$SP_OBJECT_ID" \ + --name "CUDly-CrossSubscription" \ + --no-prompt \ + --output none + +echo "" +echo "══════════════════════════════════════════════════════════════" +echo " CUDly Azure account registration values" +echo "══════════════════════════════════════════════════════════════" +echo " provider : azure" +echo " azure_auth_mode : ${MODE}" +echo " azure_subscription_id : ${SUBSCRIPTION_ID}" +echo " azure_tenant_id : ${TENANT_ID}" +echo " azure_client_id : ${APP_ID}" +if [[ "$MODE" == "client_secret" ]]; then + echo " client_secret : ${CLIENT_SECRET}" + echo "" + echo " Save the client_secret now — it will not be shown again." +else + echo "" + echo " Store the key+certificate PEM printed above as azure_wif_private_key in CUDly." + echo " Both PEM blocks are required (certificate provides x5t thumbprint for Azure AD)." + echo " Both files were deleted from disk." +fi +echo "══════════════════════════════════════════════════════════════" diff --git a/arm/CUDly-CrossSubscription/template.json b/arm/CUDly-CrossSubscription/template.json new file mode 100644 index 000000000..7a7a0c33e --- /dev/null +++ b/arm/CUDly-CrossSubscription/template.json @@ -0,0 +1,133 @@ +{ + "$schema": "https://schema.management.azure.com/schemas/2018-05-01/subscriptionDeploymentTemplate.json#", + "contentVersion": "1.0.0.0", + + "metadata": { + "description": "CUDly Cross-Subscription Role Assignments — deploy this in every target Azure subscription that CUDly should manage. It grants the CUDly service principal the permissions needed to query reservation recommendations and purchase Azure Reservations and Savings Plans.", + "consentSurface": "Every grant in this template is confined to the single subscription targeted by 'az deployment sub create'. The scope is derived from subscription().subscriptionId, the deployment target itself, so there is no scope parameter a caller can widen, and deploying to one subscription can never grant access to another. The tenant-wide '/providers/Microsoft.Capacity' scope is deliberately NOT used: an assignment there covers every reservation order in the Azure AD tenant, including subscriptions the customer never onboarded. Purchases authorise against the subscription named in the request body's billingScopeId, so subscription scope is sufficient; this matches terraform/modules/iam/azure/cudly-reservation-role, whose include_capacity_provider_scope flag defaults to false for the same reason. See known-issues.md if a tenant-wide grant is ever genuinely required: it is a separate, manual, explicitly consented step, never a default." + }, + + "parameters": { + "servicePrincipalObjectId": { + "type": "string", + "metadata": { + "description": "Object ID of the CUDly Azure AD service principal. Obtain with: az ad sp show --id --query id -o tsv" + } + }, + "roleAssignmentGuidPrefix": { + "type": "string", + "defaultValue": "[newGuid()]", + "metadata": { + "description": "Base GUID used to derive deterministic role assignment names. Leave at default to auto-generate." + } + } + }, + + "variables": { + "roles": { + "reader": "/providers/Microsoft.Authorization/roleDefinitions/acdd72a7-3385-48ef-bd42-f606fba81ae7", + "costManagementReader": "/providers/Microsoft.Authorization/roleDefinitions/72fafb9e-0641-4937-9268-a91bfd8191a3" + }, + "customRoleName": "[guid(subscription().subscriptionId, 'cudly-reservation-purchaser')]", + "customRoleDefinitionId": "[subscriptionResourceId('Microsoft.Authorization/roleDefinitions', guid(subscription().subscriptionId, 'cudly-reservation-purchaser'))]" + }, + + "resources": [ + { + "type": "Microsoft.Authorization/roleDefinitions", + "apiVersion": "2022-04-01", + "name": "[variables('customRoleName')]", + "properties": { + "roleName": "CUDly Reservation Purchaser (custom)", + "description": "Custom role granting CUDly exactly the Microsoft.Capacity and Microsoft.BillingBenefits actions required by the calculatePrice -> purchase flow. Replaces the built-in Reservation Purchaser, which lacks reservationOrders/write, calculatePrice/action and every Microsoft.BillingBenefits action.", + "type": "CustomRole", + "permissions": [ + { + "actions": [ + "Microsoft.Capacity/register/action", + "Microsoft.Capacity/calculatePrice/action", + "Microsoft.Capacity/catalogs/read", + "Microsoft.Capacity/reservationOrders/read", + "Microsoft.Capacity/reservationOrders/write", + "Microsoft.Capacity/reservationOrders/reservations/read", + "Microsoft.BillingBenefits/register/action", + "Microsoft.BillingBenefits/savingsPlanOrderAliases/write", + "Microsoft.BillingBenefits/savingsPlanOrders/read", + "Microsoft.BillingBenefits/savingsPlanOrders/savingsPlans/read", + "Microsoft.BillingBenefits/savingsPlanOrders/action" + ], + "notActions": [], + "dataActions": [], + "notDataActions": [] + } + ], + "assignableScopes": [ + "[concat('/subscriptions/', subscription().subscriptionId)]" + ] + } + }, + + { + "type": "Microsoft.Authorization/roleAssignments", + "apiVersion": "2022-04-01", + "name": "[guid(parameters('servicePrincipalObjectId'), 'cudlyCustomRole', subscription().subscriptionId)]", + "dependsOn": [ + "[variables('customRoleDefinitionId')]" + ], + "properties": { + "roleDefinitionId": "[variables('customRoleDefinitionId')]", + "principalId": "[parameters('servicePrincipalObjectId')]", + "principalType": "ServicePrincipal", + "description": "CUDly — subscription-scope assignment of custom role; grants calculatePrice/action and reservationOrders/write required by the two-step reservation purchase flow" + } + }, + + { + "type": "Microsoft.Authorization/roleAssignments", + "apiVersion": "2022-04-01", + "name": "[guid(parameters('servicePrincipalObjectId'), 'reader', subscription().subscriptionId)]", + "properties": { + "roleDefinitionId": "[variables('roles').reader]", + "principalId": "[parameters('servicePrincipalObjectId')]", + "principalType": "ServicePrincipal", + "description": "CUDly — enumerate subscription resources and locations" + } + }, + + { + "type": "Microsoft.Authorization/roleAssignments", + "apiVersion": "2022-04-01", + "name": "[guid(parameters('servicePrincipalObjectId'), 'costManagementReader', subscription().subscriptionId)]", + "properties": { + "roleDefinitionId": "[variables('roles').costManagementReader]", + "principalId": "[parameters('servicePrincipalObjectId')]", + "principalType": "ServicePrincipal", + "description": "CUDly — read cost and reservation utilisation data" + } + } + ], + + "outputs": { + "subscriptionId": { + "type": "string", + "value": "[subscription().subscriptionId]", + "metadata": { + "description": "Azure subscription ID to provide when registering this account in CUDly (azure_subscription_id field)." + } + }, + "tenantId": { + "type": "string", + "value": "[subscription().tenantId]", + "metadata": { + "description": "Azure AD tenant ID to provide when registering this account in CUDly (azure_tenant_id field)." + } + }, + "customRoleDefinitionId": { + "type": "string", + "value": "[variables('customRoleDefinitionId')]", + "metadata": { + "description": "Resource ID of the CUDly custom role definition created in this subscription." + } + } + } +} diff --git a/ci_cd_sanity_tests/cmd/azure_sanity/main.go b/ci_cd_sanity_tests/cmd/azure_sanity/main.go new file mode 100644 index 000000000..269f8be6e --- /dev/null +++ b/ci_cd_sanity_tests/cmd/azure_sanity/main.go @@ -0,0 +1,51 @@ +package main + +import ( + "context" + "flag" + "fmt" + "os" + "time" + + "github.com/LeanerCloud/CUDly/ci_cd_sanity_tests/pkg/sanity/azure" +) + +func main() { + var ( + subID = flag.String("subscription-id", "", "Azure subscription ID (or set AZURE_SUBSCRIPTION_ID)") + expectedTenant = flag.String("expected-tenant", "", "Expected Azure tenant ID (optional)") + expectedSub = flag.String("expected-subscription", "", "Expected Azure subscription ID (optional)") + outPath = flag.String("out", "azure_sanity_report.json", "Output JSON report path") + timeoutSec = flag.Int("timeout-sec", 120, "Timeout seconds") + ) + flag.Parse() + + ctx, cancel := context.WithTimeout(context.Background(), time.Duration(*timeoutSec)*time.Second) + + rep, err := azure.Run(ctx, azure.Options{ + SubscriptionID: *subID, + ExpectedTenantID: *expectedTenant, + ExpectedSubID: *expectedSub, + Timeout: time.Duration(*timeoutSec) * time.Second, + }) + if err != nil { + cancel() + fmt.Fprintf(os.Stderr, "azure sanity run failed: %v\n", err) + os.Exit(2) + } + + if err := rep.WriteJSON(*outPath); err != nil { + cancel() + fmt.Fprintf(os.Stderr, "write report failed: %v\n", err) + os.Exit(2) + } + + if rep.HasFailures() { + cancel() + fmt.Fprintf(os.Stderr, "azure sanity: FAIL (see %s)\n", *outPath) + os.Exit(1) + } + + cancel() + fmt.Printf("azure sanity: PASS (see %s)\n", *outPath) +} diff --git a/ci_cd_sanity_tests/cmd/ri-exchange/main.go b/ci_cd_sanity_tests/cmd/ri-exchange/main.go new file mode 100644 index 000000000..cdddb9c54 --- /dev/null +++ b/ci_cd_sanity_tests/cmd/ri-exchange/main.go @@ -0,0 +1,192 @@ +package main + +import ( + "context" + "encoding/json" + "flag" + "fmt" + "math" + "os" + "strings" + "time" + + "github.com/LeanerCloud/CUDly/pkg/exchange" +) + +// Output is the JSON-serialized result written to disk after a quote or +// exchange execution; it captures inputs, the AWS quote, and any error so the +// CI step can archive the artifact and surface a human-readable summary. +type Output struct { + Quote any `json:"quote"` + Mode string `json:"mode"` + Region string `json:"region"` + AccountChk string `json:"expected_account,omitempty"` + TargetOfferingID string `json:"target_offering_id"` + MaxPaymentDueUSD string `json:"max_payment_due_usd,omitempty"` + ExchangeID string `json:"exchange_id,omitempty"` + Error string `json:"error,omitempty"` + ReservedIDs []string `json:"reserved_instance_ids"` + TargetCount int32 `json:"target_count"` +} + +func parseIDs(s string) []string { + var out []string + for _, p := range strings.Split(s, ",") { + p = strings.TrimSpace(p) + if p != "" { + out = append(out, p) + } + } + return out +} + +// validateRequiredFlags checks that the required CLI flags are present and +// that --target-count can safely be narrowed to int32. It exits on the first +// failure so callers do not need to handle the error return. +func validateRequiredFlags(riIDsCSV, targetOffering string, ids []string, targetCount int) { + if len(ids) == 0 { + fmt.Fprintln(os.Stderr, "ERROR: --ri-ids is required (comma-separated)") + os.Exit(2) + } + if strings.TrimSpace(targetOffering) == "" { + fmt.Fprintln(os.Stderr, "ERROR: --target-offering-id is required") + os.Exit(2) + } + if targetCount < 1 || targetCount > math.MaxInt32 { + fmt.Fprintf(os.Stderr, "ERROR: --target-count must be between 1 and %d, got %d\n", math.MaxInt32, targetCount) + os.Exit(2) + } + _ = riIDsCSV // used indirectly via ids +} + +func main() { + var ( + region = flag.String("region", "us-east-1", "AWS region") + expectedAccount = flag.String("expected-account", "", "Safety check: expected AWS account ID (optional)") + + riIDsCSV = flag.String("ri-ids", "", "Comma-separated Convertible Reserved Instance IDs to exchange (required)") + targetOffering = flag.String("target-offering-id", "", "Target RI offering ID (required)") + targetCount = flag.Int("target-count", 1, "Target instance count (default 1)") + + // Execution gating + execute = flag.Bool("execute", false, "Actually execute the exchange (default false = quote only)") + ack = flag.String("ack", "", "Must be 'YES' to execute (safety)") + maxPaymentDue = flag.String("max-payment-due-usd", "", "Max allowed paymentDue from quote (required for execute). Example: 5.00") + + outPath = flag.String("out", "ri_exchange_result.json", "Output JSON path") + timeoutSec = flag.Int("timeout-sec", 180, "Timeout seconds") + ) + flag.Parse() + + ids := parseIDs(*riIDsCSV) + validateRequiredFlags(*riIDsCSV, *targetOffering, ids, *targetCount) + + ctx, cancel := context.WithTimeout(context.Background(), time.Duration(*timeoutSec)*time.Second) + + o := Output{ + Region: *region, + AccountChk: *expectedAccount, + ReservedIDs: ids, + TargetOfferingID: *targetOffering, + TargetCount: int32(*targetCount), // #nosec G115 -- range-validated above (1 <= targetCount <= math.MaxInt32); int->int32 cannot overflow + } + + if !*execute { + o.Mode = "dry-run" + q, err := exchange.GetExchangeQuote(ctx, exchange.ExchangeQuoteRequest{ + Region: *region, + ExpectedAccount: *expectedAccount, + ReservedIDs: ids, + TargetOfferingID: *targetOffering, + TargetCount: int32(*targetCount), // #nosec G115 -- range-validated above (1 <= targetCount <= math.MaxInt32); int->int32 cannot overflow + DryRun: false, // IAMCheckOnly: false = real quote, true = only verify IAM permissions + }) + if err != nil { + o.Error = err.Error() + o.Quote = q + writeOrExit(o, *outPath) + cancel() + fmt.Fprintf(os.Stderr, "quote: FAIL (see %s)\n", *outPath) + os.Exit(1) + } + o.Quote = q + writeOrExit(o, *outPath) + + if !q.IsValidExchange { + cancel() + fmt.Fprintf(os.Stderr, "quote: INVALID (%s) (see %s)\n", q.ValidationFailureReason, *outPath) + os.Exit(1) + } + cancel() + fmt.Printf("quote: OK (valid=%v, paymentDue=%s %s) (see %s)\n", q.IsValidExchange, q.PaymentDueRaw, q.CurrencyCode, *outPath) + os.Exit(0) + } + + // Execute path + o.Mode = "execute" + if strings.TrimSpace(*ack) != "YES" { + o.Error = "refusing to execute: pass --ack YES" + writeOrExit(o, *outPath) + cancel() + fmt.Fprintf(os.Stderr, "execute: REFUSED (see %s)\n", *outPath) + os.Exit(2) + } + if strings.TrimSpace(*maxPaymentDue) == "" { + o.Error = "refusing to execute: --max-payment-due-usd is required as a safety cap" + writeOrExit(o, *outPath) + cancel() + fmt.Fprintf(os.Stderr, "execute: REFUSED (see %s)\n", *outPath) + os.Exit(2) + } + maxRat, err := exchange.ParseDecimalRat(*maxPaymentDue) + if err != nil { + o.Error = err.Error() + writeOrExit(o, *outPath) + cancel() + fmt.Fprintf(os.Stderr, "execute: BAD INPUT (see %s)\n", *outPath) + os.Exit(2) + } + o.MaxPaymentDueUSD = maxRat.FloatString(2) + + exID, q, err := exchange.ExecuteExchange(ctx, exchange.ExchangeExecuteRequest{ + Region: *region, + ExpectedAccount: *expectedAccount, + ReservedIDs: ids, + TargetOfferingID: *targetOffering, + TargetCount: int32(*targetCount), // #nosec G115 -- range-validated above (1 <= targetCount <= math.MaxInt32); int->int32 cannot overflow + MaxPaymentDueUSD: maxRat, + }) + o.Quote = q + if err != nil { + o.Error = err.Error() + writeOrExit(o, *outPath) + cancel() + fmt.Fprintf(os.Stderr, "execute: FAIL (see %s)\n", *outPath) + os.Exit(1) + } + + o.ExchangeID = exID + writeOrExit(o, *outPath) + cancel() + fmt.Printf("execute: OK exchangeId=%s (see %s)\n", exID, *outPath) +} + +func write(v any, path string) error { + b, err := json.MarshalIndent(v, "", " ") + if err != nil { + fmt.Fprintf(os.Stderr, "failed to marshal json for %s: %v\n", path, err) + return err + } + if err := os.WriteFile(path, b, 0600); err != nil { + fmt.Fprintf(os.Stderr, "failed to write %s: %v\n", path, err) + return err + } + return nil +} + +// writeOrExit writes output to path and exits with code 1 if writing fails. +func writeOrExit(v any, path string) { + if err := write(v, path); err != nil { + os.Exit(1) + } +} diff --git a/ci_cd_sanity_tests/cmd/sanity/main.go b/ci_cd_sanity_tests/cmd/sanity/main.go new file mode 100644 index 000000000..cde88c074 --- /dev/null +++ b/ci_cd_sanity_tests/cmd/sanity/main.go @@ -0,0 +1,64 @@ +package main + +import ( + "context" + "flag" + "fmt" + "math" + "os" + "time" + + "github.com/LeanerCloud/CUDly/ci_cd_sanity_tests/pkg/sanity/aws" +) + +// requireInt32Range exits with an error when n is outside [1, math.MaxInt32]. +func requireInt32Range(flagName string, n int) { + if n < 1 || n > (1<<31-1) { + fmt.Fprintf(os.Stderr, "ERROR: %s must be between 1 and math.MaxInt32\n", flagName) + os.Exit(2) + } +} + +func main() { + var ( + region = flag.String("region", "us-east-1", "AWS region for sanity checks") + expectedAccount = flag.String("expected-account", "", "Expected AWS Account ID (optional)") + maxList = flag.Int("max-list", 5, "Max instances to list for EC2 sample (default 5). RDS uses 20..100.") + outPath = flag.String("out", "sanity_report.json", "Output JSON report path") + ) + flag.Parse() + requireInt32Range("--max-list", *maxList) + + if *maxList < 1 || *maxList > math.MaxInt32 { + fmt.Fprintf(os.Stderr, "ERROR: --max-list must be between 1 and %d, got %d\n", math.MaxInt32, *maxList) + os.Exit(2) + } + + ctx, cancel := context.WithTimeout(context.Background(), 3*time.Minute) + + rep, err := aws.Run(ctx, aws.Options{ + Region: *region, + ExpectedAccount: *expectedAccount, + MaxList: int32(*maxList), // #nosec G115 -- range-validated above (1 <= maxList <= math.MaxInt32); int->int32 cannot overflow + }) + if err != nil { + cancel() + fmt.Fprintf(os.Stderr, "sanity run failed: %v\n", err) + os.Exit(2) + } + + if err := rep.WriteJSON(*outPath); err != nil { + cancel() + fmt.Fprintf(os.Stderr, "write report failed: %v\n", err) + os.Exit(2) + } + + if rep.HasFailures() { + cancel() + fmt.Fprintf(os.Stderr, "sanity checks: FAIL (see %s)\n", *outPath) + os.Exit(1) + } + + cancel() + fmt.Printf("sanity checks: PASS (see %s)\n", *outPath) +} diff --git a/ci_cd_sanity_tests/pkg/sanity/aws/aws.go b/ci_cd_sanity_tests/pkg/sanity/aws/aws.go new file mode 100644 index 000000000..ad25852c4 --- /dev/null +++ b/ci_cd_sanity_tests/pkg/sanity/aws/aws.go @@ -0,0 +1,138 @@ +// Package aws implements read-only AWS sanity checks used in CI/CD to verify +// that deploy credentials have sufficient IAM permissions before a real deploy. +package aws + +import ( + "context" + "fmt" + "time" + + "github.com/aws/aws-sdk-go-v2/aws" + "github.com/aws/aws-sdk-go-v2/config" + "github.com/aws/aws-sdk-go-v2/service/ec2" + "github.com/aws/aws-sdk-go-v2/service/rds" + "github.com/aws/aws-sdk-go-v2/service/sts" + + "github.com/LeanerCloud/CUDly/ci_cd_sanity_tests/pkg/sanity/report" +) + +// Options controls which AWS region to target and optional safety assertions +// applied during the sanity run. +type Options struct { + Region string + ExpectedAccount string // optional safety check + MaxList int32 // used for EC2; RDS will clamp to valid range +} + +func checkIdentity(ctx context.Context, cfg aws.Config, expectedAccount string) (map[string]string, error) { + out, err := sts.NewFromConfig(cfg).GetCallerIdentity(ctx, &sts.GetCallerIdentityInput{}) + if err != nil { + return nil, err + } + d := map[string]string{ + "account": aws.ToString(out.Account), + "arn": aws.ToString(out.Arn), + "user_id": aws.ToString(out.UserId), + } + if expectedAccount != "" && aws.ToString(out.Account) != expectedAccount { + return d, fmt.Errorf("unexpected AWS account: got %s want %s", aws.ToString(out.Account), expectedAccount) + } + return d, nil +} + +func checkRegions(ctx context.Context, cfg aws.Config) (map[string]string, error) { + out, err := ec2.NewFromConfig(cfg).DescribeRegions(ctx, &ec2.DescribeRegionsInput{}) + if err != nil { + return nil, err + } + return map[string]string{"regions_count": fmt.Sprintf("%d", len(out.Regions))}, nil +} + +func checkInstances(ctx context.Context, cfg aws.Config, maxList int32) (map[string]string, error) { + if maxList <= 0 { + maxList = 5 + } + out, err := ec2.NewFromConfig(cfg).DescribeInstances(ctx, &ec2.DescribeInstancesInput{ + MaxResults: aws.Int32(maxList), + }) + if err != nil { + return nil, err + } + instances := 0 + for _, r := range out.Reservations { + instances += len(r.Instances) + } + return map[string]string{"instances_seen": fmt.Sprintf("%d", instances)}, nil +} + +func checkRDS(ctx context.Context, cfg aws.Config, maxList int32) (map[string]string, error) { + limit := maxList + if limit < 20 { + limit = 20 + } + if limit > 100 { + limit = 100 + } + out, err := rds.NewFromConfig(cfg).DescribeDBInstances(ctx, &rds.DescribeDBInstancesInput{ + MaxRecords: aws.Int32(limit), + }) + if err != nil { + return nil, err + } + return map[string]string{"db_instances_seen": fmt.Sprintf("%d", len(out.DBInstances))}, nil +} + +// Run executes all AWS sanity checks in sequence and returns a Report +// summarizing each result. It returns an error only for fatal setup failures +// (credential load, SDK init); individual check failures are captured inside +// the Report so callers can surface them all rather than stopping at the first. +func Run(ctx context.Context, opts Options) (*report.Report, error) { + if opts.Region == "" { + opts.Region = "us-east-1" + } + if opts.MaxList <= 0 { + opts.MaxList = 5 + } + + rep := &report.Report{ + RunID: fmt.Sprintf("aws-%d", time.Now().Unix()), + Cloud: "aws", + Mode: "dry-run", + StartedAt: time.Now().UTC(), + } + + cfg, err := config.LoadDefaultConfig(ctx, config.WithRegion(opts.Region)) + if err != nil { + return nil, err + } + + runCheck := func(name string, fn func() (map[string]string, error)) { + start := time.Now().UTC() + details, e := fn() + cr := report.CheckResult{Name: name, StartedAt: start, EndedAt: time.Now().UTC()} + if e == nil { + cr.Status = report.StatusPass + } else { + cr.Status = report.StatusFail + cr.Message = e.Error() + } + cr.Details = details + rep.Add(cr) + } + + runCheck("sts:GetCallerIdentity", func() (map[string]string, error) { + return checkIdentity(ctx, cfg, opts.ExpectedAccount) + }) + runCheck("ec2:DescribeRegions", func() (map[string]string, error) { + return checkRegions(ctx, cfg) + }) + runCheck("ec2:DescribeInstances (sample)", func() (map[string]string, error) { + return checkInstances(ctx, cfg, opts.MaxList) + }) + runCheck("rds:DescribeDBInstances (sample)", func() (map[string]string, error) { + return checkRDS(ctx, cfg, opts.MaxList) + }) + + rep.EndedAt = time.Now().UTC() + return rep, nil +} diff --git a/ci_cd_sanity_tests/pkg/sanity/azure/azure.go b/ci_cd_sanity_tests/pkg/sanity/azure/azure.go new file mode 100644 index 000000000..9351e4cf7 --- /dev/null +++ b/ci_cd_sanity_tests/pkg/sanity/azure/azure.go @@ -0,0 +1,342 @@ +package azure + +import ( + "context" + "encoding/json" + "fmt" + "os" + "strings" + "time" + + "github.com/Azure/azure-sdk-for-go/sdk/azcore" + "github.com/Azure/azure-sdk-for-go/sdk/azidentity" + "github.com/Azure/azure-sdk-for-go/sdk/resourcemanager/compute/armcompute/v5" + "github.com/Azure/azure-sdk-for-go/sdk/resourcemanager/resources/armresources" + "github.com/Azure/azure-sdk-for-go/sdk/resourcemanager/resources/armsubscriptions" + "github.com/LeanerCloud/CUDly/ci_cd_sanity_tests/pkg/sanity/report" +) + +type Options struct { + SubscriptionID string + ExpectedTenantID string // optional + ExpectedSubID string // optional + Timeout time.Duration +} + +// azureSubscriptionInfo holds the subscription/tenant fields extracted from the +// armsubscriptions API response. This mirrors the fields previously parsed from +// "az account show -o json" so that validateAccountExpectations is unchanged. +type azureSubscriptionInfo struct { + ID string + TenantID string + Name string + State string +} + +// azAccountShow is the JSON shape produced by "az account show -o json". It is +// retained only to support the existing validateAccountExpectations function +// which the unit tests exercise via its JSON parsing path. +type azAccountShow struct { + ID string `json:"id"` + TenantID string `json:"tenantId"` + Name string `json:"name"` + State string `json:"state"` + User struct { + Name string `json:"name"` + Type string `json:"type"` + } `json:"user"` +} + +func truncate(s string, maxLen int) string { + if len(s) <= maxLen { + return s + } + return s[:maxLen] + "...(truncated)" +} + +// validateAccountExpectations parses "az account show" JSON output and checks +// that the subscription and tenant IDs match expectations. +func validateAccountExpectations(opts Options, accountOut []byte) report.CheckResult { + start := time.Now().UTC() + check := report.CheckResult{ + Name: "azure:account:expected_checks", + StartedAt: start, + Details: map[string]string{}, + } + + var a azAccountShow + if err := json.Unmarshal(accountOut, &a); err != nil { + check.EndedAt = time.Now().UTC() + check.Status = report.StatusFail + check.Message = fmt.Sprintf("failed to parse az account show JSON: %v", err) + check.Details["raw"] = string(accountOut) + return check + } + + check.EndedAt = time.Now().UTC() + check.Details["id"] = a.ID + check.Details["tenantId"] = a.TenantID + check.Details["name"] = a.Name + check.Details["state"] = a.State + check.Details["user"] = a.User.Name + + var msgs []string + if opts.ExpectedSubID != "" && a.ID != opts.ExpectedSubID { + msgs = append(msgs, fmt.Sprintf("unexpected subscription: got %s want %s", a.ID, opts.ExpectedSubID)) + } + if opts.ExpectedTenantID != "" && a.TenantID != opts.ExpectedTenantID { + msgs = append(msgs, fmt.Sprintf("unexpected tenant: got %s want %s", a.TenantID, opts.ExpectedTenantID)) + } + + if len(msgs) == 0 { + check.Status = report.StatusPass + } else { + check.Status = report.StatusFail + check.Message = strings.Join(msgs, "; ") + } + return check +} + +// encodeAccountJSON serializes azureSubscriptionInfo into the same JSON shape +// that "az account show -o json" produced so that validateAccountExpectations +// can be reused without modification. The struct is composed of plain strings +// so json.Marshal cannot realistically fail; a nil return on the impossible +// error path lets the caller skip the expected-checks step rather than feed +// validateAccountExpectations a partially-encoded payload. +func encodeAccountJSON(info azureSubscriptionInfo) []byte { + a := azAccountShow{ + ID: info.ID, + TenantID: info.TenantID, + Name: info.Name, + State: info.State, + } + b, err := json.Marshal(a) + if err != nil { + return nil + } + return b +} + +// newCheckResult returns a CheckResult with name and timing already set. +func newCheckResult(name string, start time.Time) report.CheckResult { + return report.CheckResult{ + Name: name, + StartedAt: start, + Details: map[string]string{}, + } +} + +// checkPass records a passing check with an optional detail message and +// returns it ready to be added to the report. +func checkPass(cr *report.CheckResult, detail string) report.CheckResult { + cr.EndedAt = time.Now().UTC() + cr.Status = report.StatusPass + if detail != "" { + cr.Details["result"] = detail + } + return *cr +} + +// checkFail records a failing check and returns it. +func checkFail(cr *report.CheckResult, msg string) report.CheckResult { + cr.EndedAt = time.Now().UTC() + cr.Status = report.StatusFail + cr.Message = msg + return *cr +} + +// runGroupListCheck lists up to 10 resource groups in the subscription. +func runGroupListCheck(ctx context.Context, subscriptionID string, cred azcore.TokenCredential) report.CheckResult { + cr := newCheckResult("azure:group:list(sample)", time.Now().UTC()) + cr.Details["subscriptionID"] = subscriptionID + + rgClient, err := armresources.NewResourceGroupsClient(subscriptionID, cred, nil) + if err != nil { + return checkFail(&cr, fmt.Sprintf("failed to create resource-groups client: %v", err)) + } + + pager := rgClient.NewListPager(nil) + var names []string + for pager.More() && len(names) < 10 { + page, pageErr := pager.NextPage(ctx) + if pageErr != nil { + return checkFail(&cr, pageErr.Error()) + } + for _, rg := range page.Value { + if rg.Name != nil && rg.Location != nil { + names = append(names, fmt.Sprintf("%s (%s)", *rg.Name, *rg.Location)) + } + if len(names) >= 10 { + break + } + } + } + cr.Details["result"] = truncate(strings.Join(names, ", "), 2048) + return checkPass(&cr, "") +} + +// resourceGroupFromID extracts the resource group name from an Azure resource ID. +// The ID format is: .../resourceGroups//... +func resourceGroupFromID(id string) string { + parts := strings.Split(id, "/") + for i, p := range parts { + if strings.EqualFold(p, "resourceGroups") && i+1 < len(parts) { + return parts[i+1] + } + } + return "" +} + +// vmSummary returns a short display string for a virtual machine. +func vmSummary(vm *armcompute.VirtualMachine) string { + name := "" + rg := "" + loc := "" + if vm.Name != nil { + name = *vm.Name + } + if vm.Location != nil { + loc = *vm.Location + } + if vm.ID != nil { + rg = resourceGroupFromID(*vm.ID) + } + return fmt.Sprintf("%s (rg:%s loc:%s)", name, rg, loc) +} + +// runVMListCheck lists up to 10 virtual machines in the subscription. +func runVMListCheck(ctx context.Context, subscriptionID string, cred azcore.TokenCredential) report.CheckResult { + cr := newCheckResult("azure:vm:list(sample)", time.Now().UTC()) + cr.Details["subscriptionID"] = subscriptionID + + vmClient, err := armcompute.NewVirtualMachinesClient(subscriptionID, cred, nil) + if err != nil { + return checkFail(&cr, fmt.Sprintf("failed to create virtual-machines client: %v", err)) + } + + pager := vmClient.NewListAllPager(nil) + var items []string + for pager.More() && len(items) < 10 { + page, pageErr := pager.NextPage(ctx) + if pageErr != nil { + return checkFail(&cr, pageErr.Error()) + } + for _, vm := range page.Value { + items = append(items, vmSummary(vm)) + if len(items) >= 10 { + break + } + } + } + cr.Details["result"] = truncate(strings.Join(items, ", "), 2048) + return checkPass(&cr, "") +} + +// runAccountSetCheck verifies that the given subscription ID is reachable. +func runAccountSetCheck(ctx context.Context, subscriptionID string, cred azcore.TokenCredential) report.CheckResult { + cr := newCheckResult("azure:account:set", time.Now().UTC()) + cr.Details["subscriptionID"] = subscriptionID + + subClient, err := armsubscriptions.NewClient(cred, nil) + if err != nil { + return checkFail(&cr, fmt.Sprintf("failed to create subscriptions client: %v", err)) + } + + if _, err := subClient.Get(ctx, subscriptionID, nil); err != nil { + return checkFail(&cr, err.Error()) + } + return checkPass(&cr, "subscription reachable") +} + +// runAccountShowCheck retrieves subscription identity information. +// It returns the check result and the JSON-encoded account info (for use by +// validateAccountExpectations). The JSON is empty on failure. +func runAccountShowCheck(ctx context.Context, subscriptionID string, cred azcore.TokenCredential) (result report.CheckResult, accountJSON []byte) { + cr := newCheckResult("azure:account:show", time.Now().UTC()) + cr.Details["subscriptionID"] = subscriptionID + + subClient, err := armsubscriptions.NewClient(cred, nil) + if err != nil { + return checkFail(&cr, fmt.Sprintf("failed to create subscriptions client: %v", err)), nil + } + + resp, err := subClient.Get(ctx, subscriptionID, nil) + if err != nil { + return checkFail(&cr, err.Error()), nil + } + + sub := resp.Subscription + info := azureSubscriptionInfo{} + if sub.State != nil { + info.State = string(*sub.State) + } + if sub.SubscriptionID != nil { + info.ID = *sub.SubscriptionID + } + if sub.TenantID != nil { + info.TenantID = *sub.TenantID + } + if sub.DisplayName != nil { + info.Name = *sub.DisplayName + } + + cr.Details["id"] = info.ID + cr.Details["tenantId"] = info.TenantID + cr.Details["name"] = info.Name + cr.Details["state"] = info.State + return checkPass(&cr, "account info retrieved"), encodeAccountJSON(info) +} + +// Run performs read-only Azure sanity checks using native SDK calls. +// +// Auth: DefaultAzureCredential is used throughout. In CI this resolves via the +// AZURE_CLIENT_ID / AZURE_TENANT_ID / AZURE_CLIENT_SECRET environment +// variables (service-principal flow). On an operator workstation it falls back +// to AzureCLICredential (i.e. the session established by "az login"), so the +// behavior is identical to the previous CLI-based implementation. +func Run(ctx context.Context, opts Options) (*report.Report, error) { + if opts.SubscriptionID == "" { + opts.SubscriptionID = os.Getenv("AZURE_SUBSCRIPTION_ID") + } + if opts.SubscriptionID == "" { + return nil, fmt.Errorf("missing Azure subscription id: set AZURE_SUBSCRIPTION_ID or pass --subscription-id") + } + if opts.Timeout <= 0 { + opts.Timeout = 2 * time.Minute + } + + rctx, cancel := context.WithTimeout(ctx, opts.Timeout) + defer cancel() + + rep := &report.Report{ + RunID: fmt.Sprintf("azure-%d", time.Now().Unix()), + Cloud: "azure", + Mode: "dry-run", + StartedAt: time.Now().UTC(), + } + + cred, err := azidentity.NewDefaultAzureCredential(nil) + if err != nil { + rep.EndedAt = time.Now().UTC() + return nil, fmt.Errorf("azure: failed to build DefaultAzureCredential: %w", err) + } + + rep.Add(runAccountSetCheck(rctx, opts.SubscriptionID, cred)) + + accountShowResult, accountOut := runAccountShowCheck(rctx, opts.SubscriptionID, cred) + rep.Add(accountShowResult) + + // --- azure:account:expected_checks --- + if (opts.ExpectedSubID != "" || opts.ExpectedTenantID != "") && len(accountOut) > 0 { + rep.Add(validateAccountExpectations(opts, accountOut)) + } + + // --- azure:group:list(sample) --- + rep.Add(runGroupListCheck(rctx, opts.SubscriptionID, cred)) + + // --- azure:vm:list(sample) --- + rep.Add(runVMListCheck(rctx, opts.SubscriptionID, cred)) + + rep.EndedAt = time.Now().UTC() + return rep, nil +} diff --git a/ci_cd_sanity_tests/pkg/sanity/azure/azure_test.go b/ci_cd_sanity_tests/pkg/sanity/azure/azure_test.go new file mode 100644 index 000000000..d89a0d11d --- /dev/null +++ b/ci_cd_sanity_tests/pkg/sanity/azure/azure_test.go @@ -0,0 +1,125 @@ +package azure + +import ( + "encoding/json" + "testing" + + "github.com/LeanerCloud/CUDly/ci_cd_sanity_tests/pkg/sanity/report" + "github.com/stretchr/testify/assert" + "github.com/stretchr/testify/require" +) + +// TestEncodeAccountJSON verifies that encodeAccountJSON produces bytes that are +// accepted by validateAccountExpectations without error, and that the ID / +// TenantID fields survive the round-trip. This guards the SDK->JSON->validate +// path introduced when replacing the "az account show" CLI call. +func TestEncodeAccountJSON(t *testing.T) { + info := azureSubscriptionInfo{ + ID: "aaaabbbb-1111-2222-3333-ccccddddeeee", + TenantID: "ffffgggg-5555-6666-7777-hhhh88889999", + Name: "My Test Sub", + State: "Enabled", + } + + encoded := encodeAccountJSON(info) + require.NotEmpty(t, encoded, "encoded JSON must not be empty") + + // Must parse back as azAccountShow without error. + var parsed azAccountShow + require.NoError(t, json.Unmarshal(encoded, &parsed)) + assert.Equal(t, info.ID, parsed.ID) + assert.Equal(t, info.TenantID, parsed.TenantID) + assert.Equal(t, info.Name, parsed.Name) + assert.Equal(t, info.State, parsed.State) + + // Round-trip through validateAccountExpectations with matching expectations. + opts := Options{ + ExpectedSubID: info.ID, + ExpectedTenantID: info.TenantID, + } + result := validateAccountExpectations(opts, encoded) + assert.Equal(t, report.StatusPass, result.Status, "expected PASS for matching IDs, got: %s", result.Message) +} + +// TestTruncate verifies truncate boundary conditions. +func TestTruncate(t *testing.T) { + assert.Equal(t, "ab", truncate("ab", 5)) + assert.Equal(t, "abcde", truncate("abcde", 5)) + assert.Equal(t, "abcde...(truncated)", truncate("abcdef", 5)) +} + +func accountJSON(id, tenantID string) []byte { + b, err := json.Marshal(azAccountShow{ + ID: id, + TenantID: tenantID, + Name: "My Sub", + State: "Enabled", + }) + if err != nil { + panic(err) + } + return b +} + +func TestValidateAccountExpectations(t *testing.T) { + tests := []struct { + name string + opts Options + accountOut []byte + wantStatus report.Status + wantMsgPart string // substring expected in Message when non-empty + }{ + { + name: "valid json, no expectations", + opts: Options{}, + accountOut: accountJSON("sub-123", "tenant-456"), + wantStatus: report.StatusPass, + }, + { + name: "matching subscription and tenant", + opts: Options{ + ExpectedSubID: "sub-123", + ExpectedTenantID: "tenant-456", + }, + accountOut: accountJSON("sub-123", "tenant-456"), + wantStatus: report.StatusPass, + }, + { + name: "mismatched subscription", + opts: Options{ + ExpectedSubID: "sub-expected", + }, + accountOut: accountJSON("sub-actual", "tenant-456"), + wantStatus: report.StatusFail, + wantMsgPart: "unexpected subscription", + }, + { + name: "mismatched tenant", + opts: Options{ + ExpectedTenantID: "tenant-expected", + }, + accountOut: accountJSON("sub-123", "tenant-actual"), + wantStatus: report.StatusFail, + wantMsgPart: "unexpected tenant", + }, + { + name: "invalid json", + opts: Options{}, + accountOut: []byte(`not valid json`), + wantStatus: report.StatusFail, + wantMsgPart: "failed to parse az account show JSON", + }, + } + + for _, tt := range tests { + t.Run(tt.name, func(t *testing.T) { + result := validateAccountExpectations(tt.opts, tt.accountOut) + assert.Equal(t, tt.wantStatus, result.Status) + if tt.wantMsgPart != "" { + require.Contains(t, result.Message, tt.wantMsgPart) + } + assert.False(t, result.StartedAt.IsZero(), "StartedAt should be set") + assert.False(t, result.EndedAt.IsZero(), "EndedAt should be set") + }) + } +} diff --git a/ci_cd_sanity_tests/pkg/sanity/report/report.go b/ci_cd_sanity_tests/pkg/sanity/report/report.go new file mode 100644 index 000000000..33225ea44 --- /dev/null +++ b/ci_cd_sanity_tests/pkg/sanity/report/report.go @@ -0,0 +1,66 @@ +// Package report defines the structured output types for CI/CD sanity-test runs. +package report + +import ( + "encoding/json" + "os" + "time" +) + +// Status is the outcome of a single sanity check. +type Status string + +const ( + StatusPass Status = "PASS" + StatusFail Status = "FAIL" + StatusSkip Status = "SKIP" +) + +// CheckResult records the outcome of one named sanity check, including timing +// and optional key/value details for post-hoc debugging. +type CheckResult struct { + StartedAt time.Time `json:"started_at"` + EndedAt time.Time `json:"ended_at"` + Details map[string]string `json:"details,omitempty"` + Name string `json:"name"` + Status Status `json:"status"` + Message string `json:"message,omitempty"` +} + +// Report aggregates the results of a full sanity-test run against a single cloud +// provider and is serialized to JSON for upload to the CI artifact store. +type Report struct { + RunID string `json:"run_id"` + Cloud string `json:"cloud"` + Mode string `json:"mode"` // dry-run + StartedAt time.Time `json:"started_at"` + EndedAt time.Time `json:"ended_at"` + Results []CheckResult `json:"results"` +} + +// Add appends a single check result to the report. Not safe for concurrent +// use; callers running checks in goroutines must serialize Add calls. +func (r *Report) Add(res CheckResult) { + r.Results = append(r.Results, res) +} + +// HasFailures reports whether any recorded check ended in StatusFail. Skips +// and passes do not count as failures. +func (r *Report) HasFailures() bool { + for _, rr := range r.Results { + if rr.Status == StatusFail { + return true + } + } + return false +} + +// WriteJSON serializes the report to path with indented JSON, using 0600 +// permissions when creating the CI artifact file. +func (r *Report) WriteJSON(path string) error { + b, err := json.MarshalIndent(r, "", " ") + if err != nil { + return err + } + return os.WriteFile(path, b, 0600) +} diff --git a/cloudformation/stacks/CUDly-CrossAccount/template.yaml b/cloudformation/stacks/CUDly-CrossAccount/template.yaml new file mode 100644 index 000000000..15c88f740 --- /dev/null +++ b/cloudformation/stacks/CUDly-CrossAccount/template.yaml @@ -0,0 +1,180 @@ +AWSTemplateFormatVersion: "2010-09-09" +Description: > + CUDly Cross-Account Role — deploy this in every target AWS account that CUDly + should manage. It creates an IAM role that trusts the CUDly Lambda execution + role in the primary (hub) account to assume it via STS, then grants the + permissions needed to query and purchase Reserved Instances and Savings Plans. + +# ============================================================================= +# Parameters +# ============================================================================= + +Parameters: + PrimaryAccountId: + Type: String + Description: > + 12-digit AWS account ID of the CUDly primary (hub) account. + The Lambda execution role in this account will be trusted. + AllowedPattern: "^[0-9]{12}$" + ConstraintDescription: Must be a 12-digit AWS account ID. + + PrimaryRoleName: + Type: String + Default: CUDly-LambdaRole + Description: > + Exact name of the Lambda execution role in the primary account. + Default matches a primary CUDly stack deployed with the stack name "CUDly". + If your stack was deployed with a different name, set this to the full role + name (without path prefix). For Terraform-managed stacks this is typically + the full role name including the stack suffix. + + ExternalId: + Type: String + NoEcho: true + MinLength: 8 + Description: > + STS ExternalId for defense-in-depth (confused-deputy protection). + Generate with: openssl rand -hex 16 + Record this value — you will need it when registering the account in CUDly. + + RoleName: + Type: String + Default: CUDly + Description: > + Name for the cross-account IAM role created in this account. + Must start with "CUDly" to match the sts:AssumeRole resource scope in the + primary account's Lambda policy (arn:aws:iam::*:role/CUDly*). + +# ============================================================================= +# Resources +# ============================================================================= + +Resources: + + CUDlyRole: + Type: AWS::IAM::Role + Properties: + RoleName: !Ref RoleName + Path: / + AssumeRolePolicyDocument: + Version: "2012-10-17" + Statement: + - Effect: Allow + Principal: + AWS: !Sub "arn:aws:iam::${PrimaryAccountId}:role/${PrimaryRoleName}" + Action: sts:AssumeRole + Condition: + StringEquals: + sts:ExternalId: !Ref ExternalId + + CUDlyPolicy: + Type: AWS::IAM::Policy + Properties: + PolicyName: !Sub "${RoleName}-Policy" + Roles: + - !Ref CUDlyRole + PolicyDocument: + Version: "2012-10-17" + Statement: + # STS — caller identity (used by the /test endpoint) + - Sid: STSCallerIdentity + Effect: Allow + Action: + - sts:GetCallerIdentity + Resource: "*" + + # Cost Explorer — recommendations and utilisation + - Sid: CostExplorer + Effect: Allow + Action: + - ce:GetReservationPurchaseRecommendation + - ce:GetReservationUtilization + - ce:GetReservationCoverage + - ce:GetSavingsPlansPurchaseRecommendation + - ce:GetSavingsPlansUtilization + - ce:GetSavingsPlansCoverage + Resource: "*" + + # EC2 Reserved Instances + - Sid: EC2ReservedInstances + Effect: Allow + Action: + - ec2:DescribeReservedInstancesOfferings + - ec2:DescribeReservedInstances + - ec2:PurchaseReservedInstancesOffering + - ec2:DescribeRegions + - ec2:DescribeInstanceTypeOfferings + Resource: "*" + + # RDS Reserved Instances + - Sid: RDSReservedInstances + Effect: Allow + Action: + - rds:DescribeReservedDBInstancesOfferings + - rds:DescribeReservedDBInstances + - rds:PurchaseReservedDBInstancesOffering + - rds:DescribeDBInstances + Resource: "*" + + # ElastiCache Reserved Nodes + - Sid: ElastiCacheReservedNodes + Effect: Allow + Action: + - elasticache:DescribeReservedCacheNodesOfferings + - elasticache:DescribeReservedCacheNodes + - elasticache:PurchaseReservedCacheNodesOffering + Resource: "*" + + # OpenSearch Reserved Instances + - Sid: OpenSearchReservedInstances + Effect: Allow + Action: + - es:DescribeReservedInstanceOfferings + - es:DescribeReservedInstances + - es:PurchaseReservedInstanceOffering + Resource: "*" + + # Redshift Reserved Nodes + - Sid: RedshiftReservedNodes + Effect: Allow + Action: + - redshift:DescribeReservedNodeOfferings + - redshift:DescribeReservedNodes + - redshift:PurchaseReservedNodeOffering + Resource: "*" + + # MemoryDB Reserved Nodes + - Sid: MemoryDBReservedNodes + Effect: Allow + Action: + - memorydb:DescribeReservedNodesOfferings + - memorydb:DescribeReservedNodes + - memorydb:PurchaseReservedNodesOffering + Resource: "*" + + # Savings Plans + - Sid: SavingsPlans + Effect: Allow + Action: + - savingsplans:DescribeSavingsPlans + - savingsplans:CreateSavingsPlan + - savingsplans:DescribeSavingsPlansOfferingRates + Resource: "*" + +# ============================================================================= +# Outputs +# ============================================================================= + +Outputs: + RoleArn: + Description: > + ARN of the CUDly cross-account role. Provide this value when registering + this account in the CUDly Settings > Accounts UI or via the API + (aws_role_arn field). + Value: !GetAtt CUDlyRole.Arn + Export: + Name: !Sub "${AWS::StackName}-RoleArn" + + RoleName: + Description: Name of the created IAM role. + Value: !Ref CUDlyRole diff --git a/cloudformation/stacks/CUDly/template.yaml b/cloudformation/stacks/CUDly/template.yaml new file mode 100644 index 000000000..d2d5fd717 --- /dev/null +++ b/cloudformation/stacks/CUDly/template.yaml @@ -0,0 +1,1022 @@ +# Copyright (c) 2024 LeanerCloud +# CUDly - Multi-Cloud Commitment & Usage Discount Manager + +AWSTemplateFormatVersion: "2010-09-09" +Description: "CUDly: Automated Reserved Instance and Savings Plans purchase manager for AWS, Azure, and GCP" + +Metadata: + AWS::CloudFormation::Interface: + ParameterGroups: + - Label: + default: Notifications + Parameters: + - EmailAddress + - NotificationDaysBeforePurchase + + - Label: + default: Purchase Defaults + Parameters: + - DefaultTerm + - DefaultPaymentOption + - DefaultCoverage + - DefaultRampSchedule + + - Label: + default: Cost Dashboard + Parameters: + - EnableDashboard + - DashboardDomainName + - DashboardHostedZoneId + + - Label: + default: Deployment Configuration + Parameters: + - LambdaImageUri + - CpuArchitecture + - LambdaMemorySize + - LogRetentionPeriod + - ExecutionFrequency + + ParameterLabels: + EmailAddress: + default: Notification email address + EnableDashboard: + default: Enable the web dashboard? + DashboardDomainName: + default: Custom domain for dashboard (optional) + DashboardHostedZoneId: + default: Route53 Hosted Zone ID for custom domain + +Parameters: + EmailAddress: + Type: String + Default: user@example.com + Description: Email address for purchase notifications and savings reports + + NotificationDaysBeforePurchase: + Type: Number + Default: 3 + MinValue: 1 + MaxValue: 14 + Description: Days before scheduled purchase to send notification email + + DefaultTerm: + Type: Number + Default: 3 + AllowedValues: + - 1 + - 3 + Description: Default commitment term in years + + DefaultPaymentOption: + Type: String + Default: no-upfront + AllowedValues: + - all-upfront + - partial-upfront + - no-upfront + Description: Default payment option for commitments + + DefaultCoverage: + Type: Number + Default: 80 + MinValue: 0 + MaxValue: 100 + Description: Default percentage of recommendations to purchase + + DefaultRampSchedule: + Type: String + Default: immediate + AllowedValues: + - immediate + - weekly-25pct + - monthly-10pct + Description: Default ramp-up schedule for gradual adoption + + EnableDashboard: + Type: String + Default: "true" + AllowedValues: + - "true" + - "false" + Description: Enable the CUDly web dashboard + + DashboardDomainName: + Type: String + Default: "" + Description: Optional custom domain for the dashboard (e.g., cudly.example.com) + + DashboardHostedZoneId: + Type: String + Default: "" + Description: Route53 Hosted Zone ID for DNS validation (required if using custom domain) + + LambdaImageUri: + Type: String + Description: Full Docker image URI for the Lambda function (account.dkr.ecr.region.amazonaws.com/repo:tag) + + CpuArchitecture: + Type: String + Default: x86_64 + AllowedValues: + - x86_64 + - arm64 + Description: CPU architecture for Lambda function + + LambdaMemorySize: + Type: Number + Default: 512 + MinValue: 128 + MaxValue: 3008 + Description: Memory allocated to Lambda function (MB) + + LogRetentionPeriod: + Type: Number + Default: 30 + Description: CloudWatch Logs retention period (days) + + ExecutionFrequency: + Type: String + Default: "rate(1 day)" + Description: How often to check for new recommendations + + CrossAccountTargetAccountIds: + Type: CommaDelimitedList + Default: "" + AllowedPattern: "^$|^[0-9]{12}$" + Description: >- + Comma-separated AWS account IDs this stack itself calls sts:AssumeRole + against, for multi-account plan execution. Empty (the default) creates no + cross-account grant at all. There is no "any account" value: the previous + unconditional grant let the hub assume any CUDly* role in any AWS + account, so a mis-selected account produced a successful purchase in the + wrong account rather than an AccessDenied (#1636). For bastion-mode + accounts list the bastion's account, not the target's, because only the + first hop runs on this stack's identity. UPGRADING AN EXISTING STACK WITHOUT + SETTING THIS REMOVES THE GRANT, and multi-account collection then fails + with AccessDenied until the accounts are listed. + +Conditions: + DeployDashboard: + Fn::Equals: + - Ref: EnableDashboard + - "true" + + HasCustomDomain: + Fn::And: + - Condition: DeployDashboard + - Fn::Not: + - Fn::Equals: + - Ref: DashboardDomainName + - "" + - Fn::Not: + - Fn::Equals: + - Ref: DashboardHostedZoneId + - "" + + Arm64: + Fn::Equals: + - Ref: CpuArchitecture + - arm64 + + # Joining the list back to a string is how a CommaDelimitedList is tested for + # emptiness: the Default of "" arrives as a one-element list holding "", which + # is not equal to an empty list and cannot be compared to one. + HasCrossAccountTargets: + Fn::Not: + - Fn::Equals: + - Fn::Join: + - "" + - Ref: CrossAccountTargetAccountIds + - "" + +Resources: + # ============================================================================= + # DynamoDB Tables + # ============================================================================= + + ConfigTable: + Type: AWS::DynamoDB::Table + Properties: + TableName: !Sub "${AWS::StackName}-Config" + BillingMode: PAY_PER_REQUEST + AttributeDefinitions: + - AttributeName: PK + AttributeType: S + - AttributeName: SK + AttributeType: S + KeySchema: + - AttributeName: PK + KeyType: HASH + - AttributeName: SK + KeyType: RANGE + SSESpecification: + SSEEnabled: true + PointInTimeRecoverySpecification: + PointInTimeRecoveryEnabled: true + Tags: + - Key: Name + Value: !Sub "${AWS::StackName}-Config" + + PurchasePlansTable: + Type: AWS::DynamoDB::Table + Properties: + TableName: !Sub "${AWS::StackName}-PurchasePlans" + BillingMode: PAY_PER_REQUEST + AttributeDefinitions: + - AttributeName: PK + AttributeType: S + - AttributeName: SK + AttributeType: S + KeySchema: + - AttributeName: PK + KeyType: HASH + - AttributeName: SK + KeyType: RANGE + SSESpecification: + SSEEnabled: true + PointInTimeRecoverySpecification: + PointInTimeRecoveryEnabled: true + TimeToLiveSpecification: + AttributeName: ttl + Enabled: true + Tags: + - Key: Name + Value: !Sub "${AWS::StackName}-PurchasePlans" + + PurchaseHistoryTable: + Type: AWS::DynamoDB::Table + Properties: + TableName: !Sub "${AWS::StackName}-PurchaseHistory" + BillingMode: PAY_PER_REQUEST + AttributeDefinitions: + - AttributeName: PK + AttributeType: S + - AttributeName: SK + AttributeType: S + KeySchema: + - AttributeName: PK + KeyType: HASH + - AttributeName: SK + KeyType: RANGE + SSESpecification: + SSEEnabled: true + PointInTimeRecoverySpecification: + PointInTimeRecoveryEnabled: true + Tags: + - Key: Name + Value: !Sub "${AWS::StackName}-PurchaseHistory" + + UsersTable: + Type: AWS::DynamoDB::Table + Properties: + TableName: !Sub "${AWS::StackName}-Users" + BillingMode: PAY_PER_REQUEST + AttributeDefinitions: + - AttributeName: PK + AttributeType: S + - AttributeName: Email + AttributeType: S + - AttributeName: Role + AttributeType: S + - AttributeName: PasswordResetToken + AttributeType: S + KeySchema: + - AttributeName: PK + KeyType: HASH + SSESpecification: + SSEEnabled: true + PointInTimeRecoverySpecification: + PointInTimeRecoveryEnabled: true + GlobalSecondaryIndexes: + - IndexName: EmailIndex + KeySchema: + - AttributeName: Email + KeyType: HASH + Projection: + ProjectionType: ALL + - IndexName: RoleIndex + KeySchema: + - AttributeName: Role + KeyType: HASH + Projection: + ProjectionType: KEYS_ONLY + - IndexName: ResetTokenIndex + KeySchema: + - AttributeName: PasswordResetToken + KeyType: HASH + Projection: + ProjectionType: ALL + Tags: + - Key: Name + Value: !Sub "${AWS::StackName}-Users" + + GroupsTable: + Type: AWS::DynamoDB::Table + Properties: + TableName: !Sub "${AWS::StackName}-Groups" + BillingMode: PAY_PER_REQUEST + AttributeDefinitions: + - AttributeName: PK + AttributeType: S + KeySchema: + - AttributeName: PK + KeyType: HASH + SSESpecification: + SSEEnabled: true + PointInTimeRecoverySpecification: + PointInTimeRecoveryEnabled: true + Tags: + - Key: Name + Value: !Sub "${AWS::StackName}-Groups" + + SessionsTable: + Type: AWS::DynamoDB::Table + Properties: + TableName: !Sub "${AWS::StackName}-Sessions" + BillingMode: PAY_PER_REQUEST + AttributeDefinitions: + - AttributeName: PK + AttributeType: S + - AttributeName: UserID + AttributeType: S + KeySchema: + - AttributeName: PK + KeyType: HASH + SSESpecification: + SSEEnabled: true + PointInTimeRecoverySpecification: + PointInTimeRecoveryEnabled: true + GlobalSecondaryIndexes: + - IndexName: UserIDIndex + KeySchema: + - AttributeName: UserID + KeyType: HASH + Projection: + ProjectionType: KEYS_ONLY + TimeToLiveSpecification: + AttributeName: ExpiresAtEpoch + Enabled: true + Tags: + - Key: Name + Value: !Sub "${AWS::StackName}-Sessions" + + # ============================================================================= + # Secrets Manager + # ============================================================================= + + APIKeySecret: + Type: AWS::SecretsManager::Secret + Properties: + Name: !Sub "${AWS::StackName}-APIKey" + Description: API Key for CUDly Dashboard authentication + GenerateSecretString: + PasswordLength: 32 + ExcludePunctuation: true + + # ============================================================================= + # Note: ECR Repository is created by the deploy CLI before stack creation + # to allow pushing the Lambda image first + # ============================================================================= + + # ============================================================================= + # Lambda Execution Role + # ============================================================================= + + LambdaExecutionRole: + Type: AWS::IAM::Role + Properties: + RoleName: !Sub "${AWS::StackName}-LambdaRole" + AssumeRolePolicyDocument: + Version: "2012-10-17" + Statement: + - Effect: Allow + Principal: + Service: lambda.amazonaws.com + Action: sts:AssumeRole + Path: /lambda/ + + LambdaPolicy: + Type: AWS::IAM::Policy + Properties: + PolicyName: !Sub "${AWS::StackName}-LambdaPolicy" + Roles: + - Ref: LambdaExecutionRole + PolicyDocument: + Version: "2012-10-17" + Statement: + # CloudWatch Logs - scoped to this Lambda's log group + - Sid: CloudWatchLogs + Effect: Allow + Action: + - logs:CreateLogGroup + - logs:CreateLogStream + - logs:PutLogEvents + Resource: + - !Sub "arn:aws:logs:${AWS::Region}:${AWS::AccountId}:log-group:/aws/lambda/${AWS::StackName}-*" + - !Sub "arn:aws:logs:${AWS::Region}:${AWS::AccountId}:log-group:/aws/lambda/${AWS::StackName}-*:*" + + # Cost Explorer for recommendations + - Sid: CostExplorer + Effect: Allow + Action: + - ce:GetCostAndUsage + - ce:GetReservationPurchaseRecommendation + - ce:GetReservationUtilization + - ce:GetReservationCoverage + - ce:GetSavingsPlansPurchaseRecommendation + - ce:GetSavingsPlansUtilization + - ce:GetSavingsPlansCoverage + Resource: "*" + + # RDS Reserved Instances + - Sid: RDSReservedInstances + Effect: Allow + Action: + - rds:DescribeReservedDBInstancesOfferings + - rds:DescribeReservedDBInstances + - rds:PurchaseReservedDBInstancesOffering + - rds:DescribeDBInstances + Resource: "*" + + # ElastiCache Reserved Nodes + - Sid: ElastiCacheReservedNodes + Effect: Allow + Action: + - elasticache:DescribeReservedCacheNodesOfferings + - elasticache:DescribeReservedCacheNodes + - elasticache:PurchaseReservedCacheNodesOffering + Resource: "*" + + # EC2 Reserved Instances + - Sid: EC2ReservedInstances + Effect: Allow + Action: + - ec2:DescribeReservedInstancesOfferings + - ec2:DescribeReservedInstances + - ec2:PurchaseReservedInstancesOffering + - ec2:GetReservedInstancesExchangeQuote + - ec2:AcceptReservedInstancesExchangeQuote + - ec2:DescribeRegions + - ec2:DescribeInstanceTypeOfferings + - ec2:DescribeInstanceTypes + - ec2:CreateReservedInstancesListing + - ec2:DescribeReservedInstancesListings + - ec2:CancelReservedInstancesListing + Resource: "*" + + # Post-purchase tagging of EC2 RIs; the only action here that + # supports resource-level scoping. + - Sid: EC2ReservedInstanceTagging + Effect: Allow + Action: + - ec2:CreateTags + Resource: "arn:aws:ec2:*:*:reserved-instances/*" + + # OpenSearch Reserved Instances + - Sid: OpenSearchReservedInstances + Effect: Allow + Action: + - es:DescribeReservedInstanceOfferings + - es:DescribeReservedInstances + - es:PurchaseReservedInstanceOffering + Resource: "*" + + # Redshift Reserved Nodes + - Sid: RedshiftReservedNodes + Effect: Allow + Action: + - redshift:DescribeReservedNodeOfferings + - redshift:DescribeReservedNodes + - redshift:PurchaseReservedNodeOffering + Resource: "*" + + # MemoryDB Reserved Nodes + - Sid: MemoryDBReservedNodes + Effect: Allow + Action: + - memorydb:DescribeReservedNodesOfferings + - memorydb:DescribeReservedNodes + - memorydb:PurchaseReservedNodesOffering + Resource: "*" + + # Savings Plans + - Sid: SavingsPlans + Effect: Allow + Action: + - savingsplans:DescribeSavingsPlans + - savingsplans:CreateSavingsPlan + - savingsplans:DescribeSavingsPlansOfferings + - savingsplans:DescribeSavingsPlansOfferingRates + Resource: "*" + + # Account discovery + - Sid: AccountDiscovery + Effect: Allow + Action: + - sts:GetCallerIdentity + - organizations:ListAccounts + - organizations:DescribeAccount + - organizations:DescribeOrganization + Resource: "*" + + # Cross-account role assumption for multi-account plans. + # + # aws:ResourceAccount pins the grant to the declared accounts. The + # Resource pattern narrows which role but never whose account: + # arn:aws:iam::*:role/CUDly* matches a CUDly role in any AWS account + # on earth, so before #1636 a mis-selected account produced a + # successful AssumeRole instead of an AccessDenied. sts:ExternalId + # StringLike "*" requires the field to be present and non-empty, at + # parity with the Terraform modules; per-account values are checked + # in internal/credentials/resolver.go. + # + # Dropped entirely rather than widened when no accounts are declared. + - Fn::If: + - HasCrossAccountTargets + - Sid: CrossAccountAssumeRole + Effect: Allow + Action: + - sts:AssumeRole + Resource: "arn:aws:iam::*:role/CUDly*" + Condition: + StringEquals: + aws:ResourceAccount: + Ref: CrossAccountTargetAccountIds + StringLike: + sts:ExternalId: "*" + - Ref: AWS::NoValue + + # DynamoDB access + - Sid: DynamoDBAccess + Effect: Allow + Action: + - dynamodb:GetItem + - dynamodb:PutItem + - dynamodb:DeleteItem + - dynamodb:Query + - dynamodb:Scan + - dynamodb:UpdateItem + - dynamodb:BatchWriteItem + - dynamodb:BatchGetItem + Resource: + - !GetAtt ConfigTable.Arn + - !GetAtt PurchasePlansTable.Arn + - !GetAtt PurchaseHistoryTable.Arn + - !GetAtt UsersTable.Arn + - !Sub "${UsersTable.Arn}/index/*" + - !GetAtt GroupsTable.Arn + - !GetAtt SessionsTable.Arn + - !Sub "${SessionsTable.Arn}/index/*" + + # Secrets Manager + - Sid: SecretsManager + Effect: Allow + Action: + - secretsmanager:GetSecretValue + Resource: + - !Ref APIKeySecret + + # SNS for email notifications + - Sid: SNSPublish + Effect: Allow + Action: + - sns:Publish + Resource: + - !Ref NotificationTopic + + # SES for email (if used) + - Sid: SESEmail + Effect: Allow + Action: + - ses:SendEmail + - ses:SendTemplatedEmail + Resource: "*" + + # ============================================================================= + # Lambda Function + # ============================================================================= + + LambdaFunction: + Type: AWS::Lambda::Function + Properties: + FunctionName: !Sub "${AWS::StackName}-Handler" + PackageType: Image + Architectures: + - !If [Arm64, arm64, x86_64] + Code: + ImageUri: !Ref LambdaImageUri + Description: CUDly - Multi-cloud commitment and discount manager + Environment: + Variables: + # Configuration + CONFIG_TABLE: !Ref ConfigTable + PLANS_TABLE: !Ref PurchasePlansTable + HISTORY_TABLE: !Ref PurchaseHistoryTable + USERS_TABLE: !Ref UsersTable + GROUPS_TABLE: !Ref GroupsTable + SESSIONS_TABLE: !Ref SessionsTable + API_KEY_SECRET_ARN: !Ref APIKeySecret + NOTIFICATION_TOPIC_ARN: !Ref NotificationTopic + # Defaults + DEFAULT_TERM: !Ref DefaultTerm + DEFAULT_PAYMENT_OPTION: !Ref DefaultPaymentOption + DEFAULT_COVERAGE: !Ref DefaultCoverage + DEFAULT_RAMP_SCHEDULE: !Ref DefaultRampSchedule + NOTIFICATION_DAYS_BEFORE: !Ref NotificationDaysBeforePurchase + EMAIL_ADDRESS: !Ref EmailAddress + # Dashboard + ENABLE_DASHBOARD: !Ref EnableDashboard + DASHBOARD_BUCKET: !If [DeployDashboard, !Ref DashboardBucket, ""] + # CORS and Dashboard URL + # Note: When using a custom domain, CORS is restricted to that domain. + # Without a custom domain, CORS defaults to "*" (set at runtime). + # For production, always configure a custom domain for security. + CORS_ALLOWED_ORIGIN: !If [HasCustomDomain, !Sub "https://${DashboardDomainName}", ""] + DASHBOARD_URL: !If [HasCustomDomain, !Sub "https://${DashboardDomainName}", ""] + MemorySize: !Ref LambdaMemorySize + Role: !GetAtt LambdaExecutionRole.Arn + Timeout: 900 + Tags: + - Key: Name + Value: !Sub "${AWS::StackName}-Lambda" + + LambdaLogGroup: + Type: AWS::Logs::LogGroup + DeletionPolicy: Retain + UpdateReplacePolicy: Retain + Properties: + LogGroupName: !Sub "/aws/lambda/${LambdaFunction}" + RetentionInDays: !Ref LogRetentionPeriod + + # ============================================================================= + # Lambda Function URL (for API) + # ============================================================================= + + LambdaFunctionUrl: + Type: AWS::Lambda::Url + Condition: DeployDashboard + Properties: + AuthType: AWS_IAM + TargetFunctionArn: !GetAtt LambdaFunction.Arn + Cors: + # When a custom domain is configured, restrict CORS to that origin. + # Without a custom domain the wildcard is unavoidable for testing, + # but production deployments should always set DashboardDomainName. + # AllowMethods and AllowHeaders match the Terraform lambda module + # (terraform/modules/compute/aws/lambda/main.tf cors block). + AllowOrigins: + - !If [HasCustomDomain, !Sub "https://${DashboardDomainName}", "*"] + AllowMethods: + - GET + - POST + - PUT + - DELETE + AllowHeaders: + - Content-Type + - Authorization + - X-CSRF-Token + - X-Session-Token + + # Permission for CloudFront OAC to invoke Lambda - must be created AFTER the distribution + LambdaFunctionUrlPermission: + Type: AWS::Lambda::Permission + Condition: DeployDashboard + DependsOn: DashboardDistribution + Properties: + Action: lambda:InvokeFunctionUrl + FunctionName: !Ref LambdaFunction + Principal: cloudfront.amazonaws.com + SourceArn: !Sub "arn:${AWS::Partition}:cloudfront::${AWS::AccountId}:distribution/${DashboardDistribution}" + FunctionUrlAuthType: AWS_IAM + + # ============================================================================= + # Scheduled Execution + # ============================================================================= + + ScheduledRule: + Type: AWS::Events::Rule + Properties: + Name: !Sub "${AWS::StackName}-ScheduledExecution" + Description: Scheduled execution for CUDly recommendation collection + ScheduleExpression: !Ref ExecutionFrequency + State: ENABLED + Targets: + - Id: LambdaTarget + Arn: !GetAtt LambdaFunction.Arn + Input: '{"action": "collect_recommendations"}' + + ScheduledRulePermission: + Type: AWS::Lambda::Permission + Properties: + Action: lambda:InvokeFunction + FunctionName: !Ref LambdaFunction + Principal: events.amazonaws.com + SourceArn: !GetAtt ScheduledRule.Arn + + # Daily check for scheduled purchases + PurchaseCheckRule: + Type: AWS::Events::Rule + Properties: + Name: !Sub "${AWS::StackName}-PurchaseCheck" + Description: Daily check for scheduled automated purchases + ScheduleExpression: "rate(1 day)" + State: ENABLED + Targets: + - Id: LambdaTarget + Arn: !GetAtt LambdaFunction.Arn + Input: '{"action": "process_scheduled_purchases"}' + + PurchaseCheckPermission: + Type: AWS::Lambda::Permission + Properties: + Action: lambda:InvokeFunction + FunctionName: !Ref LambdaFunction + Principal: events.amazonaws.com + SourceArn: !GetAtt PurchaseCheckRule.Arn + + # ============================================================================= + # SNS Topic for Notifications + # ============================================================================= + + NotificationTopic: + Type: AWS::SNS::Topic + Properties: + TopicName: !Sub "${AWS::StackName}-Notifications" + KmsMasterKeyId: alias/aws/sns + Subscription: + - Endpoint: !Ref EmailAddress + Protocol: email + + # ============================================================================= + # S3 Bucket for Dashboard + # ============================================================================= + + DashboardBucket: + Type: AWS::S3::Bucket + Condition: DeployDashboard + DeletionPolicy: Delete + Properties: + PublicAccessBlockConfiguration: + BlockPublicAcls: true + BlockPublicPolicy: true + IgnorePublicAcls: true + RestrictPublicBuckets: true + BucketEncryption: + ServerSideEncryptionConfiguration: + - ServerSideEncryptionByDefault: + SSEAlgorithm: AES256 + VersioningConfiguration: + Status: Enabled + + DashboardBucketPolicy: + Type: AWS::S3::BucketPolicy + Condition: DeployDashboard + Properties: + Bucket: !Ref DashboardBucket + PolicyDocument: + Version: "2012-10-17" + Statement: + - Sid: AllowCloudFrontOAC + Effect: Allow + Principal: + Service: cloudfront.amazonaws.com + Action: s3:GetObject + Resource: !Sub "${DashboardBucket.Arn}/*" + Condition: + StringEquals: + "AWS:SourceArn": !Sub "arn:${AWS::Partition}:cloudfront::${AWS::AccountId}:distribution/${DashboardDistribution}" + + # ============================================================================= + # CloudFront Distribution + # ============================================================================= + + DashboardOAC: + Type: AWS::CloudFront::OriginAccessControl + Condition: DeployDashboard + Properties: + OriginAccessControlConfig: + Name: !Sub "${AWS::StackName}-dashboard-oac" + Description: OAC for CUDly Dashboard S3 bucket + OriginAccessControlOriginType: s3 + SigningBehavior: always + SigningProtocol: sigv4 + + # OAC for Lambda Function URL - ensures only CloudFront can invoke Lambda + LambdaOAC: + Type: AWS::CloudFront::OriginAccessControl + Condition: DeployDashboard + Properties: + OriginAccessControlConfig: + Name: !Sub "${AWS::StackName}-lambda-oac" + Description: OAC for CUDly Lambda Function URL + OriginAccessControlOriginType: lambda + SigningBehavior: always + SigningProtocol: sigv4 + + DirectoryIndexFunction: + Type: AWS::CloudFront::Function + Condition: DeployDashboard + Properties: + Name: !Sub "${AWS::StackName}-directory-index" + AutoPublish: true + FunctionConfig: + Comment: Rewrite directory requests to include index.html + Runtime: cloudfront-js-2.0 + FunctionCode: | + function handler(event) { + var request = event.request; + var uri = request.uri; + + // Check if URI ends with / or has no extension (directory request) + if (uri.endsWith('/')) { + request.uri = uri + 'index.html'; + } else if (!uri.includes('.') && !uri.startsWith('/api')) { + // No extension and not an API request - treat as directory + request.uri = uri + '/index.html'; + } + + return request; + } + + DashboardDistribution: + Type: AWS::CloudFront::Distribution + Condition: DeployDashboard + Properties: + DistributionConfig: + Enabled: true + DefaultRootObject: index.html + Aliases: + !If + - HasCustomDomain + - - !Ref DashboardDomainName + - !Ref "AWS::NoValue" + ViewerCertificate: + !If + - HasCustomDomain + - AcmCertificateArn: !Ref DashboardCertificate + SslSupportMethod: sni-only + MinimumProtocolVersion: TLSv1.2_2021 + - CloudFrontDefaultCertificate: true + Origins: + - Id: S3Origin + DomainName: !GetAtt DashboardBucket.RegionalDomainName + S3OriginConfig: + OriginAccessIdentity: "" + OriginAccessControlId: !Ref DashboardOAC + - Id: APIOrigin + DomainName: !Select [2, !Split ["/", !GetAtt LambdaFunctionUrl.FunctionUrl]] + OriginAccessControlId: !Ref LambdaOAC + CustomOriginConfig: + HTTPSPort: 443 + OriginProtocolPolicy: https-only + DefaultCacheBehavior: + TargetOriginId: S3Origin + ViewerProtocolPolicy: redirect-to-https + CachePolicyId: 658327ea-f89d-4fab-a63d-7e88639e58f6 # CachingOptimized + AllowedMethods: + - GET + - HEAD + CachedMethods: + - GET + - HEAD + FunctionAssociations: + - EventType: viewer-request + FunctionARN: !GetAtt DirectoryIndexFunction.FunctionARN + CacheBehaviors: + - PathPattern: "/api/*" + TargetOriginId: APIOrigin + ViewerProtocolPolicy: redirect-to-https + CachePolicyId: 4135ea2d-6df8-44a3-9df3-4b5a84be39ad # CachingDisabled + OriginRequestPolicyId: b689b0a8-53d0-40ab-baf2-68738e2966ac # AllViewerExceptHostHeader + AllowedMethods: + - GET + - HEAD + - OPTIONS + - PUT + - POST + - PATCH + - DELETE + - PathPattern: "/docs*" + TargetOriginId: APIOrigin + ViewerProtocolPolicy: redirect-to-https + CachePolicyId: 658327ea-f89d-4fab-a63d-7e88639e58f6 # CachingOptimized (docs can be cached) + OriginRequestPolicyId: b689b0a8-53d0-40ab-baf2-68738e2966ac # AllViewerExceptHostHeader + AllowedMethods: + - GET + - HEAD + - OPTIONS + PriceClass: PriceClass_100 + CustomErrorResponses: + - ErrorCode: 403 + ResponseCode: 200 + ResponsePagePath: /index.html + - ErrorCode: 404 + ResponseCode: 200 + ResponsePagePath: /index.html + + # ============================================================================= + # Custom Domain Resources (Optional) + # ============================================================================= + + DashboardCertificate: + Type: AWS::CertificateManager::Certificate + Condition: HasCustomDomain + Properties: + DomainName: !Ref DashboardDomainName + ValidationMethod: DNS + DomainValidationOptions: + - DomainName: !Ref DashboardDomainName + HostedZoneId: !Ref DashboardHostedZoneId + + DashboardDNSRecord: + Type: AWS::Route53::RecordSet + Condition: HasCustomDomain + Properties: + HostedZoneId: !Ref DashboardHostedZoneId + Name: !Ref DashboardDomainName + Type: A + AliasTarget: + DNSName: !GetAtt DashboardDistribution.DomainName + HostedZoneId: Z2FDTNDATAQYW2 # CloudFront hosted zone ID + EvaluateTargetHealth: false + +Outputs: + LambdaFunctionArn: + Description: ARN of the CUDly Lambda function + Value: !GetAtt LambdaFunction.Arn + Export: + Name: !Sub "${AWS::StackName}-LambdaArn" + + ConfigTableName: + Description: Name of the configuration DynamoDB table + Value: !Ref ConfigTable + Export: + Name: !Sub "${AWS::StackName}-ConfigTable" + + PurchasePlansTableName: + Description: Name of the purchase plans DynamoDB table + Value: !Ref PurchasePlansTable + Export: + Name: !Sub "${AWS::StackName}-PlansTable" + + PurchaseHistoryTableName: + Description: Name of the purchase history DynamoDB table + Value: !Ref PurchaseHistoryTable + Export: + Name: !Sub "${AWS::StackName}-HistoryTable" + + APIKeySecretArn: + Description: ARN of the API Key secret + Value: !Ref APIKeySecret + Export: + Name: !Sub "${AWS::StackName}-APIKeySecret" + + NotificationTopicArn: + Description: ARN of the SNS notification topic + Value: !Ref NotificationTopic + Export: + Name: !Sub "${AWS::StackName}-NotificationTopic" + + DashboardURL: + Condition: DeployDashboard + Description: URL of the CUDly Dashboard + Value: + !If + - HasCustomDomain + - !Sub "https://${DashboardDomainName}" + - !Sub "https://${DashboardDistribution.DomainName}" + Export: + Name: !Sub "${AWS::StackName}-DashboardURL" + + DashboardBucketName: + Condition: DeployDashboard + Description: Name of the S3 bucket for dashboard static files + Value: !Ref DashboardBucket + Export: + Name: !Sub "${AWS::StackName}-DashboardBucket" + + LambdaFunctionURL: + Condition: DeployDashboard + Description: Lambda Function URL for API access + Value: !GetAtt LambdaFunctionUrl.FunctionUrl + Export: + Name: !Sub "${AWS::StackName}-LambdaFunctionURL" + + UsersTableName: + Description: Name of the users DynamoDB table + Value: !Ref UsersTable + Export: + Name: !Sub "${AWS::StackName}-UsersTable" + + GroupsTableName: + Description: Name of the groups DynamoDB table + Value: !Ref GroupsTable + Export: + Name: !Sub "${AWS::StackName}-GroupsTable" + + SessionsTableName: + Description: Name of the sessions DynamoDB table + Value: !Ref SessionsTable + Export: + Name: !Sub "${AWS::StackName}-SessionsTable" diff --git a/cmd/cleanup-lambda/main.go b/cmd/cleanup-lambda/main.go new file mode 100644 index 000000000..09584c432 --- /dev/null +++ b/cmd/cleanup-lambda/main.go @@ -0,0 +1,106 @@ +package main + +import ( + "context" + "fmt" + "log" + "time" + + "github.com/LeanerCloud/CUDly/internal/database" + "github.com/aws/aws-lambda-go/lambda" +) + +// CleanupEvent represents the input to the cleanup function. +type CleanupEvent struct { + DryRun bool `json:"dryRun,omitempty"` +} + +// CleanupResult represents the cleanup operation results. +type CleanupResult struct { + SessionsDeleted int64 `json:"sessionsDeleted"` + ExecutionsDeleted int64 `json:"executionsDeleted"` + DryRun bool `json:"dryRun"` + Timestamp int64 `json:"timestamp"` +} + +func cleanupExpiredRecords(ctx context.Context, event CleanupEvent) (*CleanupResult, error) { + log.Printf("Starting cleanup job (dryRun=%v)", event.DryRun) + + // A new DB connection is opened per invocation (no connection reuse across warm starts). + // This is intentional: the cleanup Lambda runs infrequently and the simpler, stateless + // design is preferred over the shared-connection pattern used in cmd/lambda/main.go. + db, err := database.OpenFromEnv(ctx) + if err != nil { + return nil, err + } + defer db.Close() + + now := time.Now() + result := &CleanupResult{ + DryRun: event.DryRun, + Timestamp: now.Unix(), + } + + if event.DryRun { + if err := dryRunCount(ctx, db, now, result); err != nil { + return nil, err + } + } else { + if err := deleteExpired(ctx, db, now, result); err != nil { + return nil, err + } + } + + log.Printf("Cleanup job completed: %+v", result) + return result, nil +} + +// dryRunCount counts records that would be deleted without actually deleting them. +func dryRunCount(ctx context.Context, db *database.Connection, now time.Time, result *CleanupResult) error { + if err := db.QueryRow(ctx, "SELECT COUNT(*) FROM sessions WHERE expires_at < $1", now).Scan(&result.SessionsDeleted); err != nil { + return fmt.Errorf("failed to count expired sessions: %w", err) + } + if err := db.QueryRow(ctx, "SELECT COUNT(*) FROM purchase_executions WHERE expires_at < $1", now).Scan(&result.ExecutionsDeleted); err != nil { + return fmt.Errorf("failed to count expired executions: %w", err) + } + log.Printf("DRY RUN: Would delete %d sessions and %d executions", result.SessionsDeleted, result.ExecutionsDeleted) + return nil +} + +// deleteExpired deletes expired sessions and executions in a single transaction. +func deleteExpired(ctx context.Context, db *database.Connection, now time.Time, result *CleanupResult) (err error) { + tx, err := db.Begin(ctx) + if err != nil { + return fmt.Errorf("failed to begin transaction: %w", err) + } + defer func() { + if err != nil { + if rErr := tx.Rollback(ctx); rErr != nil { + log.Printf("rollback failed: %v", rErr) + } + } + }() + + tag, err := tx.Exec(ctx, "DELETE FROM sessions WHERE expires_at < $1", now) + if err != nil { + return fmt.Errorf("failed to cleanup sessions: %w", err) + } + result.SessionsDeleted = tag.RowsAffected() + log.Printf("Deleted %d expired sessions", result.SessionsDeleted) + + tag, err = tx.Exec(ctx, "DELETE FROM purchase_executions WHERE expires_at < $1", now) + if err != nil { + return fmt.Errorf("failed to cleanup executions: %w", err) + } + result.ExecutionsDeleted = tag.RowsAffected() + log.Printf("Deleted %d expired executions", result.ExecutionsDeleted) + + if err = tx.Commit(ctx); err != nil { + return fmt.Errorf("failed to commit cleanup transaction: %w", err) + } + return nil +} + +func main() { + lambda.Start(cleanupExpiredRecords) +} diff --git a/cmd/configure_azure.go b/cmd/configure_azure.go new file mode 100644 index 000000000..900fb345c --- /dev/null +++ b/cmd/configure_azure.go @@ -0,0 +1,602 @@ +package main + +import ( + "bufio" + "context" + "encoding/json" + "errors" + "fmt" + "io" + "log" + "os" + "os/exec" + "regexp" + "strings" + + "time" + + "github.com/Azure/azure-sdk-for-go/sdk/azcore" + "github.com/Azure/azure-sdk-for-go/sdk/azidentity" + "github.com/Azure/azure-sdk-for-go/sdk/resourcemanager/resources/armsubscriptions" + "github.com/aws/aws-sdk-go-v2/aws" + awsconfig "github.com/aws/aws-sdk-go-v2/config" + "github.com/aws/aws-sdk-go-v2/service/secretsmanager" + "github.com/spf13/cobra" + "golang.org/x/term" +) + +// azureUUIDRegex validates Azure UUIDs (subscription IDs, tenant IDs, client IDs). +var azureUUIDRegex = regexp.MustCompile(`^[0-9a-fA-F]{8}-[0-9a-fA-F]{4}-[0-9a-fA-F]{4}-[0-9a-fA-F]{4}-[0-9a-fA-F]{12}$`) + +// validateAzureUUID validates an Azure UUID to prevent command injection. +func validateAzureUUID(uuid, fieldName string) error { + if !azureUUIDRegex.MatchString(uuid) { + return fmt.Errorf("invalid %s format: must be a valid UUID (xxxxxxxx-xxxx-xxxx-xxxx-xxxxxxxxxxxx)", fieldName) + } + return nil +} + +// readTrimmedLine reads one line from reader and returns it with surrounding +// whitespace trimmed. io.EOF is tolerated when data was read — a final +// unterminated line from piped input (e.g. `printf "r" | cudly configure-azure`) +// is still valid input. io.EOF with no data, or any other error, is returned. +func readTrimmedLine(reader *bufio.Reader) (string, error) { + input, err := reader.ReadString('\n') + if err != nil && (!errors.Is(err, io.EOF) || input == "") { + return "", err + } + return strings.TrimSpace(input), nil +} + +// promptRunOrSkipListing asks whether to run an interactive SDK listing step +// (e.g. list subscriptions / projects) or skip straight to entering the ID. +// It returns true to run the listing and false to skip. Skipping is a +// deliberate operator choice for someone who already knows their ID; the +// listing still fails loud WHEN RUN (skip is not a silent fallback on error). +// Empty input or "r"/"run" runs the listing; "s"/"skip" skips it. +func promptRunOrSkipListing(reader *bufio.Reader, what string) (bool, error) { + fmt.Printf("[R]un %s, or [S]kip to enter the ID directly? ", what) + choice, err := readTrimmedLine(reader) + if err != nil { + return false, fmt.Errorf("failed to read choice: %w", err) + } + switch strings.ToLower(choice) { + case "s", "skip": + return false, nil + default: + return true, nil + } +} + +// AzureCredentials holds the Azure Service Principal credentials. +type AzureCredentials struct { + TenantID string `json:"tenant_id"` + ClientID string `json:"client_id"` + ClientSecret string `json:"client_secret"` // #nosec G117 -- operator-supplied credential input read from the user's own Azure service principal; marshaled only to store in AWS Secrets Manager, never a hardcoded secret and never logged (verified) + SubscriptionID string `json:"subscription_id"` +} + +// AzureConfigOptions holds configuration for the Azure config command. +type AzureConfigOptions struct { + StackName string + Profile string + TenantID string + ClientID string + ClientSecret string // #nosec G117 -- operator-supplied credential input from a CLI flag or interactive prompt; never a hardcoded secret and never logged (verified) + SubscriptionID string + Interactive bool + SkipSetup bool +} + +var azureOpts = AzureConfigOptions{} + +var configureAzureCmd = &cobra.Command{ + Use: "configure-azure", + Short: "Configure Azure credentials for CUDly", + Long: `Configure Azure Service Principal credentials for multi-cloud commitment management. + +This command stores your Azure credentials in AWS Secrets Manager for use by CUDly. + +You can provide credentials via flags or interactively: + cudly configure-azure --stack-name my-cudly --tenant-id xxx --client-id xxx --client-secret xxx --subscription-id xxx + cudly configure-azure --stack-name my-cudly --interactive + +To create an Azure Service Principal manually: + az login + az ad sp create-for-rbac --name "CUDly" --role "Reservations Administrator" --scopes /subscriptions/`, + RunE: runConfigureAzure, +} + +func init() { + rootCmd.AddCommand(configureAzureCmd) + + configureAzureCmd.Flags().StringVar(&azureOpts.StackName, "stack-name", "cudly", "CUDly CloudFormation stack name") + configureAzureCmd.Flags().StringVar(&azureOpts.Profile, "profile", "", "AWS profile to use") + configureAzureCmd.Flags().StringVar(&azureOpts.TenantID, "tenant-id", "", "Azure AD Tenant ID") + configureAzureCmd.Flags().StringVar(&azureOpts.ClientID, "client-id", "", "Azure Service Principal Client ID") + configureAzureCmd.Flags().StringVar(&azureOpts.ClientSecret, "client-secret", "", "Azure Service Principal Client Secret") + configureAzureCmd.Flags().StringVar(&azureOpts.SubscriptionID, "subscription-id", "", "Azure Subscription ID") + configureAzureCmd.Flags().BoolVarP(&azureOpts.Interactive, "interactive", "i", false, "Prompt for credentials interactively") + configureAzureCmd.Flags().BoolVar(&azureOpts.SkipSetup, "skip-setup", false, "Skip Azure CLI setup commands (az login, create service principal)") +} + +// validateAzureCredentialFields checks that all required fields are non-empty +// and that UUID-typed fields have the correct format. +func validateAzureCredentialFields(creds AzureCredentials) error { + if creds.TenantID == "" || creds.ClientID == "" || creds.ClientSecret == "" || creds.SubscriptionID == "" { + return fmt.Errorf("all credentials are required: tenant-id, client-id, client-secret, subscription-id") + } + if err := validateAzureUUID(creds.TenantID, "Tenant ID"); err != nil { + return err + } + if err := validateAzureUUID(creds.ClientID, "Client ID"); err != nil { + return err + } + return validateAzureUUID(creds.SubscriptionID, "Subscription ID") +} + +// storeAzureCredentials stores Azure credentials in the secrets store. +func storeAzureCredentials(ctx context.Context, store SecretsStore, stackName string, creds AzureCredentials) error { + if err := validateAzureCredentialFields(creds); err != nil { + return err + } + + // Build expected secret name pattern + secretName := fmt.Sprintf("%s-AzureCredentials", stackName) + + // Try to find the actual secret ARN by listing secrets + arns, err := store.ListSecrets(ctx, secretName) + + // Use the ARN if found, otherwise use the name (will fail if secret doesn't exist) + secretID := secretName + if err == nil && len(arns) > 0 { + secretID = arns[0] + } + + // Marshal credentials to JSON + credJSON, err := json.Marshal(creds) // #nosec G117 -- AzureCredentials marshaled intentionally for Secrets Manager storage + if err != nil { + return fmt.Errorf("failed to marshal credentials: %w", err) + } + + // Store credentials in Secrets Manager + err = store.UpdateSecret(ctx, secretID, string(credJSON)) + if err != nil { + return fmt.Errorf("failed to store credentials in Secrets Manager: %w", err) + } + + return nil +} + +func runConfigureAzure(cmd *cobra.Command, args []string) error { + ctx := context.Background() + reader := bufio.NewReader(os.Stdin) + + fmt.Println("Configure Azure Service Principal credentials for CUDly") + fmt.Println("========================================================") + fmt.Println() + + // Run Azure CLI setup if not skipped + if !azureOpts.SkipSetup { + if err := runAzureSetupCommands(ctx, reader); err != nil { + return err + } + } + + cfg, err := loadAWSConfigForAzure(ctx) + if err != nil { + return err + } + + creds, err := collectAzureCredentials(reader) + if err != nil { + return err + } + + smClient := secretsmanager.NewFromConfig(cfg) + store := NewAWSSecretsStore(smClient) + + if err := storeAzureCredentials(ctx, store, azureOpts.StackName, creds); err != nil { + return err + } + + // Zero out sensitive data from memory + azureOpts.ClientSecret = "" + creds.ClientSecret = "" + + log.Printf("Azure credentials stored successfully in Secrets Manager") + fmt.Println("\nAzure configuration complete!") + fmt.Println("CUDly can now manage Azure Reserved Instances and Savings Plans.") + + return nil +} + +// loadAWSConfigForAzure loads AWS configuration with optional profile. +func loadAWSConfigForAzure(ctx context.Context) (aws.Config, error) { + var opts []func(*awsconfig.LoadOptions) error + if azureOpts.Profile != "" { + opts = append(opts, awsconfig.WithSharedConfigProfile(azureOpts.Profile)) + } + + cfg, err := awsconfig.LoadDefaultConfig(ctx, opts...) + if err != nil { + return aws.Config{}, fmt.Errorf("failed to load AWS config: %w", err) + } + + return cfg, nil +} + +// collectAzureCredentials collects Azure credentials interactively or from flags. +func collectAzureCredentials(reader *bufio.Reader) (AzureCredentials, error) { + creds := AzureCredentials{ + TenantID: azureOpts.TenantID, + ClientID: azureOpts.ClientID, + ClientSecret: azureOpts.ClientSecret, + SubscriptionID: azureOpts.SubscriptionID, + } + + needsInput := azureOpts.Interactive || (creds.TenantID == "" || creds.ClientID == "" || creds.ClientSecret == "" || creds.SubscriptionID == "") + if !needsInput { + return creds, nil + } + + fmt.Println("\nEnter the credentials from the Service Principal output above:") + fmt.Println() + + if err := promptForAzureCredentialFields(reader, &creds); err != nil { + return AzureCredentials{}, err + } + + return creds, nil +} + +// promptForAzureCredentialFields prompts for missing credential fields. +func promptForAzureCredentialFields(reader *bufio.Reader, creds *AzureCredentials) error { + if creds.TenantID == "" { + fmt.Print("Azure Tenant ID: ") + input, err := readTrimmedLine(reader) + if err != nil { + return fmt.Errorf("failed to read tenant ID: %w", err) + } + creds.TenantID = input + } + + if creds.ClientID == "" { + fmt.Print("Client ID (appId): ") + input, err := readTrimmedLine(reader) + if err != nil { + return fmt.Errorf("failed to read client ID: %w", err) + } + creds.ClientID = input + } + + if creds.ClientSecret == "" { + fmt.Print("Client Secret (password): ") + // int(os.Stdin.Fd()) is portable: syscall.Stdin is an int on Unix but a + // Handle on Windows, so passing it to term.ReadPassword (which takes an + // int) breaks GOOS=windows builds. + secret, err := term.ReadPassword(int(os.Stdin.Fd())) // #nosec G115 -- OS file descriptors fit in int on every supported platform + if err != nil { + return fmt.Errorf("failed to read secret: %w", err) + } + fmt.Println() + creds.ClientSecret = string(secret) + } + + if creds.SubscriptionID == "" { + fmt.Print("Subscription ID: ") + input, err := readTrimmedLine(reader) + if err != nil { + return fmt.Errorf("failed to read subscription ID: %w", err) + } + creds.SubscriptionID = input + } + + return nil +} + +// newAzureWizardCredential builds the credential used by the interactive Azure +// setup wizard. It binds explicitly to the Azure CLI session (the "az login" +// the operator runs in Step 1) via AzureCLICredential rather than +// DefaultAzureCredential, whose chain prioritizes environment / workload / +// managed-identity credentials and could otherwise resolve to a different +// principal than the one the operator just signed in as. +func newAzureWizardCredential() (azcore.TokenCredential, error) { + cred, err := azidentity.NewAzureCLICredential(nil) + if err != nil { + return nil, fmt.Errorf("failed to build Azure CLI credential: %w\n"+ + "Ensure you are authenticated: run 'az login' (Step 1) before continuing", err) + } + return cred, nil +} + +// listAzureSubscriptions retrieves the operator's subscriptions via the ARM +// Subscriptions SDK and prints them in a table matching "az account list" +// output. It uses the Azure CLI credential so the listing matches the +// operator's active "az login" session. +func listAzureSubscriptions(ctx context.Context) error { + cred, err := newAzureWizardCredential() + if err != nil { + return err + } + + client, err := armsubscriptions.NewClient(cred, nil) + if err != nil { + return fmt.Errorf("failed to create subscriptions client: %w", err) + } + + fmt.Printf("%-40s %-38s %s\n", "Name", "SubscriptionId", "State") + fmt.Println(strings.Repeat("-", 95)) + + pager := client.NewListPager(nil) + for pager.More() { + page, pageErr := pager.NextPage(ctx) + if pageErr != nil { + return fmt.Errorf("failed to list subscriptions: %w", pageErr) + } + for _, sub := range page.Value { + name := "" + subID := "" + state := "" + if sub.DisplayName != nil { + name = *sub.DisplayName + } + if sub.SubscriptionID != nil { + subID = *sub.SubscriptionID + } + if sub.State != nil { + state = string(*sub.State) + } + fmt.Printf("%-40s %-38s %s\n", name, subID, state) + } + } + return nil +} + +// runAzureSetupCommands guides the operator through the Azure setup wizard. +// +// Step 1 (az login): performed via the Azure CLI. "az login" launches an +// interactive browser-based OAuth flow that cannot be replicated through the +// SDK on behalf of a human operator who does not yet have a credential. This +// is the only CLI call retained in this wizard. +// +// Step 2 (list subscriptions): performed via the ARM Subscriptions SDK using +// the Azure CLI credential, which reuses the session that "az login" just +// established. Fails loud if the SDK cannot authenticate (no CLI fallback). +// +// Step 3 (create service principal): performed via the Microsoft Graph SDK +// (application + service principal + password credential) and armauthorization +// (resolve the "Reservations Administrator" role definition and assign it at +// subscription scope). This is the create-for-rbac equivalent and fails loud +// on any SDK error. +func runAzureSetupCommands(ctx context.Context, reader *bufio.Reader) error { + if err := azureStepLogin(reader); err != nil { + return err + } + + subscriptionID, err := azureStepListSubscriptions(ctx, reader) + if err != nil { + return err + } + + return azureStepCreateServicePrincipal(ctx, reader, subscriptionID) +} + +// azureStepLogin prompts to run "az login". +func azureStepLogin(reader *bufio.Reader) error { + fmt.Println("Step 1: Azure Login") + fmt.Println("-------------------") + fmt.Println("This will open a browser window for Azure authentication.") + fmt.Println() + return promptAndRunExplicitCommand(reader, "Azure Login", "az login", "az", "login") +} + +// listAzureSubscriptionsWithTimeout runs the SDK subscription listing under a +// bounded timeout so a hung ARM call cannot stall the wizard. +func listAzureSubscriptionsWithTimeout(ctx context.Context) error { + listCtx, cancel := context.WithTimeout(ctx, 30*time.Second) + defer cancel() + return listAzureSubscriptions(listCtx) +} + +// azureStepListSubscriptions optionally lists subscriptions via SDK and prompts +// the operator to enter their subscription ID. The listing is behind a +// [R]un/[S]kip prompt so an operator who already knows their subscription ID +// can proceed even if the SDK listing would fail; when RUN it fails loud (no +// CLI fallback), instructing the operator to run "az login" first. +func azureStepListSubscriptions(ctx context.Context, reader *bufio.Reader) (string, error) { + fmt.Println() + fmt.Println("Step 2: Get Subscription ID") + fmt.Println("---------------------------") + + run, err := promptRunOrSkipListing(reader, "the Azure subscription listing (via SDK)") + if err != nil { + return "", err + } + if run { + fmt.Println("Listing your Azure subscriptions via SDK (Azure CLI credential)...") + fmt.Println() + if err = listAzureSubscriptionsWithTimeout(ctx); err != nil { + return "", fmt.Errorf("failed to list Azure subscriptions via SDK: %w\n"+ + "Ensure you are authenticated: run 'az login' (Step 1) before continuing", err) + } + fmt.Println() + } + + fmt.Print("Enter your Subscription ID: ") + subscriptionID, err := readTrimmedLine(reader) + if err != nil { + return "", fmt.Errorf("failed to read subscription ID: %w", err) + } + + if subscriptionID == "" { + return "", fmt.Errorf("subscription ID is required") + } + if err := validateAzureUUID(subscriptionID, "Subscription ID"); err != nil { + return "", err + } + return subscriptionID, nil +} + +// resolveAzureTenantID looks up the tenant ID for the given subscription via +// the ARM Subscriptions SDK, using the Azure CLI ("az login") credential. +func resolveAzureTenantID(ctx context.Context, subscriptionID string) (string, error) { + cred, err := newAzureWizardCredential() + if err != nil { + return "", err + } + client, err := armsubscriptions.NewClient(cred, nil) + if err != nil { + return "", fmt.Errorf("failed to create subscriptions client: %w", err) + } + resp, err := client.Get(ctx, subscriptionID, nil) + if err != nil { + return "", fmt.Errorf("failed to get subscription %s: %w", subscriptionID, err) + } + if resp.TenantID == nil || *resp.TenantID == "" { + return "", fmt.Errorf("subscription %s returned no tenant ID", subscriptionID) + } + return *resp.TenantID, nil +} + +// azureStepCreateServicePrincipal creates the service principal via the +// Microsoft Graph + armauthorization SDKs (the create-for-rbac equivalent) and +// prints the resulting credential material. It fails loud: any SDK error is +// returned rather than silently falling back to the CLI. +func azureStepCreateServicePrincipal(ctx context.Context, reader *bufio.Reader, subscriptionID string) error { + fmt.Println() + fmt.Println("Step 3: Create Service Principal") + fmt.Println("---------------------------------") + fmt.Println("This creates an Azure Service Principal with the") + fmt.Printf("%q role at subscription scope, via the Microsoft Graph SDK.\n", azureSPRoleName) + fmt.Println() + fmt.Printf("Create service principal %q with role %q at /subscriptions/%s?\n", azureSPName, azureSPRoleName, subscriptionID) + fmt.Printf("[R]un, [S]kip? ") + + choice, err := readTrimmedLine(reader) + if err != nil { + return fmt.Errorf("failed to read service-principal choice: %w", err) + } + choice = strings.ToLower(choice) + if choice != "r" && choice != "run" && choice != "" { + fmt.Println("Skipping Create Service Principal") + fmt.Println() + fmt.Println("Provide the appId, client secret and tenant ID for an existing") + fmt.Println("service principal in the next step.") + fmt.Println() + return nil + } + + // The step budget must exceed roleAssignRetryBudget (3 min) plus the + // pre-assignment overhead (tenant resolution + the application / password / + // service-principal / role-definition Graph calls) so the PrincipalNotFound + // retry loop in AssignRole gets its full propagation budget. A tighter + // budget would cut the retry short via ctx cancellation, then the rollback + // would delete the just-created application and reset the AAD replication + // clock on every re-run. See roleAssignRetryBudget in configure_azure_sp.go. + spCtx, cancel := context.WithTimeout(ctx, 6*time.Minute) + defer cancel() + + tenantID, err := resolveAzureTenantID(spCtx, subscriptionID) + if err != nil { + return err + } + + provisioner, err := newGraphSPProvisioner(subscriptionID) + if err != nil { + return fmt.Errorf("failed to initialize Azure SDK clients: %w\n"+ + "Ensure you are authenticated: run 'az login' (Step 1) before continuing", err) + } + + fmt.Println() + fmt.Println(strings.Repeat("-", 60)) + result, err := createAzureServicePrincipal(spCtx, provisioner, subscriptionID, tenantID) + if err != nil { + return err + } + fmt.Println(strings.Repeat("-", 60)) + + printAzureSPResult(result, subscriptionID) + return nil +} + +// printAzureSPResult prints the credential material in the same shape that +// "az ad sp create-for-rbac" prints, so the operator can feed it into the +// credential collection step that follows. +func printAzureSPResult(result azureSPResult, subscriptionID string) { + fmt.Println() + fmt.Println("Service principal created. Credential material:") + fmt.Println() + fmt.Printf(" appId (Client ID): %s\n", result.AppID) + fmt.Printf(" password (Client Secret): %s\n", result.ClientSecret) + fmt.Printf(" tenant (Tenant ID): %s\n", result.TenantID) + fmt.Println() + fmt.Println("IMPORTANT: copy the client secret now -- it cannot be retrieved later.") + fmt.Println("You'll enter these values in the next step:") + fmt.Println(" - appId -> Client ID") + fmt.Println(" - password -> Client Secret") + fmt.Println(" - tenant -> Tenant ID") + fmt.Printf(" - Subscription ID: %s\n", subscriptionID) + fmt.Println() +} + +// promptAndRunExplicitCommand shows a command and asks to run or skip. +// Takes explicit program and args to avoid command injection via string splitting. +func promptAndRunExplicitCommand(reader *bufio.Reader, name, displayCmd, program string, args ...string) error { + fmt.Printf("Command: %s\n", displayCmd) + fmt.Println() + fmt.Printf("[R]un, [S]kip? ") + + choice, err := readTrimmedLine(reader) + if err != nil { + return fmt.Errorf("failed to read choice: %w", err) + } + choice = strings.ToLower(choice) + + switch choice { + case "r", "run", "": + return executeExplicitCommand(reader, displayCmd, program, args...) + case "s", "skip": + fmt.Printf("Skipping %s\n", name) + return nil + default: + fmt.Printf("Unknown option '%s', skipping\n", choice) + return nil + } +} + +// executeExplicitCommand runs a command with explicit program and arguments. +// It is used only for the interactive "az login" auth bootstrap (Step 1), +// which has no SDK equivalent that preserves the cached-credential UX. +// The caller's reader is threaded through to the retry prompt so all input +// is consumed from one consistent buffered stream (a fresh +// bufio.NewReader(os.Stdin) here would drop input already buffered by the +// caller's reader, breaking piped input after earlier prompts). +func executeExplicitCommand(reader *bufio.Reader, displayCmd, program string, args ...string) error { + fmt.Println() + fmt.Printf("Executing: %s\n", displayCmd) + fmt.Println(strings.Repeat("-", 60)) + + // #nosec G204 -- interactive operator auth (az login): program and args are hardcoded literals from the caller (runAzureSetupCommands passes "az","login"), no shell, not attacker-controlled + cmd := exec.Command(program, args...) + cmd.Stdout = os.Stdout + cmd.Stderr = os.Stderr + cmd.Stdin = os.Stdin + + err := cmd.Run() + fmt.Println(strings.Repeat("-", 60)) + + if err != nil { + fmt.Printf("Command failed: %v\n", err) + fmt.Print("Continue anyway? [y/N]: ") + response, readErr := readTrimmedLine(reader) + if readErr != nil { + return fmt.Errorf("failed to read response: %w", readErr) + } + if !strings.EqualFold(response, "y") { + return fmt.Errorf("command failed: %w", err) + } + } + + return nil +} diff --git a/cmd/configure_azure_sp.go b/cmd/configure_azure_sp.go new file mode 100644 index 000000000..6005911c1 --- /dev/null +++ b/cmd/configure_azure_sp.go @@ -0,0 +1,345 @@ +package main + +import ( + "context" + "errors" + "fmt" + "strings" + "time" + + "github.com/Azure/azure-sdk-for-go/sdk/azcore" + armauthorization "github.com/Azure/azure-sdk-for-go/sdk/resourcemanager/authorization/armauthorization/v2" + "github.com/google/uuid" + msgraphsdk "github.com/microsoftgraph/msgraph-sdk-go" + "github.com/microsoftgraph/msgraph-sdk-go/applications" + graphmodels "github.com/microsoftgraph/msgraph-sdk-go/models" +) + +// roleAssignRetryInitial is the first back-off interval when retrying a role +// assignment that fails with PrincipalNotFound. Each subsequent interval is +// doubled up to roleAssignRetryMax. +const ( + roleAssignRetryInitial = 5 * time.Second + roleAssignRetryMax = 30 * time.Second + // roleAssignRetryBudget is the ceiling on total retry time. Azure AD + // replication is usually complete within seconds but can take up to ~10 + // minutes in the worst case (see feedback_tf_depends_on_rbac.md). Three + // minutes covers the large majority of propagation windows without + // keeping the operator waiting too long. + roleAssignRetryBudget = 3 * time.Minute +) + +// roleAssigner is the minimal subset of *armauthorization.RoleAssignmentsClient +// used by graphSPProvisioner. The interface exists solely to allow the retry +// logic in AssignRole to be exercised in unit tests without hitting Azure. +type roleAssigner interface { + Create( + ctx context.Context, + scope, roleAssignmentName string, + parameters armauthorization.RoleAssignmentCreateParameters, + options *armauthorization.RoleAssignmentsClientCreateOptions, + ) (armauthorization.RoleAssignmentsClientCreateResponse, error) +} + +// isPrincipalNotFoundErr reports whether err is an Azure ARM +// PrincipalNotFound / ServicePrincipalNotFound response -- the transient +// condition that occurs when a newly-created Entra ID service principal has +// not yet replicated to the ARM region handling the role-assignment request. +// It covers the canonical error codes as well as the message-body form +// returned by some older API versions. +func isPrincipalNotFoundErr(err error) bool { + var respErr *azcore.ResponseError + if !errors.As(err, &respErr) { + return false + } + switch respErr.ErrorCode { + case "PrincipalNotFound", "ServicePrincipalNotFound": + return true + } + // Older ARM API versions surface the condition as HTTP 400 without a + // distinct error code; instead the response body contains the canonical + // phrase "does not exist in the directory". + return strings.Contains(respErr.Error(), "does not exist in the directory") +} + +// azureSPName is the Azure AD application / service principal display name. +// It matches the name the previous "az ad sp create-for-rbac --name CUDly" +// call used so the resulting identity is interchangeable. +const azureSPName = "CUDly" + +// azureSPRoleName is the RBAC role assigned to the service principal at +// subscription scope, matching the previous create-for-rbac invocation. +const azureSPRoleName = "Reservations Administrator" + +// azureSPResult holds the credential material produced by creating the service +// principal. It mirrors the appId/password/tenant fields that +// "az ad sp create-for-rbac" prints, which the operator feeds into the +// subsequent configure step. +type azureSPResult struct { + AppID string // application (client) ID + ClientSecret string // #nosec G117 -- generated SP password (client secret): SDK-produced, surfaced once to the interactive operator on stdout to copy (mirrors az ad sp create-for-rbac); never a hardcoded secret and never persisted to logs (verified) + TenantID string // Azure AD tenant ID +} + +// azureSPProvisioner abstracts the cloud operations needed to create a service +// principal and grant it the Reservations Administrator role. It exists so the +// orchestration logic can be unit-tested with a mock without depending on the +// concrete Graph / armauthorization client types (which are not interfaces). +type azureSPProvisioner interface { + // CreateApplication creates an AAD application registration with the given + // display name and returns its object ID and application (client) ID. + CreateApplication(ctx context.Context, displayName string) (objectID, appID string, err error) + // AddPassword adds a password credential to the application identified by + // objectID and returns the generated secret text. + AddPassword(ctx context.Context, objectID string) (secretText string, err error) + // CreateServicePrincipal creates a service principal for the given + // application (client) ID and returns the service principal object ID + // (the principal ID used for role assignment). + CreateServicePrincipal(ctx context.Context, appID string) (principalID string, err error) + // ResolveRoleDefinitionID resolves the role definition ID for the given + // role display name at the given scope. + ResolveRoleDefinitionID(ctx context.Context, scope, roleName string) (roleDefinitionID string, err error) + // AssignRole creates a role assignment binding principalID to + // roleDefinitionID at the given scope. + AssignRole(ctx context.Context, scope, principalID, roleDefinitionID string) error + // DeleteApplication deletes the application registration identified by + // objectID. Deleting the application also removes its password credentials + // and the service principal created from it, so it serves as the + // compensating action for a partially completed creation flow. + DeleteApplication(ctx context.Context, objectID string) error +} + +// createAzureServicePrincipal performs the full create-for-rbac equivalent: +// it creates the application + password + service principal, resolves the +// "Reservations Administrator" role at subscription scope, assigns it, and +// returns the credential material. +// +// subscriptionID must already be validated (UUID) and tenantID resolved by the +// caller. The behavior matches: +// +// az ad sp create-for-rbac --name CUDly \ +// --role "Reservations Administrator" \ +// --scopes /subscriptions/ +func createAzureServicePrincipal(ctx context.Context, p azureSPProvisioner, subscriptionID, tenantID string) (azureSPResult, error) { + scope := fmt.Sprintf("/subscriptions/%s", subscriptionID) + + objectID, appID, err := p.CreateApplication(ctx, azureSPName) + if err != nil { + return azureSPResult{}, fmt.Errorf("failed to create application registration: %w", err) + } + + // rollback deletes the just-created application (which cascades to its + // password credentials and the derived service principal) so a failure in + // a later step does not orphan Azure AD objects. The cleanup runs on a + // fresh context in case the parent is already canceled/expired. If the + // cleanup itself fails, the operator is told exactly what to delete by hand. + rollback := func(cause error) (azureSPResult, error) { + cleanupCtx, cancel := context.WithTimeout(context.Background(), 1*time.Minute) + defer cancel() + if delErr := p.DeleteApplication(cleanupCtx, objectID); delErr != nil { + return azureSPResult{}, fmt.Errorf("%w; additionally failed to roll back application %q (appId %s) -- delete it manually: %w", + cause, objectID, appID, delErr) + } + return azureSPResult{}, fmt.Errorf("%w (rolled back: deleted application %q)", cause, objectID) + } + + secret, err := p.AddPassword(ctx, objectID) + if err != nil { + return rollback(fmt.Errorf("failed to add password credential: %w", err)) + } + + principalID, err := p.CreateServicePrincipal(ctx, appID) + if err != nil { + return rollback(fmt.Errorf("failed to create service principal: %w", err)) + } + + roleDefID, err := p.ResolveRoleDefinitionID(ctx, scope, azureSPRoleName) + if err != nil { + return rollback(fmt.Errorf("failed to resolve %q role definition: %w", azureSPRoleName, err)) + } + + if err := p.AssignRole(ctx, scope, principalID, roleDefID); err != nil { + return rollback(fmt.Errorf("failed to assign %q role at %s: %w", azureSPRoleName, scope, err)) + } + + return azureSPResult{ + AppID: appID, + ClientSecret: secret, + TenantID: tenantID, + }, nil +} + +// graphSPProvisioner is the production azureSPProvisioner backed by the +// Microsoft Graph SDK (application + service principal) and armauthorization +// (role definition + role assignment). +type graphSPProvisioner struct { + graph *msgraphsdk.GraphServiceClient + roleDefs *armauthorization.RoleDefinitionsClient + roleAsgn roleAssigner + // retryInitial and retryBudget control the PrincipalNotFound retry loop + // in AssignRole. They are set to the package constants by + // newGraphSPProvisioner and overridden in tests to keep test duration short. + retryInitial time.Duration + retryBudget time.Duration +} + +// newGraphSPProvisioner builds a graphSPProvisioner authenticated with the +// Azure CLI credential, so the session established by "az login" (wizard +// Step 1) is reused -- matching the principal used by the rest of the wizard. +// subscriptionID seeds the RoleAssignmentsClient; the actual scope is passed +// per-call to its Create method. +func newGraphSPProvisioner(subscriptionID string) (*graphSPProvisioner, error) { + cred, err := newAzureWizardCredential() + if err != nil { + return nil, err + } + + graph, err := msgraphsdk.NewGraphServiceClientWithCredentials( + cred, []string{"https://graph.microsoft.com/.default"}) + if err != nil { + return nil, fmt.Errorf("failed to create Microsoft Graph client: %w", err) + } + + roleDefs, err := armauthorization.NewRoleDefinitionsClient(cred, nil) + if err != nil { + return nil, fmt.Errorf("failed to create role definitions client: %w", err) + } + + roleAsgn, err := armauthorization.NewRoleAssignmentsClient(subscriptionID, cred, nil) + if err != nil { + return nil, fmt.Errorf("failed to create role assignments client: %w", err) + } + + return &graphSPProvisioner{ + graph: graph, + roleDefs: roleDefs, + roleAsgn: roleAsgn, + retryInitial: roleAssignRetryInitial, + retryBudget: roleAssignRetryBudget, + }, nil +} + +func (g *graphSPProvisioner) CreateApplication(ctx context.Context, displayName string) (objectID, appID string, err error) { + app := graphmodels.NewApplication() + app.SetDisplayName(&displayName) + + created, err := g.graph.Applications().Post(ctx, app, nil) + if err != nil { + return "", "", err + } + oid := created.GetId() + aid := created.GetAppId() + if oid == nil || aid == nil { + return "", "", fmt.Errorf("application created but Graph returned no id/appId") + } + return *oid, *aid, nil +} + +func (g *graphSPProvisioner) AddPassword(ctx context.Context, objectID string) (string, error) { + body := applications.NewItemAddPasswordPostRequestBody() + cred := graphmodels.NewPasswordCredential() + displayName := azureSPName + "-secret" + cred.SetDisplayName(&displayName) + body.SetPasswordCredential(cred) + + result, err := g.graph.Applications().ByApplicationId(objectID).AddPassword().Post(ctx, body, nil) + if err != nil { + return "", err + } + secret := result.GetSecretText() + if secret == nil || *secret == "" { + return "", fmt.Errorf("password credential created but Graph returned no secret text") + } + return *secret, nil +} + +func (g *graphSPProvisioner) CreateServicePrincipal(ctx context.Context, appID string) (string, error) { + sp := graphmodels.NewServicePrincipal() + sp.SetAppId(&appID) + + created, err := g.graph.ServicePrincipals().Post(ctx, sp, nil) + if err != nil { + return "", err + } + principalID := created.GetId() + if principalID == nil { + return "", fmt.Errorf("service principal created but Graph returned no id") + } + return *principalID, nil +} + +func (g *graphSPProvisioner) DeleteApplication(ctx context.Context, objectID string) error { + return g.graph.Applications().ByApplicationId(objectID).Delete(ctx, nil) +} + +func (g *graphSPProvisioner) ResolveRoleDefinitionID(ctx context.Context, scope, roleName string) (string, error) { + filter := fmt.Sprintf("roleName eq '%s'", roleName) + pager := g.roleDefs.NewListPager(scope, &armauthorization.RoleDefinitionsClientListOptions{ + Filter: &filter, + }) + for pager.More() { + page, err := pager.NextPage(ctx) + if err != nil { + return "", err + } + for _, rd := range page.Value { + if rd == nil || rd.ID == nil { + continue + } + if rd.Properties != nil && rd.Properties.RoleName != nil && *rd.Properties.RoleName == roleName { + return *rd.ID, nil + } + } + } + return "", fmt.Errorf("role definition %q not found at scope %s", roleName, scope) +} + +func (g *graphSPProvisioner) AssignRole(ctx context.Context, scope, principalID, roleDefinitionID string) error { + principalType := armauthorization.PrincipalTypeServicePrincipal + params := armauthorization.RoleAssignmentCreateParameters{ + Properties: &armauthorization.RoleAssignmentProperties{ + PrincipalID: &principalID, + RoleDefinitionID: &roleDefinitionID, + PrincipalType: &principalType, + }, + } + + // Azure AD replication is eventually consistent: a service principal + // created moments ago may not yet be visible to the ARM role-assignment + // API in a different region, returning PrincipalNotFound. Retry with + // bounded exponential back-off until the SP propagates or the budget is + // exhausted (see also: feedback_tf_depends_on_rbac.md). + deadline := time.Now().Add(g.retryBudget) + delay := g.retryInitial + for { + _, err := g.roleAsgn.Create(ctx, scope, uuid.NewString(), params, nil) + if err == nil { + return nil + } + if !isPrincipalNotFoundErr(err) { + return err + } + if time.Now().After(deadline) { + return fmt.Errorf( + "service principal %q did not propagate to ARM within %v "+ + "(Azure AD replication is eventually consistent -- "+ + "https://learn.microsoft.com/en-us/azure/role-based-access-control/troubleshooting): %w", + principalID, g.retryBudget, err) + } + // Context cancellation is terminal: do not continue retrying. Wrap + // ctx.Err() with the propagation guidance so the operator learns the + // assignment may just need more time rather than seeing a bare + // "context deadline exceeded" (errors.Is still matches the cause). + select { + case <-ctx.Done(): + return fmt.Errorf( + "stopped waiting for service principal %q to propagate to ARM "+ + "before the role assignment completed: %w -- Azure AD role "+ + "propagation can take up to ~10 minutes; re-run configure-azure "+ + "to retry (https://learn.microsoft.com/en-us/azure/role-based-access-control/troubleshooting)", + principalID, ctx.Err()) + case <-time.After(delay): + } + delay = min(delay*2, roleAssignRetryMax) + } +} diff --git a/cmd/configure_azure_sp_test.go b/cmd/configure_azure_sp_test.go new file mode 100644 index 000000000..a37c593ad --- /dev/null +++ b/cmd/configure_azure_sp_test.go @@ -0,0 +1,398 @@ +package main + +import ( + "context" + "errors" + "testing" + "time" + + "github.com/Azure/azure-sdk-for-go/sdk/azcore" + armauthorization "github.com/Azure/azure-sdk-for-go/sdk/resourcemanager/authorization/armauthorization/v2" + "github.com/stretchr/testify/assert" + "github.com/stretchr/testify/require" +) + +// mockSPProvisioner is a configurable azureSPProvisioner used to assert that +// createAzureServicePrincipal requests the correct app name, role and scope, +// and surfaces the generated secret. +type mockSPProvisioner struct { + // optional injected errors + createAppErr error + addPwErr error + createSPErr error + resolveErr error + assignErr error + deleteAppErr error + + // captured inputs + createAppName string + addPasswordObjID string + createSPAppID string + resolveRoleScope string + resolveRoleName string + assignScope string + assignPrincipalID string + assignRoleDefID string + deleteAppObjID string + + // canned outputs + appObjectID string + appID string + secret string + principalID string + roleDefID string + + // call flags + resolveRoleCalled bool + assignRoleCalled bool + deleteAppCalled bool +} + +func (m *mockSPProvisioner) CreateApplication(_ context.Context, displayName string) (string, string, error) { + m.createAppName = displayName + if m.createAppErr != nil { + return "", "", m.createAppErr + } + return m.appObjectID, m.appID, nil +} + +func (m *mockSPProvisioner) AddPassword(_ context.Context, objectID string) (string, error) { + m.addPasswordObjID = objectID + if m.addPwErr != nil { + return "", m.addPwErr + } + return m.secret, nil +} + +func (m *mockSPProvisioner) CreateServicePrincipal(_ context.Context, appID string) (string, error) { + m.createSPAppID = appID + if m.createSPErr != nil { + return "", m.createSPErr + } + return m.principalID, nil +} + +func (m *mockSPProvisioner) ResolveRoleDefinitionID(_ context.Context, scope, roleName string) (string, error) { + m.resolveRoleCalled = true + m.resolveRoleScope = scope + m.resolveRoleName = roleName + if m.resolveErr != nil { + return "", m.resolveErr + } + return m.roleDefID, nil +} + +func (m *mockSPProvisioner) AssignRole(_ context.Context, scope, principalID, roleDefinitionID string) error { + m.assignRoleCalled = true + m.assignScope = scope + m.assignPrincipalID = principalID + m.assignRoleDefID = roleDefinitionID + return m.assignErr +} + +func (m *mockSPProvisioner) DeleteApplication(_ context.Context, objectID string) error { + m.deleteAppCalled = true + m.deleteAppObjID = objectID + return m.deleteAppErr +} + +const ( + testSubID = "11111111-2222-3333-4444-555555555555" + testTenantID = "aaaaaaaa-bbbb-cccc-dddd-eeeeeeeeeeee" +) + +func TestCreateAzureServicePrincipal_Success(t *testing.T) { + m := &mockSPProvisioner{ + appObjectID: "app-object-id", + appID: "app-client-id", + secret: "super-secret-password", + principalID: "sp-principal-id", + roleDefID: "/subscriptions/" + testSubID + "/provider/.../roleDefinitions/role-guid", + } + + result, err := createAzureServicePrincipal(context.Background(), m, testSubID, testTenantID) + require.NoError(t, err) + + // App name must be exactly "CUDly". + assert.Equal(t, "CUDly", m.createAppName) + assert.Equal(t, azureSPName, m.createAppName) + + // Password added to the created application object. + assert.Equal(t, "app-object-id", m.addPasswordObjID) + + // Service principal created for the application's client ID. + assert.Equal(t, "app-client-id", m.createSPAppID) + + // Role resolved by exact display name at subscription scope. + assert.True(t, m.resolveRoleCalled) + assert.Equal(t, "Reservations Administrator", m.resolveRoleName) + assert.Equal(t, azureSPRoleName, m.resolveRoleName) + assert.Equal(t, "/subscriptions/"+testSubID, m.resolveRoleScope) + + // Role assignment binds the SP principal to the resolved role at subscription scope. + assert.True(t, m.assignRoleCalled) + assert.Equal(t, "/subscriptions/"+testSubID, m.assignScope) + assert.Equal(t, "sp-principal-id", m.assignPrincipalID) + assert.Equal(t, m.roleDefID, m.assignRoleDefID) + + // Result surfaces appId, secret and tenant (the create-for-rbac fields). + assert.Equal(t, "app-client-id", result.AppID) + assert.Equal(t, "super-secret-password", result.ClientSecret) + assert.Equal(t, testTenantID, result.TenantID) + + // No rollback on success. + assert.False(t, m.deleteAppCalled, "DeleteApplication must not be called on success") +} + +func TestCreateAzureServicePrincipal_ErrorPropagation(t *testing.T) { + tests := []struct { + name string + setup func(*mockSPProvisioner) + wantErrPart string + wantNoAssign bool + wantNoResolve bool + wantRollback bool // DeleteApplication should be called to clean up + }{ + { + name: "create application fails", + setup: func(m *mockSPProvisioner) { m.createAppErr = errors.New("graph 403") }, + wantErrPart: "failed to create application registration", + wantNoAssign: true, + wantNoResolve: true, + wantRollback: false, // nothing was created, nothing to roll back + }, + { + name: "add password fails", + setup: func(m *mockSPProvisioner) { m.addPwErr = errors.New("graph addPassword 400") }, + wantErrPart: "failed to add password credential", + wantNoAssign: true, + wantNoResolve: true, + wantRollback: true, + }, + { + name: "create service principal fails", + setup: func(m *mockSPProvisioner) { m.createSPErr = errors.New("graph sp 409") }, + wantErrPart: "failed to create service principal", + wantNoAssign: true, + wantNoResolve: true, + wantRollback: true, + }, + { + name: "resolve role fails", + setup: func(m *mockSPProvisioner) { m.resolveErr = errors.New("role not found") }, + wantErrPart: "failed to resolve", + wantNoAssign: true, + wantRollback: true, + }, + { + name: "assign role fails", + setup: func(m *mockSPProvisioner) { m.assignErr = errors.New("rbac 403") }, + wantErrPart: "failed to assign", + wantRollback: true, + }, + } + + for _, tt := range tests { + t.Run(tt.name, func(t *testing.T) { + m := &mockSPProvisioner{ + appObjectID: "app-object-id", + appID: "app-client-id", + secret: "secret", + principalID: "sp-principal-id", + roleDefID: "role-def-id", + } + tt.setup(m) + + _, err := createAzureServicePrincipal(context.Background(), m, testSubID, testTenantID) + require.Error(t, err) + assert.Contains(t, err.Error(), tt.wantErrPart) + + if tt.wantNoResolve { + assert.False(t, m.resolveRoleCalled, "ResolveRoleDefinitionID should not have been called") + } + if tt.wantNoAssign { + assert.False(t, m.assignRoleCalled, "AssignRole should not have been called") + } + assert.Equal(t, tt.wantRollback, m.deleteAppCalled, + "DeleteApplication call expectation mismatch") + if tt.wantRollback { + assert.Equal(t, "app-object-id", m.deleteAppObjID, + "rollback should delete the created application by object ID") + } + }) + } +} + +// TestCreateAzureServicePrincipal_RollbackFailureSurfaced verifies that when +// the compensating delete also fails, the error names the orphaned application +// so the operator can delete it manually. +func TestCreateAzureServicePrincipal_RollbackFailureSurfaced(t *testing.T) { + m := &mockSPProvisioner{ + appObjectID: "app-object-id", + appID: "app-client-id", + secret: "secret", + principalID: "sp-principal-id", + roleDefID: "role-def-id", + assignErr: errors.New("rbac 403"), + deleteAppErr: errors.New("delete 500"), + } + + _, err := createAzureServicePrincipal(context.Background(), m, testSubID, testTenantID) + require.Error(t, err) + assert.True(t, m.deleteAppCalled) + assert.Contains(t, err.Error(), "failed to assign") + assert.Contains(t, err.Error(), "failed to roll back") + assert.Contains(t, err.Error(), "app-object-id") +} + +func TestCreateAzureServicePrincipal_ScopeFormat(t *testing.T) { + m := &mockSPProvisioner{ + appObjectID: "o", appID: "a", secret: "s", principalID: "p", roleDefID: "r", + } + _, err := createAzureServicePrincipal(context.Background(), m, testSubID, testTenantID) + require.NoError(t, err) + // Scope must be exactly /subscriptions/ (subscription scope, not resource group). + assert.Equal(t, "/subscriptions/"+testSubID, m.resolveRoleScope) + assert.Equal(t, "/subscriptions/"+testSubID, m.assignScope) +} + +// fakeRoleAssigner is a test double for roleAssigner that fails with a +// PrincipalNotFound error for the first failsRemaining calls, then succeeds. +// If otherErr is set it is always returned instead (to test non-retryable paths). +type fakeRoleAssigner struct { + principalNotFoundErr *azcore.ResponseError + otherErr error + failsRemaining int + callCount int +} + +func (f *fakeRoleAssigner) Create( + _ context.Context, + _, _ string, + _ armauthorization.RoleAssignmentCreateParameters, + _ *armauthorization.RoleAssignmentsClientCreateOptions, +) (armauthorization.RoleAssignmentsClientCreateResponse, error) { + f.callCount++ + if f.otherErr != nil { + return armauthorization.RoleAssignmentsClientCreateResponse{}, f.otherErr + } + if f.failsRemaining > 0 { + f.failsRemaining-- + return armauthorization.RoleAssignmentsClientCreateResponse{}, f.principalNotFoundErr + } + return armauthorization.RoleAssignmentsClientCreateResponse{}, nil +} + +// newFakePrincipalNotFoundProvisioner returns a graphSPProvisioner wired to a +// fakeRoleAssigner with very short retry timing so tests complete in +// milliseconds rather than minutes. +func newFakePrincipalNotFoundProvisioner(fake *fakeRoleAssigner) *graphSPProvisioner { + return &graphSPProvisioner{ + roleAsgn: fake, + retryInitial: time.Millisecond, + retryBudget: 50 * time.Millisecond, + } +} + +// TestGraphSPProvisioner_AssignRole_RetrySucceeds verifies that AssignRole +// retries when ARM returns PrincipalNotFound and eventually succeeds. +func TestGraphSPProvisioner_AssignRole_RetrySucceeds(t *testing.T) { + principalNotFound := &azcore.ResponseError{ErrorCode: "PrincipalNotFound", StatusCode: 400} + fake := &fakeRoleAssigner{failsRemaining: 2, principalNotFoundErr: principalNotFound} + p := newFakePrincipalNotFoundProvisioner(fake) + + err := p.AssignRole(context.Background(), "/subscriptions/sub", "sp-id", "role-def-id") + + require.NoError(t, err) + assert.Equal(t, 3, fake.callCount, + "should call Create 3 times: 2 PrincipalNotFound failures then 1 success") +} + +// TestGraphSPProvisioner_AssignRole_ExhaustsRetryBudget verifies that +// AssignRole fails loud with a clear message when the service principal never +// propagates within the budget. +func TestGraphSPProvisioner_AssignRole_ExhaustsRetryBudget(t *testing.T) { + principalNotFound := &azcore.ResponseError{ErrorCode: "PrincipalNotFound", StatusCode: 400} + fake := &fakeRoleAssigner{failsRemaining: 100, principalNotFoundErr: principalNotFound} + p := newFakePrincipalNotFoundProvisioner(fake) + + err := p.AssignRole(context.Background(), "/subscriptions/sub", "sp-id", "role-def-id") + + require.Error(t, err) + assert.Contains(t, err.Error(), "did not propagate", + "error must explain that the SP did not propagate") + assert.Contains(t, err.Error(), "sp-id", + "error must identify the principal that failed to propagate") + assert.True(t, fake.callCount >= 1, "should have attempted Create at least once") +} + +// TestGraphSPProvisioner_AssignRole_NonRetryableError verifies that a +// non-PrincipalNotFound error is returned immediately without retrying. +func TestGraphSPProvisioner_AssignRole_NonRetryableError(t *testing.T) { + fake := &fakeRoleAssigner{otherErr: errors.New("authorization denied")} + p := newFakePrincipalNotFoundProvisioner(fake) + + err := p.AssignRole(context.Background(), "/subscriptions/sub", "sp-id", "role-def-id") + + require.Error(t, err) + assert.Contains(t, err.Error(), "authorization denied") + assert.Equal(t, 1, fake.callCount, "should not retry on non-PrincipalNotFound errors") +} + +// TestGraphSPProvisioner_AssignRole_ContextCancellation verifies that context +// cancellation is treated as a terminal stop and does not continue retrying. +func TestGraphSPProvisioner_AssignRole_ContextCancellation(t *testing.T) { + principalNotFound := &azcore.ResponseError{ErrorCode: "PrincipalNotFound", StatusCode: 400} + fake := &fakeRoleAssigner{failsRemaining: 100, principalNotFoundErr: principalNotFound} + p := newFakePrincipalNotFoundProvisioner(fake) + + ctx, cancel := context.WithCancel(context.Background()) + cancel() // cancel before the first retry sleep fires + + err := p.AssignRole(ctx, "/subscriptions/sub", "sp-id", "role-def-id") + + require.Error(t, err) + // The first Create call returns PrincipalNotFound; the subsequent select + // detects ctx.Done() and returns ctx.Err() immediately. + assert.Equal(t, 1, fake.callCount, "should stop retrying after context is canceled") +} + +// TestIsPrincipalNotFoundErr covers the error-code detection helper directly. +func TestIsPrincipalNotFoundErr(t *testing.T) { + tests := []struct { + err error + name string + want bool + }{ + { + name: "PrincipalNotFound error code", + err: &azcore.ResponseError{ErrorCode: "PrincipalNotFound", StatusCode: 400}, + want: true, + }, + { + name: "ServicePrincipalNotFound error code", + err: &azcore.ResponseError{ErrorCode: "ServicePrincipalNotFound", StatusCode: 400}, + want: true, + }, + { + name: "unrelated ARM error", + err: &azcore.ResponseError{ErrorCode: "AuthorizationFailed", StatusCode: 403}, + want: false, + }, + { + name: "plain error", + err: errors.New("network error"), + want: false, + }, + { + name: "nil error", + err: nil, + want: false, + }, + } + for _, tt := range tests { + t.Run(tt.name, func(t *testing.T) { + assert.Equal(t, tt.want, isPrincipalNotFoundErr(tt.err)) + }) + } +} diff --git a/cmd/configure_gcp.go b/cmd/configure_gcp.go new file mode 100644 index 000000000..a22452348 --- /dev/null +++ b/cmd/configure_gcp.go @@ -0,0 +1,827 @@ +package main + +import ( + "bufio" + "context" + "encoding/base64" + "encoding/json" + "fmt" + "log" + "os" + "os/exec" + "path/filepath" + "regexp" + "strings" + "time" + + "github.com/aws/aws-sdk-go-v2/aws" + awsconfig "github.com/aws/aws-sdk-go-v2/config" + "github.com/aws/aws-sdk-go-v2/service/secretsmanager" + "github.com/spf13/cobra" + "golang.org/x/oauth2/google" + "google.golang.org/api/cloudresourcemanager/v1" + iamv1 "google.golang.org/api/iam/v1" + "google.golang.org/api/option" +) + +// gcpProjectIDRegex validates GCP project IDs (lowercase letters, digits, hyphens, 6-30 chars). +var gcpProjectIDRegex = regexp.MustCompile(`^[a-z][a-z0-9-]{4,28}[a-z0-9]$`) + +// validateGCPProjectID validates a GCP project ID to prevent command injection. +func validateGCPProjectID(projectID string) error { + if !gcpProjectIDRegex.MatchString(projectID) { + return fmt.Errorf("invalid GCP project ID format: must be 6-30 lowercase letters, digits, or hyphens, starting with a letter") + } + return nil +} + +// GCPCredentials holds the GCP Service Account credentials. +type GCPCredentials struct { + Type string `json:"type"` + ProjectID string `json:"project_id"` + PrivateKeyID string `json:"private_key_id"` + PrivateKey string `json:"private_key"` // #nosec G117 -- operator-supplied credential input read from the user's own GCP service-account key file; marshaled only to store in AWS Secrets Manager, never a hardcoded secret and never logged (verified) + ClientEmail string `json:"client_email"` + ClientID string `json:"client_id,omitempty"` + AuthURI string `json:"auth_uri,omitempty"` + TokenURI string `json:"token_uri,omitempty"` + AuthProviderX509CertURL string `json:"auth_provider_x509_cert_url,omitempty"` + ClientX509CertURL string `json:"client_x509_cert_url,omitempty"` +} + +// GCPConfigOptions holds configuration for the GCP config command. +type GCPConfigOptions struct { + StackName string + Profile string + CredentialsFile string + ProjectID string + Interactive bool + SkipSetup bool +} + +var gcpOpts = GCPConfigOptions{} + +var configureGCPCmd = &cobra.Command{ + Use: "configure-gcp", + Short: "Configure GCP credentials for CUDly", + Long: `Configure GCP Service Account credentials for multi-cloud commitment management. + +This command stores your GCP credentials in AWS Secrets Manager for use by CUDly. + +You can provide credentials via a JSON key file: + cudly configure-gcp --stack-name my-cudly --credentials-file ~/gcp-service-account.json + +Or run interactively to create a new service account: + cudly configure-gcp --stack-name my-cudly --interactive`, + RunE: runConfigureGCP, +} + +func init() { + rootCmd.AddCommand(configureGCPCmd) + + configureGCPCmd.Flags().StringVar(&gcpOpts.StackName, "stack-name", "cudly", "CUDly CloudFormation stack name") + configureGCPCmd.Flags().StringVar(&gcpOpts.Profile, "profile", "", "AWS profile to use") + configureGCPCmd.Flags().StringVarP(&gcpOpts.CredentialsFile, "credentials-file", "f", "", "Path to GCP service account JSON key file") + configureGCPCmd.Flags().StringVar(&gcpOpts.ProjectID, "project-id", "", "GCP Project ID (overrides value in credentials file)") + configureGCPCmd.Flags().BoolVarP(&gcpOpts.Interactive, "interactive", "i", false, "Prompt for credentials file interactively") + configureGCPCmd.Flags().BoolVar(&gcpOpts.SkipSetup, "skip-setup", false, "Skip GCP CLI setup commands (gcloud login, create service account)") +} + +// storeGCPCredentials stores GCP credentials in the secrets store. +func storeGCPCredentials(ctx context.Context, store SecretsStore, stackName, credsJSON string) error { + // Validate that we have valid JSON + var creds GCPCredentials + if err := json.Unmarshal([]byte(credsJSON), &creds); err != nil { + return fmt.Errorf("failed to parse credentials: %w", err) + } + + // Validate credentials + if creds.Type != "service_account" { + return fmt.Errorf("invalid credentials file: expected type 'service_account', got '%s'", creds.Type) + } + + if creds.ProjectID == "" { + return fmt.Errorf("credentials file is missing project_id") + } + + if creds.ClientEmail == "" { + return fmt.Errorf("credentials file is missing client_email") + } + + if creds.PrivateKey == "" { + return fmt.Errorf("credentials file is missing private_key") + } + + // Build expected secret name pattern + secretName := fmt.Sprintf("%s-GCPCredentials", stackName) + + // Try to find the actual secret ARN by listing secrets + arns, err := store.ListSecrets(ctx, secretName) + + // Use the ARN if found, otherwise use the name (will fail if secret doesn't exist) + secretID := secretName + if err == nil && len(arns) > 0 { + secretID = arns[0] + } + + // Store credentials in Secrets Manager (using the original JSON format) + err = store.UpdateSecret(ctx, secretID, credsJSON) + if err != nil { + return fmt.Errorf("failed to store credentials in Secrets Manager: %w", err) + } + + return nil +} + +func runConfigureGCP(cmd *cobra.Command, args []string) error { + ctx := context.Background() + reader := bufio.NewReader(os.Stdin) + + fmt.Println("Configure GCP Service Account credentials for CUDly") + fmt.Println("===================================================") + fmt.Println() + + credsFile, err := getGCPCredentialsFilePath(ctx, reader) + if err != nil { + return err + } + + cfg, err := loadAWSConfigForGCP(ctx) + if err != nil { + return err + } + + creds, credsData, err := loadAndUpdateGCPCredentials(credsFile) + if err != nil { + return err + } + + smClient := secretsmanager.NewFromConfig(cfg) + store := NewAWSSecretsStore(smClient) + + if err := storeGCPCredentials(ctx, store, gcpOpts.StackName, string(credsData)); err != nil { + return err + } + + printGCPConfigurationSuccess(creds) + return nil +} + +// getGCPCredentialsFilePath determines the credentials file path from options or user input. +func getGCPCredentialsFilePath(ctx context.Context, reader *bufio.Reader) (string, error) { + var credsFile string + + if gcpOpts.CredentialsFile != "" { + credsFile = gcpOpts.CredentialsFile + } else if !gcpOpts.SkipSetup { + var err error + credsFile, err = runGCPSetupCommands(ctx, reader) + if err != nil { + return "", err + } + } + + if credsFile == "" { + fmt.Print("Path to GCP service account JSON key file: ") + var readErr error + credsFile, readErr = readTrimmedLine(reader) + if readErr != nil { + return "", fmt.Errorf("failed to read credentials file path: %w", readErr) + } + } + + if credsFile == "" { + return "", fmt.Errorf("credentials file is required") + } + + return credsFile, nil +} + +// loadAWSConfigForGCP loads AWS configuration with optional profile. +func loadAWSConfigForGCP(ctx context.Context) (aws.Config, error) { + var opts []func(*awsconfig.LoadOptions) error + if gcpOpts.Profile != "" { + opts = append(opts, awsconfig.WithSharedConfigProfile(gcpOpts.Profile)) + } + + cfg, err := awsconfig.LoadDefaultConfig(ctx, opts...) + if err != nil { + return aws.Config{}, fmt.Errorf("failed to load AWS config: %w", err) + } + + return cfg, nil +} + +// loadAndUpdateGCPCredentials loads, parses, and optionally updates GCP credentials. +func loadAndUpdateGCPCredentials(credsFile string) (GCPCredentials, []byte, error) { + expandedPath := filepath.Clean(expandHomeDirectory(credsFile)) + + // #nosec G304 G703 -- expandedPath is the operator's own GCP service-account key file, supplied via the --credentials-file flag or the interactive prompt of this local `configure-gcp` command; filepath.Clean applied above and it is trusted operator input, not attacker-controlled + credsData, err := os.ReadFile(expandedPath) + if err != nil { + return GCPCredentials{}, nil, fmt.Errorf("failed to read credentials file: %w", err) + } + + var creds GCPCredentials + if err = json.Unmarshal(credsData, &creds); err != nil { + return GCPCredentials{}, nil, fmt.Errorf("failed to parse credentials file: %w", err) + } + + if gcpOpts.ProjectID != "" { + creds.ProjectID = gcpOpts.ProjectID + credsData, err = json.Marshal(creds) // #nosec G117 -- GCPCredentials marshaled intentionally for Secrets Manager storage + if err != nil { + return GCPCredentials{}, nil, fmt.Errorf("failed to marshal updated credentials: %w", err) + } + } + + return creds, credsData, nil +} + +// expandHomeDirectory expands ~ to the user's home directory. +func expandHomeDirectory(path string) string { + if !strings.HasPrefix(path, "~/") { + return path + } + + home, err := os.UserHomeDir() + if err != nil { + return path + } + + return strings.Replace(path, "~", home, 1) +} + +// printGCPConfigurationSuccess prints success message with credentials info. +func printGCPConfigurationSuccess(creds GCPCredentials) { + log.Printf("GCP credentials stored successfully in Secrets Manager") + fmt.Println("\nGCP configuration complete!") + fmt.Printf("Service Account: %s\n", creds.ClientEmail) + fmt.Printf("Project ID: %s\n", creds.ProjectID) + fmt.Println("\nCUDly can now manage GCP Committed Use Discounts.") +} + +// gcpSDKCallTimeout bounds each GCP SDK helper so an ADC lookup or API call +// cannot hang indefinitely (the calls inherit context.Background()). +const gcpSDKCallTimeout = 60 * time.Second + +// newGCPAPIOption returns an oauth2 token source option for the google.golang.org +// API client, using Application Default Credentials. ADC resolves credentials in +// priority order: GOOGLE_APPLICATION_CREDENTIALS env var, gcloud ADC cache +// (populated by "gcloud auth application-default login"), Workload Identity, +// Metadata Server. +// +// NOTE: "gcloud auth login" (wizard Step 1) updates the gcloud user session but +// does NOT populate the ADC cache; the wizard's Step 1b +// ("gcloud auth application-default login") does that. If ADC is still not +// available (e.g. the operator skipped Step 1b) these calls fail loud with a +// hint to run "gcloud auth application-default login". +func newGCPAPIOption(ctx context.Context) (option.ClientOption, error) { + ts, err := google.DefaultTokenSource(ctx, + "https://www.googleapis.com/auth/cloud-platform", + "https://www.googleapis.com/auth/iam", + ) + if err != nil { + return nil, fmt.Errorf("failed to obtain GCP Application Default Credentials: %w\n"+ + "Hint: run 'gcloud auth application-default login' first", err) + } + return option.WithTokenSource(ts), nil +} + +// listGCPProjects lists GCP projects accessible to the operator via the Cloud +// Resource Manager API v1 and prints them in a table. This replaces the +// "gcloud projects list" CLI call. +func listGCPProjects(ctx context.Context) error { + ctx, cancel := context.WithTimeout(ctx, gcpSDKCallTimeout) + defer cancel() + + opt, err := newGCPAPIOption(ctx) + if err != nil { + return err + } + + svc, err := cloudresourcemanager.NewService(ctx, opt) + if err != nil { + return fmt.Errorf("failed to create resource manager client: %w", err) + } + + fmt.Printf("%-30s %-25s %s\n", "NAME", "PROJECT_ID", "PROJECT_NUMBER") + fmt.Println(strings.Repeat("-", 80)) + + req := svc.Projects.List() + if err := req.Pages(ctx, func(page *cloudresourcemanager.ListProjectsResponse) error { + for _, p := range page.Projects { + fmt.Printf("%-30s %-25s %d\n", p.Name, p.ProjectId, p.ProjectNumber) + } + return nil + }); err != nil { + return fmt.Errorf("failed to list GCP projects: %w", err) + } + return nil +} + +// createGCPServiceAccount creates a GCP IAM service account via the IAM API v1. +// This replaces "gcloud iam service-accounts create". +func createGCPServiceAccount(ctx context.Context, projectID, saName string) (string, error) { + ctx, cancel := context.WithTimeout(ctx, gcpSDKCallTimeout) + defer cancel() + + opt, err := newGCPAPIOption(ctx) + if err != nil { + return "", err + } + + svc, err := iamv1.NewService(ctx, opt) + if err != nil { + return "", fmt.Errorf("failed to create IAM client: %w", err) + } + + req := &iamv1.CreateServiceAccountRequest{ + AccountId: saName, + ServiceAccount: &iamv1.ServiceAccount{ + DisplayName: "CUDly Service Account", + Description: "Service account for CUDly commitment management", + }, + } + + sa, err := svc.Projects.ServiceAccounts.Create("projects/"+projectID, req).Context(ctx).Do() + if err != nil { + return "", fmt.Errorf("failed to create service account: %w", err) + } + + return sa.Email, nil +} + +// grantGCPIAMRole grants an IAM role to a service account on a project via the +// Cloud Resource Manager API v1. This replaces +// "gcloud projects add-iam-policy-binding". +func grantGCPIAMRole(ctx context.Context, projectID, member, role string) error { + ctx, cancel := context.WithTimeout(ctx, gcpSDKCallTimeout) + defer cancel() + + opt, err := newGCPAPIOption(ctx) + if err != nil { + return err + } + + svc, err := cloudresourcemanager.NewService(ctx, opt) + if err != nil { + return fmt.Errorf("failed to create resource manager client: %w", err) + } + + // Request policy version 3 so conditional (IAM condition) bindings are + // returned and preserved on the read-modify-write round-trip; otherwise + // the SetIamPolicy below would silently drop them. + policy, err := svc.Projects.GetIamPolicy(projectID, &cloudresourcemanager.GetIamPolicyRequest{ + Options: &cloudresourcemanager.GetPolicyOptions{RequestedPolicyVersion: 3}, + }).Context(ctx).Do() + if err != nil { + return fmt.Errorf("failed to get IAM policy for project %s: %w", projectID, err) + } + + if !addMemberToPolicyBinding(policy, member, role) { + // Member already bound to the role; nothing to write. + return nil + } + + // Write the policy back at version 3 to retain any conditional bindings. + if policy.Version < 3 { + policy.Version = 3 + } + _, err = svc.Projects.SetIamPolicy(projectID, &cloudresourcemanager.SetIamPolicyRequest{ + Policy: policy, + }).Context(ctx).Do() + if err != nil { + return fmt.Errorf("failed to set IAM policy on project %s: %w", projectID, err) + } + return nil +} + +// addMemberToPolicyBinding adds member to the binding for role in policy, +// creating the binding if absent. It returns false if member is already bound +// (no change needed) and true if the policy was modified. +func addMemberToPolicyBinding(policy *cloudresourcemanager.Policy, member, role string) bool { + for _, b := range policy.Bindings { + if b.Role != role { + continue + } + for _, m := range b.Members { + if m == member { + return false + } + } + b.Members = append(b.Members, member) + return true + } + policy.Bindings = append(policy.Bindings, &cloudresourcemanager.Binding{ + Role: role, + Members: []string{member}, + }) + return true +} + +// gcpKeyProvisioner abstracts the IAM service-account key operations used by +// writeServiceAccountKey. It exists so the reserve / mint / decode / write / +// rollback flow can be unit-tested with a mock without hitting GCP (mirrors +// azureSPProvisioner in configure_azure_sp.go). +type gcpKeyProvisioner interface { + // CreateKey mints a new JSON key for saEmail and returns the key resource + // name and the base64-encoded private key material. + CreateKey(ctx context.Context, saEmail string) (keyName, privateKeyData string, err error) + // DeleteKey deletes the key identified by keyName. It is the compensating + // action used to avoid orphaning a freshly minted key on a local failure. + DeleteKey(ctx context.Context, keyName string) error +} + +// iamKeyProvisioner is the production gcpKeyProvisioner backed by the IAM API v1. +type iamKeyProvisioner struct { + svc *iamv1.Service +} + +func (k *iamKeyProvisioner) CreateKey(ctx context.Context, saEmail string) (keyName, privateKeyData string, err error) { + resource := fmt.Sprintf("projects/-/serviceAccounts/%s", saEmail) + key, err := k.svc.Projects.ServiceAccounts.Keys.Create(resource, &iamv1.CreateServiceAccountKeyRequest{ + PrivateKeyType: "TYPE_GOOGLE_CREDENTIALS_FILE", + }).Context(ctx).Do() + if err != nil { + return "", "", err + } + return key.Name, key.PrivateKeyData, nil +} + +func (k *iamKeyProvisioner) DeleteKey(ctx context.Context, keyName string) error { + _, err := k.svc.Projects.ServiceAccounts.Keys.Delete(keyName).Context(ctx).Do() + return err +} + +// createGCPServiceAccountKey creates a JSON key for the given service account +// and writes it to keyFile. This replaces +// "gcloud iam service-accounts keys create --iam-account=". +func createGCPServiceAccountKey(ctx context.Context, saEmail, keyFile string) error { + ctx, cancel := context.WithTimeout(ctx, gcpSDKCallTimeout) + defer cancel() + + opt, err := newGCPAPIOption(ctx) + if err != nil { + return err + } + + svc, err := iamv1.NewService(ctx, opt) + if err != nil { + return fmt.Errorf("failed to create IAM client: %w", err) + } + + return writeServiceAccountKey(ctx, &iamKeyProvisioner{svc: svc}, saEmail, keyFile) +} + +// writeServiceAccountKey reserves keyFile with exclusive-create semantics +// BEFORE minting the remote key (so it never mints a key it cannot persist +// locally), then mints the key via p, decodes the base64 material and writes it +// to keyFile. If decoding or writing fails after the remote key is minted it +// deletes the remote key so it does not linger as an active, unused credential. +// Extracted from createGCPServiceAccountKey so the reserve / mint / rollback +// flow is unit-testable with a mock (no GCP credentials). +func writeServiceAccountKey(ctx context.Context, p gcpKeyProvisioner, saEmail, keyFile string) error { + // Reserve the destination file first (fails if it already exists), so we + // never mint a remote key we cannot persist locally. + // #nosec G304 -- keyFile is the sole caller's fixed path filepath.Join(os.UserHomeDir(), "cudly-gcp-key.json"); a constant filename under the operator's own home dir, program-controlled and not attacker input + f, err := os.OpenFile(keyFile, os.O_WRONLY|os.O_CREATE|os.O_EXCL, 0600) + if err != nil { + return fmt.Errorf("failed to reserve key file %s: %w", keyFile, err) + } + // Best-effort: remove the reserved file if we return before writing it. + wrote := false + defer func() { + _ = f.Close() + if !wrote { + _ = os.Remove(keyFile) + } + }() + + keyName, privateKeyData, err := p.CreateKey(ctx, saEmail) + if err != nil { + return fmt.Errorf("failed to create service account key: %w", err) + } + + // From here on, any failure must delete the newly minted remote key so it + // does not linger as an active, unused credential. + deleteRemoteKey := func(cause error) error { + // Use a fresh context: the parent may already be canceled/expired. + delCtx, delCancel := context.WithTimeout(context.Background(), gcpSDKCallTimeout) + defer delCancel() + if delErr := p.DeleteKey(delCtx, keyName); delErr != nil { + return fmt.Errorf("%w; additionally failed to delete the orphaned remote key %s: %w", cause, keyName, delErr) + } + return cause + } + + // PrivateKeyData is base64-encoded JSON. + decoded, err := base64.StdEncoding.DecodeString(privateKeyData) + if err != nil { + return deleteRemoteKey(fmt.Errorf("failed to decode key data: %w", err)) + } + + if _, err := f.Write(decoded); err != nil { + return deleteRemoteKey(fmt.Errorf("failed to write key file %s: %w", keyFile, err)) + } + wrote = true + return nil +} + +// runGCPSetupCommands guides the operator through GCP setup. +// +// Step 1 (gcloud auth login) and Step 1b (gcloud auth application-default +// login): performed via the GCP CLI. Both are interactive browser-based OAuth +// flows that cannot be replicated through SDK calls on behalf of an operator +// who does not yet have a credential. Step 1 establishes the gcloud user +// session; Step 1b populates the Application Default Credentials (ADC) cache +// that every SDK-based step below authenticates through. Both are run in one +// pass so a fresh operator completes setup without re-running the wizard. +// +// Step 2 (list projects): performed via the Cloud Resource Manager SDK v1, +// using Application Default Credentials (ADC). Fails loud if ADC is not +// available (no CLI fallback). +// +// Step 3 (gcloud config set project): sets local gcloud config state. There is +// no cloud-API equivalent for writing to the local gcloud configuration file, +// so this step remains a CLI call. +// +// Steps 4-6 (create SA, grant role, create key): performed via GCP IAM and +// Cloud Resource Manager SDK v1 APIs using ADC. Fail loud on any SDK error +// (no CLI fallback). +func runGCPSetupCommands(ctx context.Context, reader *bufio.Reader) (string, error) { + if err := gcpStepLogin(reader); err != nil { + return "", err + } + + projectID, err := gcpStepSelectProject(ctx, reader) + if err != nil { + return "", err + } + + saEmail, err := gcpStepCreateServiceAccount(ctx, reader, projectID) + if err != nil { + return "", err + } + + if err := gcpStepGrantRole(ctx, reader, projectID, saEmail); err != nil { + return "", err + } + + return gcpStepCreateKey(ctx, reader, saEmail) +} + +// gcpStepLogin runs the two interactive gcloud logins the wizard needs: +// "gcloud auth login" (user session) and "gcloud auth application-default +// login" (ADC cache). Both are browser-based OAuth flows with no SDK +// equivalent. The ADC login is required because the SDK-based steps below +// (list projects, create service account, grant role, create key) authenticate +// through Application Default Credentials, which "gcloud auth login" alone does +// NOT populate -- so without it a fresh operator would hard-abort at Step 2. +func gcpStepLogin(reader *bufio.Reader) error { + fmt.Println("Step 1: GCP Login") + fmt.Println("-----------------") + fmt.Println("This opens a browser window for GCP authentication (user session).") + fmt.Println() + if err := promptAndRunGCPCommand(reader, "GCP Login", "gcloud auth login", "gcloud", "auth", "login"); err != nil { + return err + } + + fmt.Println() + fmt.Println("Step 1b: GCP Application Default Credentials Login") + fmt.Println("-------------------------------------------------") + fmt.Println("This opens a browser window to populate the Application Default") + fmt.Println("Credentials (ADC) cache used by the SDK-based steps below") + fmt.Println("(list projects, create service account, grant role, create key).") + fmt.Println() + return promptAndRunGCPCommand(reader, "GCP ADC Login", + "gcloud auth application-default login", + "gcloud", "auth", "application-default", "login") +} + +// gcpStepSelectProject optionally lists projects and prompts for a project ID. +// The listing is behind a [R]un/[S]kip prompt so an operator who already knows +// their project ID can proceed even if the SDK listing would fail; when RUN it +// fails loud (no CLI fallback), instructing the operator to run +// "gcloud auth application-default login" first. +func gcpStepSelectProject(ctx context.Context, reader *bufio.Reader) (string, error) { + fmt.Println() + fmt.Println("Step 2: Select Project") + fmt.Println("----------------------") + + run, err := promptRunOrSkipListing(reader, "the GCP project listing (via SDK)") + if err != nil { + return "", err + } + if run { + fmt.Println("Listing your GCP projects via SDK (Application Default Credentials)...") + fmt.Println() + if err = listGCPProjects(ctx); err != nil { + return "", fmt.Errorf("failed to list GCP projects via SDK: %w\n"+ + "Ensure Application Default Credentials are set: run 'gcloud auth application-default login' first", err) + } + fmt.Println() + } + + projectID, err := readRequiredInputLine(reader, "Enter your Project ID: ", "project ID") + if err != nil { + return "", err + } + if err := validateGCPProjectID(projectID); err != nil { + return "", err + } + + // Set the project in the local gcloud config. This is a local operation + // (writes to ~/.config/gcloud/properties) with no cloud-API equivalent. + fmt.Println() + fmt.Println("Setting gcloud project context (local config)...") + // #nosec G204 G702 -- local gcloud config write: fixed argv ("gcloud config set project"), projectID pre-validated by validateGCPProjectID (strict regex) just above, passed as a discrete argv element with no shell, so it cannot inject + cmd := exec.Command("gcloud", "config", "set", "project", projectID) + cmd.Stdout = os.Stdout + cmd.Stderr = os.Stderr + if err := cmd.Run(); err != nil { + return "", fmt.Errorf("failed to set project: %w", err) + } + return projectID, nil +} + +// gcpStepCreateServiceAccount creates the CUDly service account via the IAM +// SDK. It fails loud on any SDK error (no CLI fallback). +func gcpStepCreateServiceAccount(ctx context.Context, reader *bufio.Reader, projectID string) (string, error) { + saName := "cudly-service-account" + saEmail := fmt.Sprintf("%s@%s.iam.gserviceaccount.com", saName, projectID) + + fmt.Println() + fmt.Println("Step 3: Create Service Account") + fmt.Println("------------------------------") + fmt.Println("This creates a GCP Service Account for CUDly via the IAM API.") + fmt.Println() + fmt.Printf("[R]un, [S]kip? (creates service account '%s' via SDK) ", saName) + + choice, err := reader.ReadString('\n') + if err != nil { + return "", fmt.Errorf("failed to read service-account choice: %w", err) + } + switch strings.ToLower(strings.TrimSpace(choice)) { + case "r", "run", "": + email, createErr := createGCPServiceAccount(ctx, projectID, saName) + if createErr != nil { + return "", createErr + } + saEmail = email + fmt.Printf("Service account created: %s\n", saEmail) + case "s", "skip": + fmt.Println("Skipping Create Service Account") + default: + fmt.Printf("Unknown option, skipping\n") + } + return saEmail, nil +} + +// gcpStepGrantRole grants the compute.admin role to the service account via +// the Cloud Resource Manager SDK. It fails loud on any SDK error (no CLI +// fallback). +func gcpStepGrantRole(ctx context.Context, reader *bufio.Reader, projectID, saEmail string) error { + member := fmt.Sprintf("serviceAccount:%s", saEmail) + role := "roles/compute.admin" + + fmt.Println() + fmt.Println("Step 4: Grant IAM Roles") + fmt.Println("-----------------------") + fmt.Println("Grant the required roles to the service account.") + fmt.Println() + fmt.Printf("[R]un, [S]kip? (grants %s to %s on project %s via SDK) ", role, saEmail, projectID) + + choice, err := reader.ReadString('\n') + if err != nil { + return fmt.Errorf("failed to read grant-role choice: %w", err) + } + switch strings.ToLower(strings.TrimSpace(choice)) { + case "r", "run", "": + if grantErr := grantGCPIAMRole(ctx, projectID, member, role); grantErr != nil { + return grantErr + } + fmt.Printf("Role %s granted to %s on project %s.\n", role, saEmail, projectID) + case "s", "skip": + fmt.Println("Skipping Grant IAM Roles") + default: + fmt.Printf("Unknown option, skipping\n") + } + return nil +} + +// gcpStepCreateKey creates a JSON key file for the service account. It returns +// the written key-file path only when a key was actually created; on skip or +// an unknown choice it returns an empty string so the caller knows to prompt +// for an existing credentials file instead of assuming one was written. +func gcpStepCreateKey(ctx context.Context, reader *bufio.Reader, saEmail string) (string, error) { + home, err := os.UserHomeDir() + if err != nil { + return "", fmt.Errorf("failed to get home directory: %w", err) + } + keyFile := filepath.Join(home, "cudly-gcp-key.json") + + fmt.Println() + fmt.Println("Step 5: Create and Download Key") + fmt.Println("-------------------------------") + fmt.Println("Create a JSON key file for the service account.") + fmt.Println() + fmt.Printf("[R]un, [S]kip? (creates key for %s, writes to %s via SDK) ", saEmail, keyFile) + + choice, err := reader.ReadString('\n') + if err != nil { + return "", fmt.Errorf("failed to read create-key choice: %w", err) + } + switch strings.ToLower(strings.TrimSpace(choice)) { + case "r", "run", "": + if keyErr := createGCPServiceAccountKey(ctx, saEmail, keyFile); keyErr != nil { + return "", keyErr + } + fmt.Printf("Key file written to: %s\n", keyFile) + fmt.Println() + return keyFile, nil + case "s", "skip": + fmt.Println("Skipping Create Key") + default: + fmt.Printf("Unknown option, skipping\n") + } + + // No key file was written; the caller will prompt for an existing one. + fmt.Println() + return "", nil +} + +// readRequiredInputLine prints prompt, reads a line, trims whitespace, and +// returns an error if the result is empty. +func readRequiredInputLine(reader *bufio.Reader, prompt, fieldName string) (string, error) { + fmt.Print(prompt) + value, err := readTrimmedLine(reader) + if err != nil { + return "", fmt.Errorf("failed to read %s: %w", fieldName, err) + } + if value == "" { + return "", fmt.Errorf("%s is required", fieldName) + } + return value, nil +} + +// promptAndRunGCPCommand shows a command and asks to run or skip. +// It is used only for the interactive "gcloud auth login" auth bootstrap +// (Step 1), which has no SDK equivalent that preserves the cached-credential UX. +func promptAndRunGCPCommand(reader *bufio.Reader, name, displayCmd, program string, args ...string) error { + fmt.Printf("Command: %s\n", displayCmd) + fmt.Println() + fmt.Printf("[R]un, [S]kip? ") + + choice, err := readTrimmedLine(reader) + if err != nil { + return fmt.Errorf("failed to read choice: %w", err) + } + choice = strings.ToLower(choice) + + switch choice { + case "r", "run", "": + return executeGCPCommand(reader, displayCmd, program, args...) + case "s", "skip": + fmt.Printf("Skipping %s\n", name) + return nil + default: + fmt.Printf("Unknown option '%s', skipping\n", choice) + return nil + } +} + +// executeGCPCommand runs a gcloud command with explicit program and arguments. +// It is used only for the interactive "gcloud auth login" auth bootstrap. +// The caller's reader is threaded through to the retry prompt so all input +// is consumed from one consistent buffered stream (a fresh +// bufio.NewReader(os.Stdin) here would drop input already buffered by the +// caller's reader, breaking piped input after earlier prompts). +func executeGCPCommand(reader *bufio.Reader, displayCmd, program string, args ...string) error { + fmt.Println() + fmt.Printf("Executing: %s\n", displayCmd) + fmt.Println(strings.Repeat("-", 60)) + + // #nosec G204 -- interactive operator auth (gcloud auth login): program and args are hardcoded literals from the caller (runGCPSetupCommands passes "gcloud","auth","login"), no shell, not attacker-controlled + cmd := exec.Command(program, args...) + cmd.Stdout = os.Stdout + cmd.Stderr = os.Stderr + cmd.Stdin = os.Stdin + + err := cmd.Run() + fmt.Println(strings.Repeat("-", 60)) + + if err != nil { + fmt.Printf("Command failed: %v\n", err) + fmt.Print("Continue anyway? [y/N]: ") + response, readErr := readTrimmedLine(reader) + if readErr != nil { + return fmt.Errorf("failed to read response: %w", readErr) + } + if !strings.EqualFold(response, "y") { + return fmt.Errorf("command failed: %w", err) + } + } + + return nil +} diff --git a/cmd/configure_gcp_test.go b/cmd/configure_gcp_test.go new file mode 100644 index 000000000..d3a606cdd --- /dev/null +++ b/cmd/configure_gcp_test.go @@ -0,0 +1,250 @@ +package main + +import ( + "context" + "encoding/base64" + "errors" + "os" + "path/filepath" + "testing" + + "github.com/stretchr/testify/assert" + "github.com/stretchr/testify/require" + cloudresourcemanager "google.golang.org/api/cloudresourcemanager/v1" +) + +// --- addMemberToPolicyBinding ------------------------------------------------- + +// TestAddMemberToPolicyBinding_AppendsToExistingBinding verifies a member is +// appended to an existing binding for the role and the function reports a +// change. +func TestAddMemberToPolicyBinding_AppendsToExistingBinding(t *testing.T) { + policy := &cloudresourcemanager.Policy{ + Bindings: []*cloudresourcemanager.Binding{ + {Role: "roles/viewer", Members: []string{"user:existing@example.com"}}, + }, + } + + changed := addMemberToPolicyBinding(policy, "serviceAccount:sa@proj.iam.gserviceaccount.com", "roles/viewer") + + require.True(t, changed, "adding a new member to an existing role binding must report a change") + require.Len(t, policy.Bindings, 1) + assert.Equal(t, []string{ + "user:existing@example.com", + "serviceAccount:sa@proj.iam.gserviceaccount.com", + }, policy.Bindings[0].Members) +} + +// TestAddMemberToPolicyBinding_AlreadyBoundNoChange verifies that an +// already-bound member is a no-op (returns false, no duplicate appended). +func TestAddMemberToPolicyBinding_AlreadyBoundNoChange(t *testing.T) { + member := "serviceAccount:sa@proj.iam.gserviceaccount.com" + policy := &cloudresourcemanager.Policy{ + Bindings: []*cloudresourcemanager.Binding{ + {Role: "roles/viewer", Members: []string{member}}, + }, + } + + changed := addMemberToPolicyBinding(policy, member, "roles/viewer") + + require.False(t, changed, "re-adding a member already bound to the role must report no change") + require.Len(t, policy.Bindings, 1) + assert.Equal(t, []string{member}, policy.Bindings[0].Members, + "member must not be duplicated") +} + +// TestAddMemberToPolicyBinding_CreatesMissingBinding verifies a new binding is +// appended when the role is absent from the policy. +func TestAddMemberToPolicyBinding_CreatesMissingBinding(t *testing.T) { + policy := &cloudresourcemanager.Policy{ + Bindings: []*cloudresourcemanager.Binding{ + {Role: "roles/viewer", Members: []string{"user:existing@example.com"}}, + }, + } + + member := "serviceAccount:sa@proj.iam.gserviceaccount.com" + changed := addMemberToPolicyBinding(policy, member, "roles/billing.projectManager") + + require.True(t, changed, "adding a member to an absent role must create the binding and report a change") + require.Len(t, policy.Bindings, 2) + newBinding := policy.Bindings[1] + assert.Equal(t, "roles/billing.projectManager", newBinding.Role) + assert.Equal(t, []string{member}, newBinding.Members) +} + +// TestAddMemberToPolicyBinding_PreservesConditionalBindings is the regression +// guard for the version-3 read-modify-write round-trip: adding a member to one +// role must NOT drop or mutate a conditional (IAM condition) binding on another +// role. Losing conditional bindings would silently widen access. +func TestAddMemberToPolicyBinding_PreservesConditionalBindings(t *testing.T) { + conditional := &cloudresourcemanager.Binding{ + Role: "roles/storage.objectViewer", + Members: []string{"user:auditor@example.com"}, + Condition: &cloudresourcemanager.Expr{ + Title: "only-prod-bucket", + Expression: `resource.name.startsWith("projects/_/buckets/prod-")`, + }, + } + policy := &cloudresourcemanager.Policy{ + Version: 3, + Bindings: []*cloudresourcemanager.Binding{ + conditional, + {Role: "roles/viewer", Members: []string{"user:existing@example.com"}}, + }, + } + + member := "serviceAccount:sa@proj.iam.gserviceaccount.com" + changed := addMemberToPolicyBinding(policy, member, "roles/viewer") + require.True(t, changed) + + // The conditional binding must still be present, unchanged. + require.Len(t, policy.Bindings, 2, "no binding may be dropped") + var found *cloudresourcemanager.Binding + for _, b := range policy.Bindings { + if b.Role == "roles/storage.objectViewer" { + found = b + } + } + require.NotNil(t, found, "the conditional binding must be preserved") + require.NotNil(t, found.Condition, "the IAM condition must be preserved") + assert.Equal(t, "only-prod-bucket", found.Condition.Title) + assert.Equal(t, `resource.name.startsWith("projects/_/buckets/prod-")`, found.Condition.Expression) + assert.Equal(t, []string{"user:auditor@example.com"}, found.Members, + "the conditional binding's members must be untouched") +} + +// --- writeServiceAccountKey (key-creation rollback) --------------------------- + +// mockGCPKeyProvisioner is a configurable gcpKeyProvisioner used to assert the +// reserve / mint / decode / write / rollback flow of writeServiceAccountKey. +type mockGCPKeyProvisioner struct { + createErr error + deleteErr error + keyName string + privateKeyData string // base64-encoded, as returned by the IAM API + + createCalled bool + deleteCalled bool + createSAEmail string + deletedKeyName string +} + +func (m *mockGCPKeyProvisioner) CreateKey(_ context.Context, saEmail string) (string, string, error) { + m.createCalled = true + m.createSAEmail = saEmail + if m.createErr != nil { + return "", "", m.createErr + } + return m.keyName, m.privateKeyData, nil +} + +func (m *mockGCPKeyProvisioner) DeleteKey(_ context.Context, keyName string) error { + m.deleteCalled = true + m.deletedKeyName = keyName + return m.deleteErr +} + +func TestWriteServiceAccountKey_Success(t *testing.T) { + keyMaterial := []byte(`{"type":"service_account","project_id":"proj"}`) + m := &mockGCPKeyProvisioner{ + keyName: "projects/-/serviceAccounts/sa@proj.iam.gserviceaccount.com/keys/abc123", + privateKeyData: base64.StdEncoding.EncodeToString(keyMaterial), + } + keyFile := filepath.Join(t.TempDir(), "cudly-gcp-key.json") + + err := writeServiceAccountKey(context.Background(), m, "sa@proj.iam.gserviceaccount.com", keyFile) + require.NoError(t, err) + + assert.True(t, m.createCalled) + assert.Equal(t, "sa@proj.iam.gserviceaccount.com", m.createSAEmail) + assert.False(t, m.deleteCalled, "DeleteKey must not be called on success") + + // The decoded key material must have been written to the file. + got, readErr := os.ReadFile(keyFile) // #nosec G304 -- test-controlled temp path + require.NoError(t, readErr) + assert.Equal(t, keyMaterial, got) +} + +// TestWriteServiceAccountKey_DecodeFailureRollsBack verifies that when the +// returned key material is not valid base64, the minted remote key is deleted +// (so it does not linger as an active unused credential) and the reserved local +// file is cleaned up. +func TestWriteServiceAccountKey_DecodeFailureRollsBack(t *testing.T) { + m := &mockGCPKeyProvisioner{ + keyName: "projects/-/serviceAccounts/sa@proj.iam.gserviceaccount.com/keys/abc123", + privateKeyData: "!!!not-base64!!!", + } + keyFile := filepath.Join(t.TempDir(), "cudly-gcp-key.json") + + err := writeServiceAccountKey(context.Background(), m, "sa@proj.iam.gserviceaccount.com", keyFile) + require.Error(t, err) + assert.Contains(t, err.Error(), "failed to decode key data") + + // The freshly minted remote key must have been deleted by its resource name. + assert.True(t, m.deleteCalled, "the minted key must be deleted when decoding fails") + assert.Equal(t, m.keyName, m.deletedKeyName) + + // The reserved local file must be cleaned up (nothing persisted). + _, statErr := os.Stat(keyFile) + assert.True(t, os.IsNotExist(statErr), "the reserved key file must be removed on failure") +} + +// TestWriteServiceAccountKey_RollbackFailureSurfaced verifies that when the +// compensating DeleteKey also fails, the error names the orphaned remote key so +// the operator can delete it manually, while still surfacing the original cause. +func TestWriteServiceAccountKey_RollbackFailureSurfaced(t *testing.T) { + m := &mockGCPKeyProvisioner{ + keyName: "projects/-/serviceAccounts/sa@proj.iam.gserviceaccount.com/keys/orphan999", + privateKeyData: "!!!not-base64!!!", + deleteErr: errors.New("delete 500"), + } + keyFile := filepath.Join(t.TempDir(), "cudly-gcp-key.json") + + err := writeServiceAccountKey(context.Background(), m, "sa@proj.iam.gserviceaccount.com", keyFile) + require.Error(t, err) + assert.True(t, m.deleteCalled) + assert.Contains(t, err.Error(), "failed to decode key data", "the original cause must be surfaced") + assert.Contains(t, err.Error(), "failed to delete the orphaned remote key") + assert.Contains(t, err.Error(), "orphan999", "the orphaned key name must be named for manual cleanup") +} + +// TestWriteServiceAccountKey_CreateFailureNoOrphan verifies that when minting +// the remote key fails, no rollback is attempted (nothing was minted) and no +// local file is left behind. +func TestWriteServiceAccountKey_CreateFailureNoOrphan(t *testing.T) { + m := &mockGCPKeyProvisioner{ + createErr: errors.New("iam 403"), + } + keyFile := filepath.Join(t.TempDir(), "cudly-gcp-key.json") + + err := writeServiceAccountKey(context.Background(), m, "sa@proj.iam.gserviceaccount.com", keyFile) + require.Error(t, err) + assert.Contains(t, err.Error(), "failed to create service account key") + assert.False(t, m.deleteCalled, "no remote key was minted, so DeleteKey must not be called") + + _, statErr := os.Stat(keyFile) + assert.True(t, os.IsNotExist(statErr), "the reserved key file must be removed when minting fails") +} + +// TestWriteServiceAccountKey_ReserveFailureNoMint verifies that if the +// destination file already exists (exclusive-create fails), the remote key is +// never minted, so there is nothing to orphan. +func TestWriteServiceAccountKey_ReserveFailureNoMint(t *testing.T) { + keyFile := filepath.Join(t.TempDir(), "cudly-gcp-key.json") + require.NoError(t, os.WriteFile(keyFile, []byte("pre-existing"), 0o600)) + + m := &mockGCPKeyProvisioner{ + keyName: "projects/-/serviceAccounts/sa@proj.iam.gserviceaccount.com/keys/abc123", + privateKeyData: base64.StdEncoding.EncodeToString([]byte("{}")), + } + + err := writeServiceAccountKey(context.Background(), m, "sa@proj.iam.gserviceaccount.com", keyFile) + require.Error(t, err) + assert.Contains(t, err.Error(), "failed to reserve key file") + assert.False(t, m.createCalled, "the remote key must not be minted when the file cannot be reserved") + + // The pre-existing file must be left intact (we must not clobber it). + got, readErr := os.ReadFile(keyFile) // #nosec G304 -- test-controlled temp path + require.NoError(t, readErr) + assert.Equal(t, []byte("pre-existing"), got) +} diff --git a/cmd/configure_test.go b/cmd/configure_test.go new file mode 100644 index 000000000..662b5a88a --- /dev/null +++ b/cmd/configure_test.go @@ -0,0 +1,713 @@ +package main + +import ( + "context" + "encoding/json" + "errors" + "testing" + + "github.com/stretchr/testify/assert" + "github.com/stretchr/testify/require" +) + +// MockSecretsStore is a mock implementation of SecretsStore for testing. +type MockSecretsStore struct { + listSecretsFunc func(ctx context.Context, filter string) ([]string, error) + updateSecretFunc func(ctx context.Context, secretID string, secretValue string) error + updatedSecrets map[string]string // Track updated secrets + listSecretsFilter string // Track last filter used +} + +func NewMockSecretsStore() *MockSecretsStore { + return &MockSecretsStore{ + updatedSecrets: make(map[string]string), + } +} + +func (m *MockSecretsStore) ListSecrets(ctx context.Context, filter string) ([]string, error) { + m.listSecretsFilter = filter + if m.listSecretsFunc != nil { + return m.listSecretsFunc(ctx, filter) + } + return []string{}, nil +} + +func (m *MockSecretsStore) UpdateSecret(ctx context.Context, secretID string, secretValue string) error { + m.updatedSecrets[secretID] = secretValue + if m.updateSecretFunc != nil { + return m.updateSecretFunc(ctx, secretID, secretValue) + } + return nil +} + +// TestAzureCredentials_Struct tests the AzureCredentials struct. +func TestAzureCredentials_Struct(t *testing.T) { + creds := AzureCredentials{ + TenantID: "tenant-123", + ClientID: "client-456", + ClientSecret: "secret-789", + SubscriptionID: "sub-abc", + } + + assert.Equal(t, "tenant-123", creds.TenantID) + assert.Equal(t, "client-456", creds.ClientID) + assert.Equal(t, "secret-789", creds.ClientSecret) + assert.Equal(t, "sub-abc", creds.SubscriptionID) +} + +// TestAzureConfigOptions_Defaults tests AzureConfigOptions defaults. +func TestAzureConfigOptions_Defaults(t *testing.T) { + opts := AzureConfigOptions{} + + assert.Equal(t, "", opts.StackName) + assert.Equal(t, "", opts.Profile) + assert.Equal(t, "", opts.TenantID) + assert.Equal(t, "", opts.ClientID) + assert.Equal(t, "", opts.ClientSecret) + assert.Equal(t, "", opts.SubscriptionID) + assert.False(t, opts.Interactive) +} + +// TestAzureConfigOptions_WithValues tests AzureConfigOptions with values. +func TestAzureConfigOptions_WithValues(t *testing.T) { + opts := AzureConfigOptions{ + StackName: "my-cudly", + Profile: "production", + TenantID: "tenant-id", + ClientID: "client-id", + ClientSecret: "client-secret", + SubscriptionID: "subscription-id", + Interactive: true, + } + + assert.Equal(t, "my-cudly", opts.StackName) + assert.Equal(t, "production", opts.Profile) + assert.Equal(t, "tenant-id", opts.TenantID) + assert.Equal(t, "client-id", opts.ClientID) + assert.Equal(t, "client-secret", opts.ClientSecret) + assert.Equal(t, "subscription-id", opts.SubscriptionID) + assert.True(t, opts.Interactive) +} + +// TestGCPCredentials_Struct tests the GCPCredentials struct. +func TestGCPCredentials_Struct(t *testing.T) { + creds := GCPCredentials{ + Type: "service_account", + ProjectID: "my-project", + PrivateKeyID: "key-123", + PrivateKey: "-----BEGIN PRIVATE KEY-----\n...", + ClientEmail: "sa@project.iam.gserviceaccount.com", + ClientID: "12345678901234567890", + } + + assert.Equal(t, "service_account", creds.Type) + assert.Equal(t, "my-project", creds.ProjectID) + assert.Equal(t, "key-123", creds.PrivateKeyID) + assert.Equal(t, "-----BEGIN PRIVATE KEY-----\n...", creds.PrivateKey) + assert.Equal(t, "sa@project.iam.gserviceaccount.com", creds.ClientEmail) + assert.Equal(t, "12345678901234567890", creds.ClientID) +} + +// TestGCPConfigOptions_Defaults tests GCPConfigOptions defaults. +func TestGCPConfigOptions_Defaults(t *testing.T) { + opts := GCPConfigOptions{} + + assert.Equal(t, "", opts.StackName) + assert.Equal(t, "", opts.Profile) + assert.Equal(t, "", opts.ProjectID) + assert.Equal(t, "", opts.CredentialsFile) + assert.False(t, opts.Interactive) +} + +// TestGCPConfigOptions_WithValues tests GCPConfigOptions with values. +func TestGCPConfigOptions_WithValues(t *testing.T) { + opts := GCPConfigOptions{ + StackName: "my-cudly", + Profile: "production", + ProjectID: "my-gcp-project", + CredentialsFile: "/path/to/credentials.json", + Interactive: true, + } + + assert.Equal(t, "my-cudly", opts.StackName) + assert.Equal(t, "production", opts.Profile) + assert.Equal(t, "my-gcp-project", opts.ProjectID) + assert.Equal(t, "/path/to/credentials.json", opts.CredentialsFile) + assert.True(t, opts.Interactive) +} + +// Tests for validateAzureUUID function. +func TestValidateAzureUUID(t *testing.T) { + tests := []struct { + name string + uuid string + fieldName string + wantErr bool + }{ + { + name: "Valid UUID - all lowercase", + uuid: "12345678-1234-1234-1234-123456789abc", + fieldName: "Tenant ID", + wantErr: false, + }, + { + name: "Valid UUID - all uppercase", + uuid: "12345678-1234-1234-1234-123456789ABC", + fieldName: "Client ID", + wantErr: false, + }, + { + name: "Valid UUID - mixed case", + uuid: "12345678-1234-1234-1234-123456789AbC", + fieldName: "Subscription ID", + wantErr: false, + }, + { + name: "Valid UUID - all zeros", + uuid: "00000000-0000-0000-0000-000000000000", + fieldName: "Tenant ID", + wantErr: false, + }, + { + name: "Valid UUID - all f's", + uuid: "ffffffff-ffff-ffff-ffff-ffffffffffff", + fieldName: "Client ID", + wantErr: false, + }, + { + name: "Invalid UUID - missing dashes", + uuid: "12345678123412341234123456789abc", + fieldName: "Tenant ID", + wantErr: true, + }, + { + name: "Invalid UUID - wrong dash positions", + uuid: "123456781-234-1234-1234-123456789abc", + fieldName: "Client ID", + wantErr: true, + }, + { + name: "Invalid UUID - too short", + uuid: "12345678-1234-1234-1234-123456789ab", + fieldName: "Subscription ID", + wantErr: true, + }, + { + name: "Invalid UUID - too long", + uuid: "12345678-1234-1234-1234-123456789abcd", + fieldName: "Tenant ID", + wantErr: true, + }, + { + name: "Invalid UUID - contains invalid character g", + uuid: "12345678-1234-1234-1234-123456789abg", + fieldName: "Client ID", + wantErr: true, + }, + { + name: "Invalid UUID - contains special characters", + uuid: "12345678-1234-1234-1234-123456789ab!", + fieldName: "Subscription ID", + wantErr: true, + }, + { + name: "Invalid UUID - empty string", + uuid: "", + fieldName: "Tenant ID", + wantErr: true, + }, + { + name: "Invalid UUID - command injection attempt", + uuid: "12345678-1234-1234-1234-123456789abc; rm -rf /", + fieldName: "Subscription ID", + wantErr: true, + }, + { + name: "Invalid UUID - SQL injection attempt", + uuid: "12345678-1234-1234-1234-123456789abc' OR '1'='1", + fieldName: "Client ID", + wantErr: true, + }, + { + name: "Invalid UUID - spaces", + uuid: "12345678-1234-1234-1234-123456789abc ", + fieldName: "Tenant ID", + wantErr: true, + }, + { + name: "Invalid UUID - newline", + uuid: "12345678-1234-1234-1234-123456789abc\n", + fieldName: "Client ID", + wantErr: true, + }, + } + + for _, tt := range tests { + t.Run(tt.name, func(t *testing.T) { + err := validateAzureUUID(tt.uuid, tt.fieldName) + if tt.wantErr { + assert.Error(t, err) + assert.Contains(t, err.Error(), tt.fieldName) + assert.Contains(t, err.Error(), "invalid") + } else { + assert.NoError(t, err) + } + }) + } +} + +// Tests for validateGCPProjectID function. +func TestValidateGCPProjectID(t *testing.T) { + tests := []struct { + name string + projectID string + wantErr bool + }{ + { + name: "Valid project ID - minimum length (6 chars)", + projectID: "my-pro", + wantErr: false, + }, + { + name: "Valid project ID - maximum length (30 chars)", + projectID: "my-very-long-project-id-123456", + wantErr: false, + }, + { + name: "Valid project ID - all lowercase letters", + projectID: "myproject", + wantErr: false, + }, + { + name: "Valid project ID - with numbers", + projectID: "project123", + wantErr: false, + }, + { + name: "Valid project ID - with hyphens", + projectID: "my-project-123", + wantErr: false, + }, + { + name: "Valid project ID - starts with letter", + projectID: "a12345", + wantErr: false, + }, + { + name: "Valid project ID - ends with number", + projectID: "myproject1", + wantErr: false, + }, + { + name: "Valid project ID - ends with letter", + projectID: "project-a", + wantErr: false, + }, + { + name: "Invalid project ID - too short (5 chars)", + projectID: "short", + wantErr: true, + }, + { + name: "Invalid project ID - too long (31 chars)", + projectID: "my-very-very-long-project-id-31", + wantErr: true, + }, + { + name: "Invalid project ID - starts with number", + projectID: "123project", + wantErr: true, + }, + { + name: "Invalid project ID - starts with hyphen", + projectID: "-myproject", + wantErr: true, + }, + { + name: "Invalid project ID - ends with hyphen", + projectID: "myproject-", + wantErr: true, + }, + { + name: "Invalid project ID - contains uppercase", + projectID: "MyProject", + wantErr: true, + }, + { + name: "Invalid project ID - contains underscore", + projectID: "my_project", + wantErr: true, + }, + { + name: "Invalid project ID - contains space", + projectID: "my project", + wantErr: true, + }, + { + name: "Invalid project ID - contains special characters", + projectID: "my-project!", + wantErr: true, + }, + { + name: "Invalid project ID - empty string", + projectID: "", + wantErr: true, + }, + { + name: "Invalid project ID - command injection attempt", + projectID: "myproject; rm -rf /", + wantErr: true, + }, + { + name: "Invalid project ID - path traversal attempt", + projectID: "../../../etc/passwd", + wantErr: true, + }, + { + name: "Invalid project ID - contains dot", + projectID: "my.project", + wantErr: true, + }, + } + + for _, tt := range tests { + t.Run(tt.name, func(t *testing.T) { + err := validateGCPProjectID(tt.projectID) + if tt.wantErr { + assert.Error(t, err) + assert.Contains(t, err.Error(), "invalid GCP project ID format") + } else { + assert.NoError(t, err) + } + }) + } +} + +// Tests for storeAzureCredentials function. +func TestStoreAzureCredentials(t *testing.T) { + tests := []struct { + mockSetup func(*MockSecretsStore) + validateStore func(*testing.T, *MockSecretsStore) + creds AzureCredentials + name string + stackName string + wantErrMsg string + wantErr bool + }{ + { + name: "Successfully store valid credentials", + stackName: "my-stack", + creds: AzureCredentials{ + TenantID: "12345678-1234-1234-1234-123456789abc", + ClientID: "87654321-4321-4321-4321-fedcba987654", + ClientSecret: "my-secret", + SubscriptionID: "abcdef12-3456-7890-abcd-ef1234567890", + }, + mockSetup: func(m *MockSecretsStore) { + m.listSecretsFunc = func(ctx context.Context, filter string) ([]string, error) { + return []string{"arn:aws:secretsmanager:us-east-1:123456789012:secret:my-stack-AzureCredentials-abc123"}, nil + } + }, + wantErr: false, + validateStore: func(t *testing.T, m *MockSecretsStore) { + assert.Equal(t, "my-stack-AzureCredentials", m.listSecretsFilter) + assert.Len(t, m.updatedSecrets, 1) + + secretID := "arn:aws:secretsmanager:us-east-1:123456789012:secret:my-stack-AzureCredentials-abc123" + secretValue, ok := m.updatedSecrets[secretID] + assert.True(t, ok, "Secret should be stored") + + var storedCreds AzureCredentials + err := json.Unmarshal([]byte(secretValue), &storedCreds) + require.NoError(t, err) + assert.Equal(t, "12345678-1234-1234-1234-123456789abc", storedCreds.TenantID) + assert.Equal(t, "87654321-4321-4321-4321-fedcba987654", storedCreds.ClientID) + assert.Equal(t, "my-secret", storedCreds.ClientSecret) + assert.Equal(t, "abcdef12-3456-7890-abcd-ef1234567890", storedCreds.SubscriptionID) + }, + }, + { + name: "Store credentials when secret ARN not found", + stackName: "my-stack", + creds: AzureCredentials{ + TenantID: "12345678-1234-1234-1234-123456789abc", + ClientID: "87654321-4321-4321-4321-fedcba987654", + ClientSecret: "my-secret", + SubscriptionID: "abcdef12-3456-7890-abcd-ef1234567890", + }, + mockSetup: func(m *MockSecretsStore) { + m.listSecretsFunc = func(ctx context.Context, filter string) ([]string, error) { + return []string{}, nil + } + }, + wantErr: false, + validateStore: func(t *testing.T, m *MockSecretsStore) { + assert.Equal(t, "my-stack-AzureCredentials", m.listSecretsFilter) + assert.Len(t, m.updatedSecrets, 1) + + secretValue, ok := m.updatedSecrets["my-stack-AzureCredentials"] + assert.True(t, ok, "Secret should be stored with name") + + var storedCreds AzureCredentials + err := json.Unmarshal([]byte(secretValue), &storedCreds) + require.NoError(t, err) + assert.Equal(t, "12345678-1234-1234-1234-123456789abc", storedCreds.TenantID) + }, + }, + { + // No mockSetup needed for error-path cases: credential validation runs + // before any store operation, so UpdateSecret is never called. + name: "Error when tenant ID is missing", + stackName: "my-stack", + creds: AzureCredentials{ + TenantID: "", + ClientID: "87654321-4321-4321-4321-fedcba987654", + ClientSecret: "my-secret", + SubscriptionID: "abcdef12-3456-7890-abcd-ef1234567890", + }, + wantErr: true, + wantErrMsg: "all credentials are required", + }, + { + name: "Error when client ID is missing", + stackName: "my-stack", + creds: AzureCredentials{ + TenantID: "12345678-1234-1234-1234-123456789abc", + ClientID: "", + ClientSecret: "my-secret", + SubscriptionID: "abcdef12-3456-7890-abcd-ef1234567890", + }, + wantErr: true, + wantErrMsg: "all credentials are required", + }, + { + name: "Error when client secret is missing", + stackName: "my-stack", + creds: AzureCredentials{ + TenantID: "12345678-1234-1234-1234-123456789abc", + ClientID: "87654321-4321-4321-4321-fedcba987654", + ClientSecret: "", + SubscriptionID: "abcdef12-3456-7890-abcd-ef1234567890", + }, + wantErr: true, + wantErrMsg: "all credentials are required", + }, + { + name: "Error when subscription ID is missing", + stackName: "my-stack", + creds: AzureCredentials{ + TenantID: "12345678-1234-1234-1234-123456789abc", + ClientID: "87654321-4321-4321-4321-fedcba987654", + ClientSecret: "my-secret", + SubscriptionID: "", + }, + wantErr: true, + wantErrMsg: "all credentials are required", + }, + { + name: "Error when UpdateSecret fails", + stackName: "my-stack", + creds: AzureCredentials{ + TenantID: "12345678-1234-1234-1234-123456789abc", + ClientID: "87654321-4321-4321-4321-fedcba987654", + ClientSecret: "my-secret", + SubscriptionID: "abcdef12-3456-7890-abcd-ef1234567890", + }, + mockSetup: func(m *MockSecretsStore) { + m.updateSecretFunc = func(ctx context.Context, secretID string, secretValue string) error { + return errors.New("failed to update secret") + } + }, + wantErr: true, + wantErrMsg: "failed to store credentials in Secrets Manager", + }, + } + + for _, tt := range tests { + t.Run(tt.name, func(t *testing.T) { + ctx := context.Background() + mockStore := NewMockSecretsStore() + + if tt.mockSetup != nil { + tt.mockSetup(mockStore) + } + + err := storeAzureCredentials(ctx, mockStore, tt.stackName, tt.creds) + + if tt.wantErr { + assert.Error(t, err) + if tt.wantErrMsg != "" { + assert.Contains(t, err.Error(), tt.wantErrMsg) + } + } else { + assert.NoError(t, err) + if tt.validateStore != nil { + tt.validateStore(t, mockStore) + } + } + }) + } +} + +// Tests for storeGCPCredentials function. +func TestStoreGCPCredentials(t *testing.T) { + // private_key validation is presence-only; the key content is not parsed or + // validated as a real PEM block by storeGCPCredentials. + validGCPJSON := `{ + "type": "service_account", + "project_id": "my-project", + "private_key_id": "key123", + "private_key": "-----BEGIN PRIVATE KEY-----\nMIIEvQIBADANBg...\n-----END PRIVATE KEY-----\n", + "client_email": "cudly@my-project.iam.gserviceaccount.com", + "client_id": "123456789", + "auth_uri": "https://accounts.google.com/o/oauth2/auth", + "token_uri": "https://oauth2.googleapis.com/token" + }` + + tests := []struct { + mockSetup func(*MockSecretsStore) + validateStore func(*testing.T, *MockSecretsStore) + name string + stackName string + credsJSON string + wantErrMsg string + wantErr bool + }{ + { + name: "Successfully store valid GCP credentials", + stackName: "my-stack", + credsJSON: validGCPJSON, + mockSetup: func(m *MockSecretsStore) { + m.listSecretsFunc = func(ctx context.Context, filter string) ([]string, error) { + return []string{"arn:aws:secretsmanager:us-east-1:123456789012:secret:my-stack-GCPCredentials-xyz789"}, nil + } + }, + wantErr: false, + validateStore: func(t *testing.T, m *MockSecretsStore) { + assert.Equal(t, "my-stack-GCPCredentials", m.listSecretsFilter) + assert.Len(t, m.updatedSecrets, 1) + + secretID := "arn:aws:secretsmanager:us-east-1:123456789012:secret:my-stack-GCPCredentials-xyz789" + secretValue, ok := m.updatedSecrets[secretID] + assert.True(t, ok, "Secret should be stored") + + var storedCreds GCPCredentials + err := json.Unmarshal([]byte(secretValue), &storedCreds) + require.NoError(t, err) + assert.Equal(t, "service_account", storedCreds.Type) + assert.Equal(t, "my-project", storedCreds.ProjectID) + assert.Equal(t, "cudly@my-project.iam.gserviceaccount.com", storedCreds.ClientEmail) + }, + }, + { + name: "Store credentials when secret ARN not found", + stackName: "my-stack", + credsJSON: validGCPJSON, + mockSetup: func(m *MockSecretsStore) { + m.listSecretsFunc = func(ctx context.Context, filter string) ([]string, error) { + return []string{}, nil + } + }, + wantErr: false, + validateStore: func(t *testing.T, m *MockSecretsStore) { + assert.Equal(t, "my-stack-GCPCredentials", m.listSecretsFilter) + secretValue, ok := m.updatedSecrets["my-stack-GCPCredentials"] + assert.True(t, ok, "Secret should be stored with name") + + var storedCreds GCPCredentials + err := json.Unmarshal([]byte(secretValue), &storedCreds) + require.NoError(t, err) + assert.Equal(t, "my-project", storedCreds.ProjectID) + }, + }, + { + name: "Error when JSON is invalid", + stackName: "my-stack", + credsJSON: `{invalid json`, + wantErr: true, + wantErrMsg: "failed to parse credentials", + }, + { + name: "Error when type is not service_account", + stackName: "my-stack", + credsJSON: `{ + "type": "user_account", + "project_id": "my-project", + "private_key": "key", + "client_email": "test@example.com" + }`, + wantErr: true, + wantErrMsg: "expected type 'service_account'", + }, + { + name: "Error when project_id is missing", + stackName: "my-stack", + credsJSON: `{ + "type": "service_account", + "private_key": "key", + "client_email": "test@example.com" + }`, + wantErr: true, + wantErrMsg: "missing project_id", + }, + { + name: "Error when client_email is missing", + stackName: "my-stack", + credsJSON: `{ + "type": "service_account", + "project_id": "my-project", + "private_key": "key" + }`, + wantErr: true, + wantErrMsg: "missing client_email", + }, + { + name: "Error when private_key is missing", + stackName: "my-stack", + credsJSON: `{ + "type": "service_account", + "project_id": "my-project", + "client_email": "test@example.com" + }`, + wantErr: true, + wantErrMsg: "missing private_key", + }, + { + name: "Error when UpdateSecret fails", + stackName: "my-stack", + credsJSON: validGCPJSON, + mockSetup: func(m *MockSecretsStore) { + m.updateSecretFunc = func(ctx context.Context, secretID string, secretValue string) error { + return errors.New("failed to update secret") + } + }, + wantErr: true, + wantErrMsg: "failed to store credentials in Secrets Manager", + }, + } + + for _, tt := range tests { + t.Run(tt.name, func(t *testing.T) { + ctx := context.Background() + mockStore := NewMockSecretsStore() + + if tt.mockSetup != nil { + tt.mockSetup(mockStore) + } + + err := storeGCPCredentials(ctx, mockStore, tt.stackName, tt.credsJSON) + + if tt.wantErr { + assert.Error(t, err) + if tt.wantErrMsg != "" { + assert.Contains(t, err.Error(), tt.wantErrMsg) + } + } else { + assert.NoError(t, err) + if tt.validateStore != nil { + tt.validateStore(t, mockStore) + } + } + }) + } +} diff --git a/cmd/cudly-mcp/main.go b/cmd/cudly-mcp/main.go new file mode 100644 index 000000000..3d5a4d2c8 --- /dev/null +++ b/cmd/cudly-mcp/main.go @@ -0,0 +1,42 @@ +// Command cudly-mcp runs the CUDly MCP server on stdio, exposing CUDly's +// RI/SP/CUD search and purchase tools to any MCP client (e.g. Claude Code +// via ~/.claude/mcp.json). See mcp/README.md for setup and usage. +// +// This binary is intentionally separate from the ri-helper CLI (cmd/main.go): +// it is a local/desktop MCP server, never a Lambda handler, and is kept out +// of iac/ and terraform/ (see mcp/README.md "Deployment model"). +package main + +import ( + "context" + "log" + "os" + + gosdk "github.com/modelcontextprotocol/go-sdk/mcp" + + cudlymcp "github.com/LeanerCloud/CUDly/mcp" + _ "github.com/LeanerCloud/CUDly/providers/aws" + _ "github.com/LeanerCloud/CUDly/providers/azure" + _ "github.com/LeanerCloud/CUDly/providers/gcp" +) + +// Version is overridable at build time via: +// +// go build -ldflags "-X main.Version=1.2.3" ./cmd/cudly-mcp +// +// `make build-mcp` sets it from the same $(LDFLAGS)/$(VERSION) the other +// binaries (build-server, ...) use, so a release build's tag flows through to +// the MCP initialize response's Implementation.Version. +var Version = "dev" + +func main() { + server, err := cudlymcp.NewServer(Version) + if err != nil { + log.Fatalf("cudly-mcp: failed to build server: %v", err) + } + + if err := server.Run(context.Background(), &gosdk.StdioTransport{}); err != nil { + log.Printf("cudly-mcp: server exited with error: %v", err) + os.Exit(1) + } +} diff --git a/cmd/cudly-mcp/main_test.go b/cmd/cudly-mcp/main_test.go new file mode 100644 index 000000000..6e9ecd82b --- /dev/null +++ b/cmd/cudly-mcp/main_test.go @@ -0,0 +1,351 @@ +package main + +import ( + "bytes" + "context" + "encoding/json" + "log" + "os" + "os/exec" + "path/filepath" + "strings" + "testing" + "time" + + "github.com/LeanerCloud/CUDly/pkg/common" + gosdk "github.com/modelcontextprotocol/go-sdk/mcp" + "github.com/stretchr/testify/assert" + "github.com/stretchr/testify/require" + "golang.org/x/sys/unix" + + cudlymcp "github.com/LeanerCloud/CUDly/mcp" + "github.com/LeanerCloud/CUDly/mcp/tools" +) + +const runAsMCPServerEnv = "CUDLY_MCP_TEST_HELPER_PROCESS" + +func TestMain(m *testing.M) { + if os.Getenv(runAsMCPServerEnv) == "1" { + main() + os.Exit(0) + } + + auditDir, err := os.MkdirTemp("", "cudly-mcp-audit-testmain") + if err != nil { + log.Printf("create MCP audit test directory: %v", err) + os.Exit(1) + } + if err := os.Setenv(tools.EnvAuditLog, filepath.Join(auditDir, "mcp-audit.jsonl")); err != nil { + log.Printf("set MCP audit log test path: %v", err) + if cleanupErr := os.RemoveAll(auditDir); cleanupErr != nil { + log.Printf("remove MCP audit test directory after setup failure: %v", cleanupErr) + } + os.Exit(1) + } + + code := m.Run() + os.RemoveAll(auditDir) + os.Exit(code) +} + +// isolateFromAmbientAWS points the AWS SDK at deliberately nonexistent +// profile/config/credentials so config.LoadDefaultConfig cannot resolve any +// real credentials -- neither from a dev machine's ~/.aws files nor from the +// network (IMDS, ECS/EKS container credential endpoints, web identity). This +// is required for TestRealPurchasePastProviderRegistration below: that test +// drives a real (non-dry-run) purchase call, so it must be impossible for it +// to reach an actual AWS account or make an actual network call, in this or +// any other environment the test happens to run in. +func isolateFromAmbientAWS(t *testing.T) { + t.Helper() + t.Setenv("AWS_PROFILE", "cudly-mcp-regression-test-nonexistent-profile") + t.Setenv("AWS_SHARED_CREDENTIALS_FILE", filepath.Join(t.TempDir(), "no-credentials")) + t.Setenv("AWS_CONFIG_FILE", filepath.Join(t.TempDir(), "no-config")) + t.Setenv("AWS_ACCESS_KEY_ID", "") + t.Setenv("AWS_SECRET_ACCESS_KEY", "") + t.Setenv("AWS_SESSION_TOKEN", "") + t.Setenv("AWS_EC2_METADATA_DISABLED", "true") + t.Setenv("AWS_CONTAINER_CREDENTIALS_RELATIVE_URI", "") + t.Setenv("AWS_CONTAINER_CREDENTIALS_FULL_URI", "") + t.Setenv("AWS_ROLE_ARN", "") + t.Setenv("AWS_WEB_IDENTITY_TOKEN_FILE", "") +} + +func holdMCPAuditLog(t *testing.T, path string) func() { + t.Helper() + f, err := os.OpenFile(path, os.O_CREATE|os.O_RDWR, 0o600) + require.NoError(t, err) + release := func() { + if f != nil { + assert.NoError(t, f.Close()) + f = nil + } + } + t.Cleanup(release) + require.NoError(t, unix.Flock(int(f.Fd()), unix.LOCK_EX|unix.LOCK_NB)) + return release +} + +func mcpChildEnv(auditPath string) []string { + childEnv := make([]string, 0, len(os.Environ())+3) + for _, entry := range os.Environ() { + if strings.HasPrefix(entry, tools.EnvAuditLog+"=") || strings.HasPrefix(entry, tools.EnvEnableRealPurchases+"=") { + continue + } + childEnv = append(childEnv, entry) + } + return append(childEnv, + runAsMCPServerEnv+"=1", + tools.EnvAuditLog+"="+auditPath, + tools.EnvEnableRealPurchases+"=", + ) +} + +func callMCPPreview(t *testing.T, session *gosdk.ClientSession) { + t.Helper() + ctx, cancel := context.WithTimeout(context.Background(), 10*time.Second) + defer cancel() + result, err := session.CallTool(ctx, &gosdk.CallToolParams{ + Name: "cudly_aws_ec2_ri_purchase", + Arguments: map[string]any{ + "region": "us-east-1", + "instance_type": "m5.large", + "count": 1, + "term_years": 1, + "payment_option": "no-upfront", + "aws_profile": "cudly-mcp-synthetic-profile", + "dry_run": true, + "confirm": false, + }, + }) + require.NoError(t, err) + require.False(t, result.IsError) + structured, err := json.Marshal(result.StructuredContent) + require.NoError(t, err) + var response tools.PurchaseResponse + require.NoError(t, json.Unmarshal(structured, &response)) + require.True(t, response.Success) + require.True(t, response.DryRun) +} + +func TestMainAuditLockTimeout(t *testing.T) { + t.Run("startup", func(t *testing.T) { + path := filepath.Join(t.TempDir(), "mcp-audit.jsonl") + original := []byte("existing-record\n") + require.NoError(t, os.WriteFile(path, original, 0o600)) + holdMCPAuditLog(t, path) + + exe, err := os.Executable() + require.NoError(t, err) + ctx, cancel := context.WithTimeout(context.Background(), 10*time.Second) + defer cancel() + cmd := exec.CommandContext(ctx, exe) + cmd.Env = mcpChildEnv(path) + var stderr bytes.Buffer + cmd.Stderr = &stderr + client := gosdk.NewClient(&gosdk.Implementation{Name: "test-client"}, nil) + session, err := client.Connect(ctx, &gosdk.CommandTransport{Command: cmd}, nil) + if session != nil { + defer func() { assert.NoError(t, session.Close()) }() + } + require.Error(t, err) + require.NoError(t, ctx.Err()) + assert.Contains(t, stderr.String(), "timed out acquiring audit lock") + data, readErr := os.ReadFile(path) + require.NoError(t, readErr) + require.Equal(t, original, data) + }) + + t.Run("preview", func(t *testing.T) { + isolateFromAmbientAWS(t) + path := filepath.Join(t.TempDir(), "mcp-audit.jsonl") + original := []byte("existing-record\n") + require.NoError(t, os.WriteFile(path, original, 0o600)) + exe, err := os.Executable() + require.NoError(t, err) + + ctx, cancel := context.WithTimeout(context.Background(), 20*time.Second) + defer cancel() + cmd := exec.CommandContext(ctx, exe) + cmd.Env = mcpChildEnv(path) + var stderr bytes.Buffer + cmd.Stderr = &stderr + client := gosdk.NewClient(&gosdk.Implementation{Name: "test-client"}, nil) + session, err := client.Connect(ctx, &gosdk.CommandTransport{Command: cmd}, nil) + if session != nil { + defer func() { assert.NoError(t, session.Close()) }() + } + require.NoError(t, err) + releaseHolder := holdMCPAuditLog(t, path) + callMCPPreview(t, session) + require.NoError(t, session.Close()) + require.NoError(t, ctx.Err()) + assert.Contains(t, stderr.String(), "timed out acquiring audit lock") + data, readErr := os.ReadFile(path) + require.NoError(t, readErr) + require.Equal(t, original, data) + releaseHolder() + + freshCmd := exec.CommandContext(ctx, exe) + freshCmd.Env = mcpChildEnv(path) + var freshStderr bytes.Buffer + freshCmd.Stderr = &freshStderr + freshClient := gosdk.NewClient(&gosdk.Implementation{Name: "test-client"}, nil) + freshSession, err := freshClient.Connect(ctx, &gosdk.CommandTransport{Command: freshCmd}, nil) + if freshSession != nil { + defer func() { assert.NoError(t, freshSession.Close()) }() + } + require.NoError(t, err) + callMCPPreview(t, freshSession) + require.NoError(t, freshSession.Close()) + require.NoError(t, ctx.Err()) + + data, readErr = os.ReadFile(path) + require.NoError(t, readErr) + require.True(t, bytes.HasPrefix(data, original)) + lines := bytes.Split(bytes.TrimSuffix(data, []byte{'\n'}), []byte{'\n'}) + require.Len(t, lines, 2) + var record common.AuditRecord + require.NoError(t, json.Unmarshal(lines[1], &record)) + assert.Equal(t, "skipped", record.Status) + assert.True(t, record.DryRun) + assert.Equal(t, "cudly-mcp-synthetic-profile", record.CredentialScope) + assert.Equal(t, common.PurchaseSourceMCP, record.Source) + assert.Equal(t, common.ProviderAWS, record.Provider) + assert.Equal(t, string(common.ServiceEC2), record.Service) + }) +} + +func TestMainRejectsStdoutAuditLogBeforeProtocolTraffic(t *testing.T) { + info, err := os.Stat("/dev/stdout") + if err != nil { + t.Skipf("/dev/stdout unavailable: %v", err) + } + if info.Mode().IsRegular() { + t.Skip("/dev/stdout is a regular file in this environment") + } + + exe, err := os.Executable() + require.NoError(t, err) + + ctx, cancel := context.WithTimeout(context.Background(), 10*time.Second) + defer cancel() + + cmd := exec.CommandContext(ctx, exe) + childEnv := make([]string, 0, len(os.Environ())+2) + for _, entry := range os.Environ() { + if strings.HasPrefix(entry, tools.EnvAuditLog+"=") { + continue + } + childEnv = append(childEnv, entry) + } + cmd.Env = append(childEnv, runAsMCPServerEnv+"=1", tools.EnvAuditLog+"=/dev/stdout") + var stderr bytes.Buffer + cmd.Stderr = &stderr + + client := gosdk.NewClient(&gosdk.Implementation{Name: "test-client"}, nil) + session, err := client.Connect(ctx, &gosdk.CommandTransport{ + Command: cmd, + TerminateDuration: time.Second, + }, nil) + if err == nil { + _, err = session.CallTool(ctx, &gosdk.CallToolParams{ + Name: "cudly_aws_ec2_ri_purchase", + Arguments: map[string]any{ + "region": "us-east-1", + "instance_type": "m5.large", + "count": 1, + "term_years": 1, + "payment_option": "no-upfront", + }, + }) + closeErr := session.Close() + if err == nil { + err = closeErr + } + } + + require.Error(t, err) + assert.NotContains(t, err.Error(), "invalid message version tag") + assert.NoError(t, ctx.Err()) + stderrText := stderr.String() + assert.Contains(t, stderrText, "failed to build server") + assert.Contains(t, stderrText, "non-regular audit log target") +} + +// TestRealPurchasePastProviderRegistration is the regression guard for the +// bug this file's blank imports fix: cudly-mcp never imported +// providers/aws|azure|gcp, so their init()-registered factories were never +// added to provider.CreateProvider's registry, and every real (non-dry-run) +// purchase failed at ResolveClient with "provider aws is not registered" +// before ever reaching AWS. +// +// This test MUST live in package main under cmd/cudly-mcp/ -- go test +// ./mcp/... does not catch this bug even with the blank imports reverted, +// because a test binary for the mcp or mcp/tools package never pulls in +// cmd/cudly-mcp's imports. Only a test in this package has the blank +// imports in its own dependency graph, so only here does reverting them +// actually flip provider.CreateProvider("aws") back to unregistered. +// +// The test asserts the purchase attempt gets PAST registration and fails for +// a completely different, credentials-shaped reason ("AWS is not +// configured", from providers/aws/provider.go's GetServiceClient) instead of +// "not registered". It never reaches AWS: isolateFromAmbientAWS makes +// config.LoadDefaultConfig fail to resolve the (deliberately nonexistent) +// named profile before any credential lookup or network call happens. +func TestRealPurchasePastProviderRegistration(t *testing.T) { + isolateFromAmbientAWS(t) + // This test's whole point is driving a real (non-dry-run) purchase call + // far enough to observe what happens after provider registration, so it + // must clear the operator-side EnvEnableRealPurchases gate too, or every + // assertion below would instead observe the gate's own refusal. + t.Setenv(tools.EnvEnableRealPurchases, "1") + + server, err := cudlymcp.NewServer("test-regression") + require.NoError(t, err) + + clientTransport, serverTransport := gosdk.NewInMemoryTransports() + + ctx, cancel := context.WithTimeout(context.Background(), 10*time.Second) + defer cancel() + + go func() { + _ = server.Run(ctx, serverTransport) + }() + + client := gosdk.NewClient(&gosdk.Implementation{Name: "test-client"}, nil) + session, err := client.Connect(ctx, clientTransport, nil) + require.NoError(t, err) + defer session.Close() + + result, err := session.CallTool(ctx, &gosdk.CallToolParams{ + Name: "cudly_aws_ec2_ri_purchase", + Arguments: map[string]any{ + "region": "us-east-1", + "instance_type": "m5.large", + "count": 1, + "term_years": 1, + "payment_option": "no-upfront", + "dry_run": false, + "confirm": true, + }, + }) + require.NoError(t, err, "CallTool itself must not return a transport-level error") + require.True(t, result.IsError, "a failed real purchase must surface as a tool error, not a transport error") + + // Guard both the index and the type assertion: a bare + // result.Content[0].(*gosdk.TextContent) panics on an empty Content slice + // or a non-text block, which aborts the whole package's test run instead + // of failing this one assertion readably. + require.NotEmpty(t, result.Content, "tool error result must carry at least one content block") + textContent, ok := result.Content[0].(*gosdk.TextContent) + require.True(t, ok, "first content block must be text, got %T", result.Content[0]) + text := textContent.Text + assert.NotContains(t, strings.ToLower(text), "not registered", + "provider must be registered for the cudly-mcp binary: got %q", text) + // providers/aws/provider.go's GetServiceClient (via AWSProvider.IsConfigured) + // returns exactly this string when config.LoadDefaultConfig cannot resolve + // the requested profile -- observed and confirmed stable in this test run. + assert.Contains(t, text, "AWS is not configured", + "expected a credentials/config-shaped failure once past registration: got %q", text) +} diff --git a/cmd/cudly-mcp/version_test.go b/cmd/cudly-mcp/version_test.go new file mode 100644 index 000000000..8f50688af --- /dev/null +++ b/cmd/cudly-mcp/version_test.go @@ -0,0 +1,52 @@ +package main + +// Linker injection is exercised only by building and running a subprocess. + +import ( + "context" + "os" + "os/exec" + "path/filepath" + "testing" + "time" + + gosdk "github.com/modelcontextprotocol/go-sdk/mcp" + "github.com/stretchr/testify/require" +) + +const ( + buildTimeout = 5 * time.Minute + runTimeout = 15 * time.Second + injectedTestVersion = "v0.0.0-version-test" +) + +func TestBuiltBinaryReportsInjectedVersion(t *testing.T) { + binPath := os.Getenv("CUDLY_MCP_TEST_BINARY") + if binPath == "" { + binPath = filepath.Join(t.TempDir(), "cudly-mcp") + ctx, cancel := context.WithTimeout(context.Background(), buildTimeout) + defer cancel() + + cmd := exec.CommandContext(ctx, "go", "build", + "-ldflags", "-X main.Version="+injectedTestVersion, + "-o", binPath, ".") + cmd.Dir = "." + out, err := cmd.CombinedOutput() + require.NoErrorf(t, err, "build cmd/cudly-mcp with version ldflags:\n%s", out) + } + + ctx, cancel := context.WithTimeout(context.Background(), runTimeout) + defer cancel() + + transport := &gosdk.CommandTransport{Command: exec.CommandContext(ctx, binPath)} + client := gosdk.NewClient(&gosdk.Implementation{Name: "version-test-client"}, nil) + session, err := client.Connect(ctx, transport, nil) + require.NoError(t, err, "connect to the built binary over stdio") + defer session.Close() + + result := session.InitializeResult() + require.NotNil(t, result, "InitializeResult") + require.NotNil(t, result.ServerInfo, "InitializeResult.ServerInfo") + require.Equal(t, injectedTestVersion, result.ServerInfo.Version, + "built binary must report the injected version") +} diff --git a/cmd/effective_dry_run_test.go b/cmd/effective_dry_run_test.go new file mode 100644 index 000000000..5e2195c77 --- /dev/null +++ b/cmd/effective_dry_run_test.go @@ -0,0 +1,43 @@ +package main + +import "testing" + +// TestEffectiveDryRun documents the single-flag purchase contract: a run is a +// dry run unless the user opts into real purchases with --purchase. This guards +// the original regression (issue surfaced on #1364) where --purchase alone +// silently stayed in dry-run because a separate --dry-run flag defaulted to +// true. That flag has since been removed; --purchase is now the only control, +// so moving money is always an explicit opt-in and a bare run is always safe. +func TestEffectiveDryRun(t *testing.T) { + tests := []struct { + name string + actualPurchase bool // --purchase + want bool + }{ + {name: "bare invocation is dry-run", actualPurchase: false, want: true}, + {name: "--purchase executes real purchases", actualPurchase: true, want: false}, + } + + for _, tt := range tests { + t.Run(tt.name, func(t *testing.T) { + got := effectiveDryRun(Config{ActualPurchase: tt.actualPurchase}) + if got != tt.want { + t.Errorf("effectiveDryRun(ActualPurchase=%v) = %v, want %v", + tt.actualPurchase, got, tt.want) + } + }) + } +} + +// TestDryRunFlagRemoved guards against reintroducing the --dry-run flag. It was +// removed because it was a footgun: as a default-true flag it silently +// suppressed real purchases even with --purchase (the #1364 regression), and +// once its default was flipped to false it became a redundant "force dry-run +// even with --purchase" override that only muddied the single-flag contract. +// --purchase is now the sole purchase control; a bare run is always a dry run. +func TestDryRunFlagRemoved(t *testing.T) { + if f := rootCmd.Flags().Lookup("dry-run"); f != nil { + t.Errorf("--dry-run flag is registered again (default %q); it was intentionally removed. "+ + "--purchase is the only purchase control; a bare run is always a dry run.", f.DefValue) + } +} diff --git a/cmd/gen-permissions/main.go b/cmd/gen-permissions/main.go new file mode 100644 index 000000000..ef5bf8730 --- /dev/null +++ b/cmd/gen-permissions/main.go @@ -0,0 +1,103 @@ +// gen-permissions generates frontend/src/permissions.generated.ts from the +// backend's DefaultAdminPermissions / DefaultUserPermissions / +// DefaultReadOnlyPermissions / DefaultPurchaserPermissions constants in +// internal/auth/types.go. +// +// The generated file is imported by the hand-written +// frontend/src/permissions.ts wrapper so the small data surface that +// mirrors backend state (the three default permission sets) stays in +// lockstep with the Go source of truth. A pre-commit hook re-runs this +// binary and `git diff --exit-code` catches a stale committed file. +// +// Run from repo root: +// +// go run ./cmd/gen-permissions +// +// Re-running produces deterministic output: entries are sorted by +// (action, resource) before emission. +package main + +import ( + "bytes" + "fmt" + "os" + "path/filepath" + "sort" + + "github.com/LeanerCloud/CUDly/internal/auth" +) + +const outputRelPath = "frontend/src/permissions.generated.ts" + +// permEntry is a sortable mirror of auth.Permission for stable emission. +type permEntry struct { + Action string + Resource string +} + +func collect(perms []auth.Permission) []permEntry { + out := make([]permEntry, 0, len(perms)) + for _, p := range perms { + out = append(out, permEntry{Action: p.Action, Resource: p.Resource}) + } + sort.Slice(out, func(i, j int) bool { + if out[i].Action != out[j].Action { + return out[i].Action < out[j].Action + } + return out[i].Resource < out[j].Resource + }) + return out +} + +func render(name string, entries []permEntry, buf *bytes.Buffer) { + fmt.Fprintf(buf, "export const %s: ReadonlySet = new Set([\n", name) + for _, e := range entries { + fmt.Fprintf(buf, " '%s:%s',\n", e.Action, e.Resource) + } + fmt.Fprintln(buf, "]);") +} + +func main() { + var buf bytes.Buffer + buf.WriteString(`// CODE GENERATED by ` + "`go run ./cmd/gen-permissions`" + `. DO NOT EDIT MANUALLY. +// +// Source of truth: internal/auth/types.go (DefaultAdminPermissions, +// DefaultUserPermissions, DefaultReadOnlyPermissions, +// DefaultPurchaserPermissions). To regenerate after editing the Go +// defaults, run: +// +// go run ./cmd/gen-permissions +// +// The pre-commit hook 'permissions-codegen' re-runs this generator and +// 'git diff --exit-code' on this file; CI fails if the committed copy is +// stale. +// +// Entries are sorted by (action, resource) for a stable diff. Imported +// by ./permissions.ts which adds the hand-written closed-union types and +// the canAccess / isAdmin / getRolePermissions helpers. + +`) + + render("ADMIN_PERMS", collect(auth.DefaultAdminPermissions()), &buf) + buf.WriteString("\n") + render("USER_PERMS", collect(auth.DefaultUserPermissions()), &buf) + buf.WriteString("\n") + render("READONLY_PERMS", collect(auth.DefaultReadOnlyPermissions()), &buf) + buf.WriteString("\n") + render("PURCHASER_PERMS", collect(auth.DefaultPurchaserPermissions()), &buf) + + // Resolve the output path relative to the repo root. The generator is + // always invoked from the repo root (the comment block on the package + // documents this), so a relative path is correct. We still clean it + // for safety. + outPath := filepath.Clean(outputRelPath) + // 0600 satisfies gosec G306; the file is regenerated on demand and + // also committed to git (where git stores the blob content, not the + // fs permission), so anyone who runs the build pipeline produces a + // fresh local copy. + if err := os.WriteFile(outPath, buf.Bytes(), 0o600); err != nil { + fmt.Fprintf(os.Stderr, "gen-permissions: write %s: %v\n", outPath, err) + os.Exit(1) + } + fmt.Printf("gen-permissions: wrote %s (%d bytes)\n", outPath, buf.Len()) +} diff --git a/cmd/helpers.go b/cmd/helpers.go new file mode 100644 index 000000000..a88c6ce26 --- /dev/null +++ b/cmd/helpers.go @@ -0,0 +1,256 @@ +package main + +import ( + "bufio" + "context" + "fmt" + "log" + "os" + "strings" + "sync" + + "github.com/LeanerCloud/CUDly/pkg/common" + "github.com/LeanerCloud/CUDly/pkg/recfilter" + "github.com/aws/aws-sdk-go-v2/aws" + "github.com/aws/aws-sdk-go-v2/service/organizations" + "golang.org/x/term" +) + +// Constants for purchase processing. +const ( + // PurchaseDelaySeconds is the delay between consecutive purchases to avoid rate limiting. + PurchaseDelaySeconds = 2 + + // DefaultDuplicateCheckLookbackHours is re-exported from pkg/recfilter. + DefaultDuplicateCheckLookbackHours = recfilter.DefaultDuplicateCheckLookbackHours +) + +// AppLogger is a simple logger for application output. +var AppLogger = log.New(os.Stdout, "", 0) + +// OrganizationsAPI interface for describing accounts. +type OrganizationsAPI interface { + DescribeAccount(ctx context.Context, params *organizations.DescribeAccountInput, optFns ...func(*organizations.Options)) (*organizations.DescribeAccountOutput, error) +} + +// AccountAliasGetter is an interface for getting account aliases. +type AccountAliasGetter interface { + GetAccountAlias(ctx context.Context, accountID string) string +} + +// AccountAliasCache caches account ID to alias mappings. +type AccountAliasCache struct { + orgClient OrganizationsAPI + cache map[string]string + mu sync.RWMutex +} + +// NewAccountAliasCache creates a new account alias cache. +func NewAccountAliasCache(cfg aws.Config) *AccountAliasCache { + return &AccountAliasCache{ + cache: make(map[string]string), + orgClient: organizations.NewFromConfig(cfg), + } +} + +// NewAccountAliasCacheWithClient creates a new account alias cache with a custom client +// This is useful for testing with mocked clients. +func NewAccountAliasCacheWithClient(orgClient OrganizationsAPI) *AccountAliasCache { + return &AccountAliasCache{ + cache: make(map[string]string), + orgClient: orgClient, + } +} + +// GetAccountAlias returns the account alias for an account ID. +func (c *AccountAliasCache) GetAccountAlias(ctx context.Context, accountID string) string { + if accountID == "" { + return "" + } + + c.mu.RLock() + if alias, ok := c.cache[accountID]; ok { + c.mu.RUnlock() + return alias + } + c.mu.RUnlock() + + // Try to fetch from Organizations + c.mu.Lock() + defer c.mu.Unlock() + + // Double-check after acquiring write lock + if alias, ok := c.cache[accountID]; ok { + return alias + } + + // Try to describe the account + result, err := c.orgClient.DescribeAccount(ctx, &organizations.DescribeAccountInput{ + AccountId: aws.String(accountID), + }) + if err != nil { + c.cache[accountID] = accountID // Use ID as fallback + return accountID + } + + if result.Account != nil && result.Account.Name != nil { + c.cache[accountID] = *result.Account.Name + return *result.Account.Name + } + + c.cache[accountID] = accountID + return accountID +} + +// CalculateTotalInstances calculates the total instance count across recommendations. +func CalculateTotalInstances(recs []common.Recommendation) int { + total := 0 + for _rvc := range recs { + rec := recs[_rvc] + total += rec.Count + } + return total +} + +// ApplyCoverage delegates to recfilter.ApplyCoverage, wiring AppLogger as the +// logging sink. Substantive documentation lives on recfilter.ApplyCoverage. +func ApplyCoverage(recs []common.Recommendation, coverage float64) []common.Recommendation { + return recfilter.ApplyCoverage(recs, coverage, AppLogger.Printf, nil) +} + +// applyCoverage delegates to recfilter.ApplyCoverage, wiring AppLogger as the +// logging sink. Substantive documentation lives on recfilter.ApplyCoverage. +func applyCoverage(recs []common.Recommendation, coverage float64, drops *common.DropSummary) []common.Recommendation { + return recfilter.ApplyCoverage(recs, coverage, AppLogger.Printf, drops) +} + +// ApplyTargetCoverage delegates to recfilter.ApplyTargetCoverage, wiring +// AppLogger as the logging sink. Substantive documentation (the RI/SP sizing +// formulas and the #338 flag-name history) lives on recfilter.ApplyTargetCoverage. +func ApplyTargetCoverage(recs []common.Recommendation, targetPct float64, drops *common.DropSummary) []common.Recommendation { + return recfilter.ApplyTargetCoverage(recs, targetPct, AppLogger.Printf, drops) +} + +// applySizing chooses target-coverage or coverage sizing. +// +// coverage is the effective % to apply when target-coverage is unset +// (the main path passes cfg.Coverage; the CSV path passes csvModeCoverage, +// which substitutes the default 80% with 100% so CSV-driven counts aren't +// silently dropped). +// +// drops accumulates per-reason drop counts for the end-of-run summary. +// Pass nil to skip tracking. +func applySizing(recs []common.Recommendation, cfg Config, coverage float64, drops *common.DropSummary) []common.Recommendation { + if cfg.TargetCoverage > 0 { + return ApplyTargetCoverage(recs, cfg.TargetCoverage, drops) + } + return applyCoverage(recs, coverage, drops) +} + +// ApplyInstanceLimit truncates recs so their total Count does not exceed +// maxInstances. It is a single-shot cap over whatever slice it is handed: the +// caller is responsible for handing it the complete run-wide set, because +// applying it to a subset (one service, one region) caps that subset only and +// multiplies the effective cap by the number of subsets. See +// applyGlobalInstanceLimit in multi_service.go for the run-wide call site. +// +// Recommendations are consumed in slice order, so the caller controls which +// ones survive by ordering the slice (the main path caps the scorer's +// savings-sorted output, keeping the highest-value commitments). +// +// A truncated recommendation has its extensive money fields scaled by the +// discrete count ratio, like every other sizing path (see +// common.ScaleRecommendationCosts). EstimatedSavings and friends are +// whole-row totals for the count the provider proposed, so cutting Count +// alone leaves the row claiming the savings of instances the run will not +// buy. The run summary and the purchase report then overstate the benefit +// of a capped run, which is the wrong direction to be wrong in on a money +// path (#1830). +func ApplyInstanceLimit(recs []common.Recommendation, maxInstances int32) []common.Recommendation { + if maxInstances <= 0 { + return recs + } + + result := make([]common.Recommendation, 0) + remaining := int(maxInstances) + + for _rvc := range recs { + rec := recs[_rvc] + if remaining <= 0 { + break + } + adjusted := rec + // rec.Count > remaining and remaining >= 1 together imply rec.Count + // >= 2, so the denominator is always positive here. A non-positive + // Count can never enter this branch and so is never rescaled: it + // buys nothing, there is nothing to scale down to, and a zero or + // negative denominator would produce NaN or a sign flip. + if rec.Count > remaining { + adjusted = common.ScaleRecommendationCosts(rec, float64(remaining)/float64(rec.Count)) + adjusted.Count = remaining + } + result = append(result, adjusted) + // Only a positive Count consumes budget. Subtracting a non-positive + // Count would credit budget back and let later recommendations push + // the run past the cap. + if adjusted.Count > 0 { + remaining -= adjusted.Count + } + } + return result +} + +// ConfirmPurchase asks the user for confirmation before proceeding. +// totalSavings is the estimated monthly savings from the purchase (not the purchase cost), +// matching the EstimatedSavings column and the "Estimated monthly savings" summary. +// Returns false without prompting if stdin is not a TTY and skipConfirmation is false. +func ConfirmPurchase(totalInstances int, totalSavings float64, skipConfirmation bool) bool { + if skipConfirmation { + return true + } + + if !term.IsTerminal(int(os.Stdin.Fd())) { //nolint:gosec // G115: uintptr->int for file descriptor; FD values are always small positive integers + log.Printf("stdin is not a terminal and --yes was not set; skipping purchase") + return false + } + + fmt.Printf("\n⚠️ About to purchase %d instances with estimated monthly savings: $%.2f\n", totalInstances, totalSavings) + fmt.Print("Do you want to proceed? (yes/no): ") + + reader := bufio.NewReader(os.Stdin) + response, err := reader.ReadString('\n') + if err != nil { + return false + } + + response = strings.TrimSpace(strings.ToLower(response)) + return response == "yes" || response == "y" +} + +// CheckAuditLogWritable reports whether the audit log and its immediate parent +// directories satisfy the read, append, and durability requirements. +// Thin wrapper over common.CheckAuditLogWritable; kept so cmd's existing call sites +// and tests are unchanged. +func CheckAuditLogWritable(path string) error { return common.CheckAuditLogWritable(path) } + +// DuplicateChecker is re-exported from pkg/recfilter so cmd's existing call +// sites and tests are unchanged. +type DuplicateChecker = recfilter.DuplicateChecker + +// NewDuplicateChecker creates a new duplicate checker. Pass 0 to use the +// default lookback period. Logf is wired to log.Printf so the CLI's +// decision trail keeps going to stderr exactly as it does today. +func NewDuplicateChecker(hours int) *DuplicateChecker { + d := recfilter.NewDuplicateChecker(hours) + d.Logf = log.Printf + return d +} + +// GetRecommendationDescription returns a human-readable description. +func GetRecommendationDescription(rec common.Recommendation) string { + desc := fmt.Sprintf("%s %s", rec.Service, rec.ResourceType) + if rec.Details != nil { + desc += " " + rec.Details.GetDetailDescription() + } + return desc +} diff --git a/cmd/helpers_count_override.go b/cmd/helpers_count_override.go new file mode 100644 index 000000000..b4f2e2107 --- /dev/null +++ b/cmd/helpers_count_override.go @@ -0,0 +1,92 @@ +package main + +import ( + "github.com/LeanerCloud/CUDly/pkg/common" +) + +// ApplyCountOverride replaces the count on every count-denominated +// recommendation with overrideCount, rescaling the money that count derives so +// the row describes the quantity the run will actually buy. +// +// EstimatedSavings, CommitmentCost, OnDemandCost and RecurringMonthlyCost are +// whole-row totals for the count the provider proposed, so replacing Count +// alone leaves the row claiming the money of a quantity nobody will buy +// (#1844). The scaling goes through common.ScaleRecommendationCosts, the same +// helper ApplyInstanceLimit and the coverage paths use, so the arithmetic +// cannot drift between the flags. ProjectedCoverage / ProjectedUtilization are +// count-linear but not re-derived here, exactly as after a cap (#1845). +// +// Savings Plans are left entirely alone, Count included. The flag is +// documented as an override for "all selected RIs", an SP commitment is +// dollar-denominated rather than count-denominated, its Count is a fixed +// placeholder 1 set by the parser, and the SP purchase call reads +// HourlyCommitment rather than Count. Scaling an SP by an instance count would +// multiply the dollars actually committed by an unrelated number. +// +// A row with a non-positive Count is left alone for the same reason +// ApplyInstanceLimit never rescales one: there is no denominator to form a +// ratio from, and setting Count without scaling would reintroduce precisely +// the misstatement above. +// +// Scaling down is unambiguous. Scaling up past the quantity the provider's own +// figures cover is an extrapolation: unit prices are linear so CommitmentCost +// stays true, but savings only accrue on hours a matching resource actually +// runs, so units beyond the observed demand cost money and may save nothing. +// Those rows are reported rather than passed off as measured savings. +// +// Both call sites run this before the run-wide --max-instances cap, so a run +// using both flags scales once from the override ratio and once from the cap +// ratio, which compose to a single net ratio against the provider's figures. +func ApplyCountOverride(recs []common.Recommendation, overrideCount int32) []common.Recommendation { + if overrideCount <= 0 { + return recs + } + result := make([]common.Recommendation, len(recs)) + var skippedSP, skippedNonPositive, extrapolated int + for i := range recs { + rec := recs[i] + switch { + case common.IsSavingsPlan(rec.Service): + result[i] = rec + skippedSP++ + case rec.Count <= 0: + result[i] = rec + skippedNonPositive++ + default: + result[i] = common.ScaleRecommendationCosts(rec, float64(overrideCount)/float64(rec.Count)) + result[i].Count = int(overrideCount) + if int(overrideCount) > evidencedCount(rec) { + extrapolated++ + } + } + } + reportCountOverride(overrideCount, skippedSP, skippedNonPositive, extrapolated) + return result +} + +// evidencedCount is the largest count a recommendation's money figures are +// evidence for: the provider's own pre-sizing proposal when it recorded one, +// otherwise the count the row currently carries. RecommendedCount is populated +// only on the AWS RI path, so the fallback is what the --input-csv and +// non-AWS paths use. +func evidencedCount(rec common.Recommendation) int { + if rec.RecommendedCount > rec.Count { + return rec.RecommendedCount + } + return rec.Count +} + +// reportCountOverride names what --override-count could not honor and what it +// honored only by extrapolation. Nothing here is fatal, but every case would +// otherwise change or fail to change a money figure without telling anyone. +func reportCountOverride(overrideCount int32, skippedSP, skippedNonPositive, extrapolated int) { + if skippedSP > 0 { + AppLogger.Printf("⚠️ --override-count left %d Savings Plans recommendation(s) unchanged: an SP commitment is priced in dollars per hour, not in instances, so an instance count cannot size it.\n", skippedSP) + } + if skippedNonPositive > 0 { + AppLogger.Printf("⚠️ --override-count left %d recommendation(s) with a non-positive count unchanged: there is no quantity to rescale their costs from.\n", skippedNonPositive) + } + if extrapolated > 0 { + AppLogger.Printf("⚠️ --override-count=%d exceeds the quantity the provider's figures cover on %d recommendation(s). Their costs scale with the count and stay accurate, but the savings are extrapolated past the observed demand: instances beyond it are billed and may save nothing.\n", overrideCount, extrapolated) + } +} diff --git a/cmd/helpers_count_override_rescale_test.go b/cmd/helpers_count_override_rescale_test.go new file mode 100644 index 000000000..1871aa60d --- /dev/null +++ b/cmd/helpers_count_override_rescale_test.go @@ -0,0 +1,309 @@ +package main + +import ( + "math" + "testing" + + "github.com/LeanerCloud/CUDly/pkg/common" + "github.com/stretchr/testify/assert" + "github.com/stretchr/testify/require" +) + +// countDenominatedRec is the fixture the override tests scale. Every money +// field is a whole-row total for Count, which is exactly the property +// --override-count has to preserve. +func countDenominatedRec(count int) common.Recommendation { + return common.Recommendation{ + Service: common.ServiceEC2, + Region: "us-east-1", + ResourceType: "m5.large", + Count: count, + RecommendedCount: count, + EstimatedSavings: 600, + CommitmentCost: 2400, + OnDemandCost: 3000, + SavingsPercentage: 20, + RecurringMonthlyCost: float64Ptr(200), + } +} + +// assertScaledBy checks that every extensive money field of got is in's value +// times ratio, and that the intensive ones are untouched. The assertion is the +// ratio rather than a literal figure, so it keeps its meaning if the fixture's +// dollar values change. +func assertScaledBy(t *testing.T, in, got common.Recommendation, ratio float64) { + t.Helper() + + assert.InDelta(t, in.EstimatedSavings*ratio, got.EstimatedSavings, 0.0001, + "EstimatedSavings is a whole-row total and must scale with the overridden Count") + assert.InDelta(t, in.CommitmentCost*ratio, got.CommitmentCost, 0.0001, + "CommitmentCost is a whole-row total and must scale with the overridden Count") + assert.InDelta(t, in.OnDemandCost*ratio, got.OnDemandCost, 0.0001, + "OnDemandCost is a whole-row total and must scale with the overridden Count") + + require.NotNil(t, got.RecurringMonthlyCost, + "a present monthly breakdown must stay present after the override") + assert.InDelta(t, *in.RecurringMonthlyCost*ratio, *got.RecurringMonthlyCost, 0.0001, + "RecurringMonthlyCost is a whole-row total and must scale with the overridden Count") + + // SavingsPercentage is a ratio of two figures that scale together, so it is + // invariant. Scaling it would be the mirror-image bug. + assert.InDelta(t, in.SavingsPercentage, got.SavingsPercentage, 0.0001, + "SavingsPercentage is intensive and must not be scaled") + + // RecommendedCount is a frozen record of the provider's own proposal. The + // override replaces what we will buy, not what the provider proposed. + assert.Equal(t, in.RecommendedCount, got.RecommendedCount, + "RecommendedCount records the provider's proposal and must survive the override") +} + +// TestApplyCountOverrideRescalesDownscaledRow pins the #1844 invariant in the +// direction the issue calls unambiguous: a row overridden from N to a smaller M +// carries M/N of the money it entered with, so the per-instance rate the row +// implies is unchanged. +func TestApplyCountOverrideRescalesDownscaledRow(t *testing.T) { + const ( + origCount = 100 + override = 5 + ratio = float64(override) / float64(origCount) + ) + + rec := countDenominatedRec(origCount) + perInstanceBefore := rec.EstimatedSavings / float64(rec.Count) + + got := ApplyCountOverride([]common.Recommendation{rec}, override) + + require.Len(t, got, 1) + require.Equal(t, override, got[0].Count, "the override must replace Count") + assertScaledBy(t, rec, got[0], ratio) + + assert.InDelta(t, perInstanceBefore, got[0].EstimatedSavings/float64(got[0].Count), 0.0001, + "savings per instance is the rate the row asserts and must survive the override") +} + +// TestApplyCountOverrideRescalesUpscaledRow covers the direction --max-instances +// never exercises. An override can raise the count above the provider's +// proposal, and leaving the money at the smaller quantity understates both the +// savings and, more dangerously, what the run will be charged. +func TestApplyCountOverrideRescalesUpscaledRow(t *testing.T) { + const ( + origCount = 10 + override = 20 + ratio = float64(override) / float64(origCount) + ) + + rec := countDenominatedRec(origCount) + + got := ApplyCountOverride([]common.Recommendation{rec}, override) + + require.Len(t, got, 1) + require.Equal(t, override, got[0].Count) + assertScaledBy(t, rec, got[0], ratio) + + assert.Greater(t, got[0].CommitmentCost, rec.CommitmentCost, + "buying more than the provider proposed must not report the smaller quantity's cost") +} + +// TestApplyCountOverrideLeavesSavingsPlansUntouched is the both-directions +// assertion, in one call so a fix that rescales every row cannot pass it. An SP +// commitment is priced in dollars per hour rather than in instances: its Count +// is a placeholder the SP purchase call never reads, so sizing it by an +// instance count would multiply the dollars actually committed by an unrelated +// number. +func TestApplyCountOverrideLeavesSavingsPlansUntouched(t *testing.T) { + const override = 20 + + ri := countDenominatedRec(10) + sp := common.Recommendation{ + Service: common.ServiceSavingsPlansCompute, + Region: "us-east-1", + Count: 1, + EstimatedSavings: 500, + CommitmentCost: 1200, + Details: &common.SavingsPlanDetails{ + PlanType: "Compute", + HourlyCommitment: 10, + }, + } + + got := ApplyCountOverride([]common.Recommendation{ri, sp}, override) + + require.Len(t, got, 2) + + // The count-denominated row is rescaled. + require.Equal(t, override, got[0].Count) + assertScaledBy(t, ri, got[0], float64(override)/float64(ri.Count)) + + // The dollar-denominated row is not touched at all, Count included. + assert.Equal(t, 1, got[1].Count, + "an SP is one commitment, not N instances, so the override must not set its Count") + assert.InDelta(t, 500.0, got[1].EstimatedSavings, 0.0001, "an SP's savings must not be scaled by an instance count") + assert.InDelta(t, 1200.0, got[1].CommitmentCost, 0.0001, "an SP's cost must not be scaled by an instance count") + + details, ok := got[1].Details.(*common.SavingsPlanDetails) + require.True(t, ok, "SP details must survive with their concrete type") + assert.InDelta(t, 10.0, details.HourlyCommitment, 0.0001, + "the hourly commitment is what the SP purchase actually buys and must not move with --override-count") +} + +// TestApplyCountOverrideNonPositiveCountIsNotRescaled guards the divide-by-zero +// denominator. A Count of 0 or below carries no per-unit rate to scale from, so +// the row passes through untouched rather than through a ratio computed from a +// zero denominator (Inf) or a negative one (sign flip). Setting Count without +// scaling would reintroduce the very misstatement #1844 is about. +func TestApplyCountOverrideNonPositiveCountIsNotRescaled(t *testing.T) { + recs := []common.Recommendation{ + {ResourceType: "zero-count", Count: 0, EstimatedSavings: 600, OnDemandCost: 100}, + {ResourceType: "negative-count", Count: -5, EstimatedSavings: 300, OnDemandCost: 50}, + } + + got := ApplyCountOverride(recs, 10) + + require.Len(t, got, 2) + for i := range got { + assert.False(t, math.IsNaN(got[i].EstimatedSavings) || math.IsInf(got[i].EstimatedSavings, 0), + "%s: savings must not become NaN/Inf via a non-positive denominator", got[i].ResourceType) + assert.False(t, math.IsNaN(got[i].OnDemandCost) || math.IsInf(got[i].OnDemandCost, 0), + "%s: on-demand cost must not become NaN/Inf via a non-positive denominator", got[i].ResourceType) + } + + assert.Equal(t, 0, got[0].Count, "a row with no quantity to scale from keeps its count") + assert.InDelta(t, 600.0, got[0].EstimatedSavings, 0.0001, "a zero-count row is not rescaled") + assert.InDelta(t, 100.0, got[0].OnDemandCost, 0.0001, "a zero-count row is not rescaled") + assert.Equal(t, -5, got[1].Count, "a row with no quantity to scale from keeps its count") + assert.InDelta(t, 300.0, got[1].EstimatedSavings, 0.0001, "a negative-count row is not rescaled") + assert.InDelta(t, 50.0, got[1].OnDemandCost, 0.0001, "a negative-count row is not rescaled") +} + +// TestApplyCountOverridePreservesNilRecurringMonthlyCost pins the +// absent-versus-zero rule: nil means "the provider returned no monthly +// breakdown" and renders as an em dash rather than $0. The override must not +// turn that into a confident zero. +func TestApplyCountOverridePreservesNilRecurringMonthlyCost(t *testing.T) { + recs := []common.Recommendation{ + { + Service: common.ServiceEC2, + ResourceType: "no-monthly-breakdown", + Count: 100, + EstimatedSavings: 600, + RecurringMonthlyCost: nil, + }, + } + + got := ApplyCountOverride(recs, 5) + + require.Len(t, got, 1) + require.Equal(t, 5, got[0].Count) + assert.Nil(t, got[0].RecurringMonthlyCost, + "a nil monthly cost means absent, not zero, and must survive the override as nil") +} + +// TestApplyCountOverrideDoesNotAliasCallerRecs pins that the caller's slice +// survives the override intact. The pipeline hands the pre-override slice on to +// the cap, whose reporter diffs one against the other, so a shared pointer +// target would corrupt what the operator is shown was reduced. +func TestApplyCountOverrideDoesNotAliasCallerRecs(t *testing.T) { + before := []common.Recommendation{countDenominatedRec(100)} + + got := ApplyCountOverride(before, 5) + + require.Len(t, got, 1) + require.NotNil(t, got[0].RecurringMonthlyCost) + require.NotNil(t, before[0].RecurringMonthlyCost) + + assert.NotSame(t, before[0].RecurringMonthlyCost, got[0].RecurringMonthlyCost, + "the scaled monthly cost must be a fresh pointer, not a write through the caller's") + assert.Equal(t, 100, before[0].Count, "the input rec must not be mutated") + assert.InDelta(t, 600.0, before[0].EstimatedSavings, 0.0001, "the input rec must not be mutated") + assert.InDelta(t, 200.0, *before[0].RecurringMonthlyCost, 0.0001, + "the input rec's pointer target must not be mutated") +} + +// TestApplyCountOverrideReportsExtrapolationPastProviderEvidence pins the +// disclosure that makes scaling up defensible. Costs stay accurate at any +// quantity because unit prices are linear, but savings only accrue on hours a +// matching instance runs, so an override past the demand the provider measured +// produces a savings figure nobody measured. Silently presenting it would be +// the fabricated figure #1844 set out to remove. +// +// The boundary is the provider's own proposal, not the row's current count: a +// row sized down to 80 by --coverage from a proposal of 100 is still inside +// what the provider's figures cover at 100, so overriding back up to 100 is +// interpolation and must stay quiet. +func TestApplyCountOverrideReportsExtrapolationPastProviderEvidence(t *testing.T) { + sizedDownFromProposal := common.Recommendation{ + Service: common.ServiceEC2, + ResourceType: "within-proposal", + Count: 80, + RecommendedCount: 100, + EstimatedSavings: 480, + } + + quiet := captureAppOutput(t, func() { + got := ApplyCountOverride([]common.Recommendation{sizedDownFromProposal}, 100) + require.Len(t, got, 1) + require.Equal(t, 100, got[0].Count) + }) + assert.NotContains(t, quiet, "exceeds the quantity", + "overriding back up to the provider's own proposal is interpolation, not extrapolation") + + loud := captureAppOutput(t, func() { + got := ApplyCountOverride([]common.Recommendation{sizedDownFromProposal}, 101) + require.Len(t, got, 1) + require.Equal(t, 101, got[0].Count) + }) + assert.Contains(t, loud, "exceeds the quantity", + "an override past the provider's proposal extrapolates the savings and must say so") +} + +// TestApplyCountOverrideReportsSkippedRows pins that the two exempt classes are +// named rather than silently passed through. An operator who set +// --override-count and got a differently-sized run than they asked for has to +// be told which rows the flag could not size and why. +func TestApplyCountOverrideReportsSkippedRows(t *testing.T) { + recs := []common.Recommendation{ + {Service: common.ServiceSavingsPlansCompute, Count: 1, EstimatedSavings: 500}, + {Service: common.ServiceEC2, ResourceType: "zero-count", Count: 0, EstimatedSavings: 600}, + } + + out := captureAppOutput(t, func() { + got := ApplyCountOverride(recs, 10) + require.Len(t, got, 2) + }) + + assert.Contains(t, out, "Savings Plans recommendation(s) unchanged", + "an SP the override could not size must be named") + assert.Contains(t, out, "non-positive count unchanged", + "a row with no quantity to rescale from must be named") +} + +// TestApplyCountOverrideThenInstanceLimitScalesOnce pins the ordering both call +// sites establish: the override runs first and the run-wide cap second. The two +// ratios have to compose to a single net ratio against the provider's figures, +// rather than the cap re-deriving its ratio from a base the override already +// corrupted. +// +// The expectation is computed independently of either function, so it cannot +// agree with a wrong implementation. +func TestApplyCountOverrideThenInstanceLimitScalesOnce(t *testing.T) { + const ( + origCount = 100 + override = 20 + capTo = 8 + ) + + rec := countDenominatedRec(origCount) + + overridden := ApplyCountOverride([]common.Recommendation{rec}, override) + require.Len(t, overridden, 1) + require.Equal(t, override, overridden[0].Count) + + capped := ApplyInstanceLimit(overridden, capTo) + require.Len(t, capped, 1) + require.Equal(t, capTo, capped[0].Count, "the cap truncates the overridden count") + + // Net ratio is capTo/origCount: the override's 20/100 composed with the + // cap's 8/20. Anything else means one stage scaled against a wrong base. + netRatio := float64(capTo) / float64(origCount) + assertScaledBy(t, rec, capped[0], netRatio) +} diff --git a/cmd/helpers_instance_limit_rescale_test.go b/cmd/helpers_instance_limit_rescale_test.go new file mode 100644 index 000000000..12358640c --- /dev/null +++ b/cmd/helpers_instance_limit_rescale_test.go @@ -0,0 +1,220 @@ +package main + +import ( + "math" + "testing" + + "github.com/LeanerCloud/CUDly/pkg/common" + "github.com/stretchr/testify/assert" + "github.com/stretchr/testify/require" +) + +// float64Ptr is local to this file so the rescale tests can express "the +// provider returned a monthly breakdown" without borrowing a helper whose +// nil-vs-zero semantics might change elsewhere. +func float64Ptr(v float64) *float64 { return &v } + +// TestApplyInstanceLimitRescalesTruncatedRow pins the #1830 invariant: a row +// the cap truncates from N to M carries M/N of the money it entered with. +// +// The assertion is the ratio, not a literal figure, so the test keeps +// meaning if the fixture's dollar values are ever changed. +func TestApplyInstanceLimitRescalesTruncatedRow(t *testing.T) { + const ( + origCount = 100 + maxKept = 10 + ratio = float64(maxKept) / float64(origCount) + ) + + rec := common.Recommendation{ + Service: common.ServiceEC2, + Region: "us-east-1", + ResourceType: "m5.large", + Count: origCount, + EstimatedSavings: 600, + CommitmentCost: 2400, + OnDemandCost: 3000, + SavingsPercentage: 20, + RecurringMonthlyCost: float64Ptr(200), + } + + got := ApplyInstanceLimit([]common.Recommendation{rec}, maxKept) + + require.Len(t, got, 1) + require.Equal(t, maxKept, got[0].Count, "the cap must truncate Count to the budget") + + assert.InDelta(t, rec.EstimatedSavings*ratio, got[0].EstimatedSavings, 0.0001, + "EstimatedSavings is a whole-row total and must scale with the truncated Count") + assert.InDelta(t, rec.CommitmentCost*ratio, got[0].CommitmentCost, 0.0001, + "CommitmentCost is a whole-row total and must scale with the truncated Count") + assert.InDelta(t, rec.OnDemandCost*ratio, got[0].OnDemandCost, 0.0001, + "OnDemandCost is a whole-row total and must scale with the truncated Count") + + require.NotNil(t, got[0].RecurringMonthlyCost, + "a present monthly breakdown must stay present after truncation") + assert.InDelta(t, *rec.RecurringMonthlyCost*ratio, *got[0].RecurringMonthlyCost, 0.0001, + "RecurringMonthlyCost is a whole-row total and must scale with the truncated Count") + + // SavingsPercentage is a ratio of two figures that scale together, so it + // is invariant under truncation. Scaling it would be the mirror-image bug. + assert.InDelta(t, rec.SavingsPercentage, got[0].SavingsPercentage, 0.0001, + "SavingsPercentage is intensive and must not be scaled") + + // The caller's rec must not be mutated: RecurringMonthlyCost is a pointer + // and a shared target would corrupt the pre-cap slice the reporter diffs + // against. + assert.Equal(t, origCount, rec.Count, "the input rec must not be mutated") + assert.InDelta(t, 600.0, rec.EstimatedSavings, 0.0001, "the input rec must not be mutated") + require.NotNil(t, rec.RecurringMonthlyCost) + assert.InDelta(t, 200.0, *rec.RecurringMonthlyCost, 0.0001, + "the input rec's pointer target must not be mutated") +} + +// TestApplyInstanceLimitLeavesUntruncatedRowsUntouched is the other direction: +// a fix that rescaled every row would satisfy the truncation test above while +// silently shrinking rows that fit inside the budget. +func TestApplyInstanceLimitLeavesUntruncatedRowsUntouched(t *testing.T) { + monthly := float64Ptr(200) + rec := common.Recommendation{ + Service: common.ServiceEC2, + Region: "us-east-1", + ResourceType: "m5.large", + Count: 4, + EstimatedSavings: 600, + CommitmentCost: 2400, + OnDemandCost: 3000, + SavingsPercentage: 20, + RecurringMonthlyCost: monthly, + } + + // A budget strictly larger than the row's Count, so the row fits whole. + got := ApplyInstanceLimit([]common.Recommendation{rec}, 10) + + require.Len(t, got, 1) + assert.Equal(t, 4, got[0].Count, "a row that fits the budget keeps its Count") + assert.InDelta(t, 600.0, got[0].EstimatedSavings, 0.0001, + "an untruncated row must keep its savings; rescaling everything is the mirror-image bug") + assert.InDelta(t, 2400.0, got[0].CommitmentCost, 0.0001, "an untruncated row must keep its commitment cost") + assert.InDelta(t, 3000.0, got[0].OnDemandCost, 0.0001, "an untruncated row must keep its on-demand cost") + require.NotNil(t, got[0].RecurringMonthlyCost) + assert.InDelta(t, 200.0, *got[0].RecurringMonthlyCost, 0.0001, + "an untruncated row must keep its monthly cost") +} + +// TestApplyInstanceLimitPreservesNilRecurringMonthlyCost pins the +// absent-versus-zero rule: nil means "the provider returned no monthly +// breakdown" and the frontend renders it as "—". Truncation must not turn +// that into a confident $0. +func TestApplyInstanceLimitPreservesNilRecurringMonthlyCost(t *testing.T) { + recs := []common.Recommendation{ + { + Service: common.ServiceEC2, + Region: "us-east-1", + ResourceType: "truncated", + Count: 100, + EstimatedSavings: 600, + RecurringMonthlyCost: nil, + }, + } + + got := ApplyInstanceLimit(recs, 10) + + require.Len(t, got, 1) + require.Equal(t, 10, got[0].Count) + assert.Nil(t, got[0].RecurringMonthlyCost, + "a nil monthly cost means absent, not zero, and must survive truncation as nil") +} + +// TestApplyInstanceLimitNonPositiveCountIsNotRescaled guards the divide-by-zero +// denominator. A Count of 0 or below buys nothing and cannot be truncated, so +// it must pass through with its money untouched rather than through a ratio +// computed from a zero denominator (NaN) or a negative one (sign flip). +func TestApplyInstanceLimitNonPositiveCountIsNotRescaled(t *testing.T) { + recs := []common.Recommendation{ + {ResourceType: "zero-count", Count: 0, EstimatedSavings: 600, OnDemandCost: 100}, + {ResourceType: "negative-count", Count: -5, EstimatedSavings: 300, OnDemandCost: 50}, + } + + got := ApplyInstanceLimit(recs, 10) + + require.Len(t, got, 2) + for i := range got { + assert.False(t, math.IsNaN(got[i].EstimatedSavings) || math.IsInf(got[i].EstimatedSavings, 0), + "%s: savings must not become NaN/Inf via a non-positive denominator", got[i].ResourceType) + } + assert.Equal(t, 0, got[0].Count) + assert.InDelta(t, 600.0, got[0].EstimatedSavings, 0.0001, "a zero-count row is not truncated, so it is not rescaled") + assert.InDelta(t, 100.0, got[0].OnDemandCost, 0.0001, "a zero-count row is not truncated, so it is not rescaled") + assert.Equal(t, -5, got[1].Count) + assert.InDelta(t, 300.0, got[1].EstimatedSavings, 0.0001, "a negative-count row is not truncated, so it is not rescaled") + assert.InDelta(t, 50.0, got[1].OnDemandCost, 0.0001, "a negative-count row is not truncated, so it is not rescaled") +} + +// TestApplyInstanceLimitRescalesSavingsPlanHourlyCommitment covers the Savings +// Plan case. The provider parser pins SP recs at Count 1, so they are normally +// undivisible, but the --input-csv path builds Service straight from the CSV +// column (parseCSVRecords), so a file naming a savingsplans service at a count +// above the budget reaches this branch. HourlyCommitment is the SP's actual +// money quantity, so leaving it whole while the cost fields shrink produces +// exactly the internally-inconsistent row #1830 warns about. +// +// --override-count no longer routes SPs here: it leaves them alone precisely +// because an SP is dollar-denominated rather than count-denominated (#1844). +func TestApplyInstanceLimitRescalesSavingsPlanHourlyCommitment(t *testing.T) { + const ratio = 2.0 / 5.0 + + rec := common.Recommendation{ + Service: common.ServiceSavingsPlansCompute, + Region: "us-east-1", + Count: 5, + EstimatedSavings: 500, + Details: &common.SavingsPlanDetails{ + PlanType: "Compute", + HourlyCommitment: 10, + }, + } + + got := ApplyInstanceLimit([]common.Recommendation{rec}, 2) + + require.Len(t, got, 1) + require.Equal(t, 2, got[0].Count) + assert.InDelta(t, 500*ratio, got[0].EstimatedSavings, 0.0001) + + details, ok := got[0].Details.(*common.SavingsPlanDetails) + require.True(t, ok, "SP details must survive truncation with their concrete type") + assert.InDelta(t, 10*ratio, details.HourlyCommitment, 0.0001, + "an SP's hourly commitment must scale with the truncated Count, like every other extensive figure") + + // The caller's Details must not be mutated through the shared pointer. + orig, ok := rec.Details.(*common.SavingsPlanDetails) + require.True(t, ok) + assert.InDelta(t, 10.0, orig.HourlyCommitment, 0.0001, + "the input rec's Details pointer target must not be mutated") +} + +// TestApplyInstanceLimitTotalMatchesSumOfKeptRows is the run-summary invariant: +// what the summary adds up must equal what the run will actually buy. It is +// asserted against an independently computed expectation rather than against +// the function's own output, so it cannot agree with a wrong implementation. +func TestApplyInstanceLimitTotalMatchesSumOfKeptRows(t *testing.T) { + recs := []common.Recommendation{ + {ResourceType: "whole", Count: 6, EstimatedSavings: 500}, // fits whole + {ResourceType: "truncated", Count: 6, EstimatedSavings: 100}, // truncated 6 -> 4 + {ResourceType: "dropped", Count: 6, EstimatedSavings: 900}, // budget exhausted + } + + got := ApplyInstanceLimit(recs, 10) + + require.Len(t, got, 2, "the third row must be dropped once the budget is spent") + assert.Equal(t, 10, CalculateTotalInstances(got), "the cap is a hard budget") + + var total float64 + for i := range got { + total += got[i].EstimatedSavings + } + // 500 for the whole row + 100*(4/6) for the truncated one. The dropped + // row contributes nothing. + want := 500.0 + 100.0*(4.0/6.0) + assert.InDelta(t, want, total, 0.0001, + "the reported total must equal the savings of the instances the run will actually buy") +} diff --git a/cmd/helpers_test.go b/cmd/helpers_test.go new file mode 100644 index 000000000..357d01292 --- /dev/null +++ b/cmd/helpers_test.go @@ -0,0 +1,1464 @@ +package main + +import ( + "context" + "errors" + "math" + "strings" + "sync" + "testing" + "time" + + "github.com/LeanerCloud/CUDly/pkg/common" + "github.com/aws/aws-sdk-go-v2/aws" + "github.com/aws/aws-sdk-go-v2/service/organizations" + "github.com/aws/aws-sdk-go-v2/service/organizations/types" + "github.com/stretchr/testify/assert" + "github.com/stretchr/testify/mock" + "github.com/stretchr/testify/require" +) + +func TestCalculateTotalInstances(t *testing.T) { + tests := []struct { + name string + recs []common.Recommendation + expected int + }{ + { + name: "multiple recommendations", + recs: []common.Recommendation{ + {Count: 5}, + {Count: 3}, + {Count: 2}, + }, + expected: 10, + }, + { + name: "empty recommendations", + recs: []common.Recommendation{}, + expected: 0, + }, + { + name: "single recommendation", + recs: []common.Recommendation{ + {Count: 7}, + }, + expected: 7, + }, + } + + for _, tt := range tests { + t.Run(tt.name, func(t *testing.T) { + total := CalculateTotalInstances(tt.recs) + assert.Equal(t, tt.expected, total) + }) + } +} + +func TestNewAccountAliasCacheWithClient(t *testing.T) { + mockOrg := &MockOrganizationsClient{} + cache := NewAccountAliasCacheWithClient(mockOrg) + + assert.NotNil(t, cache) + assert.NotNil(t, cache.cache) + assert.Equal(t, mockOrg, cache.orgClient) + assert.Equal(t, 0, len(cache.cache)) +} + +func TestGetAccountAlias(t *testing.T) { + ctx := context.Background() + + tests := []struct { + name string + accountID string + mockSetup func(m *MockOrganizationsClient) + expected string + shouldCache bool + }{ + { + name: "Empty account ID returns empty", + accountID: "", + mockSetup: func(m *MockOrganizationsClient) { + // No calls expected + }, + expected: "", + shouldCache: false, + }, + { + name: "Successful account lookup", + accountID: "123456789012", + mockSetup: func(m *MockOrganizationsClient) { + m.On("DescribeAccount", ctx, &organizations.DescribeAccountInput{ + AccountId: aws.String("123456789012"), + }).Return(&organizations.DescribeAccountOutput{ + Account: &types.Account{ + Name: aws.String("Production Account"), + }, + }, nil).Once() + }, + expected: "Production Account", + shouldCache: true, + }, + { + name: "Account not found - uses ID as fallback", + accountID: "999888777666", + mockSetup: func(m *MockOrganizationsClient) { + m.On("DescribeAccount", ctx, &organizations.DescribeAccountInput{ + AccountId: aws.String("999888777666"), + }).Return(nil, errors.New("account not found")).Once() + }, + expected: "999888777666", + shouldCache: true, + }, + { + name: "Account with nil name - uses ID as fallback", + accountID: "111222333444", + mockSetup: func(m *MockOrganizationsClient) { + m.On("DescribeAccount", ctx, &organizations.DescribeAccountInput{ + AccountId: aws.String("111222333444"), + }).Return(&organizations.DescribeAccountOutput{ + Account: &types.Account{ + Name: nil, + }, + }, nil).Once() + }, + expected: "111222333444", + shouldCache: true, + }, + } + + for _, tt := range tests { + t.Run(tt.name, func(t *testing.T) { + mockOrg := &MockOrganizationsClient{} + if tt.mockSetup != nil { + tt.mockSetup(mockOrg) + } + + cache := NewAccountAliasCacheWithClient(mockOrg) + + result := cache.GetAccountAlias(ctx, tt.accountID) + assert.Equal(t, tt.expected, result) + + if tt.shouldCache && tt.accountID != "" { + // Verify caching - second call should not hit the API + result2 := cache.GetAccountAlias(ctx, tt.accountID) + assert.Equal(t, tt.expected, result2) + } + + mockOrg.AssertExpectations(t) + }) + } +} + +func TestGetAccountAliasConcurrency(t *testing.T) { + // This test relies on AccountAliasCache.GetAccountAlias using double-checked locking: + // first acquire a read lock to check the cache, then acquire a write lock and re-check + // before calling the API. Without this pattern, multiple goroutines could pass the + // read-lock cache-miss check concurrently and issue multiple API calls. + // Run with -race to surface any data races in the cache map. + ctx := context.Background() + mockOrg := &MockOrganizationsClient{} + + // Setup mock to return account name + mockOrg.On("DescribeAccount", ctx, mock.AnythingOfType("*organizations.DescribeAccountInput")). + Return(&organizations.DescribeAccountOutput{ + Account: &types.Account{ + Name: aws.String("Test Account"), + }, + }, nil).Once() + + cache := NewAccountAliasCacheWithClient(mockOrg) + + // Test concurrent access to ensure proper locking. Worker goroutines must + // not call assert/require directly; funnel results back and assert on the + // test goroutine after wg.Wait(). + const numGoroutines = 10 + var wg sync.WaitGroup + wg.Add(numGoroutines) + results := make(chan string, numGoroutines) + for i := 0; i < numGoroutines; i++ { + go func() { + defer wg.Done() + results <- cache.GetAccountAlias(ctx, "123456789012") + }() + } + + // Wait for all goroutines to complete, then assert on the test goroutine. + wg.Wait() + close(results) + for result := range results { + assert.Equal(t, "Test Account", result) + } + + // Mock should only be called once due to double-checked locking in GetAccountAlias + mockOrg.AssertExpectations(t) +} + +func TestGetAccountAliasRealFunction(t *testing.T) { + // Skip integration test that requires real AWS API + t.Skip("Skipping integration test - GetAccountAlias tested via mock tests") + + // This test would validate GetAccountAlias with real AWS API + // but the functionality is already tested via the mock tests above +} + +func TestApplyCountOverride(t *testing.T) { + tests := []struct { + name string + recs []common.Recommendation + expectedCounts []int + overrideCount int32 + }{ + { + name: "Override with positive value", + recs: []common.Recommendation{ + {Count: 5, ResourceType: "db.t3.small"}, + {Count: 10, ResourceType: "db.t3.medium"}, + {Count: 3, ResourceType: "db.t3.large"}, + }, + overrideCount: 2, + expectedCounts: []int{2, 2, 2}, + }, + { + name: "Override with zero - no change", + recs: []common.Recommendation{ + {Count: 5, ResourceType: "db.t3.small"}, + {Count: 10, ResourceType: "db.t3.medium"}, + }, + overrideCount: 0, + expectedCounts: []int{5, 10}, + }, + { + name: "Override with negative value - no change", + recs: []common.Recommendation{ + {Count: 5, ResourceType: "db.t3.small"}, + }, + overrideCount: -1, + expectedCounts: []int{5}, + }, + { + name: "Empty recommendations", + recs: []common.Recommendation{}, + overrideCount: 5, + expectedCounts: []int{}, + }, + } + + for _, tt := range tests { + t.Run(tt.name, func(t *testing.T) { + result := ApplyCountOverride(tt.recs, tt.overrideCount) + assert.Equal(t, len(tt.expectedCounts), len(result)) + for i, rec := range result { + assert.Equal(t, tt.expectedCounts[i], rec.Count) + } + }) + } +} + +func TestApplyCoverage(t *testing.T) { + tests := []struct { + name string + recs []common.Recommendation + expectedCounts []int + coverage float64 + expectedLen int + }{ + { + name: "100% coverage - no change", + recs: []common.Recommendation{ + {Count: 10, EstimatedSavings: 100}, + {Count: 5, EstimatedSavings: 50}, + }, + coverage: 100.0, + expectedCounts: []int{10, 5}, + expectedLen: 2, + }, + { + name: "50% coverage", + recs: []common.Recommendation{ + {Count: 10, EstimatedSavings: 100}, + {Count: 6, EstimatedSavings: 60}, + }, + coverage: 50.0, + expectedCounts: []int{5, 3}, + expectedLen: 2, + }, + { + name: "0% coverage - returns empty", + recs: []common.Recommendation{ + {Count: 10, EstimatedSavings: 100}, + }, + coverage: 0.0, + expectedCounts: []int{}, + expectedLen: 0, + }, + { + name: "Negative coverage - returns empty", + recs: []common.Recommendation{ + {Count: 10, EstimatedSavings: 100}, + }, + coverage: -10.0, + expectedCounts: []int{}, + expectedLen: 0, + }, + { + name: "Coverage reduces to zero - filters out", + recs: []common.Recommendation{ + {Count: 1, EstimatedSavings: 10}, + {Count: 10, EstimatedSavings: 100}, + }, + coverage: 10.0, // 1*0.1 = 0, 10*0.1 = 1 + expectedCounts: []int{1}, + expectedLen: 1, + }, + { + name: "Savings Plans - reduces hourly commitment", + recs: []common.Recommendation{ + { + Service: common.ServiceSavingsPlansAll, + Count: 1, + EstimatedSavings: 100, + Details: &common.SavingsPlanDetails{ + HourlyCommitment: 10.0, + PlanType: "Compute", + }, + }, + }, + coverage: 50.0, + expectedCounts: []int{1}, // Count stays the same for SPs + expectedLen: 1, + }, + } + + for _, tt := range tests { + t.Run(tt.name, func(t *testing.T) { + result := ApplyCoverage(tt.recs, tt.coverage) + assert.Equal(t, tt.expectedLen, len(result)) + for i := range result { + if i < len(tt.expectedCounts) { + assert.Equal(t, tt.expectedCounts[i], result[i].Count) + } + } + + // For Savings Plans, verify hourly commitment is adjusted + if tt.name == "Savings Plans - reduces hourly commitment" && len(result) > 0 { + details, ok := result[0].Details.(*common.SavingsPlanDetails) + require.True(t, ok, "expected *common.SavingsPlanDetails in result Details") + assert.Equal(t, 5.0, details.HourlyCommitment) // 10 * 0.5 + assert.Equal(t, 50.0, result[0].EstimatedSavings) // 100 * 0.5 + } + }) + } +} + +// TestApplyCoverage_RICostScaling locks the fix for the CR finding on +// helpers.go:159-166: cost-bearing fields must scale by the DISCRETE +// count ratio (newCount / rec.Count), not the raw coverage ratio. With +// rec.Count=3 and coverage=50%, newCount=int(1.5)=1 (33% of instances) +// so costs must drop to 33% of original, not 50%, otherwise the sized +// purchase reads ~50% more expensive than what was actually bought. +func TestApplyCoverage_RICostScaling(t *testing.T) { + monthly := 60.0 + recs := []common.Recommendation{ + { + Service: common.ServiceEC2, + CommitmentType: common.CommitmentReservedInstance, + Count: 3, + CommitmentCost: 900, + OnDemandCost: 1800, + EstimatedSavings: 300, + RecurringMonthlyCost: &monthly, + }, + } + out := ApplyCoverage(recs, 50.0) + require.Len(t, out, 1) + // newCount = int(3 * 0.5) = 1. sizedRatio = 1/3. + assert.Equal(t, 1, out[0].Count) + assert.InDelta(t, 300.0, out[0].CommitmentCost, 0.01, "CommitmentCost scales by 1/3 (newCount/rec.Count), NOT 0.5 (raw ratio)") + assert.InDelta(t, 600.0, out[0].OnDemandCost, 0.01) + assert.InDelta(t, 100.0, out[0].EstimatedSavings, 0.01) + require.NotNil(t, out[0].RecurringMonthlyCost) + assert.InDelta(t, 20.0, *out[0].RecurringMonthlyCost, 0.01, "RecurringMonthlyCost scales by sized ratio too") + // Original pointer not mutated. + assert.Equal(t, 60.0, monthly) +} + +func TestApplySizing_LegacyCoverageRecordsCountsFlooredToZero(t *testing.T) { + recs := []common.Recommendation{ + {Service: common.ServiceEC2, Count: 1}, + {Service: common.ServiceRDS, Count: 10}, + } + drops := common.NewDropSummary() + + out := applySizing(recs, Config{}, 10, drops) + + require.Len(t, out, 1) + assert.Equal(t, 1, out[0].Count) + assert.Equal(t, 1, drops.Total()) + assert.Contains(t, drops.FormatOneLine(), common.DropTargetSizedToZero) +} + +func TestAdjustRecommendationsForExisting(t *testing.T) { + ctx := context.Background() + + tests := []struct { + name string + inputRecs []common.Recommendation + existingRIs []common.Commitment + expectedCounts []int + expectedLen int + }{ + { + name: "No existing RIs - all recommendations kept", + inputRecs: []common.Recommendation{ + {ResourceType: "db.t3.small", Region: "us-east-1", Count: 5}, + {ResourceType: "db.t3.medium", Region: "us-west-2", Count: 3}, + }, + existingRIs: []common.Commitment{}, + expectedLen: 2, + expectedCounts: []int{5, 3}, + }, + { + name: "Recent RI - partial adjustment", + inputRecs: []common.Recommendation{ + {ResourceType: "db.t3.small", Region: "us-east-1", Count: 10, Details: &common.DatabaseDetails{Engine: "mysql"}}, + }, + existingRIs: []common.Commitment{ + {ResourceType: "db.t3.small", Region: "us-east-1", Engine: "mysql", Count: 3, State: "active", StartDate: time.Now().Add(-1 * time.Hour)}, + }, + expectedLen: 1, + expectedCounts: []int{7}, // 10 - 3 + }, + { + name: "Recent RI - complete coverage", + inputRecs: []common.Recommendation{ + {ResourceType: "db.t3.small", Region: "us-east-1", Count: 5, Details: &common.DatabaseDetails{Engine: "postgresql"}}, + }, + existingRIs: []common.Commitment{ + {ResourceType: "db.t3.small", Region: "us-east-1", Engine: "postgresql", Count: 10, State: "active", StartDate: time.Now().Add(-1 * time.Hour)}, + }, + expectedLen: 0, // All covered + expectedCounts: []int{}, + }, + { + name: "Old RI - not recent, no adjustment", + inputRecs: []common.Recommendation{ + {ResourceType: "db.t3.small", Region: "us-east-1", Count: 5, Details: &common.DatabaseDetails{Engine: "mysql"}}, + }, + existingRIs: []common.Commitment{ + {ResourceType: "db.t3.small", Region: "us-east-1", Engine: "mysql", Count: 10, State: "active", StartDate: time.Now().Add(-48 * time.Hour)}, + }, + expectedLen: 1, + expectedCounts: []int{5}, // No adjustment - RI is too old + }, + { + // Boundary: just inside the 24-hour window — should be treated as recent + name: "RI just inside lookback threshold - adjusted", + inputRecs: []common.Recommendation{ + {ResourceType: "db.t3.small", Region: "us-east-1", Count: 5, Details: &common.DatabaseDetails{Engine: "mysql"}}, + }, + existingRIs: []common.Commitment{ + {ResourceType: "db.t3.small", Region: "us-east-1", Engine: "mysql", Count: 10, State: "active", StartDate: time.Now().Add(-23*time.Hour - 59*time.Minute)}, + }, + expectedLen: 0, // Recent RI fully covers the recommendation + expectedCounts: []int{}, + }, + { + // Boundary: just outside the 24-hour window — should not be treated as recent + name: "RI just outside lookback threshold - not adjusted", + inputRecs: []common.Recommendation{ + {ResourceType: "db.t3.small", Region: "us-east-1", Count: 5, Details: &common.DatabaseDetails{Engine: "mysql"}}, + }, + existingRIs: []common.Commitment{ + {ResourceType: "db.t3.small", Region: "us-east-1", Engine: "mysql", Count: 10, State: "active", StartDate: time.Now().Add(-24*time.Hour - 1*time.Minute)}, + }, + expectedLen: 1, // RI is outside lookback window, no adjustment + expectedCounts: []int{5}, + }, + { + name: "Different engine - no adjustment", + inputRecs: []common.Recommendation{ + {ResourceType: "db.t3.small", Region: "us-east-1", Count: 5, Details: &common.DatabaseDetails{Engine: "postgresql"}}, + }, + existingRIs: []common.Commitment{ + {ResourceType: "db.t3.small", Region: "us-east-1", Engine: "mysql", Count: 10, State: "active", StartDate: time.Now().Add(-1 * time.Hour)}, + }, + expectedLen: 1, + expectedCounts: []int{5}, // Different engine + }, + } + + for _, tt := range tests { + t.Run(tt.name, func(t *testing.T) { + mockClient := &MockServiceClient{} + mockClient.On("GetExistingCommitments", ctx).Return(tt.existingRIs, nil) + + checker := NewDuplicateChecker(0) + result, _, err := checker.AdjustRecommendationsForExisting(ctx, tt.inputRecs, mockClient) + + assert.NoError(t, err) + assert.Equal(t, tt.expectedLen, len(result)) + for i := range result { + if i < len(tt.expectedCounts) { + assert.Equal(t, tt.expectedCounts[i], result[i].Count) + } + } + + mockClient.AssertExpectations(t) + }) + } +} + +func TestGetRecommendationDescription(t *testing.T) { + tests := []struct { + name string + expected string + rec common.Recommendation + }{ + { + name: "RDS recommendation with database details", + rec: common.Recommendation{ + Service: common.ServiceRDS, + ResourceType: "db.t3.small", + Details: &common.DatabaseDetails{ + Engine: "mysql", + }, + }, + // GetDetailDescription returns "engine/AZConfig"; AZConfig is empty so trailing slash is included + expected: "rds db.t3.small mysql/", + }, + { + name: "EC2 recommendation without details", + rec: common.Recommendation{ + Service: common.ServiceEC2, + ResourceType: "t3.medium", + }, + expected: "ec2 t3.medium", + }, + { + name: "ElastiCache recommendation with cache details", + rec: common.Recommendation{ + Service: common.ServiceElastiCache, + ResourceType: "cache.t3.micro", + Details: &common.CacheDetails{ + Engine: "redis", + }, + }, + // GetDetailDescription returns "engine/NodeType"; NodeType is empty so trailing slash is included + expected: "elasticache cache.t3.micro redis/", + }, + } + + for _, tt := range tests { + t.Run(tt.name, func(t *testing.T) { + result := GetRecommendationDescription(tt.rec) + assert.Equal(t, tt.expected, result) + }) + } +} + +func TestNormalizeEngineName(t *testing.T) { + tests := []struct { + input string + expected string + }{ + {"Aurora PostgreSQL", "aurora-postgresql"}, + {"Aurora MySQL", "aurora-mysql"}, + {"MySQL", "mysql"}, + {"PostgreSQL", "postgresql"}, + {"postgres", "postgresql"}, + {"MariaDB", "mariadb"}, + {"Oracle", "oracle"}, + {"oracle-se", "oracle"}, + {"oracle-se1", "oracle"}, + {"oracle-se2", "oracle"}, + {"oracle-ee", "oracle"}, + {"SQL Server", "sqlserver"}, + {"sqlserver-se", "sqlserver"}, + {"sqlserver-ee", "sqlserver"}, + {"sqlserver-ex", "sqlserver"}, + {"sqlserver-web", "sqlserver"}, + {"unknown-engine", "unknown-engine"}, + } + + for _, tt := range tests { + t.Run(tt.input, func(t *testing.T) { + result := common.NormalizeEngineName(tt.input) + assert.Equal(t, tt.expected, result) + }) + } +} + +// TestGetEngineFromRecommendation only exercises pointer-typed +// DatabaseDetails/CacheDetails: every producer (AWS, Azure, the CSV +// loader, and the JSON codec) constructs them that way, so there is no +// value-typed case to cover -- see +// pkg/common/service_details_codec.go's package doc for the invariant. +func TestGetEngineFromRecommendation(t *testing.T) { + tests := []struct { + name string + expected string + rec common.Recommendation + }{ + { + name: "DatabaseDetails pointer type", + rec: common.Recommendation{ + Details: &common.DatabaseDetails{Engine: "postgresql"}, + }, + expected: "postgresql", + }, + { + name: "CacheDetails pointer type", + rec: common.Recommendation{ + Details: &common.CacheDetails{Engine: "valkey"}, + }, + expected: "valkey", + }, + { + name: "No details", + rec: common.Recommendation{ + Details: nil, + }, + expected: "", + }, + { + name: "ComputeDetails - returns empty (no engine)", + rec: common.Recommendation{ + Details: &common.ComputeDetails{Platform: "Linux/UNIX"}, + }, + expected: "", + }, + // Typed nils: rec.Details != nil (the interface holds a type) but the + // pointer inside is nil, so the `rec.Details == nil` guard does not + // catch it and the field read would panic without the per-case check. + { + name: "typed nil *DatabaseDetails - returns empty, no panic", + rec: common.Recommendation{ + Details: (*common.DatabaseDetails)(nil), + }, + expected: "", + }, + { + name: "typed nil *CacheDetails - returns empty, no panic", + rec: common.Recommendation{ + Details: (*common.CacheDetails)(nil), + }, + expected: "", + }, + } + + for _, tt := range tests { + t.Run(tt.name, func(t *testing.T) { + result := common.EngineFromDetails(tt.rec.Details) + assert.Equal(t, tt.expected, result) + }) + } +} + +// confirmPurchaseWithInput is a testable variant of ConfirmPurchase that reads +// from the provided reader rather than os.Stdin, allowing stdin to be mocked in tests. +func confirmPurchaseWithInput(skipConfirmation bool, input string) bool { + if skipConfirmation { + return true + } + response := strings.TrimSpace(strings.ToLower(strings.SplitN(input, "\n", 2)[0])) + return response == "yes" || response == "y" +} + +func TestConfirmPurchase(t *testing.T) { + tests := []struct { + name string + totalInstances int + totalCost float64 + skipConfirmation bool + expected bool + }{ + { + name: "Skip confirmation returns true", + totalInstances: 10, + totalCost: 100.50, + skipConfirmation: true, + expected: true, + }, + { + name: "Skip confirmation with zero cost", + totalInstances: 0, + totalCost: 0.0, + skipConfirmation: true, + expected: true, + }, + { + name: "Skip confirmation with high cost", + totalInstances: 1000, + totalCost: 999999.99, + skipConfirmation: true, + expected: true, + }, + } + + for _, tt := range tests { + t.Run(tt.name, func(t *testing.T) { + result := ConfirmPurchase(tt.totalInstances, tt.totalCost, tt.skipConfirmation) + assert.Equal(t, tt.expected, result) + }) + } +} + +func TestConfirmPurchaseInput(t *testing.T) { + // Tests for the interactive stdin branch of ConfirmPurchase logic + tests := []struct { + name string + input string + expected bool + }{ + {name: "yes accepts", input: "yes\n", expected: true}, + {name: "y accepts", input: "y\n", expected: true}, + {name: "YES accepts (case insensitive)", input: "YES\n", expected: true}, + {name: "Y accepts (case insensitive)", input: "Y\n", expected: true}, + {name: "no rejects", input: "no\n", expected: false}, + {name: "n rejects", input: "n\n", expected: false}, + {name: "empty string rejects", input: "\n", expected: false}, + {name: "arbitrary text rejects", input: "maybe\n", expected: false}, + } + + for _, tt := range tests { + t.Run(tt.name, func(t *testing.T) { + result := confirmPurchaseWithInput(false, tt.input) + assert.Equal(t, tt.expected, result) + }) + } +} + +func TestAdjustRecommendationsForExistingRIsEdgeCases(t *testing.T) { + ctx := context.Background() + + tests := []struct { + name string + inputRecs []common.Recommendation + existingRIs []common.Commitment + expectedCounts []int + expectedLen int + }{ + { + name: "Multiple RIs same instance type different regions", + inputRecs: []common.Recommendation{ + {ResourceType: "db.t3.small", Region: "us-east-1", Count: 10, Details: &common.DatabaseDetails{Engine: "mysql"}}, + {ResourceType: "db.t3.small", Region: "eu-west-1", Count: 8, Details: &common.DatabaseDetails{Engine: "mysql"}}, + }, + existingRIs: []common.Commitment{ + {ResourceType: "db.t3.small", Region: "us-east-1", Engine: "mysql", Count: 3, State: "active", StartDate: time.Now().Add(-1 * time.Hour)}, + {ResourceType: "db.t3.small", Region: "eu-west-1", Engine: "mysql", Count: 2, State: "active", StartDate: time.Now().Add(-1 * time.Hour)}, + }, + expectedLen: 2, // Both regions should have adjusted counts + expectedCounts: []int{7, 6}, + }, + { + name: "Retired RI should not affect recommendations", + inputRecs: []common.Recommendation{ + {ResourceType: "db.t3.small", Region: "us-east-1", Count: 5, Details: &common.DatabaseDetails{Engine: "mysql"}}, + }, + existingRIs: []common.Commitment{ + {ResourceType: "db.t3.small", Region: "us-east-1", Engine: "mysql", Count: 10, State: "retired", StartDate: time.Now().Add(-1 * time.Hour)}, + }, + expectedLen: 1, // Retired RI should not affect + expectedCounts: []int{5}, + }, + { + name: "Payment pending RI should adjust", + inputRecs: []common.Recommendation{ + {ResourceType: "db.t3.small", Region: "us-east-1", Count: 10, Details: &common.DatabaseDetails{Engine: "mysql"}}, + }, + existingRIs: []common.Commitment{ + {ResourceType: "db.t3.small", Region: "us-east-1", Engine: "mysql", Count: 4, State: "payment-pending", StartDate: time.Now().Add(-1 * time.Hour)}, + }, + expectedLen: 1, + expectedCounts: []int{6}, // 10 - 4 = 6 + }, + } + + for _, tt := range tests { + t.Run(tt.name, func(t *testing.T) { + mockClient := &MockServiceClient{} + mockClient.On("GetExistingCommitments", ctx).Return(tt.existingRIs, nil) + + checker := NewDuplicateChecker(0) + result, _, err := checker.AdjustRecommendationsForExisting(ctx, tt.inputRecs, mockClient) + + assert.NoError(t, err) + assert.Equal(t, tt.expectedLen, len(result)) + for i := range result { + if i < len(tt.expectedCounts) { + assert.Equal(t, tt.expectedCounts[i], result[i].Count) + } + } + + mockClient.AssertExpectations(t) + }) + } +} + +func TestApplyInstanceLimit(t *testing.T) { + tests := []struct { + name string + recs []common.Recommendation + expectedCounts []int + expectedLen int + maxInstances int32 + }{ + { + name: "No limit - all recommendations kept", + recs: []common.Recommendation{ + {Count: 5, ResourceType: "db.t3.small"}, + {Count: 3, ResourceType: "db.t3.medium"}, + }, + maxInstances: 0, + expectedLen: 2, + expectedCounts: []int{5, 3}, + }, + { + name: "Limit exceeds total - all kept", + recs: []common.Recommendation{ + {Count: 5, ResourceType: "db.t3.small"}, + {Count: 3, ResourceType: "db.t3.medium"}, + }, + maxInstances: 20, + expectedLen: 2, + expectedCounts: []int{5, 3}, + }, + { + name: "Limit applies to first recommendation", + recs: []common.Recommendation{ + {Count: 10, ResourceType: "db.t3.small"}, + {Count: 5, ResourceType: "db.t3.medium"}, + }, + maxInstances: 7, + expectedLen: 1, + expectedCounts: []int{7}, + }, + { + name: "Limit applies across recommendations", + recs: []common.Recommendation{ + {Count: 5, ResourceType: "db.t3.small"}, + {Count: 5, ResourceType: "db.t3.medium"}, + {Count: 5, ResourceType: "db.t3.large"}, + }, + maxInstances: 12, + expectedLen: 3, + expectedCounts: []int{5, 5, 2}, + }, + { + name: "Negative limit - all kept", + recs: []common.Recommendation{ + {Count: 5, ResourceType: "db.t3.small"}, + }, + maxInstances: -1, + expectedLen: 1, + expectedCounts: []int{5}, + }, + { + // math.MinInt32 should be treated the same as any negative value: no limit applied + name: "math.MinInt32 limit - all kept (no int32 wrap)", + recs: []common.Recommendation{ + {Count: 5, ResourceType: "db.t3.small"}, + }, + maxInstances: math.MinInt32, + expectedLen: 1, + expectedCounts: []int{5}, + }, + } + + for _, tt := range tests { + t.Run(tt.name, func(t *testing.T) { + result := ApplyInstanceLimit(tt.recs, tt.maxInstances) + assert.Equal(t, tt.expectedLen, len(result)) + for i := range result { + if i < len(tt.expectedCounts) { + assert.Equal(t, tt.expectedCounts[i], result[i].Count) + } + } + }) + } +} + +func TestNewDuplicateChecker_CustomWindow(t *testing.T) { + checker := NewDuplicateChecker(48) + assert.Equal(t, 48, checker.LookbackHours) +} + +func TestNewDuplicateChecker_ZeroUsesDefault(t *testing.T) { + checker := NewDuplicateChecker(0) + assert.Equal(t, DefaultDuplicateCheckLookbackHours, checker.LookbackHours) +} + +func TestAdjustRecommendationsForExisting_WithinWindow(t *testing.T) { + ctx := context.Background() + rec := common.Recommendation{ + ResourceType: "db.t3.small", Region: "us-east-1", Count: 5, + Details: &common.DatabaseDetails{Engine: "mysql"}, + } + existing := []common.Commitment{ + {ResourceType: "db.t3.small", Region: "us-east-1", Engine: "mysql", + Count: 5, State: "active", StartDate: time.Now().Add(-1 * time.Hour)}, + } + + mockClient := &MockServiceClient{} + mockClient.On("GetExistingCommitments", ctx).Return(existing, nil) + + checker := NewDuplicateChecker(24) + passed, filtered, err := checker.AdjustRecommendationsForExisting(ctx, []common.Recommendation{rec}, mockClient) + + require.NoError(t, err) + assert.Empty(t, passed) + assert.Len(t, filtered, 1) + mockClient.AssertExpectations(t) +} + +func TestAdjustRecommendationsForExisting_OutsideWindow(t *testing.T) { + ctx := context.Background() + rec := common.Recommendation{ + ResourceType: "db.t3.small", Region: "us-east-1", Count: 5, + Details: &common.DatabaseDetails{Engine: "mysql"}, + } + existing := []common.Commitment{ + {ResourceType: "db.t3.small", Region: "us-east-1", Engine: "mysql", + Count: 5, State: "active", StartDate: time.Now().Add(-30 * 24 * time.Hour)}, + } + + mockClient := &MockServiceClient{} + mockClient.On("GetExistingCommitments", ctx).Return(existing, nil) + + checker := NewDuplicateChecker(24) + passed, filtered, err := checker.AdjustRecommendationsForExisting(ctx, []common.Recommendation{rec}, mockClient) + + require.NoError(t, err) + assert.Len(t, passed, 1) + assert.Equal(t, 5, passed[0].Count) + assert.Empty(t, filtered) + mockClient.AssertExpectations(t) +} + +func TestAdjustRecommendationsForExisting_PartialCoverage(t *testing.T) { + ctx := context.Background() + rec := common.Recommendation{ + ResourceType: "db.t3.small", Region: "us-east-1", Count: 10, + Details: &common.DatabaseDetails{Engine: "mysql"}, + } + existing := []common.Commitment{ + {ResourceType: "db.t3.small", Region: "us-east-1", Engine: "mysql", + Count: 3, State: "active", StartDate: time.Now().Add(-1 * time.Hour)}, + } + + mockClient := &MockServiceClient{} + mockClient.On("GetExistingCommitments", ctx).Return(existing, nil) + + checker := NewDuplicateChecker(24) + passed, filtered, err := checker.AdjustRecommendationsForExisting(ctx, []common.Recommendation{rec}, mockClient) + + require.NoError(t, err) + require.Len(t, passed, 1) + assert.Equal(t, 7, passed[0].Count) // 10 - 3 = 7 + assert.Empty(t, filtered) // partial coverage stays in passed, not filtered + mockClient.AssertExpectations(t) +} + +// TestApplyTargetCoverage covers the RI sizing branch of issue #338's +// --target-coverage flag, now under-buy semantics: n = floor(avg*target). +// Confirms: floor (not ceil) selection so coverage stays at-most target, +// drop-when-target-too-low (avg*target < 1), no-signal pass-through, and +// projected utilization (typically 100% since we under-buy) / coverage +// (tracks target%) outputs. +func TestApplyTargetCoverage_RI(t *testing.T) { + mkRI := func(count int, avg, recUtil float64) common.Recommendation { + return common.Recommendation{ + Service: common.ServiceEC2, + Region: "us-east-1", + ResourceType: "t3.medium", + Count: count, + CommitmentType: common.CommitmentReservedInstance, + CommitmentCost: 1000, + OnDemandCost: 2000, + EstimatedSavings: 500, + AverageInstancesUsedPerHour: avg, + RecommendedUtilization: recUtil, + } + } + + tests := []struct { + name string + rec common.Recommendation + target float64 + wantDropped bool + wantCount int + wantProjUtil float64 // 0 means "don't assert" + wantProjCovGTE float64 // we assert coverage >= this (handles the float clamping) + }{ + { + // avg=8.5, target=95%, existing=0%. + // gap=95. n = floor(8.5 * 95/100) = floor(8.075) = 8. + // Projected util = 8.5/8 = 106.25 → clamped to 100. + // Projected cov = 0 + 8/8.5*100 = 94.117…% + name: "RI: target 95 buys 8 (floor of avg*0.95)", + rec: mkRI(10, 8.5, 0), + target: 95, + wantCount: 8, + wantProjUtil: 100, + wantProjCovGTE: 94.0, + }, + { + // avg=10, target=50%, existing=0%. n=floor(10*0.5)=5. + // Projected cov = 5/10*100 = 50.0%. + name: "RI: target 50 buys half of avg demand", + rec: mkRI(10, 10, 0), + target: 50, + wantCount: 5, + wantProjUtil: 100, // 10/5 clamped + wantProjCovGTE: 50.0, + }, + { + // avg=0.4, target=50%, existing=0%. + // n = floor(0.4 * 50/100) = floor(0.2) = 0 → DROPPED. + // Tiny pools where avg×gap%<100 produce 0 RIs under the + // coverage-anchored formula. --min-pool-size upstream is the + // intended filter for these; the drop here is the fallback + // when the upstream filter wasn't applied. + name: "RI: tiny avg below 1-RI threshold drops", + rec: mkRI(5, 0.4, 0), + target: 50, + wantDropped: true, + }, + { + // avg=0 (no signal) → passed through unchanged, counted in skip summary. + // Projection metrics never set on the pass-through path. + name: "RI: no signal → passed through unmodified", + rec: mkRI(5, 0, 0), + target: 80, + wantCount: 5, + wantProjUtil: 0, // never set in pass-through + }, + { + // avg=4, target=80%, existing=0%. n=floor(4*0.8)=3. + // Projected util = 4/3 = 133% clamped to 100. + // Projected cov = 3/4*100 = 75.0%. + name: "RI: target 80 buys floor(avg*0.8)", + rec: mkRI(5, 4, 0), + target: 80, + wantCount: 3, + wantProjUtil: 100, + wantProjCovGTE: 75.0, + }, + } + + for _, tt := range tests { + t.Run(tt.name, func(t *testing.T) { + recs := []common.Recommendation{tt.rec} + out := ApplyTargetCoverage(recs, tt.target, nil) + if tt.wantDropped { + if len(out) != 0 { + t.Fatalf("expected drop; got %d recs", len(out)) + } + return + } + if len(out) != 1 { + t.Fatalf("expected 1 rec; got %d", len(out)) + } + if out[0].Count != tt.wantCount { + t.Errorf("Count: got %d, want %d", out[0].Count, tt.wantCount) + } + if tt.wantProjUtil > 0 { + if math.Abs(out[0].ProjectedUtilization-tt.wantProjUtil) > 0.01 { + t.Errorf("ProjectedUtilization: got %.4f, want %.4f", + out[0].ProjectedUtilization, tt.wantProjUtil) + } + } + // Zero means "don't assert" (matches the wantProjUtil convention) + // since the pass-through path leaves ProjectedCoverage at zero. + if tt.wantProjCovGTE > 0 { + if out[0].ProjectedCoverage < tt.wantProjCovGTE-0.01 { + t.Errorf("ProjectedCoverage: got %.4f, want >= %.4f", + out[0].ProjectedCoverage, tt.wantProjCovGTE) + } + } + }) + } +} + +// TestApplyTargetCoverage_RI_CostScaling verifies RI cost-bearing fields +// scale by the sized-to-original count ratio. SavingsPercentage is invariant. +// The scaled values let downstream consumers (CSV writer, reporter, audit +// log) trust rec.CommitmentCost / rec.EstimatedSavings as the sized purchase +// rather than AWS's pre-sized proposal. +func TestApplyTargetCoverage_RI_CostScaling(t *testing.T) { + rec := common.Recommendation{ + Service: common.ServiceEC2, + Count: 10, + CommitmentType: common.CommitmentReservedInstance, + CommitmentCost: 1000, + OnDemandCost: 2000, + EstimatedSavings: 500, + SavingsPercentage: 25, + AverageInstancesUsedPerHour: 8, + } + // target=80, existing=0, avg=8 → gap=80. + // n = floor(8 * 80/100) = 6. Ratio = 6/10 = 0.6 (cost scaling still + // uses rec.Count to convert AWS's quoted cost-for-rec.Count into + // cost-for-nTarget). + out := ApplyTargetCoverage([]common.Recommendation{rec}, 80, nil) + require.Len(t, out, 1) + assert.Equal(t, 6, out[0].Count) + assert.InDelta(t, 600.0, out[0].CommitmentCost, 0.001, "CommitmentCost scales by nTarget/rec.Count") + assert.InDelta(t, 1200.0, out[0].OnDemandCost, 0.001, "OnDemandCost scales by nTarget/rec.Count") + assert.InDelta(t, 300.0, out[0].EstimatedSavings, 0.001, "EstimatedSavings scales by nTarget/rec.Count") + assert.Equal(t, 25.0, out[0].SavingsPercentage, "SavingsPercentage is invariant under count scaling") + + t.Run("RecurringMonthlyCost scales by ratio when populated", func(t *testing.T) { + // AWS populated RecurringStandardMonthlyCost for partial/no-upfront + // recs; sized purchase must scale this monthly fee by the same + // nTarget/rec.Count ratio so total cost (upfront + monthly × term) + // reflects what the user actually buys. + monthly := 50.0 + recWithMonthly := rec + recWithMonthly.RecurringMonthlyCost = &monthly + out := ApplyTargetCoverage([]common.Recommendation{recWithMonthly}, 80, nil) + require.Len(t, out, 1) + require.NotNil(t, out[0].RecurringMonthlyCost, "scaled pointer should be non-nil") + assert.InDelta(t, 30.0, *out[0].RecurringMonthlyCost, 0.001, "monthly cost scales by 6/10") + // Original pointer untouched (we allocated a new one). + assert.Equal(t, 50.0, monthly, "original RecurringMonthlyCost target should not be mutated") + }) + + t.Run("RecurringMonthlyCost stays nil when not populated", func(t *testing.T) { + // AWS API didn't return RecurringStandardMonthlyCost (all-upfront, + // or field missing). The sized rec should also have nil so + // downstream renders "unknown" rather than zero. + out := ApplyTargetCoverage([]common.Recommendation{rec}, 80, nil) + require.Len(t, out, 1) + assert.Nil(t, out[0].RecurringMonthlyCost, "nil input → nil output") + }) +} + +// TestApplyTargetCoverage_RI_ExistingCoverage covers the under-buy formula's +// existing-commitment branch: gap = (target - existing_cov)/100, then +// n_target = floor(avg * gap). Matches the worked example from the #338 design +// thread (20 instances, 10 existing RIs at 50% coverage, target 80% → buy 6). +func TestApplyTargetCoverage_RI_ExistingCoverage(t *testing.T) { + mkRI := func(count int, avg, existingCov float64) common.Recommendation { + return common.Recommendation{ + Service: common.ServiceEC2, + Region: "us-east-1", + ResourceType: "t3.medium", + Count: count, + RecommendedCount: count, + CommitmentType: common.CommitmentReservedInstance, + CommitmentCost: 1000, + OnDemandCost: 2000, + EstimatedSavings: 500, + AverageInstancesUsedPerHour: avg, + ExistingCoveragePct: existingCov, + } + } + + tests := []struct { + name string + rec common.Recommendation + target float64 + wantDropped bool + wantCount int + wantTotalCov float64 // ProjectedCoverage = existing + new contribution + }{ + { + // User's worked example: avg=20, existing=50%, target=80%. + // gap=30. n=ceil(20*0.30)=6. Total cov = 50 + 6/20*100 = 80. + name: "User example: 50% existing, 80% target on avg=20 → buy 6", + rec: mkRI(10, 20, 50), + target: 80, + wantCount: 6, + wantTotalCov: 80, + }, + { + // existing=0%, target=70%, avg=10. gap=70. n=ceil(7)=7. + name: "Zero existing: ceil(avg*target/100)", + rec: mkRI(10, 10, 0), + target: 70, + wantCount: 7, + wantTotalCov: 70, + }, + { + // existing=80% meets target=80% → drop. + name: "Existing meets target exactly: drop", + rec: mkRI(10, 10, 80), + target: 80, + wantDropped: true, + }, + { + // existing=95% exceeds target=80% → drop. + name: "Existing exceeds target: drop", + rec: mkRI(10, 10, 95), + target: 80, + wantDropped: true, + }, + { + // avg=2, existing=70%, target=80%. gap=10. + // n = floor(2 * 10/100) = floor(0.2) = 0 → DROPPED. + // Small pool + thin gap: no integer buy can approximate the + // target. --min-pool-size upstream is the intended filter. + name: "Small gap on tiny avg drops", + rec: mkRI(5, 2, 70), + target: 80, + wantDropped: true, + }, + { + // avg=10, existing=60%, target=70%. gap=10. + // n = floor(10 * 10/100) = 1. Total cov = 60 + 1/10*100 = 70. + // Coverage-anchored: 1 RI exactly closes the 10-point gap + // because avg=10 and each RI is worth 10% of avg demand. + name: "Small top-up: 1 RI exactly closes 10-pt gap on avg=10", + rec: mkRI(10, 10, 60), + target: 70, + wantCount: 1, + wantTotalCov: 70, + }, + } + + for _, tt := range tests { + t.Run(tt.name, func(t *testing.T) { + out := ApplyTargetCoverage([]common.Recommendation{tt.rec}, tt.target, nil) + if tt.wantDropped { + assert.Len(t, out, 0, "expected drop") + return + } + require.Len(t, out, 1) + assert.Equal(t, tt.wantCount, out[0].Count, "Count") + assert.InDelta(t, tt.wantTotalCov, out[0].ProjectedCoverage, 0.01, "ProjectedCoverage is TOTAL (existing + new)") + }) + } +} + +// TestApplyTargetCoverage_SP covers the SP sizing branch under under-buy +// semantics: HourlyCommitment and EstimatedSavings scale by targetPct/100 +// regardless of AWS's projected utilization. CommitmentCost / OnDemandCost / +// SavingsPercentage must NOT change. +func TestApplyTargetCoverage_SP(t *testing.T) { + mkSP := func(recUtil float64) common.Recommendation { + return common.Recommendation{ + Service: common.ServiceSavingsPlansCompute, + CommitmentType: common.CommitmentSavingsPlan, + CommitmentCost: 1000, + OnDemandCost: 5000, + EstimatedSavings: 1500, + SavingsPercentage: 30, + RecommendedUtilization: recUtil, + Details: &common.SavingsPlanDetails{HourlyCommitment: 2.0}, + } + } + + t.Run("AWS above target — still scales by target (under-buy)", func(t *testing.T) { + // RecUtil=95, target=80. Even though AWS projects above target, the + // flag's intent is "leave 20% headroom", so the commitment shrinks + // to 80% of AWS rec. All cost-bearing fields scale by 0.8. + // Projected util = 95/0.80 = 118.75 clamped to 100. + out := ApplyTargetCoverage([]common.Recommendation{mkSP(95)}, 80, nil) + require.Len(t, out, 1) + assert.InDelta(t, 1.6, out[0].Details.(*common.SavingsPlanDetails).HourlyCommitment, 0.001) + assert.InDelta(t, 800.0, out[0].CommitmentCost, 0.001, "CommitmentCost scales by target/100") + assert.InDelta(t, 4000.0, out[0].OnDemandCost, 0.001, "OnDemandCost scales by target/100") + assert.InDelta(t, 1200.0, out[0].EstimatedSavings, 0.001) + assert.Equal(t, 30.0, out[0].SavingsPercentage, "SavingsPercentage is invariant") + assert.InDelta(t, 100.0, out[0].ProjectedUtilization, 0.001, "RecUtil/ratio = 95/0.80 = 118.75 clamps to 100") + assert.Equal(t, 0.0, out[0].ProjectedCoverage, "SPs intentionally leave ProjectedCoverage at zero") + }) + + t.Run("AWS below target — scale down by target (under-buy)", func(t *testing.T) { + // RecUtil=50, target=80. All cost-bearing fields shrink to 80%. + // Projected util = 50/0.80 = 62.5 (no clamp needed). + out := ApplyTargetCoverage([]common.Recommendation{mkSP(50)}, 80, nil) + require.Len(t, out, 1) + details := out[0].Details.(*common.SavingsPlanDetails) + assert.InDelta(t, 1.6, details.HourlyCommitment, 0.001) + assert.InDelta(t, 800.0, out[0].CommitmentCost, 0.001, "CommitmentCost scales by target/100") + assert.InDelta(t, 4000.0, out[0].OnDemandCost, 0.001, "OnDemandCost scales by target/100") + assert.InDelta(t, 1200.0, out[0].EstimatedSavings, 0.001) + assert.Equal(t, 30.0, out[0].SavingsPercentage, "SavingsPercentage is invariant") + assert.InDelta(t, 62.5, out[0].ProjectedUtilization, 0.001, "RecUtil/ratio = 50/0.80 = 62.5") + assert.Equal(t, 0.0, out[0].ProjectedCoverage) + }) + + t.Run("no signal → passed through unchanged", func(t *testing.T) { + out := ApplyTargetCoverage([]common.Recommendation{mkSP(0)}, 80, nil) + require.Len(t, out, 1) + // Original recommendation values intact. + assert.Equal(t, 2.0, out[0].Details.(*common.SavingsPlanDetails).HourlyCommitment) + assert.Equal(t, 1500.0, out[0].EstimatedSavings) + assert.Equal(t, 0.0, out[0].ProjectedUtilization) + }) +} + +// TestApplySizing checks the routing helper picks the right sizer based +// on cfg.TargetCoverage being >0 vs ==0. +func TestApplySizing(t *testing.T) { + ri := common.Recommendation{ + Service: common.ServiceEC2, + Count: 10, + CommitmentType: common.CommitmentReservedInstance, + AverageInstancesUsedPerHour: 8, + } + + t.Run("TargetCoverage > 0 → ApplyTargetCoverage", func(t *testing.T) { + cfg := Config{TargetCoverage: 80, Coverage: 100} + out := applySizing([]common.Recommendation{ri}, cfg, cfg.Coverage, nil) + require.Len(t, out, 1) + // avg=8, target=80%, existing=0%. gap=80. + // n = floor(8 * 80/100) = floor(6.4) = 6. ProjUtil = 8/6 = 133% → 100. + assert.Equal(t, 6, out[0].Count) + assert.Equal(t, 100.0, out[0].ProjectedUtilization) + }) + + t.Run("TargetCoverage == 0 → ApplyCoverage", func(t *testing.T) { + cfg := Config{TargetCoverage: 0, Coverage: 50} + out := applySizing([]common.Recommendation{ri}, cfg, cfg.Coverage, nil) + require.Len(t, out, 1) + // ApplyCoverage(50) on count=10 → 5. ProjectedUtilization NOT set + // (zero) because we took the coverage branch. + assert.Equal(t, 5, out[0].Count) + assert.Equal(t, 0.0, out[0].ProjectedUtilization) + }) +} + +// TestApplyTargetCoverage_RI_Target100 covers the target == 100 boundary. +// With the coverage-anchored formula, target=100 (existing=0) yields +// n = floor(avg * 100/100) = floor(avg) — operators get a buy sized to +// the pool's average concurrent demand, not AWS's rec.Count (which may +// be sized to peak/ROI-curated). Pools where avg<1 drop; --min-pool-size +// is the intended upstream filter. +func TestApplyTargetCoverage_RI_Target100(t *testing.T) { + mkRI := func(count int, avg float64) common.Recommendation { + return common.Recommendation{ + Service: common.ServiceEC2, + Region: "us-east-1", + ResourceType: "t3.medium", + Count: count, + CommitmentType: common.CommitmentReservedInstance, + AverageInstancesUsedPerHour: avg, + } + } + + tests := []struct { + name string + rec common.Recommendation + wantDropped bool + wantCount int + }{ + // avg=0.999 → floor(0.999)=0 → drop. + {name: "target 100, avg=0.999 → drop (avg<1)", rec: mkRI(5, 0.999), wantDropped: true}, + // avg=1.0 → buy 1 (matches avg demand). + {name: "target 100, avg=1 → buy 1", rec: mkRI(5, 1.0), wantCount: 1}, + // avg=8.7 → floor(8.7) = 8. + {name: "target 100, avg=8.7 → buy floor(avg)=8", rec: mkRI(10, 8.7), wantCount: 8}, + // avg=10 → buy 10 (avg=10 means 10 concurrent instances on average). + {name: "target 100, avg=10 → buy 10", rec: mkRI(10, 10.0), wantCount: 10}, + } + + for _, tt := range tests { + t.Run(tt.name, func(t *testing.T) { + out := ApplyTargetCoverage([]common.Recommendation{tt.rec}, 100, nil) + if tt.wantDropped { + assert.Len(t, out, 0, "expected drop at target=100 for avg=%.3f", tt.rec.AverageInstancesUsedPerHour) + return + } + require.Len(t, out, 1) + assert.Equal(t, tt.wantCount, out[0].Count) + }) + } +} + +// TestApplyTargetCoverage_SP_NoSignalGuards covers the two SP no-signal +// branches: RecommendedUtilization <= 0 (already covered by other tests) and +// the new HourlyCommitment <= 0 guard (CE occasionally returns $0 +// placeholder recs). +func TestApplyTargetCoverage_SP_NoSignalGuards(t *testing.T) { + t.Run("HourlyCommitment=0 with positive RecommendedUtilization → pass through unscaled", func(t *testing.T) { + rec := common.Recommendation{ + Service: common.ServiceSavingsPlansCompute, + CommitmentType: common.CommitmentSavingsPlan, + EstimatedSavings: 1500, + RecommendedUtilization: 50, + Details: &common.SavingsPlanDetails{HourlyCommitment: 0}, + } + out := ApplyTargetCoverage([]common.Recommendation{rec}, 80, nil) + require.Len(t, out, 1, "$0 SP rec should still be in output (pass-through)") + // Pass-through — projection fields must NOT be set, savings unchanged. + assert.Equal(t, 0.0, out[0].ProjectedUtilization, "ProjectedUtilization must NOT be set for $0-commitment pass-through") + assert.Equal(t, 1500.0, out[0].EstimatedSavings, "EstimatedSavings unchanged on pass-through") + assert.Equal(t, 0.0, out[0].Details.(*common.SavingsPlanDetails).HourlyCommitment, "HourlyCommitment unchanged") + }) + + t.Run("Details is wrong type → pass through unscaled, no projection metric set", func(t *testing.T) { + // Defensive case: SP rec with non-SP Details (a parser bug). The + // scaling can't proceed, and we MUST NOT set ProjectedUtilization + // to target% because the underlying cost fields aren't scaled — + // that would mislead the operator into thinking the rec was sized + // to the target when in fact it's the original unscaled commitment. + rec := common.Recommendation{ + Service: common.ServiceSavingsPlansCompute, + CommitmentType: common.CommitmentSavingsPlan, + EstimatedSavings: 1500, + RecommendedUtilization: 50, + Details: common.ComputeDetails{Platform: "Linux/UNIX"}, // wrong type + } + out := ApplyTargetCoverage([]common.Recommendation{rec}, 80, nil) + require.Len(t, out, 1) + assert.Equal(t, 0.0, out[0].ProjectedUtilization, "must NOT set projection when scaling failed") + assert.Equal(t, 1500.0, out[0].EstimatedSavings, "EstimatedSavings must remain unscaled when scaling failed") + }) +} + +// TestApplyTargetCoverage_DropTargetAlreadyMet verifies that when existing +// coverage already meets or exceeds the target, the recommendation is +// dropped and the drop is recorded in a non-nil DropSummary under the +// DropTargetAlreadyMet category. If the drops.Add call for that branch +// were removed, d.Total() would stay at 0 and the assertion below would fail. +func TestApplyTargetCoverage_DropTargetAlreadyMet(t *testing.T) { + // ExistingCoveragePct=90 >= target=80: gapPct=80-90=-10 <= 0 -> drop. + rec := common.Recommendation{ + Service: common.ServiceEC2, + Region: "us-east-1", + ResourceType: "t3.medium", + Count: 5, + CommitmentType: common.CommitmentReservedInstance, + AverageInstancesUsedPerHour: 8.0, + ExistingCoveragePct: 90.0, + } + + d := common.NewDropSummary() + out := ApplyTargetCoverage([]common.Recommendation{rec}, 80, d) + + assert.Empty(t, out, "already-covered rec should be dropped") + assert.Equal(t, 1, d.Total(), "drop summary should record 1 drop") + assert.Contains(t, d.FormatOneLine(), common.DropTargetAlreadyMet, + "drop summary should name the target-already-met category") +} + +// TestApplyTargetCoverage_DropTargetSizedToZero verifies that when the +// floor(avg * gapPct / 100) formula produces 0, the recommendation is +// dropped and the drop is recorded in a non-nil DropSummary under the +// DropTargetSizedToZero category. If the drops.Add call for that branch +// were removed, d.Total() would stay at 0 and the assertion below would fail. +func TestApplyTargetCoverage_DropTargetSizedToZero(t *testing.T) { + // avg=0.4, target=80%, existing=0%: floor(0.4 * 80/100) = floor(0.32) = 0 -> drop. + rec := common.Recommendation{ + Service: common.ServiceEC2, + Region: "us-east-1", + ResourceType: "t3.micro", + Count: 1, + CommitmentType: common.CommitmentReservedInstance, + AverageInstancesUsedPerHour: 0.4, + ExistingCoveragePct: 0.0, + } + + d := common.NewDropSummary() + out := ApplyTargetCoverage([]common.Recommendation{rec}, 80, d) + + assert.Empty(t, out, "floor-to-zero rec should be dropped") + assert.Equal(t, 1, d.Total(), "drop summary should record 1 drop") + assert.Contains(t, d.FormatOneLine(), common.DropTargetSizedToZero, + "drop summary should name the target-sized-to-zero category") +} diff --git a/cmd/helpers_typed_nil_details_test.go b/cmd/helpers_typed_nil_details_test.go new file mode 100644 index 000000000..b45214fc0 --- /dev/null +++ b/cmd/helpers_typed_nil_details_test.go @@ -0,0 +1,144 @@ +package main + +import ( + "testing" + + "github.com/LeanerCloud/CUDly/pkg/common" + "github.com/stretchr/testify/assert" + "github.com/stretchr/testify/require" +) + +// typedNilSPDetails returns a ServiceDetails interface holding a TYPED nil +// *SavingsPlanDetails. This is not the same value as a plain nil interface: +// a type assertion to *SavingsPlanDetails succeeds on this one (ok == true) +// and yields a nil pointer, so any unguarded dereference panics. A plain nil +// interface fails the assertion and takes the warning path instead, which is +// why only the typed form reproduces the crash. +func typedNilSPDetails() common.ServiceDetails { + var d *common.SavingsPlanDetails // nil pointer of concrete type + return d +} + +// TestTypedNilSPDetailsAssertionSucceeds pins the Go semantic the other tests +// in this file depend on. If this ever stops holding, the guards below are +// guarding nothing and the tests would pass vacuously. +func TestTypedNilSPDetailsAssertionSucceeds(t *testing.T) { + details, ok := typedNilSPDetails().(*common.SavingsPlanDetails) + require.True(t, ok, "a typed nil must still satisfy the type assertion") + require.Nil(t, details, "and must yield a nil pointer, which is what makes an unguarded deref panic") + + var plain common.ServiceDetails + _, plainOK := plain.(*common.SavingsPlanDetails) + require.False(t, plainOK, "a plain nil interface must NOT satisfy it; the two values differ") +} + +// TestScaleRecommendationCostsSurvivesTypedNilDetails covers the helper +// itself. A malformed recommendation must degrade, not crash: the cost fields +// still scale (they do not depend on Details) and Details is left exactly as +// it was rather than being replaced with a fabricated zero-value struct. +func TestScaleRecommendationCostsSurvivesTypedNilDetails(t *testing.T) { + rec := common.Recommendation{ + Service: common.ServiceSavingsPlansCompute, + Count: 4, + EstimatedSavings: 500, + CommitmentCost: 200, + OnDemandCost: 700, + Details: typedNilSPDetails(), + } + + var got common.Recommendation + require.NotPanics(t, func() { + got = common.ScaleRecommendationCosts(rec, 0.5) + }, "a typed nil *SavingsPlanDetails must not panic the sizing helper") + + assert.InDelta(t, 250.0, got.EstimatedSavings, 0.0001, "cost fields do not depend on Details and must still scale") + assert.InDelta(t, 100.0, got.CommitmentCost, 0.0001) + assert.InDelta(t, 350.0, got.OnDemandCost, 0.0001) + + details, ok := got.Details.(*common.SavingsPlanDetails) + require.True(t, ok) + assert.Nil(t, details, + "a nil commitment must stay nil rather than becoming a fabricated zero-value struct") +} + +// TestApplyInstanceLimitSurvivesTypedNilSPDetails is the #1830 path. A capped +// run must still produce a report when one recommendation is malformed; +// panicking mid-run leaves the operator with no report at all rather than a +// degraded one. +func TestApplyInstanceLimitSurvivesTypedNilSPDetails(t *testing.T) { + recs := []common.Recommendation{ + { + Service: common.ServiceSavingsPlansCompute, + ResourceType: "malformed", + Count: 10, + EstimatedSavings: 500, + Details: typedNilSPDetails(), + }, + { + Service: common.ServiceEC2, + ResourceType: "healthy", + Count: 4, + EstimatedSavings: 100, + }, + } + + var got []common.Recommendation + require.NotPanics(t, func() { + got = ApplyInstanceLimit(recs, 4) + }, "a malformed row must not crash the whole capped run") + + require.Len(t, got, 1, "the budget is spent by the first row") + assert.Equal(t, 4, got[0].Count) + assert.InDelta(t, 500.0*4.0/10.0, got[0].EstimatedSavings, 0.0001, + "the truncated row's money still rescales even though its Details are malformed") +} + +// TestApplyCoverageSurvivesTypedNilSPDetails covers the applyCoverage SP +// branch, which discards the asserted pointer and so cannot see the nil +// itself. A typed nil must take the same warning-and-pass-through path as a +// wrong Details type, since neither can have its commitment scaled. +func TestApplyCoverageSurvivesTypedNilSPDetails(t *testing.T) { + rec := common.Recommendation{ + Service: common.ServiceSavingsPlansCompute, + Count: 1, + EstimatedSavings: 500, + CommitmentCost: 200, + Details: typedNilSPDetails(), + } + + var got []common.Recommendation + require.NotPanics(t, func() { + got = ApplyCoverage([]common.Recommendation{rec}, 50) + }, "a typed nil must not panic the coverage sizing path") + + require.Len(t, got, 1, "a malformed rec is preserved, not dropped") + assert.InDelta(t, 500.0, got[0].EstimatedSavings, 0.0001, + "an unscalable commitment must pass through UNSCALED rather than scaling costs it cannot match") + assert.InDelta(t, 200.0, got[0].CommitmentCost, 0.0001) +} + +// TestApplyTargetCoverageSurvivesTypedNilSPDetails covers applyTargetCoverageSP, +// which reads HourlyCommitment off the asserted pointer for its no-signal +// guard and so dereferences before any nil check. +func TestApplyTargetCoverageSurvivesTypedNilSPDetails(t *testing.T) { + rec := common.Recommendation{ + Service: common.ServiceSavingsPlansCompute, + Count: 1, + RecommendedUtilization: 90, + EstimatedSavings: 500, + CommitmentCost: 200, + Details: typedNilSPDetails(), + } + + var got []common.Recommendation + require.NotPanics(t, func() { + got = ApplyTargetCoverage([]common.Recommendation{rec}, 80, nil) + }, "a typed nil must not panic the target-coverage sizing path") + + require.Len(t, got, 1, "a malformed rec is preserved, not dropped") + assert.InDelta(t, 500.0, got[0].EstimatedSavings, 0.0001, + "an unscalable commitment must pass through UNSCALED") + assert.InDelta(t, 200.0, got[0].CommitmentCost, 0.0001) + assert.InDelta(t, 0.0, got[0].ProjectedUtilization, 0.0001, + "projection fields stay zero on a rec whose commitment could not be scaled") +} diff --git a/cmd/lambda/clear-rate-limit/main.go b/cmd/lambda/clear-rate-limit/main.go new file mode 100644 index 000000000..ff2ed891b --- /dev/null +++ b/cmd/lambda/clear-rate-limit/main.go @@ -0,0 +1,67 @@ +package main + +import ( + "context" + "fmt" + "log" + "os" + + "github.com/LeanerCloud/CUDly/internal/database" + "github.com/aws/aws-lambda-go/lambda" +) + +const defaultDomain = "leanercloud.com" + +func getDomain() string { + if domain := os.Getenv("RATE_LIMIT_DOMAIN"); domain != "" { + return domain + } + return defaultDomain +} + +// Response is the Lambda function's return value, reporting how many +// forgot_password rate-limit rows were deleted and how many rate-limit +// rows (across all endpoints) remain in the table. +type Response struct { + Message string `json:"message"` + DeletedCount int `json:"deleted_count"` + RemainingCount int `json:"remaining_count"` +} + +func clearRateLimit(ctx context.Context) (Response, error) { + db, err := database.OpenFromEnv(ctx) + if err != nil { + return Response{}, err + } + defer db.Close() + + // Clear rate limits for forgot_password endpoint + domain := getDomain() + tag, err := db.Exec(ctx, + "DELETE FROM rate_limits WHERE id LIKE $1", + "EMAIL#%@"+domain+"#ENDPOINT#forgot_password") + if err != nil { + return Response{}, fmt.Errorf("failed to delete rate limits: %w", err) + } + + deletedCount := tag.RowsAffected() + + // Get remaining count + var remainingCount int64 + err = db.QueryRow(ctx, "SELECT COUNT(*) FROM rate_limits").Scan(&remainingCount) + if err != nil { + return Response{}, fmt.Errorf("failed to count remaining rate limits: %w", err) + } + + log.Printf("Successfully cleared %d rate limit(s), %d remaining", deletedCount, remainingCount) + + return Response{ + Message: fmt.Sprintf("Successfully cleared %d rate limit(s)", deletedCount), + DeletedCount: int(deletedCount), + RemainingCount: int(remainingCount), + }, nil +} + +func main() { + lambda.Start(clearRateLimit) +} diff --git a/cmd/lambda/main.go b/cmd/lambda/main.go new file mode 100644 index 000000000..515000d8f --- /dev/null +++ b/cmd/lambda/main.go @@ -0,0 +1,74 @@ +// Package main provides the Lambda entry point for CUDly. +// This handler uses the unified server package with PostgreSQL backend. +// It processes multiple event types: +// - Scheduled events for recommendation collection +// - HTTP requests for the dashboard API +// - Purchase approval workflow events +package main + +import ( + "context" + "encoding/json" + "fmt" + "log" + "sync" + + "github.com/LeanerCloud/CUDly/internal/server" + "github.com/aws/aws-lambda-go/lambda" +) + +// Version is set at build time. +var Version = "dev" + +var ( + app *server.Application + appMu sync.Mutex +) + +// initApp initializes the application using the unified server package. +// Uses a mutex to protect against concurrent initialization. +// +// Note: The mutex is held for the entire initialization duration. If initialization +// is slow (e.g., DB timeout), concurrent Lambda invocations (possible with provisioned +// concurrency) will block until the first goroutine completes. In practice, Lambda +// serializes cold starts, so this is not an issue. For provisioned concurrency, +// consider implementing a leader election pattern if initialization time becomes a concern. +func initApp(ctx context.Context) (*server.Application, error) { + appMu.Lock() + defer appMu.Unlock() + + if app != nil { + return app, nil + } + + log.Printf("CUDly Lambda Handler starting, version: %s", Version) + + // Initialize using the unified server package (PostgreSQL-based). + // Pass Version directly to avoid the os.Setenv round-trip (04-N1). + var err error + app, err = server.NewApplication(ctx, Version) + if err != nil { + return nil, fmt.Errorf("failed to initialize application: %w", err) + } + + log.Println("Lambda handler initialized successfully") + return app, nil +} + +// Handler is the main Lambda handler function +// This delegates to Application.HandleLambdaEvent which handles all event types. +func Handler(ctx context.Context, rawEvent json.RawMessage) (interface{}, error) { + // Initialize app on first request (lazy initialization) + application, err := initApp(ctx) + if err != nil { + // Lambda runtime will log the returned error, so no need to log here + return nil, fmt.Errorf("initialization failed: %w", err) + } + + // Delegate to the unified server package + return application.HandleLambdaEvent(ctx, rawEvent) +} + +func main() { + lambda.Start(Handler) +} diff --git a/cmd/lambda/main_test.go b/cmd/lambda/main_test.go new file mode 100644 index 000000000..d54e151d2 --- /dev/null +++ b/cmd/lambda/main_test.go @@ -0,0 +1,168 @@ +package main + +import ( + "context" + "encoding/json" + "os" + "testing" + + "github.com/LeanerCloud/CUDly/internal/api" + "github.com/LeanerCloud/CUDly/internal/server" + "github.com/LeanerCloud/CUDly/internal/testutil" + "github.com/stretchr/testify/assert" + "github.com/stretchr/testify/require" +) + +// createTestApp creates a minimal Application for testing with no DB dependency. +func createTestApp() *server.Application { + apiHandler := api.NewHandler(api.HandlerConfig{}) + return &server.Application{ + API: apiHandler, + Scheduler: &testutil.MockScheduler{}, + Purchase: &testutil.MockPurchaseManager{}, + } +} + +func TestInitApp_Cached(t *testing.T) { + // Save and restore the global app + origApp := app + defer func() { app = origApp }() + + testApp := createTestApp() + app = testApp + + result, err := initApp(context.Background()) + require.NoError(t, err) + assert.Equal(t, testApp, result, "should return cached app") +} + +// TestInitApp_SetsVersion verifies that the ldflags-stamped Version is passed +// directly to NewApplication (04-N1) rather than round-tripping through +// os.Setenv("VERSION",...) / os.Getenv("VERSION"). We verify indirectly: +// initApp is expected to fail because DB_HOST is unset, which means +// NewApplication(ctx, Version) was called with the correct value. A later +// successful init path (TestNewApplicationFromDeps in internal/server) confirms +// the field is stored on ApplicationConfig.Version. +func TestInitApp_SetsVersion(t *testing.T) { + origApp := app + origVersion := Version + origDBHost := os.Getenv("DB_HOST") + defer func() { + app = origApp + Version = origVersion + if origDBHost != "" { + os.Setenv("DB_HOST", origDBHost) + } else { + os.Unsetenv("DB_HOST") + } + }() + + app = nil + Version = "test-v1.2.3" + os.Unsetenv("DB_HOST") + + _, err := initApp(context.Background()) + // Expected to fail because DB_HOST is not set; the Version is now passed + // directly to NewApplication, not via the VERSION env var (04-N1). + require.Error(t, err) + // VERSION env var is intentionally no longer set by initApp. + assert.NotEqual(t, "test-v1.2.3", os.Getenv("VERSION"), + "VERSION env var must not be set by initApp after 04-N1 refactor") +} + +func TestInitApp_FailsWithoutDB(t *testing.T) { + origApp := app + origDBHost := os.Getenv("DB_HOST") + defer func() { + app = origApp + if origDBHost != "" { + os.Setenv("DB_HOST", origDBHost) + } else { + os.Unsetenv("DB_HOST") + } + }() + + app = nil + os.Unsetenv("DB_HOST") + + _, err := initApp(context.Background()) + require.Error(t, err) + assert.Contains(t, err.Error(), "failed to initialize application") +} + +func TestHandler_InitFailure(t *testing.T) { + origApp := app + origDBHost := os.Getenv("DB_HOST") + defer func() { + app = origApp + if origDBHost != "" { + os.Setenv("DB_HOST", origDBHost) + } else { + os.Unsetenv("DB_HOST") + } + }() + + app = nil + os.Unsetenv("DB_HOST") + + rawEvent := json.RawMessage(`{"action":"collect_recommendations"}`) + _, err := Handler(context.Background(), rawEvent) + require.Error(t, err) + assert.Contains(t, err.Error(), "initialization failed") +} + +func TestHandler_ScheduledEvent(t *testing.T) { + origApp := app + defer func() { app = origApp }() + + app = createTestApp() + + // Scheduled event - will be processed by the app + rawEvent := json.RawMessage(`{"source":"aws.events","detail-type":"Scheduled Event","action":"collect_recommendations"}`) + result, err := Handler(context.Background(), rawEvent) + require.NoError(t, err) + assert.NotNil(t, result) +} + +func TestHandler_SQSEvent(t *testing.T) { + origApp := app + defer func() { app = origApp }() + + app = createTestApp() + + rawEvent := json.RawMessage(`{"Records":[{"eventSource":"aws:sqs","messageId":"msg-1","body":"{}"}]}`) + result, err := Handler(context.Background(), rawEvent) + require.NoError(t, err) + assert.NotNil(t, result) +} + +func TestHandler_HTTPEvent(t *testing.T) { + origApp := app + defer func() { app = origApp }() + + app = createTestApp() + + rawEvent := json.RawMessage(`{"requestContext":{"http":{"method":"GET","path":"/api/health"}},"rawPath":"/api/health","headers":{}}`) + result, err := Handler(context.Background(), rawEvent) + require.NoError(t, err) + assert.NotNil(t, result) +} + +func TestHandler_ReusesApp(t *testing.T) { + origApp := app + defer func() { app = origApp }() + + app = createTestApp() + + rawEvent := json.RawMessage(`{"source":"aws.events","action":"collect_recommendations"}`) + + // Call twice - should reuse the same app + result1, err1 := Handler(context.Background(), rawEvent) + require.NoError(t, err1) + + result2, err2 := Handler(context.Background(), rawEvent) + require.NoError(t, err2) + + assert.NotNil(t, result1) + assert.NotNil(t, result2) +} diff --git a/cmd/main.go b/cmd/main.go new file mode 100644 index 000000000..509b1d58b --- /dev/null +++ b/cmd/main.go @@ -0,0 +1,387 @@ +package main + +import ( + "context" + "fmt" + "log" + "regexp" + "strings" + "time" + + "github.com/LeanerCloud/CUDly/pkg/common" + "github.com/LeanerCloud/CUDly/pkg/provider" + _ "github.com/LeanerCloud/CUDly/providers/aws" + "github.com/LeanerCloud/CUDly/providers/aws/recommendations" + "github.com/LeanerCloud/CUDly/providers/aws/services/ec2" + "github.com/LeanerCloud/CUDly/providers/aws/services/elasticache" + "github.com/LeanerCloud/CUDly/providers/aws/services/memorydb" + "github.com/LeanerCloud/CUDly/providers/aws/services/opensearch" + "github.com/LeanerCloud/CUDly/providers/aws/services/rds" + "github.com/LeanerCloud/CUDly/providers/aws/services/redshift" + "github.com/LeanerCloud/CUDly/providers/aws/services/savingsplans" + _ "github.com/LeanerCloud/CUDly/providers/azure" + _ "github.com/LeanerCloud/CUDly/providers/gcp" + "github.com/aws/aws-sdk-go-v2/aws" + "github.com/google/uuid" + "github.com/spf13/cobra" +) + +const ( + // MaxReasonableInstances is the maximum number of instances that can be processed + // This is a safety limit to prevent accidental large purchases. + MaxReasonableInstances = 10000 +) + +// Config holds all configuration for the RI helper tool. +type Config struct { + AuditLog string + CSVInput string + IdempotencyWindow string + ValidationProfile string + Profile string + PaymentOption string + CSVOutput string + Regions []string + Services []string + ExcludeAccounts []string + Providers []string + IncludeRegions []string + ExcludeRegions []string + IncludeInstanceTypes []string + ExcludeInstanceTypes []string + IncludeEngines []string + IncludeAccounts []string + ExcludeEngines []string + ExcludeSPTypes []string + IncludeSPTypes []string + MaxBreakEvenMonths int + TargetCoverage float64 + Coverage float64 + MinPoolSize float64 + RebuyWindowDays int + TermYears int + CoverageLookbackDays int + MinCount int + MinSavingsPct float64 + OverrideCount int32 + MaxInstances int32 + IncludeExtendedSupport bool + AllServices bool + ActualPurchase bool + SkipConfirmation bool + // RecLookbackPeriod controls the LookbackPeriodInDays passed to + // GetReservationPurchaseRecommendation. Valid values: "7d", "30d", "60d" + // (recommendations.DefaultRecLookbackPeriod is the shared default). + // A longer window smooths seasonal spikes; a shorter window weights + // recent demand more heavily. + RecLookbackPeriod string +} + +func main() { + if err := rootCmd.Execute(); err != nil { + log.Fatalf("Error executing command: %v", err) + } +} + +var rootCmd = &cobra.Command{ + Use: "ri-helper", + Short: "AWS Reserved Instance purchase tool based on Cost Explorer recommendations", + Long: `A tool that fetches Reserved Instance recommendations from AWS Cost Explorer +for multiple services (RDS, ElastiCache, EC2, OpenSearch, Redshift, MemoryDB) and +purchases them based on specified coverage percentage. Supports multiple regions.`, + PreRunE: validateFlags, + Run: runTool, +} + +func init() { + // Note: We still bind to package-level variables here for cobra's flag system + // These will be copied into a ToolConfig in runTool + rootCmd.Flags().StringSliceVarP(&toolCfg.Regions, "regions", "r", []string{}, "AWS regions (comma-separated or multiple flags). If empty, auto-discovers regions from recommendations") + rootCmd.Flags().StringSliceVarP(&toolCfg.Services, "services", "s", []string{"rds"}, "Services to process (rds, elasticache, ec2, opensearch, redshift, memorydb, savingsplans)") + rootCmd.Flags().BoolVar(&toolCfg.AllServices, "all-services", false, "Process all supported services") + rootCmd.Flags().Float64VarP(&toolCfg.Coverage, "coverage", "c", 80.0, "Percentage of recommendations to purchase (0-100)") + rootCmd.Flags().Float64VarP(&toolCfg.TargetCoverage, "target-coverage", "u", 0, + "Target % (0-100) of historical avg-hourly usage to cover with commitments. "+ + "When >0, sizes each rec to floor(avg * target/100) so projected coverage "+ + "approximates target and projected utilization stays near 100%, leaving "+ + "(100-target)% on-demand headroom. Overrides --coverage. Default 0 = disabled.") + rootCmd.Flags().BoolVar(&toolCfg.ActualPurchase, "purchase", false, "Actually purchase RIs instead of just printing the data") + rootCmd.Flags().StringVarP(&toolCfg.CSVOutput, "output", "o", "", "Output CSV file path (if not specified, auto-generates filename)") + rootCmd.Flags().StringVarP(&toolCfg.CSVInput, "input-csv", "i", "", "Input CSV file with recommendations to purchase") + rootCmd.Flags().StringVarP(&toolCfg.PaymentOption, "payment", "p", "no-upfront", "Payment option (all-upfront, partial-upfront, no-upfront)") + rootCmd.Flags().IntVarP(&toolCfg.TermYears, "term", "t", 3, "Term in years (1 or 3)") + rootCmd.Flags().StringVar(&toolCfg.Profile, "profile", "", "AWS profile to use (defaults to AWS_PROFILE env var or default profile)") + + // Filter flags + rootCmd.Flags().StringSliceVar(&toolCfg.IncludeRegions, "include-regions", []string{}, "Only include recommendations for these regions (comma-separated)") + rootCmd.Flags().StringSliceVar(&toolCfg.ExcludeRegions, "exclude-regions", []string{}, "Exclude recommendations for these regions (comma-separated)") + rootCmd.Flags().StringSliceVar(&toolCfg.IncludeInstanceTypes, "include-instance-types", []string{}, "Only include these instance types (comma-separated, e.g., 'db.t3.micro,cache.t3.small')") + rootCmd.Flags().StringSliceVar(&toolCfg.ExcludeInstanceTypes, "exclude-instance-types", []string{}, "Exclude these instance types (comma-separated)") + rootCmd.Flags().StringSliceVar(&toolCfg.IncludeEngines, "include-engines", []string{}, "Only include these engines (comma-separated, e.g., 'redis,mysql,postgresql')") + rootCmd.Flags().StringSliceVar(&toolCfg.ExcludeEngines, "exclude-engines", []string{}, "Exclude these engines (comma-separated)") + rootCmd.Flags().StringSliceVar(&toolCfg.IncludeAccounts, "include-accounts", []string{}, "Only include recommendations for these account names (comma-separated)") + rootCmd.Flags().StringSliceVar(&toolCfg.ExcludeAccounts, "exclude-accounts", []string{}, "Exclude recommendations for these account names (comma-separated)") + rootCmd.Flags().BoolVar(&toolCfg.SkipConfirmation, "yes", false, "Skip confirmation prompt for purchases (use with caution)") + rootCmd.Flags().Int32Var(&toolCfg.MaxInstances, "max-instances", 0, "Maximum total number of instances to purchase (0 = no limit)") + rootCmd.Flags().Int32Var(&toolCfg.OverrideCount, "override-count", 0, "Override recommendation count with fixed number for all selected RIs (0 = use recommendation or coverage)") + rootCmd.Flags().StringVar(&toolCfg.ValidationProfile, "validation-profile", "", "AWS profile to use for validating running instances (if different from main profile)") + rootCmd.Flags().BoolVar(&toolCfg.IncludeExtendedSupport, "include-extended-support", false, "Include instances running on extended support engine versions (by default they are excluded)") + + // Savings Plans specific filters + rootCmd.Flags().StringSliceVar(&toolCfg.IncludeSPTypes, "include-sp-types", []string{}, "Only include these Savings Plan types (comma-separated: Compute, EC2Instance, SageMaker, Database)") + rootCmd.Flags().StringSliceVar(&toolCfg.ExcludeSPTypes, "exclude-sp-types", []string{}, "Exclude these Savings Plan types (comma-separated: Compute, EC2Instance, SageMaker, Database)") + + // Purchase pipeline flags + rootCmd.Flags().StringVar(&toolCfg.AuditLog, "audit-log", "./cudly-audit.jsonl", "Path to JSONL audit log file") + rootCmd.Flags().StringVar(&toolCfg.IdempotencyWindow, "idempotency-window", "24h", "Lookback window for duplicate purchase detection") + rootCmd.Flags().Float64Var(&toolCfg.MinSavingsPct, "min-savings-pct", 0, "Minimum savings percentage to include a recommendation (0 = no filter)") + rootCmd.Flags().IntVar(&toolCfg.MaxBreakEvenMonths, "max-break-even-months", 0, "Maximum break-even period in months (0 = no filter)") + rootCmd.Flags().IntVar(&toolCfg.MinCount, "min-count", 0, "Minimum instance count to include a recommendation (0 = no filter)") + rootCmd.Flags().IntVar(&toolCfg.CoverageLookbackDays, "coverage-lookback-days", 30, + "Number of calendar days of historical demand fed to GetReservationCoverage "+ + "when computing the existing-RI coverage map for --target-coverage sizing. "+ + "Match this to your AWS console coverage report window to reconcile "+ + "CUDly's ExistingCoverage column against the console export. Default 30.") + rootCmd.Flags().IntVar(&toolCfg.RebuyWindowDays, "rebuy-window-days", 0, + "When >0, treat existing RIs expiring within this many days as already "+ + "uncovered, so --target-coverage sizes recommendations to replace them "+ + "before they expire. Default 0 = trust existing coverage fully.") + rootCmd.Flags().Float64Var(&toolCfg.MinPoolSize, "min-pool-size", 0, + "When >0, drop RI recommendations for pools with AverageInstancesUsedPerHour "+ + "below this threshold. Useful with --target-coverage to skip tiny pools "+ + "that integer arithmetic forces above target (e.g. avg=1 cannot hit 80%%). "+ + "Default 0 = no filter.") + rootCmd.Flags().StringVar(&toolCfg.RecLookbackPeriod, "rec-lookback-period", recommendations.DefaultRecLookbackPeriod, + "Historical window for GetReservationPurchaseRecommendation. "+ + "Valid values: 7d, 30d, 60d. A longer window smooths seasonal spikes; "+ + "a shorter window weights recent demand more heavily. Default 7d.") +} + +// Package-level Config that cobra flags bind to. +var toolCfg = Config{} + +// validateFlags is now defined in validators.go + +// parseServices converts service names to ServiceType. The legacy +// "savingsplans" / "sp" aliases fan out to all four per-plan-type SP slugs so +// existing CLI scripts that pass --services savingsplans keep covering every +// plan type. Specific slugs (savingsplans-compute, etc.) are also accepted for +// targeted runs. +// +// Duplicates are silently dropped via a `seen` set so combinations like +// `--services savingsplans,savingsplans-compute` don't double-process Compute +// SP through both the fan-out path and the explicit-slug path. +func parseServices(serviceNames []string) []common.ServiceType { + var result []common.ServiceType + seen := make(map[common.ServiceType]struct{}) + add := func(service common.ServiceType) { + if _, ok := seen[service]; ok { + return + } + seen[service] = struct{}{} + result = append(result, service) + } + allSPSlugs := []common.ServiceType{ + common.ServiceSavingsPlansCompute, + common.ServiceSavingsPlansEC2Instance, + common.ServiceSavingsPlansSageMaker, + common.ServiceSavingsPlansDatabase, + } + serviceMap := map[string]common.ServiceType{ + "rds": common.ServiceRDS, + "elasticache": common.ServiceElastiCache, + "ec2": common.ServiceEC2, + "opensearch": common.ServiceOpenSearch, + "elasticsearch": common.ServiceOpenSearch, // Legacy alias maps to OpenSearch + "redshift": common.ServiceRedshift, + "memorydb": common.ServiceMemoryDB, + "savingsplans-compute": common.ServiceSavingsPlansCompute, + "savingsplans-ec2instance": common.ServiceSavingsPlansEC2Instance, + "savingsplans-sagemaker": common.ServiceSavingsPlansSageMaker, + "savingsplans-database": common.ServiceSavingsPlansDatabase, + "savings-plans-compute": common.ServiceSavingsPlansCompute, + "savings-plans-ec2instance": common.ServiceSavingsPlansEC2Instance, + "savings-plans-sagemaker": common.ServiceSavingsPlansSageMaker, + "savings-plans-database": common.ServiceSavingsPlansDatabase, + } + + for _, name := range serviceNames { + key := strings.ToLower(name) + if key == "savingsplans" || key == "savings-plans" || key == "sp" { + for _, service := range allSPSlugs { + add(service) + } + continue + } + if service, ok := serviceMap[key]; ok { + add(service) + } else { + log.Printf("Warning: Unknown service '%s', skipping", name) + } + } + + return result +} + +// getAllServices returns all supported services. +func getAllServices() []common.ServiceType { + return []common.ServiceType{ + common.ServiceRDS, + common.ServiceElastiCache, + common.ServiceEC2, + common.ServiceOpenSearch, + common.ServiceRedshift, + common.ServiceMemoryDB, + common.ServiceSavingsPlansCompute, + common.ServiceSavingsPlansEC2Instance, + common.ServiceSavingsPlansSageMaker, + common.ServiceSavingsPlansDatabase, + } +} + +// createServiceClient creates the appropriate service client for a service. +func createServiceClient(service common.ServiceType, cfg aws.Config) provider.ServiceClient { + switch service { + case common.ServiceRDS: + return rds.NewClient(cfg) + case common.ServiceElastiCache: + return elasticache.NewClient(cfg) + case common.ServiceEC2: + return ec2.NewClient(cfg) + case common.ServiceOpenSearch: + return opensearch.NewClient(cfg) + case common.ServiceRedshift: + return redshift.NewClient(cfg) + case common.ServiceMemoryDB: + return memorydb.NewClient(cfg) + case common.ServiceSavingsPlansCompute, + common.ServiceSavingsPlansEC2Instance, + common.ServiceSavingsPlansSageMaker, + common.ServiceSavingsPlansDatabase: + pt, ok := savingsplans.PlanTypeForServiceType(service) + if !ok { + return nil + } + return savingsplans.NewClient(cfg, pt) + default: + return nil + } +} + +// effectiveSizingPct returns the sizing % actually applied to recommendations: +// cfg.TargetCoverage when set (>0), else cfg.Coverage. Use this when +// emitting human-facing labels (purchase IDs, audit-log fields) so the label +// reflects the value that drove the sizing, not the unused default. +func effectiveSizingPct(cfg Config) float64 { + if cfg.TargetCoverage > 0 { + return cfg.TargetCoverage + } + return cfg.Coverage +} + +// extractEngineLabel returns the engine or platform string from the +// polymorphic Details field. DatabaseDetails/CacheDetails are always +// pointers (every producer constructs them that way; see +// pkg/common/service_details_codec.go's package doc for the pointer +// invariant). ComputeDetails is still accepted as a value because the +// Azure compute client and the GCP compute-engine client both construct +// it that way. +func extractEngineLabel(details interface{}) string { + switch d := details.(type) { + case *common.DatabaseDetails: + if d != nil { + return d.Engine + } + case *common.CacheDetails: + if d != nil { + return d.Engine + } + case common.ComputeDetails: + return d.Platform + case *common.ComputeDetails: + if d != nil { + return d.Platform + } + } + return "" +} + +// generatePurchaseID creates a descriptive purchase ID with UUID for uniqueness. +// sizingPct is the percentage that actually drove the sizing decision (see +// effectiveSizingPct); it appears in the ID as e.g. "80pct" purely for human +// readability and audit traceability. +func generatePurchaseID(rec common.Recommendation, region string, _ int, isDryRun bool, sizingPct float64) string { + // Generate a short UUID suffix (first 8 characters) for uniqueness + uuidSuffix := uuid.New().String()[:8] + timestamp := time.Now().Format("20060102-150405") + prefix := "ri" + if isDryRun { + prefix = "dryrun" + } + + service := strings.ToLower(string(rec.Service)) + instanceType := strings.ReplaceAll(rec.ResourceType, ".", "-") + + raw := extractEngineLabel(rec.Details) + engine := strings.ToLower(raw) + engine = strings.ReplaceAll(engine, " ", "-") + engine = strings.ReplaceAll(engine, "_", "-") + engine = strings.ReplaceAll(engine, "/", "-") + + // Add account name if available + accountName := sanitizeAccountName(rec.AccountName) + coveragePct := fmt.Sprintf("%.0fpct", sizingPct) + if accountName != "" { + if engine != "" { + return fmt.Sprintf("%s-%s-%s-%s-%s-%s-%dx-%s-%s-%s", + prefix, accountName, service, engine, region, instanceType, rec.Count, coveragePct, timestamp, uuidSuffix) + } + return fmt.Sprintf("%s-%s-%s-%s-%s-%dx-%s-%s-%s", + prefix, accountName, service, region, instanceType, rec.Count, coveragePct, timestamp, uuidSuffix) + } + + // Fallback without account name + if engine != "" { + return fmt.Sprintf("%s-%s-%s-%s-%s-%dx-%s-%s-%s", + prefix, service, engine, region, instanceType, rec.Count, coveragePct, timestamp, uuidSuffix) + } + return fmt.Sprintf("%s-%s-%s-%s-%dx-%s-%s-%s", + prefix, service, region, instanceType, rec.Count, coveragePct, timestamp, uuidSuffix) +} + +// sanitizeAccountName converts account name to a filesystem/ID-safe format. +func sanitizeAccountName(accountName string) string { + if accountName == "" { + return "" + } + + // Convert to lowercase + clean := strings.ToLower(accountName) + + // Replace spaces and special chars with hyphens + clean = strings.ReplaceAll(clean, " ", "-") + clean = strings.ReplaceAll(clean, "_", "-") + clean = strings.ReplaceAll(clean, ".", "-") + + // Remove any characters that aren't alphanumeric or hyphens + var b strings.Builder + for _, r := range clean { + if (r >= 'a' && r <= 'z') || (r >= '0' && r <= '9') || r == '-' { + b.WriteRune(r) + } + } + result := b.String() + + // Remove leading/trailing hyphens and collapse multiple hyphens + result = strings.Trim(result, "-") + result = regexp.MustCompile("-{2,}").ReplaceAllString(result, "-") + + return result +} + +func runTool(cmd *cobra.Command, args []string) { + ctx := context.Background() + + // Always use the multi-service implementation + runToolMultiService(ctx, toolCfg) +} diff --git a/cmd/main_test.go b/cmd/main_test.go new file mode 100644 index 000000000..4432a749e --- /dev/null +++ b/cmd/main_test.go @@ -0,0 +1,1239 @@ +package main + +import ( + "fmt" + "strings" + "testing" + + "github.com/LeanerCloud/CUDly/pkg/common" + "github.com/aws/aws-sdk-go-v2/aws" + "github.com/stretchr/testify/assert" +) + +func TestParseServices(t *testing.T) { + tests := []struct { + name string + input []string + expected []common.ServiceType + }{ + { + name: "Valid services", + input: []string{"rds", "elasticache", "ec2"}, + expected: []common.ServiceType{ + common.ServiceRDS, + common.ServiceElastiCache, + common.ServiceEC2, + }, + }, + { + name: "Mixed case services", + input: []string{"RDS", "ElastiCache", "EC2"}, + expected: []common.ServiceType{ + common.ServiceRDS, + common.ServiceElastiCache, + common.ServiceEC2, + }, + }, + { + name: "Invalid services", + input: []string{"invalid", "unknown"}, + expected: nil, + }, + { + name: "Mix of valid and invalid", + input: []string{"rds", "invalid", "ec2"}, + expected: []common.ServiceType{ + common.ServiceRDS, + common.ServiceEC2, + }, + }, + { + name: "All supported services", + input: []string{"rds", "elasticache", "ec2", "opensearch", "redshift", "memorydb"}, + expected: []common.ServiceType{ + common.ServiceRDS, + common.ServiceElastiCache, + common.ServiceEC2, + common.ServiceOpenSearch, + common.ServiceRedshift, + common.ServiceMemoryDB, + }, + }, + { + name: "Legacy elasticsearch alias", + input: []string{"elasticsearch"}, + expected: []common.ServiceType{ + common.ServiceElasticsearch, + }, + }, + } + + for _, tt := range tests { + t.Run(tt.name, func(t *testing.T) { + result := parseServices(tt.input) + assert.Equal(t, tt.expected, result) + }) + } +} + +func TestGetAllServices(t *testing.T) { + services := getAllServices() + + expected := []common.ServiceType{ + common.ServiceRDS, + common.ServiceElastiCache, + common.ServiceEC2, + common.ServiceOpenSearch, + common.ServiceRedshift, + common.ServiceMemoryDB, + common.ServiceSavingsPlansCompute, + common.ServiceSavingsPlansEC2Instance, + common.ServiceSavingsPlansSageMaker, + common.ServiceSavingsPlansDatabase, + } + + assert.Equal(t, expected, services) +} + +func TestGeneratePurchaseID(t *testing.T) { + tests := []struct { + name string + region string + expectedPrefix string + rec common.Recommendation + index int + coverage float64 + isDryRun bool + }{ + { + name: "RDS Recommendation - dry run", + rec: common.Recommendation{ + Service: common.ServiceRDS, + ResourceType: "db.t3.micro", + Count: 2, + Details: &common.DatabaseDetails{ + Engine: "mysql", + AZConfig: "single-az", + }, + }, + region: "us-east-1", + index: 1, + isDryRun: true, + coverage: 80.0, + expectedPrefix: "dryrun-rds-mysql-us-east-1-db-t3-micro-2x", + }, + { + name: "EC2 Recommendation - actual purchase", + rec: common.Recommendation{ + Service: common.ServiceEC2, + ResourceType: "t3.large", + Count: 5, + }, + region: "eu-west-1", + index: 3, + isDryRun: false, + coverage: 80.0, + expectedPrefix: "ri-ec2-eu-west-1-t3-large-5x", + }, + { + name: "ElastiCache Recommendation - dry run", + rec: common.Recommendation{ + Service: common.ServiceElastiCache, + ResourceType: "cache.r5.large", + Count: 1, + Details: &common.CacheDetails{ + Engine: "redis", + }, + }, + region: "us-west-2", + index: 2, + isDryRun: true, + coverage: 80.0, + expectedPrefix: "dryrun-elasticache-redis-us-west-2-cache-r5-large-1x", + }, + { + name: "RDS Recommendation - multi-AZ", + rec: common.Recommendation{ + Service: common.ServiceRDS, + ResourceType: "db.m5.xlarge", + Count: 3, + Details: &common.DatabaseDetails{ + Engine: "postgres", + AZConfig: "multi-az", + }, + }, + region: "ap-southeast-1", + index: 5, + isDryRun: false, + coverage: 80.0, + expectedPrefix: "ri-rds-postgres-ap-southeast-1-db-m5-xlarge-3x", + }, + } + + for _, tt := range tests { + t.Run(tt.name, func(t *testing.T) { + result := generatePurchaseID(tt.rec, tt.region, tt.index, tt.isDryRun, tt.coverage) + assert.Contains(t, result, tt.expectedPrefix) + // Should contain timestamp (YYYYMMDD-HHMMSS) and UUID suffix (8 chars) + assert.Regexp(t, `-\d{8}-\d{6}-[a-f0-9]{8}$`, result) + }) + } +} + +func TestCreateServiceClient(t *testing.T) { + cfg := aws.Config{ + Region: "us-east-1", + } + + tests := []struct { + name string + service common.ServiceType + expectNil bool + }{ + { + name: "RDS service", + service: common.ServiceRDS, + expectNil: false, + }, + { + name: "ElastiCache service", + service: common.ServiceElastiCache, + expectNil: false, + }, + { + name: "EC2 service", + service: common.ServiceEC2, + expectNil: false, + }, + { + name: "OpenSearch service", + service: common.ServiceOpenSearch, + expectNil: false, + }, + { + name: "Redshift service", + service: common.ServiceRedshift, + expectNil: false, + }, + { + name: "MemoryDB service", + service: common.ServiceMemoryDB, + expectNil: false, + }, + { + name: "Compute Savings Plans service", + service: common.ServiceSavingsPlansCompute, + expectNil: false, + }, + { + name: "EC2 Instance Savings Plans service", + service: common.ServiceSavingsPlansEC2Instance, + expectNil: false, + }, + { + name: "SageMaker Savings Plans service", + service: common.ServiceSavingsPlansSageMaker, + expectNil: false, + }, + { + name: "Database Savings Plans service", + service: common.ServiceSavingsPlansDatabase, + expectNil: false, + }, + { + // Umbrella sentinel is no longer dispatched to a client — the + // per-plan-type slugs above carry the actual work. Keep the case + // to lock the contract: createServiceClient returns nil for the + // umbrella so the caller knows to fan out. + name: "Savings Plans umbrella sentinel (returns nil)", + service: common.ServiceSavingsPlansAll, + expectNil: true, + }, + { + name: "Unknown service", + service: common.ServiceType("unknown"), + expectNil: true, + }, + } + + for _, tt := range tests { + t.Run(tt.name, func(t *testing.T) { + client := createServiceClient(tt.service, cfg) + if tt.expectNil { + assert.Nil(t, client) + } else { + assert.NotNil(t, client) + } + }) + } +} + +func TestRunTool(t *testing.T) { + // Skip this integration test that requires AWS credentials + t.Skip("Skipping integration test that requires AWS credentials - functionality tested in TestProcessServiceWithMocks") + + // This test would validate that runTool correctly delegates to runToolMultiService + // but it requires actual AWS credentials to run. + // The actual functionality is tested in TestProcessServiceWithMocks in multi_service_test.go + // which uses mocked AWS clients and doesn't require credentials. +} + +func TestInit(t *testing.T) { + // Test that init properly sets up command flags + // This is called automatically, so we just verify the flags exist + + assert.NotNil(t, rootCmd) + assert.Equal(t, "ri-helper", rootCmd.Use) + + // Check that flags are defined + flag := rootCmd.Flags().Lookup("regions") + assert.NotNil(t, flag) + assert.Equal(t, "r", flag.Shorthand) + + flag = rootCmd.Flags().Lookup("services") + assert.NotNil(t, flag) + assert.Equal(t, "s", flag.Shorthand) + + flag = rootCmd.Flags().Lookup("all-services") + assert.NotNil(t, flag) + + flag = rootCmd.Flags().Lookup("coverage") + assert.NotNil(t, flag) + assert.Equal(t, "c", flag.Shorthand) + + flag = rootCmd.Flags().Lookup("purchase") + assert.NotNil(t, flag) + + flag = rootCmd.Flags().Lookup("output") + assert.NotNil(t, flag) + assert.Equal(t, "o", flag.Shorthand) + + flag = rootCmd.Flags().Lookup("payment") + assert.NotNil(t, flag) + assert.Equal(t, "p", flag.Shorthand) + + flag = rootCmd.Flags().Lookup("term") + assert.NotNil(t, flag) + assert.Equal(t, "t", flag.Shorthand) +} + +func TestMainFunction(t *testing.T) { + // Save original args + origArgs := rootCmd.Args + + // Test with help flag to avoid actual execution + rootCmd.SetArgs([]string{"--help"}) + + // Run main should not panic + defer func() { + if r := recover(); r != nil { + t.Errorf("main() panicked: %v", r) + } + // Restore + rootCmd.Args = origArgs + }() + + // We can't easily test main() directly due to log.Fatalf + // but we can test the command structure + assert.NotNil(t, rootCmd) +} + +func TestCommandFlags(t *testing.T) { + tests := []struct { + name string + flagName string + shorthand string + defaultValue string + }{ + {"regions flag", "regions", "r", "[]"}, + {"services flag", "services", "s", "[rds]"}, + {"coverage flag", "coverage", "c", "80"}, + {"payment flag", "payment", "p", "no-upfront"}, + {"term flag", "term", "t", "3"}, + {"output flag", "output", "o", ""}, + } + + for _, tt := range tests { + t.Run(tt.name, func(t *testing.T) { + flag := rootCmd.Flags().Lookup(tt.flagName) + assert.NotNil(t, flag) + if tt.shorthand != "" { + assert.Equal(t, tt.shorthand, flag.Shorthand) + } + assert.Equal(t, tt.defaultValue, flag.DefValue) + }) + } +} + +func TestGeneratePurchaseIDEdgeCases(t *testing.T) { + testCoverage := 80.0 + + // Test with recommendations that have special characters + rec := common.Recommendation{ + Service: common.ServiceRDS, + ResourceType: "db.r5b.2xlarge", + Count: 10, + Details: &common.DatabaseDetails{ + Engine: "MySQL 8.0", + AZConfig: "single-az", + }, + } + + id := generatePurchaseID(rec, "us-east-1", 999, false, testCoverage) + assert.Contains(t, id, "rds") + assert.Contains(t, id, "r5b-2xlarge") + assert.Contains(t, id, "10x") + // Index is no longer included due to UUID replacement + + // Test with empty region + id = generatePurchaseID(rec, "", 1, true, testCoverage) + assert.Contains(t, id, "dryrun") + + // Test with very long instance type + rec.ResourceType = "db.x2gd.metal.16xlarge" + id = generatePurchaseID(rec, "ap-south-1", 1, false, testCoverage) + assert.Contains(t, id, "x2gd-metal") +} + +func TestGeneratePurchaseIDComprehensive(t *testing.T) { + // Use a test coverage value + testCoverage := 75.0 + + tests := []struct { + name string + region string + rec common.Recommendation + expectedContains []string + expectedNotContains []string + isDryRun bool + }{ + { + name: "RDS with account name and engine", + rec: common.Recommendation{ + Service: common.ServiceRDS, + ResourceType: "db.r5.large", + Count: 3, + AccountName: "Production Account", + Details: &common.DatabaseDetails{ + Engine: "PostgreSQL", + }, + }, + region: "eu-west-1", + isDryRun: false, + expectedContains: []string{ + "ri-", "production-account", "rds", "postgresql", "eu-west-1", + "db-r5-large", "3x", "75pct", + }, + }, + { + name: "ElastiCache with Redis engine", + rec: common.Recommendation{ + Service: common.ServiceElastiCache, + ResourceType: "cache.r5.xlarge", + Count: 5, + Details: &common.CacheDetails{ + Engine: "Redis", + }, + }, + region: "us-west-2", + isDryRun: false, + expectedContains: []string{ + "ri-", "elasticache", "redis", "us-west-2", + "cache-r5-xlarge", "5x", "75pct", + }, + }, + { + name: "EC2 with platform", + rec: common.Recommendation{ + Service: common.ServiceEC2, + ResourceType: "m5.2xlarge", + Count: 10, + Details: common.ComputeDetails{ + Platform: "Linux/UNIX", + }, + }, + region: "ap-southeast-1", + isDryRun: true, + expectedContains: []string{ + "dryrun-", "ec2", "linux-unix", "ap-southeast-1", + "m5-2xlarge", "10x", "75pct", + }, + }, + { + name: "MemoryDB recommendation", + rec: common.Recommendation{ + Service: common.ServiceMemoryDB, + ResourceType: "db.r6g.large", + Count: 2, + Details: &common.CacheDetails{Engine: "redis"}, + }, + region: "us-east-1", + isDryRun: false, + expectedContains: []string{ + "ri-", "memorydb", "memorydb", "us-east-1", + "db-r6g-large", "2x", "75pct", + }, + }, + { + name: "OpenSearch without engine", + rec: common.Recommendation{ + Service: common.ServiceOpenSearch, + ResourceType: "r5.large.search", + Count: 4, + }, + region: "eu-central-1", + isDryRun: false, + expectedContains: []string{ + "ri-", "opensearch", "eu-central-1", + "r5-large-search", "4x", "75pct", + }, + }, + { + name: "Elasticsearch alias (same as OpenSearch)", + rec: common.Recommendation{ + Service: common.ServiceElasticsearch, // Should work as alias for OpenSearch + ResourceType: "m5.xlarge.elasticsearch", + Count: 3, + }, + region: "us-west-1", + isDryRun: false, + expectedContains: []string{ + "ri-", "opensearch", "us-west-1", + "m5-xlarge-elasticsearch", "3x", "75pct", + }, + }, + { + name: "Redshift without engine", + rec: common.Recommendation{ + Service: common.ServiceRedshift, + ResourceType: "dc2.large", + Count: 8, + }, + region: "us-east-2", + isDryRun: false, + expectedContains: []string{ + "ri-", "redshift", "us-east-2", + "dc2-large", "8x", "75pct", + }, + }, + { + name: "RDS recommendation with account", + rec: common.Recommendation{ + Service: common.ServiceRDS, + ResourceType: "db.r6g.xlarge", + Count: 15, + AccountName: "Staging", + Details: &common.DatabaseDetails{ + Engine: "aurora-mysql", + AZConfig: "multi-az", + }, + }, + region: "ca-central-1", + isDryRun: false, + expectedContains: []string{ + "ri-", "staging", "aurora-mysql", "r6g-xlarge", + "15x", "75pct", "ca-central-1", + }, + }, + { + name: "ElastiCache single-AZ recommendation", + rec: common.Recommendation{ + Service: common.ServiceElastiCache, + ResourceType: "cache.m5.large", + Count: 1, + Details: &common.CacheDetails{ + Engine: "redis", + }, + }, + region: "ap-northeast-1", + isDryRun: true, + expectedContains: []string{ + "dryrun-", "redis", "m5-large", + "1x", "75pct", "ap-northeast-1", + }, + }, + { + name: "Recommendation with special characters in engine", + rec: common.Recommendation{ + Service: common.ServiceRDS, + ResourceType: "db.t3.micro", + Count: 20, + Details: &common.DatabaseDetails{ + Engine: "MySQL_8.0_Community", + }, + }, + region: "us-west-1", + isDryRun: false, + expectedContains: []string{ + "ri-", "rds", "mysql-8.0-community", + "db-t3-micro", "20x", "75pct", + }, + }, + { + name: "Large count recommendation", + rec: common.Recommendation{ + Service: common.ServiceEC2, + ResourceType: "t3.nano", + Count: 999, + }, + region: "eu-west-2", + isDryRun: false, + expectedContains: []string{ + "ri-", "ec2", "eu-west-2", + "t3-nano", "999x", "75pct", + }, + }, + } + + for _, tt := range tests { + t.Run(tt.name, func(t *testing.T) { + result := generatePurchaseID(tt.rec, tt.region, 1, tt.isDryRun, testCoverage) + + // Check expected contains + for _, expected := range tt.expectedContains { + assert.Contains(t, result, expected, "Expected ID to contain '%s'", expected) + } + + // Check expected not contains + for _, notExpected := range tt.expectedNotContains { + assert.NotContains(t, result, notExpected, "Expected ID not to contain '%s'", notExpected) + } + + // Should always contain timestamp and UUID + assert.Regexp(t, `-\d{8}-\d{6}-[a-f0-9]{8}$`, result) + }) + } +} + +func TestGeneratePurchaseIDCoverageVariations(t *testing.T) { + rec := common.Recommendation{ + Service: common.ServiceRDS, + ResourceType: "db.t3.small", + Count: 1, + Details: &common.DatabaseDetails{ + Engine: "mysql", + }, + } + + tests := []struct { + name string + expectedCoverage string + coverage float64 + }{ + {"Coverage 0%", "0pct", 0.0}, + {"Coverage 50%", "50pct", 50.0}, + {"Coverage 75.5%", "76pct", 75.5}, // Rounds to nearest integer + {"Coverage 99%", "99pct", 99.0}, + {"Coverage 100%", "100pct", 100.0}, + {"Coverage 33.3%", "33pct", 33.3}, + } + + for _, tt := range tests { + t.Run(tt.name, func(t *testing.T) { + result := generatePurchaseID(rec, "us-east-1", 1, false, tt.coverage) + assert.Contains(t, result, tt.expectedCoverage) + }) + } +} + +// TestEffectiveSizingPct locks the rule that target-coverage wins over +// coverage when both are configured. Without this, dry-run purchase IDs (and +// any other audit label that calls effectiveSizingPct) print the unused +// default Coverage=80 instead of the actual target value the user passed. +// Regression guard for the Aurora-target / RDS-target CLI dry-run runs. +func TestEffectiveSizingPct(t *testing.T) { + tests := []struct { + name string + cfg Config + want float64 + }{ + {"target unset, coverage default", Config{Coverage: 80}, 80}, + {"target unset, coverage custom", Config{Coverage: 50}, 50}, + {"target set, coverage default ignored", Config{TargetCoverage: 70, Coverage: 80}, 70}, + {"target set, coverage explicit ignored", Config{TargetCoverage: 95, Coverage: 30}, 95}, + {"target zero falls back to coverage", Config{TargetCoverage: 0, Coverage: 60}, 60}, + } + for _, tt := range tests { + t.Run(tt.name, func(t *testing.T) { + assert.Equal(t, tt.want, effectiveSizingPct(tt.cfg)) + }) + } +} + +func TestParseServicesWithEmptyAndNil(t *testing.T) { + // Empty slice + result := parseServices([]string{}) + assert.Empty(t, result) + + // Slice with empty strings + result = parseServices([]string{"", "rds", ""}) + assert.Len(t, result, 1) + assert.Equal(t, common.ServiceRDS, result[0]) + + // All invalid + result = parseServices([]string{"foo", "bar", "baz"}) + assert.Empty(t, result) +} + +func TestFilterFlagValidation(t *testing.T) { + // Save original toolCfg values + origCfg := toolCfg + + defer func() { + toolCfg = origCfg + }() + + tests := []struct { + name string + errorContains string + includeRegions []string + excludeRegions []string + includeInstanceTypes []string + excludeInstanceTypes []string + expectError bool + }{ + { + name: "No conflicts", + includeRegions: []string{"us-east-1"}, + excludeRegions: []string{"us-west-2"}, + includeInstanceTypes: []string{"db.t3.micro"}, + excludeInstanceTypes: []string{"db.t3.large"}, + expectError: false, + }, + { + name: "Region conflict", + includeRegions: []string{"us-east-1", "us-west-2"}, + excludeRegions: []string{"us-west-2"}, + includeInstanceTypes: []string{}, + excludeInstanceTypes: []string{}, + expectError: true, + errorContains: "region 'us-west-2' cannot be both included and excluded", + }, + { + name: "Instance type conflict", + includeRegions: []string{}, + excludeRegions: []string{}, + includeInstanceTypes: []string{"db.t3.small"}, + excludeInstanceTypes: []string{"db.t3.small"}, + expectError: true, + errorContains: "instance type 'db.t3.small' cannot be both included and excluded", + }, + { + name: "Empty filters valid", + includeRegions: []string{}, + excludeRegions: []string{}, + includeInstanceTypes: []string{}, + excludeInstanceTypes: []string{}, + expectError: false, + }, + } + + for _, tt := range tests { + t.Run(tt.name, func(t *testing.T) { + // Set test values in toolCfg + toolCfg.IncludeRegions = tt.includeRegions + toolCfg.ExcludeRegions = tt.excludeRegions + toolCfg.IncludeInstanceTypes = tt.includeInstanceTypes + toolCfg.ExcludeInstanceTypes = tt.excludeInstanceTypes + toolCfg.Coverage = 80.0 + toolCfg.PaymentOption = "no-upfront" + toolCfg.TermYears = 3 + + // Call validateFlags + err := validateFlags(nil, nil) + + if tt.expectError { + assert.Error(t, err) + if err != nil && tt.errorContains != "" { + assert.Contains(t, err.Error(), tt.errorContains) + } + } else { + assert.NoError(t, err) + } + }) + } +} + +func TestCreateServiceClientAllServices(t *testing.T) { + cfg := aws.Config{ + Region: "eu-central-1", + } + + // Test that all services return non-nil clients now + services := getAllServices() + for _, service := range services { + client := createServiceClient(service, cfg) + assert.NotNil(t, client, "Service %s should have a client", service) + } +} + +func TestValidateFlags(t *testing.T) { + tests := []struct { + name string + setPayment string + setCoverage float64 + setTerm int + expectError bool + }{ + { + name: "Valid flags", + setCoverage: 80.0, + setTerm: 1, + setPayment: "partial-upfront", + expectError: false, + }, + { + name: "Coverage too high", + setCoverage: 150.0, + setTerm: 1, + setPayment: "partial-upfront", + expectError: true, + }, + { + name: "Coverage negative", + setCoverage: -10.0, + setTerm: 1, + setPayment: "partial-upfront", + expectError: true, + }, + { + name: "Invalid term", + setCoverage: 80.0, + setTerm: 2, + setPayment: "partial-upfront", + expectError: true, + }, + { + name: "Invalid payment option", + setCoverage: 80.0, + setTerm: 1, + setPayment: "invalid", + expectError: true, + }, + } + + for _, tt := range tests { + t.Run(tt.name, func(t *testing.T) { + // Save original values + origCfg := toolCfg + + // Set test values + toolCfg.Coverage = tt.setCoverage + toolCfg.TermYears = tt.setTerm + toolCfg.PaymentOption = tt.setPayment + + // Call validateFlags + err := validateFlags(nil, []string{}) + + // Restore original values + toolCfg = origCfg + + if tt.expectError { + assert.Error(t, err) + } else { + assert.NoError(t, err) + } + }) + } +} + +func TestValidateFlagsExtended(t *testing.T) { + // Save original toolCfg + origCfg := toolCfg + + defer func() { + toolCfg = origCfg + }() + + tests := []struct { + setCSVInput string + setCSVOutput string + errorContains string + setPayment string + name string + setIncludeEngines []string + setExcludeAccounts []string + setIncludeAccounts []string + setExcludeEngines []string + setIncludeTypes []string + setExcludeTypes []string + setCoverage float64 + setTerm int + setMaxInstances int32 + expectError bool + }{ + // Coverage boundary tests + { + name: "Coverage at minimum boundary (0)", + setCoverage: 0.0, + setTerm: 1, + setPayment: "no-upfront", + expectError: false, + }, + { + name: "Coverage at maximum boundary (100)", + setCoverage: 100.0, + setTerm: 1, + setPayment: "no-upfront", + expectError: false, + }, + { + name: "Coverage below minimum", + setCoverage: -0.001, + setTerm: 1, + setPayment: "no-upfront", + expectError: true, + errorContains: "coverage percentage must be between 0 and 100", + }, + { + name: "Coverage above maximum", + setCoverage: 100.001, + setTerm: 1, + setPayment: "no-upfront", + expectError: true, + errorContains: "coverage percentage must be between 0 and 100", + }, + { + name: "Coverage with decimals", + setCoverage: 75.5, + setTerm: 3, + setPayment: "partial-upfront", + expectError: false, + }, + + // Max instances tests + { + name: "Max instances zero (no limit)", + setCoverage: 80.0, + setTerm: 1, + setPayment: "no-upfront", + setMaxInstances: 0, + expectError: false, + }, + { + name: "Max instances positive", + setCoverage: 80.0, + setTerm: 1, + setPayment: "no-upfront", + setMaxInstances: 100, + expectError: false, + }, + { + name: "Max instances negative", + setCoverage: 80.0, + setTerm: 1, + setPayment: "no-upfront", + setMaxInstances: -5, + expectError: true, + errorContains: "max-instances must be 0", + }, + + // Payment option tests + { + name: "Payment all-upfront", + setCoverage: 80.0, + setTerm: 3, + setPayment: "all-upfront", + expectError: false, + }, + { + name: "Payment invalid mixed case", + setCoverage: 80.0, + setTerm: 1, + setPayment: "All-Upfront", + expectError: true, + errorContains: "invalid payment option", + }, + { + name: "Payment empty string", + setCoverage: 80.0, + setTerm: 1, + setPayment: "", + expectError: true, + errorContains: "invalid payment option", + }, + + // Term tests + { + name: "Term zero", + setCoverage: 80.0, + setTerm: 0, + setPayment: "no-upfront", + expectError: true, + errorContains: "invalid term", + }, + { + name: "Term negative", + setCoverage: 80.0, + setTerm: -1, + setPayment: "no-upfront", + expectError: true, + errorContains: "invalid term", + }, + { + name: "Term five years", + setCoverage: 80.0, + setTerm: 5, + setPayment: "no-upfront", + expectError: true, + errorContains: "invalid term", + }, + + // Engine conflict tests + { + name: "Engine conflict", + setCoverage: 80.0, + setTerm: 1, + setPayment: "no-upfront", + setIncludeEngines: []string{"mysql", "postgres"}, + setExcludeEngines: []string{"postgres", "redis"}, + expectError: true, + errorContains: "engine 'postgres' cannot be both included and excluded", + }, + { + name: "No engine conflict", + setCoverage: 80.0, + setTerm: 1, + setPayment: "no-upfront", + setIncludeEngines: []string{"mysql"}, + setExcludeEngines: []string{"postgres"}, + expectError: false, + }, + + // Instance type validation tests + { + name: "Invalid include instance type", + setCoverage: 80.0, + setTerm: 1, + setPayment: "no-upfront", + setIncludeTypes: []string{"invalidtype"}, // No dot, should fail validation + expectError: true, + errorContains: "invalid include-instance-types", + }, + { + name: "Invalid exclude instance type", + setCoverage: 80.0, + setTerm: 1, + setPayment: "no-upfront", + setExcludeTypes: []string{"badinstance"}, // No dot, should fail validation + expectError: true, + errorContains: "invalid exclude-instance-types", + }, + { + name: "Valid instance types", + setCoverage: 80.0, + setTerm: 1, + setPayment: "no-upfront", + setIncludeTypes: []string{"db.t3.small", "cache.t3.small"}, + setExcludeTypes: []string{"db.m5.large"}, + expectError: false, + }, + + // Combined validations + { + name: "All valid flags combined", + setCoverage: 85.5, + setTerm: 3, + setPayment: "partial-upfront", + setMaxInstances: 50, + setIncludeTypes: []string{"db.t3.small"}, + setExcludeTypes: []string{"db.m5.large"}, + setIncludeEngines: []string{"mysql"}, + setExcludeEngines: []string{"postgres"}, + expectError: false, + }, + } + + for _, tt := range tests { + t.Run(tt.name, func(t *testing.T) { + // Set test values + toolCfg.Coverage = tt.setCoverage + toolCfg.TermYears = tt.setTerm + toolCfg.PaymentOption = tt.setPayment + toolCfg.MaxInstances = tt.setMaxInstances + toolCfg.CSVOutput = tt.setCSVOutput + toolCfg.CSVInput = tt.setCSVInput + toolCfg.IncludeEngines = tt.setIncludeEngines + toolCfg.ExcludeEngines = tt.setExcludeEngines + toolCfg.IncludeAccounts = tt.setIncludeAccounts + toolCfg.ExcludeAccounts = tt.setExcludeAccounts + toolCfg.IncludeInstanceTypes = tt.setIncludeTypes + toolCfg.ExcludeInstanceTypes = tt.setExcludeTypes + + // Call validateFlags + err := validateFlags(nil, []string{}) + + if tt.expectError { + assert.Error(t, err) + if err != nil && tt.errorContains != "" { + assert.Contains(t, err.Error(), tt.errorContains) + } + } else { + assert.NoError(t, err) + } + }) + } +} + +func TestSanitizeAccountName(t *testing.T) { + tests := []struct { + name string + input string + expected string + }{ + {"Simple name", "production", "production"}, + {"Spaces to hyphens", "my account", "my-account"}, + {"Underscores to hyphens", "my_account", "my-account"}, + {"Uppercase to lowercase", "PRODUCTION", "production"}, + {"Special chars removed", "my@account#123", "myaccount123"}, + {"Dots to hyphens", "my.account.com", "my-account-com"}, + {"Long name preserved", "very-long-production-environment-name", "very-long-production-environment-name"}, + {"Empty string", "", ""}, + {"Only special chars", "@#$%", ""}, + {"Multiple hyphens collapsed", "my---account", "my-account"}, + {"Leading/trailing hyphens removed", "-account-", "account"}, + } + + for _, tt := range tests { + t.Run(tt.name, func(t *testing.T) { + result := sanitizeAccountName(tt.input) + assert.Equal(t, tt.expected, result) + }) + } +} +func TestGeneratePurchaseID_EdgeCases(t *testing.T) { + tests := []struct { + name string + region string + rec common.Recommendation + index int + isDryRun bool + }{ + { + name: "RDS dry run", + rec: common.Recommendation{ + Service: common.ServiceRDS, + ResourceType: "db.t3.micro", + Count: 2, + }, + region: "us-east-1", + index: 1, + isDryRun: true, + }, + { + name: "EC2 actual purchase", + rec: common.Recommendation{ + Service: common.ServiceEC2, + ResourceType: "t3.large", + Count: 5, + }, + region: "eu-west-1", + index: 99, + isDryRun: false, + }, + { + name: "ElastiCache with dots in instance type", + rec: common.Recommendation{ + Service: common.ServiceElastiCache, + ResourceType: "cache.r6g.2xlarge", + Count: 1, + }, + region: "ap-southeast-1", + index: 1000, + isDryRun: false, + }, + { + name: "Unknown service", + rec: common.Recommendation{ + Service: common.ServiceType("future-service"), + ResourceType: "unknown.large", + Count: 10, + }, + region: "us-west-2", + index: 1, + isDryRun: true, + }, + } + + for _, tt := range tests { + t.Run(tt.name, func(t *testing.T) { + testCoverage := 80.0 + id := generatePurchaseID(tt.rec, tt.region, tt.index, tt.isDryRun, testCoverage) + + // Verify ID contains expected parts + if tt.isDryRun { + assert.Contains(t, id, "dryrun") + } else { + assert.Contains(t, id, "ri") + } + + assert.Contains(t, id, tt.region) + assert.Contains(t, id, strings.ReplaceAll(tt.rec.ResourceType, ".", "-")) + assert.Contains(t, id, fmt.Sprintf("%dx", tt.rec.Count)) + // Should contain timestamp (YYYYMMDD-HHMMSS) and UUID suffix (8 chars) + assert.Regexp(t, `-\d{8}-\d{6}-[a-f0-9]{8}$`, id) + }) + } +} + +func TestValidateInstanceTypes(t *testing.T) { + tests := []struct { + name string + errorContains string + instanceTypes []string + expectError bool + }{ + { + name: "Empty slice is valid", + instanceTypes: []string{}, + expectError: false, + }, + { + name: "Valid instance types", + instanceTypes: []string{"db.t3.micro", "cache.r5.large", "t3.medium"}, + expectError: false, + }, + { + name: "Invalid - no dot", + instanceTypes: []string{"t3micro"}, + expectError: true, + errorContains: "invalid instance type format", + }, + { + name: "Invalid - empty string", + instanceTypes: []string{"db.t3.micro", "", "cache.r5.large"}, + expectError: true, + errorContains: "empty instance type", + }, + { + name: "Valid with multiple dots", + instanceTypes: []string{"r5.xlarge.search", "db.r6g.2xlarge"}, + expectError: false, + }, + { + name: "Single valid instance type", + instanceTypes: []string{"m5.large"}, + expectError: false, + }, + { + name: "Mix of valid and invalid", + instanceTypes: []string{"db.t3.small", "invalidtype"}, + expectError: true, + errorContains: "invalid instance type format", + }, + } + + for _, tt := range tests { + t.Run(tt.name, func(t *testing.T) { + err := validateInstanceTypes(tt.instanceTypes) + if tt.expectError { + assert.Error(t, err) + if tt.errorContains != "" { + assert.Contains(t, err.Error(), tt.errorContains) + } + } else { + assert.NoError(t, err) + } + }) + } +} diff --git a/cmd/multi_service.go b/cmd/multi_service.go new file mode 100644 index 000000000..2dd0112dd --- /dev/null +++ b/cmd/multi_service.go @@ -0,0 +1,730 @@ +package main + +import ( + "context" + "fmt" + "log" + "os" + "os/signal" + "sync/atomic" + "time" + + "github.com/LeanerCloud/CUDly/internal/reporter" + "github.com/LeanerCloud/CUDly/pkg/common" + "github.com/LeanerCloud/CUDly/pkg/provider" + "github.com/LeanerCloud/CUDly/pkg/scorer" + awsprovider "github.com/LeanerCloud/CUDly/providers/aws" + "github.com/LeanerCloud/CUDly/providers/aws/recommendations" + "github.com/aws/aws-sdk-go-v2/aws" + awsconfig "github.com/aws/aws-sdk-go-v2/config" + "github.com/google/uuid" +) + +// fetchExistingCoverage retrieves the existing-RI coverage map from Cost +// Explorer so --target-coverage sizing can subtract what's already owned in +// each pool. Best-effort: a transient CE failure logs a warning and returns +// an empty map, which the sizing path treats as "no signal" — recs sized +// without subtracting existing commitments. Skipping the fetch entirely +// when --target-coverage is not in play avoids the per-region CE charges +// for users on the --coverage path. +// +// Coverage is fetched per-region per-account so CE's org-wide aggregate +// doesn't bleed one account's coverage into another in multi-account orgs. +// Regions come from cfg.Regions if set, otherwise from EC2 DescribeRegions. +// +// The lookback window is cfg.CoverageLookbackDays (default 30, matching the +// CE UI default). Operators reconciling against the AWS console coverage +// report should match this value to the report's own time window. +func fetchExistingCoverage(ctx context.Context, awsCfg aws.Config, recClient provider.RecommendationsClient, cfg Config) recommendations.PoolCoverageMap { + if cfg.TargetCoverage <= 0 { + return nil + } + adapter, ok := recClient.(*awsprovider.RecommendationsClientAdapter) + if !ok { + // Non-AWS provider: feature not wired up. Sizing degenerates to + // the no-existing-commitments path. + return nil + } + lookbackDays := cfg.CoverageLookbackDays + if lookbackDays <= 0 { + lookbackDays = 30 + } + regions := cfg.Regions + if len(regions) == 0 { + allRegions, err := getAllAWSRegions(ctx, awsCfg) + if err != nil { + AppLogger.Printf(" ⚠️ Could not list AWS regions for coverage fetch (%v); skipping existing-coverage subtraction\n", err) + return nil + } + regions = allRegions + } + AppLogger.Printf("\n🔎 Fetching existing-RI coverage from Cost Explorer per-account across %d regions (lookback %d days)...\n", len(regions), lookbackDays) + cov, err := adapter.GetRICoverageMap(ctx, lookbackDays, regions) + if err != nil { + AppLogger.Printf(" ⚠️ Could not fetch existing-RI coverage (%v); sizing will assume zero existing coverage\n", err) + return nil + } + AppLogger.Printf(" ✅ Fetched coverage for %d (region, instance-type, engine, account) entries\n", len(cov)) + return cov +} + +// shutdownRequested is set to true when SIGINT is received during a purchase run. +var shutdownRequested atomic.Bool + +// effectiveDryRun reports whether the run must stay in dry-run mode. A run is +// dry-run unless the user opts into real purchases with --purchase; that single +// flag is the only control. It defaults to false, so a bare invocation is a +// dry run and moving money is always an explicit opt-in. Both the non-CSV and +// CSV code paths use this helper so the guard is consistent and defined in one +// place. +func effectiveDryRun(cfg Config) bool { + return !cfg.ActualPurchase +} + +// runToolMultiService is the main entry point for processing multiple services. +// It runs a two-phase pipeline: (1) fetch+filter all recommendations, then +// (2) score, display, confirm, and purchase. +func runToolMultiService(ctx context.Context, cfg Config) { + if cfg.CSVInput != "" { + runCSVPathOrFatal(ctx, cfg) + return + } + + servicesToProcess := determineServicesToProcess(cfg) + if len(servicesToProcess) == 0 { + log.Fatalf("No valid services specified") + } + + isDryRun := effectiveDryRun(cfg) + + // Register SIGINT handler so a running purchase loop can be interrupted cleanly. + shutdownRequested.Store(false) + sigCh := make(chan os.Signal, 1) + signal.Notify(sigCh, os.Interrupt) + go func() { <-sigCh; shutdownRequested.Store(true) }() + defer signal.Stop(sigCh) + + // Verify the audit log and its immediate parents before making cloud API calls. + if err := CheckAuditLogWritable(cfg.AuditLog); err != nil { + log.Fatalf("Cannot prepare durable audit log: %v", err) //nolint:gocritic // exitAfterDefer: intentional startup fatal before cleanup matters + } + + printRunMode(isDryRun) + AppLogger.Printf("📊 Processing services: %s\n", formatServices(servicesToProcess)) + printPaymentAndTerm(cfg) + + awsCfg, err := loadAWSConfig(ctx, cfg) + if err != nil { + log.Fatalf("Failed to load AWS config: %v", err) + } + + accountCache := NewAccountAliasCache(awsCfg) + recClient := awsprovider.NewRecommendationsClient(awsCfg) + if adapter, ok := recClient.(*awsprovider.RecommendationsClientAdapter); ok && cfg.RecLookbackPeriod != "" { + adapter.SetRecLookbackPeriod(cfg.RecLookbackPeriod) + } + engineData := fetchEngineVersionData(ctx, cfg) + + // Fetch existing-RI coverage so --target-coverage can subtract what + // the user already owns. Best-effort: a failure here logs a warning + // and continues with an empty map, which makes sizing degenerate to + // the no-existing-commitments path (matches behavior when no recs + // are matched in the map). + coverageMap := fetchExistingCoverage(ctx, awsCfg, recClient, cfg) + + // Phase 1: collect all recommendations without purchasing. + AppLogger.Printf("\n📥 Fetching recommendations from all services...\n") + allRecs, drops := fetchAllRecs(ctx, awsCfg, recClient, accountCache, servicesToProcess, engineData, cfg, coverageMap) + + // Phase 2: score, enforce the run-wide instance cap, and display. + scoredResult := scoreLimitAndDisplay(allRecs, cfg, drops) + if len(scoredResult.Passed) == 0 { + printDropSummary(drops) + AppLogger.Printf("\nℹ️ No recommendations passed filters. Nothing to purchase.\n") + return + } + + // Phases 3-4: confirm, purchase, and produce summary outputs. + runPurchaseAndReport(ctx, awsCfg, scoredResult, isDryRun, cfg, drops) +} + +// runPurchaseAndReport handles the confirm, execute, and report phases of +// the multi-service pipeline. It is a separate function to keep +// runToolMultiService within the cyclomatic-complexity limit. +func runPurchaseAndReport(ctx context.Context, awsCfg aws.Config, scoredResult scorer.ScoredResult, isDryRun bool, cfg Config, drops *common.DropSummary) { + runID := uuid.New().String() + if !isDryRun { + totalInstances, totalSavings := sumPassedRecs(scoredResult.Passed) + if !ConfirmPurchase(totalInstances, totalSavings, cfg.SkipConfirmation) { + printDropSummary(drops) + AppLogger.Printf("\n❌ Purchase canceled.\n") + return + } + } + + allResults := executePurchasePipeline(ctx, awsCfg, scoredResult.Passed, isDryRun, runID, cfg) + + // Produce summary outputs. + writeReportAndSummary(scoredResult.Passed, allResults, isDryRun, cfg, drops) +} + +// writeReportAndSummary writes the CSV report and prints the final summary. +func writeReportAndSummary(passed []common.Recommendation, allResults []common.PurchaseResult, isDryRun bool, cfg Config, drops *common.DropSummary) { + serviceStats := buildServiceStats(passed, allResults) + finalCSVOutput := generateCSVFilename(isDryRun, cfg) + if err := writeMultiServiceCSVReport(allResults, finalCSVOutput); err != nil { + log.Printf("Warning: Failed to write CSV output: %v", err) + } else { + AppLogger.Printf("\n📋 CSV report written to: %s\n", finalCSVOutput) + } + printDropSummary(drops) + printMultiServiceSummary(passed, allResults, serviceStats, isDryRun) +} + +// printDropSummary writes the accumulated drop reasons at the terminal summary +// boundary. A nil or empty summary produces no output. +func printDropSummary(drops *common.DropSummary) { + if drops == nil { + return + } + if line := drops.FormatOneLine(); line != "" { + AppLogger.Printf("\n%s\n", line) + } +} + +// loadAWSConfig builds an aws.Config from the tool config. +func loadAWSConfig(ctx context.Context, cfg Config) (aws.Config, error) { + var opts []func(*awsconfig.LoadOptions) error + opts = append(opts, awsconfig.WithRegion("us-east-1")) + if cfg.Profile != "" { + opts = append(opts, awsconfig.WithSharedConfigProfile(cfg.Profile)) + } + return awsconfig.LoadDefaultConfig(ctx, opts...) +} + +// scoreLimitAndDisplay runs the scorer on recs, enforces the run-wide +// --max-instances cap on the survivors, and prints the scored table and +// summary. +// +// The cap runs between scoring and rendering so the table, the confirmation +// prompt and the purchase loop all describe the same post-cap set, and so the +// instances that survive are the highest-savings ones run-wide (scorer.Score +// sorts Passed by savings percentage descending). +func scoreLimitAndDisplay(recs []common.Recommendation, cfg Config, drops *common.DropSummary) scorer.ScoredResult { + scorerCfg := scorer.Config{ + MinSavingsPct: cfg.MinSavingsPct, + MaxBreakEvenMonths: cfg.MaxBreakEvenMonths, + MinCount: cfg.MinCount, + } + result := scorer.Score(recs, scorerCfg) + result.Passed = applyGlobalInstanceLimit(result.Passed, cfg, rankBySavingsPercentage, drops) + fmt.Print(reporter.RenderTable(result)) + fmt.Print(reporter.RenderExcluded(result)) + fmt.Print(reporter.RenderSummary(result)) + return result +} + +// rankingRule names the ordering --max-instances consumes, so the cap can say +// which rule decided what survived. The two paths rank on different keys +// because they carry different data, and an operator reading the drop list has +// to know which one applied. Typed rather than a bare string so the call sites +// cannot invent a third wording that does not match any implemented ordering. +type rankingRule string + +const ( + // rankBySavingsPercentage is the recommendation-driven path: scorer.Score + // sorts on SavingsPercentage, an intensive rate, before the cap runs. + rankBySavingsPercentage rankingRule = "highest savings-percentage recommendations first" + // rankBySavingsPerInstance is the --input-csv path: a CSV row carries no + // savings percentage, so sortBySavingsPerInstance derives the rate from + // the EstimatedSavings and Count columns it does carry. + rankBySavingsPerInstance rankingRule = "highest savings-per-instance rows first (a CSV row carries no savings percentage)" +) + +// capBinds reports whether --max-instances will actually remove instances from +// recs, rather than being unset or already satisfied. +// +// applyGlobalInstanceLimit and requireRankingSignal have to agree on this +// exactly: the second refuses precisely the runs whose outcome the first would +// otherwise decide on an ordering that carries no information. If the two +// conditions drift apart, either a run is refused for a cap that changes +// nothing, or an unrankable run reaches the cap after all. +func capBinds(recs []common.Recommendation, cfg Config) bool { + return cfg.MaxInstances > 0 && CalculateTotalInstances(recs) > int(cfg.MaxInstances) +} + +// applyGlobalInstanceLimit enforces --max-instances once across the entire run. +// +// The flag is documented as a hard cap on the total number of instances +// purchased across all recommendations, so it has to see every service and +// every region together. Applying it inside the per-region fetch instead caps +// each (service, region) pair independently and multiplies the operator's cap +// by the number of pairs. +// +// passed must already be ordered best-first by rule: ApplyInstanceLimit +// consumes the slice in order and drops the tail, so the ordering decides +// which commitments survive. scorer.Score guarantees the +// rankBySavingsPercentage ordering; scoreAndLimitCSVRecs establishes the +// rankBySavingsPerInstance one. +// +// Truncation can push a recommendation under --min-count, which is a hard +// floor rather than advice, so dropTruncatedBelowMinCount removes any such +// recommendation instead of purchasing it short. +// +// Nothing is truncated silently. Every reduced or dropped recommendation is +// named on stdout, and the drops are counted into the end-of-run summary. +func applyGlobalInstanceLimit(passed []common.Recommendation, cfg Config, rule rankingRule, drops *common.DropSummary) []common.Recommendation { + if !capBinds(passed, cfg) { + return passed + } + + totalBefore := CalculateTotalInstances(passed) + limited := ApplyInstanceLimit(passed, cfg.MaxInstances) + limited, belowMin := dropTruncatedBelowMinCount(limited, cfg.MinCount) + reportInstanceLimit(passed, limited, len(belowMin), totalBefore, cfg.MaxInstances, rule, drops) + reportMinCountDrops(belowMin, cfg.MinCount, drops) + return limited +} + +// dropTruncatedBelowMinCount removes recommendations that the cap truncated to +// fewer instances than --min-count allows, returning the survivors and the +// removed recommendations at their truncated counts. +// +// --min-count is a hard floor everywhere else in the codebase, never advice: +// the scorer rejects recommendations under it outright +// (scorer.filterReason, "count %d below minimum %d"), the scheduler's +// meetsMinCount drops them, and both `docs/cli/filtering.md` and +// `docs/cli/README.md` describe it as dropping recommendations below the +// number. filtering.md applies it to "the adjusted instance count (after +// coverage scaling)", so the floor is meant to gate the *sized* count, and +// truncation by --max-instances is another form of sizing. +// +// Buying a commitment smaller than the operator's stated minimum can be worse +// than buying nothing, which is the whole reason the floor exists, so a +// truncated recommendation is dropped rather than purchased short. The freed +// budget is deliberately not redistributed: the next recommendation would have +// to fit in an even smaller remainder and would fail the same floor. +// +// Removal is always from the tail. ApplyInstanceLimit reduces at most one +// recommendation (the one where the budget runs out, which is the last it +// keeps), and every earlier one still carries the full count that already +// cleared the scorer's floor. Taking only from the tail keeps the result a +// prefix of the input, which reportInstanceLimit relies on. +func dropTruncatedBelowMinCount(limited []common.Recommendation, minCount int) (kept, removed []common.Recommendation) { + if minCount <= 0 { + return limited, nil + } + kept = limited + for len(kept) > 0 && kept[len(kept)-1].Count < minCount { + removed = append(removed, kept[len(kept)-1]) + kept = kept[:len(kept)-1] + } + return kept, removed +} + +// reportMinCountDrops names each recommendation the --min-count floor rejected +// after --max-instances truncated it. It continues the reportInstanceLimit +// listing, where these already appear as dropped, and explains why they were +// not simply purchased at the reduced count. +func reportMinCountDrops(removed []common.Recommendation, minCount int, drops *common.DropSummary) { + for i := range removed { + rec := removed[i] + AppLogger.Printf(" ↳ %s %s %s: the cap left room for only %d instances, below --min-count %d, so it is dropped rather than purchased short\n", + rec.Service, rec.Region, rec.ResourceType, rec.Count, minCount) + } + drops.Add(common.DropMinCountAfterCap, len(removed)) +} + +// reportInstanceLimit prints what --max-instances removed from the run. +// after must be the prefix of before produced by ApplyInstanceLimit, so +// after[i] and before[i] describe the same recommendation. +// +// rule is the ordering the caller sorted before by, and is printed verbatim: +// the drop list is unreadable without knowing which key decided it, and the +// two paths do not rank on the same key. +// +// belowMinCount is how many of the missing entries were removed by the +// --min-count floor rather than by the budget. They are still listed here as +// dropped (they were), but they are attributed to --min-count-after-cap by +// reportMinCountDrops, so excluding them from this tally keeps each dropped +// recommendation counted exactly once in the end-of-run summary. +func reportInstanceLimit(before, after []common.Recommendation, belowMinCount, totalBefore int, maxInstances int32, rule rankingRule, drops *common.DropSummary) { + AppLogger.Printf("\n🔒 --max-instances=%d caps the whole run: the %d recommendations that passed scoring total %d instances.\n", + maxInstances, len(before), totalBefore) + AppLogger.Printf(" Keeping the %s. The following are reduced or dropped:\n", rule) + + reduced, dropped := 0, 0 + for i := range before { + rec := before[i] + kept := 0 + if i < len(after) { + kept = after[i].Count + } + switch { + case kept == rec.Count: + continue + case kept > 0: + reduced++ + AppLogger.Printf(" • reduced: %s %s %s %d → %d instances\n", rec.Service, rec.Region, rec.ResourceType, rec.Count, kept) + default: + dropped++ + AppLogger.Printf(" • dropped: %s %s %s (%d instances)\n", rec.Service, rec.Region, rec.ResourceType, rec.Count) + } + } + + drops.Add(common.DropMaxInstances, dropped-belowMinCount) + AppLogger.Printf(" Proceeding with %d instances across %d recommendations (%d reduced, %d dropped).\n", + CalculateTotalInstances(after), len(after), reduced, dropped) +} + +// sumPassedRecs returns total instance count and total estimated savings for passed recs. +func sumPassedRecs(recs []common.Recommendation) (total int, totalSavings float64) { + for _rvc := range recs { + r := recs[_rvc] + total += r.Count + totalSavings += r.EstimatedSavings + } + return +} + +// executePurchasePipeline purchases each rec in the passed list (or dry-runs) and writes audit records. +func executePurchasePipeline(ctx context.Context, awsCfg aws.Config, recs []common.Recommendation, isDryRun bool, runID string, cfg Config) []common.PurchaseResult { + results := make([]common.PurchaseResult, 0, len(recs)) + for i := range recs { + rec := recs[i] + if shutdownRequested.Load() { + log.Printf("Shutdown requested — skipping %d remaining recommendations", len(recs)-i) + break + } + result, status := purchaseSingleRec(ctx, awsCfg, rec, i+1, isDryRun, cfg) + results = append(results, result) + auditRec := common.NewAuditRecord(runID, rec, result, status, isDryRun, common.PurchaseSourceCLI) + if err := common.WriteAuditRecord(auditRec, cfg.AuditLog); err != nil { + log.Printf("Warning: failed to write audit record: %v", err) + } + if !isDryRun && i < len(recs)-1 && os.Getenv("DISABLE_PURCHASE_DELAY") != "true" { + time.Sleep(PurchaseDelaySeconds * time.Second) + } + } + return results +} + +// purchaseSingleRec executes or dry-runs a single purchase and returns the result + audit status. +func purchaseSingleRec(ctx context.Context, awsCfg aws.Config, rec common.Recommendation, index int, isDryRun bool, cfg Config) (purchaseResult common.PurchaseResult, auditStatus string) { + AppLogger.Printf(" [%d] %s %s %s (count=%d)\n", index, rec.Service, rec.Region, rec.ResourceType, rec.Count) + if isDryRun { + result := createDryRunResult(rec, rec.Region, index, cfg) + AppLogger.Printf(" [dry-run] %s\n", result.CommitmentID) + return result, "skipped" + } + + regionalCfg := awsCfg.Copy() + regionalCfg.Region = rec.Region + serviceClient := createServiceClient(rec.Service, regionalCfg) + if serviceClient == nil { + AppLogger.Printf(" ⚠️ No service client for %s\n", rec.Service) + return common.PurchaseResult{Success: false}, "error" + } + + result := executePurchase(ctx, rec, rec.Region, index, serviceClient, cfg) + status := "success" + if !result.Success { + status = "error" + AppLogger.Printf(" ❌ %v\n", result.Error) + } else { + AppLogger.Printf(" ✅ %s\n", result.CommitmentID) + } + return result, status +} + +// buildServiceStats computes per-service statistics from a purchase run. +// Results are assumed to be in the same order as recs (1:1 correspondence). +func buildServiceStats(recs []common.Recommendation, results []common.PurchaseResult) map[common.ServiceType]ServiceProcessingStats { + byService := make(map[common.ServiceType][]common.Recommendation) + resultsByService := make(map[common.ServiceType][]common.PurchaseResult) + for i := range recs { + rec := recs[i] + byService[rec.Service] = append(byService[rec.Service], rec) + if i < len(results) { + resultsByService[rec.Service] = append(resultsByService[rec.Service], results[i]) + } + } + stats := make(map[common.ServiceType]ServiceProcessingStats) + for service, serviceRecs := range byService { + stats[service] = calculateServiceStats(service, serviceRecs, resultsByService[service]) + } + return stats +} + +// runCSVPathOrFatal runs the CSV purchase path and exits fatally on error. +// It isolates the error-to-fatal glue from runToolMultiService so that path +// stays under the cyclomatic-complexity budget while runToolFromCSV remains +// unit-testable via its returned error. +func runCSVPathOrFatal(ctx context.Context, cfg Config) { + if err := runToolFromCSV(ctx, cfg); err != nil { + log.Fatalf("%v", err) + } +} + +// runToolFromCSV processes recommendations from a CSV input file. +// It returns an error instead of exiting so the orchestration glue is +// unit-testable; the caller (runCSVPathOrFatal) turns errors fatal. +func runToolFromCSV(ctx context.Context, cfg Config) error { + isDryRun := effectiveDryRun(cfg) + printRunMode(isDryRun) + + csvModeCoverage := determineCSVCoverage(cfg) + + AppLogger.Printf("📄 Reading recommendations from CSV: %s\n", cfg.CSVInput) + + // Read recommendations from CSV + recs, err := loadRecommendationsFromCSV(cfg.CSVInput) + if err != nil { + return fmt.Errorf("failed to read CSV file: %w", err) + } + + AppLogger.Printf("✅ Loaded %d recommendations from CSV\n", len(recs)) + + // Filter and adjust recommendations + recs, err = filterAndAdjustRecommendations(recs, csvModeCoverage, cfg) + if err != nil { + return err + } + + if len(recs) == 0 { + AppLogger.Println("⚠️ No recommendations to process after filtering") + return nil + } + + awsCfg, err := loadAWSConfig(ctx, cfg) + if err != nil { + return fmt.Errorf("failed to load AWS config: %w", err) + } + + // Create account alias cache for lookup + accountCache := NewAccountAliasCache(awsCfg) + + // Populate account names from account IDs + populateAccountNames(ctx, recs, accountCache) + + // Group recommendations by service and region + recsByServiceRegion := groupRecommendationsByServiceRegion(recs) + + // Process purchases + allResults := make([]common.PurchaseResult, 0) + serviceResults := make([]common.PurchaseResult, 0) + serviceStats := make(map[common.ServiceType]ServiceProcessingStats) + // allAdjustedRecs accumulates post-dedup recommendations so the final summary + // reflects what was actually processed rather than the pre-dedup input slice. + allAdjustedRecs := make([]common.Recommendation, 0) + + for service, regionRecs := range recsByServiceRegion { + // Reset service results for each service + serviceResults = serviceResults[:0] + + AppLogger.Printf("\n━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━\n") + AppLogger.Printf("🎯 Processing %s\n", getServiceDisplayName(service)) + AppLogger.Printf("━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━\n") + + serviceRecs := make([]common.Recommendation, 0) + for region, recs := range regionRecs { + AppLogger.Printf("\n 📍 Region: %s (%d recommendations)\n", region, len(recs)) + + // Get service client for this region + regionalCfg := awsCfg.Copy() + regionalCfg.Region = region + serviceClient := createServiceClient(service, regionalCfg) + + if serviceClient == nil { + AppLogger.Printf(" ⚠️ Service client not yet implemented for %s\n", getServiceDisplayName(service)) + AppLogger.Printf(" (Skipping purchase phase for this service)\n") + continue + } + + // Check for duplicate RIs to avoid double purchasing + adjustedRecs, err := adjustRecsForDuplicates(ctx, recs, serviceClient) + if err != nil { + AppLogger.Printf(" ⚠️ Warning: Could not check for existing RIs: %v\n", err) + adjustedRecs = recs // Continue with original recommendations if check fails + } + // Deducting existing commitments shrinks Count, which can push a + // row that cleared the floor in filterAndAdjustRecommendations back + // under it (--min-count 5, a row of 6, and 5 matching recent + // commitments would otherwise be purchased at 1). --min-count is a + // floor on what gets bought, so it is re-applied to whatever the + // deduction left, not only to the pre-deduction counts. + recs = applyMinCountFloor(adjustedRecs, cfg.MinCount) + + serviceRecs = append(serviceRecs, recs...) + allAdjustedRecs = append(allAdjustedRecs, recs...) + + // Process purchases for this region + regionResults := processPurchaseLoop(ctx, recs, region, isDryRun, serviceClient, cfg) + serviceResults = append(serviceResults, regionResults...) + } + + // Add service results to overall results + allResults = append(allResults, serviceResults...) + + // Calculate service statistics (using only this service's results) + stats := calculateServiceStats(service, serviceRecs, serviceResults) + serviceStats[service] = stats + printServiceSummary(service, stats) + } + + // Generate CSV filename and write report + finalCSVOutput := generateCSVFilename(isDryRun, cfg) + + // Write CSV report + if err := writeMultiServiceCSVReport(allResults, finalCSVOutput); err != nil { + log.Printf("Warning: Failed to write CSV output: %v", err) + } else { + AppLogger.Printf("\n📋 CSV report written to: %s\n", finalCSVOutput) + } + + // Print final summary using the post-dedup slice so counts match what was + // actually processed, not the pre-dedup input passed into the outer loop. + printMultiServiceSummary(allAdjustedRecs, allResults, serviceStats, isDryRun) + return nil +} + +// filterAndAdjustRecommendations applies filters, coverage, count override, +// the --min-count floor and the run-wide --max-instances cap to +// recommendations loaded from --input-csv. +// +// It returns an error when the cap cannot be enforced honestly; see +// requireRankingSignal. +func filterAndAdjustRecommendations(recs []common.Recommendation, csvModeCoverage float64, cfg Config) ([]common.Recommendation, error) { + // Query running instances for engine version validation + log.Printf("🔍 Querying running RDS instances across all regions to validate engine versions...") + instanceVersions, err := queryRunningInstanceEngineVersions(context.Background(), cfg) + if err != nil { + log.Printf("⚠️ Warning: Failed to query running instances for engine version validation: %v", err) + log.Printf(" Continuing without engine version filtering") + instanceVersions = make(map[string][]InstanceEngineVersion) + } else { + log.Printf("✅ Found %d instance types with version information across all regions", len(instanceVersions)) + } + + // Query major engine versions for extended support detection + log.Printf("🔍 Querying AWS RDS major engine versions for extended support information...") + versionInfo, err := queryMajorEngineVersions(context.Background(), cfg) + if err != nil { + log.Printf("⚠️ Warning: Failed to query major engine versions: %v", err) + log.Printf(" Continuing without extended support detection") + versionInfo = make(map[string]MajorEngineVersionInfo) + } else { + log.Printf("✅ Found support information for %d major engine versions", len(versionInfo)) + } + + // Apply filters (empty currentRegion since we're processing from CSV, not iterating regions). + // Drop tracking is skipped on the CSV path (nil drops). + originalCount := len(recs) + recs = applyFilters(recs, &cfg, instanceVersions, versionInfo, "", nil) + if len(recs) < originalCount { + AppLogger.Printf("🔍 After filters: %d recs (filtered out %d)\n", len(recs), originalCount-len(recs)) + } + + // Apply sizing — target-coverage if set, otherwise coverage. + // Coverage 100% is a no-op (early-returned inside ApplyCoverage), but + // --target-coverage always applies even at coverage 100%, so the + // CSV-path short-circuit is conditional on TargetCoverage == 0. + if cfg.TargetCoverage > 0 || csvModeCoverage < 100 { + beforeSize := len(recs) + recs = applySizing(recs, cfg, csvModeCoverage, nil) + if cfg.TargetCoverage > 0 { + AppLogger.Printf("🎯 Applying %.1f%% target-coverage: %d recs selected (from %d)\n", cfg.TargetCoverage, len(recs), beforeSize) + } else { + AppLogger.Printf("📈 Applying %.1f%% coverage: %d recs selected (from %d)\n", csvModeCoverage, len(recs), beforeSize) + } + } + + // Apply count override if specified + if cfg.OverrideCount > 0 { + recs = ApplyCountOverride(recs, cfg.OverrideCount) + } + + // Enforce --min-count and the run-wide --max-instances cap. + return scoreAndLimitCSVRecs(recs, cfg) +} + +// processService processes a single service and returns recommendations and results. +// Used by legacy callers; new code should use fetchAllRecs + executePurchasePipeline. +func processService(ctx context.Context, awsCfg aws.Config, recClient provider.RecommendationsClient, accountCache *AccountAliasCache, service common.ServiceType, isDryRun bool, cfg Config, engineData engineVersionData) ([]common.Recommendation, []common.PurchaseResult) { //nolint:unparam // engineData always nil at current callsites but param is part of the API + regionsToProcess, err := determineRegionsForService(ctx, awsCfg, recClient, service, cfg.Regions) + if err != nil { + log.Printf("❌ Failed to determine regions: %v", err) + return nil, nil + } + + serviceRecs := make([]common.Recommendation, 0) + serviceResults := make([]common.PurchaseResult, 0) + + for i, region := range regionsToProcess { + // Legacy single-service entry point — no coverage map is fetched here, + // so sizing falls back to the no-existing-commitments formula. The new + // path (runToolMultiService) fetches coverage once and threads it through. + regionResult := processRegionRecommendations( + ctx, awsCfg, recClient, accountCache, + service, region, i+1, len(regionsToProcess), + engineData, isDryRun, cfg, nil, + ) + serviceRecs = append(serviceRecs, regionResult.recommendations...) + serviceResults = append(serviceResults, regionResult.results...) + } + + return serviceRecs, serviceResults +} + +// processPurchaseLoop processes purchases for a single region (used by CSV mode). +func processPurchaseLoop(ctx context.Context, recs []common.Recommendation, region string, isDryRun bool, serviceClient provider.ServiceClient, cfg Config) []common.PurchaseResult { + results := make([]common.PurchaseResult, 0, len(recs)) + + for j := range recs { + rec := recs[j] + AppLogger.Printf(" [%d/%d] Processing: %s %s\n", j+1, len(recs), rec.Service, rec.ResourceType) + AppLogger.Printf(" 💳 Purchasing %d instances\n", rec.Count) + + var result common.PurchaseResult + if isDryRun { + result = createDryRunResult(rec, region, j+1, cfg) + } else { + // Ask for confirmation before proceeding with purchases (only on first item) + if j == 0 { + totalInstances := CalculateTotalInstances(recs) + totalSavings := 0.0 + for _rvc := range recs { + r := recs[_rvc] + totalSavings += r.EstimatedSavings + } + + if !ConfirmPurchase(totalInstances, totalSavings, cfg.SkipConfirmation) { + // User canceled - return canceled results for all + return createCancelledResults(recs, region, cfg) + } + } + + // Execute actual purchase + result = executePurchase(ctx, rec, region, j+1, serviceClient, cfg) + + // Add delay between purchases to avoid rate limiting + if j < len(recs)-1 && os.Getenv("DISABLE_PURCHASE_DELAY") != "true" { + time.Sleep(PurchaseDelaySeconds * time.Second) + } + } + + results = append(results, result) + + if result.Success { + AppLogger.Printf(" ✅ Success: %s\n", result.CommitmentID) + } else { + errMsg := "unknown error" + if result.Error != nil { + errMsg = result.Error.Error() + } + AppLogger.Printf(" ❌ Failed: %s\n", errMsg) + } + } + + return results +} diff --git a/cmd/multi_service_coverage_test.go b/cmd/multi_service_coverage_test.go new file mode 100644 index 000000000..ad82791f8 --- /dev/null +++ b/cmd/multi_service_coverage_test.go @@ -0,0 +1,842 @@ +package main + +import ( + "context" + "errors" + "os" + "testing" + "time" + + "github.com/LeanerCloud/CUDly/pkg/common" + "github.com/aws/aws-sdk-go-v2/aws" + awsrds "github.com/aws/aws-sdk-go-v2/service/rds" + rdstypes "github.com/aws/aws-sdk-go-v2/service/rds/types" + "github.com/stretchr/testify/assert" + "github.com/stretchr/testify/mock" + "github.com/stretchr/testify/require" +) + +// ==================== Tests for coverage improvement ==================== +// This file contains additional tests to improve coverage for functions +// identified as having low coverage + +// ==================== Tests for queryMajorEngineVersions ==================== + +// engineKeyedRDSMajorVersionsStub implements RDSMajorVersionsClient with +// canned per-engine responses and records which engines were queried, so +// queryMajorEngineVersionsWithClient can be tested hermetically (no AWS +// credentials, no network). +type engineKeyedRDSMajorVersionsStub struct { + versionsByEngine map[string][]rdstypes.DBMajorEngineVersion + errByEngine map[string]error + enginesQueried []string +} + +func (s *engineKeyedRDSMajorVersionsStub) DescribeDBMajorEngineVersions( + _ context.Context, + params *awsrds.DescribeDBMajorEngineVersionsInput, + _ ...func(*awsrds.Options), +) (*awsrds.DescribeDBMajorEngineVersionsOutput, error) { + engine := aws.ToString(params.Engine) + s.enginesQueried = append(s.enginesQueried, engine) + if err := s.errByEngine[engine]; err != nil { + return nil, err + } + return &awsrds.DescribeDBMajorEngineVersionsOutput{ + DBMajorEngineVersions: s.versionsByEngine[engine], + }, nil +} + +// TestQueryMajorEngineVersionsWithClient_Success replaces a former +// live-credential test: it stubs the RDS client and asserts the merged +// engine:major-version map deterministically. +func TestQueryMajorEngineVersionsWithClient_Success(t *testing.T) { + stub := &engineKeyedRDSMajorVersionsStub{ + versionsByEngine: map[string][]rdstypes.DBMajorEngineVersion{ + "mysql": {rdsMajorVersion("mysql", "5.7"), rdsMajorVersion("mysql", "8.0")}, + "postgres": {rdsMajorVersion("postgres", "13")}, + }, + } + + result, err := queryMajorEngineVersionsWithClient(context.Background(), stub) + require.NoError(t, err) + + assert.ElementsMatch(t, + []string{"mysql", "postgres", "aurora-mysql", "aurora-postgresql"}, + stub.enginesQueried, + "must query every supported engine exactly once") + + require.Len(t, result, 3) + assert.Equal(t, "mysql", result["mysql:5.7"].Engine) + assert.Equal(t, "5.7", result["mysql:5.7"].MajorEngineVersion) + assert.Equal(t, "8.0", result["mysql:8.0"].MajorEngineVersion) + assert.Equal(t, "13", result["postgres:13"].MajorEngineVersion) + require.Len(t, result["postgres:13"].SupportedEngineLifecycles, 1, + "lifecycle data must be carried through from the API response") +} + +// TestQueryMajorEngineVersionsWithClient_EngineErrorContinues asserts the +// warn-and-continue contract: one engine failing must not drop the results +// of the others, and the overall call still succeeds. +func TestQueryMajorEngineVersionsWithClient_EngineErrorContinues(t *testing.T) { + stub := &engineKeyedRDSMajorVersionsStub{ + versionsByEngine: map[string][]rdstypes.DBMajorEngineVersion{ + "postgres": {rdsMajorVersion("postgres", "15")}, + }, + errByEngine: map[string]error{ + "mysql": errors.New("DescribeDBMajorEngineVersions throttled"), + }, + } + + result, err := queryMajorEngineVersionsWithClient(context.Background(), stub) + require.NoError(t, err, "per-engine API failures are warn-and-continue") + + assert.ElementsMatch(t, + []string{"mysql", "postgres", "aurora-mysql", "aurora-postgresql"}, + stub.enginesQueried, + "a failing engine must not stop the remaining engine queries") + require.Len(t, result, 1) + assert.Equal(t, "15", result["postgres:15"].MajorEngineVersion) +} + +// TestQueryMajorEngineVersions_ProfileSelection asserts validation-profile +// precedence deterministically: the configured profiles point at names +// guaranteed not to exist in the (redirected, empty) shared config, so +// config loading fails fast without any live AWS call and the error names +// the profile the function actually selected. +func TestQueryMajorEngineVersions_ProfileSelection(t *testing.T) { + ctx := context.Background() + + tests := []struct { + name string + profile string + validationProfile string + wantProfileInErr string + }{ + { + name: "validation profile takes precedence", + profile: "cudly-test-main-profile-does-not-exist", + validationProfile: "cudly-test-validation-profile-does-not-exist", + wantProfileInErr: "cudly-test-validation-profile-does-not-exist", + }, + { + name: "falls back to main profile", + profile: "cudly-test-main-profile-does-not-exist", + wantProfileInErr: "cudly-test-main-profile-does-not-exist", + }, + } + + for _, tt := range tests { + t.Run(tt.name, func(t *testing.T) { + t.Setenv("AWS_CONFIG_FILE", os.DevNull) + t.Setenv("AWS_SHARED_CREDENTIALS_FILE", os.DevNull) + + _, err := queryMajorEngineVersions(ctx, Config{ + Profile: tt.profile, + ValidationProfile: tt.validationProfile, + }) + + require.Error(t, err) + assert.Contains(t, err.Error(), "failed to load AWS config") + assert.Contains(t, err.Error(), tt.wantProfileInErr, + "error must name the selected profile") + }) + } +} + +// ==================== Tests for extractMajorVersion ==================== + +func TestExtractMajorVersion_ComprehensiveTests(t *testing.T) { + tests := []struct { + name string + engine string + fullVersion string + expectedMajor string + }{ + // Aurora MySQL special handling + { + name: "Aurora MySQL 2.x format", + engine: "aurora-mysql", + fullVersion: "mysql_aurora.2.11.3", + expectedMajor: "5.7", + }, + { + name: "Aurora MySQL 3.x format", + engine: "aurora-mysql", + fullVersion: "mysql_aurora.3.04.0", + expectedMajor: "8.0", + }, + { + name: "Aurora MySQL direct 5.7", + engine: "aurora-mysql", + fullVersion: "5.7.mysql_aurora.2.11.1", + expectedMajor: "5.7", + }, + { + name: "Aurora MySQL direct 8.0", + engine: "aurora-mysql", + fullVersion: "8.0.mysql_aurora.3.04.0", + expectedMajor: "8.0", + }, + + // Standard MySQL/PostgreSQL + { + name: "MySQL 5.7", + engine: "mysql", + fullVersion: "5.7.44", + expectedMajor: "5.7", + }, + { + name: "MySQL 8.0", + engine: "mysql", + fullVersion: "8.0.35", + expectedMajor: "8.0", + }, + { + name: "PostgreSQL 13", + engine: "postgres", + fullVersion: "13.12", + expectedMajor: "13.12", + }, + { + name: "PostgreSQL 15", + engine: "postgres", + fullVersion: "15.4", + expectedMajor: "15.4", + }, + + // Edge cases + { + name: "Empty version", + engine: "mysql", + fullVersion: "", + expectedMajor: "", + }, + { + name: "Single digit version", + engine: "postgres", + fullVersion: "14", + expectedMajor: "14", + }, + { + name: "Version with patch suffix", + engine: "mysql", + fullVersion: "8.0.35-rds.1", + expectedMajor: "8.0", + }, + } + + for _, tt := range tests { + t.Run(tt.name, func(t *testing.T) { + result := extractMajorVersion(tt.engine, tt.fullVersion) + assert.Equal(t, tt.expectedMajor, result) + }) + } +} + +// ==================== Tests for isInExtendedSupport ==================== + +func TestIsInExtendedSupport_EdgeCases(t *testing.T) { + now := time.Now() + pastDate := now.AddDate(0, -6, 0) // 6 months ago + futureDate := now.AddDate(3, 0, 0) // 3 years from now + + versionInfo := map[string]MajorEngineVersionInfo{ + "mysql:5.7": { + Engine: "mysql", + MajorEngineVersion: "5.7", + SupportedEngineLifecycles: []EngineLifecycleInfo{ + { + LifecycleSupportName: string(rdstypes.LifecycleSupportNameOpenSourceRdsExtendedSupport), + LifecycleSupportStartDate: pastDate, + LifecycleSupportEndDate: futureDate, + }, + }, + }, + "mysql:8.0": { + Engine: "mysql", + MajorEngineVersion: "8.0", + SupportedEngineLifecycles: []EngineLifecycleInfo{ + { + LifecycleSupportName: string(rdstypes.LifecycleSupportNameOpenSourceRdsStandardSupport), + LifecycleSupportStartDate: now.AddDate(-2, 0, 0), + LifecycleSupportEndDate: futureDate, + }, + }, + }, + } + + tests := []struct { + name string + engine string + fullVersion string + expectedExtended bool + }{ + { + name: "MySQL 5.7 in extended support", + engine: "mysql", + fullVersion: "5.7.44", + expectedExtended: true, + }, + { + name: "MySQL 8.0 in standard support", + engine: "mysql", + fullVersion: "8.0.35", + expectedExtended: false, + }, + { + name: "Unknown version", + engine: "mysql", + fullVersion: "9.0.0", + expectedExtended: false, + }, + { + name: "Empty version", + engine: "mysql", + fullVersion: "", + expectedExtended: false, + }, + } + + for _, tt := range tests { + t.Run(tt.name, func(t *testing.T) { + result := isInExtendedSupport(tt.engine, tt.fullVersion, versionInfo) + assert.Equal(t, tt.expectedExtended, result) + }) + } +} + +// ==================== Tests for adjustRecommendationForExcludedVersions ==================== + +func TestAdjustRecommendationForExcludedVersions_AdditionalCases(t *testing.T) { + now := time.Now() + pastDate := now.AddDate(0, -6, 0) + futureDate := now.AddDate(3, 0, 0) + + versionInfo := map[string]MajorEngineVersionInfo{ + "mysql:5.7": { + Engine: "mysql", + MajorEngineVersion: "5.7", + SupportedEngineLifecycles: []EngineLifecycleInfo{ + { + LifecycleSupportName: string(rdstypes.LifecycleSupportNameOpenSourceRdsExtendedSupport), + LifecycleSupportStartDate: pastDate, + LifecycleSupportEndDate: futureDate, + }, + }, + }, + } + + tests := []struct { + instanceVersions map[string][]InstanceEngineVersion + name string + rec common.Recommendation + expectedCount int + }{ + { + name: "No running instances - no adjustment", + rec: common.Recommendation{ + ResourceType: "db.t3.small", + Count: 10, + Region: "us-east-1", + Details: &common.DatabaseDetails{ + Engine: "mysql", + }, + }, + instanceVersions: map[string][]InstanceEngineVersion{}, + expectedCount: 10, + }, + { + name: "Running instances with extended support - adjust count", + rec: common.Recommendation{ + ResourceType: "db.t3.small", + Count: 10, + Region: "us-east-1", + Details: &common.DatabaseDetails{ + Engine: "mysql", + }, + }, + instanceVersions: map[string][]InstanceEngineVersion{ + "db.t3.small": { + { + Engine: "mysql", + EngineVersion: "5.7.44", + InstanceClass: "db.t3.small", + Region: "us-east-1", + }, + { + Engine: "mysql", + EngineVersion: "5.7.42", + InstanceClass: "db.t3.small", + Region: "us-east-1", + }, + }, + }, + expectedCount: 8, // 10 - 2 extended support instances + }, + { + name: "Running instances in different region - no adjustment", + rec: common.Recommendation{ + ResourceType: "db.t3.small", + Count: 10, + Region: "us-east-1", + Details: &common.DatabaseDetails{ + Engine: "mysql", + }, + }, + instanceVersions: map[string][]InstanceEngineVersion{ + "db.t3.small": { + { + Engine: "mysql", + EngineVersion: "5.7.44", + InstanceClass: "db.t3.small", + Region: "us-west-2", // Different region + }, + }, + }, + expectedCount: 10, // No adjustment + }, + { + name: "Non-RDS recommendation - no adjustment", + rec: common.Recommendation{ + ResourceType: "t3.small", + Count: 5, + Region: "us-east-1", + Details: common.ComputeDetails{ + Platform: "Linux", + }, + }, + instanceVersions: map[string][]InstanceEngineVersion{}, + expectedCount: 5, + }, + } + + for _, tt := range tests { + t.Run(tt.name, func(t *testing.T) { + result := adjustRecommendationForExcludedVersions(tt.rec, tt.instanceVersions, versionInfo) + assert.Equal(t, tt.expectedCount, result.Count) + }) + } +} + +// ==================== Tests for validateFlags ==================== + +func TestValidateFlags_Coverage(t *testing.T) { + origCfg := toolCfg + defer func() { toolCfg = origCfg }() + + tests := []struct { + setupCfg func() + name string + errorMsg string + expectError bool + }{ + { + name: "Valid configuration", + setupCfg: func() { + toolCfg.Coverage = 80.0 + toolCfg.PaymentOption = "partial-upfront" + toolCfg.TermYears = 3 + toolCfg.MaxInstances = 100 + toolCfg.OverrideCount = 5 + }, + expectError: false, + }, + { + name: "Coverage below 0", + setupCfg: func() { + toolCfg.Coverage = -10.0 + toolCfg.PaymentOption = "all-upfront" + toolCfg.TermYears = 1 + }, + expectError: true, + errorMsg: "coverage percentage must be between 0 and 100", + }, + { + name: "Coverage above 100", + setupCfg: func() { + toolCfg.Coverage = 150.0 + toolCfg.PaymentOption = "all-upfront" + toolCfg.TermYears = 1 + }, + expectError: true, + errorMsg: "coverage percentage must be between 0 and 100", + }, + { + name: "Invalid payment option", + setupCfg: func() { + toolCfg.Coverage = 80.0 + toolCfg.PaymentOption = "invalid-option" + toolCfg.TermYears = 3 + }, + expectError: true, + errorMsg: "invalid payment option", + }, + { + name: "Invalid term years", + setupCfg: func() { + toolCfg.Coverage = 80.0 + toolCfg.PaymentOption = "partial-upfront" + toolCfg.TermYears = 2 // Only 1 or 3 allowed + }, + expectError: true, + errorMsg: "invalid term", + }, + { + name: "Negative max instances", + setupCfg: func() { + toolCfg.Coverage = 80.0 + toolCfg.PaymentOption = "all-upfront" + toolCfg.TermYears = 3 + toolCfg.MaxInstances = -5 + }, + expectError: true, + errorMsg: "max-instances must be 0", + }, + { + name: "Max instances exceeds limit", + setupCfg: func() { + toolCfg.Coverage = 80.0 + toolCfg.PaymentOption = "all-upfront" + toolCfg.TermYears = 3 + toolCfg.MaxInstances = MaxReasonableInstances + 1 + }, + expectError: true, + errorMsg: "exceeds reasonable limit", + }, + { + name: "Negative override count", + setupCfg: func() { + toolCfg.Coverage = 80.0 + toolCfg.PaymentOption = "all-upfront" + toolCfg.TermYears = 3 + toolCfg.MaxInstances = 0 + toolCfg.OverrideCount = -3 + }, + expectError: true, + errorMsg: "override-count must be 0", + }, + { + name: "Override count exceeds limit", + setupCfg: func() { + toolCfg.Coverage = 80.0 + toolCfg.PaymentOption = "all-upfront" + toolCfg.TermYears = 3 + toolCfg.MaxInstances = 0 + toolCfg.OverrideCount = MaxReasonableInstances + 1 + }, + expectError: true, + errorMsg: "exceeds reasonable limit", + }, + } + + for _, tt := range tests { + t.Run(tt.name, func(t *testing.T) { + tt.setupCfg() + + err := validateFlags(nil, nil) + + if tt.expectError { + assert.Error(t, err) + assert.Contains(t, err.Error(), tt.errorMsg) + } else { + assert.NoError(t, err) + } + }) + } +} + +func TestValidateFlags_CSVPaths(t *testing.T) { + origCfg := toolCfg + defer func() { toolCfg = origCfg }() + + // Setup base valid config + setupBaseCfg := func() { + toolCfg.Coverage = 80.0 + toolCfg.PaymentOption = "all-upfront" + toolCfg.TermYears = 3 + toolCfg.MaxInstances = 0 + toolCfg.OverrideCount = 0 + toolCfg.CSVOutput = "" + toolCfg.CSVInput = "" + } + + t.Run("Valid CSV output path", func(t *testing.T) { + setupBaseCfg() + toolCfg.CSVOutput = "/tmp/test-output.csv" + + err := validateFlags(nil, nil) + assert.NoError(t, err) + }) + + t.Run("CSV output with non-existent directory", func(t *testing.T) { + setupBaseCfg() + toolCfg.CSVOutput = "/nonexistent/directory/output.csv" + + err := validateFlags(nil, nil) + assert.Error(t, err) + assert.Contains(t, err.Error(), "output directory does not exist") + }) + + t.Run("CSV input with non-existent file", func(t *testing.T) { + setupBaseCfg() + toolCfg.CSVInput = "/nonexistent/file.csv" + + err := validateFlags(nil, nil) + assert.Error(t, err) + assert.Contains(t, err.Error(), "input CSV file does not exist") + }) + + t.Run("CSV input without .csv extension", func(t *testing.T) { + setupBaseCfg() + // Create a temp file without .csv extension + tmpFile, err := os.CreateTemp("", "test-input-*.txt") + assert.NoError(t, err) + defer os.Remove(tmpFile.Name()) + tmpFile.Close() + + toolCfg.CSVInput = tmpFile.Name() + + err = validateFlags(nil, nil) + assert.Error(t, err) + assert.Contains(t, err.Error(), "must have .csv extension") + }) + + t.Run("Valid CSV input file", func(t *testing.T) { + setupBaseCfg() + // Create a temp CSV file + tmpFile, err := os.CreateTemp("", "test-input-*.csv") + assert.NoError(t, err) + defer os.Remove(tmpFile.Name()) + tmpFile.Close() + + toolCfg.CSVInput = tmpFile.Name() + + err = validateFlags(nil, nil) + assert.NoError(t, err) + }) +} + +// ==================== Tests for processService Error Paths ==================== + +func TestProcessService_GetRegionsError(t *testing.T) { + ctx := context.Background() + awsCfg := aws.Config{Region: "us-east-1"} + + origCfg := toolCfg + defer func() { toolCfg = origCfg }() + + toolCfg.Coverage = 100.0 + toolCfg.PaymentOption = "all-upfront" + toolCfg.TermYears = 3 + toolCfg.Regions = []string{} // Empty - will trigger auto-discovery + + mockClient := &MockRecommendationsClient{} + accountCache := NewAccountAliasCache(awsCfg) + + // This test verifies behavior when region discovery is needed + // Since getAllAWSRegions requires real AWS config, we test with explicit regions instead + toolCfg.Regions = []string{"us-east-1"} + + // Setup mock to return empty recommendations + mockClient.On("GetRecommendations", ctx, mock.AnythingOfType("*common.RecommendationParams")).Return([]common.Recommendation{}, nil) + + recs, results := processService(ctx, awsCfg, mockClient, accountCache, common.ServiceRDS, true, toolCfg, engineVersionData{}) + + // Should return empty recommendations + assert.Empty(t, recs) + assert.Empty(t, results) + + mockClient.AssertExpectations(t) +} + +func TestProcessService_GetRecommendationsError(t *testing.T) { + ctx := context.Background() + awsCfg := aws.Config{Region: "us-east-1"} + + origCfg := toolCfg + defer func() { toolCfg = origCfg }() + + toolCfg.Coverage = 100.0 + toolCfg.PaymentOption = "partial-upfront" + toolCfg.TermYears = 1 + toolCfg.Regions = []string{"us-east-1"} + + mockClient := &MockRecommendationsClient{} + accountCache := NewAccountAliasCache(awsCfg) + + // Setup mock to return error + mockClient.On("GetRecommendations", ctx, mock.AnythingOfType("*common.RecommendationParams")).Return([]common.Recommendation(nil), errors.New("API error")) + + recs, results := processService(ctx, awsCfg, mockClient, accountCache, common.ServiceEC2, true, toolCfg, engineVersionData{}) + + // Should continue with empty results after error + assert.Empty(t, recs) + assert.Empty(t, results) + + mockClient.AssertExpectations(t) +} + +func TestProcessService_AllRecommendationsFilteredOut(t *testing.T) { + ctx := context.Background() + awsCfg := aws.Config{Region: "us-east-1"} + + origCfg := toolCfg + defer func() { toolCfg = origCfg }() + + toolCfg.Coverage = 100.0 + toolCfg.PaymentOption = "no-upfront" + toolCfg.TermYears = 1 + toolCfg.Regions = []string{"us-east-1"} + toolCfg.IncludeInstanceTypes = []string{"db.r5.large"} // Filter to specific type + + mockClient := &MockRecommendationsClient{} + accountCache := NewAccountAliasCache(awsCfg) + + // Return recommendations that don't match the filter + mockRecs := []common.Recommendation{ + {ResourceType: "db.t3.small", Count: 5, Region: "us-east-1", EstimatedSavings: 100}, + {ResourceType: "db.t3.medium", Count: 3, Region: "us-east-1", EstimatedSavings: 200}, + } + mockClient.On("GetRecommendations", ctx, mock.AnythingOfType("*common.RecommendationParams")).Return(mockRecs, nil) + + recs, results := processService(ctx, awsCfg, mockClient, accountCache, common.ServiceRDS, true, toolCfg, engineVersionData{}) + + // All recommendations should be filtered out + assert.Empty(t, recs) + assert.Empty(t, results) + + mockClient.AssertExpectations(t) +} + +// ==================== Tests for filterAndAdjustRecommendations Edge Cases ==================== + +func TestFilterAndAdjustRecommendations_ZeroCoverage(t *testing.T) { + saved := saveGlobalVars() + defer saved.restore() + + recommendations := []common.Recommendation{ + {Service: common.ServiceRDS, ResourceType: "db.t3.small", Count: 5}, + {Service: common.ServiceRDS, ResourceType: "db.t3.medium", Count: 3}, + } + + toolCfg.MaxInstances = 0 + toolCfg.OverrideCount = 0 + + result, err := filterAndAdjustRecommendations(recommendations, 0.0, toolCfg) + require.NoError(t, err) + + // 0% coverage should return empty + assert.Empty(t, result) +} + +func TestFilterAndAdjustRecommendations_WithEngineVersionFiltering(t *testing.T) { + saved := saveGlobalVars() + defer saved.restore() + + recommendations := []common.Recommendation{ + { + Service: common.ServiceRDS, + ResourceType: "db.t3.small", + Count: 5, + Region: "us-east-1", + Details: &common.DatabaseDetails{ + Engine: "mysql", + }, + }, + } + + toolCfg.MaxInstances = 0 + toolCfg.OverrideCount = 0 + toolCfg.IncludeExtendedSupport = false + + result, err := filterAndAdjustRecommendations(recommendations, 100.0, toolCfg) + require.NoError(t, err) + + // Should return recommendations (engine version filtering is done inside the function) + assert.NotEmpty(t, result) +} + +func TestFilterAndAdjustRecommendations_MaxInstancesApplied(t *testing.T) { + saved := saveGlobalVars() + defer saved.restore() + + recommendations := []common.Recommendation{ + // EstimatedSavings is set because a binding --max-instances has to rank + // the rows it chooses between; requireRankingSignal refuses a run whose + // rows carry no savings value rather than capping them by name. + {Service: common.ServiceRDS, ResourceType: "db.t3.small", Count: 20, EstimatedSavings: 100}, + {Service: common.ServiceRDS, ResourceType: "db.t3.medium", Count: 20, EstimatedSavings: 200}, + } + + toolCfg.MaxInstances = 15 + toolCfg.OverrideCount = 0 + + result, err := filterAndAdjustRecommendations(recommendations, 100.0, toolCfg) + require.NoError(t, err) + + // Total instances should not exceed maxInstances + totalInstances := 0 + for _, rec := range result { + totalInstances += rec.Count + } + assert.LessOrEqual(t, totalInstances, int(toolCfg.MaxInstances)) +} + +func TestFilterAndAdjustRecommendations_OverrideCountApplied(t *testing.T) { + saved := saveGlobalVars() + defer saved.restore() + + recommendations := []common.Recommendation{ + {Service: common.ServiceRDS, ResourceType: "db.t3.small", Count: 20}, + {Service: common.ServiceRDS, ResourceType: "db.t3.medium", Count: 15}, + } + + toolCfg.MaxInstances = 0 + toolCfg.OverrideCount = 5 + + result, err := filterAndAdjustRecommendations(recommendations, 100.0, toolCfg) + require.NoError(t, err) + + // All recommendations should have count = OverrideCount + for _, rec := range result { + assert.Equal(t, int(toolCfg.OverrideCount), rec.Count) + } +} + +// TestFetchExistingCoverage_LookbackDays verifies that fetchExistingCoverage +// honors cfg.CoverageLookbackDays (issue #360). The test uses the +// MockRecommendationsClient which fails the *awsprovider.RecommendationsClientAdapter +// type assertion, exercising the non-AWS-provider early-return path. The key +// assertions are: +// - TargetCoverage=0 always returns nil regardless of lookback. +// - TargetCoverage>0 with a non-AWS client returns nil (non-AWS path, +// no CE call is made). +// +// Threading through the actual lookback value to GetRICoverageMap is +// integration-tested in providers/aws/recommendations (TestGetRICoverageMap_*). +func TestFetchExistingCoverage_LookbackDays(t *testing.T) { + ctx := context.Background() + awsCfg := aws.Config{} + mockClient := &MockRecommendationsClient{} + + t.Run("zero TargetCoverage returns nil regardless of lookback", func(t *testing.T) { + cfg := Config{TargetCoverage: 0, CoverageLookbackDays: 14, Regions: []string{"us-east-1"}} + got := fetchExistingCoverage(ctx, awsCfg, mockClient, cfg) + assert.Nil(t, got, "TargetCoverage=0 must short-circuit before any CE call") + }) + + t.Run("non-AWS adapter returns nil, lookback not needed", func(t *testing.T) { + cfg := Config{TargetCoverage: 80, CoverageLookbackDays: 14, Regions: []string{"us-east-1"}} + got := fetchExistingCoverage(ctx, awsCfg, mockClient, cfg) + assert.Nil(t, got, "non-AWS provider must return nil (no CE integration)") + }) + + // The previous "custom lookback stored in Config" subcase asserted only that + // a struct field equals what was just assigned -- a tautology that passes even + // if CoverageLookbackDays is never forwarded to GetRICoverageMap. The real + // assertion -- that the lookback value reaches the CE TimePeriod -- is covered + // by TestGetRICoverageMap_LookbackWindowWidth in + // providers/aws/recommendations/coverage_test.go, which directly verifies + // end-start == lookbackDays on the actual CE input. No redundant subcase here. +} diff --git a/cmd/multi_service_csv.go b/cmd/multi_service_csv.go new file mode 100644 index 000000000..b230d56ae --- /dev/null +++ b/cmd/multi_service_csv.go @@ -0,0 +1,509 @@ +package main + +import ( + "encoding/csv" + "errors" + "fmt" + "io" + "log" + "os" + "sort" + "time" + + "github.com/LeanerCloud/CUDly/pkg/common" + "github.com/LeanerCloud/CUDly/providers/aws/recommendations" +) + +// determineCSVCoverage determines the coverage percentage to use for CSV mode. +func determineCSVCoverage(cfg Config) float64 { + // When using CSV input, default to 100% coverage (use exact numbers from CSV) + // unless user explicitly provided a different coverage value + if cfg.Coverage == 80.0 { + // User didn't override the default, so use 100% for CSV mode + return 100.0 + } + return cfg.Coverage +} + +// loadRecommendationsFromCSV reads and returns recommendations from a CSV file. +func loadRecommendationsFromCSV(csvPath string) ([]common.Recommendation, error) { + file, err := os.Open(csvPath) // #nosec G304 -- CLI tool: csvPath is an operator-supplied command-line argument + if err != nil { + return nil, fmt.Errorf("failed to open CSV file: %w", err) + } + defer func() { + if closeErr := file.Close(); closeErr != nil { + log.Printf("Warning: failed to close CSV file %s: %v", csvPath, closeErr) + } + }() + + reader := csv.NewReader(file) + + // Read header + header, err := reader.Read() + if err != nil { + return nil, fmt.Errorf("failed to read CSV header: %w", err) + } + + // Build column index map + colIdx := buildColumnIndexMap(header) + + // Parse all records + parsed, err := parseCSVRecords(reader, colIdx) + if err != nil { + return nil, err + } + + return parsed, nil +} + +// buildColumnIndexMap creates a map from column names to indices. +func buildColumnIndexMap(header []string) map[string]int { + colIdx := make(map[string]int) + for i, col := range header { + colIdx[col] = i + } + return colIdx +} + +// parseCSVRecords reads and parses all CSV records. +func parseCSVRecords(reader *csv.Reader, colIdx map[string]int) ([]common.Recommendation, error) { + var recs []common.Recommendation + + for { + record, err := reader.Read() + if errors.Is(err, io.EOF) { + break + } + if err != nil { + return nil, fmt.Errorf("failed to read CSV record: %w", err) + } + + // Skip the trailing TOTAL summary row that writeMultiServiceCSVReport + // emits (label in the Service column). Without this, feeding the tool + // its own output parses TOTAL as a bogus recommendation with an + // unknown service type. + if getCSVField(record, colIdx, "Service") == "TOTAL" { + continue + } + + rec, err := parseCSVRecord(record, colIdx) + if err != nil { + return nil, err + } + + recs = append(recs, rec) + } + + return recs, nil +} + +// parseCSVRecord parses a single CSV record into a Recommendation. +func parseCSVRecord(record []string, colIdx map[string]int) (common.Recommendation, error) { + rec := common.Recommendation{} + + // Parse string fields + rec.Service = common.ServiceType(getCSVField(record, colIdx, "Service")) + rec.Region = getCSVField(record, colIdx, "Region") + rec.ResourceType = getCSVField(record, colIdx, "ResourceType") + rec.Account = getCSVField(record, colIdx, "Account") + rec.AccountName = getCSVField(record, colIdx, "AccountName") + rec.Term = getCSVField(record, colIdx, "Term") + rec.PaymentOption = getCSVField(record, colIdx, "PaymentOption") + + // Parse integer fields + if err := parseCSVInt(record, colIdx, "Count", &rec.Count); err != nil { + return rec, err + } + + // Parse float fields + if err := parseCSVFloat(record, colIdx, "EstimatedSavings", &rec.EstimatedSavings); err != nil { + return rec, err + } + + // Reconstruct the service Details from the Engine/Deployment columns. The + // purchase path needs them: RDS findOfferingID rejects a rec with nil + // Details ("invalid service details for RDS"), and RI offerings are keyed + // by engine and Multi-AZ. This mirrors the writer side (extractEngine / + // extractDeployment emit DatabaseDetails / CacheDetails / ComputeDetails), + // so a CSV the tool wrote round-trips losslessly. Engine is stored in Cost + // Explorer format ("Aurora MySQL"); findOfferingID normalizes it. Guarded + // on a non-empty Engine so minimal CSVs and Savings Plans rows (no Engine + // column) keep their previous nil-Details behavior. + if engine := getCSVField(record, colIdx, "Engine"); engine != "" { + deployment := getCSVField(record, colIdx, "Deployment") + switch rec.Service { + case common.ServiceRDS, common.ServiceRelationalDB: + rec.Details = &common.DatabaseDetails{ + Engine: engine, + AZConfig: deployment, + InstanceClass: rec.ResourceType, + } + case common.ServiceElastiCache, common.ServiceCache: + rec.Details = &common.CacheDetails{ + Engine: engine, + NodeType: rec.ResourceType, + } + case common.ServiceEC2, common.ServiceCompute: + rec.Details = &common.ComputeDetails{ + InstanceType: rec.ResourceType, + Platform: engine, + } + } + } + + return rec, nil +} + +// getCSVField safely retrieves a string field from a CSV record. +func getCSVField(record []string, colIdx map[string]int, fieldName string) string { + if idx, ok := colIdx[fieldName]; ok && idx < len(record) { + return record[idx] + } + return "" +} + +// parseCSVInt parses an integer field from a CSV record. +func parseCSVInt(record []string, colIdx map[string]int, fieldName string, target *int) error { + value := getCSVField(record, colIdx, fieldName) + if value == "" { + return nil + } + + if _, err := fmt.Sscanf(value, "%d", target); err != nil { + return fmt.Errorf("invalid %s value '%s': %w", fieldName, value, err) + } + return nil +} + +// parseCSVFloat parses a float field from a CSV record. +func parseCSVFloat(record []string, colIdx map[string]int, fieldName string, target *float64) error { + value := getCSVField(record, colIdx, fieldName) + if value == "" { + return nil + } + + if _, err := fmt.Sscanf(value, "%f", target); err != nil { + return fmt.Errorf("invalid %s value '%s': %w", fieldName, value, err) + } + return nil +} + +// writeMultiServiceCSVReport writes purchase results to a CSV file. +func writeMultiServiceCSVReport(results []common.PurchaseResult, filepath string) error { + if len(results) == 0 { + return nil + } + + file, err := os.Create(filepath) // #nosec G304 -- CLI tool: filepath is an operator-supplied output path argument + if err != nil { + return fmt.Errorf("failed to create CSV file: %w", err) + } + defer func() { + if err := file.Close(); err != nil { + log.Printf("Warning: failed to close CSV file %s: %v", filepath, err) + } + }() + + writer := csv.NewWriter(file) + defer writer.Flush() + + // Write header. RecommendedCount shows AWS's pre-sizing count alongside + // Count (the post-sizing value); UpfrontPayment is rec.CommitmentCost, + // which ApplyCoverage / ApplyTargetCoverage now scale at sizing time so + // the value already reflects the sized purchase. ExistingCoverage shows + // the % of demand already covered by commitments in the same pool (from + // CE GetReservationCoverage); ProjectedCoverage shows where the purchase + // landed (total coverage after adding the new RIs). All four optional + // columns render blank when zero so users on the straight --coverage + // path don't see noise. ProjectedUtilization and RecommendedUtilization + // are NOT emitted because under under-buy sizing both land at ~100% on + // every row, which adds noise without information; the underlying fields + // stay on the Recommendation struct for internal use (SP no-signal + // guard, etc.). + header := []string{ + "Service", "Region", "ResourceType", "Family", "Engine", "Deployment", + "Instances", "CoveredInstances", + "Count", "NormalizedUnits", "RecommendedCount", + "Account", "AccountName", "Term", "PaymentOption", + "UpfrontPayment", "RecurringMonthlyCost", "EstimatedSavings", + "CommitmentID", "Success", "Error", "Timestamp", + "ExistingCoverage", "ProjectedCoverage", + } + if err := writer.Write(header); err != nil { + return fmt.Errorf("failed to write CSV header: %w", err) + } + + // Sort by upfront DESC so the biggest-dollar decisions surface at + // the top of the file rather than wherever AWS rec API happened to + // return them. Operators reading top-down see the rows that matter + // most for budget review first. Copy the slice so the caller's + // ordering isn't mutated (some callers iterate results twice). + sorted := make([]common.PurchaseResult, len(results)) + copy(sorted, results) + sort.SliceStable(sorted, func(i, j int) bool { + return sorted[i].Recommendation.CommitmentCost > sorted[j].Recommendation.CommitmentCost + }) + + for i := range sorted { + r := sorted[i] + rec := r.Recommendation + errStr := "" + if r.Error != nil { + errStr = r.Error.Error() + } + + row := []string{ + string(rec.Service), + rec.Region, + rec.ResourceType, + extractRDSFamily(rec), + extractEngine(rec), + extractDeployment(rec), + formatAvgInstancesOrBlank(rec.AverageInstancesUsedPerHour), + formatCoveredInstancesOrBlank(rec), + fmt.Sprintf("%d", rec.Count), + formatNormalizedUnitsOrBlank(rec), + formatIntOrBlank(rec.RecommendedCount), + rec.Account, + rec.AccountName, + rec.Term, + rec.PaymentOption, + formatCurrencyOrBlank(rec.CommitmentCost), + formatRecurringMonthlyOrBlank(rec.RecurringMonthlyCost), + fmt.Sprintf("%.2f", rec.EstimatedSavings), + r.CommitmentID, + fmt.Sprintf("%t", r.Success), + errStr, + r.Timestamp.Format(time.RFC3339), + formatExistingCoverage(rec), + formatPercentOrBlank(rec.ProjectedCoverage), + } + if err := writer.Write(row); err != nil { + return fmt.Errorf("failed to write CSV row: %w", err) + } + } + + // TOTAL row aggregates the sum-able fields (Count, NormalizedUnits, + // UpfrontPayment, RecurringMonthlyCost, EstimatedSavings) so operators + // don't have to recompute in a spreadsheet. The "TOTAL" label lands in + // the Service column for easy spotting; columns that don't aggregate + // meaningfully (per-rec identifiers, timestamps, %) stay blank. + if len(sorted) > 0 { + totalRow := buildTotalRow(sorted) + if err := writer.Write(totalRow); err != nil { + return fmt.Errorf("failed to write CSV total row: %w", err) + } + } + + return nil +} + +// buildTotalRow sums the count + currency columns across results and +// returns a row aligned to the same header order as writeCSVRowsOrdered. +// Non-summable cells (per-rec identifiers, percentages, timestamps) are +// blank; the "TOTAL" label lands in Service so the row reads as a +// summary at first glance. +func buildTotalRow(results []common.PurchaseResult) []string { + var totalCount int + var totalNU, totalUpfront, totalRecurring, totalSavings float64 + hasRecurring := false + for i := range results { + r := results[i] + totalCount += r.Recommendation.Count + totalNU += float64(r.Recommendation.Count) * recommendations.RDSInstanceNUFromType(r.Recommendation.ResourceType) + totalUpfront += r.Recommendation.CommitmentCost + totalSavings += r.Recommendation.EstimatedSavings + if r.Recommendation.RecurringMonthlyCost != nil { + totalRecurring += *r.Recommendation.RecurringMonthlyCost + hasRecurring = true + } + } + recurringCell := "" + if hasRecurring { + recurringCell = fmt.Sprintf("%.2f", totalRecurring) + } + nuCell := "" + if totalNU > 0 { + nuCell = fmt.Sprintf("%g", totalNU) + } + return []string{ + "TOTAL", "", "", "", "", "", // Service through Deployment + "", "", // Instances, CoveredInstances + fmt.Sprintf("%d", totalCount), nuCell, "", // Count, NormalizedUnits, RecommendedCount + "", "", "", "", // Account, AccountName, Term, PaymentOption + fmt.Sprintf("%.2f", totalUpfront), recurringCell, fmt.Sprintf("%.2f", totalSavings), + "", "", "", "", // CommitmentID, Success, Error, Timestamp + "", "", // ExistingCoverage, ProjectedCoverage + } +} + +// formatIntOrBlank renders an int as its decimal string when non-zero, "" +// otherwise. SP recommendations leave RecommendedCount at zero (SPs are +// dollar-denominated, not count-denominated), so blanking matches the +// "0 = unknown / not applicable" convention used elsewhere in the CSV. +func formatIntOrBlank(v int) string { + if v == 0 { + return "" + } + return fmt.Sprintf("%d", v) +} + +// extractRDSFamily returns the RDS instance-family prefix (e.g. +// "db.r7g") for an RDS recommendation, empty for any service whose +// instance type doesn't follow the RDS three-part naming. Useful for +// grouping rows in the CSV by family-NU bucket so operators can see at +// a glance which recs belong to the same size-flex family. +func extractRDSFamily(rec common.Recommendation) string { + if rec.Service != common.ServiceRDS && rec.Service != common.ServiceRelationalDB { + return "" + } + return recommendations.RDSFamilyFromType(rec.ResourceType) +} + +// formatNormalizedUnitsOrBlank renders the per-rec NU contribution +// (rec.Count × NU(size)) for RDS rows: e.g. 15 × db.r7g.large = 60 NU. +// Surfaces the size-flex math AWS rec API uses to bundle family demand +// into a single rec at one size — without this column, operators have +// to compute NU by hand to verify the bundling. Renders blank for +// non-RDS rows and for sizes not in the standard NU scale. +func formatNormalizedUnitsOrBlank(rec common.Recommendation) string { + if rec.Service != common.ServiceRDS && rec.Service != common.ServiceRelationalDB { + return "" + } + nu := recommendations.RDSInstanceNUFromType(rec.ResourceType) + if nu == 0 || rec.Count == 0 { + return "" + } + return fmt.Sprintf("%g", float64(rec.Count)*nu) +} + +// extractDeployment returns the RDS deployment-option string +// ("single-az" / "multi-az") for an RDS recommendation, empty for any +// service that doesn't carry a deployment dimension. Critical for RDS +// price verification: Multi-AZ list prices are roughly 2x Single-AZ, so +// operators need to see the deployment alongside the upfront figure to +// confirm a $X upfront row is for the deployment they expect. +// +// DatabaseDetails is always a pointer: every producer (AWS, Azure, the CSV +// loader, and the JSON codec used on the purchase-execution round trip) +// constructs it that way -- see pkg/common/service_details_codec.go's +// package doc for the invariant. +func extractDeployment(rec common.Recommendation) string { + if details, ok := rec.Details.(*common.DatabaseDetails); ok && details != nil { + return details.AZConfig + } + return "" +} + +// extractEngine returns the engine / platform string for a recommendation's +// polymorphic Details: Engine for RDS / ElastiCache (DatabaseDetails, +// CacheDetails), Platform for EC2 (ComputeDetails), empty for SP and other +// commitment types that don't carry an engine field. +// +// DatabaseDetails/CacheDetails are always pointers (see extractDeployment's +// godoc for the invariant). ComputeDetails is still accepted as a value +// because the Azure compute client and the GCP compute-engine client both +// construct it that way; without that case the column silently blanks every +// Azure VM and GCP compute row. +func extractEngine(rec common.Recommendation) string { + switch details := rec.Details.(type) { + case *common.DatabaseDetails: + if details != nil { + return details.Engine + } + case *common.CacheDetails: + if details != nil { + return details.Engine + } + case *common.ComputeDetails: + if details != nil { + return details.Platform + } + case common.ComputeDetails: + return details.Platform + } + return "" +} + +// formatExistingCoverage renders the existing-RI coverage cell with +// three distinct states: +// - "n/a" when CE returned no data for the rec's pool (rec parser was +// able to surface a recommendation from some other signal but CE's +// coverage view doesn't see the pool yet — e.g. recently-launched +// instances within CUDly's run window but outside CE's lookback) +// - "0.0" when CE confirms the pool exists but has zero RI coverage +// (the legitimate "buy for uncovered demand" case) +// - "X.X" with one decimal for any non-zero coverage percentage +// +// Previously both the no-data and the genuine-zero cases rendered as a +// blank cell, conflating "we don't know" with "definitely zero" and +// making it impossible to spot pools where the CE signal was missing. +func formatExistingCoverage(rec common.Recommendation) string { + if !rec.ExistingCoverageKnown { + return "n/a" + } + return fmt.Sprintf("%.1f", rec.ExistingCoveragePct) +} + +// formatRecurringMonthlyOrBlank renders rec.RecurringMonthlyCost (the +// per-month fee on top of any upfront payment, populated by the AWS +// parser when CE returns RecurringStandardMonthlyCost). Distinguishes +// "no recurring fee" (all-upfront RIs, where the pointer is set to +// zero) from "unknown" (pointer is nil because CE didn't return the +// field): zero renders as "0.00", nil renders as blank. +// +// Operators on partial-upfront / no-upfront plans need this to compute +// total cost (upfront + monthly × 36); without it the CSV only shows +// the upfront portion and over-states ROI. +func formatRecurringMonthlyOrBlank(p *float64) string { + if p == nil { + return "" + } + return fmt.Sprintf("%.2f", *p) +} + +// formatCurrencyOrBlank renders a currency value as "%.2f" when non-zero, +// "" otherwise. Used for UpfrontPayment so a no-upfront / unknown-upfront +// rec renders as a blank cell rather than "$0.00". +func formatCurrencyOrBlank(v float64) string { + if v == 0 { + return "" + } + return fmt.Sprintf("%.2f", v) +} + +// formatAvgInstancesOrBlank renders the average instances-per-hour signal +// (AverageInstancesUsedPerHour from CE) with one decimal so operators can +// see the pool's running demand without losing the fractional precision +// CE returns. Blank when zero, matching the "0 = no signal" convention. +func formatAvgInstancesOrBlank(v float64) string { + if v == 0 { + return "" + } + return fmt.Sprintf("%.1f", v) +} + +// formatCoveredInstancesOrBlank renders the instances in the pool already +// covered by existing commitments: avg × existing_coverage / 100. Useful +// next to Instances so operators can read "you have X running, Y are +// already covered, this rec adds N more" without doing the arithmetic. +// Blank when either signal is zero (we can't compute a meaningful value). +func formatCoveredInstancesOrBlank(rec common.Recommendation) string { + if rec.AverageInstancesUsedPerHour <= 0 || rec.ExistingCoveragePct <= 0 { + return "" + } + covered := rec.AverageInstancesUsedPerHour * rec.ExistingCoveragePct / 100.0 + return fmt.Sprintf("%.1f", covered) +} + +// formatPercentOrBlank renders a % value as "%.1f" when non-zero, "" otherwise. +// Zero means "unknown / not applicable" — we don't want "0.0" in cells where +// the metric simply wasn't computed (e.g. ProjectedCoverage for SP rows, or +// any utilization field when --target-coverage wasn't used). +func formatPercentOrBlank(v float64) string { + if v == 0 { + return "" + } + return fmt.Sprintf("%.1f", v) +} diff --git a/cmd/multi_service_csv_cap.go b/cmd/multi_service_csv_cap.go new file mode 100644 index 000000000..163c0a69e --- /dev/null +++ b/cmd/multi_service_csv_cap.go @@ -0,0 +1,163 @@ +package main + +import ( + "fmt" + "sort" + "strings" + + "github.com/LeanerCloud/CUDly/pkg/common" + "github.com/LeanerCloud/CUDly/pkg/scorer" +) + +// scoreAndLimitCSVRecs enforces --min-count and the run-wide --max-instances +// cap on recommendations loaded from --input-csv, so both spend guards behave +// on the CSV path as they do on the recommendation-driven path. Both are +// documented in docs/cli/filtering.md as flags of the tool, not of a mode. +// +// Previously the CSV path handed the load-ordered slice straight to +// ApplyInstanceLimit, which consumes its input in slice order and drops the +// tail: whichever rows appeared first in the file spent the budget, and +// --min-count was never consulted, so a row the cap truncated below the floor +// was purchased short. +// +// Only MinCount is enforced here. A CSV row carries no savings percentage and +// no break-even figure (writeMultiServiceCSVReport emits neither column and +// parseCSVRecord reads neither), so both fields load as zero and gating on +// them would reject every row of every file. --min-savings-pct and +// --max-break-even-months are therefore refused up front by +// validateCSVModeFilterFlags rather than accepted and ignored here; #1819 +// tracks teaching the CSV format to carry those columns. +// +// scorer.Score's own ordering is useless here: SavingsPercentage is uniformly +// zero on loaded rows, so it resolves on EstimatedSavings, a whole-row dollar +// total, while --max-instances is a budget in instances. Ranking a total +// against a per-instance budget spends the whole budget on whichever row is +// merely biggest, not on the rows that return the most per instance bought. +// sortBySavingsPerInstance therefore re-orders the survivors on the rate +// derived from the two columns a CSV does carry, which is the greedy the flag +// implies, and applyGlobalInstanceLimit consumes that order: the cap keeps the +// best-value rows, names every row it reduces or drops, and drops rather than +// shortens anything truncated below --min-count. +func scoreAndLimitCSVRecs(recs []common.Recommendation, cfg Config) ([]common.Recommendation, error) { + passed := applyMinCountFloor(recs, cfg.MinCount) + if err := requireRankingSignal(passed, cfg); err != nil { + return nil, err + } + sortBySavingsPerInstance(passed) + return applyGlobalInstanceLimit(passed, cfg, rankBySavingsPerInstance, nil), nil +} + +// applyMinCountFloor drops recommendations below the --min-count floor and +// names each drop on stdout. A floor of 0 disables the flag, and returns recs +// untouched rather than re-ordering them for nothing. +// +// The floor itself is scorer.Score's, so the CSV path and the +// recommendation-driven path reject on the identical predicate and reason +// string. MinCount is the only scorer.Config field this helper ever sets, so +// the "--min-count dropped" prefix cannot come to describe some other filter. +func applyMinCountFloor(recs []common.Recommendation, minCount int) []common.Recommendation { + if minCount <= 0 { + return recs + } + scored := scorer.Score(recs, scorer.Config{MinCount: minCount}) + for i := range scored.Filtered { + f := scored.Filtered[i] + AppLogger.Printf("🔒 --min-count dropped %s %s %s: %s\n", + f.Recommendation.Service, f.Recommendation.Region, f.Recommendation.ResourceType, f.FilterReason) + } + return scored.Passed +} + +// savingsPerInstance is the ranking key for CSV rows: the row's monthly +// savings divided by the instances it would buy. --max-instances is a budget +// in instances, so the rows worth keeping are the ones returning the most per +// instance, not the ones whose total happens to be largest. +// +// A non-positive Count buys nothing and has no rate, so it ranks last rather +// than dividing by zero. ApplyInstanceLimit already refuses to credit budget +// back for such a row. +func savingsPerInstance(rec common.Recommendation) float64 { + if rec.Count <= 0 { + return 0 + } + return rec.EstimatedSavings / float64(rec.Count) +} + +// sortBySavingsPerInstance orders recs best-value-first, in place. +// +// Rows with equal rates fall back to the same Service|Region|ResourceType key +// scorer.Score tie-breaks on, so the selection is deterministic whatever order +// the file listed them in. The tie-break is spelled out here rather than +// inherited from an upstream sort because --min-count 0 skips the scorer +// entirely, which would otherwise leave file order deciding between equals on +// exactly the path #1741 is about. +// +// It must run before the cap, never after: ApplyInstanceLimit consumes its +// input in slice order, so by the time it returns the selection has already +// been made and re-ordering the survivors decides nothing. The rate values +// themselves are invariant across the cap since #1830 (a truncated row's +// savings and Count are scaled by the same ratio, and savings/count is +// unchanged by scaling both), but that only means a post-cap sort would be +// harmless rather than useful. Ordering still has to happen first. +func sortBySavingsPerInstance(recs []common.Recommendation) { + sort.SliceStable(recs, func(i, j int) bool { + a, b := recs[i], recs[j] + if rateA, rateB := savingsPerInstance(a), savingsPerInstance(b); rateA != rateB { + return rateA > rateB + } + keyA := string(a.Service) + "|" + a.Region + "|" + a.ResourceType + keyB := string(b.Service) + "|" + b.Region + "|" + b.ResourceType + return keyA < keyB + }) +} + +// maxNamedUnrankableRows bounds how many offending rows requireRankingSignal +// names before summarizing the rest, so a large file produces a readable error. +const maxNamedUnrankableRows = 5 + +// requireRankingSignal refuses a run whose --max-instances cap has to choose +// between rows it cannot rank. +// +// parseCSVFloat leaves EstimatedSavings at zero for a blank cell, and +// getCSVField returns "" for a column that is not in the header at all, so a +// CSV written without an EstimatedSavings column loads every row at zero. +// Nothing downstream can tell that apart from a row genuinely worth $0: the +// value is absent, not zero. With every rate equal, sortBySavingsPerInstance +// and scorer.Score both fall through to the Service|Region|ResourceType +// tie-break, and the cap silently buys by instance-type name while stdout and +// docs/cli/filtering.md both promise it is buying by savings. +// +// Ranking only decides anything when the cap actually binds, so that is the +// only case this refuses; a file with no savings column still runs uncapped, +// and so does one whose total already fits. Money paths in this project fail +// loud rather than picking a defensible-looking default, and #1741's own +// framing is that silent partial enforcement of a spend guard is worse than +// not offering the guard. +func requireRankingSignal(recs []common.Recommendation, cfg Config) error { + if !capBinds(recs, cfg) { + return nil + } + + unrankable := make([]string, 0) + for i := range recs { + if recs[i].EstimatedSavings <= 0 { + unrankable = append(unrankable, fmt.Sprintf("%s %s %s", + recs[i].Service, recs[i].Region, recs[i].ResourceType)) + } + } + if len(unrankable) == 0 { + return nil + } + + named := unrankable + suffix := "" + if len(named) > maxNamedUnrankableRows { + named = named[:maxNamedUnrankableRows] + suffix = fmt.Sprintf(" (and %d more)", len(unrankable)-maxNamedUnrankableRows) + } + return fmt.Errorf( + "--max-instances=%d has to choose which of %d recommendations to buy, but %d row(s) of %s carry no usable EstimatedSavings value: %s%s. "+ + "A blank or missing EstimatedSavings cell is indistinguishable from $0 of savings, so capping on it would pick by instance-type name rather than by value. "+ + "Populate EstimatedSavings for every row, or drop --max-instances and cap the file itself", + cfg.MaxInstances, len(recs), len(unrankable), cfg.CSVInput, strings.Join(named, ", "), suffix) +} diff --git a/cmd/multi_service_csv_test.go b/cmd/multi_service_csv_test.go new file mode 100644 index 000000000..6fc24eae6 --- /dev/null +++ b/cmd/multi_service_csv_test.go @@ -0,0 +1,853 @@ +package main + +import ( + "encoding/csv" + "errors" + "os" + "path/filepath" + "strings" + "testing" + "time" + + "github.com/LeanerCloud/CUDly/pkg/common" + "github.com/stretchr/testify/assert" + "github.com/stretchr/testify/require" +) + +func TestDetermineCSVCoverage(t *testing.T) { + tests := []struct { + name string + cfg Config + expected float64 + }{ + { + name: "Default coverage (80) changed to 100 for CSV", + cfg: Config{ + Coverage: 80.0, + }, + expected: 100.0, + }, + { + name: "User-specified coverage preserved", + cfg: Config{ + Coverage: 75.0, + }, + expected: 75.0, + }, + { + name: "User-specified 100% coverage preserved", + cfg: Config{ + Coverage: 100.0, + }, + expected: 100.0, + }, + { + name: "User-specified 50% coverage preserved", + cfg: Config{ + Coverage: 50.0, + }, + expected: 50.0, + }, + { + name: "User-specified 0% coverage preserved", + cfg: Config{ + Coverage: 0.0, + }, + expected: 0.0, + }, + } + + for _, tt := range tests { + t.Run(tt.name, func(t *testing.T) { + result := determineCSVCoverage(tt.cfg) + assert.Equal(t, tt.expected, result) + }) + } +} + +func TestWriteMultiServiceCSVReport(t *testing.T) { + tests := []struct { + name string + filename string + results []common.PurchaseResult + wantErr bool + }{ + { + name: "RDS results", + results: []common.PurchaseResult{ + { + Recommendation: common.Recommendation{ + Service: common.ServiceRDS, + Region: "us-east-1", + ResourceType: "db.t3.micro", + Count: 2, + Term: "3yr", + PaymentOption: "partial-upfront", + EstimatedSavings: 100, + SavingsPercentage: 30, + Timestamp: time.Now(), + Details: &common.DatabaseDetails{ + Engine: "mysql", + AZConfig: "multi-az", + }, + }, + Success: true, + CommitmentID: "test-001", + Timestamp: time.Now(), + }, + }, + filename: "test-rds.csv", + wantErr: false, + }, + { + name: "ElastiCache results", + results: []common.PurchaseResult{ + { + Recommendation: common.Recommendation{ + Service: common.ServiceElastiCache, + Region: "us-west-2", + ResourceType: "cache.t3.micro", + Count: 1, + Term: "1yr", + Details: &common.CacheDetails{ + Engine: "redis", + NodeType: "cache.t3.micro", + }, + }, + Success: true, + CommitmentID: "test-002", + Timestamp: time.Now(), + }, + }, + filename: "test-cache.csv", + wantErr: false, + }, + { + name: "EC2 results", + results: []common.PurchaseResult{ + { + Recommendation: common.Recommendation{ + Service: common.ServiceEC2, + Region: "eu-west-1", + ResourceType: "t3.medium", + Count: 5, + Term: "3yr", + Details: common.ComputeDetails{ + Platform: "Linux/UNIX", + Tenancy: "shared", + Scope: "region", + }, + }, + Success: false, + CommitmentID: "test-003", + Error: errors.New("Insufficient capacity"), + Timestamp: time.Now(), + }, + }, + filename: "test-ec2.csv", + wantErr: false, + }, + { + name: "Empty results", + results: []common.PurchaseResult{}, + filename: "test-empty.csv", + wantErr: false, + }, + { + name: "Unknown service type", + results: []common.PurchaseResult{ + { + Recommendation: common.Recommendation{ + Service: common.ServiceType("unknown"), + Region: "us-east-1", + ResourceType: "unknown.large", + Count: 1, + Term: "3yr", + }, + Success: true, + }, + }, + filename: "test-unknown.csv", + wantErr: false, + }, + } + + for _, tt := range tests { + t.Run(tt.name, func(t *testing.T) { + tmpDir := t.TempDir() + filepath := tmpDir + "/" + tt.filename + + err := writeMultiServiceCSVReport(tt.results, filepath) + + if tt.wantErr { + assert.Error(t, err) + } else { + assert.NoError(t, err) + } + }) + } +} + +// TestWriteMultiServiceCSVReport_CoverageColumn confirms the ProjectedCoverage +// and RecommendedCount columns added for --target-coverage (#338) are emitted +// with the "blank-when-zero" formatting (matches the "0 = unknown" convention +// shared with the JSON-level omitempty tags). The sibling ProjectedUtilization +// and RecommendedUtilization fields are intentionally NOT emitted to CSV +// (both land at ~100% on every under-buy row, so they add noise without +// information). +func TestWriteMultiServiceCSVReport_CoverageColumn(t *testing.T) { + tmpDir := t.TempDir() + filepath := tmpDir + "/util.csv" + + results := []common.PurchaseResult{ + { + Recommendation: common.Recommendation{ + Service: common.ServiceEC2, + Region: "us-east-1", + ResourceType: "t3.medium", + Count: 7, // post-sizing + RecommendedCount: 10, // AWS pre-sizing + CommitmentCost: 700, // already scaled at sizing time + Term: "1yr", + ProjectedUtilization: 95.0, + ProjectedCoverage: 87.5, + ExistingCoveragePct: 20.0, + ExistingCoverageKnown: true, + RecommendedUtilization: 80.0, + AverageInstancesUsedPerHour: 10.0, + // Pointer form matches the live parser (parser_services.go + // stores &common.ComputeDetails{...}); extractEngine must + // handle both pointer and value Details. + Details: &common.ComputeDetails{Platform: "Linux/UNIX"}, + }, + Success: true, + }, + { + // All sizing-related fields zero — ProjectedCoverage and + // RecommendedCount cells should both be blank (SP rec or a + // pre-target rec that never went through sizing). UpfrontPayment + // is also blank when CommitmentCost is zero. No Details either, + // so the Engine column is blank. + Recommendation: common.Recommendation{ + Service: common.ServiceEC2, + Region: "us-east-1", + ResourceType: "m5.large", + Count: 5, + Term: "1yr", + }, + Success: true, + }, + } + + err := writeMultiServiceCSVReport(results, filepath) + require.NoError(t, err) + + content, err := os.ReadFile(filepath) + require.NoError(t, err) + csvText := string(content) + + // Header contains ProjectedCoverage, ExistingCoverage, RecommendedCount, + // and UpfrontPayment but NOT the always-100% utilization siblings. + assert.Contains(t, csvText, "ProjectedCoverage") + assert.Contains(t, csvText, "ExistingCoverage") + assert.Contains(t, csvText, "RecommendedCount") + assert.Contains(t, csvText, "UpfrontPayment") + assert.NotContains(t, csvText, "ProjectedUtilization", "column was removed; it's ~100% on every under-buy row") + assert.NotContains(t, csvText, "RecommendedUtilization", "column was removed; it's ~99-100% on every row") + + // First data row (populated rec) has the coverage and AWS-count values. + assert.Contains(t, csvText, "87.5", "ProjectedCoverage should render with one decimal") + + r := csv.NewReader(strings.NewReader(csvText)) + rows, err := r.ReadAll() + require.NoError(t, err) + // Header + 2 data rows + 1 TOTAL row. + require.Len(t, rows, 4) + // TOTAL row sits at the bottom with "TOTAL" in the Service column and + // summed Count / UpfrontPayment / EstimatedSavings. + totalRow := rows[3] + assert.Equal(t, "TOTAL", totalRow[0], "TOTAL label lands in Service column") + // Data rows are sorted by UpfrontPayment DESC; populated rec ($700) + // comes before the empty rec ($0). + header := rows[0] + idxProjCov, idxRecCount, idxUpfront, idxExisting := -1, -1, -1, -1 + idxEngine, idxInstances, idxCovered := -1, -1, -1 + for i, h := range header { + switch h { + case "ProjectedCoverage": + idxProjCov = i + case "RecommendedCount": + idxRecCount = i + case "UpfrontPayment": + idxUpfront = i + case "ExistingCoverage": + idxExisting = i + case "Engine": + idxEngine = i + case "Instances": + idxInstances = i + case "CoveredInstances": + idxCovered = i + } + } + require.NotEqual(t, -1, idxProjCov, "ProjectedCoverage column not found") + require.NotEqual(t, -1, idxRecCount, "RecommendedCount column not found") + require.NotEqual(t, -1, idxUpfront, "UpfrontPayment column not found") + require.NotEqual(t, -1, idxExisting, "ExistingCoverage column not found") + require.NotEqual(t, -1, idxEngine, "Engine column not found") + require.NotEqual(t, -1, idxInstances, "Instances column not found") + require.NotEqual(t, -1, idxCovered, "CoveredInstances column not found") + + // Populated row: RecommendedCount=10 renders as "10", UpfrontPayment + // emits CommitmentCost as-is (sizing already scaled it; see + // ApplyTargetCoverage), ProjectedCoverage=87.5 renders, ExistingCoverage=20.0, + // Engine pulled from *ComputeDetails.Platform. Instances = avg = 10.0. + // CoveredInstances = 10.0 × 20% = 2.0. + populatedRow := rows[1] + assert.Equal(t, "10", populatedRow[idxRecCount], "RecommendedCount should render as decimal") + assert.Equal(t, "700.00", populatedRow[idxUpfront], "UpfrontPayment should render rec.CommitmentCost as-is") + assert.Equal(t, "87.5", populatedRow[idxProjCov]) + assert.Equal(t, "20.0", populatedRow[idxExisting], "ExistingCoverage should render with one decimal") + assert.Equal(t, "Linux/UNIX", populatedRow[idxEngine], "Engine should pull from *ComputeDetails.Platform") + assert.Equal(t, "10.0", populatedRow[idxInstances], "Instances should render avg with one decimal") + assert.Equal(t, "2.0", populatedRow[idxCovered], "CoveredInstances = avg * existing_cov / 100") + + // Zero-fields row: optional cells blank, Engine blank when Details is nil. + // ExistingCoverage shows "n/a" because ExistingCoverageKnown wasn't set: + // CE had no data for this pool (distinct from "0% covered", which would + // be ExistingCoverageKnown=true, Pct=0 rendering as "0.0"). + zeroRow := rows[2] + assert.Equal(t, "", zeroRow[idxProjCov], "zero ProjectedCoverage should be blank") + assert.Equal(t, "", zeroRow[idxRecCount], "zero RecommendedCount should be blank (SP rec or pre-sizing)") + assert.Equal(t, "", zeroRow[idxUpfront], "zero CommitmentCost should leave UpfrontPayment blank") + assert.Equal(t, "n/a", zeroRow[idxExisting], "ExistingCoverage should render n/a when CE had no signal") + assert.Equal(t, "", zeroRow[idxEngine], "missing Details should leave Engine blank") + assert.Equal(t, "", zeroRow[idxInstances], "zero avg should leave Instances blank") + assert.Equal(t, "", zeroRow[idxCovered], "missing avg or existing_cov should leave CoveredInstances blank") +} + +// TestWriteMultiServiceCSVReport_SortAndTotal confirms data rows are +// sorted by UpfrontPayment DESC and that a TOTAL summary row lands at +// the bottom with the column sums. Operators reading the file top-down +// want the biggest-dollar decisions surfaced first; the TOTAL row +// removes the need to copy-paste columns into a spreadsheet to add +// them up. +func TestWriteMultiServiceCSVReport_SortAndTotal(t *testing.T) { + tmpDir := t.TempDir() + fp := tmpDir + "/sort-total.csv" + + // Three recs: $5K, $20K, $1K upfront. After DESC sort the order + // should be 20K, 5K, 1K. + results := []common.PurchaseResult{ + {Recommendation: common.Recommendation{Service: "rds", ResourceType: "db.r6g.large", Count: 5, CommitmentCost: 5000, EstimatedSavings: 500}}, + {Recommendation: common.Recommendation{Service: "rds", ResourceType: "db.r6g.2xlarge", Count: 2, CommitmentCost: 20000, EstimatedSavings: 1500}}, + {Recommendation: common.Recommendation{Service: "rds", ResourceType: "db.t4g.medium", Count: 4, CommitmentCost: 1000, EstimatedSavings: 80}}, + } + require.NoError(t, writeMultiServiceCSVReport(results, fp)) + content, err := os.ReadFile(fp) + require.NoError(t, err) + + r := csv.NewReader(strings.NewReader(string(content))) + rows, err := r.ReadAll() + require.NoError(t, err) + require.Len(t, rows, 5) // header + 3 data + TOTAL + + // Find UpfrontPayment column index. + header := rows[0] + idxUpfront := -1 + idxCount := -1 + idxService := -1 + idxNU := -1 + idxSavings := -1 + for i, h := range header { + switch h { + case "UpfrontPayment": + idxUpfront = i + case "Count": + idxCount = i + case "Service": + idxService = i + case "NormalizedUnits": + idxNU = i + case "EstimatedSavings": + idxSavings = i + } + } + + // Sort order: $20K, $5K, $1K. + assert.Equal(t, "20000.00", rows[1][idxUpfront], "row 1 has the largest upfront") + assert.Equal(t, "5000.00", rows[2][idxUpfront]) + assert.Equal(t, "1000.00", rows[3][idxUpfront]) + + // TOTAL row aggregates: count=11, upfront=$26K, savings=$2,080. + // NU = 5×4 + 2×16 + 4×2 = 20 + 32 + 8 = 60. + totalRow := rows[4] + assert.Equal(t, "TOTAL", totalRow[idxService]) + assert.Equal(t, "11", totalRow[idxCount]) + assert.Equal(t, "60", totalRow[idxNU]) + assert.Equal(t, "26000.00", totalRow[idxUpfront]) + assert.Equal(t, "2080.00", totalRow[idxSavings]) +} + +// TestFormatExistingCoverage locks the three-state rendering: +// - ExistingCoverageKnown=false → "n/a" (CE has no data for this pool) +// - ExistingCoverageKnown=true, Pct=0 → "0.0" (CE confirms zero coverage) +// - ExistingCoverageKnown=true, Pct>0 → formatted with one decimal +// +// Critical for operators interpreting the column: a blank or zero cell +// previously meant either "CE was queried but returned 0%" or "CE +// returned nothing", with no way to tell which. +func TestFormatExistingCoverage(t *testing.T) { + tests := []struct { + name string + rec common.Recommendation + want string + }{ + {"unknown (CE no data)", common.Recommendation{}, "n/a"}, + {"known zero coverage", common.Recommendation{ExistingCoverageKnown: true}, "0.0"}, + {"known partial coverage", common.Recommendation{ExistingCoverageKnown: true, ExistingCoveragePct: 37.74}, "37.7"}, + {"known full coverage", common.Recommendation{ExistingCoverageKnown: true, ExistingCoveragePct: 100.0}, "100.0"}, + } + for _, tt := range tests { + t.Run(tt.name, func(t *testing.T) { + assert.Equal(t, tt.want, formatExistingCoverage(tt.rec)) + }) + } +} + +// TestFormatRecurringMonthlyOrBlank locks the nil-vs-zero distinction: +// nil pointer (AWS API didn't return RecurringStandardMonthlyCost) +// renders as blank, zero value (genuinely no monthly fee, e.g. +// all-upfront RIs) renders as "0.00". Operators need to tell "we don't +// know" apart from "definitely zero" to compute total cost correctly. +func TestFormatRecurringMonthlyOrBlank(t *testing.T) { + zero := 0.0 + twenty := 20.5 + tests := []struct { + name string + in *float64 + want string + }{ + {"nil → blank (unknown)", nil, ""}, + {"zero pointer → 0.00 (definitely zero)", &zero, "0.00"}, + {"non-zero → formatted", &twenty, "20.50"}, + } + for _, tt := range tests { + t.Run(tt.name, func(t *testing.T) { + assert.Equal(t, tt.want, formatRecurringMonthlyOrBlank(tt.in)) + }) + } +} + +// TestExtractRDSFamily covers the family-prefix extraction used by the +// CSV writer to group rows by size-flex family. +func TestExtractRDSFamily(t *testing.T) { + tests := []struct { + name string + rec common.Recommendation + want string + }{ + {"RDS db.r7g.large", common.Recommendation{Service: common.ServiceRDS, ResourceType: "db.r7g.large"}, "db.r7g"}, + {"RDS db.t4g.medium", common.Recommendation{Service: common.ServiceRDS, ResourceType: "db.t4g.medium"}, "db.t4g"}, + {"RelationalDB alias", common.Recommendation{Service: common.ServiceRelationalDB, ResourceType: "db.m5.xlarge"}, "db.m5"}, + // Non-RDS services blank even when ResourceType looks RDS-shaped. + {"EC2 ignored", common.Recommendation{Service: common.ServiceEC2, ResourceType: "m5.large"}, ""}, + {"ElastiCache ignored", common.Recommendation{Service: common.ServiceElastiCache, ResourceType: "cache.t3.micro"}, ""}, + // Malformed RDS type. + {"RDS bare type", common.Recommendation{Service: common.ServiceRDS, ResourceType: "db.r7g"}, ""}, + } + for _, tt := range tests { + t.Run(tt.name, func(t *testing.T) { + assert.Equal(t, tt.want, extractRDSFamily(tt.rec)) + }) + } +} + +// TestFormatNormalizedUnitsOrBlank confirms NU values land for RDS rows +// with known sizes and stay blank for non-RDS / zero-count / unknown-size +// inputs, matching the "0/empty = unknown" convention used elsewhere. +func TestFormatNormalizedUnitsOrBlank(t *testing.T) { + tests := []struct { + name string + rec common.Recommendation + want string + }{ + // 15 × db.r7g.large = 15 × 4 NU = 60 NU + {"r7g.large × 15", common.Recommendation{Service: common.ServiceRDS, ResourceType: "db.r7g.large", Count: 15}, "60"}, + // 3 × db.t4g.medium = 3 × 2 NU = 6 NU + {"t4g.medium × 3", common.Recommendation{Service: common.ServiceRDS, ResourceType: "db.t4g.medium", Count: 3}, "6"}, + // Fractional NU survives via %g (db.t4g.micro = 0.5 NU) + {"t4g.micro × 3", common.Recommendation{Service: common.ServiceRDS, ResourceType: "db.t4g.micro", Count: 3}, "1.5"}, + // Non-RDS service → blank + {"EC2 row blank", common.Recommendation{Service: common.ServiceEC2, ResourceType: "m5.large", Count: 5}, ""}, + // Zero count → blank + {"zero count blank", common.Recommendation{Service: common.ServiceRDS, ResourceType: "db.r7g.large", Count: 0}, ""}, + // Unknown size → blank + {"unknown size blank", common.Recommendation{Service: common.ServiceRDS, ResourceType: "db.r7g.bogus", Count: 5}, ""}, + } + for _, tt := range tests { + t.Run(tt.name, func(t *testing.T) { + assert.Equal(t, tt.want, formatNormalizedUnitsOrBlank(tt.rec)) + }) + } +} + +// TestExtractDeployment covers the deployment-extraction helper used by +// the RDS row in the CSV. Single-AZ / Multi-AZ is critical context for +// pricing verification (Multi-AZ list price is ~2x Single-AZ) so the +// column should land for every RDS rec. DatabaseDetails is always a +// pointer (every producer constructs it that way); no value-typed case +// is needed. +func TestExtractDeployment(t *testing.T) { + tests := []struct { + name string + rec common.Recommendation + want string + }{ + {"*DatabaseDetails Single-AZ", common.Recommendation{Details: &common.DatabaseDetails{AZConfig: "single-az"}}, "single-az"}, + {"*DatabaseDetails Multi-AZ", common.Recommendation{Details: &common.DatabaseDetails{AZConfig: "multi-az"}}, "multi-az"}, + {"DatabaseDetails empty AZConfig", common.Recommendation{Details: &common.DatabaseDetails{Engine: "mysql"}}, ""}, + // Non-RDS Details → blank (column is RDS-only data). + {"CacheDetails -> empty", common.Recommendation{Details: &common.CacheDetails{Engine: "redis"}}, ""}, + {"ComputeDetails -> empty", common.Recommendation{Details: &common.ComputeDetails{Platform: "Linux/UNIX"}}, ""}, + {"nil Details -> empty", common.Recommendation{}, ""}, + {"nil *DatabaseDetails -> empty", common.Recommendation{Details: (*common.DatabaseDetails)(nil)}, ""}, + } + for _, tt := range tests { + t.Run(tt.name, func(t *testing.T) { + assert.Equal(t, tt.want, extractDeployment(tt.rec)) + }) + } +} + +// TestExtractEngine covers the four cases the helper dispatches on: +// DatabaseDetails (RDS engine), CacheDetails (ElastiCache engine), +// ComputeDetails (EC2 platform), and unset/other Details (blank). +// DatabaseDetails/CacheDetails are always pointers (every producer +// constructs them that way); ComputeDetails still accepts a value because +// the Azure compute client and the GCP compute-engine client both construct +// it that way. +func TestExtractEngine(t *testing.T) { + tests := []struct { + name string + rec common.Recommendation + want string + }{ + {"*DatabaseDetails -> Engine", common.Recommendation{Details: &common.DatabaseDetails{Engine: "aurora-postgresql"}}, "aurora-postgresql"}, + {"*CacheDetails -> Engine", common.Recommendation{Details: &common.CacheDetails{Engine: "redis"}}, "redis"}, + {"*ComputeDetails -> Platform", common.Recommendation{Details: &common.ComputeDetails{Platform: "Linux/UNIX"}}, "Linux/UNIX"}, + // The Azure compute client and GCP's compute-engine client both + // still construct ComputeDetails as a value; keep coverage for + // that form. + {"ComputeDetails (value) -> Platform", common.Recommendation{Details: common.ComputeDetails{Platform: "Windows"}}, "Windows"}, + // Fallbacks. + {"nil Details -> empty", common.Recommendation{}, ""}, + {"SavingsPlanDetails -> empty", common.Recommendation{Details: &common.SavingsPlanDetails{HourlyCommitment: 1.0}}, ""}, + {"nil *DatabaseDetails -> empty", common.Recommendation{Details: (*common.DatabaseDetails)(nil)}, ""}, + } + for _, tt := range tests { + t.Run(tt.name, func(t *testing.T) { + assert.Equal(t, tt.want, extractEngine(tt.rec)) + }) + } +} + +// TestFormatCurrencyOrBlank locks the blank-when-zero behavior for the +// UpfrontPayment column. Non-zero renders with two decimals; zero renders +// as an empty cell so users can distinguish "no upfront due" from "actual +// $0 upfront", consistent with the rest of the optional CSV columns. +func TestFormatCurrencyOrBlank(t *testing.T) { + tests := []struct { + name string + in float64 + want string + }{ + {"non-zero renders with two decimals", 1234.56, "1234.56"}, + {"integer value gets .00", 700, "700.00"}, + {"zero blanks the cell", 0, ""}, + } + for _, tt := range tests { + t.Run(tt.name, func(t *testing.T) { + assert.Equal(t, tt.want, formatCurrencyOrBlank(tt.in)) + }) + } +} + +// Tests for loadRecommendationsFromCSV function. +func TestLoadRecommendationsFromCSV(t *testing.T) { + tests := []struct { + validate func(t *testing.T, recs []common.Recommendation) + name string + csvContent string + errContains string + wantErr bool + }{ + { + name: "Valid CSV with all fields", + csvContent: `Service,Region,ResourceType,Count,Account,AccountName,Term,PaymentOption,EstimatedSavings +rds,us-east-1,db.t3.micro,5,123456789012,Production,3yr,partial-upfront,1500.50 +ec2,us-west-2,t3.medium,10,123456789012,Development,1yr,all-upfront,2000.75`, + wantErr: false, + validate: func(t *testing.T, recs []common.Recommendation) { + require.Len(t, recs, 2) + + // Validate first recommendation + assert.Equal(t, common.ServiceRDS, recs[0].Service) + assert.Equal(t, "us-east-1", recs[0].Region) + assert.Equal(t, "db.t3.micro", recs[0].ResourceType) + assert.Equal(t, 5, recs[0].Count) + assert.Equal(t, "123456789012", recs[0].Account) + assert.Equal(t, "Production", recs[0].AccountName) + assert.Equal(t, "3yr", recs[0].Term) + assert.Equal(t, "partial-upfront", recs[0].PaymentOption) + assert.InDelta(t, 1500.50, recs[0].EstimatedSavings, 0.01) + + // Validate second recommendation + assert.Equal(t, common.ServiceEC2, recs[1].Service) + assert.Equal(t, "us-west-2", recs[1].Region) + assert.Equal(t, "t3.medium", recs[1].ResourceType) + assert.Equal(t, 10, recs[1].Count) + }, + }, + { + name: "Valid CSV with minimal fields", + csvContent: `Service,Region,ResourceType,Count +elasticache,eu-west-1,cache.t3.micro,3`, + wantErr: false, + validate: func(t *testing.T, recs []common.Recommendation) { + require.Len(t, recs, 1) + assert.Equal(t, common.ServiceElastiCache, recs[0].Service) + assert.Equal(t, "eu-west-1", recs[0].Region) + assert.Equal(t, "cache.t3.micro", recs[0].ResourceType) + assert.Equal(t, 3, recs[0].Count) + }, + }, + { + name: "Valid CSV with empty optional Account field", + csvContent: `Service,Region,ResourceType,Count,Account +rds,us-east-1,db.t3.micro,2,`, + wantErr: false, + validate: func(t *testing.T, recs []common.Recommendation) { + require.Len(t, recs, 1) + assert.Equal(t, common.ServiceRDS, recs[0].Service) + assert.Equal(t, "us-east-1", recs[0].Region) + assert.Equal(t, "db.t3.micro", recs[0].ResourceType) + assert.Equal(t, 2, recs[0].Count) + assert.Equal(t, "", recs[0].Account) + }, + }, + { + name: "Engine and Deployment reconstruct service Details", + csvContent: `Service,Region,ResourceType,Engine,Deployment,Count,Term,PaymentOption +rds,us-east-1,db.t4g.medium,Aurora MySQL,single-az,3,3yr,partial-upfront +rds,eu-west-2,db.r7g.large,MySQL,multi-az,2,3yr,partial-upfront +elasticache,us-east-1,cache.r6g.large,redis,,4,1yr,no-upfront +ec2,us-west-2,m5.large,Linux/UNIX,,5,1yr,all-upfront`, + wantErr: false, + validate: func(t *testing.T, recs []common.Recommendation) { + require.Len(t, recs, 4) + + // RDS Aurora MySQL, single-az -> DatabaseDetails + rds0, ok := recs[0].Details.(*common.DatabaseDetails) + require.True(t, ok, "RDS rec should carry *DatabaseDetails") + assert.Equal(t, "Aurora MySQL", rds0.Engine) + assert.Equal(t, "single-az", rds0.AZConfig) + assert.Equal(t, "db.t4g.medium", rds0.InstanceClass) + + // RDS MySQL, multi-az -> AZConfig carried through for offering lookup + rds1, ok := recs[1].Details.(*common.DatabaseDetails) + require.True(t, ok) + assert.Equal(t, "MySQL", rds1.Engine) + assert.Equal(t, "multi-az", rds1.AZConfig) + + // ElastiCache -> CacheDetails (no Deployment column) + cache, ok := recs[2].Details.(*common.CacheDetails) + require.True(t, ok, "ElastiCache rec should carry *CacheDetails") + assert.Equal(t, "redis", cache.Engine) + assert.Equal(t, "cache.r6g.large", cache.NodeType) + + // EC2 -> ComputeDetails, platform from the Engine column + ec2, ok := recs[3].Details.(*common.ComputeDetails) + require.True(t, ok, "EC2 rec should carry *ComputeDetails") + assert.Equal(t, "Linux/UNIX", ec2.Platform) + assert.Equal(t, "m5.large", ec2.InstanceType) + }, + }, + { + name: "No Engine column leaves Details nil (Savings Plans / minimal CSV)", + csvContent: `Service,Region,ResourceType,Count,Term,PaymentOption +rds,us-east-1,db.t3.micro,2,3yr,partial-upfront`, + wantErr: false, + validate: func(t *testing.T, recs []common.Recommendation) { + require.Len(t, recs, 1) + assert.Nil(t, recs[0].Details, "Details must stay nil when Engine column is absent") + }, + }, + { + name: "TOTAL summary row is skipped", + csvContent: `Service,Region,ResourceType,Engine,Deployment,Count,Term,PaymentOption +rds,us-east-1,db.t4g.medium,Aurora MySQL,single-az,3,3yr,partial-upfront +TOTAL,,,,,3,,`, + wantErr: false, + validate: func(t *testing.T, recs []common.Recommendation) { + require.Len(t, recs, 1, "the trailing TOTAL row must not become a recommendation") + assert.Equal(t, common.ServiceRDS, recs[0].Service) + }, + }, + { + name: "CSV with different column order", + csvContent: `Count,Service,Region,ResourceType +7,rds,ap-south-1,db.r5.large`, + wantErr: false, + validate: func(t *testing.T, recs []common.Recommendation) { + require.Len(t, recs, 1) + assert.Equal(t, common.ServiceRDS, recs[0].Service) + assert.Equal(t, "ap-south-1", recs[0].Region) + assert.Equal(t, "db.r5.large", recs[0].ResourceType) + assert.Equal(t, 7, recs[0].Count) + }, + }, + { + name: "Empty CSV file (only header)", + csvContent: `Service,Region,ResourceType,Count,Account,AccountName,Term,PaymentOption,EstimatedSavings +`, + wantErr: false, + validate: func(t *testing.T, recs []common.Recommendation) { + assert.Len(t, recs, 0) + }, + }, + { + name: "Invalid Count value - non-numeric", + csvContent: `Service,Region,ResourceType,Count +rds,us-east-1,db.t3.micro,abc`, + wantErr: true, + errContains: "invalid Count value", + }, + { + name: "Invalid EstimatedSavings value - non-numeric", + csvContent: `Service,Region,ResourceType,Count,EstimatedSavings +rds,us-east-1,db.t3.micro,5,invalid`, + wantErr: true, + errContains: "invalid EstimatedSavings value", + }, + { + name: "Multiple rows with various services", + csvContent: `Service,Region,ResourceType,Count,EstimatedSavings +rds,us-east-1,db.t3.micro,5,100.00 +ec2,us-west-2,t3.medium,10,200.50 +elasticache,eu-west-1,cache.t3.micro,3,50.25 +opensearch,ap-southeast-1,t3.small.search,2,75.00`, + wantErr: false, + validate: func(t *testing.T, recs []common.Recommendation) { + require.Len(t, recs, 4) + assert.Equal(t, common.ServiceRDS, recs[0].Service) + assert.Equal(t, common.ServiceEC2, recs[1].Service) + assert.Equal(t, common.ServiceElastiCache, recs[2].Service) + assert.Equal(t, common.ServiceOpenSearch, recs[3].Service) + }, + }, + { + name: "CSV with large Count values", + csvContent: `Service,Region,ResourceType,Count +ec2,us-east-1,t3.large,1000`, + wantErr: false, + validate: func(t *testing.T, recs []common.Recommendation) { + require.Len(t, recs, 1) + assert.Equal(t, 1000, recs[0].Count) + }, + }, + { + name: "CSV with decimal savings", + csvContent: `Service,Region,ResourceType,Count,EstimatedSavings +rds,us-east-1,db.t3.micro,5,1234.5678`, + wantErr: false, + validate: func(t *testing.T, recs []common.Recommendation) { + require.Len(t, recs, 1) + assert.InDelta(t, 1234.5678, recs[0].EstimatedSavings, 0.0001) + }, + }, + { + name: "CSV with zero values", + csvContent: `Service,Region,ResourceType,Count,EstimatedSavings +rds,us-east-1,db.t3.micro,0,0`, + wantErr: false, + validate: func(t *testing.T, recs []common.Recommendation) { + require.Len(t, recs, 1) + assert.Equal(t, 0, recs[0].Count) + assert.Equal(t, float64(0), recs[0].EstimatedSavings) + }, + }, + } + + for _, tt := range tests { + t.Run(tt.name, func(t *testing.T) { + // Create temporary CSV file + tmpDir := t.TempDir() + csvPath := filepath.Join(tmpDir, "test.csv") + err := os.WriteFile(csvPath, []byte(tt.csvContent), 0644) + require.NoError(t, err) + + // Call function + recs, err := loadRecommendationsFromCSV(csvPath) + + // Validate results + if tt.wantErr { + assert.Error(t, err) + if tt.errContains != "" { + assert.Contains(t, err.Error(), tt.errContains) + } + } else { + assert.NoError(t, err) + if tt.validate != nil { + tt.validate(t, recs) + } + } + }) + } +} + +// Test loadRecommendationsFromCSV with file errors. +func TestLoadRecommendationsFromCSV_FileErrors(t *testing.T) { + tests := []struct { + name string + setup func(t *testing.T) string + errContains string + }{ + { + name: "Non-existent file", + setup: func(t *testing.T) string { + return "/nonexistent/path/to/file.csv" + }, + errContains: "failed to open CSV file", + }, + { + name: "Directory instead of file", + setup: func(t *testing.T) string { + return t.TempDir() + }, + errContains: "", + }, + { + name: "Empty file (no header)", + setup: func(t *testing.T) string { + tmpDir := t.TempDir() + csvPath := filepath.Join(tmpDir, "empty.csv") + err := os.WriteFile(csvPath, []byte(""), 0644) + require.NoError(t, err) + return csvPath + }, + errContains: "failed to read CSV header", + }, + } + + for _, tt := range tests { + t.Run(tt.name, func(t *testing.T) { + path := tt.setup(t) + _, err := loadRecommendationsFromCSV(path) + assert.Error(t, err) + if tt.errContains != "" { + assert.Contains(t, err.Error(), tt.errContains) + } + }) + } +} diff --git a/cmd/multi_service_engine_versions.go b/cmd/multi_service_engine_versions.go new file mode 100644 index 000000000..d6f395b55 --- /dev/null +++ b/cmd/multi_service_engine_versions.go @@ -0,0 +1,491 @@ +package main + +import ( + "context" + "fmt" + "log" + "runtime" + "strings" + "sync" + "time" + + "github.com/LeanerCloud/CUDly/pkg/common" + "github.com/aws/aws-sdk-go-v2/aws" + "github.com/aws/aws-sdk-go-v2/config" + awsec2 "github.com/aws/aws-sdk-go-v2/service/ec2" + ec2types "github.com/aws/aws-sdk-go-v2/service/ec2/types" + awsrds "github.com/aws/aws-sdk-go-v2/service/rds" + rdstypes "github.com/aws/aws-sdk-go-v2/service/rds/types" +) + +// InstanceEngineVersion stores engine version information for an instance. +type InstanceEngineVersion struct { + Engine string + EngineVersion string + InstanceClass string + Region string +} + +// EngineLifecycleInfo stores lifecycle support information for a major engine version. +type EngineLifecycleInfo struct { + LifecycleSupportStartDate time.Time + LifecycleSupportEndDate time.Time + LifecycleSupportName string +} + +// MajorEngineVersionInfo stores support information for a major engine version. +type MajorEngineVersionInfo struct { + Engine string + MajorEngineVersion string + SupportedEngineLifecycles []EngineLifecycleInfo +} + +// queryRunningInstanceEngineVersions queries all running RDS instances and returns their engine versions. +func queryRunningInstanceEngineVersions(ctx context.Context, cfg Config) (map[string][]InstanceEngineVersion, error) { + awsCfg, err := loadValidationAWSConfig(ctx, cfg) + if err != nil { + return nil, err + } + + regions, err := getAWSRegions(ctx, awsCfg) + if err != nil { + return nil, err + } + + return queryRDSInstancesInRegions(ctx, awsCfg, regions) +} + +// loadValidationAWSConfig loads AWS configuration for validation. +func loadValidationAWSConfig(ctx context.Context, cfg Config) (aws.Config, error) { + validationProfile := cfg.ValidationProfile + if validationProfile == "" { + validationProfile = cfg.Profile + } + + var configOptions []func(*config.LoadOptions) error + configOptions = append(configOptions, config.WithRegion("us-east-1")) + if validationProfile != "" { + configOptions = append(configOptions, config.WithSharedConfigProfile(validationProfile)) + } + + awsCfg, err := config.LoadDefaultConfig(ctx, configOptions...) + if err != nil { + return aws.Config{}, fmt.Errorf("failed to load validation AWS config: %w", err) + } + + return awsCfg, nil +} + +// getAWSRegions retrieves all AWS regions. +func getAWSRegions(ctx context.Context, awsCfg aws.Config) ([]ec2types.Region, error) { + ec2Client := awsec2.NewFromConfig(awsCfg) + regionsOutput, err := ec2Client.DescribeRegions(ctx, &awsec2.DescribeRegionsInput{}) + if err != nil { + return nil, fmt.Errorf("failed to describe regions: %w", err) + } + return regionsOutput.Regions, nil +} + +// maxConcurrentRegionQueries limits the number of concurrent AWS API calls across regions. +const maxConcurrentRegionQueries = 10 + +// maxEngineVersionPages caps DescribeDBMajorEngineVersions pagination per engine. +// 20 pages x ~100 records/page = ~2000 records, enough for any engine list (issue #692). +const maxEngineVersionPages = 20 + +// RDSMajorVersionsClient is the subset of the RDS API needed by +// queryMajorEngineVersionsWithClient, extracted so tests can inject a mock. +type RDSMajorVersionsClient interface { + DescribeDBMajorEngineVersions(ctx context.Context, params *awsrds.DescribeDBMajorEngineVersionsInput, optFns ...func(*awsrds.Options)) (*awsrds.DescribeDBMajorEngineVersionsOutput, error) +} + +// queryRDSInstancesInRegions queries RDS instances in all regions concurrently. +func queryRDSInstancesInRegions(ctx context.Context, awsCfg aws.Config, regions []ec2types.Region) (map[string][]InstanceEngineVersion, error) { + instanceVersions := make(map[string][]InstanceEngineVersion) + var mu sync.Mutex + var wg sync.WaitGroup + + sem := make(chan struct{}, maxConcurrentRegionQueries) + + for _, region := range regions { + wg.Add(1) + sem <- struct{}{} // acquire semaphore + go func(regionName string) { + defer wg.Done() + defer func() { <-sem }() // release semaphore + defer func() { + if r := recover(); r != nil { + buf := make([]byte, 4096) + n := runtime.Stack(buf, false) + log.Printf("ERROR: panic in region worker (region=%s): %v\n%s", regionName, r, buf[:n]) + } + }() + queryRDSInstancesInRegion(ctx, awsCfg, regionName, instanceVersions, &mu) + }(aws.ToString(region.RegionName)) + } + + wg.Wait() + return instanceVersions, nil +} + +// queryRDSInstancesInRegion queries RDS instances in a single region. +func queryRDSInstancesInRegion(ctx context.Context, awsCfg aws.Config, regionName string, instanceVersions map[string][]InstanceEngineVersion, mu *sync.Mutex) { + regionCfg := awsCfg.Copy() + regionCfg.Region = regionName + rdsClient := awsrds.NewFromConfig(regionCfg) + + var marker *string + for { + localVersions, nextMarker, err := queryRDSInstancesPage(ctx, rdsClient, marker, regionName) + if err != nil { + log.Printf("⚠️ Warning: Failed to describe RDS instances in %s: %v", regionName, err) + break + } + + // Merge into shared map with mutex protection + mu.Lock() + for instanceType, versions := range localVersions { + instanceVersions[instanceType] = append(instanceVersions[instanceType], versions...) + } + mu.Unlock() + + if nextMarker == nil { + break + } + marker = nextMarker + } +} + +// queryRDSInstancesPage queries a single page of RDS instances. +func queryRDSInstancesPage(ctx context.Context, rdsClient *awsrds.Client, marker *string, regionName string) (versions map[string][]InstanceEngineVersion, nextMarker *string, err error) { + input := &awsrds.DescribeDBInstancesInput{Marker: marker} + output, err := rdsClient.DescribeDBInstances(ctx, input) + if err != nil { + return nil, nil, err + } + + versions = make(map[string][]InstanceEngineVersion) + for _rvc := range output.DBInstances { + dbInstance := output.DBInstances[_rvc] + instanceClass := aws.ToString(dbInstance.DBInstanceClass) + engine := aws.ToString(dbInstance.Engine) + engineVersion := aws.ToString(dbInstance.EngineVersion) + + versions[instanceClass] = append(versions[instanceClass], InstanceEngineVersion{ + Engine: engine, + EngineVersion: engineVersion, + InstanceClass: instanceClass, + Region: regionName, + }) + } + + if output.Marker != nil && aws.ToString(output.Marker) != "" { + nextMarker = output.Marker + } + + return +} + +// queryMajorEngineVersions queries AWS for major engine version lifecycle support information. +func queryMajorEngineVersions(ctx context.Context, cfg Config) (map[string]MajorEngineVersionInfo, error) { + // Determine which profile to use + profile := cfg.ValidationProfile + if profile == "" { + profile = cfg.Profile + } + + // Load AWS configuration + var configOptions []func(*config.LoadOptions) error + configOptions = append(configOptions, config.WithRegion("us-east-1")) + if profile != "" { + configOptions = append(configOptions, config.WithSharedConfigProfile(profile)) + } + awsCfg, err := config.LoadDefaultConfig(ctx, configOptions...) + if err != nil { + return nil, fmt.Errorf("failed to load AWS config: %w", err) + } + + return queryMajorEngineVersionsWithClient(ctx, awsrds.NewFromConfig(awsCfg)) +} + +// queryMajorEngineVersionsWithClient is the testable core of queryMajorEngineVersions. +// It accepts a RDSMajorVersionsClient so tests can inject a mock without real AWS creds. +func queryMajorEngineVersionsWithClient(ctx context.Context, rdsClient RDSMajorVersionsClient) (map[string]MajorEngineVersionInfo, error) { + // Map of "engine:majorVersion" -> MajorEngineVersionInfo + versionInfo := make(map[string]MajorEngineVersionInfo) + + // Query all engine types we care about + engines := []string{"mysql", "postgres", "aurora-mysql", "aurora-postgresql"} + + for _, engine := range engines { + if err := fetchMajorEngineVersionsForEngine(ctx, rdsClient, engine, versionInfo); err != nil { + log.Printf("Warning: Failed to describe major engine versions for %s: %v", engine, err) + } + } + + return versionInfo, nil +} + +// fetchMajorEngineVersionsForEngine fetches all pages of major engine version +// info for a single engine and merges results into versionInfo. Returns an error +// only on API failure or pagination cap exceeded (issue #692). +func fetchMajorEngineVersionsForEngine(ctx context.Context, rdsClient RDSMajorVersionsClient, engine string, versionInfo map[string]MajorEngineVersionInfo) error { + var marker *string + + for pageIdx := 0; ; pageIdx++ { + if err := ctx.Err(); err != nil { + return err + } + if pageIdx >= maxEngineVersionPages { + return fmt.Errorf( + "pagination cap reached after %d pages for engine %s (issue #692)", + maxEngineVersionPages, engine, + ) + } + + output, err := rdsClient.DescribeDBMajorEngineVersions(ctx, &awsrds.DescribeDBMajorEngineVersionsInput{ + Engine: aws.String(engine), + Marker: marker, + }) + if err != nil { + return err + } + + for _, version := range output.DBMajorEngineVersions { + info := parseDBMajorEngineVersion(version) + key := fmt.Sprintf("%s:%s", info.Engine, info.MajorEngineVersion) + versionInfo[key] = info + } + + if output.Marker == nil || aws.ToString(output.Marker) == "" { + break + } + marker = output.Marker + } + + return nil +} + +// parseDBMajorEngineVersion converts an RDS DBMajorEngineVersion into a +// MajorEngineVersionInfo, extracting lifecycle support dates. Extracted from +// fetchMajorEngineVersionsForEngine to keep its cyclomatic complexity below +// the gocyclo cap. +func parseDBMajorEngineVersion(version rdstypes.DBMajorEngineVersion) MajorEngineVersionInfo { + info := MajorEngineVersionInfo{ + Engine: aws.ToString(version.Engine), + MajorEngineVersion: aws.ToString(version.MajorEngineVersion), + } + + for _, lifecycle := range version.SupportedEngineLifecycles { + lifecycleInfo := EngineLifecycleInfo{ + LifecycleSupportName: string(lifecycle.LifecycleSupportName), + } + + if lifecycle.LifecycleSupportStartDate != nil { + lifecycleInfo.LifecycleSupportStartDate = *lifecycle.LifecycleSupportStartDate + } + if lifecycle.LifecycleSupportEndDate != nil { + lifecycleInfo.LifecycleSupportEndDate = *lifecycle.LifecycleSupportEndDate + } + + info.SupportedEngineLifecycles = append(info.SupportedEngineLifecycles, lifecycleInfo) + } + + return info +} + +// extractMajorVersion extracts the major version from a full engine version string +// Handles special cases like Aurora MySQL version mapping. +func extractMajorVersion(engine, fullVersion string) string { + if fullVersion == "" { + return "" + } + + normalizedEngine := normalizeEngineNameForVersion(engine) + + // Handle Aurora MySQL special format + if normalizedEngine == "auroramysql" { + if auroraVersion := extractAuroraMySQLVersion(fullVersion); auroraVersion != "" { + return auroraVersion + } + } + + // For standard versions, extract "X.Y" or "X" + return extractStandardVersion(fullVersion) +} + +// normalizeEngineNameForVersion normalizes an engine name by removing spaces and hyphens. +func normalizeEngineNameForVersion(engine string) string { + normalized := strings.ToLower(engine) + normalized = strings.ReplaceAll(normalized, "-", "") + normalized = strings.ReplaceAll(normalized, " ", "") + return normalized +} + +// extractAuroraMySQLVersion extracts the MySQL-compatible version from Aurora MySQL. +func extractAuroraMySQLVersion(fullVersion string) string { + // Aurora MySQL 2.x is compatible with MySQL 5.7 + if strings.Contains(fullVersion, "mysql_aurora.2.") { + return "5.7" + } + // Aurora MySQL 3.x is compatible with MySQL 8.0 + if strings.Contains(fullVersion, "mysql_aurora.3.") { + return "8.0" + } + // Check if it starts with a version number + if strings.HasPrefix(fullVersion, "5.7") { + return "5.7" + } + if strings.HasPrefix(fullVersion, "8.0") { + return "8.0" + } + return "" +} + +// extractStandardVersion extracts major.minor version from a standard version string. +func extractStandardVersion(fullVersion string) string { + parts := strings.Split(fullVersion, ".") + if len(parts) >= 2 { + return extractMajorMinorVersion(parts[0], parts[1]) + } + if len(parts) >= 1 { + return parts[0] + } + return "" +} + +// extractMajorMinorVersion combines major and minor version parts. +func extractMajorMinorVersion(major, minor string) string { + // Filter out non-numeric parts in minor version + numericMinor := extractNumericPrefix(minor) + if numericMinor != "" { + return major + "." + numericMinor + } + return major +} + +// extractNumericPrefix extracts the numeric prefix from a string. +func extractNumericPrefix(s string) string { + numericPrefix := "" + for _, ch := range s { + if ch >= '0' && ch <= '9' { + numericPrefix += string(ch) + } else { + break + } + } + return numericPrefix +} + +// isInExtendedSupport checks if a version is currently in extended support based on lifecycle dates. +func isInExtendedSupport(engine, fullVersion string, versionInfo map[string]MajorEngineVersionInfo) bool { + majorVersion := extractMajorVersion(engine, fullVersion) + if majorVersion == "" { + return false + } + + // Normalize engine name for lookup + normalizedEngine := strings.ToLower(engine) + normalizedEngine = strings.ReplaceAll(normalizedEngine, " ", "") + + // Look up the version info + key := fmt.Sprintf("%s:%s", normalizedEngine, majorVersion) + info, exists := versionInfo[key] + if !exists { + // If we don't have info, assume not in extended support + return false + } + + // Check if current date falls within extended support period + now := time.Now() + for _, lifecycle := range info.SupportedEngineLifecycles { + if lifecycle.LifecycleSupportName != string(rdstypes.LifecycleSupportNameOpenSourceRdsExtendedSupport) { + continue + } + // Not in extended support before its start date + if now.Before(lifecycle.LifecycleSupportStartDate) { + continue + } + // Past the end date means extended support is over; a zero end date + // is open-ended (issue #1182) + if lifecycle.LifecycleSupportEndDate.IsZero() || now.Before(lifecycle.LifecycleSupportEndDate) { + return true + } + } + + return false +} + +// adjustRecommendationForExcludedVersions reduces the instance count in a recommendation +// by the number of instances running versions in extended support. +func adjustRecommendationForExcludedVersions(rec common.Recommendation, instanceVersions map[string][]InstanceEngineVersion, versionInfo map[string]MajorEngineVersionInfo) common.Recommendation { + // Check if this instance type has any running instances + versions, exists := instanceVersions[rec.ResourceType] + if !exists { + // No running instances of this type, return unchanged + return rec + } + + // Get the engine name from the recommendation. DatabaseDetails is + // always a pointer (every producer constructs it that way; see + // pkg/common/service_details_codec.go's package doc). A typed nil + // pointer is treated like a non-RDS rec rather than panicking on the + // field read. + var recEngine string + switch details := rec.Details.(type) { + case *common.DatabaseDetails: + if details == nil { + return rec + } + recEngine = details.Engine + default: + return rec // Not RDS, no engine version filtering + } + + // Count how many instances in this region are running versions in extended support + excludedCount := 0 + + for _, version := range versions { + // Only count instances in the same region + if version.Region != rec.Region { + continue + } + + // Match engine (normalize by removing spaces/hyphens and comparing lowercase) + normalizeEngine := func(engine string) string { + normalized := strings.ToLower(engine) + normalized = strings.ReplaceAll(normalized, "-", "") + normalized = strings.ReplaceAll(normalized, " ", "") + return normalized + } + + versionEngineNorm := normalizeEngine(version.Engine) + recEngineNorm := normalizeEngine(recEngine) + + if versionEngineNorm != recEngineNorm { + continue + } + + // Check if this version is in extended support + if isInExtendedSupport(version.Engine, version.EngineVersion, versionInfo) { + majorVersion := extractMajorVersion(version.Engine, version.EngineVersion) + excludedCount++ + log.Printf("🚫 Found extended support instance: %s %s in %s running version %s (major version %s is in extended support)", + recEngine, rec.ResourceType, rec.Region, version.EngineVersion, majorVersion) + } + } + + // If we found excluded instances, reduce the recommendation count + if excludedCount > 0 { + originalCount := rec.Count + newCount := max(0, rec.Count-excludedCount) + + if newCount != originalCount { + log.Printf("📉 Adjusting recommendation for %s %s in %s: %d instances → %d instances (excluded %d extended support instances)", + recEngine, rec.ResourceType, rec.Region, originalCount, newCount, excludedCount) + rec.Count = newCount + } + } + + return rec +} diff --git a/cmd/multi_service_engine_versions_paginate_test.go b/cmd/multi_service_engine_versions_paginate_test.go new file mode 100644 index 000000000..c8ad2296a --- /dev/null +++ b/cmd/multi_service_engine_versions_paginate_test.go @@ -0,0 +1,153 @@ +package main + +import ( + "context" + "fmt" + "testing" + "time" + + "github.com/aws/aws-sdk-go-v2/aws" + awsrds "github.com/aws/aws-sdk-go-v2/service/rds" + rdstypes "github.com/aws/aws-sdk-go-v2/service/rds/types" + "github.com/stretchr/testify/assert" + "github.com/stretchr/testify/require" +) + +// multiPageRDSMajorVersionsMock implements RDSMajorVersionsClient and returns +// distinct pages based on the Marker in the incoming request. +type multiPageRDSMajorVersionsMock struct { + pages []*awsrds.DescribeDBMajorEngineVersionsOutput + tokens []string // tokens[i] triggers pages[i+1]; first call has empty marker + calls int +} + +func (m *multiPageRDSMajorVersionsMock) DescribeDBMajorEngineVersions( + _ context.Context, + params *awsrds.DescribeDBMajorEngineVersionsInput, + _ ...func(*awsrds.Options), +) (*awsrds.DescribeDBMajorEngineVersionsOutput, error) { + idx := 0 + incoming := aws.ToString(params.Marker) + for i, tok := range m.tokens { + if tok == incoming { + idx = i + 1 + break + } + } + if incoming == "" { + idx = 0 + } + m.calls++ + if idx >= len(m.pages) { + return nil, fmt.Errorf("unexpected RDS Marker %q", incoming) + } + return m.pages[idx], nil +} + +// rdsMajorVersion builds a minimal DBMajorEngineVersion for tests. +func rdsMajorVersion(engine, major string) rdstypes.DBMajorEngineVersion { //nolint:unparam // param intentional for interface consistency/future use + return rdstypes.DBMajorEngineVersion{ + Engine: aws.String(engine), + MajorEngineVersion: aws.String(major), + SupportedEngineLifecycles: []rdstypes.SupportedEngineLifecycle{ + { + LifecycleSupportName: rdstypes.LifecycleSupportNameOpenSourceRdsExtendedSupport, + LifecycleSupportStartDate: aws.Time(time.Now().AddDate(-1, 0, 0)), + LifecycleSupportEndDate: aws.Time(time.Now().AddDate(2, 0, 0)), + }, + }, + } +} + +// TestFetchMajorEngineVersionsForEngine_Paginates asserts that all pages are +// fetched and results accumulated (issue #692 regression test). +func TestFetchMajorEngineVersionsForEngine_Paginates(t *testing.T) { + mock := &multiPageRDSMajorVersionsMock{ + pages: []*awsrds.DescribeDBMajorEngineVersionsOutput{ + { + DBMajorEngineVersions: []rdstypes.DBMajorEngineVersion{ + rdsMajorVersion("mysql", "5.7"), + rdsMajorVersion("mysql", "8.0"), + }, + Marker: aws.String("tok1"), + }, + { + DBMajorEngineVersions: []rdstypes.DBMajorEngineVersion{ + rdsMajorVersion("mysql", "8.4"), + rdsMajorVersion("mysql", "9.0"), + }, + Marker: aws.String("tok2"), + }, + { + DBMajorEngineVersions: []rdstypes.DBMajorEngineVersion{ + rdsMajorVersion("mysql", "9.1"), + }, + Marker: nil, + }, + }, + tokens: []string{"tok1", "tok2"}, + } + + versionInfo := make(map[string]MajorEngineVersionInfo) + err := fetchMajorEngineVersionsForEngine(context.Background(), mock, "mysql", versionInfo) + require.NoError(t, err) + // 2 + 2 + 1 = 5 versions across 3 pages + assert.Len(t, versionInfo, 5, "must accumulate all versions across pages") + assert.Equal(t, 3, mock.calls, "must call API once per page") + assert.Contains(t, versionInfo, "mysql:5.7") + assert.Contains(t, versionInfo, "mysql:9.1") +} + +// TestFetchMajorEngineVersionsForEngine_EmptyMarkerTerminates asserts that an +// empty-string Marker is treated as terminal (parity with PR #690). +func TestFetchMajorEngineVersionsForEngine_EmptyMarkerTerminates(t *testing.T) { + mock := &multiPageRDSMajorVersionsMock{ + pages: []*awsrds.DescribeDBMajorEngineVersionsOutput{ + { + DBMajorEngineVersions: []rdstypes.DBMajorEngineVersion{ + rdsMajorVersion("mysql", "8.0"), + }, + Marker: aws.String(""), // empty string -- must terminate + }, + }, + tokens: []string{}, + } + + versionInfo := make(map[string]MajorEngineVersionInfo) + err := fetchMajorEngineVersionsForEngine(context.Background(), mock, "mysql", versionInfo) + require.NoError(t, err) + assert.Len(t, versionInfo, 1) + assert.Equal(t, 1, mock.calls, "empty-string Marker must terminate after page 1") +} + +// alwaysNextPageRDSMock returns pages each carrying a non-nil non-empty Marker. +type alwaysNextPageRDSMock struct { + calls int +} + +func (m *alwaysNextPageRDSMock) DescribeDBMajorEngineVersions( + _ context.Context, + _ *awsrds.DescribeDBMajorEngineVersionsInput, + _ ...func(*awsrds.Options), +) (*awsrds.DescribeDBMajorEngineVersionsOutput, error) { + m.calls++ + return &awsrds.DescribeDBMajorEngineVersionsOutput{ + DBMajorEngineVersions: []rdstypes.DBMajorEngineVersion{ + rdsMajorVersion("mysql", fmt.Sprintf("5.%d", m.calls)), + }, + Marker: aws.String(fmt.Sprintf("tok%d", m.calls)), + }, nil +} + +// TestFetchMajorEngineVersionsForEngine_PaginationCapError asserts that +// exceeding maxEngineVersionPages returns a diagnostic error (issue #692). +func TestFetchMajorEngineVersionsForEngine_PaginationCapError(t *testing.T) { + mock := &alwaysNextPageRDSMock{} + versionInfo := make(map[string]MajorEngineVersionInfo) + + err := fetchMajorEngineVersionsForEngine(context.Background(), mock, "mysql", versionInfo) + require.Error(t, err) + assert.Contains(t, err.Error(), "pagination cap reached") + assert.Equal(t, maxEngineVersionPages, mock.calls, + "must stop exactly at the cap") +} diff --git a/cmd/multi_service_engine_versions_test.go b/cmd/multi_service_engine_versions_test.go new file mode 100644 index 000000000..d1273570b --- /dev/null +++ b/cmd/multi_service_engine_versions_test.go @@ -0,0 +1,797 @@ +package main + +import ( + "context" + "testing" + "time" + + "github.com/LeanerCloud/CUDly/pkg/common" + rdstypes "github.com/aws/aws-sdk-go-v2/service/rds/types" + "github.com/stretchr/testify/assert" +) + +func TestAdjustRecommendationForExcludedVersions(t *testing.T) { + tests := []struct { + versionInfo map[string]MajorEngineVersionInfo + instanceVersions map[string][]InstanceEngineVersion + name string + recommendation common.Recommendation + expectedCount int + expectedAdjusted bool + }{ + { + name: "No running instances - recommendation unchanged", + recommendation: common.Recommendation{ + Service: common.ServiceRDS, + Region: "us-east-1", + ResourceType: "db.r5.large", + Count: 10, + Details: &common.DatabaseDetails{ + Engine: "Aurora MySQL", + }, + }, + versionInfo: createTestVersionInfo(), + instanceVersions: map[string][]InstanceEngineVersion{}, + expectedCount: 10, + expectedAdjusted: false, + }, + { + name: "Exclude 1 MySQL 5.7 instance in extended support", + recommendation: common.Recommendation{ + Service: common.ServiceRDS, + Region: "us-east-1", + ResourceType: "db.r5.large", + Count: 10, + Details: &common.DatabaseDetails{ + Engine: "Aurora MySQL", + }, + }, + versionInfo: createTestVersionInfo(), + instanceVersions: map[string][]InstanceEngineVersion{ + "db.r5.large": { + {Engine: "aurora-mysql", EngineVersion: "5.7.mysql_aurora.2.11.1", InstanceClass: "db.r5.large", Region: "us-east-1"}, + {Engine: "aurora-mysql", EngineVersion: "8.0.mysql_aurora.3.04.0", InstanceClass: "db.r5.large", Region: "us-east-1"}, + {Engine: "aurora-mysql", EngineVersion: "8.0.mysql_aurora.3.04.0", InstanceClass: "db.r5.large", Region: "us-east-1"}, + }, + }, + expectedCount: 9, // 10 - 1 MySQL 5.7 instance in extended support + expectedAdjusted: true, + }, + { + name: "Exclude all MySQL 5.7 instances in extended support", + recommendation: common.Recommendation{ + Service: common.ServiceRDS, + Region: "eu-west-2", + ResourceType: "db.t3.small", + Count: 2, + Details: &common.DatabaseDetails{ + Engine: "Aurora MySQL", + }, + }, + versionInfo: createTestVersionInfo(), + instanceVersions: map[string][]InstanceEngineVersion{ + "db.t3.small": { + {Engine: "aurora-mysql", EngineVersion: "5.7.mysql_aurora.2.11.2", InstanceClass: "db.t3.small", Region: "eu-west-2"}, + {Engine: "aurora-mysql", EngineVersion: "5.7.mysql_aurora.2.11.2", InstanceClass: "db.t3.small", Region: "eu-west-2"}, + }, + }, + expectedCount: 0, // All excluded (both in extended support) + expectedAdjusted: true, + }, + { + name: "Different engine - no adjustment", + recommendation: common.Recommendation{ + Service: common.ServiceRDS, + Region: "us-east-1", + ResourceType: "db.r5.large", + Count: 5, + Details: &common.DatabaseDetails{ + Engine: "Aurora PostgreSQL", + }, + }, + versionInfo: createTestVersionInfo(), + instanceVersions: map[string][]InstanceEngineVersion{ + "db.r5.large": { + {Engine: "aurora-mysql", EngineVersion: "5.7.mysql_aurora.2.11.1", InstanceClass: "db.r5.large", Region: "us-east-1"}, + }, + }, + expectedCount: 5, // Different engine, no adjustment + expectedAdjusted: false, + }, + { + name: "Different region - no adjustment", + recommendation: common.Recommendation{ + Service: common.ServiceRDS, + Region: "us-east-1", + ResourceType: "db.r5.large", + Count: 5, + Details: &common.DatabaseDetails{ + Engine: "Aurora MySQL", + }, + }, + versionInfo: createTestVersionInfo(), + instanceVersions: map[string][]InstanceEngineVersion{ + "db.r5.large": { + {Engine: "aurora-mysql", EngineVersion: "5.7.mysql_aurora.2.11.1", InstanceClass: "db.r5.large", Region: "eu-west-2"}, + }, + }, + expectedCount: 5, // Different region, no adjustment + expectedAdjusted: false, + }, + { + name: "MySQL (not Aurora) with standard mysql engine name", + recommendation: common.Recommendation{ + Service: common.ServiceRDS, + Region: "eu-west-2", + ResourceType: "db.r5.4xlarge", + Count: 8, + Details: &common.DatabaseDetails{ + Engine: "MySQL", + }, + }, + versionInfo: map[string]MajorEngineVersionInfo{ + "mysql:5.7": { + Engine: "mysql", + MajorEngineVersion: "5.7", + SupportedEngineLifecycles: []EngineLifecycleInfo{ + { + LifecycleSupportName: string(rdstypes.LifecycleSupportNameOpenSourceRdsExtendedSupport), + LifecycleSupportStartDate: time.Now().AddDate(0, -6, 0), + LifecycleSupportEndDate: time.Now().AddDate(3, 0, 0), + }, + }, + }, + }, + instanceVersions: map[string][]InstanceEngineVersion{ + "db.r5.4xlarge": { + {Engine: "mysql", EngineVersion: "5.7.44", InstanceClass: "db.r5.4xlarge", Region: "eu-west-2"}, + {Engine: "mysql", EngineVersion: "8.0.35", InstanceClass: "db.r5.4xlarge", Region: "eu-west-2"}, + }, + }, + expectedCount: 7, // 8 - 1 MySQL 5.7 instance in extended support + expectedAdjusted: true, + }, + { + name: "Engine name normalization - spaces vs hyphens", + recommendation: common.Recommendation{ + Service: common.ServiceRDS, + Region: "us-west-2", + ResourceType: "db.r6g.large", + Count: 3, + Details: &common.DatabaseDetails{ + Engine: "Aurora MySQL", // Space in name + }, + }, + versionInfo: createTestVersionInfo(), + instanceVersions: map[string][]InstanceEngineVersion{ + "db.r6g.large": { + {Engine: "aurora-mysql", EngineVersion: "5.7.mysql_aurora.2.12.0", InstanceClass: "db.r6g.large", Region: "us-west-2"}, // Hyphen in name + }, + }, + expectedCount: 2, // Should match despite space vs hyphen + expectedAdjusted: true, + }, + } + + for _, tt := range tests { + t.Run(tt.name, func(t *testing.T) { + result := adjustRecommendationForExcludedVersions(tt.recommendation, tt.instanceVersions, tt.versionInfo) + + assert.Equal(t, tt.expectedCount, result.Count, "Instance count mismatch") + + if tt.expectedAdjusted { + assert.NotEqual(t, tt.recommendation.Count, result.Count, "Count should have been adjusted") + } else { + assert.Equal(t, tt.recommendation.Count, result.Count, "Count should not have been adjusted") + } + }) + } +} + +func TestAdjustRecommendationForExcludedVersions_MultipleVersionsInExtendedSupport(t *testing.T) { + recommendation := common.Recommendation{ + Service: common.ServiceRDS, + Region: "us-east-1", + ResourceType: "db.r5.large", + Count: 10, + Details: &common.DatabaseDetails{ + Engine: "Aurora MySQL", + }, + } + + instanceVersions := map[string][]InstanceEngineVersion{ + "db.r5.large": { + {Engine: "aurora-mysql", EngineVersion: "5.6.mysql_aurora.1.22.5", InstanceClass: "db.r5.large", Region: "us-east-1"}, + {Engine: "aurora-mysql", EngineVersion: "5.7.mysql_aurora.2.11.1", InstanceClass: "db.r5.large", Region: "us-east-1"}, + {Engine: "aurora-mysql", EngineVersion: "8.0.mysql_aurora.3.04.0", InstanceClass: "db.r5.large", Region: "us-east-1"}, + }, + } + + // Version info with both 5.6 and 5.7 in extended support + versionInfo := map[string]MajorEngineVersionInfo{ + "aurora-mysql:5.6": { + Engine: "aurora-mysql", + MajorEngineVersion: "5.6", + SupportedEngineLifecycles: []EngineLifecycleInfo{ + { + LifecycleSupportName: string(rdstypes.LifecycleSupportNameOpenSourceRdsExtendedSupport), + LifecycleSupportStartDate: time.Now().AddDate(0, -12, 0), + LifecycleSupportEndDate: time.Now().AddDate(2, 0, 0), + }, + }, + }, + "aurora-mysql:5.7": { + Engine: "aurora-mysql", + MajorEngineVersion: "5.7", + SupportedEngineLifecycles: []EngineLifecycleInfo{ + { + LifecycleSupportName: string(rdstypes.LifecycleSupportNameOpenSourceRdsExtendedSupport), + LifecycleSupportStartDate: time.Now().AddDate(0, -6, 0), + LifecycleSupportEndDate: time.Now().AddDate(3, 0, 0), + }, + }, + }, + } + + result := adjustRecommendationForExcludedVersions(recommendation, instanceVersions, versionInfo) + + assert.Equal(t, 8, result.Count, "Should exclude 2 instances (5.6 and 5.7 both in extended support)") +} + +func TestAdjustRecommendationForExcludedVersions_NonRDSService(t *testing.T) { + recommendation := common.Recommendation{ + Service: common.ServiceEC2, + Region: "us-east-1", + ResourceType: "m5.large", + Count: 5, + Details: nil, // Not RDS + } + + instanceVersions := map[string][]InstanceEngineVersion{} + versionInfo := createTestVersionInfo() + + result := adjustRecommendationForExcludedVersions(recommendation, instanceVersions, versionInfo) + + assert.Equal(t, 5, result.Count, "Non-RDS services should not be adjusted") +} + +// TestAdjustRecommendationForExcludedVersions_TypedNilDetails pins the typed-nil +// guard on the *common.DatabaseDetails case. A nil interface is caught by the +// caller's own nil checks, but a (*common.DatabaseDetails)(nil) stored in the +// interface reaches the type switch, and reading details.Engine there panics. +// instanceVersions must contain an entry for the rec's ResourceType so the +// function gets past its early "no running instances" return and actually +// reaches the switch. +func TestAdjustRecommendationForExcludedVersions_TypedNilDetails(t *testing.T) { + recommendation := common.Recommendation{ + Service: common.ServiceRDS, + Region: "us-east-1", + ResourceType: "db.r5.large", + Count: 7, + Details: (*common.DatabaseDetails)(nil), + } + + instanceVersions := map[string][]InstanceEngineVersion{ + "db.r5.large": { + {Engine: "aurora-mysql", EngineVersion: "5.7.mysql_aurora.2.11.1", InstanceClass: "db.r5.large", Region: "us-east-1"}, + }, + } + + assert.NotPanics(t, func() { + result := adjustRecommendationForExcludedVersions(recommendation, instanceVersions, createTestVersionInfo()) + assert.Equal(t, 7, result.Count, "typed-nil Details must pass through unadjusted") + }) +} + +func TestExtractMajorVersion_Additional(t *testing.T) { + tests := []struct { + name string + engine string + version string + expected string + }{ + { + name: "MySQL 5.7.44 extracts 5.7", + engine: "mysql", + version: "5.7.44", + expected: "5.7", + }, + { + name: "MySQL 8.0.35 extracts 8.0", + engine: "mysql", + version: "8.0.35", + expected: "8.0", + }, + { + name: "PostgreSQL 13.10 extracts 13.10", + engine: "postgres", + version: "13.10", + expected: "13.10", + }, + { + name: "PostgreSQL 15.4 extracts 15.4", + engine: "postgres", + version: "15.4", + expected: "15.4", + }, + { + name: "Aurora MySQL compatible 5.7.mysql_aurora.2.11.3", + engine: "aurora-mysql", + version: "5.7.mysql_aurora.2.11.3", + expected: "5.7", + }, + { + name: "Aurora PostgreSQL 14.6", + engine: "aurora-postgresql", + version: "14.6", + expected: "14.6", + }, + { + name: "Empty version", + engine: "mysql", + version: "", + expected: "", + }, + } + + for _, tt := range tests { + t.Run(tt.name, func(t *testing.T) { + result := extractMajorVersion(tt.engine, tt.version) + assert.Equal(t, tt.expected, result) + }) + } +} + +// Comprehensive tests for extractMajorVersion function. +func TestExtractMajorVersion_Comprehensive(t *testing.T) { + tests := []struct { + name string + engine string + version string + expected string + }{ + // Aurora MySQL special formats + { + name: "Aurora MySQL 2.x format (MySQL 5.7 compatible)", + engine: "aurora-mysql", + version: "mysql_aurora.2.11.1", + expected: "5.7", + }, + { + name: "Aurora MySQL 3.x format (MySQL 8.0 compatible)", + engine: "aurora-mysql", + version: "mysql_aurora.3.04.0", + expected: "8.0", + }, + { + name: "Aurora MySQL 2.x with full version", + engine: "aurora-mysql", + version: "5.7.mysql_aurora.2.12.0", + expected: "5.7", + }, + { + name: "Aurora MySQL 3.x with full version", + engine: "aurora-mysql", + version: "8.0.mysql_aurora.3.05.2", + expected: "8.0", + }, + { + name: "Aurora MySQL with only major.minor", + engine: "aurora-mysql", + version: "5.7", + expected: "5.7", + }, + { + name: "Aurora MySQL with only major.minor v8", + engine: "aurora-mysql", + version: "8.0", + expected: "8.0", + }, + { + name: "Aurora MySQL engine name with spaces", + engine: "Aurora MySQL", + version: "mysql_aurora.2.11.1", + expected: "5.7", + }, + { + name: "Aurora MySQL engine name normalized", + engine: "AuroraMYSQL", + version: "mysql_aurora.3.04.0", + expected: "8.0", + }, + + // Standard MySQL versions + { + name: "MySQL 5.6.x", + engine: "mysql", + version: "5.6.51", + expected: "5.6", + }, + { + name: "MySQL 5.7.x", + engine: "mysql", + version: "5.7.40", + expected: "5.7", + }, + { + name: "MySQL 8.0.x", + engine: "mysql", + version: "8.0.33", + expected: "8.0", + }, + { + name: "MySQL with only major.minor", + engine: "mysql", + version: "5.7", + expected: "5.7", + }, + + // PostgreSQL versions + { + name: "PostgreSQL 11.x", + engine: "postgres", + version: "11.19", + expected: "11.19", + }, + { + name: "PostgreSQL 12.x", + engine: "postgres", + version: "12.15", + expected: "12.15", + }, + { + name: "PostgreSQL 13.x", + engine: "postgres", + version: "13.11", + expected: "13.11", + }, + { + name: "PostgreSQL 14.x", + engine: "postgres", + version: "14.8", + expected: "14.8", + }, + { + name: "PostgreSQL 15.x", + engine: "postgres", + version: "15.3", + expected: "15.3", + }, + + // Aurora PostgreSQL versions + { + name: "Aurora PostgreSQL 11.x", + engine: "aurora-postgresql", + version: "11.18", + expected: "11.18", + }, + { + name: "Aurora PostgreSQL 13.x", + engine: "aurora-postgresql", + version: "13.10", + expected: "13.10", + }, + { + name: "Aurora PostgreSQL 14.x", + engine: "aurora-postgresql", + version: "14.7", + expected: "14.7", + }, + + // Edge cases + { + name: "Version with only major number", + engine: "mysql", + version: "8", + expected: "8", + }, + { + name: "Version with patch containing letters", + engine: "mysql", + version: "5.7.40a", + expected: "5.7", + }, + { + name: "Version with non-numeric minor (extracts numeric part)", + engine: "mysql", + version: "8.0rc1", + expected: "8.0", + }, + { + name: "Empty version string", + engine: "mysql", + version: "", + expected: "", + }, + { + name: "Version with extra dots", + engine: "postgres", + version: "13.10.1.2", + expected: "13.10", + }, + { + name: "Engine name with hyphens", + engine: "aurora-mysql", + version: "5.7.44", + expected: "5.7", + }, + } + + for _, tt := range tests { + t.Run(tt.name, func(t *testing.T) { + result := extractMajorVersion(tt.engine, tt.version) + assert.Equal(t, tt.expected, result, "extractMajorVersion(%q, %q) should return %q", tt.engine, tt.version, tt.expected) + }) + } +} + +func TestIsInExtendedSupport(t *testing.T) { + now := time.Now() + pastDate := now.AddDate(0, -6, 0) + futureDate := now.AddDate(3, 0, 0) + + tests := []struct { + versionInfo map[string]MajorEngineVersionInfo + name string + engine string + version string + expected bool + }{ + { + name: "Version in extended support", + engine: "aurora-mysql", + version: "5.7.mysql_aurora.2.11.1", + versionInfo: map[string]MajorEngineVersionInfo{ + "aurora-mysql:5.7": { + Engine: "aurora-mysql", + MajorEngineVersion: "5.7", + SupportedEngineLifecycles: []EngineLifecycleInfo{ + { + LifecycleSupportName: string(rdstypes.LifecycleSupportNameOpenSourceRdsExtendedSupport), + LifecycleSupportStartDate: pastDate, + LifecycleSupportEndDate: futureDate, + }, + }, + }, + }, + expected: true, + }, + { + name: "Version not in extended support - still in standard support", + engine: "aurora-mysql", + version: "8.0.mysql_aurora.3.04.0", + versionInfo: map[string]MajorEngineVersionInfo{ + "aurora-mysql:8.0": { + Engine: "aurora-mysql", + MajorEngineVersion: "8.0", + SupportedEngineLifecycles: []EngineLifecycleInfo{ + { + LifecycleSupportName: string(rdstypes.LifecycleSupportNameOpenSourceRdsStandardSupport), + LifecycleSupportStartDate: now.AddDate(-2, 0, 0), + LifecycleSupportEndDate: futureDate, + }, + }, + }, + }, + expected: false, + }, + { + name: "Version info not found", + engine: "mysql", + version: "5.7.44", + versionInfo: map[string]MajorEngineVersionInfo{ + "postgres:13": { + Engine: "postgres", + MajorEngineVersion: "13", + }, + }, + expected: false, + }, + { + name: "Empty version info", + engine: "mysql", + version: "5.7.44", + versionInfo: map[string]MajorEngineVersionInfo{}, + expected: false, + }, + { + name: "Extended support not started yet", + engine: "mysql", + version: "5.7.44", + versionInfo: map[string]MajorEngineVersionInfo{ + "mysql:5.7": { + Engine: "mysql", + MajorEngineVersion: "5.7", + SupportedEngineLifecycles: []EngineLifecycleInfo{ + { + LifecycleSupportName: string(rdstypes.LifecycleSupportNameOpenSourceRdsExtendedSupport), + LifecycleSupportStartDate: futureDate, + LifecycleSupportEndDate: futureDate.AddDate(1, 0, 0), + }, + }, + }, + }, + expected: false, + }, + { + name: "Extended support started on current date", + engine: "postgres", + version: "11.19", + versionInfo: map[string]MajorEngineVersionInfo{ + "postgres:11.19": { + Engine: "postgres", + MajorEngineVersion: "11.19", + SupportedEngineLifecycles: []EngineLifecycleInfo{ + { + LifecycleSupportName: string(rdstypes.LifecycleSupportNameOpenSourceRdsExtendedSupport), + LifecycleSupportStartDate: now, + LifecycleSupportEndDate: futureDate, + }, + }, + }, + }, + expected: true, + }, + { + name: "Extended support already ended", + engine: "mysql", + version: "5.7.44", + versionInfo: map[string]MajorEngineVersionInfo{ + "mysql:5.7": { + Engine: "mysql", + MajorEngineVersion: "5.7", + SupportedEngineLifecycles: []EngineLifecycleInfo{ + { + LifecycleSupportName: string(rdstypes.LifecycleSupportNameOpenSourceRdsExtendedSupport), + LifecycleSupportStartDate: now.AddDate(-3, 0, 0), + LifecycleSupportEndDate: pastDate, + }, + }, + }, + }, + expected: false, + }, + { + name: "Zero end date treated as open-ended", + engine: "mysql", + version: "5.7.44", + versionInfo: map[string]MajorEngineVersionInfo{ + "mysql:5.7": { + Engine: "mysql", + MajorEngineVersion: "5.7", + SupportedEngineLifecycles: []EngineLifecycleInfo{ + { + LifecycleSupportName: string(rdstypes.LifecycleSupportNameOpenSourceRdsExtendedSupport), + LifecycleSupportStartDate: pastDate, + }, + }, + }, + }, + expected: true, + }, + { + name: "Engine name normalization with spaces", + engine: "Aurora MySQL", + version: "5.7.mysql_aurora.2.11.1", + versionInfo: map[string]MajorEngineVersionInfo{ + "auroramysql:5.7": { + Engine: "auroramysql", + MajorEngineVersion: "5.7", + SupportedEngineLifecycles: []EngineLifecycleInfo{ + { + LifecycleSupportName: string(rdstypes.LifecycleSupportNameOpenSourceRdsExtendedSupport), + LifecycleSupportStartDate: pastDate, + LifecycleSupportEndDate: futureDate, + }, + }, + }, + }, + expected: true, + }, + } + + for _, tt := range tests { + t.Run(tt.name, func(t *testing.T) { + result := isInExtendedSupport(tt.engine, tt.version, tt.versionInfo) + assert.Equal(t, tt.expected, result) + }) + } +} + +func TestQueryMajorEngineVersions_ErrorHandling(t *testing.T) { + // This test validates the error handling logic when AWS config fails + ctx := context.Background() + + tests := []struct { + name string + cfg Config + wantErr bool + }{ + { + name: "Empty profile - should use default credentials", + cfg: Config{ + Profile: "", + ValidationProfile: "", + }, + wantErr: false, // Will fail with real AWS but tests error path exists + }, + { + name: "With validation profile", + cfg: Config{ + Profile: "default", + ValidationProfile: "validation-profile", + }, + wantErr: false, + }, + { + name: "Fallback to main profile", + cfg: Config{ + Profile: "main-profile", + ValidationProfile: "", + }, + wantErr: false, + }, + } + + for _, tt := range tests { + t.Run(tt.name, func(t *testing.T) { + // This will likely fail in test environment without real AWS credentials + // but it validates the function signature and basic logic paths + _, err := queryMajorEngineVersions(ctx, tt.cfg) + // We expect an error in test environment (no real AWS creds) + // The important thing is that the function doesn't panic + if err != nil { + assert.Contains(t, err.Error(), "failed to load AWS config") + } + }) + } +} + +func TestQueryRunningInstanceEngineVersions_ErrorHandling(t *testing.T) { + // This test validates the error handling logic + ctx := context.Background() + + tests := []struct { + name string + cfg Config + wantErr bool + }{ + { + name: "Empty profile - should use default credentials", + cfg: Config{ + Profile: "", + ValidationProfile: "", + }, + wantErr: false, + }, + { + name: "With validation profile", + cfg: Config{ + Profile: "default", + ValidationProfile: "validation-profile", + }, + wantErr: false, + }, + { + name: "Fallback to main profile", + cfg: Config{ + Profile: "main-profile", + ValidationProfile: "", + }, + wantErr: false, + }, + } + + for _, tt := range tests { + t.Run(tt.name, func(t *testing.T) { + // This will likely fail in test environment without real AWS credentials + // but it validates the function signature and basic logic paths + _, err := queryRunningInstanceEngineVersions(ctx, tt.cfg) + // We expect an error in test environment (no real AWS creds) + // The important thing is that the function doesn't panic + if err != nil { + assert.Contains(t, err.Error(), "failed to") + } + }) + } +} diff --git a/cmd/multi_service_filters.go b/cmd/multi_service_filters.go new file mode 100644 index 000000000..66eb43e7d --- /dev/null +++ b/cmd/multi_service_filters.go @@ -0,0 +1,206 @@ +package main + +import ( + "log" + "strings" + + "github.com/LeanerCloud/CUDly/pkg/common" + "github.com/LeanerCloud/CUDly/pkg/recfilter" + awsprovider "github.com/LeanerCloud/CUDly/providers/aws" +) + +// filtersFromConfig maps the CLI Config's dimension-filter and min-pool-size +// fields onto a recfilter.Filters value. Account filtering stays cmd-only +// (see shouldIncludeAccount) so it is not part of recfilter.Filters. +func filtersFromConfig(cfg *Config) recfilter.Filters { + return recfilter.Filters{ + IncludeRegions: cfg.IncludeRegions, + ExcludeRegions: cfg.ExcludeRegions, + IncludeInstanceTypes: cfg.IncludeInstanceTypes, + ExcludeInstanceTypes: cfg.ExcludeInstanceTypes, + IncludeEngines: cfg.IncludeEngines, + ExcludeEngines: cfg.ExcludeEngines, + MinPoolSize: cfg.MinPoolSize, + } +} + +// applyFilters applies region, instance type, engine, and engine version filters to recommendations. +// currentRegion is the region being processed in the current loop iteration; if non-empty, only +// recommendations for that region are included. +// drops accumulates per-reason drop counts for the end-of-run summary; pass nil to skip tracking. +func applyFilters(recs []common.Recommendation, cfg *Config, instanceVersions map[string][]InstanceEngineVersion, versionInfo map[string]MajorEngineVersionInfo, currentRegion string, drops *common.DropSummary) []common.Recommendation { + survivors := filtersFromConfig(cfg).ApplyMinPoolSize(recs, log.Printf, drops) + + var filtered []common.Recommendation + for i := range survivors { + adjusted, include, dropReason := processRecommendation(&survivors[i], cfg, instanceVersions, versionInfo, currentRegion) + if include { + filtered = append(filtered, adjusted) + } else if dropReason != "" { + drops.Add(dropReason, 1) + } + } + + return filtered +} + +// processRecommendation applies all filters to a recommendation and returns +// (adjusted, include, dropReason). dropReason is non-empty only when +// include is false and the drop is worth surfacing in the end-of-run +// summary (dimension-filter mismatches such as region/account/engine are +// expected exclusions and are not counted). The flat boolean-filter checks +// are delegated to passesDimensionFilters to keep this function under +// gocyclo's complexity threshold. +func processRecommendation(rec *common.Recommendation, cfg *Config, instanceVersions map[string][]InstanceEngineVersion, versionInfo map[string]MajorEngineVersionInfo, currentRegion string) (result common.Recommendation, include bool, dropReason string) { + // Filter to only recommendations for the current region being processed. + // This prevents duplicating recommendations across all regions. + // Skip for Savings Plans (account-level, not regional). No drop reason: + // same rec will be returned by its own region's pass. + if currentRegion != "" && rec.Region != currentRegion && !common.IsSavingsPlan(rec.Service) { + return *rec, false, "" + } + + if !passesDimensionFilters(rec, cfg) { + // Dimension mismatches (region/account/engine/instance-type) are expected + // operator-scoping choices, not drops worth surfacing in the summary. + // --min-pool-size drops are counted separately in applyFilters before + // this function runs, so no drop reason is surfaced here. + return *rec, false, "" + } + + // Apply engine version filters - adjust instance count by subtracting extended support versions. + if !cfg.IncludeExtendedSupport { + adjusted := adjustRecommendationForExcludedVersions(*rec, instanceVersions, versionInfo) + // Skip if all instances were excluded (count reduced to 0). + if adjusted.Count <= 0 { + return adjusted, false, common.DropExtendedSupport + } + return adjusted, true, "" + } + + return *rec, true, "" +} + +// passesDimensionFilters runs the stateless include/exclude checks on +// region, instance type, engine, and account. Returns false on +// the first failing filter. Split out of processRecommendation to keep +// each function's cyclomatic complexity under the gocyclo limit; the +// dimension filters here are pure functions of rec + cfg with no side +// effects. Pool-size filtering is handled with logging in applyFilters. +// +// Region and account stay here rather than moving into recfilter: the +// region-agnostic handling (#1881) needs providers/aws, which the pkg module +// cannot import, and account filtering is name-substring matching backed by +// AccountAliasCache. Only the instance-type and engine checks are portable. +func passesDimensionFilters(rec *common.Recommendation, cfg *Config) bool { + if !shouldIncludeRecommendationRegion(rec, cfg) { + return false + } + if !shouldIncludeInstanceType(rec.ResourceType, cfg) { + return false + } + if !shouldIncludeEngine(rec, cfg) { + return false + } + return shouldIncludeAccount(rec.AccountName, cfg) +} + +// shouldIncludePoolSize checks if a recommendation's pool size meets cfg.MinPoolSize. +// Thin wrapper over recfilter.Filters.IncludesPoolSize; kept so cmd's existing call sites +// and tests are unchanged. +func shouldIncludePoolSize(rec *common.Recommendation, cfg *Config) bool { + return filtersFromConfig(cfg).IncludesPoolSize(rec) +} + +// shouldIncludeRecommendationRegion applies the region filters to a whole +// recommendation rather than to its bare Region field. Savings Plans +// recommendations leave the top-level Region empty and carry the +// EC2Instance-scoped region in Details instead, so matching on rec.Region +// alone dropped every SP recommendation under --include-regions and leaked +// region-scoped ones past --exclude-regions (#1582). +// +// Both predicates are the provider package's, which already filters AWS +// recommendations on these semantics, rather than a second implementation. +// Recommendations from other providers are unaffected: both predicates are +// gated on common.CommitmentSavingsPlan and fall through to the plain +// rec.Region comparison for everything else. +func shouldIncludeRecommendationRegion(rec *common.Recommendation, cfg *Config) bool { + if awsprovider.IsRegionAgnostic(*rec) { + return true + } + return shouldIncludeRegion(awsprovider.EffectiveRegion(*rec), cfg) +} + +// shouldIncludeRegion checks if a region should be included based on filters. +// Thin wrapper over recfilter.Filters.IncludesRegion; kept so cmd's existing call sites +// and tests are unchanged. +func shouldIncludeRegion(region string, cfg *Config) bool { + return filtersFromConfig(cfg).IncludesRegion(region) +} + +// shouldIncludeInstanceType checks if an instance type should be included based on filters. +// Thin wrapper over recfilter.Filters.IncludesInstanceType; kept so cmd's existing call sites +// and tests are unchanged. +func shouldIncludeInstanceType(instanceType string, cfg *Config) bool { + return filtersFromConfig(cfg).IncludesInstanceType(instanceType) +} + +// shouldIncludeEngine checks if a recommendation should be included based on engine filters. +// Thin wrapper over recfilter.Filters.IncludesEngine; kept so cmd's existing call sites +// and tests are unchanged. +func shouldIncludeEngine(rec *common.Recommendation, cfg *Config) bool { + return filtersFromConfig(cfg).IncludesEngine(rec) +} + +// shouldIncludeAccount checks if an account should be included based on filters. +func shouldIncludeAccount(accountName string, cfg *Config) bool { + // If account name is empty and there are filters, skip it (unless include list is empty). + if accountName == "" { + return len(cfg.IncludeAccounts) == 0 && len(cfg.ExcludeAccounts) == 0 + } + + accountLower := strings.ToLower(accountName) + + // Check include list. + if !checkIncludeList(accountLower, cfg.IncludeAccounts) { + return false + } + + // Check exclude list. + if checkExcludeList(accountLower, cfg.ExcludeAccounts) { + return false + } + + return true +} + +// checkIncludeList checks if an account matches the include filters. +func checkIncludeList(accountLower string, includeAccounts []string) bool { + if len(includeAccounts) == 0 { + return true + } + + for _, filter := range includeAccounts { + if accountMatchesFilter(accountLower, filter) { + return true + } + } + + return false +} + +// checkExcludeList checks if an account matches any exclude filters. +func checkExcludeList(accountLower string, excludeAccounts []string) bool { + for _, filter := range excludeAccounts { + if accountMatchesFilter(accountLower, filter) { + return true + } + } + return false +} + +// accountMatchesFilter checks if an account matches a filter pattern (exact or substring match). +func accountMatchesFilter(accountLower, filter string) bool { + filterLower := strings.ToLower(filter) + return filterLower == accountLower || strings.Contains(accountLower, filterLower) +} diff --git a/cmd/multi_service_filters_test.go b/cmd/multi_service_filters_test.go new file mode 100644 index 000000000..58a666130 --- /dev/null +++ b/cmd/multi_service_filters_test.go @@ -0,0 +1,755 @@ +package main + +import ( + "fmt" + "log" + "testing" + "time" + + "github.com/LeanerCloud/CUDly/pkg/common" + "github.com/stretchr/testify/assert" + "github.com/stretchr/testify/require" +) + +func TestApplyFilters(t *testing.T) { + // Save original values + origCfg := toolCfg + + // Restore after test + defer func() { + toolCfg = origCfg + }() + + tests := []struct { + name string + recommendations []common.Recommendation + includeRegions []string + excludeRegions []string + includeInstanceTypes []string + excludeInstanceTypes []string + expectedCount int + }{ + { + name: "No filters - all pass through", + recommendations: []common.Recommendation{ + {Region: "us-east-1", ResourceType: "db.t3.micro", Count: 1}, + {Region: "us-west-2", ResourceType: "db.t3.small", Count: 1}, + }, + includeRegions: []string{}, + excludeRegions: []string{}, + includeInstanceTypes: []string{}, + excludeInstanceTypes: []string{}, + expectedCount: 2, + }, + { + name: "Include specific regions only", + recommendations: []common.Recommendation{ + {Region: "us-east-1", ResourceType: "db.t3.micro", Count: 1}, + {Region: "us-west-2", ResourceType: "db.t3.small", Count: 1}, + {Region: "eu-west-1", ResourceType: "db.t3.medium", Count: 1}, + }, + includeRegions: []string{"us-east-1", "eu-west-1"}, + excludeRegions: []string{}, + includeInstanceTypes: []string{}, + excludeInstanceTypes: []string{}, + expectedCount: 2, + }, + { + name: "Exclude specific regions", + recommendations: []common.Recommendation{ + {Region: "us-east-1", ResourceType: "db.t3.micro", Count: 1}, + {Region: "us-west-2", ResourceType: "db.t3.small", Count: 1}, + }, + includeRegions: []string{}, + excludeRegions: []string{"us-west-2"}, + includeInstanceTypes: []string{}, + excludeInstanceTypes: []string{}, + expectedCount: 1, + }, + { + name: "Include specific instance types", + recommendations: []common.Recommendation{ + {Region: "us-east-1", ResourceType: "db.t3.micro", Count: 1}, + {Region: "us-west-2", ResourceType: "db.t3.small", Count: 1}, + {Region: "eu-west-1", ResourceType: "db.t3.micro", Count: 1}, + }, + includeRegions: []string{}, + excludeRegions: []string{}, + includeInstanceTypes: []string{"db.t3.micro"}, + excludeInstanceTypes: []string{}, + expectedCount: 2, + }, + { + name: "Combined filters", + recommendations: []common.Recommendation{ + {Region: "us-east-1", ResourceType: "db.t3.micro", Count: 1}, + {Region: "us-east-1", ResourceType: "db.t3.small", Count: 1}, + {Region: "us-west-2", ResourceType: "db.t3.micro", Count: 1}, + }, + includeRegions: []string{}, + excludeRegions: []string{}, + includeInstanceTypes: []string{}, + excludeInstanceTypes: []string{"db.t3.micro"}, + expectedCount: 1, // Only us-east-1 with db.t3.small + }, + { + name: "Include and exclude same instance type - exclude takes precedence", + recommendations: []common.Recommendation{ + {Region: "us-east-1", ResourceType: "db.t3.micro", Count: 1}, + {Region: "us-west-2", ResourceType: "db.t3.small", Count: 1}, + }, + includeRegions: []string{}, + excludeRegions: []string{}, + includeInstanceTypes: []string{"db.t3.micro", "db.t3.small"}, + excludeInstanceTypes: []string{"db.t3.micro"}, + expectedCount: 1, // db.t3.micro excluded, only db.t3.small remains + }, + } + + for _, tt := range tests { + t.Run(tt.name, func(t *testing.T) { + // Set toolCfg fields + toolCfg.IncludeRegions = tt.includeRegions + toolCfg.ExcludeRegions = tt.excludeRegions + toolCfg.IncludeInstanceTypes = tt.includeInstanceTypes + toolCfg.ExcludeInstanceTypes = tt.excludeInstanceTypes + + // Apply filters with Config (empty currentRegion for test) + result := applyFilters(tt.recommendations, &toolCfg, make(map[string][]InstanceEngineVersion), make(map[string]MajorEngineVersionInfo), "", nil) + + // Check count + assert.Equal(t, tt.expectedCount, len(result)) + }) + } +} + +func TestShouldIncludeRegion(t *testing.T) { + // Save original values + origCfg := toolCfg + + defer func() { + toolCfg = origCfg + }() + + tests := []struct { + name string + region string + includeRegions []string + excludeRegions []string + expected bool + }{ + { + name: "No filters - should include", + region: "us-east-1", + includeRegions: []string{}, + excludeRegions: []string{}, + expected: true, + }, + { + name: "In include list", + region: "us-east-1", + includeRegions: []string{"us-east-1", "us-west-2"}, + excludeRegions: []string{}, + expected: true, + }, + { + name: "Not in include list", + region: "eu-west-1", + includeRegions: []string{"us-east-1"}, + excludeRegions: []string{}, + expected: false, + }, + { + name: "In exclude list", + region: "us-east-1", + includeRegions: []string{}, + excludeRegions: []string{"us-east-1"}, + expected: false, + }, + } + + for _, tt := range tests { + t.Run(tt.name, func(t *testing.T) { + toolCfg.IncludeRegions = tt.includeRegions + toolCfg.ExcludeRegions = tt.excludeRegions + + result := shouldIncludeRegion(tt.region, &toolCfg) + assert.Equal(t, tt.expected, result) + }) + } +} + +func TestShouldIncludeInstanceType(t *testing.T) { + // Save original values + origCfg := toolCfg + + defer func() { + toolCfg = origCfg + }() + + tests := []struct { + name string + instanceType string + includeInstanceTypes []string + excludeInstanceTypes []string + expected bool + }{ + { + name: "No filters - should include", + instanceType: "db.t3.micro", + includeInstanceTypes: []string{}, + excludeInstanceTypes: []string{}, + expected: true, + }, + { + name: "In include list", + instanceType: "cache.t3.micro", + includeInstanceTypes: []string{"cache.t3.micro"}, + excludeInstanceTypes: []string{}, + expected: true, + }, + { + name: "In exclude list", + instanceType: "db.t3.large", + includeInstanceTypes: []string{}, + excludeInstanceTypes: []string{"db.t3.large"}, + expected: false, + }, + { + name: "Not in include list - excluded via whitelist", + instanceType: "db.r5.large", + includeInstanceTypes: []string{"db.t3.micro"}, + excludeInstanceTypes: []string{}, + expected: false, + }, + } + + for _, tt := range tests { + t.Run(tt.name, func(t *testing.T) { + toolCfg.IncludeInstanceTypes = tt.includeInstanceTypes + toolCfg.ExcludeInstanceTypes = tt.excludeInstanceTypes + + result := shouldIncludeInstanceType(tt.instanceType, &toolCfg) + assert.Equal(t, tt.expected, result) + }) + } +} + +func TestShouldIncludeEngine(t *testing.T) { + // Save original values + origCfg := toolCfg + + defer func() { + toolCfg = origCfg + }() + + tests := []struct { + name string + recommendation common.Recommendation + includeEngines []string + excludeEngines []string + expected bool + }{ + { + name: "ElastiCache Redis - no filters", + recommendation: common.Recommendation{ + Service: common.ServiceElastiCache, + Details: &common.CacheDetails{ + Engine: "redis", + }, + }, + includeEngines: []string{}, + excludeEngines: []string{}, + expected: true, + }, + { + name: "ElastiCache Redis - in include list", + recommendation: common.Recommendation{ + Service: common.ServiceElastiCache, + Details: &common.CacheDetails{ + Engine: "redis", + }, + }, + includeEngines: []string{"redis"}, + excludeEngines: []string{}, + expected: true, + }, + { + name: "ElastiCache Valkey - not in include list", + recommendation: common.Recommendation{ + Service: common.ServiceElastiCache, + Details: &common.CacheDetails{ + Engine: "valkey", + }, + }, + includeEngines: []string{"redis"}, + excludeEngines: []string{}, + expected: false, + }, + { + name: "ElastiCache Redis - in exclude list", + recommendation: common.Recommendation{ + Service: common.ServiceElastiCache, + Details: &common.CacheDetails{ + Engine: "redis", + }, + }, + includeEngines: []string{}, + excludeEngines: []string{"redis"}, + expected: false, + }, + { + name: "RDS with nil Details", + recommendation: common.Recommendation{ + Service: common.ServiceRDS, + Details: nil, + }, + includeEngines: []string{"mysql"}, + excludeEngines: []string{}, + expected: false, // nil Details with include list - exclude unknown engines + }, + { + name: "RDS with nil Details - no filters", + recommendation: common.Recommendation{ + Service: common.ServiceRDS, + Details: nil, + }, + includeEngines: []string{}, + excludeEngines: []string{}, + expected: true, // nil Details with no filters - include by default + }, + { + name: "RDS MySQL - with ServiceDetails", + recommendation: common.Recommendation{ + Service: common.ServiceRDS, + Details: &common.DatabaseDetails{ + Engine: "mysql", + }, + }, + includeEngines: []string{"mysql", "postgresql"}, + excludeEngines: []string{}, + expected: true, + }, + { + name: "Case insensitive matching", + recommendation: common.Recommendation{ + Service: common.ServiceElastiCache, + Details: &common.CacheDetails{ + Engine: "Redis", + }, + }, + includeEngines: []string{"REDIS"}, + excludeEngines: []string{}, + expected: true, + }, + } + + for _, tt := range tests { + t.Run(tt.name, func(t *testing.T) { + toolCfg.IncludeEngines = tt.includeEngines + toolCfg.ExcludeEngines = tt.excludeEngines + + result := shouldIncludeEngine(&tt.recommendation, &toolCfg) + assert.Equal(t, tt.expected, result) + }) + } +} + +func TestShouldIncludeAccount(t *testing.T) { + // Save original values + origCfg := toolCfg + + defer func() { + toolCfg = origCfg + }() + + tests := []struct { + name string + accountID string + includeAccounts []string + excludeAccounts []string + expected bool + }{ + { + name: "No filters - should include", + accountID: "123456789012", + includeAccounts: []string{}, + excludeAccounts: []string{}, + expected: true, + }, + { + name: "In include list", + accountID: "123456789012", + includeAccounts: []string{"123456789012", "210987654321"}, + excludeAccounts: []string{}, + expected: true, + }, + { + name: "Not in include list", + accountID: "999888777666", + includeAccounts: []string{"123456789012"}, + excludeAccounts: []string{}, + expected: false, + }, + { + name: "In exclude list", + accountID: "123456789012", + includeAccounts: []string{}, + excludeAccounts: []string{"123456789012"}, + expected: false, + }, + { + name: "Not in exclude list", + accountID: "999888777666", + includeAccounts: []string{}, + excludeAccounts: []string{"123456789012"}, + expected: true, + }, + } + + for _, tt := range tests { + t.Run(tt.name, func(t *testing.T) { + toolCfg.IncludeAccounts = tt.includeAccounts + toolCfg.ExcludeAccounts = tt.excludeAccounts + + result := shouldIncludeAccount(tt.accountID, &toolCfg) + assert.Equal(t, tt.expected, result) + }) + } +} + +func TestShouldIncludePoolSize(t *testing.T) { + tests := []struct { + name string + avg float64 + minPool float64 + expected bool + }{ + {"filter disabled (0)", 0.5, 0, true}, + {"avg=0 passes through", 0, 2.0, true}, + {"avg below threshold", 1.5, 2.0, false}, + {"avg equal to threshold", 2.0, 2.0, true}, + {"avg above threshold", 3.0, 2.0, true}, + } + + for _, tt := range tests { + t.Run(tt.name, func(t *testing.T) { + rec := common.Recommendation{AverageInstancesUsedPerHour: tt.avg} + cfg := Config{MinPoolSize: tt.minPool} + assert.Equal(t, tt.expected, shouldIncludePoolSize(&rec, &cfg)) + }) + } +} + +// TestApplyFilters_DropMinPoolSize verifies that a recommendation whose +// AverageInstancesUsedPerHour is below --min-pool-size is recorded in a +// non-nil DropSummary under the DropMinPoolSize category. If the +// drops.Add call for that path were removed, d.Total() would stay at 0 +// and the first assertion below would fail. +func TestApplyFilters_DropMinPoolSize(t *testing.T) { + origCfg := toolCfg + defer func() { toolCfg = origCfg }() + + // Pool with avg=1.5 is below min-pool-size=2; it should be dropped. + rec := common.Recommendation{ + Region: "us-east-1", + ResourceType: "db.t3.micro", + Count: 2, + AverageInstancesUsedPerHour: 1.5, + } + toolCfg.MinPoolSize = 2.0 + + d := common.NewDropSummary() + result := applyFilters( + []common.Recommendation{rec}, + &toolCfg, + make(map[string][]InstanceEngineVersion), + make(map[string]MajorEngineVersionInfo), + "", + d, + ) + + assert.Empty(t, result, "below-min-pool-size rec should be filtered out") + assert.Equal(t, 1, d.Total(), "drop summary should record 1 drop") + assert.Contains(t, d.FormatOneLine(), common.DropMinPoolSize, + "drop summary should name the --min-pool-size category") +} + +// TestApplyFilters_DropExtendedSupport verifies that a recommendation whose +// entire instance count is on an engine version in extended support is +// recorded in a non-nil DropSummary under DropExtendedSupport when +// --include-extended-support is false. If the drops.Add call for that +// path were removed, d.Total() would stay at 0 and the assertion below +// would fail. +func TestApplyFilters_DropExtendedSupport(t *testing.T) { + origCfg := toolCfg + defer func() { toolCfg = origCfg }() + + toolCfg.IncludeExtendedSupport = false + + // One RDS MySQL 5.7.42 instance in us-east-1; 5.7 is in extended support. + rec := common.Recommendation{ + Region: "us-east-1", + ResourceType: "db.t3.micro", + Count: 1, + Service: common.ServiceRDS, + Details: &common.DatabaseDetails{Engine: "mysql"}, + } + + instanceVersions := map[string][]InstanceEngineVersion{ + "db.t3.micro": { + {Engine: "mysql", EngineVersion: "5.7.42", InstanceClass: "db.t3.micro", Region: "us-east-1"}, + }, + } + + // Mark mysql 5.7 as in extended support (start date well in the past). + versionInfo := map[string]MajorEngineVersionInfo{ + "mysql:5.7": { + Engine: "mysql", + MajorEngineVersion: "5.7", + SupportedEngineLifecycles: []EngineLifecycleInfo{ + { + LifecycleSupportName: "open-source-rds-extended-support", + LifecycleSupportStartDate: time.Now().Add(-365 * 24 * time.Hour), + }, + }, + }, + } + + d := common.NewDropSummary() + result := applyFilters( + []common.Recommendation{rec}, + &toolCfg, + instanceVersions, + versionInfo, + "", + d, + ) + + assert.Empty(t, result, "all-extended-support rec should be filtered out") + assert.Equal(t, 1, d.Total(), "drop summary should record 1 drop") + assert.Contains(t, d.FormatOneLine(), common.DropExtendedSupport, + "drop summary should name the --include-extended-support category") +} + +// TestApplyFilters_SavingsPlansRegionFilters covers the CLI half of #1582. +// Savings Plans recommendations never populate the top-level rec.Region +// (parser_sp.go stores the CE-supplied region in Details.Region instead), so +// filtering on the bare field dropped every SP recommendation whenever +// --include-regions was set, and leaked region-scoped EC2Instance SPs past +// --exclude-regions. The provider-side filters were fixed first; these cases +// pin the same semantics on the CLI path. +func TestApplyFilters_SavingsPlansRegionFilters(t *testing.T) { + ec2SP := func(region string) common.Recommendation { + return common.Recommendation{ + Provider: common.ProviderAWS, + Service: common.ServiceSavingsPlansEC2Instance, + CommitmentType: common.CommitmentSavingsPlan, + Count: 1, + Details: &common.SavingsPlanDetails{ + PlanType: "EC2Instance", + InstanceFamily: "m5", + Region: region, + }, + } + } + computeSP := func() common.Recommendation { + return common.Recommendation{ + Provider: common.ProviderAWS, + Service: common.ServiceSavingsPlansCompute, + CommitmentType: common.CommitmentSavingsPlan, + Count: 1, + Details: &common.SavingsPlanDetails{PlanType: "Compute"}, + } + } + + tests := []struct { + name string + rec common.Recommendation + includeRegions []string + excludeRegions []string + wantKept bool + }{ + { + name: "EC2Instance SP in the included region survives", + rec: ec2SP("us-east-1"), + includeRegions: []string{"us-east-1"}, + wantKept: true, + }, + { + name: "EC2Instance SP outside the included region is dropped", + rec: ec2SP("eu-west-1"), + includeRegions: []string{"us-east-1"}, + wantKept: false, + }, + { + name: "region-agnostic Compute SP survives an include filter", + rec: computeSP(), + includeRegions: []string{"us-east-1"}, + wantKept: true, + }, + { + name: "EC2Instance SP in an excluded region is dropped", + rec: ec2SP("eu-west-1"), + excludeRegions: []string{"eu-west-1"}, + wantKept: false, + }, + { + name: "region-agnostic Compute SP survives an exclude filter", + rec: computeSP(), + excludeRegions: []string{"eu-west-1"}, + wantKept: true, + }, + { + // Cost Explorer omitted SavingsPlansDetails.Region. An EC2Instance + // SP is region-scoped, so an unknown region must not be treated as + // region-agnostic and waved past an explicit region filter. + name: "EC2Instance SP with an unknown region is not over-included", + rec: ec2SP(""), + includeRegions: []string{"us-east-1"}, + wantKept: false, + }, + { + name: "reservation rec with an unknown region is still dropped", + rec: common.Recommendation{Service: common.ServiceRDS, CommitmentType: common.CommitmentReservedInstance, Count: 1}, + includeRegions: []string{"us-east-1"}, + wantKept: false, + }, + } + + for _, tt := range tests { + t.Run(tt.name, func(t *testing.T) { + cfg := Config{ + IncludeRegions: tt.includeRegions, + ExcludeRegions: tt.excludeRegions, + } + want := 0 + if tt.wantKept { + want = 1 + } + got := applyFilters([]common.Recommendation{tt.rec}, &cfg, nil, nil, "", common.NewDropSummary()) + assert.Len(t, got, want, "region filter kept the wrong number of recommendations") + }) + } +} + +// applyFiltersPreExtraction is the pre-refactor applyFilters implementation, +// kept verbatim (from origin/main, before the recfilter.Filters.ApplyMinPoolSize +// extraction) as a differential oracle. It applies the --min-pool-size check +// inline, in the same loop and same order as processRecommendation, rather +// than as a separate pass over the full slice. +func applyFiltersPreExtraction(recs []common.Recommendation, cfg *Config, instanceVersions map[string][]InstanceEngineVersion, versionInfo map[string]MajorEngineVersionInfo, currentRegion string, drops *common.DropSummary) []common.Recommendation { + var filtered []common.Recommendation + var poolDropCount int + var poolDropInstances float64 + + for i := range recs { + if cfg.MinPoolSize > 0 && !shouldIncludePoolSize(&recs[i], cfg) { + poolDropInstances += recs[i].AverageInstancesUsedPerHour + label := fmt.Sprintf("%s/%s/%s", recs[i].Service, recs[i].Region, recs[i].ResourceType) + log.Printf("INFO: --min-pool-size=%.1f dropped %s (avg=%.2f < threshold)", cfg.MinPoolSize, label, recs[i].AverageInstancesUsedPerHour) + poolDropCount++ + drops.Add(common.DropMinPoolSize, 1) + continue + } + adjusted, include, dropReason := processRecommendation(&recs[i], cfg, instanceVersions, versionInfo, currentRegion) + if include { + filtered = append(filtered, adjusted) + } else if dropReason != "" { + drops.Add(dropReason, 1) + } + } + + if poolDropCount > 0 { + log.Printf("INFO: --min-pool-size dropped %d recommendation(s) (%.2f avg instances/hr total)", poolDropCount, poolDropInstances) + } + + return filtered +} + +// TestApplyFilters_MinPoolSizeMultiRegionMatchesPreExtractionBehaviour is a +// differential regression test against a reviewer's claim that extracting +// the --min-pool-size check into recfilter.Filters.ApplyMinPoolSize changed +// its ordering relative to the currentRegion guard in processRecommendation +// (inflating common.DropMinPoolSize in multi-region runs). It runs the +// current applyFilters and the pre-extraction oracle above side by side, +// once per region, over the same multi-region recommendation set, and +// asserts both the survivors and the drop accounting match exactly. +func TestApplyFilters_MinPoolSizeMultiRegionMatchesPreExtractionBehaviour(t *testing.T) { + origCfg := toolCfg + defer func() { toolCfg = origCfg }() + + toolCfg = Config{MinPoolSize: 2.0} + + regions := []string{"us-east-1", "eu-west-1", "ap-southeast-1"} + const distinctBelowThreshold = 3 // one below-threshold rec per region below + + // Fresh copies per call: applyFilters mutates nothing in place today, but + // the oracle and the refactored code must each see their own slice so a + // hypothetical future in-place adjustment on one side can't leak into the + // other's input and mask a real divergence. + makeRecs := func() []common.Recommendation { + return []common.Recommendation{ + {Region: "us-east-1", ResourceType: "db.t3.micro", Count: 3, AverageInstancesUsedPerHour: 5.0}, + {Region: "us-east-1", ResourceType: "db.t3.small", Count: 3, AverageInstancesUsedPerHour: 1.0}, // below threshold + {Region: "eu-west-1", ResourceType: "db.t3.micro", Count: 3, AverageInstancesUsedPerHour: 5.0}, + {Region: "eu-west-1", ResourceType: "db.t3.small", Count: 3, AverageInstancesUsedPerHour: 1.0}, // below threshold + {Region: "ap-southeast-1", ResourceType: "db.t3.micro", Count: 3, AverageInstancesUsedPerHour: 5.0}, + {Region: "ap-southeast-1", ResourceType: "db.t3.small", Count: 3, AverageInstancesUsedPerHour: 1.0}, // below threshold + {Region: "us-east-1", ResourceType: "db.t3.large", Count: 3, AverageInstancesUsedPerHour: 0}, // no-signal passthrough + { + Service: common.ServiceSavingsPlansCompute, + ResourceType: "ec2-instance", + Count: 10, + AverageInstancesUsedPerHour: 5.0, // account-level: bypasses the currentRegion guard + }, + } + } + + var totalDropMinPoolSizeNew, totalDropMinPoolSizeOld int + + for _, region := range regions { + t.Run(region, func(t *testing.T) { + dropsNew := common.NewDropSummary() + dropsOld := common.NewDropSummary() + + resultNew := applyFilters(makeRecs(), &toolCfg, make(map[string][]InstanceEngineVersion), make(map[string]MajorEngineVersionInfo), region, dropsNew) + resultOld := applyFiltersPreExtraction(makeRecs(), &toolCfg, make(map[string][]InstanceEngineVersion), make(map[string]MajorEngineVersionInfo), region, dropsOld) + + assert.Equal(t, resultOld, resultNew, + "region %s: refactored applyFilters diverged from the pre-extraction oracle -- the ApplyMinPoolSize extraction changed observable CLI filtering behavior", region) + assert.Equal(t, dropsOld.FormatOneLine(), dropsNew.FormatOneLine(), + "region %s: drop summaries diverged between pre- and post-extraction implementations", region) + assert.Equal(t, dropsOld.Total(), dropsNew.Total(), + "region %s: total drop counts diverged between pre- and post-extraction implementations", region) + + // The fixture is built so --min-pool-size is the ONLY drop reason + // either implementation can record, which is what lets the summed + // Total() below stand in for the min-pool count specifically. + // Asserting the exact per-pass count keeps that guarantee honest: + // if some other filter started dropping rows, Total() would exceed + // the below-threshold count and this would fail. + require.Contains(t, dropsNew.FormatOneLine(), common.DropMinPoolSize, + "region %s: fixture must exercise the --min-pool-size drop path", region) + require.Equal(t, distinctBelowThreshold, dropsNew.Total(), + "region %s: --min-pool-size must be the only drop reason this fixture records", region) + + totalDropMinPoolSizeNew += dropsNew.Total() + totalDropMinPoolSizeOld += dropsOld.Total() + }) + } + + require.Equal(t, totalDropMinPoolSizeOld, totalDropMinPoolSizeNew, + "old and new implementations must agree on the summed multi-region --min-pool-size drop count") + + // Pre-existing behavior, identical in both implementations (not introduced + // by the ApplyMinPoolSize extraction): each per-region call re-scans the + // FULL recommendation set passed to it, not a per-region subset, so a + // below-threshold recommendation from one region is re-counted as dropped + // during every other region's pass too. Summed across N region passes this + // is distinctBelowThreshold * N, not distinctBelowThreshold. This assertion + // pins today's (inflated) value rather than an aspirational deduplicated + // one -- see the test's summary report for the actual vs. distinct counts. + assert.Equal(t, distinctBelowThreshold*len(regions), totalDropMinPoolSizeNew, + "summed multi-region --min-pool-size drop count should match today's pre-existing inflation factor") +} diff --git a/cmd/multi_service_helpers.go b/cmd/multi_service_helpers.go new file mode 100644 index 000000000..ca5df213e --- /dev/null +++ b/cmd/multi_service_helpers.go @@ -0,0 +1,705 @@ +package main + +import ( + "context" + "fmt" + "log" + "sort" + "strings" + "time" + + "github.com/LeanerCloud/CUDly/pkg/common" + "github.com/LeanerCloud/CUDly/pkg/provider" + "github.com/LeanerCloud/CUDly/providers/aws/recommendations" + azureprovider "github.com/LeanerCloud/CUDly/providers/azure" + "github.com/aws/aws-sdk-go-v2/aws" + awsec2 "github.com/aws/aws-sdk-go-v2/service/ec2" +) + +// EC2ClientInterface defines the interface for EC2 operations. +type EC2ClientInterface interface { + DescribeRegions(ctx context.Context, params *awsec2.DescribeRegionsInput, optFns ...func(*awsec2.Options)) (*awsec2.DescribeRegionsOutput, error) +} + +// formatServices formats a list of services for display. +func formatServices(services []common.ServiceType) string { + names := make([]string, len(services)) + for i, s := range services { + names[i] = getServiceDisplayName(s) + } + return strings.Join(names, ", ") +} + +// getServiceDisplayName returns the display name for a service type. +func getServiceDisplayName(service common.ServiceType) string { + switch service { + case common.ServiceRDS: + return "RDS" + case common.ServiceElastiCache: + return "ElastiCache" + case common.ServiceEC2: + return "EC2" + case common.ServiceOpenSearch: + return "OpenSearch" + case common.ServiceRedshift: + return "Redshift" + case common.ServiceMemoryDB: + return "MemoryDB" + } + if name, ok := savingsPlanDisplayName(service); ok { + return name + } + return string(service) +} + +// savingsPlanDisplayName returns a friendly label for the four per-plan-type +// SP slugs and the legacy umbrella. Hoisted out of getServiceDisplayName to +// keep the main switch under the gocyclo limit; using a small lookup map +// instead of a five-arm switch also makes adding a fifth plan type a +// one-line change. +func savingsPlanDisplayName(service common.ServiceType) (string, bool) { + labels := map[common.ServiceType]string{ + common.ServiceSavingsPlansAll: "Savings Plans", + common.ServiceSavingsPlansCompute: "Compute Savings Plans", + common.ServiceSavingsPlansEC2Instance: "EC2 Instance Savings Plans", + common.ServiceSavingsPlansSageMaker: "SageMaker Savings Plans", + common.ServiceSavingsPlansDatabase: "Database Savings Plans", + } + name, ok := labels[service] + return name, ok +} + +// getAllAWSRegions retrieves all available AWS regions. +func getAllAWSRegions(ctx context.Context, cfg aws.Config) ([]string, error) { + // Create EC2 client to get regions + ec2Client := awsec2.NewFromConfig(cfg) + return getAllAWSRegionsWithClient(ctx, ec2Client) +} + +// getAllAWSRegionsWithClient retrieves all available AWS regions using the provided client. +func getAllAWSRegionsWithClient(ctx context.Context, ec2Client EC2ClientInterface) ([]string, error) { + // Describe all regions + result, err := ec2Client.DescribeRegions(ctx, &awsec2.DescribeRegionsInput{ + AllRegions: aws.Bool(false), // Only get opted-in regions + }) + if err != nil { + return nil, fmt.Errorf("failed to describe regions: %w", err) + } + + regions := make([]string, 0, len(result.Regions)) + for _, region := range result.Regions { + if region.RegionName != nil { + regions = append(regions, *region.RegionName) + } + } + + sort.Strings(regions) + return regions, nil +} + +// discoverRegionsForService discovers regions that have recommendations for a specific service. +func discoverRegionsForService(ctx context.Context, client provider.RecommendationsClient, service common.ServiceType) ([]string, error) { + recs, err := client.GetRecommendationsForService(ctx, service) + if partial := azureprovider.AsPartialSubscriptionFailure(err); partial != nil { + // Region discovery is best-effort: the subscriptions that answered + // still tell us where to look. Report the gap rather than dropping + // the discovered regions or failing outright. + AppLogger.Printf(" ⚠️ Region discovery incomplete: %d of %d Azure subscriptions succeeded\n", + partial.Succeeded, partial.Attempted) + } else if err != nil { + return nil, err + } + + regionSet := make(map[string]bool) + for _rvc := range recs { + rec := recs[_rvc] + if rec.Region != "" { + regionSet[rec.Region] = true + } + } + + regions := make([]string, 0, len(regionSet)) + for region := range regionSet { + regions = append(regions, region) + } + + sort.Strings(regions) + return regions, nil +} + +// applyCommonCoverage applies coverage percentage to recommendations. +func applyCommonCoverage(recs []common.Recommendation, coverage float64) []common.Recommendation { + return ApplyCoverage(recs, coverage) +} + +// determineServicesToProcess returns the list of services to process based on flags. +func determineServicesToProcess(cfg Config) []common.ServiceType { + if cfg.AllServices { + return getAllServices() + } + if len(cfg.Services) > 0 { + return parseServices(cfg.Services) + } + // Default to RDS only for backward compatibility + return []common.ServiceType{common.ServiceRDS} +} + +// printRunMode prints the current run mode (dry run or purchase). +func printRunMode(isDryRun bool) { + if isDryRun { + AppLogger.Println("🔍 DRY RUN MODE - No actual purchases will be made") + } else { + AppLogger.Println("💰 PURCHASE MODE - Reserved Instances will be purchased") + } +} + +// printPaymentAndTerm prints the payment option and term information. +func printPaymentAndTerm(cfg Config) { + AppLogger.Printf("💳 Payment option: %s, Term: %d year(s)\n", cfg.PaymentOption, cfg.TermYears) +} + +// generateCSVFilename generates a CSV filename based on the mode and timestamp. +func generateCSVFilename(isDryRun bool, cfg Config) string { + if cfg.CSVOutput != "" { + return cfg.CSVOutput + } + timestamp := time.Now().Format("20060102-150405") + mode := "dryrun" + if !isDryRun { + mode = "purchase" + } + return fmt.Sprintf("ri-helper-%s-%s.csv", mode, timestamp) +} + +// groupRecommendationsByServiceRegion groups recommendations by service and region. +func groupRecommendationsByServiceRegion(recs []common.Recommendation) map[common.ServiceType]map[string][]common.Recommendation { + recsByServiceRegion := make(map[common.ServiceType]map[string][]common.Recommendation) + for _rvc := range recs { + rec := recs[_rvc] + if _, ok := recsByServiceRegion[rec.Service]; !ok { + recsByServiceRegion[rec.Service] = make(map[string][]common.Recommendation) + } + recsByServiceRegion[rec.Service][rec.Region] = append(recsByServiceRegion[rec.Service][rec.Region], rec) + } + return recsByServiceRegion +} + +// populateAccountNames populates account names from account IDs using the cache. +func populateAccountNames(ctx context.Context, recs []common.Recommendation, accountCache *AccountAliasCache) { + for i := range recs { + if recs[i].Account != "" { + recs[i].AccountName = accountCache.GetAccountAlias(ctx, recs[i].Account) + } + } +} + +// adjustRecsForDuplicates checks for existing RIs and adjusts recommendations to avoid duplicates. +func adjustRecsForDuplicates(ctx context.Context, recs []common.Recommendation, serviceClient provider.ServiceClient) ([]common.Recommendation, error) { + duplicateChecker := NewDuplicateChecker(0) + adjustedRecs, _, err := duplicateChecker.AdjustRecommendationsForExisting(ctx, recs, serviceClient) + if err != nil { + return recs, err // Return original recommendations with error + } + + originalInstances := CalculateTotalInstances(recs) + adjustedInstances := CalculateTotalInstances(adjustedRecs) + if originalInstances != adjustedInstances { + AppLogger.Printf(" 🔍 Adjusted recommendations: %d instances → %d instances to avoid duplicate purchases\n", originalInstances, adjustedInstances) + } + + return adjustedRecs, nil +} + +// createDryRunResult creates a purchase result for dry run mode. +func createDryRunResult(rec common.Recommendation, region string, index int, cfg Config) common.PurchaseResult { + return common.PurchaseResult{ + Recommendation: rec, + Success: true, + CommitmentID: generatePurchaseID(rec, region, index, true, effectiveSizingPct(cfg)), + DryRun: true, + Timestamp: time.Now(), + } +} + +// createCancelledResults creates purchase results for canceled purchases. +func createCancelledResults(recs []common.Recommendation, region string, cfg Config) []common.PurchaseResult { + results := make([]common.PurchaseResult, len(recs)) + for k := range recs { + results[k] = common.PurchaseResult{ + Recommendation: recs[k], + Success: false, + CommitmentID: generatePurchaseID(recs[k], region, k+1, false, effectiveSizingPct(cfg)), + Error: fmt.Errorf("purchase canceled by user"), + Timestamp: time.Now(), + } + } + return results +} + +// executePurchase executes an actual RI purchase. +func executePurchase(ctx context.Context, rec common.Recommendation, region string, index int, serviceClient provider.ServiceClient, cfg Config) common.PurchaseResult { + AppLogger.Printf(" ⚠️ ACTUAL PURCHASE: About to buy %d instances of %s\n", rec.Count, rec.ResourceType) + // Compute the descriptive commitment ID up front and hand it to the + // provider so the purchased commitment is named descriptively at AWS + // (e.g. RDS ReservedDBInstanceId), not just in our local report. + commitmentID := generatePurchaseID(rec, region, index, false, effectiveSizingPct(cfg)) + opts := common.PurchaseOptions{Source: common.PurchaseSourceCLI, ReservationID: commitmentID} + result, err := serviceClient.PurchaseCommitment(ctx, rec, opts) + if err != nil { + result.Success = false + result.Error = err + } + if result.CommitmentID == "" { + result.CommitmentID = commitmentID + } + return result +} + +// determineRegionsForService determines which regions to process for a given service. +func determineRegionsForService(ctx context.Context, awsCfg aws.Config, recClient provider.RecommendationsClient, service common.ServiceType, configuredRegions []string) ([]string, error) { + // If regions are explicitly configured, use those + if len(configuredRegions) > 0 { + return configuredRegions, nil + } + + // Savings Plans are account-level, not regional - only query once + if common.IsSavingsPlan(service) { + AppLogger.Printf("🌍 Fetching account-level Savings Plans recommendations...\n") + return []string{"us-east-1"}, nil // Single query for account-level data + } + + // Default to all AWS regions for other services + AppLogger.Printf("🌍 Processing all AWS regions for %s...\n", getServiceDisplayName(service)) + allRegions, err := getAllAWSRegions(ctx, awsCfg) + if err != nil { + return handleRegionDiscoveryError(ctx, recClient, service, err) + } + + AppLogger.Printf("📍 Processing %d region(s)\n", len(allRegions)) + return allRegions, nil +} + +// handleRegionDiscoveryError handles errors during region discovery by falling back to auto-discovery. +func handleRegionDiscoveryError(ctx context.Context, recClient provider.RecommendationsClient, service common.ServiceType, originalErr error) ([]string, error) { + AppLogger.Printf("❌ Failed to get AWS regions: %v\n", originalErr) + AppLogger.Printf("🔍 Falling back to auto-discovery...\n") + + discoveredRegions, err := discoverRegionsForService(ctx, recClient, service) + if err != nil { + return nil, fmt.Errorf("failed to discover regions: %w", err) + } + + return discoveredRegions, nil +} + +// engineVersionData holds the results of engine version queries. +type engineVersionData struct { + instanceVersions map[string][]InstanceEngineVersion + versionInfo map[string]MajorEngineVersionInfo +} + +// fetchEngineVersionData queries running instances and major engine versions for validation. +func fetchEngineVersionData(ctx context.Context, cfg Config) engineVersionData { + data := engineVersionData{ + instanceVersions: make(map[string][]InstanceEngineVersion), + versionInfo: make(map[string]MajorEngineVersionInfo), + } + + // Query running instances for engine version validation + data.instanceVersions = queryInstanceVersions(ctx, cfg) + + // Query major engine versions for extended support detection + data.versionInfo = queryMajorVersions(ctx, cfg) + + return data +} + +// queryInstanceVersions queries running instances for engine version validation. +func queryInstanceVersions(ctx context.Context, cfg Config) map[string][]InstanceEngineVersion { + AppLogger.Printf("🔍 Querying running RDS instances across all regions to validate engine versions...\n") + instanceVersions, err := queryRunningInstanceEngineVersions(ctx, cfg) + if err != nil { + AppLogger.Printf("⚠️ Warning: Failed to query running instances for engine version validation: %v\n", err) + AppLogger.Printf(" Continuing without engine version filtering\n") + return make(map[string][]InstanceEngineVersion) + } + + AppLogger.Printf("✅ Found %d instance types with version information across all regions\n", len(instanceVersions)) + return instanceVersions +} + +// queryMajorVersions queries major engine versions for extended support detection. +func queryMajorVersions(ctx context.Context, cfg Config) map[string]MajorEngineVersionInfo { + AppLogger.Printf("🔍 Querying AWS RDS major engine versions for extended support information...\n") + versionInfo, err := queryMajorEngineVersions(ctx, cfg) + if err != nil { + AppLogger.Printf("⚠️ Warning: Failed to query major engine versions: %v\n", err) + AppLogger.Printf(" Continuing without extended support detection\n") + return make(map[string]MajorEngineVersionInfo) + } + + AppLogger.Printf("✅ Found support information for %d major engine versions\n", len(versionInfo)) + return versionInfo +} + +// regionRecommendations holds the processed recommendations for a single region. +type regionRecommendations struct { + recommendations []common.Recommendation + results []common.PurchaseResult +} + +// processRegionRecommendations fetches and processes recommendations for a single region. +func processRegionRecommendations( + ctx context.Context, + awsCfg aws.Config, + recClient provider.RecommendationsClient, + accountCache *AccountAliasCache, + service common.ServiceType, + region string, + regionIndex, totalRegions int, + engineData engineVersionData, + isDryRun bool, + cfg Config, + coverageMap recommendations.PoolCoverageMap, +) regionRecommendations { + result := regionRecommendations{ + recommendations: make([]common.Recommendation, 0), + results: make([]common.PurchaseResult, 0), + } + + AppLogger.Printf("\n 📍 [%d/%d] Region: %s\n", regionIndex, totalRegions, region) + + // Fetch recommendations + recs := fetchRecommendationsForRegion(ctx, recClient, service, region, cfg) + if len(recs) == 0 { + AppLogger.Printf(" ℹ️ No recommendations found\n") + return result + } + + AppLogger.Printf(" ✅ Found %d recommendations\n", len(recs)) + + // Populate account names + populateRecommendationAccountNames(ctx, recs, accountCache) + + // Apply filters (drops not tracked on this legacy test-only path). + recs = applyRegionFilters(recs, engineData, region, cfg, nil) + if len(recs) == 0 { + AppLogger.Printf(" ℹ️ No recommendations after applying filters\n") + return result + } + + // Apply coverage and overrides. This legacy path (test-only) doesn't + // fetch expiring commitments — expiry-aware sizing only runs via the + // main pipeline in fetchAndFilterRegionRecs, which has access to the + // regional service client at the right moment. Drop tracking skipped (nil). + filteredRecs := applyCoverageAndOverrides(recs, cfg, coverageMap, nil, nil) + + result.recommendations = filteredRecs + + // --max-instances is a run-wide cap, and this legacy per-region entry point + // has no view of the other services and regions in the run, so it cannot + // evaluate the cap. Refuse to spend rather than purchase uncapped: an + // over-purchase of reserved capacity is not reversible. Dry runs continue + // so the recommendations are still reported. + // + // This is deliberately the first thing after the recommendations are + // recorded, ahead of building the service client and of the duplicate + // check. The decision depends only on cfg.MaxInstances and isDryRun, both + // already known, so reaching it through a cloud API call would be work + // done to arrive at an answer that was already determined -- and it would + // make the refusal path fail differently depending on whether the describe + // call happened to succeed. + if cfg.MaxInstances > 0 && !isDryRun { + log.Printf("❌ Refusing to purchase %s/%s: --max-instances is a run-wide cap and cannot be enforced on the per-region path. Use the multi-service pipeline (the default entry point).", + getServiceDisplayName(service), region) + return result + } + + // Get service client and process purchases + regionalCfg := awsCfg.Copy() + regionalCfg.Region = region + serviceClient := createServiceClient(service, regionalCfg) + + if serviceClient == nil { + AppLogger.Printf(" ⚠️ Service client not yet implemented for %s\n", getServiceDisplayName(service)) + AppLogger.Printf(" (Skipping purchase phase for this service)\n") + return result + } + + // Check for duplicate RIs. Drop tracking skipped (nil). + adjustedRecs := checkDuplicates(ctx, filteredRecs, serviceClient, nil) + + // Process purchases + regionResults := processPurchaseLoop(ctx, adjustedRecs, region, isDryRun, serviceClient, cfg) + result.results = regionResults + + return result +} + +// fetchRecommendationsForRegion fetches recommendations from AWS for a specific region. +func fetchRecommendationsForRegion( + ctx context.Context, + recClient provider.RecommendationsClient, + service common.ServiceType, + region string, + cfg Config, +) []common.Recommendation { + termStr := "1yr" + if cfg.TermYears == 3 { + termStr = "3yr" + } + + lookback := cfg.RecLookbackPeriod + if lookback == "" { + lookback = recommendations.DefaultRecLookbackPeriod + } + params := common.RecommendationParams{ + Service: service, + Region: region, + PaymentOption: cfg.PaymentOption, + Term: termStr, + LookbackPeriod: lookback, + // Savings Plans specific filters + IncludeSPTypes: cfg.IncludeSPTypes, + ExcludeSPTypes: cfg.ExcludeSPTypes, + } + + recs, err := recClient.GetRecommendations(ctx, ¶ms) + if partial := azureprovider.AsPartialSubscriptionFailure(err); partial != nil { + // Keep the subscriptions that did answer, but say plainly that the + // sweep was incomplete: without this the operator would read a short + // list as "little to buy here" rather than "some subscriptions were + // never queried". + AppLogger.Printf(" ⚠️ Incomplete: %d of %d Azure subscriptions succeeded; %d could not be queried\n", + partial.Succeeded, partial.Attempted, len(partial.Failed)) + for _, f := range partial.Failed { + AppLogger.Printf(" subscription %s: %v\n", f.SubscriptionID, f.Err) + } + return recs + } + if err != nil { + AppLogger.Printf(" ❌ Failed to fetch recommendations: %v\n", err) + return nil + } + + return recs +} + +// populateRecommendationAccountNames populates account names from account IDs. +func populateRecommendationAccountNames(ctx context.Context, recs []common.Recommendation, accountCache *AccountAliasCache) { + for i := range recs { + if recs[i].Account != "" { + recs[i].AccountName = accountCache.GetAccountAlias(ctx, recs[i].Account) + } + } +} + +// applyRegionFilters applies region and instance type filters to recommendations. +func applyRegionFilters( + recs []common.Recommendation, + engineData engineVersionData, + region string, + cfg Config, + drops *common.DropSummary, +) []common.Recommendation { + originalCount := len(recs) + recs = applyFilters(recs, &cfg, engineData.instanceVersions, engineData.versionInfo, region, drops) + + if len(recs) < originalCount { + AppLogger.Printf(" 🔍 After filters: %d recommendations (filtered out %d)\n", len(recs), originalCount-len(recs)) + } + + return recs +} + +// applyCoverageAndOverrides applies sizing (coverage % or target-coverage) +// and count overrides. Sizing mode is selected by cfg.TargetCoverage: > 0 +// routes to ApplyTargetCoverage; otherwise the legacy ApplyCoverage path. +// coverageMap (when non-nil) populates Recommendation.ExistingCoveragePct +// before sizing so the under-buy formula can subtract pool coverage already +// owned. Nil map = no-op; recs keep their default zero values and the +// sizing path falls back to the no-existing-commitments formula. +// +// expiringCommitments (when non-empty and cfg.RebuyWindowDays > 0) reduces +// ExistingCoveragePct further by the share of pool demand attributable to +// RIs expiring within the window, so --target-coverage recommends +// replacements before the cliff. Doesn't run if either input is empty. +// +// drops accumulates per-reason drop counts for the end-of-run summary. +// Pass nil to skip tracking. +func applyCoverageAndOverrides(recs []common.Recommendation, cfg Config, coverageMap recommendations.PoolCoverageMap, expiringCommitments []common.Commitment, drops *common.DropSummary) []common.Recommendation { + recommendations.ApplyCoverageMapToRecommendations(recs, coverageMap) + if cfg.RebuyWindowDays > 0 && len(expiringCommitments) > 0 { + n := recommendations.AdjustExistingCoverageForExpiringCommitments(recs, expiringCommitments, cfg.RebuyWindowDays) + if n > 0 { + AppLogger.Printf(" ⏰ Treating %d recs as partially uncovered (RIs expiring within %d days)\n", n, cfg.RebuyWindowDays) + } + } + // Family-NU sizing for RDS recs: AWS rec API already bundles size-flex + // demand within a family into one rec at one size, so per-pool sizing + // under-buys. Run the family-NU pass first (RDS only, target-coverage + // mode only); non-RDS recs flow through the per-pool path unchanged. + // When TargetCoverage isn't set (legacy --coverage path) the per-pool + // flow handles everything as before. + var sizedRDS []common.Recommendation + rest := recs + if cfg.TargetCoverage > 0 { + var familyDrops recommendations.FamilyDropCounts + sizedRDS, rest, familyDrops = recommendations.ApplyFamilyNUSizingRDS(recs, coverageMap, cfg.TargetCoverage) + drops.Add(common.DropFamilyAlreadyAtTarget, familyDrops.AlreadyAtTarget) + drops.Add(common.DropFamilyNoNUSignal, familyDrops.NoNUSignal) + drops.Add(common.DropFamilySizedToZero, familyDrops.SizedToZero) + } + filteredRecs := applySizing(rest, cfg, cfg.Coverage, drops) + filteredRecs = append(filteredRecs, sizedRDS...) + if cfg.TargetCoverage > 0 { + AppLogger.Printf(" 🎯 Applying %.1f%% target-coverage: %d recommendations selected (%d via family-NU, %d via per-pool)\n", + cfg.TargetCoverage, len(filteredRecs), len(sizedRDS), len(filteredRecs)-len(sizedRDS)) + } else { + AppLogger.Printf(" 📈 Applying %.1f%% coverage: %d recommendations selected\n", cfg.Coverage, len(filteredRecs)) + } + + // Apply count override if specified + if cfg.OverrideCount > 0 { + filteredRecs = ApplyCountOverride(filteredRecs, cfg.OverrideCount) + } + + return filteredRecs +} + +// checkDuplicates adjusts recommendations against already-owned RIs so the run +// does not double-purchase existing capacity. +// +// It deliberately does NOT apply --max-instances. This function runs once per +// (service, region), and capping here caps each region independently, which +// multiplies the operator's cap by the number of service/region pairs. The cap +// is applied once run-wide instead, after every region has been fetched +// (applyGlobalInstanceLimit in multi_service.go). +// +// drops accumulates per-reason drop counts for the end-of-run summary; pass nil to skip. +func checkDuplicates( + ctx context.Context, + filteredRecs []common.Recommendation, + serviceClient provider.ServiceClient, + drops *common.DropSummary, +) []common.Recommendation { + // Check for duplicate RIs to avoid double purchasing + duplicateChecker := NewDuplicateChecker(0) + adjustedRecs, dedupedOut, err := duplicateChecker.AdjustRecommendationsForExistingRIs(ctx, filteredRecs, serviceClient) + if err != nil { + AppLogger.Printf(" ⚠️ Warning: Could not check for existing RIs: %v\n", err) + // Continue with original filteredRecs on error; adjustedRecs is not used in this branch. + } else { + // Always use the adjusted recommendations (they might have different counts even if same length) + originalInstances := CalculateTotalInstances(filteredRecs) + adjustedInstances := CalculateTotalInstances(adjustedRecs) + if originalInstances != adjustedInstances { + AppLogger.Printf(" 🔍 Adjusted recommendations: %d instances → %d instances to avoid duplicate purchases\n", originalInstances, adjustedInstances) + } + drops.Add(common.DropDuplicateDedup, len(dedupedOut)) + filteredRecs = adjustedRecs + } + + return filteredRecs +} + +// fetchAndFilterRegionRecs fetches, filters, applies coverage, and deduplicates +// recommendations for a single service+region. No purchases are made. +// coverageMap (when non-nil) feeds existing-pool coverage into the sizing +// step for --target-coverage. drops accumulates per-reason drop counts for +// the end-of-run summary; pass nil to skip tracking. +func fetchAndFilterRegionRecs( + ctx context.Context, + awsCfg aws.Config, + recClient provider.RecommendationsClient, + accountCache *AccountAliasCache, + service common.ServiceType, + region string, + regionIndex, totalRegions int, + engineData engineVersionData, + cfg Config, + coverageMap recommendations.PoolCoverageMap, + drops *common.DropSummary, +) []common.Recommendation { + AppLogger.Printf("\n 📍 [%d/%d] Region: %s\n", regionIndex, totalRegions, region) + + recs := fetchRecommendationsForRegion(ctx, recClient, service, region, cfg) + if len(recs) == 0 { + AppLogger.Printf(" ℹ️ No recommendations found\n") + return nil + } + AppLogger.Printf(" ✅ Found %d recommendations\n", len(recs)) + + populateRecommendationAccountNames(ctx, recs, accountCache) + recs = applyRegionFilters(recs, engineData, region, cfg, drops) + if len(recs) == 0 { + AppLogger.Printf(" ℹ️ No recommendations after applying filters\n") + return nil + } + + // Build the regional service client once and reuse for both expiry-aware + // coverage adjustment and the downstream duplicate check. + regionalCfg := awsCfg.Copy() + regionalCfg.Region = region + serviceClient := createServiceClient(service, regionalCfg) + + // Fetch existing commitments up front when --rebuy-window-days is set so + // the sizing step can deduct expiring RIs from ExistingCoveragePct. + // Best-effort: a failure here logs and continues; expiry adjustment + // becomes a no-op for this region. + var expiringCommitments []common.Commitment + if cfg.RebuyWindowDays > 0 && serviceClient != nil { + commits, err := serviceClient.GetExistingCommitments(ctx) + if err != nil { + AppLogger.Printf(" ⚠️ Could not fetch existing commitments for expiry check: %v\n", err) + } else { + expiringCommitments = commits + } + } + + recs = applyCoverageAndOverrides(recs, cfg, coverageMap, expiringCommitments, drops) + + // Deduplication: skip recs matching recently-purchased commitments. + // --max-instances is NOT applied here; it is enforced once run-wide by the + // caller so the cap covers every service and region together. + if serviceClient != nil { + recs = checkDuplicates(ctx, recs, serviceClient, drops) + } + + return recs +} + +// fetchAllRecs collects recommendations from all services and regions without +// purchasing. coverageMap (when non-nil) populates Recommendation.ExistingCoveragePct +// on each rec before sizing, so --target-coverage can subtract what's already +// owned in the same pool. The returned DropSummary accumulates per-reason +// drop counts across every service and region for the end-of-run summary. +func fetchAllRecs( + ctx context.Context, + awsCfg aws.Config, + recClient provider.RecommendationsClient, + accountCache *AccountAliasCache, + servicesToProcess []common.ServiceType, + engineData engineVersionData, + cfg Config, + coverageMap recommendations.PoolCoverageMap, +) ([]common.Recommendation, *common.DropSummary) { + all := make([]common.Recommendation, 0) + drops := common.NewDropSummary() + for _, service := range servicesToProcess { + AppLogger.Printf("\n━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━\n") + AppLogger.Printf("🔍 Fetching %s recommendations\n", getServiceDisplayName(service)) + AppLogger.Printf("━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━\n") + + regions, err := determineRegionsForService(ctx, awsCfg, recClient, service, cfg.Regions) + if err != nil { + log.Printf("❌ Failed to determine regions for %s: %v", getServiceDisplayName(service), err) + continue + } + for i, region := range regions { + recs := fetchAndFilterRegionRecs(ctx, awsCfg, recClient, accountCache, service, region, i+1, len(regions), engineData, cfg, coverageMap, drops) + all = append(all, recs...) + } + } + return all, drops +} diff --git a/cmd/multi_service_helpers_test.go b/cmd/multi_service_helpers_test.go new file mode 100644 index 000000000..d8f6a029f --- /dev/null +++ b/cmd/multi_service_helpers_test.go @@ -0,0 +1,832 @@ +package main + +import ( + "context" + "errors" + "os" + "testing" + "time" + + "github.com/LeanerCloud/CUDly/pkg/common" + "github.com/aws/aws-sdk-go-v2/aws" + "github.com/aws/aws-sdk-go-v2/service/ec2" + "github.com/aws/aws-sdk-go-v2/service/ec2/types" + "github.com/aws/aws-sdk-go-v2/service/organizations" + orgtypes "github.com/aws/aws-sdk-go-v2/service/organizations/types" + "github.com/stretchr/testify/assert" + "github.com/stretchr/testify/mock" +) + +func TestGetAllAWSRegions(t *testing.T) { + ctx := context.Background() + + tests := []struct { + name string + mockOutput *ec2.DescribeRegionsOutput + mockError error + expectRegions []string + expectError bool + }{ + { + name: "Success with multiple regions", + mockOutput: &ec2.DescribeRegionsOutput{ + Regions: []types.Region{ + {RegionName: aws.String("us-east-1")}, + {RegionName: aws.String("eu-west-1")}, + {RegionName: aws.String("ap-south-1")}, + }, + }, + expectRegions: []string{"ap-south-1", "eu-west-1", "us-east-1"}, // Sorted + expectError: false, + }, + { + name: "Error from AWS API", + mockOutput: nil, + mockError: errors.New("AWS API error"), + expectRegions: nil, + expectError: true, + }, + { + name: "Empty regions list", + mockOutput: &ec2.DescribeRegionsOutput{ + Regions: []types.Region{}, + }, + expectRegions: []string{}, + expectError: false, + }, + { + name: "Regions with nil names", + mockOutput: &ec2.DescribeRegionsOutput{ + Regions: []types.Region{ + {RegionName: aws.String("us-east-1")}, + {RegionName: nil}, + {RegionName: aws.String("eu-west-1")}, + }, + }, + expectRegions: []string{"eu-west-1", "us-east-1"}, // Sorted, nil excluded + expectError: false, + }, + } + + for _, tt := range tests { + t.Run(tt.name, func(t *testing.T) { + mockEC2 := &MockEC2Client{} + mockEC2.On("DescribeRegions", ctx, mock.Anything).Return(tt.mockOutput, tt.mockError) + + // Use the new interface-based function + regions, err := getAllAWSRegionsWithClient(ctx, mockEC2) + + if tt.expectError { + assert.Error(t, err) + assert.Nil(t, regions) + } else { + assert.NoError(t, err) + assert.Equal(t, tt.expectRegions, regions) + } + + mockEC2.AssertExpectations(t) + }) + } + + t.Run("Integration test", func(t *testing.T) { + // This test requires actual AWS credentials + if os.Getenv("AWS_ACCESS_KEY_ID") == "" { + t.Skip("Skipping integration test: AWS credentials not present") + } + + cfg := aws.Config{Region: "us-east-1"} + regions, err := getAllAWSRegions(ctx, cfg) + + if err == nil { + assert.NotNil(t, regions) + assert.Greater(t, len(regions), 0) + + // Verify regions are sorted + for i := 1; i < len(regions); i++ { + assert.LessOrEqual(t, regions[i-1], regions[i]) + } + } + }) +} + +func TestDiscoverRegionsForService(t *testing.T) { + ctx := context.Background() + + tests := []struct { + name string + service common.ServiceType + mockReturns []common.Recommendation + expectedRegions []string + expectError bool + }{ + { + name: "Multiple unique regions", + service: common.ServiceRDS, + mockReturns: []common.Recommendation{ + {Region: "us-east-1", ResourceType: "db.t3.micro"}, + {Region: "us-west-2", ResourceType: "db.t3.small"}, + {Region: "eu-west-1", ResourceType: "db.t3.medium"}, + }, + expectedRegions: []string{"eu-west-1", "us-east-1", "us-west-2"}, + }, + { + name: "Duplicate regions", + service: common.ServiceEC2, + mockReturns: []common.Recommendation{ + {Region: "us-east-1", ResourceType: "t3.micro"}, + {Region: "us-east-1", ResourceType: "t3.small"}, + {Region: "us-west-2", ResourceType: "t3.medium"}, + }, + expectedRegions: []string{"us-east-1", "us-west-2"}, + }, + { + name: "No recommendations", + service: common.ServiceElastiCache, + mockReturns: []common.Recommendation{}, + expectedRegions: []string{}, + }, + { + name: "Recommendations with empty regions filtered", + service: common.ServiceRedshift, + mockReturns: []common.Recommendation{ + {Region: "us-east-1", ResourceType: "ra3.xlplus"}, + {Region: "", ResourceType: "ra3.4xlarge"}, + {Region: "us-west-2", ResourceType: "ra3.16xlarge"}, + }, + expectedRegions: []string{"us-east-1", "us-west-2"}, + }, + } + + for _, tt := range tests { + t.Run(tt.name, func(t *testing.T) { + mockClient := &MockRecommendationsClient{} + mockClient.On("GetRecommendationsForService", ctx, tt.service).Return(tt.mockReturns, nil) + + // Now we can use the actual function directly since it accepts an interface + regions, err := discoverRegionsForService(ctx, mockClient, tt.service) + + assert.NoError(t, err) + assert.Equal(t, tt.expectedRegions, regions) + + mockClient.AssertExpectations(t) + }) + } +} + +func TestFormatServices(t *testing.T) { + tests := []struct { + name string + expected string + services []common.ServiceType + }{ + { + name: "Empty list", + services: []common.ServiceType{}, + expected: "", + }, + { + name: "Single service", + services: []common.ServiceType{common.ServiceRDS}, + expected: "RDS", + }, + { + name: "Multiple services", + services: []common.ServiceType{common.ServiceRDS, common.ServiceEC2, common.ServiceElastiCache}, + expected: "RDS, EC2, ElastiCache", + }, + { + name: "All services", + services: getAllServices(), + expected: formatServices(getAllServices()), // Build expected from getAllServices() to avoid test breakage on service list changes + }, + } + + for _, tt := range tests { + t.Run(tt.name, func(t *testing.T) { + result := formatServices(tt.services) + assert.Equal(t, tt.expected, result) + }) + } +} + +func TestGetServiceDisplayName(t *testing.T) { + tests := []struct { + service common.ServiceType + expected string + }{ + {common.ServiceRDS, "RDS"}, + {common.ServiceElastiCache, "ElastiCache"}, + {common.ServiceEC2, "EC2"}, + {common.ServiceOpenSearch, "OpenSearch"}, + {common.ServiceElasticsearch, "OpenSearch"}, + {common.ServiceRedshift, "Redshift"}, + {common.ServiceMemoryDB, "MemoryDB"}, + {common.ServiceType("custom"), "custom"}, + {common.ServiceType(""), ""}, + } + + for _, tt := range tests { + t.Run(string(tt.service), func(t *testing.T) { + result := getServiceDisplayName(tt.service) + assert.Equal(t, tt.expected, result) + }) + } +} + +func TestApplyCommonCoverage(t *testing.T) { + recs := []common.Recommendation{ + {Count: 10, EstimatedSavings: 100}, + {Count: 5, EstimatedSavings: 50}, + {Count: 2, EstimatedSavings: 20}, + } + + tests := []struct { + name string + expectedInstances []int + coverage float64 + expectedCount int + }{ + { + name: "100% coverage", + coverage: 100.0, + expectedCount: 3, + expectedInstances: []int{10, 5, 2}, + }, + { + name: "50% coverage", + coverage: 50.0, + expectedCount: 3, + expectedInstances: []int{5, 2, 1}, // Using floor: 10*0.5=5, 5*0.5=2.5→2, 2*0.5=1 + }, + { + name: "0% coverage", + coverage: 0.0, + expectedCount: 0, + expectedInstances: []int{}, + }, + { + name: "75% coverage", + coverage: 75.0, + expectedCount: 3, + expectedInstances: []int{7, 3, 1}, // Using floor: 10*0.75=7.5→7, 5*0.75=3.75→3, 2*0.75=1.5→1 + }, + } + + for _, tt := range tests { + t.Run(tt.name, func(t *testing.T) { + result := applyCommonCoverage(recs, tt.coverage) + assert.Equal(t, tt.expectedCount, len(result)) + + for i, rec := range result { + if i < len(tt.expectedInstances) { + assert.Equal(t, tt.expectedInstances[i], rec.Count) + } + } + }) + } +} + +func TestCreateDryRunResult(t *testing.T) { + // Save original values + origCfg := toolCfg + + defer func() { + toolCfg = origCfg + }() + + toolCfg.Coverage = 75.0 + + rec := common.Recommendation{ + Service: common.ServiceRDS, + ResourceType: "db.t3.small", + Count: 5, + Region: "us-east-1", + } + + result := createDryRunResult(rec, "us-east-1", 1, toolCfg) + + assert.True(t, result.Success) + assert.Equal(t, rec, result.Recommendation) + assert.Nil(t, result.Error) // Dry runs are successful, so no error + assert.True(t, result.DryRun) + // Check format: dryrun-{service}-{region}-{type}-{count} + assert.Regexp(t, "^dryrun-rds-us-east-1-.*-5x-", result.CommitmentID) + assert.NotEmpty(t, result.Timestamp) +} + +func TestCreateCancelledResults(t *testing.T) { + // Save original values + origCfg := toolCfg + + defer func() { + toolCfg = origCfg + }() + + toolCfg.Coverage = 80.0 + + recs := []common.Recommendation{ + {Service: common.ServiceRDS, ResourceType: "db.t3.small", Count: 2}, + {Service: common.ServiceRDS, ResourceType: "db.t3.medium", Count: 3}, + {Service: common.ServiceRDS, ResourceType: "db.t3.large", Count: 1}, + } + + results := createCancelledResults(recs, "us-west-2", toolCfg) + + assert.Len(t, results, 3) + for i, result := range results { + assert.False(t, result.Success) + assert.Equal(t, recs[i], result.Recommendation) + assert.NotNil(t, result.Error) + assert.Contains(t, result.Error.Error(), "canceled") + assert.Contains(t, result.CommitmentID, "us-west-2") + } +} + +func TestExecutePurchase(t *testing.T) { + ctx := context.Background() + // Save original values + origCfg := toolCfg + + defer func() { + toolCfg = origCfg + }() + + toolCfg.Coverage = 90.0 + + rec := common.Recommendation{ + Service: common.ServiceEC2, + ResourceType: "t3.medium", + Count: 10, + } + + mockClient := &MockServiceClient{} + expectedResult := common.PurchaseResult{ + Recommendation: rec, + Success: true, + CommitmentID: "test-purchase-id-123", + Error: nil, + Timestamp: time.Now(), + } + var capturedOpts common.PurchaseOptions + mockClient.On("PurchaseCommitment", ctx, rec, mock.MatchedBy(func(o common.PurchaseOptions) bool { + capturedOpts = o + return o.Source == common.PurchaseSourceCLI + })).Return(expectedResult, nil) + + result := executePurchase(ctx, rec, "eu-west-1", 5, mockClient, toolCfg) + + assert.True(t, result.Success) + assert.Equal(t, "test-purchase-id-123", result.CommitmentID) + assert.Nil(t, result.Error) + + // #617: executePurchase hands the provider a descriptive reservation ID + // (account/service/region/size aware) rather than leaving it generic. + assert.NotEmpty(t, capturedOpts.ReservationID) + assert.Contains(t, capturedOpts.ReservationID, "t3-medium") + + mockClient.AssertExpectations(t) +} + +func TestAdjustRecsForDuplicates(t *testing.T) { + ctx := context.Background() + + tests := []struct { + name string + inputRecs []common.Recommendation + existingRIs []common.Commitment + expectedCount int + expectedError bool + }{ + { + name: "No duplicates", + inputRecs: []common.Recommendation{ + {ResourceType: "db.t3.small", Count: 5}, + {ResourceType: "db.t3.medium", Count: 3}, + }, + existingRIs: []common.Commitment{}, + expectedCount: 2, + expectedError: false, + }, + { + name: "With duplicates - adjusts count", + inputRecs: []common.Recommendation{ + {ResourceType: "db.t3.small", Count: 10}, + }, + existingRIs: []common.Commitment{ + {ResourceType: "db.t3.small", Count: 3}, + }, + expectedCount: 1, // Should still have 1 recommendation but with adjusted count + expectedError: false, + }, + } + + for _, tt := range tests { + t.Run(tt.name, func(t *testing.T) { + mockClient := &MockServiceClient{} + mockClient.On("GetExistingCommitments", ctx).Return(tt.existingRIs, nil) + + results, err := adjustRecsForDuplicates(ctx, tt.inputRecs, mockClient) + + if tt.expectedError { + assert.Error(t, err) + } else { + assert.NoError(t, err) + assert.LessOrEqual(t, len(results), len(tt.inputRecs)) + } + + mockClient.AssertExpectations(t) + }) + } +} + +func TestAdjustRecsForDuplicatesError(t *testing.T) { + ctx := context.Background() + + recs := []common.Recommendation{ + {ResourceType: "db.t3.small", Count: 5}, + } + + mockClient := &MockServiceClient{} + mockClient.On("GetExistingCommitments", ctx).Return([]common.Commitment(nil), errors.New("API error")) + + // Logger output disabled for testing + + results, err := adjustRecsForDuplicates(ctx, recs, mockClient) + + // Should return original recommendations with error (error is propagated) + assert.Error(t, err) + assert.Contains(t, err.Error(), "API error") + assert.Equal(t, recs, results) // Still returns original recommendations + + mockClient.AssertExpectations(t) +} + +func TestGroupRecommendationsByServiceRegion(t *testing.T) { + tests := []struct { + expectedGroups map[common.ServiceType]map[string]int + name string + recommendations []common.Recommendation + }{ + { + name: "Single service single region", + recommendations: []common.Recommendation{ + {Service: common.ServiceRDS, Region: "us-east-1", ResourceType: "db.t3.small", Count: 5}, + {Service: common.ServiceRDS, Region: "us-east-1", ResourceType: "db.t3.medium", Count: 3}, + }, + expectedGroups: map[common.ServiceType]map[string]int{ + common.ServiceRDS: {"us-east-1": 2}, + }, + }, + { + name: "Single service multiple regions", + recommendations: []common.Recommendation{ + {Service: common.ServiceRDS, Region: "us-east-1", ResourceType: "db.t3.small", Count: 5}, + {Service: common.ServiceRDS, Region: "us-west-2", ResourceType: "db.t3.medium", Count: 3}, + {Service: common.ServiceRDS, Region: "eu-west-1", ResourceType: "db.t3.large", Count: 2}, + }, + expectedGroups: map[common.ServiceType]map[string]int{ + common.ServiceRDS: {"us-east-1": 1, "us-west-2": 1, "eu-west-1": 1}, + }, + }, + { + name: "Multiple services multiple regions", + recommendations: []common.Recommendation{ + {Service: common.ServiceRDS, Region: "us-east-1", ResourceType: "db.t3.small", Count: 5}, + {Service: common.ServiceRDS, Region: "us-west-2", ResourceType: "db.t3.medium", Count: 3}, + {Service: common.ServiceElastiCache, Region: "us-east-1", ResourceType: "cache.t3.small", Count: 2}, + {Service: common.ServiceElastiCache, Region: "eu-west-1", ResourceType: "cache.t3.medium", Count: 4}, + {Service: common.ServiceEC2, Region: "us-east-1", ResourceType: "m5.large", Count: 10}, + }, + expectedGroups: map[common.ServiceType]map[string]int{ + common.ServiceRDS: {"us-east-1": 1, "us-west-2": 1}, + common.ServiceElastiCache: {"us-east-1": 1, "eu-west-1": 1}, + common.ServiceEC2: {"us-east-1": 1}, + }, + }, + { + name: "Empty recommendations", + recommendations: []common.Recommendation{}, + expectedGroups: map[common.ServiceType]map[string]int{}, + }, + } + + for _, tt := range tests { + t.Run(tt.name, func(t *testing.T) { + result := groupRecommendationsByServiceRegion(tt.recommendations) + + // Verify the structure matches expected + assert.Equal(t, len(tt.expectedGroups), len(result)) + + for service, regions := range tt.expectedGroups { + assert.Contains(t, result, service) + assert.Equal(t, len(regions), len(result[service])) + + for region, expectedCount := range regions { + assert.Contains(t, result[service], region) + assert.Equal(t, expectedCount, len(result[service][region])) + } + } + }) + } +} + +func TestGenerateCSVFilename(t *testing.T) { + tests := []struct { + check func(t *testing.T, filename string) + name string + cfg Config + isDryRun bool + }{ + { + name: "Dry run mode generates dryrun filename", + isDryRun: true, + cfg: Config{}, + check: func(t *testing.T, filename string) { + assert.Contains(t, filename, "ri-helper-dryrun-") + assert.Contains(t, filename, ".csv") + }, + }, + { + name: "Purchase mode generates purchase filename", + isDryRun: false, + cfg: Config{}, + check: func(t *testing.T, filename string) { + assert.Contains(t, filename, "ri-helper-purchase-") + assert.Contains(t, filename, ".csv") + }, + }, + { + name: "Custom output overrides default", + isDryRun: true, + cfg: Config{CSVOutput: "custom-output.csv"}, + check: func(t *testing.T, filename string) { + assert.Equal(t, "custom-output.csv", filename) + }, + }, + } + + for _, tt := range tests { + t.Run(tt.name, func(t *testing.T) { + result := generateCSVFilename(tt.isDryRun, tt.cfg) + tt.check(t, result) + }) + } +} + +func TestPrintRunMode(t *testing.T) { + // Capture output by disabling logger + // Logger output disabled for testing + + // Just ensure no panic - the function primarily prints + printRunMode(true) + printRunMode(false) +} + +func TestPrintPaymentAndTerm(t *testing.T) { //nolint:unparam // param intentional for interface consistency/future use + // Capture output by disabling logger + // Logger output disabled for testing + + cfg := Config{ + PaymentOption: "partial-upfront", + TermYears: 3, + } + + // Just ensure no panic - the function primarily prints + printPaymentAndTerm(cfg) +} + +func TestDetermineServicesToProcess_AllServices(t *testing.T) { + cfg := Config{ + AllServices: true, + } + + result := determineServicesToProcess(cfg) + + // Should contain all supported services + assert.Contains(t, result, common.ServiceRDS) + assert.Contains(t, result, common.ServiceElastiCache) + assert.Contains(t, result, common.ServiceEC2) + assert.Contains(t, result, common.ServiceOpenSearch) + assert.Contains(t, result, common.ServiceRedshift) + assert.Contains(t, result, common.ServiceMemoryDB) +} + +func TestDetermineServicesToProcess_SpecificServices(t *testing.T) { + cfg := Config{ + AllServices: false, + Services: []string{"rds", "elasticache"}, + } + + result := determineServicesToProcess(cfg) + + assert.Equal(t, 2, len(result)) + assert.Contains(t, result, common.ServiceRDS) + assert.Contains(t, result, common.ServiceElastiCache) +} + +func TestPopulateAccountNames(t *testing.T) { + ctx := context.Background() + + tests := []struct { + name string + recommendations []common.Recommendation + setupMock func(m *MockOrganizationsClient) + expectedNames []string + }{ + { + name: "Populates account names from IDs", + recommendations: []common.Recommendation{ + {Account: "123456789012", AccountName: ""}, + {Account: "210987654321", AccountName: ""}, + }, + setupMock: func(m *MockOrganizationsClient) { + m.On("DescribeAccount", ctx, &organizations.DescribeAccountInput{ + AccountId: aws.String("123456789012"), + }).Return(&organizations.DescribeAccountOutput{ + Account: &orgtypes.Account{ + Name: aws.String("Production"), + }, + }, nil).Once() + m.On("DescribeAccount", ctx, &organizations.DescribeAccountInput{ + AccountId: aws.String("210987654321"), + }).Return(&organizations.DescribeAccountOutput{ + Account: &orgtypes.Account{ + Name: aws.String("Development"), + }, + }, nil).Once() + }, + expectedNames: []string{"Production", "Development"}, + }, + { + name: "Handles empty account IDs", + recommendations: []common.Recommendation{ + {Account: "", AccountName: ""}, + {Account: "123456789012", AccountName: ""}, + }, + setupMock: func(m *MockOrganizationsClient) { + m.On("DescribeAccount", ctx, &organizations.DescribeAccountInput{ + AccountId: aws.String("123456789012"), + }).Return(&organizations.DescribeAccountOutput{ + Account: &orgtypes.Account{ + Name: aws.String("Production"), + }, + }, nil).Once() + }, + expectedNames: []string{"", "Production"}, + }, + { + name: "Uses cached values for repeated accounts", + recommendations: []common.Recommendation{ + {Account: "123456789012", AccountName: ""}, + {Account: "123456789012", AccountName: ""}, + {Account: "123456789012", AccountName: ""}, + }, + setupMock: func(m *MockOrganizationsClient) { + // Should only be called once due to caching + m.On("DescribeAccount", ctx, &organizations.DescribeAccountInput{ + AccountId: aws.String("123456789012"), + }).Return(&organizations.DescribeAccountOutput{ + Account: &orgtypes.Account{ + Name: aws.String("Production"), + }, + }, nil).Once() + }, + expectedNames: []string{"Production", "Production", "Production"}, + }, + { + name: "Handles empty recommendations", + recommendations: []common.Recommendation{}, + setupMock: func(m *MockOrganizationsClient) { + // No calls expected + }, + expectedNames: []string{}, + }, + } + + for _, tt := range tests { + t.Run(tt.name, func(t *testing.T) { + mockOrg := &MockOrganizationsClient{} + tt.setupMock(mockOrg) + + cache := &TestAccountAliasCache{ + cache: make(map[string]string), + orgClient: mockOrg, + } + + // Manually populate account names using our test cache + for i := range tt.recommendations { + if tt.recommendations[i].Account != "" { + tt.recommendations[i].AccountName = cache.GetAccountAlias(ctx, tt.recommendations[i].Account) + } + } + + assert.Equal(t, len(tt.expectedNames), len(tt.recommendations)) + for i, rec := range tt.recommendations { + assert.Equal(t, tt.expectedNames[i], rec.AccountName) + } + + mockOrg.AssertExpectations(t) + }) + } +} + +// TestPopulateAccountNamesLogic tests the logic of populateAccountNames +// by verifying it populates the AccountName field correctly. +func TestPopulateAccountNamesLogic(t *testing.T) { + ctx := context.Background() + + t.Run("Correctly populates account names", func(t *testing.T) { + mockOrg := &MockOrganizationsClient{} + mockOrg.On("DescribeAccount", ctx, &organizations.DescribeAccountInput{ + AccountId: aws.String("123456789012"), + }).Return(&organizations.DescribeAccountOutput{ + Account: &orgtypes.Account{ + Name: aws.String("Production"), + }, + }, nil).Once() + + cache := &TestAccountAliasCache{ + cache: make(map[string]string), + orgClient: mockOrg, + } + + recs := []common.Recommendation{ + {Account: "123456789012", AccountName: ""}, + } + + // Simulate what populateAccountNames does - call GetAccountAlias for each rec + for i := range recs { + if recs[i].Account != "" { + recs[i].AccountName = cache.GetAccountAlias(ctx, recs[i].Account) + } + } + + assert.Equal(t, "Production", recs[0].AccountName) + mockOrg.AssertExpectations(t) + }) + + t.Run("Skips empty account IDs", func(t *testing.T) { + cache := &TestAccountAliasCache{ + cache: make(map[string]string), + orgClient: &MockOrganizationsClient{}, // No calls expected + } + + recs := []common.Recommendation{ + {Account: "", AccountName: ""}, + {Account: "", AccountName: "initial"}, + } + + // Simulate what populateAccountNames does + for i := range recs { + if recs[i].Account != "" { + recs[i].AccountName = cache.GetAccountAlias(ctx, recs[i].Account) + } + } + + // Empty accounts should not be modified (or return empty from GetAccountAlias) + assert.Equal(t, "", recs[0].AccountName) + assert.Equal(t, "initial", recs[1].AccountName) + }) + + t.Run("Handles multiple accounts with caching", func(t *testing.T) { + mockOrg := &MockOrganizationsClient{} + // Should only call once per unique account + mockOrg.On("DescribeAccount", ctx, &organizations.DescribeAccountInput{ + AccountId: aws.String("111222333444"), + }).Return(&organizations.DescribeAccountOutput{ + Account: &orgtypes.Account{ + Name: aws.String("Dev"), + }, + }, nil).Once() + mockOrg.On("DescribeAccount", ctx, &organizations.DescribeAccountInput{ + AccountId: aws.String("555666777888"), + }).Return(&organizations.DescribeAccountOutput{ + Account: &orgtypes.Account{ + Name: aws.String("Staging"), + }, + }, nil).Once() + + cache := &TestAccountAliasCache{ + cache: make(map[string]string), + orgClient: mockOrg, + } + + recs := []common.Recommendation{ + {Account: "111222333444", AccountName: ""}, + {Account: "111222333444", AccountName: ""}, // Same account - should use cache + {Account: "555666777888", AccountName: ""}, + } + + // Simulate what populateAccountNames does + for i := range recs { + if recs[i].Account != "" { + recs[i].AccountName = cache.GetAccountAlias(ctx, recs[i].Account) + } + } + + assert.Equal(t, "Dev", recs[0].AccountName) + assert.Equal(t, "Dev", recs[1].AccountName) + assert.Equal(t, "Staging", recs[2].AccountName) + mockOrg.AssertExpectations(t) + }) +} diff --git a/cmd/multi_service_max_instances_test.go b/cmd/multi_service_max_instances_test.go new file mode 100644 index 000000000..2b113a9ae --- /dev/null +++ b/cmd/multi_service_max_instances_test.go @@ -0,0 +1,904 @@ +package main + +import ( + "context" + "path/filepath" + "testing" + + "github.com/LeanerCloud/CUDly/pkg/common" + "github.com/aws/aws-sdk-go-v2/aws" + "github.com/stretchr/testify/assert" + "github.com/stretchr/testify/mock" + "github.com/stretchr/testify/require" +) + +// maxInstancesFixtureServices and maxInstancesFixtureRegions describe the +// multi-service, multi-region fan-out the run-wide cap has to survive. A +// single-service or single-region fixture stays green with the per-region bug +// present, so both dimensions are needed. +var ( + maxInstancesFixtureServices = []common.ServiceType{common.ServiceRDS, common.ServiceElastiCache} + maxInstancesFixtureRegions = []string{"us-east-1", "us-west-2", "eu-west-1"} +) + +// countPerFixtureRec is the instance count each (service, region) pair returns. +const countPerFixtureRec = 8 + +// newMaxInstancesMockClientWith returns a recommendations client that answers +// one recommendation per (service, region) pair. savingsFor maps the fan-out +// position (service index, region index) to a savings percentage, which lets a +// test decide whether the best recommendations are fetched first or last. +func newMaxInstancesMockClientWith(count int, savingsFor func(serviceIdx, regionIdx int) float64) *MockRecommendationsClient { + mockClient := &MockRecommendationsClient{} + for i, svc := range maxInstancesFixtureServices { + for j, region := range maxInstancesFixtureRegions { + svc, region := svc, region + savings := savingsFor(i, j) + rec := common.Recommendation{ + Service: svc, + Region: region, + ResourceType: "db.t3.small", + Count: count, + EstimatedSavings: savings * 10, + SavingsPercentage: savings, + } + mockClient.On("GetRecommendations", mock.Anything, + mock.MatchedBy(func(p *common.RecommendationParams) bool { + return p != nil && p.Service == svc && p.Region == region + }), + ).Return([]common.Recommendation{rec}, nil).Once() + } + } + return mockClient +} + +// newMaxInstancesMockClient answers with savings descending in fan-out order +// (50, 45, 40, ...), so the best recommendations are also the first fetched. +func newMaxInstancesMockClient() *MockRecommendationsClient { + return newMaxInstancesMockClientWith(countPerFixtureRec, func(i, j int) float64 { + return 50.0 - 5*float64(i*len(maxInstancesFixtureRegions)+j) + }) +} + +// TestMaxInstancesCapsWholeRunAcrossServicesAndRegions is the regression test +// for #1608. +// +// --max-instances is documented (cmd/main.go, docs/cli/README.md, +// docs/cli/filtering.md) as a hard cap on the *total* number of instances +// purchased across all recommendations. It used to be applied inside the +// per-(service, region) fetch, so every pair independently kept up to +// MaxInstances and the run bought roughly cap x services x regions. +// +// This is the issue's scenario in miniature: 2 services x 3 regions, each +// answering with 8 instances (48 natural total), capped at 10. Pre-fix the run +// kept all 48 (4.8x the cap); post-fix the sum across every service and region +// is exactly 10. +func TestMaxInstancesCapsWholeRunAcrossServicesAndRegions(t *testing.T) { + const maxInstances = 10 + + ctx := context.Background() + awsCfg := aws.Config{Region: "us-east-1"} + + origCfg := toolCfg + t.Cleanup(func() { toolCfg = origCfg }) + + toolCfg.Coverage = 100.0 + toolCfg.PaymentOption = "partial-upfront" + toolCfg.TermYears = 1 + toolCfg.Regions = maxInstancesFixtureRegions + toolCfg.MaxInstances = maxInstances + + mockClient := newMaxInstancesMockClient() + t.Cleanup(func() { mockClient.AssertExpectations(t) }) + + accountCache := NewAccountAliasCache(awsCfg) + allRecs, drops := fetchAllRecs(ctx, awsCfg, mockClient, accountCache, + maxInstancesFixtureServices, engineVersionData{}, toolCfg, nil) + + pairs := len(maxInstancesFixtureServices) * len(maxInstancesFixtureRegions) + naturalTotal := pairs * countPerFixtureRec + + // The fetch stage must hand the *uncapped* set to the run-wide cap. If this + // fails, the cap has been pushed back down into the per-region path. + require.Len(t, allRecs, pairs, "fetch stage must not drop recommendations") + require.Equal(t, naturalTotal, CalculateTotalInstances(allRecs), + "fetch stage must not apply the cap per region") + + scored := scoreLimitAndDisplay(allRecs, toolCfg, drops) + + total := CalculateTotalInstances(scored.Passed) + assert.LessOrEqual(t, total, maxInstances, + "--max-instances must cap the sum across every service and region, got %d instances from %d service/region pairs", + total, pairs) + // The cap is a budget to spend, not just a ceiling: with 48 instances + // available it should be consumed exactly. + assert.Equal(t, maxInstances, total) + + // Highest-savings recommendations survive, run-wide: the 50% rec keeps all + // 8 instances and the 45% rec is reduced to the remaining 2. + require.Len(t, scored.Passed, 2) + assert.InDelta(t, 50.0, scored.Passed[0].SavingsPercentage, 0.001) + assert.Equal(t, countPerFixtureRec, scored.Passed[0].Count) + assert.InDelta(t, 45.0, scored.Passed[1].SavingsPercentage, 0.001) + assert.Equal(t, maxInstances-countPerFixtureRec, scored.Passed[1].Count) + + // No silent clamping: the four fully-dropped recommendations are counted + // into the end-of-run summary. + assert.Contains(t, drops.FormatOneLine(), common.DropMaxInstances+"=4") +} + +// TestMaxInstancesKeepsHighestSavingsNotFirstFetched pins the *selection* +// property of the cap, which TestMaxInstancesCapsWholeRunAcrossServicesAndRegions +// cannot see. +// +// ApplyInstanceLimit consumes its input in slice order and drops the tail, so +// whatever ordering it is handed decides which commitments are bought. Capping +// the merged set before scoring and capping the scorer's savings-sorted output +// both satisfy "total <= cap", so the total-focused test passes either way. +// Only this fixture distinguishes them: the first-fetched service and region +// carry the *worst* recommendations, so a cap that consumes in fetch order +// spends the entire budget on them and leaves the best ones unbought. +// +// RDS is fetched first (index 0 of maxInstancesFixtureServices) with 10-12% +// savings; ElastiCache is fetched second with 40-50%. The cap admits 10 of the +// 36 available instances, and every one of them must come from ElastiCache. +func TestMaxInstancesKeepsHighestSavingsNotFirstFetched(t *testing.T) { + const ( + maxInstances = 10 + countPerRec = 6 + ) + + ctx := context.Background() + awsCfg := aws.Config{Region: "us-east-1"} + + origCfg := toolCfg + t.Cleanup(func() { toolCfg = origCfg }) + + toolCfg.Coverage = 100.0 + toolCfg.PaymentOption = "partial-upfront" + toolCfg.TermYears = 1 + toolCfg.Regions = maxInstancesFixtureRegions + toolCfg.MaxInstances = maxInstances + toolCfg.MinSavingsPct = 0 // the low-savings recs must reach the cap, not be scored out + + // Worst-first: the first-fetched service gets 10/11/12%, the second gets + // 40/45/50%. Best-value selection and fetch-order selection now disagree. + mockClient := newMaxInstancesMockClientWith(countPerRec, func(i, j int) float64 { + if i == 0 { + return 10.0 + float64(j) + } + return 40.0 + 5*float64(j) + }) + t.Cleanup(func() { mockClient.AssertExpectations(t) }) + + accountCache := NewAccountAliasCache(awsCfg) + allRecs, drops := fetchAllRecs(ctx, awsCfg, mockClient, accountCache, + maxInstancesFixtureServices, engineVersionData{}, toolCfg, nil) + require.Len(t, allRecs, len(maxInstancesFixtureServices)*len(maxInstancesFixtureRegions)) + + scored := scoreLimitAndDisplay(allRecs, toolCfg, drops) + + // The total is respected under either placement, so it proves nothing on + // its own here. It is asserted only to keep the fixture honest. + require.Equal(t, maxInstances, CalculateTotalInstances(scored.Passed)) + + // The property under test: every purchased instance comes from the + // highest-savings recommendations run-wide, not from the ones fetched + // first. A pre-scoring cap spends the whole budget on the 10-11% RDS recs. + for i := range scored.Passed { + rec := scored.Passed[i] + assert.Equal(t, common.ServiceElastiCache, rec.Service, + "the cap must consume the scorer's savings-sorted order, not fetch order; "+ + "%s %s at %.0f%% savings was bought while better recommendations were dropped", + rec.Service, rec.Region, rec.SavingsPercentage) + assert.GreaterOrEqual(t, rec.SavingsPercentage, 40.0) + } + + // Exact survivors: the 50% rec keeps all 6 instances, the 45% rec is + // reduced to the remaining 4, and everything at or below 40% is dropped. + require.Len(t, scored.Passed, 2) + assert.InDelta(t, 50.0, scored.Passed[0].SavingsPercentage, 0.001) + assert.Equal(t, countPerRec, scored.Passed[0].Count) + assert.InDelta(t, 45.0, scored.Passed[1].SavingsPercentage, 0.001) + assert.Equal(t, maxInstances-countPerRec, scored.Passed[1].Count) +} + +// TestMaxInstancesNeverPurchasesBelowMinCount pins that the cap cannot deliver +// a purchase smaller than the operator's --min-count floor. +// +// Moving the cap downstream of the scorer means truncation is no longer +// re-filtered by the scorer's --min-count gate, so a survivor the cap trims to +// fit the budget could land under a floor the operator explicitly set. Asking +// for "at least 5" and being sold 2 is wrong regardless of which direction it +// errs in, and a commitment below the minimum can be worse than none at all. +// +// Fixture: 6 recommendations of 6 instances each, cap 10, --min-count 5. The +// budget leaves room for 6 + 4, and that 4 is below the floor, so the second +// recommendation must be dropped rather than purchased short. The run buys 6. +func TestMaxInstancesNeverPurchasesBelowMinCount(t *testing.T) { + const ( + maxInstances = 10 + minCount = 5 + countPerRec = 6 + ) + + ctx := context.Background() + awsCfg := aws.Config{Region: "us-east-1"} + + origCfg := toolCfg + t.Cleanup(func() { toolCfg = origCfg }) + + toolCfg.Coverage = 100.0 + toolCfg.PaymentOption = "partial-upfront" + toolCfg.TermYears = 1 + toolCfg.Regions = maxInstancesFixtureRegions + toolCfg.MaxInstances = maxInstances + toolCfg.MinCount = minCount + + mockClient := newMaxInstancesMockClientWith(countPerRec, func(i, j int) float64 { + return 50.0 - 5*float64(i*len(maxInstancesFixtureRegions)+j) + }) + t.Cleanup(func() { mockClient.AssertExpectations(t) }) + + accountCache := NewAccountAliasCache(awsCfg) + allRecs, drops := fetchAllRecs(ctx, awsCfg, mockClient, accountCache, + maxInstancesFixtureServices, engineVersionData{}, toolCfg, nil) + require.Len(t, allRecs, len(maxInstancesFixtureServices)*len(maxInstancesFixtureRegions)) + + scored := scoreLimitAndDisplay(allRecs, toolCfg, drops) + + // The property under test. Every purchased recommendation honors the + // floor; none is bought at a truncated count below it. + for i := range scored.Passed { + assert.GreaterOrEqual(t, scored.Passed[i].Count, minCount, + "--min-count %d was set, but %s %s would be purchased at count=%d", + minCount, scored.Passed[i].Service, scored.Passed[i].Region, scored.Passed[i].Count) + } + + // The truncated second recommendation is dropped, not reduced to 4. + require.Len(t, scored.Passed, 1) + assert.Equal(t, countPerRec, scored.Passed[0].Count) + assert.Equal(t, countPerRec, CalculateTotalInstances(scored.Passed)) + + // Dropped for the min-count reason, not silently, and counted exactly + // once: 5 of the 6 scored recommendations are gone, 1 of them to the + // floor and 4 to the budget. Double-counting the floor drop under both + // reasons would report 6 drops for 5 recommendations. + summary := drops.FormatOneLine() + assert.Contains(t, summary, common.DropMinCountAfterCap+"=1") + assert.Contains(t, summary, common.DropMaxInstances+"=4") + assert.Equal(t, 5, drops.Total()) +} + +// TestDropTruncatedBelowMinCount covers the floor helper directly, including +// the boundary (a truncated count exactly equal to the floor is kept) and the +// disabled case. +func TestDropTruncatedBelowMinCount(t *testing.T) { + tests := []struct { + name string + counts []int + minCount int + wantKept []int + wantRemoved int + }{ + {name: "floor disabled keeps everything", counts: []int{5, 2}, minCount: 0, wantKept: []int{5, 2}}, + {name: "truncated tail below floor is dropped", counts: []int{6, 4}, minCount: 5, wantKept: []int{6}, wantRemoved: 1}, + {name: "truncated tail exactly at floor is kept", counts: []int{6, 5}, minCount: 5, wantKept: []int{6, 5}}, + {name: "tail above floor is kept", counts: []int{6, 6}, minCount: 5, wantKept: []int{6, 6}}, + {name: "every rec below floor leaves nothing", counts: []int{2}, minCount: 5, wantKept: []int{}, wantRemoved: 1}, + } + + for _, tt := range tests { + t.Run(tt.name, func(t *testing.T) { + recs := make([]common.Recommendation, len(tt.counts)) + for i, c := range tt.counts { + recs[i] = common.Recommendation{Service: common.ServiceRDS, Region: "us-east-1", Count: c} + } + + kept, removed := dropTruncatedBelowMinCount(recs, tt.minCount) + + gotCounts := make([]int, len(kept)) + for i := range kept { + gotCounts[i] = kept[i].Count + } + assert.Equal(t, tt.wantKept, gotCounts) + assert.Len(t, removed, tt.wantRemoved) + }) + } +} + +// TestMaxInstancesNotAppliedWhenUnset guards the other direction: with the flag +// unset the fan-out is purchased in full, so the cap cannot silently shrink a +// run that never asked for one. +func TestMaxInstancesNotAppliedWhenUnset(t *testing.T) { + ctx := context.Background() + awsCfg := aws.Config{Region: "us-east-1"} + + origCfg := toolCfg + t.Cleanup(func() { toolCfg = origCfg }) + + toolCfg.Coverage = 100.0 + toolCfg.PaymentOption = "partial-upfront" + toolCfg.TermYears = 1 + toolCfg.Regions = maxInstancesFixtureRegions + toolCfg.MaxInstances = 0 + + mockClient := newMaxInstancesMockClient() + t.Cleanup(func() { mockClient.AssertExpectations(t) }) + + accountCache := NewAccountAliasCache(awsCfg) + allRecs, drops := fetchAllRecs(ctx, awsCfg, mockClient, accountCache, + maxInstancesFixtureServices, engineVersionData{}, toolCfg, nil) + + scored := scoreLimitAndDisplay(allRecs, toolCfg, drops) + + pairs := len(maxInstancesFixtureServices) * len(maxInstancesFixtureRegions) + assert.Len(t, scored.Passed, pairs) + assert.Equal(t, pairs*countPerFixtureRec, CalculateTotalInstances(scored.Passed)) + assert.NotContains(t, drops.FormatOneLine(), common.DropMaxInstances) +} + +func TestApplyGlobalInstanceLimit(t *testing.T) { + recs := []common.Recommendation{ + {Service: common.ServiceRDS, Region: "us-east-1", ResourceType: "db.t3.small", Count: 5}, + {Service: common.ServiceEC2, Region: "us-west-2", ResourceType: "m5.large", Count: 4}, + {Service: common.ServiceEC2, Region: "eu-west-1", ResourceType: "m5.xlarge", Count: 3}, + } + + tests := []struct { + name string + maxInstances int32 + expectedTotal int + expectedLen int + expectedDrops int + }{ + {name: "unset leaves the run untouched", maxInstances: 0, expectedTotal: 12, expectedLen: 3}, + {name: "cap above the total leaves the run untouched", maxInstances: 99, expectedTotal: 12, expectedLen: 3}, + {name: "cap truncates the tail", maxInstances: 7, expectedTotal: 7, expectedLen: 2, expectedDrops: 1}, + {name: "cap below the first rec keeps one reduced rec", maxInstances: 2, expectedTotal: 2, expectedLen: 1, expectedDrops: 2}, + {name: "cap equal to the total leaves the run untouched", maxInstances: 12, expectedTotal: 12, expectedLen: 3}, + } + + for _, tt := range tests { + t.Run(tt.name, func(t *testing.T) { + drops := common.NewDropSummary() + got := applyGlobalInstanceLimit(recs, Config{MaxInstances: tt.maxInstances}, rankBySavingsPercentage, drops) + + assert.Len(t, got, tt.expectedLen) + assert.Equal(t, tt.expectedTotal, CalculateTotalInstances(got)) + + if tt.expectedDrops == 0 { + assert.Empty(t, drops.FormatOneLine()) + return + } + assert.Contains(t, drops.FormatOneLine(), common.DropMaxInstances) + assert.Equal(t, tt.expectedDrops, drops.Total()) + }) + } + + // The input slice must not be mutated: the caller still renders it. + assert.Equal(t, 5, recs[0].Count) + assert.Equal(t, 4, recs[1].Count) + assert.Equal(t, 3, recs[2].Count) +} + +// TestApplyInstanceLimitNonPositiveCountDoesNotCreditBudget pins that a +// non-positive Count cannot raise the remaining budget and let later +// recommendations push the run past the cap. +func TestApplyInstanceLimitNonPositiveCountDoesNotCreditBudget(t *testing.T) { + recs := []common.Recommendation{ + {ResourceType: "a", Count: 4}, + {ResourceType: "b", Count: -10}, + {ResourceType: "c", Count: 100}, + } + + got := ApplyInstanceLimit(recs, 5) + + total := 0 + for i := range got { + if got[i].Count > 0 { + total += got[i].Count + } + } + assert.LessOrEqual(t, total, 5, "a negative Count must not raise the remaining budget") +} + +// TestReportInstanceLimitNamesEveryChange checks the no-silent-clamping +// contract: each reduced and each dropped recommendation is named on stdout. +func TestReportInstanceLimitNamesEveryChange(t *testing.T) { + before := []common.Recommendation{ + {Service: common.ServiceRDS, Region: "us-east-1", ResourceType: "db.t3.small", Count: 5}, + {Service: common.ServiceEC2, Region: "us-west-2", ResourceType: "m5.large", Count: 4}, + {Service: common.ServiceEC2, Region: "eu-west-1", ResourceType: "m5.xlarge", Count: 3}, + } + after := ApplyInstanceLimit(before, 7) + + drops := common.NewDropSummary() + out := captureAppOutput(t, func() { + reportInstanceLimit(before, after, 0, CalculateTotalInstances(before), 7, rankBySavingsPercentage, drops) + }) + + assert.Contains(t, out, "--max-instances=7") + assert.Contains(t, out, "reduced") + assert.Contains(t, out, "m5.large") + assert.Contains(t, out, "dropped") + assert.Contains(t, out, "m5.xlarge") +} + +// ==================== --input-csv path (#1741) ==================== + +// csvFixtureHeader matches writeMultiServiceCSVReport's column names for the +// fields these tests exercise, so the fixtures parse the way a real +// round-tripped report does. +const csvFixtureHeader = "Service,Region,ResourceType,Engine,Count,EstimatedSavings,Term,PaymentOption,Account\n" + +// csvNoSavingsHeader is csvFixtureHeader without the EstimatedSavings column, +// the shape of a hand-written minimal CSV. getCSVField returns "" for a column +// that is not in the header and parseCSVFloat leaves the field at zero, so +// every row of such a file loads with EstimatedSavings == 0. +const csvNoSavingsHeader = "Service,Region,ResourceType,Engine,Count,Term,PaymentOption,Account\n" + +// csvRecsFrom writes rows to a temporary recommendations CSV, loads them back +// through the production parser, and returns the recommendations exactly as a +// --input-csv run sees them. +// +// Building the fixture through loadRecommendationsFromCSV rather than by hand +// keeps these tests honest about what a CSV row actually carries: neither +// writeMultiServiceCSVReport nor parseCSVRecord knows a savings-percentage +// column, so every loaded row has SavingsPercentage == 0 and any +// savings-ordered selection has to resolve on EstimatedSavings. A hand-built +// fixture could quietly set SavingsPercentage and prove nothing about the real +// input. +func csvRecsFrom(t *testing.T, rows string) []common.Recommendation { + t.Helper() + return csvRecsFromHeader(t, csvFixtureHeader, rows) +} + +// csvRecsFromHeader is csvRecsFrom over an arbitrary header, for fixtures that +// exercise a CSV missing a column the selection depends on. +func csvRecsFromHeader(t *testing.T, header, rows string) []common.Recommendation { + t.Helper() + recs, err := loadRecommendationsFromCSV(writeTestRecommendationsCSV(t, header+rows)) + require.NoError(t, err) + require.NotEmpty(t, recs) + for i := range recs { + require.Zero(t, recs[i].SavingsPercentage, + "a CSV row carries no savings percentage; the ordering under test must come from EstimatedSavings") + } + return recs +} + +// TestCSVCapKeepsHighestSavingsNotFileOrder is the selection-order regression +// for #1741. +// +// filterAndAdjustRecommendations used to hand the load-ordered slice straight +// to ApplyInstanceLimit, which consumes its input in slice order and drops the +// tail, so whichever rows appeared first in the file spent the whole budget. +// Both the old and the new behavior respect the cap total, so only a fixture +// whose best-value rows are *not* first can tell them apart: the $10 row leads +// the file and the $500 row sits in the middle. +func TestCSVCapKeepsHighestSavingsNotFileOrder(t *testing.T) { + isolateAWSEnv(t) + const ( + maxInstances = 10 + countPerRow = 6 + ) + + recs := csvRecsFrom(t, `rds,us-east-1,db.t3.small,postgres,6,10.00,1yr,All Upfront,123456789012 +rds,us-east-1,db.t3.medium,postgres,6,500.00,1yr,All Upfront,123456789012 +rds,us-east-1,db.t3.large,postgres,6,100.00,1yr,All Upfront,123456789012 +`) + + var ( + got []common.Recommendation + err error + ) + out := captureAppOutput(t, func() { + got, err = filterAndAdjustRecommendations(recs, 100.0, Config{MaxInstances: maxInstances}) + }) + require.NoError(t, err) + + // Asserted only to keep the fixture honest: a file-ordered cap satisfies + // the total too, so the total on its own proves nothing here. + require.Equal(t, maxInstances, CalculateTotalInstances(got)) + + // The property under test is the selection ORDER. The budget must go to + // the highest-savings rows, best first, not to the rows the file happened + // to list first. + require.Len(t, got, 2) + assert.Equal(t, "db.t3.medium", got[0].ResourceType, + "the cap must consume savings order, not file order; got %s first", got[0].ResourceType) + assert.InDelta(t, 500.00, got[0].EstimatedSavings, 0.001) + assert.Equal(t, countPerRow, got[0].Count) + assert.Equal(t, "db.t3.large", got[1].ResourceType) + assert.Equal(t, maxInstances-countPerRow, got[1].Count) + // This row is truncated, so its savings are the file's figure scaled by + // the share of the row the budget actually buys (#1830). Asserting the + // unscaled 100.00 here is what the bug looked like. + assert.InDelta(t, 100.00*float64(maxInstances-countPerRow)/float64(countPerRow), + got[1].EstimatedSavings, 0.001) + + // Nothing shrinks silently: the row the cap dropped is named. + assert.Contains(t, out, "db.t3.small") +} + +// TestCSVCapDropsTruncationBelowMinCount pins that the CSV path cannot deliver +// a purchase smaller than the operator's --min-count floor, the property #1608 +// and #1725 established on the recommendation-driven path. +// +// Fixture: two rows of 6 instances, cap 10, floor 5. The budget leaves room +// for 6 + 4, and 4 is under the floor, so the second row is dropped rather +// than purchased short. The run buys 6. +func TestCSVCapDropsTruncationBelowMinCount(t *testing.T) { + isolateAWSEnv(t) + const ( + maxInstances = 10 + minCount = 5 + countPerRow = 6 + ) + + recs := csvRecsFrom(t, `rds,us-east-1,db.t3.large,postgres,6,100.00,1yr,All Upfront,123456789012 +rds,us-east-1,db.t3.medium,postgres,6,500.00,1yr,All Upfront,123456789012 +`) + + var ( + got []common.Recommendation + err error + ) + out := captureAppOutput(t, func() { + got, err = filterAndAdjustRecommendations(recs, 100.0, Config{MaxInstances: maxInstances, MinCount: minCount}) + }) + require.NoError(t, err) + + for i := range got { + assert.GreaterOrEqual(t, got[i].Count, minCount, + "--min-count %d was set, but %s would be purchased at count=%d", + minCount, got[i].ResourceType, got[i].Count) + } + + require.Len(t, got, 1) + assert.Equal(t, "db.t3.medium", got[0].ResourceType) + assert.Equal(t, countPerRow, got[0].Count) + assert.Equal(t, countPerRow, CalculateTotalInstances(got)) + assert.Contains(t, out, "below --min-count 5") +} + +// TestCSVMinCountDropsRowsUnderTheFloor pins the floor with no cap in play: a +// CSV row that is already under --min-count is dropped, not purchased. The +// higher-value row is the one under the floor, so a selection that sorts by +// savings without gating on the floor still buys it. +func TestCSVMinCountDropsRowsUnderTheFloor(t *testing.T) { + isolateAWSEnv(t) + const minCount = 5 + + recs := csvRecsFrom(t, `rds,us-east-1,db.t3.small,postgres,2,500.00,1yr,All Upfront,123456789012 +rds,us-east-1,db.t3.medium,postgres,6,100.00,1yr,All Upfront,123456789012 +`) + + var ( + got []common.Recommendation + err error + ) + out := captureAppOutput(t, func() { + got, err = filterAndAdjustRecommendations(recs, 100.0, Config{MinCount: minCount}) + }) + require.NoError(t, err) + + require.Len(t, got, 1, "the row at count=2 is below --min-count 5 and must not be purchased") + assert.Equal(t, "db.t3.medium", got[0].ResourceType) + assert.Equal(t, 6, got[0].Count) + assert.Contains(t, out, "--min-count dropped") + assert.Contains(t, out, "db.t3.small") +} + +// TestRunToolFromCSVEnforcesMinCountAndCap drives the guards end to end +// through the real --input-csv entry point and asserts on the purchase report, +// in both directions: the guards bind when they should, and a legitimate run +// under the same flags still purchases every row at its full count. +func TestRunToolFromCSVEnforcesMinCountAndCap(t *testing.T) { + origCfg := toolCfg + t.Cleanup(func() { toolCfg = origCfg }) + isolateAWSEnv(t) + + // Worst-value row first, so file order and savings order disagree. + csvPath := writeTestRecommendationsCSV(t, csvFixtureHeader+ + `rds,us-east-1,db.t3.small,postgres,6,10.00,1yr,All Upfront,123456789012 +rds,us-east-1,db.t3.medium,postgres,6,500.00,1yr,All Upfront,123456789012 +rds,us-east-1,db.t3.large,postgres,6,100.00,1yr,All Upfront,123456789012 +`) + + tests := []struct { + name string + wantRows []string + maxInstances int32 + minCount int + }{ + { + // Budget 10 across rows of 6: the best row takes 6, the next is + // truncated to 4, and 4 is under the floor, so it is dropped. + // A file-ordered cap would instead buy db.t3.small at 6. + name: "cap and floor bind", + maxInstances: 10, + minCount: 5, + wantRows: []string{"db.t3.medium@6"}, + }, + { + // Same floor, cap above the natural total: nothing is dropped and + // every row is purchased at its CSV count. A guard that + // over-blocks is as wrong as one that never fires. + name: "legitimate run purchases every row", + maxInstances: 100, + minCount: 5, + wantRows: []string{"db.t3.small@6", "db.t3.medium@6", "db.t3.large@6"}, + }, + } + + for _, tt := range tests { + t.Run(tt.name, func(t *testing.T) { + reportPath := filepath.Join(t.TempDir(), "report.csv") + toolCfg.CSVInput = csvPath + toolCfg.CSVOutput = reportPath + toolCfg.ActualPurchase = false + toolCfg.Coverage = 100.0 + toolCfg.TargetCoverage = 0 + toolCfg.OverrideCount = 0 + toolCfg.MaxInstances = tt.maxInstances + toolCfg.MinCount = tt.minCount + + require.NoError(t, runToolFromCSV(context.Background(), toolCfg)) + + header, rows := readPurchaseReport(t, reportPath) + assert.ElementsMatch(t, tt.wantRows, reportRowPairs(t, header, rows)) + }) + } +} + +// reportRowPairs pairs each purchase-report row's ResourceType with its Count. +// Asserting the two columns independently would pass a result that put the +// right counts against the wrong rows, which is precisely what a wrong +// selection order produces. +func reportRowPairs(t *testing.T, header []string, rows [][]string) []string { + t.Helper() + types := reportColumn(t, header, rows, "ResourceType") + counts := reportColumn(t, header, rows, "Count") + require.Len(t, counts, len(types)) + + pairs := make([]string, 0, len(types)) + for i := range types { + pairs = append(pairs, types[i]+"@"+counts[i]) + } + return pairs +} + +// TestCSVCapRanksBySavingsPerInstance pins the ranking key the cap consumes. +// +// --max-instances is a budget in instances, so the rows worth keeping are the +// ones returning the most savings per instance bought. Ranking on the +// EstimatedSavings column directly ranks a whole-row dollar total against a +// per-instance budget: the fixture's bulk row is larger in total ($600) and +// worse per instance ($6) than the rich row ($500 total, $83 each), so a +// total-ranked cap spends the entire 10-instance budget on bulk for about $60 +// of real value and drops rich outright. +// +// Every earlier fixture in this file uses equal counts, where ranking by total +// and ranking by rate are indistinguishable. The counts here are deliberately +// unequal, which is the only way to tell the two rules apart. +func TestCSVCapRanksBySavingsPerInstance(t *testing.T) { + isolateAWSEnv(t) + const maxInstances = 10 + + recs := csvRecsFrom(t, `rds,us-east-1,db.bulk.large,postgres,100,600.00,1yr,All Upfront,123456789012 +rds,us-east-1,db.rich.large,postgres,6,500.00,1yr,All Upfront,123456789012 +`) + + var ( + got []common.Recommendation + err error + ) + out := captureAppOutput(t, func() { + got, err = filterAndAdjustRecommendations(recs, 100.0, Config{MaxInstances: maxInstances}) + }) + require.NoError(t, err) + + require.Equal(t, maxInstances, CalculateTotalInstances(got), + "the cap total is respected by both the right and the wrong ranking, and is asserted only to keep the fixture honest") + + // The substance: ORDER, and the count each row is bought at. The + // best-value row must be taken whole before the bulk row gets the + // remainder. + require.Len(t, got, 2) + assert.Equal(t, "db.rich.large", got[0].ResourceType, + "the budget must go to the highest savings-per-instance row first; got %s", got[0].ResourceType) + assert.Equal(t, 6, got[0].Count) + assert.Equal(t, "db.bulk.large", got[1].ResourceType) + assert.Equal(t, maxInstances-6, got[1].Count, "the bulk row takes only the remaining budget") + + assert.Contains(t, out, "savings-per-instance", + "the cap must name the ranking rule it applied, not claim a savings percentage a CSV row does not carry") +} + +// TestCSVCapRefusesRowsWithoutARankingSignal covers the absent-vs-zero trap. +// +// A CSV with no EstimatedSavings column, or with a blank cell in it, loads +// every affected row at zero savings. Nothing downstream can tell that apart +// from a row genuinely worth $0, so a binding cap would rank on a value that +// is not there and silently fall through to the Service|Region|ResourceType +// tie-break, buying by instance-type name while stdout promises it is buying +// by savings. The run is refused instead. +// +// Both directions are covered: the two refusal cases, a case proving the +// refusal is scoped to a cap that actually binds, and a legitimate file that +// still ranks and caps normally. +func TestCSVCapRefusesRowsWithoutARankingSignal(t *testing.T) { + isolateAWSEnv(t) + + t.Run("no EstimatedSavings column at all", func(t *testing.T) { + // Worst alphabetically first, so a name-ordered cap is visible: it + // keeps db.aaa.large and drops db.zzz.large regardless of file order. + recs := csvRecsFromHeader(t, csvNoSavingsHeader, + `rds,us-east-1,db.zzz.large,postgres,6,1yr,All Upfront,123456789012 +rds,us-east-1,db.aaa.large,postgres,6,1yr,All Upfront,123456789012 +`) + + var err error + captureAppOutput(t, func() { + _, err = filterAndAdjustRecommendations(recs, 100.0, Config{MaxInstances: 6}) + }) + + require.Error(t, err, "a binding cap with nothing to rank on must refuse, not pick by name") + assert.Contains(t, err.Error(), "EstimatedSavings") + assert.Contains(t, err.Error(), "db.zzz.large", "the error must name the rows it cannot rank") + assert.Contains(t, err.Error(), "db.aaa.large") + }) + + t.Run("one blank EstimatedSavings cell among populated rows", func(t *testing.T) { + // The partial case is the worse one: the blank row loads as $0, ranks + // last, and is dropped first, with nothing saying the value was absent + // rather than genuinely zero. + recs := csvRecsFrom(t, `rds,us-east-1,db.t3.medium,postgres,6,500.00,1yr,All Upfront,123456789012 +rds,us-east-1,db.t3.small,postgres,6,,1yr,All Upfront,123456789012 +`) + + var err error + captureAppOutput(t, func() { + _, err = filterAndAdjustRecommendations(recs, 100.0, Config{MaxInstances: 10}) + }) + + require.Error(t, err) + assert.Contains(t, err.Error(), "db.t3.small", "the blank row must be named") + assert.NotContains(t, err.Error(), "db.t3.medium", "a row with a real savings value is rankable and must not be blamed") + }) + + t.Run("no savings column but the cap does not bind", func(t *testing.T) { + // Nothing has to be chosen between, so nothing has to be ranked. A + // guard that over-blocks is as wrong as one that never fires. + recs := csvRecsFromHeader(t, csvNoSavingsHeader, + `rds,us-east-1,db.zzz.large,postgres,6,1yr,All Upfront,123456789012 +rds,us-east-1,db.aaa.large,postgres,6,1yr,All Upfront,123456789012 +`) + + var ( + got []common.Recommendation + err error + ) + captureAppOutput(t, func() { + got, err = filterAndAdjustRecommendations(recs, 100.0, Config{MaxInstances: 12}) + }) + + require.NoError(t, err) + assert.Len(t, got, 2) + assert.Equal(t, 12, CalculateTotalInstances(got)) + }) + + t.Run("populated savings column still ranks and caps", func(t *testing.T) { + recs := csvRecsFrom(t, `rds,us-east-1,db.t3.small,postgres,6,10.00,1yr,All Upfront,123456789012 +rds,us-east-1,db.t3.medium,postgres,6,500.00,1yr,All Upfront,123456789012 +`) + + var ( + got []common.Recommendation + err error + ) + captureAppOutput(t, func() { + got, err = filterAndAdjustRecommendations(recs, 100.0, Config{MaxInstances: 6}) + }) + + require.NoError(t, err) + require.Len(t, got, 1) + assert.Equal(t, "db.t3.medium", got[0].ResourceType) + assert.Equal(t, 6, got[0].Count) + }) +} + +// TestApplyMinCountFloorAfterDuplicateAdjustment covers the floor's second +// call site, the one inside runToolFromCSV's purchase loop. +// +// adjustRecsForDuplicates subtracts recent matching commitments from Count +// after filterAndAdjustRecommendations has already applied the floor, so a row +// that cleared --min-count 5 at Count 6 with 5 existing commitments arrives at +// the purchase loop at Count 1. Without a second application the operator's +// floor is defeated on the very path #1741 is about. +// +// This exercises applyMinCountFloor on the post-deduction counts rather than +// through runToolFromCSV, because createServiceClient is not injectable: there +// is no seam to drive a fake GetExistingCommitments through the CSV entry +// point, and with invalid credentials adjustRecsForDuplicates errors and falls +// back to the unadjusted slice. +func TestApplyMinCountFloorAfterDuplicateAdjustment(t *testing.T) { + const minCount = 5 + + // Counts as adjustRecsForDuplicates leaves them: the first row had 6 and + // 5 existing commitments deducted, the second was untouched. + adjusted := []common.Recommendation{ + {Service: common.ServiceRDS, Region: "us-east-1", ResourceType: "db.t3.large", Count: 1, EstimatedSavings: 500}, + {Service: common.ServiceRDS, Region: "us-east-1", ResourceType: "db.t3.medium", Count: 6, EstimatedSavings: 100}, + } + + // Snapshot before the first call. applyMinCountFloor returns its input slice + // untouched at a floor of 0, and scorer.Score may reorder adjusted below, so + // asserting against adjusted itself would compare the result to itself. + original := append([]common.Recommendation(nil), adjusted...) + + var got []common.Recommendation + out := captureAppOutput(t, func() { + got = applyMinCountFloor(adjusted, minCount) + }) + + require.Len(t, got, 1, "the row the deduction left at 1 is below --min-count 5 and must not be purchased") + assert.Equal(t, "db.t3.medium", got[0].ResourceType) + assert.Equal(t, 6, got[0].Count) + assert.Contains(t, out, "db.t3.large", "the drop must be named") + assert.Contains(t, out, "--min-count") + + // A floor of 0 disables the flag: every row survives, in the order given. + var unfiltered []common.Recommendation + quiet := captureAppOutput(t, func() { + unfiltered = applyMinCountFloor(adjusted, 0) + }) + assert.Equal(t, original, unfiltered, "--min-count 0 must not drop or reorder anything") + assert.Empty(t, quiet) +} + +// TestCSVCapOrderIsIndependentOfFileOrder pins that equal-rate rows are not +// separated by where they sit in the file. +// +// --min-count 0 skips the scorer, so the per-instance sort is the only +// ordering the cap sees. If it left equal rates in input order, file order +// would still decide which of two equally-valuable rows the budget buys, which +// is the defect #1741 reported, surviving one layer down. +func TestCSVCapOrderIsIndependentOfFileOrder(t *testing.T) { + isolateAWSEnv(t) + + // Two rows at the identical rate ($100 for 6 instances), listed + // worst-alphabetically-first. + rows := []string{ + "rds,us-east-1,db.t3.zulu,postgres,6,100.00,1yr,All Upfront,123456789012\n", + "rds,us-east-1,db.t3.alpha,postgres,6,100.00,1yr,All Upfront,123456789012\n", + } + + selection := func(t *testing.T, rows []string) []string { + t.Helper() + recs := csvRecsFrom(t, rows[0]+rows[1]) + var ( + got []common.Recommendation + err error + ) + captureAppOutput(t, func() { + got, err = filterAndAdjustRecommendations(recs, 100.0, Config{MaxInstances: 6}) + }) + require.NoError(t, err) + + names := make([]string, 0, len(got)) + for i := range got { + names = append(names, got[i].ResourceType) + } + return names + } + + forward := selection(t, rows) + reversed := selection(t, []string{rows[1], rows[0]}) + + assert.Equal(t, []string{"db.t3.alpha"}, forward, + "equal rates must tie-break on Service|Region|ResourceType, not on file position") + assert.Equal(t, forward, reversed, "reversing the file must not change what the cap buys") +} diff --git a/cmd/multi_service_stats.go b/cmd/multi_service_stats.go new file mode 100644 index 000000000..fe7ee024d --- /dev/null +++ b/cmd/multi_service_stats.go @@ -0,0 +1,246 @@ +package main + +import ( + "github.com/LeanerCloud/CUDly/pkg/common" +) + +// ServiceProcessingStats holds statistics for each service. +type ServiceProcessingStats struct { + Service common.ServiceType + RegionsProcessed int + RecommendationsFound int + RecommendationsSelected int + InstancesProcessed int + SuccessfulPurchases int + FailedPurchases int + TotalEstimatedSavings float64 +} + +// calculateServiceStats calculates statistics for a service based on recommendations and results. +func calculateServiceStats(service common.ServiceType, recs []common.Recommendation, results []common.PurchaseResult) ServiceProcessingStats { + stats := ServiceProcessingStats{ + Service: service, + RecommendationsFound: len(recs), + RecommendationsSelected: len(recs), + } + + regionSet := make(map[string]bool) + for _rvc := range recs { + rec := recs[_rvc] + regionSet[rec.Region] = true + stats.InstancesProcessed += rec.Count + stats.TotalEstimatedSavings += rec.EstimatedSavings + } + stats.RegionsProcessed = len(regionSet) + + for _rvc := range results { + result := results[_rvc] + if result.Success { + stats.SuccessfulPurchases++ + } else { + stats.FailedPurchases++ + } + } + + return stats +} + +// printServiceSummary prints a summary for a single service. +func printServiceSummary(service common.ServiceType, stats ServiceProcessingStats) { + AppLogger.Printf("\n📊 %s Summary:\n", getServiceDisplayName(service)) + AppLogger.Printf(" Regions processed: %d\n", stats.RegionsProcessed) + AppLogger.Printf(" Recommendations: %d\n", stats.RecommendationsSelected) + AppLogger.Printf(" Instances: %d\n", stats.InstancesProcessed) + AppLogger.Printf(" Successful: %d, Failed: %d\n", stats.SuccessfulPurchases, stats.FailedPurchases) + if stats.TotalEstimatedSavings > 0 { + AppLogger.Printf(" Estimated monthly savings: $%.2f\n", stats.TotalEstimatedSavings) + } +} + +// printMultiServiceSummary prints the final summary for all services. +func printMultiServiceSummary(allRecommendations []common.Recommendation, allResults []common.PurchaseResult, serviceStats map[common.ServiceType]ServiceProcessingStats, isDryRun bool) { //nolint:unparam // param intentional for interface consistency/future use + printSummaryHeader(isDryRun) + + spStats, riStats, riAggregates := separateAndAggregateStats(serviceStats) + + printReservedInstancesSection(riStats, riAggregates) + + if spStats.RecommendationsSelected > 0 { + printSavingsPlansSection(allRecommendations, spStats) + } + + if len(riStats) > 0 && spStats.RecommendationsSelected > 0 { + printComparisonSection(allRecommendations, riStats, riAggregates.savings) + } + + printSuccessRate(riAggregates.success, riAggregates.failed) + printFinalMessage(isDryRun, riAggregates.success) +} + +// riAggregateStats holds aggregated RI statistics. +type riAggregateStats struct { + recommendations int + instances int + savings float64 + success int + failed int +} + +// printSummaryHeader prints the summary header with mode indication. +func printSummaryHeader(isDryRun bool) { + AppLogger.Println("\n🎯 Final Summary:") + AppLogger.Println("==========================================") + if isDryRun { + AppLogger.Println("Mode: DRY RUN") + } else { + AppLogger.Println("Mode: ACTUAL PURCHASE") + } +} + +// separateAndAggregateStats separates SP from RI stats and aggregates RI totals. +func separateAndAggregateStats(serviceStats map[common.ServiceType]ServiceProcessingStats) (ServiceProcessingStats, map[common.ServiceType]ServiceProcessingStats, riAggregateStats) { + spStats := ServiceProcessingStats{} + riStats := make(map[common.ServiceType]ServiceProcessingStats) + aggregates := riAggregateStats{} + + // Aggregate Savings Plans stats across all matching slugs (legacy + // umbrella + the four per-plan-type slugs from issue #22). Pre-split, + // only one SP slug existed so a plain assignment was correct; post- + // split, multiple SP slugs can land in serviceStats and Go map + // iteration order is non-deterministic, so a last-write-wins + // assignment would discard data unpredictably. Sum the relevant + // counters into spStats so the printed Savings Plans summary + // reflects every plan type's contribution. + for service, stats := range serviceStats { + if common.IsSavingsPlan(service) { + if spStats.Service == "" { + spStats.Service = common.ServiceSavingsPlansAll + } + if stats.RegionsProcessed > spStats.RegionsProcessed { + spStats.RegionsProcessed = stats.RegionsProcessed + } + spStats.RecommendationsFound += stats.RecommendationsFound + spStats.RecommendationsSelected += stats.RecommendationsSelected + spStats.InstancesProcessed += stats.InstancesProcessed + spStats.SuccessfulPurchases += stats.SuccessfulPurchases + spStats.FailedPurchases += stats.FailedPurchases + spStats.TotalEstimatedSavings += stats.TotalEstimatedSavings + } else { + riStats[service] = stats + aggregates.recommendations += stats.RecommendationsSelected + aggregates.instances += stats.InstancesProcessed + aggregates.savings += stats.TotalEstimatedSavings + aggregates.success += stats.SuccessfulPurchases + aggregates.failed += stats.FailedPurchases + } + } + + return spStats, riStats, aggregates +} + +// printReservedInstancesSection prints the RI section with per-service and total stats. +func printReservedInstancesSection(riStats map[common.ServiceType]ServiceProcessingStats, aggregates riAggregateStats) { + if len(riStats) == 0 { + return + } + + AppLogger.Println("\n💰 RESERVED INSTANCES:") + AppLogger.Println("--------------------------------------------------") + for service, stats := range riStats { + AppLogger.Printf("%-15s | Recs: %3d | Instances: %3d | Savings: $%8.2f/mo\n", + getServiceDisplayName(service), + stats.RecommendationsSelected, + stats.InstancesProcessed, + stats.TotalEstimatedSavings) + } + AppLogger.Printf("%-15s | Recs: %3d | Instances: %3d | Savings: $%8.2f/mo\n", + "TOTAL RIs", + aggregates.recommendations, + aggregates.instances, + aggregates.savings) +} + +// printSuccessRate prints the overall success rate if results exist. +func printSuccessRate(success, failed int) { + totalResults := success + failed + if totalResults > 0 { + successRate := (float64(success) / float64(totalResults)) * 100 + AppLogger.Printf("\nOverall success rate: %.1f%%\n", successRate) + } +} + +// archeraSignupURL is the Archera signup link with CUDly attribution. Now an +// alias for the shared constant in pkg/common, which the MCP server's +// post-purchase offer reads too, so the link and the two disclosures below +// are written once rather than once per binary. Kept identical to the +// frontend's ARCHERA_SIGNUP_URL (frontend/src/archera.ts), which cannot +// import Go. +const archeraSignupURL = common.ArcheraSignupURL + +// printFinalMessage prints the final message based on mode and results. +func printFinalMessage(isDryRun bool, riSuccess int) { + if isDryRun { + AppLogger.Println("\n💡 To actually purchase these RIs, run with --purchase flag") + AppLogger.Println(" Note: Savings Plans purchasing not yet implemented") + } else if riSuccess > 0 { + AppLogger.Println("\n🎉 Purchase operations completed!") + AppLogger.Println("⏰ Allow up to 15 minutes for RIs to appear in your account") + printArcheraPitch() + } +} + +// printArcheraPitch prints a soft, non-blocking suggestion to insure the +// just-purchased commitments through Archera, followed by the two Archera +// partnership disclosures CUDly commits to keeping visible everywhere the +// integration is surfaced (see project_archera_partnership memory): +// 1. Non-gating: CUDly works fully without Archera. +// 2. Sponsorship: Archera funds part of CUDly's development. +// +// Shown only after a successful real purchase. Archera is a third-party +// service independent of CUDly. +func printArcheraPitch() { + AppLogger.Println("\n🛡️ Want to push your coverage to 100% without the risk that a future") + AppLogger.Println(" capacity decrease leaves you paying for commitments you no longer use?") + AppLogger.Println(" You can buy underutilization insurance for Reserved Instances and") + AppLogger.Println(" Savings Plans from Archera by signing up at:") + AppLogger.Printf(" %s\n", archeraSignupURL) + AppLogger.Printf(" within the first %d days of the purchase.\n", common.ArcheraEnrollmentWindowDays) + AppLogger.Printf("\n %s\n", common.ArcheraNonGatingDisclosure) + AppLogger.Printf("\n %s\n", common.ArcheraSponsorshipDisclosure) +} + +// printSavingsPlansSection prints the Savings Plans summary section. +func printSavingsPlansSection(allRecommendations []common.Recommendation, spStats ServiceProcessingStats) { //nolint:unparam // param intentional for interface consistency/future use + AppLogger.Println("\n📊 SAVINGS PLANS:") + AppLogger.Println("--------------------------------------------------") + + // Categorize recommendations by SP type + breakdown := categorizeSPRecommendations(allRecommendations) + + // Print summary for each type + printSPTypeSummaries(breakdown) + + // Show best options by category + printBestSPOptions(breakdown) +} + +// printComparisonSection prints the comparison between RIs and Savings Plans. +func printComparisonSection(allRecommendations []common.Recommendation, riStats map[common.ServiceType]ServiceProcessingStats, riSavings float64) { + AppLogger.Println("\n🔄 COMPARISON:") + AppLogger.Println("--------------------------------------------------") + + // Collect SP savings by type + spSavings := collectSPSavings(allRecommendations) + + // Collect RI savings by service + risByService := collectRISavings(riStats) + + // Calculate comparison options + opts := calculateComparisonOptions(riSavings, spSavings, risByService) + + // Print all options + printComparisonOptions(opts) + + // Determine and print the best option + determineBestOption(opts) +} diff --git a/cmd/multi_service_stats_helpers.go b/cmd/multi_service_stats_helpers.go new file mode 100644 index 000000000..62f23aa6b --- /dev/null +++ b/cmd/multi_service_stats_helpers.go @@ -0,0 +1,237 @@ +package main + +import ( + "github.com/LeanerCloud/CUDly/pkg/common" +) + +// SPTypeBreakdown holds savings information broken down by Savings Plan type. +type SPTypeBreakdown struct { + ComputeSavings float64 + EC2InstanceSavings float64 + SageMakerSavings float64 + DatabaseSavings float64 + ComputeCount int + EC2InstanceCount int + SageMakerCount int + DatabaseCount int +} + +// categorizeSPRecommendations categorizes Savings Plan recommendations by type. +func categorizeSPRecommendations(recommendations []common.Recommendation) SPTypeBreakdown { + breakdown := SPTypeBreakdown{} + + for _rvc := range recommendations { + rec := recommendations[_rvc] + if common.IsSavingsPlan(rec.Service) { + if details, ok := rec.Details.(*common.SavingsPlanDetails); ok { + switch details.PlanType { + case "Compute": + breakdown.ComputeSavings += rec.EstimatedSavings + breakdown.ComputeCount++ + case "EC2Instance": + breakdown.EC2InstanceSavings += rec.EstimatedSavings + breakdown.EC2InstanceCount++ + case "SageMaker": + breakdown.SageMakerSavings += rec.EstimatedSavings + breakdown.SageMakerCount++ + case "Database": + breakdown.DatabaseSavings += rec.EstimatedSavings + breakdown.DatabaseCount++ + } + } + } + } + + return breakdown +} + +// printSPTypeSummaries prints the summary for each Savings Plan type. +func printSPTypeSummaries(breakdown SPTypeBreakdown) { + if breakdown.ComputeCount > 0 { + AppLogger.Printf(" Compute SP | Recs: %3d | Covers: EC2, Fargate, Lambda | $%8.2f/mo\n", + breakdown.ComputeCount, breakdown.ComputeSavings) + } + if breakdown.EC2InstanceCount > 0 { + AppLogger.Printf(" EC2 Inst SP | Recs: %3d | Covers: EC2 only (better rate) | $%8.2f/mo\n", + breakdown.EC2InstanceCount, breakdown.EC2InstanceSavings) + } + if breakdown.SageMakerCount > 0 { + AppLogger.Printf(" SageMaker SP | Recs: %3d | Covers: SageMaker instances | $%8.2f/mo\n", + breakdown.SageMakerCount, breakdown.SageMakerSavings) + } + if breakdown.DatabaseCount > 0 { + AppLogger.Printf(" Database SP | Recs: %3d | Covers: RDS, Aurora, ElastiCache, etc. | $%8.2f/mo\n", + breakdown.DatabaseCount, breakdown.DatabaseSavings) + } +} + +// printBestSPOptions prints the best Savings Plan options by category. +func printBestSPOptions(breakdown SPTypeBreakdown) { + AppLogger.Println() + + // Best for EC2/Compute + if breakdown.EC2InstanceSavings > 0 || breakdown.ComputeSavings > 0 { + if breakdown.EC2InstanceSavings > breakdown.ComputeSavings { + AppLogger.Printf(" ⭐ Best for EC2: EC2 Instance SP ($%.2f/mo)\n", breakdown.EC2InstanceSavings) + } else if breakdown.ComputeSavings > 0 { + AppLogger.Printf(" ⭐ Best for Compute: Compute SP ($%.2f/mo) - more flexible\n", breakdown.ComputeSavings) + } + } + + // Best for Databases + if breakdown.DatabaseSavings > 0 { + AppLogger.Printf(" ⭐ Best for Databases: Database SP ($%.2f/mo)\n", breakdown.DatabaseSavings) + } + + // Best for ML + if breakdown.SageMakerSavings > 0 { + AppLogger.Printf(" ⭐ Best for ML: SageMaker SP ($%.2f/mo)\n", breakdown.SageMakerSavings) + } +} + +// SPSavingsByType holds Savings Plan savings categorized by plan type. +type SPSavingsByType struct { + EC2SPSavings float64 + ComputeSPSavings float64 + DatabaseSPSavings float64 +} + +// collectSPSavings collects Savings Plan savings by type. +func collectSPSavings(recommendations []common.Recommendation) SPSavingsByType { + savings := SPSavingsByType{} + + for _rvc := range recommendations { + rec := recommendations[_rvc] + if common.IsSavingsPlan(rec.Service) { + if details, ok := rec.Details.(*common.SavingsPlanDetails); ok { + switch details.PlanType { + case "EC2Instance": + savings.EC2SPSavings += rec.EstimatedSavings + case "Compute": + savings.ComputeSPSavings += rec.EstimatedSavings + case "Database": + savings.DatabaseSPSavings += rec.EstimatedSavings + } + } + } + } + + return savings +} + +// RISavingsByService holds Reserved Instance savings categorized by service. +type RISavingsByService struct { + EC2RISavings float64 + DBRISavings float64 +} + +// collectRISavings collects Reserved Instance savings by service. +func collectRISavings(riStats map[common.ServiceType]ServiceProcessingStats) RISavingsByService { + savings := RISavingsByService{} + + // EC2 RIs + if stats, ok := riStats[common.ServiceEC2]; ok { + savings.EC2RISavings = stats.TotalEstimatedSavings + } + + // Database RIs (RDS, ElastiCache, MemoryDB, Redshift) + for service, stats := range riStats { + if service == common.ServiceRDS || service == common.ServiceElastiCache || + service == common.ServiceMemoryDB || service == common.ServiceRedshift { + savings.DBRISavings += stats.TotalEstimatedSavings + } + } + + return savings +} + +// ComparisonOptions holds the calculated savings for different purchasing options. +type ComparisonOptions struct { + BestComputeSPName string + Option1Savings float64 + Option2Savings float64 + Option3Savings float64 + BestComputeSP float64 + HasDatabaseSP bool +} + +// calculateComparisonOptions calculates savings for all comparison options. +func calculateComparisonOptions(riSavings float64, spSavings SPSavingsByType, risByService RISavingsByService) ComparisonOptions { + opts := ComparisonOptions{ + Option1Savings: riSavings, + HasDatabaseSP: spSavings.DatabaseSPSavings > 0, + } + + // Determine best compute SP + opts.BestComputeSP = spSavings.EC2SPSavings + opts.BestComputeSPName = "EC2 Instance SP" + if spSavings.ComputeSPSavings > spSavings.EC2SPSavings { + opts.BestComputeSP = spSavings.ComputeSPSavings + opts.BestComputeSPName = "Compute SP" + } + + // Option 2: Best compute SP + non-EC2 RIs + opts.Option2Savings = riSavings - risByService.EC2RISavings + opts.BestComputeSP + + // Option 3: Compute SP + Database SP (if available) + if opts.HasDatabaseSP { + opts.Option3Savings = riSavings - risByService.EC2RISavings - risByService.DBRISavings + + opts.BestComputeSP + spSavings.DatabaseSPSavings + } + + return opts +} + +// printComparisonOptions prints all comparison options. +func printComparisonOptions(opts ComparisonOptions) { + // Option 1: All RIs + AppLogger.Printf("Option 1 (All RIs):\n") + AppLogger.Printf(" Total monthly savings: $%.2f\n", opts.Option1Savings) + AppLogger.Printf(" Pros: Highest discount for specific instance types\n") + AppLogger.Printf(" Cons: Less flexible, locked to instance family/engine\n") + + // Option 2: Best compute SP + non-EC2 RIs + AppLogger.Printf("\nOption 2 (%s for compute + RIs for databases):\n", opts.BestComputeSPName) + AppLogger.Printf(" Total monthly savings: $%.2f\n", opts.Option2Savings) + AppLogger.Printf(" Pros: Flexible compute (can change EC2 families)\n") + AppLogger.Printf(" Cons: DB RIs still locked to engine/instance type\n") + + // Option 3: If we have Database SP recommendations + if opts.HasDatabaseSP { + AppLogger.Printf("\nOption 3 (%s + Database SP):\n", opts.BestComputeSPName) + AppLogger.Printf(" Total monthly savings: $%.2f\n", opts.Option3Savings) + AppLogger.Printf(" Pros: Maximum flexibility for both compute and databases\n") + AppLogger.Printf(" Cons: May have slightly lower discount than targeted RIs\n") + } +} + +// determineBestOption determines and prints the best purchasing option. +func determineBestOption(opts ComparisonOptions) { + if !opts.HasDatabaseSP { + // Only 2 options available + if opts.Option2Savings > opts.Option1Savings { + AppLogger.Printf("\n ⭐ RECOMMENDATION: Use Option 2 (saves $%.2f/mo more)\n", + opts.Option2Savings-opts.Option1Savings) + } else { + AppLogger.Printf("\n ⭐ RECOMMENDATION: Use Option 1 (saves $%.2f/mo more)\n", + opts.Option1Savings-opts.Option2Savings) + } + return + } + + // All 3 options available - find the best + best := "Option 1 (All RIs)" + bestSavings := opts.Option1Savings + + if opts.Option2Savings > bestSavings { + best = "Option 2 (Compute SP + DB RIs)" + bestSavings = opts.Option2Savings + } + + if opts.Option3Savings > bestSavings { + best = "Option 3 (Compute SP + Database SP)" + bestSavings = opts.Option3Savings + } + + AppLogger.Printf("\n ⭐ RECOMMENDATION: %s ($%.2f/mo)\n", best, bestSavings) +} diff --git a/cmd/multi_service_stats_test.go b/cmd/multi_service_stats_test.go new file mode 100644 index 000000000..32c7839b0 --- /dev/null +++ b/cmd/multi_service_stats_test.go @@ -0,0 +1,560 @@ +package main + +import ( + "bytes" + "fmt" + "io" + "log" + "os" + "testing" + + "github.com/LeanerCloud/CUDly/pkg/common" + "github.com/stretchr/testify/assert" +) + +// captureAppOutput captures output from AppLogger and returns the captured string. +// Usage: output := captureAppOutput(t, func() { printSomething() }). +func captureAppOutput(t *testing.T, fn func()) string { + t.Helper() + old := os.Stdout + oldLogger := AppLogger + r, w, err := os.Pipe() + if err != nil { + t.Fatalf("os.Pipe failed: %v", err) + } + os.Stdout = w + AppLogger = log.New(w, "", 0) + + defer func() { + os.Stdout = old + AppLogger = oldLogger + }() + + fn() + + w.Close() // Must close writer before reading to unblock io.Copy + var buf bytes.Buffer + _, _ = io.Copy(&buf, r) + return buf.String() +} + +func TestCalculateServiceStats(t *testing.T) { + tests := []struct { + name string + service common.ServiceType + recs []common.Recommendation + results []common.PurchaseResult + expected ServiceProcessingStats + }{ + { + name: "Empty inputs", + service: common.ServiceRDS, + recs: []common.Recommendation{}, + results: []common.PurchaseResult{}, + expected: ServiceProcessingStats{ + Service: common.ServiceRDS, + RegionsProcessed: 0, + RecommendationsFound: 0, + RecommendationsSelected: 0, + InstancesProcessed: 0, + SuccessfulPurchases: 0, + FailedPurchases: 0, + TotalEstimatedSavings: 0, + }, + }, + { + name: "Multiple regions with mixed results", + service: common.ServiceEC2, + recs: []common.Recommendation{ + {Region: "us-east-1", Count: 2, EstimatedSavings: 100}, + {Region: "us-west-2", Count: 3, EstimatedSavings: 200}, + {Region: "eu-west-1", Count: 1, EstimatedSavings: 50}, + }, + results: []common.PurchaseResult{ + {Success: true}, + {Success: true}, + {Success: false}, + }, + expected: ServiceProcessingStats{ + Service: common.ServiceEC2, + RegionsProcessed: 3, + RecommendationsFound: 3, + RecommendationsSelected: 3, + InstancesProcessed: 6, + SuccessfulPurchases: 2, + FailedPurchases: 1, + TotalEstimatedSavings: 350, + }, + }, + { + name: "Same region multiple recommendations", + service: common.ServiceElastiCache, + recs: []common.Recommendation{ + {Region: "us-east-1", Count: 1, EstimatedSavings: 100}, + {Region: "us-east-1", Count: 2, EstimatedSavings: 200}, + {Region: "us-east-1", Count: 3, EstimatedSavings: 300}, + }, + results: []common.PurchaseResult{ + {Success: true}, + {Success: true}, + {Success: true}, + }, + expected: ServiceProcessingStats{ + Service: common.ServiceElastiCache, + RegionsProcessed: 1, + RecommendationsFound: 3, + RecommendationsSelected: 3, + InstancesProcessed: 6, + SuccessfulPurchases: 3, + FailedPurchases: 0, + TotalEstimatedSavings: 600, + }, + }, + } + + for _, tt := range tests { + t.Run(tt.name, func(t *testing.T) { + result := calculateServiceStats(tt.service, tt.recs, tt.results) + assert.Equal(t, tt.expected, result) + }) + } +} + +func TestPrintServiceSummary(t *testing.T) { + tests := []struct { + name string + service common.ServiceType + stats ServiceProcessingStats + }{ + { + name: "With savings", + service: common.ServiceRDS, + stats: ServiceProcessingStats{ + Service: common.ServiceRDS, + RegionsProcessed: 2, + RecommendationsSelected: 5, + InstancesProcessed: 10, + SuccessfulPurchases: 4, + FailedPurchases: 1, + TotalEstimatedSavings: 1500.50, + }, + }, + { + name: "Without savings", + service: common.ServiceEC2, + stats: ServiceProcessingStats{ + Service: common.ServiceEC2, + RegionsProcessed: 1, + RecommendationsSelected: 0, + InstancesProcessed: 0, + SuccessfulPurchases: 0, + FailedPurchases: 0, + TotalEstimatedSavings: 0, + }, + }, + } + + for _, tt := range tests { + t.Run(tt.name, func(t *testing.T) { + output := captureAppOutput(t, func() { + printServiceSummary(tt.service, tt.stats) + }) + + // Verify output contains expected information + assert.Contains(t, output, getServiceDisplayName(tt.service)) + assert.Contains(t, output, fmt.Sprintf("Regions processed: %d", tt.stats.RegionsProcessed)) + assert.Contains(t, output, fmt.Sprintf("Recommendations: %d", tt.stats.RecommendationsSelected)) + assert.Contains(t, output, fmt.Sprintf("Instances: %d", tt.stats.InstancesProcessed)) + + if tt.stats.TotalEstimatedSavings > 0 { + assert.Contains(t, output, fmt.Sprintf("$%.2f", tt.stats.TotalEstimatedSavings)) + } + }) + } +} + +func TestPrintMultiServiceSummary(t *testing.T) { + tests := []struct { + stats map[common.ServiceType]ServiceProcessingStats + name string + recs []common.Recommendation + results []common.PurchaseResult + isDryRun bool + }{ + { + name: "Dry run with multiple services", + recs: []common.Recommendation{ + {Service: common.ServiceRDS, Count: 2}, + {Service: common.ServiceEC2, Count: 3}, + }, + results: []common.PurchaseResult{ + {Success: true, Recommendation: common.Recommendation{Count: 2}}, + {Success: false, Recommendation: common.Recommendation{Count: 3}}, + }, + stats: map[common.ServiceType]ServiceProcessingStats{ + common.ServiceRDS: { + Service: common.ServiceRDS, + RecommendationsSelected: 1, + InstancesProcessed: 2, + SuccessfulPurchases: 1, + TotalEstimatedSavings: 500.0, + }, + common.ServiceEC2: { + Service: common.ServiceEC2, + RecommendationsSelected: 1, + InstancesProcessed: 3, + FailedPurchases: 1, + TotalEstimatedSavings: 300.0, + }, + }, + isDryRun: true, + }, + { + name: "Actual purchase with success", + recs: []common.Recommendation{ + {Service: common.ServiceElastiCache, Count: 5}, + }, + results: []common.PurchaseResult{ + {Success: true, Recommendation: common.Recommendation{Count: 5}}, + }, + stats: map[common.ServiceType]ServiceProcessingStats{ + common.ServiceElastiCache: { + Service: common.ServiceElastiCache, + RecommendationsSelected: 1, + InstancesProcessed: 5, + SuccessfulPurchases: 1, + TotalEstimatedSavings: 1000.0, + }, + }, + isDryRun: false, + }, + { + name: "Empty results", + recs: []common.Recommendation{}, + results: []common.PurchaseResult{}, + stats: map[common.ServiceType]ServiceProcessingStats{}, + isDryRun: true, + }, + } + + for _, tt := range tests { + t.Run(tt.name, func(t *testing.T) { + output := captureAppOutput(t, func() { + printMultiServiceSummary(tt.recs, tt.results, tt.stats, tt.isDryRun) + }) + + // Verify output contains expected information + assert.Contains(t, output, "Final Summary") + if tt.isDryRun { + assert.Contains(t, output, "DRY RUN") + } else { + assert.Contains(t, output, "ACTUAL PURCHASE") + } + + if len(tt.stats) > 0 { + assert.Contains(t, output, "RESERVED INSTANCES:") + } + + if len(tt.results) > 0 { + assert.Contains(t, output, "success rate") + } + }) + } +} + +func TestPrintSavingsPlansSection(t *testing.T) { + tests := []struct { + checkOutput func(t *testing.T, output string) + name string + recommendations []common.Recommendation + stats ServiceProcessingStats + }{ + { + name: "Prints Compute Savings Plans", + recommendations: []common.Recommendation{ + { + Service: common.ServiceSavingsPlansAll, + EstimatedSavings: 500.0, + Details: &common.SavingsPlanDetails{ + PlanType: "Compute", + HourlyCommitment: 1.5, + }, + }, + }, + stats: ServiceProcessingStats{ + Service: common.ServiceSavingsPlansAll, + RecommendationsSelected: 1, + TotalEstimatedSavings: 500.0, + }, + checkOutput: func(t *testing.T, output string) { + assert.Contains(t, output, "SAVINGS PLANS:") + assert.Contains(t, output, "Compute SP") + assert.Contains(t, output, "500.00") + }, + }, + { + name: "Prints EC2 Instance Savings Plans", + recommendations: []common.Recommendation{ + { + Service: common.ServiceSavingsPlansAll, + EstimatedSavings: 300.0, + Details: &common.SavingsPlanDetails{ + PlanType: "EC2Instance", + HourlyCommitment: 1.0, + }, + }, + }, + stats: ServiceProcessingStats{ + Service: common.ServiceSavingsPlansAll, + RecommendationsSelected: 1, + TotalEstimatedSavings: 300.0, + }, + checkOutput: func(t *testing.T, output string) { + assert.Contains(t, output, "EC2 Inst SP") + assert.Contains(t, output, "300.00") + }, + }, + { + name: "Prints Database Savings Plans", + recommendations: []common.Recommendation{ + { + Service: common.ServiceSavingsPlansAll, + EstimatedSavings: 400.0, + Details: &common.SavingsPlanDetails{ + PlanType: "Database", + HourlyCommitment: 1.2, + }, + }, + }, + stats: ServiceProcessingStats{ + Service: common.ServiceSavingsPlansAll, + RecommendationsSelected: 1, + TotalEstimatedSavings: 400.0, + }, + checkOutput: func(t *testing.T, output string) { + assert.Contains(t, output, "Database SP") + assert.Contains(t, output, "400.00") + }, + }, + { + name: "Prints SageMaker Savings Plans", + recommendations: []common.Recommendation{ + { + Service: common.ServiceSavingsPlansAll, + EstimatedSavings: 250.0, + Details: &common.SavingsPlanDetails{ + PlanType: "SageMaker", + HourlyCommitment: 0.8, + }, + }, + }, + stats: ServiceProcessingStats{ + Service: common.ServiceSavingsPlansAll, + RecommendationsSelected: 1, + TotalEstimatedSavings: 250.0, + }, + checkOutput: func(t *testing.T, output string) { + assert.Contains(t, output, "SageMaker SP") + assert.Contains(t, output, "250.00") + }, + }, + { + name: "Prints multiple SP types with recommendations", + recommendations: []common.Recommendation{ + { + Service: common.ServiceSavingsPlansAll, + EstimatedSavings: 500.0, + Details: &common.SavingsPlanDetails{ + PlanType: "Compute", + HourlyCommitment: 1.5, + }, + }, + { + Service: common.ServiceSavingsPlansAll, + EstimatedSavings: 600.0, + Details: &common.SavingsPlanDetails{ + PlanType: "EC2Instance", + HourlyCommitment: 1.8, + }, + }, + }, + stats: ServiceProcessingStats{ + Service: common.ServiceSavingsPlansAll, + RecommendationsSelected: 2, + TotalEstimatedSavings: 1100.0, + }, + checkOutput: func(t *testing.T, output string) { + assert.Contains(t, output, "Compute SP") + assert.Contains(t, output, "EC2 Inst SP") + assert.Contains(t, output, "500.00") + assert.Contains(t, output, "600.00") + assert.Contains(t, output, "Best for EC2") + }, + }, + } + + for _, tt := range tests { + t.Run(tt.name, func(t *testing.T) { + output := captureAppOutput(t, func() { + printSavingsPlansSection(tt.recommendations, tt.stats) + }) + + tt.checkOutput(t, output) + }) + } +} + +func TestPrintComparisonSection(t *testing.T) { + tests := []struct { + riStats map[common.ServiceType]ServiceProcessingStats + checkOutput func(t *testing.T, output string) + name string + recommendations []common.Recommendation + riSavings float64 + }{ + { + name: "Comparison with EC2 RIs and EC2 Instance SP", + recommendations: []common.Recommendation{ + { + Service: common.ServiceSavingsPlansAll, + EstimatedSavings: 600.0, + Details: &common.SavingsPlanDetails{ + PlanType: "EC2Instance", + }, + }, + }, + riStats: map[common.ServiceType]ServiceProcessingStats{ + common.ServiceEC2: { + TotalEstimatedSavings: 500.0, + }, + }, + riSavings: 500.0, + checkOutput: func(t *testing.T, output string) { + assert.Contains(t, output, "COMPARISON:") + assert.Contains(t, output, "Option 1 (All RIs)") + assert.Contains(t, output, "500.00") + assert.Contains(t, output, "Option 2") + }, + }, + { + name: "Comparison with Database RIs and Database SP", + recommendations: []common.Recommendation{ + { + Service: common.ServiceSavingsPlansAll, + EstimatedSavings: 800.0, + Details: &common.SavingsPlanDetails{ + PlanType: "Database", + }, + }, + }, + riStats: map[common.ServiceType]ServiceProcessingStats{ + common.ServiceRDS: { + TotalEstimatedSavings: 700.0, + }, + common.ServiceElastiCache: { + TotalEstimatedSavings: 200.0, + }, + }, + riSavings: 900.0, + checkOutput: func(t *testing.T, output string) { + assert.Contains(t, output, "COMPARISON:") + assert.Contains(t, output, "Option 3") + assert.Contains(t, output, "Database SP") + }, + }, + { + name: "Compute SP better than EC2 Instance SP", + recommendations: []common.Recommendation{ + { + Service: common.ServiceSavingsPlansAll, + EstimatedSavings: 500.0, + Details: &common.SavingsPlanDetails{ + PlanType: "EC2Instance", + }, + }, + { + Service: common.ServiceSavingsPlansAll, + EstimatedSavings: 700.0, + Details: &common.SavingsPlanDetails{ + PlanType: "Compute", + }, + }, + }, + riStats: map[common.ServiceType]ServiceProcessingStats{ + common.ServiceEC2: { + TotalEstimatedSavings: 600.0, + }, + }, + riSavings: 600.0, + checkOutput: func(t *testing.T, output string) { + assert.Contains(t, output, "Compute SP") + assert.Contains(t, output, "700.00") + }, + }, + } + + for _, tt := range tests { + t.Run(tt.name, func(t *testing.T) { + // Capture stdout + output := captureAppOutput(t, func() { + printComparisonSection(tt.recommendations, tt.riStats, tt.riSavings) + }) + + tt.checkOutput(t, output) + }) + } +} + +func TestPrintFinalMessage(t *testing.T) { + tests := []struct { + name string + wantOutput []string + notWanted []string + riSuccess int + isDryRun bool + }{ + { + name: "Dry run shows no Archera pitch", + isDryRun: true, + riSuccess: 5, + notWanted: []string{"Archera", archeraSignupURL}, + }, + { + name: "No successful purchases shows no pitch", + isDryRun: false, + riSuccess: 0, + notWanted: []string{"Purchase operations completed", "Archera", archeraSignupURL}, + }, + { + name: "Successful purchase shows Archera pitch with URL and disclosure", + isDryRun: false, + riSuccess: 3, + wantOutput: []string{ + "Purchase operations completed", + "underutilization insurance", + "https://www.archera.ai/cudly", + "first 7 days", + // Non-gating partnership disclosure (project_archera_partnership memory). + "entirely optional", + "work fully without Archera", + // Sponsorship partnership disclosure (project_archera_partnership memory). + "sponsors CUDly's Open Source", + }, + }, + } + + for _, tt := range tests { + t.Run(tt.name, func(t *testing.T) { + output := captureAppOutput(t, func() { + printFinalMessage(tt.isDryRun, tt.riSuccess) + }) + + for _, want := range tt.wantOutput { + assert.Contains(t, output, want) + } + for _, notWant := range tt.notWanted { + assert.NotContains(t, output, notWant) + } + }) + } +} diff --git a/cmd/multi_service_test.go b/cmd/multi_service_test.go new file mode 100644 index 000000000..eb8ae8880 --- /dev/null +++ b/cmd/multi_service_test.go @@ -0,0 +1,1932 @@ +package main + +import ( + "bytes" + "context" + "encoding/csv" + "fmt" + "log" + "os" + "path/filepath" + "strconv" + "strings" + "testing" + "time" + + "github.com/LeanerCloud/CUDly/pkg/common" + "github.com/LeanerCloud/CUDly/pkg/scorer" + "github.com/aws/aws-sdk-go-v2/aws" + rdstypes "github.com/aws/aws-sdk-go-v2/service/rds/types" + "github.com/stretchr/testify/assert" + "github.com/stretchr/testify/mock" + "github.com/stretchr/testify/require" +) + +func TestRunToolMultiService_Validation(t *testing.T) { + // Save original values + origCfg := toolCfg + + // Restore after test + defer func() { + toolCfg = origCfg + }() + + tests := []struct { + setupVars func() + name string + expectPanic bool + }{ + { + name: "Valid input - all services", + setupVars: func() { + toolCfg.Coverage = 75.0 + toolCfg.PaymentOption = "partial-upfront" + toolCfg.TermYears = 3 + toolCfg.AllServices = true + toolCfg.Services = nil + }, + expectPanic: false, + }, + { + name: "Valid input - specific services", + setupVars: func() { + toolCfg.Coverage = 50.0 + toolCfg.PaymentOption = "no-upfront" + toolCfg.TermYears = 1 + toolCfg.AllServices = false + toolCfg.Services = []string{"rds", "ec2"} + }, + expectPanic: false, + }, + { + name: "Invalid coverage - too high", + setupVars: func() { + toolCfg.Coverage = 150.0 + toolCfg.PaymentOption = "partial-upfront" + toolCfg.TermYears = 3 + }, + expectPanic: true, + }, + { + name: "Invalid coverage - negative", + setupVars: func() { + toolCfg.Coverage = -10.0 + toolCfg.PaymentOption = "all-upfront" + toolCfg.TermYears = 1 + }, + expectPanic: true, + }, + { + name: "Invalid payment option", + setupVars: func() { + toolCfg.Coverage = 80.0 + toolCfg.PaymentOption = "invalid-payment" + toolCfg.TermYears = 3 + }, + expectPanic: true, + }, + { + name: "Invalid term years", + setupVars: func() { + toolCfg.Coverage = 80.0 + toolCfg.PaymentOption = "partial-upfront" + toolCfg.TermYears = 2 // Only 1 or 3 allowed + }, + expectPanic: true, + }, + { + name: "Default to RDS when no services", + setupVars: func() { + toolCfg.Coverage = 80.0 + toolCfg.PaymentOption = "all-upfront" + toolCfg.TermYears = 3 + toolCfg.AllServices = false + toolCfg.Services = nil + }, + expectPanic: false, + }, + } + + for _, tt := range tests { + t.Run(tt.name, func(t *testing.T) { + tt.setupVars() + + if tt.expectPanic { + // Can't easily test log.Fatalf, so we skip execution + // In production code, would use dependency injection + return + } + + // For non-panic tests, verify the setup is valid + assert.GreaterOrEqual(t, toolCfg.Coverage, 0.0) + assert.LessOrEqual(t, toolCfg.Coverage, 100.0) + assert.Contains(t, []string{"all-upfront", "partial-upfront", "no-upfront"}, toolCfg.PaymentOption) + assert.Contains(t, []int{1, 3}, toolCfg.TermYears) + }) + } +} + +func TestProcessService_EdgeCases(t *testing.T) { + // Save original values + origCfg := toolCfg + + defer func() { + toolCfg = origCfg + }() + + // Set test values + toolCfg.PaymentOption = "partial-upfront" + toolCfg.TermYears = 3 + + tests := []struct { + name string + setupFunc func() + service common.ServiceType + isDryRun bool + expectRecs int + }{ + { + name: "With explicit regions", + setupFunc: func() { + toolCfg.Regions = []string{"us-east-1"} + toolCfg.Coverage = 100.0 + }, + service: common.ServiceRDS, + isDryRun: true, + expectRecs: 0, // Would need mock to return actual recs + }, + { + name: "No regions triggers discovery", + setupFunc: func() { + toolCfg.Regions = []string{} + toolCfg.Coverage = 75.0 + }, + service: common.ServiceEC2, + isDryRun: false, + expectRecs: 0, // Would need mock + }, + { + name: "Zero coverage", + setupFunc: func() { + toolCfg.Regions = []string{"us-west-2"} + toolCfg.Coverage = 0.0 + }, + service: common.ServiceElastiCache, + isDryRun: true, + expectRecs: 0, + }, + } + + for _, tt := range tests { + t.Run(tt.name, func(t *testing.T) { + tt.setupFunc() + + // Verify test configuration was applied correctly + // Note: Full integration tests would require AWS credentials or mocks + // These tests verify the configuration setup is working as expected + + switch tt.name { + case "With explicit regions": + assert.Equal(t, []string{"us-east-1"}, toolCfg.Regions) + assert.Equal(t, 100.0, toolCfg.Coverage) + case "No regions triggers discovery": + assert.Empty(t, toolCfg.Regions) + assert.Equal(t, 75.0, toolCfg.Coverage) + case "Zero coverage": + assert.Equal(t, []string{"us-west-2"}, toolCfg.Regions) + assert.Equal(t, 0.0, toolCfg.Coverage) + } + + // Verify the test case service is valid + assert.NotEmpty(t, tt.service, "service should not be empty") + assert.GreaterOrEqual(t, tt.expectRecs, 0, "expected recommendations should be non-negative") + }) + } +} + +// TestProcessServiceWithMocks tests the processService function using mocks. +func TestProcessServiceWithMocks(t *testing.T) { + ctx := context.Background() + awsCfg := aws.Config{Region: "us-east-1"} + + // Save original values + origCfg := toolCfg + + defer func() { + toolCfg = origCfg + }() + + tests := []struct { + setupFunc func() + name string + service common.ServiceType + testRegions []string + mockRecs []common.Recommendation + isDryRun bool + }{ + { + name: "RDS dry run with recommendations", + service: common.ServiceRDS, + isDryRun: true, + testRegions: []string{"us-east-1"}, + mockRecs: []common.Recommendation{ + {ResourceType: "db.t3.micro", Count: 2, Region: "us-east-1", EstimatedSavings: 100}, + {ResourceType: "db.t3.small", Count: 1, Region: "us-east-1", EstimatedSavings: 200}, + }, + setupFunc: func() { + toolCfg.Coverage = 100.0 + toolCfg.PaymentOption = "partial-upfront" + toolCfg.TermYears = 3 + }, + }, + { + name: "EC2 with no recommendations", + service: common.ServiceEC2, + isDryRun: true, + testRegions: []string{"us-west-2"}, + mockRecs: []common.Recommendation{}, + setupFunc: func() { + toolCfg.Coverage = 80.0 + toolCfg.PaymentOption = "no-upfront" + toolCfg.TermYears = 1 + }, + }, + { + name: "ElastiCache with 50% coverage", + service: common.ServiceElastiCache, + isDryRun: false, + testRegions: []string{"eu-west-1"}, + mockRecs: []common.Recommendation{ + {ResourceType: "cache.t3.micro", Count: 3, Region: "eu-west-1", EstimatedSavings: 150}, + {ResourceType: "cache.t3.small", Count: 2, Region: "eu-west-1", EstimatedSavings: 250}, + }, + setupFunc: func() { + toolCfg.Coverage = 50.0 + toolCfg.PaymentOption = "all-upfront" + toolCfg.TermYears = 3 + }, + }, + } + + for _, tt := range tests { + t.Run(tt.name, func(t *testing.T) { + tt.setupFunc() + + // Create mock client + mockClient := &MockRecommendationsClient{} + + // Setup expectations + for range tt.testRegions { + mockClient.On("GetRecommendations", ctx, mock.AnythingOfType("*common.RecommendationParams")).Return(tt.mockRecs, nil) + } + + // Set regions in toolCfg for this test + toolCfg.Regions = tt.testRegions + + // Now we can use the actual function directly since it accepts an interface + accountCache := NewAccountAliasCache(awsCfg) + recs, results := processService(ctx, awsCfg, mockClient, accountCache, tt.service, tt.isDryRun, toolCfg, engineVersionData{}) + + if len(tt.mockRecs) > 0 { + // Should have recommendations based on coverage + expectedCount := int(float64(len(tt.mockRecs)) * toolCfg.Coverage / 100.0) + if expectedCount > 0 { + assert.NotEmpty(t, recs) + assert.LessOrEqual(t, len(recs), len(tt.mockRecs)) + } else { + assert.Empty(t, recs) + } + + // Check results for dry run + if tt.isDryRun && len(recs) > 0 { + assert.Equal(t, len(recs), len(results)) + for _, result := range results { + assert.True(t, result.Success) + assert.Nil(t, result.Error) // Dry runs are successful, so no error + assert.True(t, result.DryRun) + } + } + } else { + assert.Empty(t, recs) + assert.Empty(t, results) + } + + mockClient.AssertExpectations(t) + }) + } +} + +func TestProcessService_SavingsPlansAccountLevel(t *testing.T) { + ctx := context.Background() + awsCfg := aws.Config{Region: "us-east-1"} + + origCfg := toolCfg + defer func() { toolCfg = origCfg }() + + toolCfg.Coverage = 100.0 + toolCfg.PaymentOption = "all-upfront" + toolCfg.TermYears = 3 + toolCfg.Regions = []string{} // Empty - should auto-detect for Savings Plans + + mockClient := &MockRecommendationsClient{} + + // Savings Plans should only query us-east-1 once (account-level). Use the + // per-plan-type Compute slug now that the legacy umbrella has been + // retired from createServiceClient dispatch. + mockRecs := []common.Recommendation{ + {Service: common.ServiceSavingsPlansCompute, ResourceType: "ComputeSP", Count: 1, Region: "us-east-1", EstimatedSavings: 1000}, + } + mockClient.On("GetRecommendations", ctx, mock.AnythingOfType("*common.RecommendationParams")).Return(mockRecs, nil) + + accountCache := NewAccountAliasCache(awsCfg) + recs, results := processService(ctx, awsCfg, mockClient, accountCache, common.ServiceSavingsPlansCompute, true, toolCfg, engineVersionData{}) + + // Should get recommendations + assert.NotEmpty(t, recs) + assert.NotEmpty(t, results) + + // Verify Savings Plans queried only once + mockClient.AssertNumberOfCalls(t, "GetRecommendations", 1) + mockClient.AssertExpectations(t) +} + +func TestProcessService_WithInstanceLimit(t *testing.T) { + // --max-instances is a run-wide cap, so the per-region path deliberately + // does NOT apply it (that was #1608: applying it here capped each + // service/region pair independently). This legacy entry point therefore + // returns the uncapped recommendations; the cap is enforced once by + // scoreLimitAndDisplay on the real pipeline, covered by + // TestMaxInstancesCapsWholeRunAcrossServicesAndRegions. + // + // This run is a dry run, so the fail-closed purchase guard does not fire; + // TestProcessService_InstanceLimitRefusesRealPurchase covers that. + + ctx := context.Background() + awsCfg := aws.Config{Region: "us-east-1"} + + origCfg := toolCfg + defer func() { toolCfg = origCfg }() + + toolCfg.Coverage = 100.0 + toolCfg.PaymentOption = "partial-upfront" + toolCfg.TermYears = 1 + toolCfg.Regions = []string{"us-east-1"} + toolCfg.MaxInstances = 15 // Limit to 15 total instances + + mockClient := &MockRecommendationsClient{} + + mockRecs := []common.Recommendation{ + {ResourceType: "db.t3.micro", Count: 10, Region: "us-east-1", EstimatedSavings: 100}, + {ResourceType: "db.t3.small", Count: 10, Region: "us-east-1", EstimatedSavings: 200}, + } + mockClient.On("GetRecommendations", ctx, mock.AnythingOfType("*common.RecommendationParams")).Return(mockRecs, nil) + + accountCache := NewAccountAliasCache(awsCfg) + recs, results := processService(ctx, awsCfg, mockClient, accountCache, common.ServiceRDS, true, toolCfg, engineVersionData{}) + + assert.Len(t, recs, 2, "Should return recommendations") + + // Assert on the *results*, not on recs. processRegionRecommendations assigns + // result.recommendations before the cap block and did so pre-fix too, so a + // count taken from recs passes either way and would be a vacuous guard. + // The dry-run results carry the recommendations that were actually handed + // to processPurchaseLoop, which is what the cap used to shrink: pre-fix + // these totalled 15 (10 + 5 truncated), post-fix they total the full 20. + resultInstances := 0 + for i := range results { + resultInstances += results[i].Recommendation.Count + } + assert.Len(t, results, 2) + assert.Equal(t, 20, resultInstances, + "the per-region path must not apply --max-instances; the cap is run-wide") + + mockClient.AssertExpectations(t) +} + +// TestProcessService_InstanceLimitRefusesRealPurchase pins the fail-closed +// contract on the legacy per-region entry point: it cannot see the other +// services and regions in the run, so it cannot evaluate the run-wide +// --max-instances cap. Rather than purchase uncapped it must make no purchases +// at all. An uncapped over-purchase of reserved capacity is not reversible. +func TestProcessService_InstanceLimitRefusesRealPurchase(t *testing.T) { + ctx := context.Background() + awsCfg := aws.Config{Region: "us-east-1"} + + origCfg := toolCfg + t.Cleanup(func() { toolCfg = origCfg }) + + toolCfg.Coverage = 100.0 + toolCfg.PaymentOption = "partial-upfront" + toolCfg.TermYears = 1 + toolCfg.Regions = []string{"us-east-1"} + toolCfg.MaxInstances = 15 + + mockClient := &MockRecommendationsClient{} + mockRecs := []common.Recommendation{ + {ResourceType: "db.t3.micro", Count: 10, Region: "us-east-1", EstimatedSavings: 100}, + {ResourceType: "db.t3.small", Count: 10, Region: "us-east-1", EstimatedSavings: 200}, + } + mockClient.On("GetRecommendations", ctx, mock.AnythingOfType("*common.RecommendationParams")).Return(mockRecs, nil) + t.Cleanup(func() { mockClient.AssertExpectations(t) }) + + accountCache := NewAccountAliasCache(awsCfg) + + var recs []common.Recommendation + var results []common.PurchaseResult + out := captureAppOutput(t, func() { + recs, results = processService(ctx, awsCfg, mockClient, accountCache, common.ServiceRDS, false, toolCfg, engineVersionData{}) + }) + + assert.NotEmpty(t, recs, "recommendations are still reported") + assert.Empty(t, results, "no purchase may be attempted when the cap cannot be evaluated") + + // The refusal must be reached without touching AWS. The duplicate check is + // the only cloud call on this path, and without credentials it logs this + // warning, so its absence is positive evidence that the guard returned + // before any client was built. Before the guard was hoisted above + // createServiceClient this assertion failed: the describe ran, 403'd, and + // emitted the warning on the way to a refusal that was already decided. + assert.NotContains(t, out, "Could not check for existing RIs", + "the refusal path must not reach a cloud API call") +} + +func TestProcessService_WithOverrideCount(t *testing.T) { + ctx := context.Background() + awsCfg := aws.Config{Region: "us-east-1"} + + origCfg := toolCfg + defer func() { toolCfg = origCfg }() + + toolCfg.Coverage = 100.0 + toolCfg.PaymentOption = "no-upfront" + toolCfg.TermYears = 1 + toolCfg.Regions = []string{"us-east-1"} + toolCfg.OverrideCount = 3 // Override all counts to 3 + + mockClient := &MockRecommendationsClient{} + + mockRecs := []common.Recommendation{ + {ResourceType: "cache.t3.micro", Count: 10, Region: "us-east-1", EstimatedSavings: 100}, + {ResourceType: "cache.t3.small", Count: 5, Region: "us-east-1", EstimatedSavings: 200}, + } + mockClient.On("GetRecommendations", ctx, mock.AnythingOfType("*common.RecommendationParams")).Return(mockRecs, nil) + + accountCache := NewAccountAliasCache(awsCfg) + recs, _ := processService(ctx, awsCfg, mockClient, accountCache, common.ServiceElastiCache, true, toolCfg, engineVersionData{}) + + // All recommendations should have count=3 (override) + for _, rec := range recs { + assert.Equal(t, 3, rec.Count) + } + + mockClient.AssertExpectations(t) +} + +func TestProcessService_MultipleRegions(t *testing.T) { + ctx := context.Background() + awsCfg := aws.Config{Region: "us-east-1"} + + origCfg := toolCfg + defer func() { toolCfg = origCfg }() + + toolCfg.Coverage = 100.0 + toolCfg.PaymentOption = "all-upfront" + toolCfg.TermYears = 3 + toolCfg.Regions = []string{"us-east-1", "us-west-2", "eu-west-1"} + + mockClient := &MockRecommendationsClient{} + + // Setup mock for each region call in order; Once() ensures testify cycles + // through the returns so each region's call gets region-appropriate results. + for _, region := range toolCfg.Regions { + mockRecs := []common.Recommendation{ + {ResourceType: "db.t3.small", Count: 2, Region: region, EstimatedSavings: 100}, + } + mockClient.On("GetRecommendations", ctx, mock.AnythingOfType("*common.RecommendationParams")).Return(mockRecs, nil).Once() + } + + accountCache := NewAccountAliasCache(awsCfg) + recs, results := processService(ctx, awsCfg, mockClient, accountCache, common.ServiceRDS, true, toolCfg, engineVersionData{}) + + // Should get recommendations from all 3 regions + assert.NotEmpty(t, recs) + assert.Len(t, recs, 3) // One from each region + assert.Len(t, results, 3) + + // Verify each region was queried + mockClient.AssertNumberOfCalls(t, "GetRecommendations", 3) + mockClient.AssertExpectations(t) +} + +// ==================== Helper Function Tests ==================== + +func TestApplyCoverageToRecommendations(t *testing.T) { + tests := []struct { + name string + recs []common.Recommendation + coverage float64 + expectedRecs int + }{ + { + name: "50% coverage of 4 recommendations", + recs: []common.Recommendation{ + {ResourceType: "type1", Count: 2}, + {ResourceType: "type2", Count: 3}, + {ResourceType: "type3", Count: 1}, + {ResourceType: "type4", Count: 4}, + }, + coverage: 0.5, + expectedRecs: 2, + }, + { + name: "100% coverage", + recs: []common.Recommendation{ + {ResourceType: "type1", Count: 2}, + {ResourceType: "type2", Count: 3}, + }, + coverage: 1.0, + expectedRecs: 2, + }, + { + name: "0% coverage", + recs: []common.Recommendation{ + {ResourceType: "type1", Count: 2}, + {ResourceType: "type2", Count: 3}, + }, + coverage: 0.0, + expectedRecs: 0, + }, + { + name: "75% coverage of 3 recommendations", + recs: []common.Recommendation{ + {ResourceType: "type1", Count: 2}, + {ResourceType: "type2", Count: 2}, + {ResourceType: "type3", Count: 2}, + }, + coverage: 0.75, + expectedRecs: 2, + }, + } + + for _, tt := range tests { + t.Run(tt.name, func(t *testing.T) { + result := applyCoverageToRecommendations(tt.recs, tt.coverage) + assert.Equal(t, tt.expectedRecs, len(result)) + }) + } +} + +func TestPrintDropSummaryImmediatelyPrecedesTerminalOutput(t *testing.T) { + for _, terminal := range []string{ + "ℹ️ No recommendations passed filters. Nothing to purchase.", + "FINAL SUMMARY", + } { + t.Run(terminal, func(t *testing.T) { + oldLogger := AppLogger + var output bytes.Buffer + AppLogger = log.New(&output, "", 0) + t.Cleanup(func() { AppLogger = oldLogger }) + + drops := common.NewDropSummary() + drops.Add(common.DropTargetSizedToZero, 1) + + printDropSummary(drops) + AppLogger.Printf("\n%s\n", terminal) + + assert.Contains(t, output.String(), + "Dropped 1 recs: target-sized-to-zero=1\n\n"+terminal) + assert.Equal(t, 1, strings.Count(output.String(), "Dropped 1 recs")) + }) + } +} + +func TestRunPurchaseAndReportDeclinedConfirmationPrintsDropsBeforeCancellation(t *testing.T) { + drops := common.NewDropSummary() + drops.Add(common.DropTargetSizedToZero, 1) + scored := scorer.ScoredResult{Passed: []common.Recommendation{{Count: 1, EstimatedSavings: 10}}} + + output := captureAppOutput(t, func() { + runPurchaseAndReport(context.Background(), aws.Config{}, scored, false, Config{}, drops) + }) + + assert.Contains(t, output, + "Dropped 1 recs: target-sized-to-zero=1\n\n❌ Purchase canceled.") + assert.Equal(t, 1, strings.Count(output, "Dropped 1 recs")) +} + +func TestServiceProcessingOrder(t *testing.T) { + // Test that services are processed in a consistent order + services := []common.ServiceType{ + common.ServiceRDS, + common.ServiceElastiCache, + common.ServiceEC2, + common.ServiceOpenSearch, + common.ServiceRedshift, + common.ServiceMemoryDB, + } + + // Verify all expected services are present + expectedServices := map[common.ServiceType]bool{ + common.ServiceRDS: false, + common.ServiceElastiCache: false, + common.ServiceEC2: false, + common.ServiceOpenSearch: false, + common.ServiceRedshift: false, + common.ServiceMemoryDB: false, + } + + for _, service := range services { + expectedServices[service] = true + } + + // Check all services were found + for service, found := range expectedServices { + assert.True(t, found, "Service %s should be in processing list", service) + } +} + +func TestGenerateCSVFilenameHelper(t *testing.T) { + tests := []struct { + name string + service common.ServiceType + payment string + expectParts []string + dryRun bool + }{ + { + name: "RDS dry run", + service: common.ServiceRDS, + payment: "no-upfront", + dryRun: true, + expectParts: []string{"rds", "no-upfront", "dryrun"}, + }, + { + name: "EC2 actual purchase", + service: common.ServiceEC2, + payment: "all-upfront", + dryRun: false, + expectParts: []string{"ec2", "all-upfront", "purchase"}, + }, + } + + for _, tt := range tests { + t.Run(tt.name, func(t *testing.T) { + filename := generateCSVFilenameTestHelper(tt.service, tt.payment, tt.dryRun) + + for _, part := range tt.expectParts { + assert.Contains(t, filename, part) + } + + // Should end with .csv + assert.Contains(t, filename, ".csv") + }) + } +} + +func TestMultiServiceConfig(t *testing.T) { + cfg := MultiServiceConfig{ + Services: map[common.ServiceType]ServiceConfig{ + common.ServiceRDS: { + Enabled: true, + Coverage: 0.5, + }, + common.ServiceElastiCache: { + Enabled: true, + Coverage: 0.8, + }, + common.ServiceEC2: { + Enabled: false, + Coverage: 0.0, + }, + }, + PaymentOption: "no-upfront", + TermYears: 3, + DryRun: true, + } + + // Test enabled services count + enabledCount := 0 + for _, svcConfig := range cfg.Services { + if svcConfig.Enabled { + enabledCount++ + } + } + assert.Equal(t, 2, enabledCount) + + // Test coverage values + assert.Equal(t, 0.5, cfg.Services[common.ServiceRDS].Coverage) + assert.Equal(t, 0.8, cfg.Services[common.ServiceElastiCache].Coverage) +} + +// ==================== Benchmark Tests ==================== + +func BenchmarkCalculateTotalInstances(b *testing.B) { + recs := make([]common.Recommendation, 100) + for i := range recs { + recs[i] = common.Recommendation{Count: i%10 + 1} + } + + b.ResetTimer() + for i := 0; i < b.N; i++ { + _ = calculateTotalInstances(recs) + } +} + +func BenchmarkApplyCoverageToRecommendations(b *testing.B) { + recs := make([]common.Recommendation, 100) + for i := range recs { + recs[i] = common.Recommendation{ + ResourceType: "type", + Count: i%5 + 1, + } + } + + b.ResetTimer() + for i := 0; i < b.N; i++ { + _ = applyCoverageToRecommendations(recs, 0.5) + } +} + +// ==================== Helper Functions for Tests ==================== + +func calculateTotalInstances(recs []common.Recommendation) int { + var total int + for _, rec := range recs { + total += rec.Count + } + return total +} + +func applyCoverageToRecommendations(recs []common.Recommendation, coverage float64) []common.Recommendation { + if coverage <= 0 { + return []common.Recommendation{} + } + if coverage >= 1.0 { + return recs + } + + targetCount := int(float64(len(recs)) * coverage) + if targetCount == 0 && coverage > 0 && len(recs) > 0 { + targetCount = 1 + } + + if targetCount >= len(recs) { + return recs + } + + return recs[:targetCount] +} + +func generateCSVFilenameTestHelper(service common.ServiceType, payment string, dryRun bool) string { + mode := "purchase" + if dryRun { + mode = "dryrun" + } + + serviceStr := "" + switch service { + case common.ServiceRDS: + serviceStr = "rds" + case common.ServiceElastiCache: + serviceStr = "elasticache" + case common.ServiceEC2: + serviceStr = "ec2" + case common.ServiceOpenSearch: + serviceStr = "opensearch" + case common.ServiceRedshift: + serviceStr = "redshift" + case common.ServiceMemoryDB: + serviceStr = "memorydb" + default: + serviceStr = "unknown" + } + + return serviceStr + "-" + payment + "-" + mode + ".csv" +} + +// Test types. +type MultiServiceConfig struct { + Services map[common.ServiceType]ServiceConfig + PaymentOption string + TermYears int + DryRun bool +} + +type ServiceConfig struct { + Enabled bool + Coverage float64 +} + +// ==================== Filter Function Tests ==================== + +func TestApplyFilters_RegionFiltering(t *testing.T) { + tests := []struct { + name string + currentRegion string + recs []common.Recommendation + includeRegions []string + excludeRegions []string + expectedCount int + }{ + { + name: "No region filters - all pass", + recs: []common.Recommendation{ + {Region: "us-east-1", ResourceType: "db.t3.small", Count: 1, Service: common.ServiceRDS}, + {Region: "us-west-2", ResourceType: "db.t3.medium", Count: 2, Service: common.ServiceRDS}, + {Region: "eu-west-1", ResourceType: "db.t3.large", Count: 3, Service: common.ServiceRDS}, + }, + includeRegions: []string{}, + excludeRegions: []string{}, + currentRegion: "", + expectedCount: 3, + }, + { + name: "Include specific regions", + recs: []common.Recommendation{ + {Region: "us-east-1", ResourceType: "db.t3.small", Count: 1, Service: common.ServiceRDS}, + {Region: "us-west-2", ResourceType: "db.t3.medium", Count: 2, Service: common.ServiceRDS}, + {Region: "eu-west-1", ResourceType: "db.t3.large", Count: 3, Service: common.ServiceRDS}, + }, + includeRegions: []string{"us-east-1", "us-west-2"}, + excludeRegions: []string{}, + currentRegion: "", + expectedCount: 2, + }, + { + name: "Exclude specific regions", + recs: []common.Recommendation{ + {Region: "us-east-1", ResourceType: "db.t3.small", Count: 1, Service: common.ServiceRDS}, + {Region: "us-west-2", ResourceType: "db.t3.medium", Count: 2, Service: common.ServiceRDS}, + {Region: "eu-west-1", ResourceType: "db.t3.large", Count: 3, Service: common.ServiceRDS}, + }, + includeRegions: []string{}, + excludeRegions: []string{"us-west-2"}, + currentRegion: "", + expectedCount: 2, + }, + { + name: "Current region filter with RDS", + recs: []common.Recommendation{ + {Region: "us-east-1", ResourceType: "db.t3.small", Count: 1, Service: common.ServiceRDS}, + {Region: "us-west-2", ResourceType: "db.t3.medium", Count: 2, Service: common.ServiceRDS}, + }, + includeRegions: []string{}, + excludeRegions: []string{}, + currentRegion: "us-east-1", + expectedCount: 1, + }, + { + name: "Savings Plans bypass region filter", + recs: []common.Recommendation{ + {Region: "us-east-1", ResourceType: "ComputeSP", Count: 1, Service: common.ServiceSavingsPlansAll}, + {Region: "us-west-2", ResourceType: "ComputeSP", Count: 2, Service: common.ServiceSavingsPlansAll}, + }, + includeRegions: []string{}, + excludeRegions: []string{}, + currentRegion: "us-east-1", + expectedCount: 2, // Both should pass, SP is account-level + }, + } + + for _, tt := range tests { + t.Run(tt.name, func(t *testing.T) { + cfg := Config{ + IncludeRegions: tt.includeRegions, + ExcludeRegions: tt.excludeRegions, + } + + result := applyFilters(tt.recs, &cfg, map[string][]InstanceEngineVersion{}, map[string]MajorEngineVersionInfo{}, tt.currentRegion, nil) + assert.Equal(t, tt.expectedCount, len(result), "Expected %d recommendations, got %d", tt.expectedCount, len(result)) + }) + } +} + +func TestApplyFilters_InstanceTypeFiltering(t *testing.T) { + tests := []struct { + name string + recs []common.Recommendation + includeInstanceTypes []string + excludeInstanceTypes []string + expectedCount int + }{ + { + name: "Include specific instance types", + recs: []common.Recommendation{ + {Region: "us-east-1", ResourceType: "db.t3.small", Count: 1, Service: common.ServiceRDS}, + {Region: "us-east-1", ResourceType: "db.t3.medium", Count: 2, Service: common.ServiceRDS}, + {Region: "us-east-1", ResourceType: "db.r5.large", Count: 3, Service: common.ServiceRDS}, + }, + includeInstanceTypes: []string{"db.t3.small", "db.t3.medium"}, + excludeInstanceTypes: []string{}, + expectedCount: 2, + }, + { + name: "Exclude specific instance types", + recs: []common.Recommendation{ + {Region: "us-east-1", ResourceType: "db.t3.small", Count: 1, Service: common.ServiceRDS}, + {Region: "us-east-1", ResourceType: "db.t3.medium", Count: 2, Service: common.ServiceRDS}, + {Region: "us-east-1", ResourceType: "db.r5.large", Count: 3, Service: common.ServiceRDS}, + }, + includeInstanceTypes: []string{}, + excludeInstanceTypes: []string{"db.r5.large"}, + expectedCount: 2, + }, + } + + for _, tt := range tests { + t.Run(tt.name, func(t *testing.T) { + cfg := Config{ + IncludeInstanceTypes: tt.includeInstanceTypes, + ExcludeInstanceTypes: tt.excludeInstanceTypes, + } + + result := applyFilters(tt.recs, &cfg, map[string][]InstanceEngineVersion{}, map[string]MajorEngineVersionInfo{}, "", nil) + assert.Equal(t, tt.expectedCount, len(result)) + }) + } +} + +func TestApplyFilters_EngineFiltering(t *testing.T) { + tests := []struct { + name string + recs []common.Recommendation + includeEngines []string + excludeEngines []string + expectedCount int + }{ + { + name: "Include specific engines", + recs: []common.Recommendation{ + {Region: "us-east-1", ResourceType: "db.t3.small", Count: 1, Service: common.ServiceRDS, Details: &common.DatabaseDetails{Engine: "postgresql"}}, + {Region: "us-east-1", ResourceType: "db.t3.medium", Count: 2, Service: common.ServiceRDS, Details: &common.DatabaseDetails{Engine: "mysql"}}, + {Region: "us-east-1", ResourceType: "cache.t3.small", Count: 3, Service: common.ServiceElastiCache, Details: &common.CacheDetails{Engine: "redis"}}, + }, + includeEngines: []string{"postgresql", "mysql"}, + excludeEngines: []string{}, + expectedCount: 2, + }, + { + name: "Exclude specific engines", + recs: []common.Recommendation{ + {Region: "us-east-1", ResourceType: "db.t3.small", Count: 1, Service: common.ServiceRDS, Details: &common.DatabaseDetails{Engine: "postgresql"}}, + {Region: "us-east-1", ResourceType: "db.t3.medium", Count: 2, Service: common.ServiceRDS, Details: &common.DatabaseDetails{Engine: "mysql"}}, + {Region: "us-east-1", ResourceType: "cache.t3.small", Count: 3, Service: common.ServiceElastiCache, Details: &common.CacheDetails{Engine: "redis"}}, + }, + includeEngines: []string{}, + excludeEngines: []string{"redis"}, + expectedCount: 2, + }, + { + name: "Case insensitive engine matching", + recs: []common.Recommendation{ + {Region: "us-east-1", ResourceType: "db.t3.small", Count: 1, Service: common.ServiceRDS, Details: &common.DatabaseDetails{Engine: "PostgreSQL"}}, + {Region: "us-east-1", ResourceType: "db.t3.medium", Count: 2, Service: common.ServiceRDS, Details: &common.DatabaseDetails{Engine: "MySQL"}}, + }, + includeEngines: []string{"postgresql", "mysql"}, + excludeEngines: []string{}, + expectedCount: 2, + }, + { + name: "No engine details - with include list", + recs: []common.Recommendation{ + {Region: "us-east-1", ResourceType: "db.t3.small", Count: 1, Service: common.ServiceRDS}, + {Region: "us-east-1", ResourceType: "db.t3.medium", Count: 2, Service: common.ServiceRDS, Details: &common.DatabaseDetails{Engine: "mysql"}}, + }, + includeEngines: []string{"mysql"}, + excludeEngines: []string{}, + expectedCount: 1, // Only the one with engine details + }, + { + name: "No engine details - no include list", + recs: []common.Recommendation{ + {Region: "us-east-1", ResourceType: "db.t3.small", Count: 1, Service: common.ServiceRDS}, + {Region: "us-east-1", ResourceType: "db.t3.medium", Count: 2, Service: common.ServiceRDS}, + }, + includeEngines: []string{}, + excludeEngines: []string{}, + expectedCount: 2, // All pass + }, + } + + for _, tt := range tests { + t.Run(tt.name, func(t *testing.T) { + cfg := Config{ + IncludeEngines: tt.includeEngines, + ExcludeEngines: tt.excludeEngines, + } + + result := applyFilters(tt.recs, &cfg, map[string][]InstanceEngineVersion{}, map[string]MajorEngineVersionInfo{}, "", nil) + assert.Equal(t, tt.expectedCount, len(result)) + }) + } +} + +func TestApplyFilters_AccountFiltering(t *testing.T) { + tests := []struct { + name string + recs []common.Recommendation + includeAccounts []string + excludeAccounts []string + expectedCount int + }{ + { + name: "Include specific accounts", + recs: []common.Recommendation{ + {Region: "us-east-1", ResourceType: "db.t3.small", Count: 1, Service: common.ServiceRDS, AccountName: "prod-account"}, + {Region: "us-east-1", ResourceType: "db.t3.medium", Count: 2, Service: common.ServiceRDS, AccountName: "dev-account"}, + {Region: "us-east-1", ResourceType: "db.t3.large", Count: 3, Service: common.ServiceRDS, AccountName: "staging-account"}, + }, + includeAccounts: []string{"prod", "dev"}, + excludeAccounts: []string{}, + expectedCount: 2, // Substring match + }, + { + name: "Exclude specific accounts", + recs: []common.Recommendation{ + {Region: "us-east-1", ResourceType: "db.t3.small", Count: 1, Service: common.ServiceRDS, AccountName: "prod-account"}, + {Region: "us-east-1", ResourceType: "db.t3.medium", Count: 2, Service: common.ServiceRDS, AccountName: "dev-account"}, + }, + includeAccounts: []string{}, + excludeAccounts: []string{"dev"}, + expectedCount: 1, + }, + { + name: "Empty account name with filters", + recs: []common.Recommendation{ + {Region: "us-east-1", ResourceType: "db.t3.small", Count: 1, Service: common.ServiceRDS, AccountName: ""}, + {Region: "us-east-1", ResourceType: "db.t3.medium", Count: 2, Service: common.ServiceRDS, AccountName: "prod"}, + }, + includeAccounts: []string{"prod"}, + excludeAccounts: []string{}, + expectedCount: 1, // Empty account name is filtered out + }, + { + name: "Empty account name without filters", + recs: []common.Recommendation{ + {Region: "us-east-1", ResourceType: "db.t3.small", Count: 1, Service: common.ServiceRDS, AccountName: ""}, + {Region: "us-east-1", ResourceType: "db.t3.medium", Count: 2, Service: common.ServiceRDS, AccountName: "prod"}, + }, + includeAccounts: []string{}, + excludeAccounts: []string{}, + expectedCount: 2, // All pass + }, + } + + for _, tt := range tests { + t.Run(tt.name, func(t *testing.T) { + cfg := Config{ + IncludeAccounts: tt.includeAccounts, + ExcludeAccounts: tt.excludeAccounts, + } + + result := applyFilters(tt.recs, &cfg, map[string][]InstanceEngineVersion{}, map[string]MajorEngineVersionInfo{}, "", nil) + assert.Equal(t, tt.expectedCount, len(result)) + }) + } +} + +func TestApplyFilters_CombinedFilters(t *testing.T) { + recs := []common.Recommendation{ + {Region: "us-east-1", ResourceType: "db.t3.small", Count: 1, Service: common.ServiceRDS, Details: &common.DatabaseDetails{Engine: "postgresql"}, AccountName: "prod-account"}, + {Region: "us-west-2", ResourceType: "db.t3.medium", Count: 2, Service: common.ServiceRDS, Details: &common.DatabaseDetails{Engine: "mysql"}, AccountName: "dev-account"}, + {Region: "us-east-1", ResourceType: "db.r5.large", Count: 3, Service: common.ServiceRDS, Details: &common.DatabaseDetails{Engine: "postgresql"}, AccountName: "staging-account"}, + {Region: "eu-west-1", ResourceType: "cache.t3.small", Count: 4, Service: common.ServiceElastiCache, Details: &common.CacheDetails{Engine: "redis"}, AccountName: "prod-account"}, + } + + cfg := Config{ + IncludeRegions: []string{"us-east-1", "us-west-2"}, + IncludeEngines: []string{"postgresql", "mysql"}, + ExcludeInstanceTypes: []string{"db.r5.large"}, + IncludeAccounts: []string{"prod", "dev"}, + } + + result := applyFilters(recs, &cfg, map[string][]InstanceEngineVersion{}, map[string]MajorEngineVersionInfo{}, "", nil) + + // Only the first two should pass all filters + assert.Equal(t, 2, len(result)) + if len(result) >= 2 { + assert.Equal(t, "db.t3.small", result[0].ResourceType) + assert.Equal(t, "db.t3.medium", result[1].ResourceType) + } +} + +// ==================== CSV Report Tests ==================== + +func TestWriteMultiServiceCSVReport_EmptyResults(t *testing.T) { + tmpFile := "/tmp/test_empty_results.csv" + defer os.Remove(tmpFile) + + err := writeMultiServiceCSVReport([]common.PurchaseResult{}, tmpFile) + assert.NoError(t, err) + + // File should not be created for empty results + _, err = os.Stat(tmpFile) + assert.True(t, os.IsNotExist(err)) +} + +func TestWriteMultiServiceCSVReport_Success(t *testing.T) { + tmpFile := "/tmp/test_csv_success.csv" + defer os.Remove(tmpFile) + + results := []common.PurchaseResult{ + { + Recommendation: common.Recommendation{ + Service: common.ServiceRDS, + Region: "us-east-1", + ResourceType: "db.t3.small", + Count: 2, + Account: "123456789012", + AccountName: "test-account", + Term: "1yr", + PaymentOption: "all-upfront", + EstimatedSavings: 100.50, + }, + Success: true, + CommitmentID: "test-commitment-id", + Timestamp: time.Now(), + }, + } + + err := writeMultiServiceCSVReport(results, tmpFile) + assert.NoError(t, err) + + // Verify file was created and has content + content, err := os.ReadFile(tmpFile) + assert.NoError(t, err) + assert.Contains(t, string(content), "Service,Region,ResourceType") + assert.Contains(t, string(content), "rds") + assert.Contains(t, string(content), "us-east-1") + assert.Contains(t, string(content), "db.t3.small") + assert.Contains(t, string(content), "test-commitment-id") +} + +func TestWriteMultiServiceCSVReport_WithError(t *testing.T) { + tmpFile := "/tmp/test_csv_with_error.csv" + defer os.Remove(tmpFile) + + results := []common.PurchaseResult{ + { + Recommendation: common.Recommendation{ + Service: common.ServiceEC2, + Region: "us-west-2", + ResourceType: "t3.medium", + Count: 1, + }, + Success: false, + CommitmentID: "", + Error: fmt.Errorf("purchase failed: insufficient quota"), + Timestamp: time.Now(), + }, + } + + err := writeMultiServiceCSVReport(results, tmpFile) + assert.NoError(t, err) + + // Verify error is included in CSV + content, err := os.ReadFile(tmpFile) + assert.NoError(t, err) + assert.Contains(t, string(content), "false") + assert.Contains(t, string(content), "purchase failed: insufficient quota") +} + +func TestWriteMultiServiceCSVReport_InvalidPath(t *testing.T) { + // Use an invalid path (directory that doesn't exist) + invalidPath := "/nonexistent/directory/test.csv" + + results := []common.PurchaseResult{ + { + Recommendation: common.Recommendation{ + Service: common.ServiceRDS, + Region: "us-east-1", + ResourceType: "db.t3.small", + Count: 1, + }, + Success: true, + Timestamp: time.Now(), + }, + } + + err := writeMultiServiceCSVReport(results, invalidPath) + assert.Error(t, err) + assert.Contains(t, err.Error(), "failed to create CSV file") +} + +func TestWriteMultiServiceCSVReport_MultipleResults(t *testing.T) { + tmpFile := "/tmp/test_csv_multiple.csv" + defer os.Remove(tmpFile) + + results := []common.PurchaseResult{ + { + Recommendation: common.Recommendation{ + Service: common.ServiceRDS, + Region: "us-east-1", + ResourceType: "db.t3.small", + Count: 2, + EstimatedSavings: 150.25, + }, + Success: true, + CommitmentID: "commitment-1", + Timestamp: time.Now(), + }, + { + Recommendation: common.Recommendation{ + Service: common.ServiceElastiCache, + Region: "us-west-2", + ResourceType: "cache.t3.micro", + Count: 3, + EstimatedSavings: 75.50, + }, + Success: true, + CommitmentID: "commitment-2", + Timestamp: time.Now(), + }, + } + + err := writeMultiServiceCSVReport(results, tmpFile) + assert.NoError(t, err) + + // Verify both results are in CSV + content, err := os.ReadFile(tmpFile) + assert.NoError(t, err) + assert.Contains(t, string(content), "commitment-1") + assert.Contains(t, string(content), "commitment-2") + assert.Contains(t, string(content), "150.25") + assert.Contains(t, string(content), "75.50") +} + +// ==================== Additional ProcessPurchaseLoop Tests ==================== + +func TestProcessPurchaseLoopPurchaseFailure(t *testing.T) { + ctx := context.Background() + origCfg := toolCfg + defer func() { toolCfg = origCfg }() + + toolCfg.Coverage = 80.0 + toolCfg.SkipConfirmation = true + + recs := []common.Recommendation{ + {Service: common.ServiceRDS, ResourceType: "db.t3.large", Count: 1, EstimatedSavings: 500}, + } + + mockClient := &MockServiceClient{} + // Simulate a purchase failure + failureResult := common.PurchaseResult{ + Recommendation: recs[0], + Success: false, + CommitmentID: "", + Error: fmt.Errorf("API error: quota exceeded"), + Timestamp: time.Now(), + } + mockClient.On("PurchaseCommitment", ctx, recs[0], mock.MatchedBy(func(o common.PurchaseOptions) bool { return o.Source == common.PurchaseSourceCLI })).Return(failureResult, nil) + + t.Setenv("DISABLE_PURCHASE_DELAY", "true") + + results := processPurchaseLoop(ctx, recs, "ap-south-1", false, mockClient, toolCfg) + + assert.Len(t, results, 1) + assert.False(t, results[0].Success) + assert.NotNil(t, results[0].Error) + assert.Contains(t, results[0].Error.Error(), "quota exceeded") + + mockClient.AssertExpectations(t) +} + +func TestProcessPurchaseLoopUserCancellation(t *testing.T) { + ctx := context.Background() + origCfg := toolCfg + defer func() { toolCfg = origCfg }() + + toolCfg.Coverage = 90.0 + toolCfg.SkipConfirmation = false // User will be prompted + + recs := []common.Recommendation{ + {Service: common.ServiceEC2, ResourceType: "m5.large", Count: 10, EstimatedSavings: 5000}, + {Service: common.ServiceEC2, ResourceType: "m5.xlarge", Count: 5, EstimatedSavings: 3000}, + } + + mockClient := &MockServiceClient{} + // No expectations - should not be called if user cancels + + // Since we can't mock user input easily, we'll skip confirmation instead + // But the test verifies the cancellation logic is present + toolCfg.SkipConfirmation = true // Actually proceed for test + + // Setup mock to succeed + for _, rec := range recs { + result := common.PurchaseResult{ + Recommendation: rec, + Success: true, + CommitmentID: "test-id", + Timestamp: time.Now(), + } + mockClient.On("PurchaseCommitment", ctx, rec, mock.MatchedBy(func(o common.PurchaseOptions) bool { return o.Source == common.PurchaseSourceCLI })).Return(result, nil) + } + + t.Setenv("DISABLE_PURCHASE_DELAY", "true") + + results := processPurchaseLoop(ctx, recs, "eu-central-1", false, mockClient, toolCfg) + + assert.Len(t, results, 2) + for _, result := range results { + assert.True(t, result.Success) + } + + mockClient.AssertExpectations(t) +} + +func TestProcessPurchaseLoopEmptyRecommendations(t *testing.T) { + ctx := context.Background() + origCfg := toolCfg + defer func() { toolCfg = origCfg }() + + toolCfg.Coverage = 100.0 + + mockClient := &MockServiceClient{} + + results := processPurchaseLoop(ctx, []common.Recommendation{}, "us-east-1", false, mockClient, toolCfg) + + assert.Empty(t, results) + mockClient.AssertNotCalled(t, "PurchaseCommitment", mock.Anything, mock.Anything, mock.Anything) +} + +func TestProcessServicePurchasesUserCancellation(t *testing.T) { + ctx := context.Background() + origCfg := toolCfg + defer func() { toolCfg = origCfg }() + + toolCfg.Coverage = 85.0 + toolCfg.SkipConfirmation = true // Skip for testing + + recs := []common.Recommendation{ + {Service: common.ServiceElastiCache, ResourceType: "cache.r6g.large", Count: 2, EstimatedSavings: 200}, + } + + mockClient := &MockServiceClient{} + result := common.PurchaseResult{ + Recommendation: recs[0], + Success: true, + CommitmentID: "cache-purchase-123", + Timestamp: time.Now(), + } + mockClient.On("PurchaseCommitment", ctx, recs[0], mock.MatchedBy(func(o common.PurchaseOptions) bool { return o.Source == common.PurchaseSourceCLI })).Return(result, nil) + + t.Setenv("DISABLE_PURCHASE_DELAY", "true") + + results := processPurchaseLoop(ctx, recs, "us-west-1", false, mockClient, toolCfg) + + assert.Len(t, results, 1) + assert.True(t, results[0].Success) + assert.Equal(t, "cache-purchase-123", results[0].CommitmentID) + + mockClient.AssertExpectations(t) +} + +func TestProcessServicePurchasesDryRunMultiple(t *testing.T) { + ctx := context.Background() + origCfg := toolCfg + defer func() { toolCfg = origCfg }() + + toolCfg.Coverage = 100.0 + + recs := []common.Recommendation{ + {Service: common.ServiceRDS, ResourceType: "db.r5.xlarge", Count: 5, EstimatedSavings: 1000}, + {Service: common.ServiceRDS, ResourceType: "db.r5.2xlarge", Count: 3, EstimatedSavings: 800}, + {Service: common.ServiceRDS, ResourceType: "db.r5.4xlarge", Count: 1, EstimatedSavings: 500}, + } + + mockClient := &MockServiceClient{} + // Dry run should not call PurchaseCommitment + + results := processPurchaseLoop(ctx, recs, "ap-northeast-1", true, mockClient, toolCfg) + + assert.Len(t, results, 3) + for i, result := range results { + assert.True(t, result.Success) + assert.True(t, result.DryRun) + assert.Contains(t, result.CommitmentID, "dryrun") + assert.Nil(t, result.Error) + assert.Equal(t, recs[i].ResourceType, result.Recommendation.ResourceType) + } + + mockClient.AssertNotCalled(t, "PurchaseCommitment", mock.Anything, mock.Anything, mock.Anything) +} + +// ==================== New Extracted Function Tests ==================== + +func TestExecutePurchaseWithEmptyPurchaseID(t *testing.T) { + ctx := context.Background() + // Save original values + origCfg := toolCfg + + defer func() { + toolCfg = origCfg + }() + + toolCfg.Coverage = 85.0 + + rec := common.Recommendation{ + Service: common.ServiceElastiCache, + ResourceType: "cache.r5.large", + Count: 3, + } + + mockClient := &MockServiceClient{} + // Return result without PurchaseID + expectedResult := common.PurchaseResult{ + Recommendation: rec, + Success: true, + CommitmentID: "", // Empty ID - should be generated + Error: nil, + Timestamp: time.Now(), + } + mockClient.On("PurchaseCommitment", ctx, rec, mock.MatchedBy(func(o common.PurchaseOptions) bool { return o.Source == common.PurchaseSourceCLI })).Return(expectedResult, nil) + + // Logger output disabled for testing + + result := executePurchase(ctx, rec, "ap-southeast-1", 2, mockClient, toolCfg) + + assert.True(t, result.Success) + assert.NotEmpty(t, result.CommitmentID) // Should have generated ID + assert.Contains(t, result.CommitmentID, "ap-southeast-1") + + mockClient.AssertExpectations(t) +} + +func TestProcessPurchaseLoopDryRun(t *testing.T) { + ctx := context.Background() + // Save original values + origCfg := toolCfg + + defer func() { + toolCfg = origCfg + }() + + toolCfg.Coverage = 75.0 + + recs := []common.Recommendation{ + {Service: common.ServiceRDS, ResourceType: "db.t3.small", Count: 2, SourceRecommendation: "Test 1"}, + {Service: common.ServiceRDS, ResourceType: "db.t3.medium", Count: 3, SourceRecommendation: "Test 2"}, + } + + mockClient := &MockServiceClient{} + + // Logger output disabled for testing + + results := processPurchaseLoop(ctx, recs, "us-east-1", true, mockClient, toolCfg) + + assert.Len(t, results, 2) + for _, result := range results { + assert.True(t, result.Success) + assert.Nil(t, result.Error) // Dry runs are successful, so no error + assert.True(t, result.DryRun) + assert.Contains(t, result.CommitmentID, "dryrun") + } + + // Mock should not be called in dry run mode + mockClient.AssertNotCalled(t, "PurchaseCommitment", mock.Anything, mock.Anything, mock.Anything) +} + +func TestProcessPurchaseLoopActualPurchase(t *testing.T) { + ctx := context.Background() + // Save original values + origCfg := toolCfg + + defer func() { + toolCfg = origCfg + }() + + toolCfg.Coverage = 80.0 + toolCfg.SkipConfirmation = true // Skip confirmation for testing + + recs := []common.Recommendation{ + {Service: common.ServiceEC2, ResourceType: "t3.small", Count: 1, SourceRecommendation: "EC2 Test 1", EstimatedSavings: 100}, + {Service: common.ServiceEC2, ResourceType: "t3.medium", Count: 2, SourceRecommendation: "EC2 Test 2", EstimatedSavings: 200}, + } + + mockClient := &MockServiceClient{} + for i, rec := range recs { + result := common.PurchaseResult{ + Recommendation: rec, + Success: true, + CommitmentID: fmt.Sprintf("purchase-id-%d", i), + Error: nil, + Timestamp: time.Now(), + } + mockClient.On("PurchaseCommitment", ctx, rec, mock.MatchedBy(func(o common.PurchaseOptions) bool { return o.Source == common.PurchaseSourceCLI })).Return(result, nil) + } + + // Logger output disabled for testing + + // Disable purchase delay for testing + t.Setenv("DISABLE_PURCHASE_DELAY", "true") + + results := processPurchaseLoop(ctx, recs, "eu-west-1", false, mockClient, toolCfg) + + assert.Len(t, results, 2) + for i, result := range results { + assert.True(t, result.Success) + assert.Equal(t, fmt.Sprintf("purchase-id-%d", i), result.CommitmentID) + } + + mockClient.AssertExpectations(t) +} + +func TestProcessPurchaseLoopWithConfirmation(t *testing.T) { + ctx := context.Background() + // Save original values + origCfg := toolCfg + + defer func() { + toolCfg = origCfg + }() + + toolCfg.Coverage = 80.0 + toolCfg.SkipConfirmation = true // Skip confirmation to proceed with purchase + + recs := []common.Recommendation{ + {Service: common.ServiceRDS, ResourceType: "db.r5.large", Count: 5, SourceRecommendation: "Expensive", EstimatedSavings: 1000}, + } + + mockClient := &MockServiceClient{} + // Mock the purchase since skipConfirmation=true will proceed + result := common.PurchaseResult{ + Recommendation: recs[0], + Success: true, + CommitmentID: "confirmed-purchase-123", + Error: nil, + Timestamp: time.Now(), + } + mockClient.On("PurchaseCommitment", ctx, recs[0], mock.MatchedBy(func(o common.PurchaseOptions) bool { return o.Source == common.PurchaseSourceCLI })).Return(result, nil) + + // Logger output disabled for testing + + // Disable purchase delay for testing + t.Setenv("DISABLE_PURCHASE_DELAY", "true") + + results := processPurchaseLoop(ctx, recs, "us-west-2", false, mockClient, toolCfg) + + assert.Len(t, results, 1) + assert.True(t, results[0].Success) + assert.Equal(t, "confirmed-purchase-123", results[0].CommitmentID) + + mockClient.AssertExpectations(t) +} + +func TestFilterAndAdjustRecommendations(t *testing.T) { + // Save and restore ALL global variables + saved := saveGlobalVars() + defer saved.restore() + + tests := []struct { + setupFilters func() + name string + recommendations []common.Recommendation + coverage float64 + expectedMin int + expectedMax int + }{ + { + name: "100% coverage no filters", + recommendations: []common.Recommendation{ + {Service: common.ServiceRDS, ResourceType: "db.t3.small", Count: 5}, + {Service: common.ServiceRDS, ResourceType: "db.t3.medium", Count: 3}, + }, + coverage: 100.0, + setupFilters: func() { + toolCfg.MaxInstances = 0 + toolCfg.OverrideCount = 0 + }, + expectedMin: 2, + expectedMax: 2, + }, + { + name: "50% coverage", + recommendations: []common.Recommendation{ + {Service: common.ServiceRDS, ResourceType: "db.t3.small", Count: 10}, + {Service: common.ServiceRDS, ResourceType: "db.t3.medium", Count: 6}, + }, + coverage: 50.0, + setupFilters: func() { + toolCfg.MaxInstances = 0 + toolCfg.OverrideCount = 0 + }, + expectedMin: 1, + expectedMax: 2, + }, + { + name: "Instance limit applied", + // EstimatedSavings is set because a binding --max-instances has to + // rank the rows it chooses between; requireRankingSignal refuses a + // run whose rows carry no savings value rather than capping them by + // name. + recommendations: []common.Recommendation{ + {Service: common.ServiceRDS, ResourceType: "db.t3.small", Count: 10, EstimatedSavings: 100}, + {Service: common.ServiceRDS, ResourceType: "db.t3.medium", Count: 10, EstimatedSavings: 200}, + {Service: common.ServiceRDS, ResourceType: "db.t3.large", Count: 10, EstimatedSavings: 300}, + }, + coverage: 100.0, + setupFilters: func() { + toolCfg.MaxInstances = 15 + toolCfg.OverrideCount = 0 + }, + expectedMin: 1, + expectedMax: 3, + }, + } + + for _, tt := range tests { + t.Run(tt.name, func(t *testing.T) { + // Setup filters + tt.setupFilters() + + // Suppress logger + // Logger output disabled for testing + + result, err := filterAndAdjustRecommendations(tt.recommendations, tt.coverage, toolCfg) + require.NoError(t, err) + + // Verify result is within expected range + assert.GreaterOrEqual(t, len(result), tt.expectedMin) + assert.LessOrEqual(t, len(result), tt.expectedMax) + + // Verify all results have count > 0 + for _, rec := range result { + assert.Positive(t, rec.Count) + } + }) + } +} + +// isolateAWSEnv pins the AWS environment to deterministic invalid static +// credentials so the runToolFromCSV tests behave identically on credentialed +// developer machines and bare CI runners: config loading succeeds, every +// outbound AWS API call fails and takes the warn-and-continue path, and no +// real account is ever touched. +func isolateAWSEnv(t *testing.T) { + t.Helper() + t.Setenv("AWS_ACCESS_KEY_ID", "invalid-test-key-id") + t.Setenv("AWS_SECRET_ACCESS_KEY", "invalid-test-secret") + t.Setenv("AWS_SESSION_TOKEN", "") + t.Setenv("AWS_PROFILE", "") + t.Setenv("AWS_CONFIG_FILE", os.DevNull) + t.Setenv("AWS_SHARED_CREDENTIALS_FILE", os.DevNull) + t.Setenv("AWS_EC2_METADATA_DISABLED", "true") +} + +// readPurchaseReport parses the CSV purchase report written by runToolFromCSV +// and returns the header plus per-purchase data rows. The trailing TOTAL +// summary row is skipped, mirroring loadRecommendationsFromCSV. +func readPurchaseReport(t *testing.T, path string) ([]string, [][]string) { + t.Helper() + f, err := os.Open(path) + require.NoError(t, err, "runToolFromCSV must write the purchase report") + defer func() { _ = f.Close() }() + rows, err := csv.NewReader(f).ReadAll() + require.NoError(t, err) + require.NotEmpty(t, rows, "purchase report must contain a header row") + dataRows := make([][]string, 0, len(rows)-1) + for _, row := range rows[1:] { + if len(row) > 0 && row[0] == "TOTAL" { + continue + } + dataRows = append(dataRows, row) + } + return rows[0], dataRows +} + +// reportColumn returns the values of the named column across all data rows. +func reportColumn(t *testing.T, header []string, rows [][]string, name string) []string { + t.Helper() + idx := -1 + for i, col := range header { + if col == name { + idx = i + break + } + } + require.NotEqual(t, -1, idx, "column %s missing from purchase report header", name) + vals := make([]string, 0, len(rows)) + for _, row := range rows { + vals = append(vals, row[idx]) + } + return vals +} + +// writeTestRecommendationsCSV writes a recommendations CSV into a temp dir +// and returns its path. +func writeTestRecommendationsCSV(t *testing.T, csvData string) string { + t.Helper() + path := filepath.Join(t.TempDir(), "recommendations.csv") + require.NoError(t, os.WriteFile(path, []byte(csvData), 0o600)) + return path +} + +func TestRunToolFromCSV(t *testing.T) { + origCfg := toolCfg + defer func() { toolCfg = origCfg }() + isolateAWSEnv(t) + + // Column names match writeMultiServiceCSVReport's header (the real + // round-trip shape). The previous fixture used headers the parser + // silently ignored, so every row loaded with Count=0 and the whole run + // processed nothing while the NotPanics assertion stayed green. + csvPath := writeTestRecommendationsCSV(t, `Service,Region,ResourceType,Engine,Count,Term,PaymentOption,Account +rds,us-east-1,db.t3.small,postgres,2,1yr,All Upfront,123456789012 +elasticache,us-west-2,cache.t3.micro,redis,1,1yr,All Upfront,123456789012 +`) + + tests := []struct { + name string + coverage float64 + wantResourceTypes []string + wantCounts []string + }{ + { + // 100% coverage keeps both recommendations at their CSV counts. + name: "Dry run writes a report row per recommendation", + coverage: 100.0, + wantResourceTypes: []string{"db.t3.small", "cache.t3.micro"}, + wantCounts: []string{"2", "1"}, + }, + { + // 50% coverage halves db.t3.small (2 -> 1) and drops + // cache.t3.micro (1 -> 0). + name: "Coverage sizing is applied before the purchase loop", + coverage: 50.0, + wantResourceTypes: []string{"db.t3.small"}, + wantCounts: []string{"1"}, + }, + } + + for _, tt := range tests { + t.Run(tt.name, func(t *testing.T) { + reportPath := filepath.Join(t.TempDir(), "report.csv") + toolCfg.CSVInput = csvPath + toolCfg.CSVOutput = reportPath + toolCfg.ActualPurchase = false + toolCfg.Coverage = tt.coverage + toolCfg.TargetCoverage = 0 + toolCfg.MaxInstances = 0 + toolCfg.OverrideCount = 0 + + err := runToolFromCSV(context.Background(), toolCfg) + require.NoError(t, err) + + header, rows := readPurchaseReport(t, reportPath) + assert.ElementsMatch(t, tt.wantResourceTypes, reportColumn(t, header, rows, "ResourceType")) + assert.ElementsMatch(t, tt.wantCounts, reportColumn(t, header, rows, "Count")) + for _, success := range reportColumn(t, header, rows, "Success") { + assert.Equal(t, "true", success, "dry-run purchases must be recorded as successful") + } + }) + } +} + +// TestRunToolFromCSV_NonExistentFile asserts the unreadable-input error path +// surfaces as an error (previously log.Fatalf, which made it untestable). +func TestRunToolFromCSV_NonExistentFile(t *testing.T) { + origCfg := toolCfg + defer func() { toolCfg = origCfg }() + + toolCfg.CSVInput = filepath.Join(t.TempDir(), "does-not-exist.csv") + toolCfg.ActualPurchase = false + + err := runToolFromCSV(context.Background(), toolCfg) + require.Error(t, err) + assert.Contains(t, err.Error(), "failed to read CSV file") + assert.Contains(t, err.Error(), "failed to open CSV file") +} + +// TestRunToolFromCSV_EmptyFile asserts a CSV without a header row is rejected +// with a parse error rather than being silently treated as zero rows. +func TestRunToolFromCSV_EmptyFile(t *testing.T) { + origCfg := toolCfg + defer func() { toolCfg = origCfg }() + + toolCfg.CSVInput = writeTestRecommendationsCSV(t, "") + toolCfg.ActualPurchase = false + + err := runToolFromCSV(context.Background(), toolCfg) + require.Error(t, err) + assert.Contains(t, err.Error(), "failed to read CSV file") + assert.Contains(t, err.Error(), "failed to read CSV header") +} + +func TestRunToolFromCSV_WithMaxInstances(t *testing.T) { + origCfg := toolCfg + defer func() { toolCfg = origCfg }() + isolateAWSEnv(t) + + // EstimatedSavings is populated because a binding --max-instances has to + // rank the rows it chooses between. A file without that column is refused + // rather than capped by name; that case is covered by + // TestCSVCapRefusesRowsWithoutARankingSignal. + csvPath := writeTestRecommendationsCSV(t, `Service,Region,ResourceType,Engine,Count,EstimatedSavings,Term,PaymentOption,Account +rds,us-east-1,db.t3.small,postgres,10,100.00,1yr,All Upfront,123456789012 +rds,us-east-1,db.t3.medium,mysql,10,200.00,1yr,All Upfront,123456789012 +rds,us-east-1,db.t3.large,postgres,10,300.00,1yr,All Upfront,123456789012 +`) + + reportPath := filepath.Join(t.TempDir(), "report.csv") + toolCfg.CSVInput = csvPath + toolCfg.CSVOutput = reportPath + toolCfg.ActualPurchase = false + toolCfg.Coverage = 100.0 + toolCfg.TargetCoverage = 0 + toolCfg.OverrideCount = 0 + toolCfg.MaxInstances = 15 // Should limit total instances to 15 + + err := runToolFromCSV(context.Background(), toolCfg) + require.NoError(t, err) + + header, rows := readPurchaseReport(t, reportPath) + total := 0 + for _, raw := range reportColumn(t, header, rows, "Count") { + n, convErr := strconv.Atoi(raw) + require.NoError(t, convErr) + total += n + } + assert.Positive(t, total, "some instances must survive the max-instances cap") + assert.LessOrEqual(t, total, 15, "purchased instance total must respect --max-instances") +} + +func TestRunToolFromCSV_WithOverrideCount(t *testing.T) { + origCfg := toolCfg + defer func() { toolCfg = origCfg }() + isolateAWSEnv(t) + + csvPath := writeTestRecommendationsCSV(t, `Service,Region,ResourceType,Engine,Count,Term,PaymentOption,Account +rds,us-east-1,db.t3.small,postgres,10,1yr,All Upfront,123456789012 +rds,us-east-1,db.t3.medium,mysql,5,1yr,All Upfront,123456789012 +`) + + reportPath := filepath.Join(t.TempDir(), "report.csv") + toolCfg.CSVInput = csvPath + toolCfg.CSVOutput = reportPath + toolCfg.ActualPurchase = false + toolCfg.Coverage = 100.0 + toolCfg.TargetCoverage = 0 + toolCfg.MaxInstances = 0 + toolCfg.OverrideCount = 3 // Override each recommendation to count=3 + + err := runToolFromCSV(context.Background(), toolCfg) + require.NoError(t, err) + + header, rows := readPurchaseReport(t, reportPath) + counts := reportColumn(t, header, rows, "Count") + require.Len(t, counts, 2, "both recommendations must reach the purchase loop") + for _, raw := range counts { + assert.Equal(t, "3", raw, "--override-count must replace every recommendation count") + } +} + +// ==================== Tests for adjustRecommendationForExcludedVersions ==================== + +// Helper to create test version info with extended support dates. +func createTestVersionInfo() map[string]MajorEngineVersionInfo { + now := time.Now() + pastDate := now.AddDate(0, -6, 0) // 6 months ago + futureDate := now.AddDate(3, 0, 0) // 3 years from now + + return map[string]MajorEngineVersionInfo{ + "aurora-mysql:5.7": { + Engine: "aurora-mysql", + MajorEngineVersion: "5.7", + SupportedEngineLifecycles: []EngineLifecycleInfo{ + { + LifecycleSupportName: string(rdstypes.LifecycleSupportNameOpenSourceRdsStandardSupport), + LifecycleSupportStartDate: now.AddDate(-5, 0, 0), + LifecycleSupportEndDate: pastDate, + }, + { + LifecycleSupportName: string(rdstypes.LifecycleSupportNameOpenSourceRdsExtendedSupport), + LifecycleSupportStartDate: pastDate, + LifecycleSupportEndDate: futureDate, + }, + }, + }, + "aurora-mysql:8.0": { + Engine: "aurora-mysql", + MajorEngineVersion: "8.0", + SupportedEngineLifecycles: []EngineLifecycleInfo{ + { + LifecycleSupportName: string(rdstypes.LifecycleSupportNameOpenSourceRdsStandardSupport), + LifecycleSupportStartDate: now.AddDate(-2, 0, 0), + LifecycleSupportEndDate: futureDate, + }, + }, + }, + } +} + +// ==================== generateCSVFilename Tests ==================== + +// ==================== printRunMode Tests ==================== + +// ==================== printPaymentAndTerm Tests ==================== + +// ==================== extractMajorVersion Tests ==================== + +// ==================== determineServicesToProcess Tests ==================== + +// ==================== determineCSVCoverage Tests ==================== diff --git a/cmd/multi_service_test_common_test.go b/cmd/multi_service_test_common_test.go new file mode 100644 index 000000000..24097be8d --- /dev/null +++ b/cmd/multi_service_test_common_test.go @@ -0,0 +1,199 @@ +package main + +import ( + "context" + "sync" + + "github.com/LeanerCloud/CUDly/pkg/common" + "github.com/aws/aws-sdk-go-v2/aws" + "github.com/aws/aws-sdk-go-v2/service/ec2" + "github.com/aws/aws-sdk-go-v2/service/organizations" + "github.com/stretchr/testify/mock" +) + +// ==================== Mock Implementations ==================== + +// MockEC2Client for testing getAllAWSRegions. +type MockEC2Client struct { + mock.Mock +} + +func (m *MockEC2Client) DescribeRegions(ctx context.Context, params *ec2.DescribeRegionsInput, optFns ...func(*ec2.Options)) (*ec2.DescribeRegionsOutput, error) { + args := m.Called(ctx, params) + if args.Get(0) == nil { + return nil, args.Error(1) + } + return args.Get(0).(*ec2.DescribeRegionsOutput), args.Error(1) +} + +// MockRecommendationsClient for testing. +type MockRecommendationsClient struct { + mock.Mock +} + +func (m *MockRecommendationsClient) GetRecommendations(ctx context.Context, params *common.RecommendationParams) ([]common.Recommendation, error) { + args := m.Called(ctx, params) + if args.Get(0) == nil { + return nil, args.Error(1) + } + return args.Get(0).([]common.Recommendation), args.Error(1) +} + +func (m *MockRecommendationsClient) GetRecommendationsForService(ctx context.Context, service common.ServiceType) ([]common.Recommendation, error) { + args := m.Called(ctx, service) + if args.Get(0) == nil { + return nil, args.Error(1) + } + return args.Get(0).([]common.Recommendation), args.Error(1) +} + +func (m *MockRecommendationsClient) GetAllRecommendations(ctx context.Context) ([]common.Recommendation, error) { + args := m.Called(ctx) + if args.Get(0) == nil { + return nil, args.Error(1) + } + return args.Get(0).([]common.Recommendation), args.Error(1) +} + +// MockServiceClient implements provider.ServiceClient for testing. +type MockServiceClient struct { + mock.Mock +} + +func (m *MockServiceClient) GetServiceType() common.ServiceType { + args := m.Called() + return args.Get(0).(common.ServiceType) +} + +func (m *MockServiceClient) GetRegion() string { + args := m.Called() + return args.String(0) +} + +func (m *MockServiceClient) GetRecommendations(ctx context.Context, params *common.RecommendationParams) ([]common.Recommendation, error) { + args := m.Called(ctx, params) + if args.Get(0) == nil { + return nil, args.Error(1) + } + return args.Get(0).([]common.Recommendation), args.Error(1) +} + +func (m *MockServiceClient) PurchaseCommitment(ctx context.Context, rec common.Recommendation, opts common.PurchaseOptions) (common.PurchaseResult, error) { + args := m.Called(ctx, rec, opts) + return args.Get(0).(common.PurchaseResult), args.Error(1) +} + +func (m *MockServiceClient) ValidateOffering(ctx context.Context, rec common.Recommendation) error { + args := m.Called(ctx, rec) + return args.Error(0) +} + +func (m *MockServiceClient) GetOfferingDetails(ctx context.Context, rec common.Recommendation) (*common.OfferingDetails, error) { + args := m.Called(ctx, rec) + if args.Get(0) == nil { + return nil, args.Error(1) + } + return args.Get(0).(*common.OfferingDetails), args.Error(1) +} + +func (m *MockServiceClient) GetExistingCommitments(ctx context.Context) ([]common.Commitment, error) { + args := m.Called(ctx) + if args.Get(0) == nil { + return nil, args.Error(1) + } + return args.Get(0).([]common.Commitment), args.Error(1) +} + +func (m *MockServiceClient) GetValidResourceTypes(ctx context.Context) ([]string, error) { + args := m.Called(ctx) + if args.Get(0) == nil { + return nil, args.Error(1) + } + return args.Get(0).([]string), args.Error(1) +} + +// ==================== Test Helpers ==================== + +// globalVarsSnapshot captures the toolCfg for tests. +type globalVarsSnapshot struct { + cfg Config +} + +// saveGlobalVars captures current toolCfg state. +func saveGlobalVars() *globalVarsSnapshot { + return &globalVarsSnapshot{ + cfg: toolCfg, + } +} + +// restoreGlobalVars restores toolCfg state from snapshot. +func (s *globalVarsSnapshot) restore() { + toolCfg = s.cfg +} + +// OrganizationsClientAPI is an interface for organizations client operations. +type OrganizationsClientAPI interface { + DescribeAccount(ctx context.Context, params *organizations.DescribeAccountInput, optFns ...func(*organizations.Options)) (*organizations.DescribeAccountOutput, error) +} + +// MockOrganizationsClient for testing account alias cache. +type MockOrganizationsClient struct { + mock.Mock +} + +func (m *MockOrganizationsClient) DescribeAccount(ctx context.Context, params *organizations.DescribeAccountInput, optFns ...func(*organizations.Options)) (*organizations.DescribeAccountOutput, error) { + args := m.Called(ctx, params) + if args.Get(0) == nil { + return nil, args.Error(1) + } + return args.Get(0).(*organizations.DescribeAccountOutput), args.Error(1) +} + +// TestAccountAliasCache is a test-friendly version of AccountAliasCache. +type TestAccountAliasCache struct { + orgClient OrganizationsClientAPI + cache map[string]string + mu sync.RWMutex +} + +// GetAccountAlias returns the account alias for an account ID (same logic as production). +func (c *TestAccountAliasCache) GetAccountAlias(ctx context.Context, accountID string) string { + if accountID == "" { + return "" + } + + c.mu.RLock() + if alias, ok := c.cache[accountID]; ok { + c.mu.RUnlock() + return alias + } + c.mu.RUnlock() + + // Try to fetch from Organizations + c.mu.Lock() + defer c.mu.Unlock() + + // Double-check after acquiring write lock + if alias, ok := c.cache[accountID]; ok { + return alias + } + + // Try to describe the account + result, err := c.orgClient.DescribeAccount(ctx, &organizations.DescribeAccountInput{ + AccountId: aws.String(accountID), + }) + if err != nil { + c.cache[accountID] = accountID // Use ID as fallback + return accountID + } + + if result.Account != nil && result.Account.Name != nil { + c.cache[accountID] = *result.Account.Name + return *result.Account.Name + } + + c.cache[accountID] = accountID + return accountID +} + +// ==================== Core Function Tests ==================== diff --git a/cmd/rekey/README.md b/cmd/rekey/README.md new file mode 100644 index 000000000..eb51ca360 --- /dev/null +++ b/cmd/rekey/README.md @@ -0,0 +1,144 @@ +# Rekey credentials from the zero dev-key — Azure & GCP only + +## Context + +Before the credential-encryption-key fix landed, Azure Container Apps and GCP +Cloud Run deployments silently used the all-zero AES-256 dev key for every +tenant credential write. The Go code read `CREDENTIAL_ENCRYPTION_KEY_SECRET_ARN` +but Terraform for those clouds wrote `_SECRET_NAME` (Azure) and `_SECRET_ID` +(GCP), so the loader never matched and fell through to the dev key. + +After deploying the fix, the service loads the real key from each cloud's +secret store. Existing rows in `account_credentials` remain encrypted under +the zero key and **cannot be decrypted** by the new service until they are +re-encrypted under the real key. This runbook walks operators through that +one-shot migration. + +> **AWS deployments do not need this** — the env var name matched, so AWS +> rows have always been encrypted under the real key. + +## Migration window — read this first + +Between PR deploy (step 1) and rekey completion (step 3), any read of an +existing tenant credential will fail with a decrypt error. The service stays +up; only credential-touching operations (recommendation listing for +onboarded accounts, scheduled purchases, etc.) on Azure/GCP are affected. +Schedule the run during a low-traffic window. For dev-stage Azure/GCP +deployments the impact is expected to be small; confirm with the operator +before proceeding. + +`CREDENTIAL_ENCRYPTION_ALLOW_DEV_KEY=1` is **not** part of this migration — +it exists only for local dev. Do not set it in any deployed environment. + +## Pre-flight checklist + +- [ ] PR fixing `loadKey` to read the per-cloud env vars is **deployed and + live** — confirm by tailing logs for a startup line of the form + `credentials: loaded encryption key via CREDENTIAL_ENCRYPTION_KEY_SECRET_NAME` + (Azure) or `…_SECRET_ID` (GCP). The env var name in that line MUST NOT + be `CREDENTIAL_ENCRYPTION_KEY` (raw hex, misconfiguration) or + `CREDENTIAL_ENCRYPTION_ALLOW_DEV_KEY` (dev fallback engaged). +- [ ] `/health` returns `credential_store: { status: "healthy" }` — confirms + the real key passes the round-trip self-check. +- [ ] You have the same `DATABASE_*` and `CREDENTIAL_ENCRYPTION_KEY_SECRET_*` + env vars the service uses, scoped to a one-off job runner inside the + same network as the DB (see step 2). + +## Steps + +### 1. Deploy the fix normally + +Merge the security PR and let CI deploy. No special flags. The service starts +with the real key. Existing zero-key rows remain in the DB and fail to +decrypt on read, but the service otherwise runs. + +### 2. Run `cmd/rekey` as a one-off job, in-cluster + +Build the binary into the same container image used for production (or run +via `go run ./cmd/rekey`). Launch as a short-lived job in the same +network/identity as the running service so it can reach Postgres and the +secret store: + +- **Azure**: Container Apps job, same identity + Key Vault access as the + main app. +- **GCP**: Cloud Run Job, same service account as the main app. +- **AWS**: not needed (was never broken). + +Required env vars on the job: + +```bash +CUDLY_REKEY_FROM_ZERO_KEY=1 # safety gate +CREDENTIAL_ENCRYPTION_KEY_SECRET_NAME= # Azure (or) +CREDENTIAL_ENCRYPTION_KEY_SECRET_ID= # GCP +SECRET_PROVIDER=azure # (or gcp) +DATABASE_HOST=... # same as service +DATABASE_USER=... +DATABASE_NAME=... +DATABASE_PASSWORD_SECRET=... # if used +AZURE_KEY_VAULT_URL=... # Azure only +GCP_PROJECT_ID=... # GCP only +``` + +Capture the final log line, which has the form: + +```text +rekey: scanned=N re_keyed=N skippedAlreadyReal=N errored=0 +``` + +### 3. Verify + +- `errored == 0` — every row processed. +- `scanned == reKeyed + skippedAlreadyReal` — no rows were missed. +- Spot-check one onboarded tenant in the dashboard: their AWS/Azure/GCP + account should list recommendations without an "invalid ciphertext" error. + +If `errored > 0`, do not retry blindly. Inspect the logged row IDs (no +plaintext is logged) and resolve manually before re-running. + +### 4. No redeploy required + +The service has been running with the real key since step 1; the rekey job +just transformed at-rest data. + +### 5. Post-flight verification + +- `/health` still returns `credential_store: { status: "healthy" }`. +- Application logs no longer show "decrypt: ..." ERROR lines on credential + reads (these were the symptom of pre-rekey state). + +## Operational security + +- The job decrypts plaintext credentials in memory for the duration of one + row's transaction. It never writes plaintext to logs or stdout. +- Run the job in an ephemeral container that is torn down immediately after + completion. Do not leave the job spec lying around with + `CUDLY_REKEY_FROM_ZERO_KEY=1` set. +- The runtime may surface env var values in invocation history (Azure + Activity log, GCP Cloud Run Jobs console, AWS CloudWatch). Only + `CUDLY_REKEY_FROM_ZERO_KEY=1` should be visible there as an out-of-band + flag — the secret name/ID itself is identical to what the production + service exposes already. + +## Idempotency + +The job is safe to rerun. Real-key rows fail the zero-key Decrypt check +(AES-GCM authentication tag mismatch) and fall into the +`skippedAlreadyReal` bucket. The second run reports +`reKeyed=0 skippedAlreadyReal=N`. + +## Troubleshooting + +- **`real key is the all-zero dev key — refusing to rekey`**: the job + loaded the zero key as the "real" key, which means the production env + vars are still misconfigured. Re-check step 1's pre-flight log line. +- **`failed to load credential encryption key: …no credential encryption + key configured`**: the job's env vars don't match the production service. + Re-export `CREDENTIAL_ENCRYPTION_KEY_SECRET_*` and `SECRET_PROVIDER`. +- **Many `errored` rows**: usually a DB transient. Wait, then rerun — the + re-keyed rows from the partial run land in `skippedAlreadyReal` next time. + +## Rollback + +There is no rollback. Once a row is encrypted under the real key, the zero +key cannot decrypt it. If something is wrong with the real key itself, +restore from a DB snapshot taken before the migration window. diff --git a/cmd/rekey/main.go b/cmd/rekey/main.go new file mode 100644 index 000000000..70dc6285b --- /dev/null +++ b/cmd/rekey/main.go @@ -0,0 +1,208 @@ +// Command rekey is a one-shot migration that re-encrypts every row in +// account_credentials whose ciphertext was produced under the all-zero dev key +// (because the cipher.go silent-fallback bug was active at write time on +// Azure Container Apps and GCP Cloud Run before the security fix landed). +// +// Operator runbook: cmd/rekey/README.md +// +// Refuses to run unless CUDLY_REKEY_FROM_ZERO_KEY=1 is set. Idempotent — +// running it a second time finds nothing to re-key (zero-key Decrypt fails +// on real-key rows, so they fall into the "skipped" bucket). +package main + +import ( + "context" + "flag" + "fmt" + "log" + "os" + "time" + + "github.com/jackc/pgx/v5" + + "github.com/LeanerCloud/CUDly/internal/credentials" + "github.com/LeanerCloud/CUDly/internal/database" + "github.com/LeanerCloud/CUDly/internal/secrets" +) + +const safetyEnv = "CUDLY_REKEY_FROM_ZERO_KEY" + +type counters struct { + scanned, reKeyed, skippedAlreadyReal, errored int +} + +func main() { + timeout := flag.Duration("timeout", 10*time.Minute, "overall timeout for the migration") + flag.Parse() + + if os.Getenv(safetyEnv) != "1" { + log.Fatalf("rekey: refusing to run without %s=1 (see cmd/rekey/README.md)", safetyEnv) + } + + ctx, cancel := context.WithTimeout(context.Background(), *timeout) + defer cancel() + + if err := run(ctx); err != nil { + log.Fatalf("rekey: %v", err) //nolint:gocritic // exitAfterDefer: intentional fatal; cancel() best-effort on timeout + } +} + +func run(ctx context.Context) error { + resolver, err := buildResolver(ctx) + if err != nil { + return fmt.Errorf("build resolver: %w", err) + } + defer func() { + if cerr := resolver.Close(); cerr != nil { + log.Printf("rekey: warning: resolver close: %v", cerr) + } + }() + + realKey, source, err := credentials.LoadKey(ctx, resolver) + if err != nil { + return fmt.Errorf("load real key: %w", err) + } + log.Printf("rekey: loaded real key via %s", source) + + zeroKey := credentials.DevKey() + if isEqual(realKey, zeroKey) { + return fmt.Errorf("real key is the all-zero dev key — refusing to rekey (would be a no-op)") + } + + db, err := initDB(ctx, resolver) + if err != nil { + return fmt.Errorf("connect db: %w", err) + } + defer db.Close() + + cs, err := rekeyAccountCredentials(ctx, db, zeroKey, realKey) + if err != nil { + return fmt.Errorf("rekey: %w", err) + } + log.Printf("rekey: scanned=%d re_keyed=%d skipped_already_real=%d errored=%d", + cs.scanned, cs.reKeyed, cs.skippedAlreadyReal, cs.errored) + + if cs.errored > 0 { + return fmt.Errorf("rekey: %d rows errored — see logs for IDs", cs.errored) + } + return nil +} + +// buildResolver wires the same secrets.Resolver the production server uses. +func buildResolver(ctx context.Context) (secrets.Resolver, error) { + return secrets.NewResolver(ctx, secrets.LoadConfigFromEnv()) +} + +// initDB connects using the same env-driven path the production server uses, +// so the rekey job runs against the same DB as the live service. +func initDB(ctx context.Context, resolver secrets.Resolver) (*database.Connection, error) { + dbConfig, err := database.LoadFromEnv() + if err != nil { + return nil, fmt.Errorf("load db config: %w", err) + } + var sr database.SecretResolver + if dbConfig.PasswordSecret != "" { + sr = resolver + } + return database.NewConnection(ctx, dbConfig, sr) +} + +// rekeyAccountCredentials walks account_credentials and re-encrypts every row +// whose ciphertext decrypts under the zero key. Real-key rows are detected by +// decrypt failure (AES-GCM authentication tag mismatch) and skipped. Each +// re-key is committed in its own transaction so a partial run leaves the +// database in a consistent state. +func rekeyAccountCredentials(ctx context.Context, db *database.Connection, zeroKey, realKey []byte) (counters, error) { + var cs counters + + rows, err := db.Query(ctx, `SELECT id, encrypted_blob FROM account_credentials`) + if err != nil { + return cs, fmt.Errorf("query: %w", err) + } + + type row struct { + id string + blob string + } + var pending []row + for rows.Next() { + var r row + if err := rows.Scan(&r.id, &r.blob); err != nil { + rows.Close() + return cs, fmt.Errorf("scan: %w", err) + } + pending = append(pending, r) + } + rows.Close() + if err := rows.Err(); err != nil { + return cs, fmt.Errorf("iter: %w", err) + } + + for _, r := range pending { + cs.scanned++ + outcome := rekeyOne(ctx, db, r.id, r.blob, zeroKey, realKey) + switch outcome { + case outcomeReKeyed: + cs.reKeyed++ + case outcomeSkipped: + cs.skippedAlreadyReal++ + case outcomeErrored: + cs.errored++ + } + } + return cs, nil +} + +type rekeyOutcome int + +const ( + outcomeReKeyed rekeyOutcome = iota + outcomeSkipped + outcomeErrored +) + +// rekeyOne handles a single row inside its own transaction. Returns the outcome. +// Plaintext is held in memory only for the duration of this call and is never +// logged. +func rekeyOne(ctx context.Context, db *database.Connection, id, blob string, zeroKey, realKey []byte) rekeyOutcome { + plaintext, err := credentials.Decrypt(zeroKey, blob) + if err != nil { + // Decrypt with zero key failed — assume already real-key encrypted. + return outcomeSkipped + } + newBlob, err := credentials.Encrypt(realKey, plaintext) + if err != nil { + log.Printf("rekey: encrypt id=%s: %v", id, err) + return outcomeErrored + } + tx, err := db.BeginTx(ctx, pgx.TxOptions{}) + if err != nil { + log.Printf("rekey: begin tx id=%s: %v", id, err) + return outcomeErrored + } + if _, err := tx.Exec(ctx, `UPDATE account_credentials SET encrypted_blob = $1 WHERE id = $2`, newBlob, id); err != nil { + if rErr := tx.Rollback(ctx); rErr != nil { + log.Printf("rekey: rollback id=%s: %v", id, rErr) + } + log.Printf("rekey: update id=%s: %v", id, err) + return outcomeErrored + } + if err := tx.Commit(ctx); err != nil { + log.Printf("rekey: commit id=%s: %v", id, err) + return outcomeErrored + } + return outcomeReKeyed +} + +// isEqual is a small constant-time-like equality check on key bytes. +// Used only for sanity-check that real != zero; not security-sensitive. +func isEqual(a, b []byte) bool { + if len(a) != len(b) { + return false + } + var diff byte + for i := range a { + diff |= a[i] ^ b[i] + } + return diff == 0 +} diff --git a/cmd/rekey/main_test.go b/cmd/rekey/main_test.go new file mode 100644 index 000000000..78723cb94 --- /dev/null +++ b/cmd/rekey/main_test.go @@ -0,0 +1,84 @@ +package main + +import ( + "strings" + "testing" + + "github.com/stretchr/testify/assert" + "github.com/stretchr/testify/require" + + "github.com/LeanerCloud/CUDly/internal/credentials" +) + +// TestRekeyOne_RoundTrip verifies the encrypt/decrypt loop without DB. +// It exercises the helper used by rekeyAccountCredentials per row. We run +// the production credentials.Encrypt/Decrypt directly to confirm: +// - decrypt(zeroKey, zeroBlob) → plaintext (the rekey path triggers) +// - decrypt(zeroKey, realBlob) → error (the skip path triggers) +// - encrypt(realKey, plaintext) round-trips +// +// The transaction wrapper inside rekeyOne is exercised by an integration +// test against testcontainers Postgres in CI; that's heavier than warranted +// for a unit test, so we cover the crypto half here. +func TestRekey_DecryptionRouting(t *testing.T) { + zeroKey := credentials.DevKey() + realKey := decodeHex(t, strings.Repeat("ab", 32)) + + plaintext := []byte(`{"access_key_id":"AKIA","secret_access_key":"abc"}`) + + zeroBlob, err := credentials.Encrypt(zeroKey, plaintext) + require.NoError(t, err) + + // Zero-key decrypt of a zero-encrypted blob → plaintext recovered. + got, err := credentials.Decrypt(zeroKey, zeroBlob) + require.NoError(t, err) + assert.Equal(t, plaintext, got) + + // Re-encrypt with real key, then prove zero-key decrypt fails on the + // new blob (proves the skip-path detection works). + realBlob, err := credentials.Encrypt(realKey, plaintext) + require.NoError(t, err) + _, err = credentials.Decrypt(zeroKey, realBlob) + require.Error(t, err, "zero key must NOT decrypt real-key ciphertext (this is how rekey detects already-real rows)") + + // And the real key still recovers the plaintext from the new blob. + got, err = credentials.Decrypt(realKey, realBlob) + require.NoError(t, err) + assert.Equal(t, plaintext, got) +} + +func TestIsEqual_KeyComparison(t *testing.T) { + a := credentials.DevKey() + b := credentials.DevKey() + assert.True(t, isEqual(a, b)) + + c := decodeHex(t, strings.Repeat("ff", 32)) + assert.False(t, isEqual(a, c)) + assert.False(t, isEqual(a, []byte{0, 0, 0})) // length mismatch +} + +func decodeHex(t *testing.T, s string) []byte { + t.Helper() + b := make([]byte, len(s)/2) + for i := 0; i < len(s); i += 2 { + var hi, lo byte + switch c := s[i]; { + case c >= '0' && c <= '9': + hi = c - '0' + case c >= 'a' && c <= 'f': + hi = c - 'a' + 10 + default: + t.Fatalf("bad hex char %q", c) + } + switch c := s[i+1]; { + case c >= '0' && c <= '9': + lo = c - '0' + case c >= 'a' && c <= 'f': + lo = c - 'a' + 10 + default: + t.Fatalf("bad hex char %q", c) + } + b[i/2] = hi<<4 | lo + } + return b +} diff --git a/cmd/secrets_store.go b/cmd/secrets_store.go new file mode 100644 index 000000000..b03b56e8d --- /dev/null +++ b/cmd/secrets_store.go @@ -0,0 +1,67 @@ +package main + +import ( + "context" + + "github.com/aws/aws-sdk-go-v2/aws" + "github.com/aws/aws-sdk-go-v2/service/secretsmanager" + secretsmgrtypes "github.com/aws/aws-sdk-go-v2/service/secretsmanager/types" +) + +// SecretsStore interface for storing credentials. +type SecretsStore interface { + // ListSecrets returns a list of secret ARNs matching the filter + ListSecrets(ctx context.Context, filter string) ([]string, error) + // UpdateSecret updates a secret with the given ID and value + UpdateSecret(ctx context.Context, secretID string, secretValue string) error +} + +// AWSSecretsStore implements SecretsStore using AWS Secrets Manager. +type AWSSecretsStore struct { + client *secretsmanager.Client +} + +// NewAWSSecretsStore creates a new AWS Secrets Manager store. +func NewAWSSecretsStore(client *secretsmanager.Client) *AWSSecretsStore { + return &AWSSecretsStore{ + client: client, + } +} + +// ListSecrets lists secrets matching the filter (by name). +func (s *AWSSecretsStore) ListSecrets(ctx context.Context, filter string) ([]string, error) { + input := &secretsmanager.ListSecretsInput{ + Filters: []secretsmgrtypes.Filter{ + { + Key: secretsmgrtypes.FilterNameStringTypeName, + Values: []string{filter}, + }, + }, + } + + result, err := s.client.ListSecrets(ctx, input) + if err != nil { + return nil, err + } + + arns := make([]string, 0, len(result.SecretList)) + for _rvc := range result.SecretList { + secret := result.SecretList[_rvc] + if secret.ARN != nil { + arns = append(arns, *secret.ARN) + } + } + + return arns, nil +} + +// UpdateSecret updates a secret with the given value. +func (s *AWSSecretsStore) UpdateSecret(ctx context.Context, secretID, secretValue string) error { + input := &secretsmanager.UpdateSecretInput{ + SecretId: aws.String(secretID), + SecretString: aws.String(secretValue), + } + + _, err := s.client.UpdateSecret(ctx, input) + return err +} diff --git a/cmd/server/main.go b/cmd/server/main.go new file mode 100644 index 000000000..266abd46f --- /dev/null +++ b/cmd/server/main.go @@ -0,0 +1,133 @@ +// Package main provides the unified entry point for CUDly server. +// It supports both AWS Lambda and standard HTTP server modes. +package main + +import ( + "context" + "flag" + "log" + "os" + "strconv" + "time" + + "github.com/LeanerCloud/CUDly/internal/runtime" + "github.com/LeanerCloud/CUDly/internal/server" +) + +// Version, BuildTime, and GitSHA are set at build time via ldflags. +var ( + Version = "dev" + BuildTime = "unknown" + GitSHA = "unknown" +) + +func main() { + // Parse command line flags + mode := flag.String("mode", "auto", "Runtime mode: auto, lambda, http") + port := flag.Int("port", 8080, "HTTP server port (ignored in lambda mode)") + task := flag.String("task", "", "Run a scheduled task and exit (e.g., collect_recommendations, cleanup)") + flag.Parse() + + // Print version info + log.Printf("CUDly Server v%s (git: %s, built: %s)", Version, GitSHA, BuildTime) + + // Export BUILD_TIME and GIT_SHA to the environment so the api package can + // read them without importing main (import cycle). VERSION is passed + // directly to NewApplication to avoid the env round-trip (04-N1). + if err := os.Setenv("BUILD_TIME", BuildTime); err != nil { + log.Printf("failed to export BUILD_TIME: %v", err) + } + if err := os.Setenv("GIT_SHA", GitSHA); err != nil { + log.Printf("failed to export GIT_SHA: %v", err) + } + + ctx := context.Background() + + // Initialize application; pass Version directly so it is stamped on + // ApplicationConfig without going through os.Setenv("VERSION",...). + app, err := server.NewApplication(ctx, Version) + if err != nil { + log.Fatalf("Failed to initialize application: %v", err) + } + defer app.Close() + + // If --task is provided, run the task with a timeout and exit + if *task != "" { + timeout := getTaskTimeout() + taskCtx, cancel := context.WithTimeout(ctx, timeout) + + log.Printf("Running scheduled task: %s (timeout: %v)", *task, timeout) + taskType := server.ScheduledTaskType(*task) + result, err := app.HandleScheduledTask(taskCtx, taskType, server.ScheduledTaskParams{}) + cancel() + if err != nil { + log.Fatalf("Scheduled task %q failed: %v", *task, err) //nolint:gocritic // exitAfterDefer: intentional fatal; app.Close() not needed on task failure + } + log.Printf("Scheduled task %q completed successfully: %v", *task, result) + return + } + + // Determine runtime mode + runtimeMode := determineRuntimeMode(*mode) + log.Printf("Starting CUDly server in %s mode", runtimeMode) + + // Start appropriate server + switch runtimeMode { + case "lambda": + server.StartLambdaHandler(app) + case "http": + if err := server.StartHTTPServer(app, *port); err != nil { + log.Fatalf("HTTP server failed: %v", err) + } + default: + log.Fatalf("Unknown runtime mode: %s", runtimeMode) + } +} + +// getTaskTimeout returns the task timeout from TASK_TIMEOUT env var or the default of 15 minutes. +// Logs a warning when TASK_TIMEOUT is set but cannot be parsed or is non-positive, +// so the operator knows the value was not applied. +func getTaskTimeout() time.Duration { + const defaultTimeout = 15 * time.Minute + if v := os.Getenv("TASK_TIMEOUT"); v != "" { + secs, err := strconv.Atoi(v) + if err != nil { + log.Printf("WARNING: TASK_TIMEOUT=%q is not a valid integer; using default %v", v, defaultTimeout) // #nosec G706 -- TASK_TIMEOUT env var is operator-controlled; logged for diagnostics + return defaultTimeout + } + if secs <= 0 { + log.Printf("WARNING: TASK_TIMEOUT=%q must be a positive number; using default %v", v, defaultTimeout) // #nosec G706 -- TASK_TIMEOUT env var is operator-controlled; logged for diagnostics + return defaultTimeout + } + return time.Duration(secs) * time.Second + } + return defaultTimeout +} + +// determineRuntimeMode determines the runtime mode based on flags and environment. +func determineRuntimeMode(modeFlag string) string { + // If mode is explicitly set, use it + if modeFlag != "auto" { + return modeFlag + } + + // Auto-detect based on environment using the canonical runtime helper, + // which encapsulates the detection rule so future changes stay consistent + // across all call sites (issue 04-M5). + if runtime.IsLambda() { + return "lambda" + } + + // Check for explicit RUNTIME_MODE environment variable + if runtimeMode := os.Getenv("RUNTIME_MODE"); runtimeMode != "" { + switch runtimeMode { + case "lambda", "http": + return runtimeMode + default: + log.Printf("Warning: unrecognized RUNTIME_MODE %q, falling back to auto-detection", runtimeMode) // #nosec G706 -- RUNTIME_MODE env var is operator-controlled; logged for diagnostics + } + } + + // Default to HTTP mode for containers (Fargate, Cloud Run, Container Apps) + return "http" +} diff --git a/cmd/server/main_test.go b/cmd/server/main_test.go new file mode 100644 index 000000000..29938eae4 --- /dev/null +++ b/cmd/server/main_test.go @@ -0,0 +1,70 @@ +package main + +import ( + "os" + "testing" + "time" +) + +func TestGetTaskTimeout(t *testing.T) { + tests := []struct { + name string + envValue string + expected time.Duration + }{ + { + name: "default when env not set", + envValue: "", + expected: 15 * time.Minute, + }, + { + name: "valid env value", + envValue: "60", + expected: 60 * time.Second, + }, + { + name: "invalid env value falls back to default", + envValue: "not-a-number", + expected: 15 * time.Minute, + }, + { + name: "zero falls back to default", + envValue: "0", + expected: 15 * time.Minute, + }, + { + name: "negative falls back to default", + envValue: "-10", + expected: 15 * time.Minute, + }, + { + name: "large value", + envValue: "3600", + expected: 3600 * time.Second, + }, + } + + for _, tt := range tests { + t.Run(tt.name, func(t *testing.T) { + old := os.Getenv("TASK_TIMEOUT") + defer func() { + if old == "" { + os.Unsetenv("TASK_TIMEOUT") + } else { + os.Setenv("TASK_TIMEOUT", old) + } + }() + + if tt.envValue != "" { + os.Setenv("TASK_TIMEOUT", tt.envValue) + } else { + os.Unsetenv("TASK_TIMEOUT") + } + + result := getTaskTimeout() + if result != tt.expected { + t.Errorf("getTaskTimeout() = %v, want %v", result, tt.expected) + } + }) + } +} diff --git a/cmd/validators.go b/cmd/validators.go new file mode 100644 index 000000000..36b22ec1f --- /dev/null +++ b/cmd/validators.go @@ -0,0 +1,298 @@ +package main + +import ( + "fmt" + "log" + "os" + "path/filepath" + "strings" + + "github.com/LeanerCloud/CUDly/pkg/common" + "github.com/spf13/cobra" +) + +// validateFlags performs validation on command line flags before execution. +func validateFlags(cmd *cobra.Command, args []string) error { + if err := validateNumericRanges(cmd); err != nil { + return err + } + + if err := validatePaymentAndTerm(); err != nil { + return err + } + + if err := validateFilePaths(); err != nil { + return err + } + + if err := validateFilterFlags(); err != nil { + return err + } + + if err := validateCSVModeFilterFlags(); err != nil { + return err + } + + if err := validateRecLookbackPeriod(); err != nil { + return err + } + + return nil +} + +// validateRecLookbackPeriod validates the --rec-lookback-period flag. +func validateRecLookbackPeriod() error { + valid := map[string]bool{"7d": true, "30d": true, "60d": true} + if !valid[toolCfg.RecLookbackPeriod] { + return fmt.Errorf("invalid rec-lookback-period %q: must be one of 7d, 30d, 60d", toolCfg.RecLookbackPeriod) + } + return nil +} + +// validateNumericRanges validates all numeric configuration values. +// cmd is used to detect explicitly-set flags via cobra's Changed(); pass +// nil if no flag-source-detection is required (e.g. unit-test paths that +// only need the numeric bounds checks). +func validateNumericRanges(cmd *cobra.Command) error { + // Validate coverage percentage + if toolCfg.Coverage < 0 || toolCfg.Coverage > 100 { + return fmt.Errorf("coverage percentage must be between 0 and 100, got: %.2f", toolCfg.Coverage) + } + + if err := validateTargetCoverage(cmd); err != nil { + return err + } + + // Validate coverage lookback days + if toolCfg.CoverageLookbackDays < 1 { + return fmt.Errorf("coverage-lookback-days must be >= 1, got: %d", toolCfg.CoverageLookbackDays) + } + + // Validate max instances + if toolCfg.MaxInstances < 0 { + return fmt.Errorf("max-instances must be 0 (no limit) or a positive number, got: %d", toolCfg.MaxInstances) + } + + if toolCfg.MaxInstances > MaxReasonableInstances { + return fmt.Errorf("max-instances (%d) exceeds reasonable limit of %d", toolCfg.MaxInstances, MaxReasonableInstances) + } + + // Validate override count + if toolCfg.OverrideCount < 0 { + return fmt.Errorf("override-count must be 0 (disabled) or a positive number, got: %d", toolCfg.OverrideCount) + } + + if toolCfg.OverrideCount > MaxReasonableInstances { + return fmt.Errorf("override-count (%d) exceeds reasonable limit of %d", toolCfg.OverrideCount, MaxReasonableInstances) + } + + if err := validateMinPoolSize(); err != nil { + return err + } + + return nil +} + +// validateMinPoolSize checks that --min-pool-size is a non-negative whole number. +// Instance counts are integers, so a fractional threshold is nonsensical. +func validateMinPoolSize() error { + if toolCfg.MinPoolSize < 0 { + return fmt.Errorf("min-pool-size must be 0 (disabled) or a positive whole number, got: %.2f", toolCfg.MinPoolSize) + } + if toolCfg.MinPoolSize > 0 && toolCfg.MinPoolSize != float64(int64(toolCfg.MinPoolSize)) { + return fmt.Errorf("min-pool-size must be a whole number (instance counts are integers), got: %.2f", toolCfg.MinPoolSize) + } + return nil +} + +// validateTargetCoverage validates the --target-coverage range and +// emits an info-log when it is set alongside an explicitly-overridden +// --coverage (target wins). Split out of validateNumericRanges to keep +// the parent under gocyclo's complexity threshold. +func validateTargetCoverage(cmd *cobra.Command) error { + if toolCfg.TargetCoverage < 0 || toolCfg.TargetCoverage > 100 { + return fmt.Errorf("target-coverage percentage must be between 0 and 100, got: %.2f", toolCfg.TargetCoverage) + } + + // Info-log when both flags are explicitly set — target-coverage wins. + // Detect "user explicitly set --coverage" via cobra's Changed() rather than + // comparing to the default value, so a user who happens to set --coverage 80 + // (which equals the default) still sees the notice. + if toolCfg.TargetCoverage > 0 && cmd != nil && cmd.Flags().Changed("coverage") { + log.Printf("--target-coverage=%.1f set; --coverage=%.1f is being ignored (target-coverage sizing supersedes coverage sizing)", + toolCfg.TargetCoverage, toolCfg.Coverage) + } + return nil +} + +// validatePaymentAndTerm validates payment options and term configuration. +func validatePaymentAndTerm() error { + // Validate payment option + validPaymentOptions := map[string]bool{ + "all-upfront": true, + "partial-upfront": true, + "no-upfront": true, + } + if !validPaymentOptions[toolCfg.PaymentOption] { + return fmt.Errorf("invalid payment option: %s. Must be one of: all-upfront, partial-upfront, no-upfront", toolCfg.PaymentOption) + } + + // Validate term years + if toolCfg.TermYears != 1 && toolCfg.TermYears != 3 { + return fmt.Errorf("invalid term: %d years. Must be 1 or 3", toolCfg.TermYears) + } + + // Warn about RDS 3-year no-upfront limitation + return warnRDS3YearNoUpfront() +} + +// warnRDS3YearNoUpfront warns if RDS service is selected with 3-year no-upfront. +func warnRDS3YearNoUpfront() error { + // In CSV-input mode the payment option comes from each row, not the + // --payment flag (which keeps its no-upfront default), so this flag-based + // check would fire a false alarm on a CSV full of partial-upfront rows. + if toolCfg.CSVInput != "" { + return nil + } + + if toolCfg.PaymentOption != "no-upfront" || toolCfg.TermYears != 3 { + return nil + } + + services := determineServicesToProcess(toolCfg) + hasRDS := toolCfg.AllServices || containsService(services, common.ServiceRDS) + + if hasRDS { + log.Println("⚠️ WARNING: AWS does not offer 3-year no-upfront Reserved Instances for RDS.") + log.Println(" RDS 3-year RIs only support: all-upfront, partial-upfront") + log.Println(" No RDS recommendations will be found with this combination.") + } + + return nil +} + +// validateCSVModeFilterFlags refuses the scorer thresholds that --input-csv +// cannot enforce, instead of accepting them and buying as if they were unset. +// +// A recommendations CSV carries neither a savings-percentage column nor a +// break-even column (writeMultiServiceCSVReport emits neither and +// parseCSVRecord reads neither), so both fields load as zero on every row. +// Gating on a zero field would reject every row of every file, and skipping +// the gate silently drops a spend guard the operator believes is in force. +// Refusing the combination is the only honest option until #1819 teaches the +// format to carry the columns. +// +// Both flags default to 0, so this can only fire when one was set explicitly. +func validateCSVModeFilterFlags() error { + if toolCfg.CSVInput == "" { + return nil + } + + if toolCfg.MinSavingsPct > 0 { + return fmt.Errorf("--min-savings-pct cannot be applied to --input-csv runs: a recommendations CSV carries no savings-percentage column, so the threshold would be silently ignored (see #1819). Remove --min-savings-pct, or filter the CSV before passing it in") + } + + if toolCfg.MaxBreakEvenMonths > 0 { + return fmt.Errorf("--max-break-even-months cannot be applied to --input-csv runs: a recommendations CSV carries no break-even column, so the threshold would be silently ignored (see #1819). Remove --max-break-even-months, or filter the CSV before passing it in") + } + + return nil +} + +// containsService checks if a service exists in the slice. +func containsService(services []common.ServiceType, service common.ServiceType) bool { + for _, svc := range services { + if svc == service { + return true + } + } + return false +} + +// validateFilePaths validates CSV input/output paths. +func validateFilePaths() error { + // Validate CSV output path if provided + if toolCfg.CSVOutput != "" { + dir := filepath.Dir(toolCfg.CSVOutput) + if dir != "." && dir != "" { + if _, err := os.Stat(dir); os.IsNotExist(err) { + return fmt.Errorf("output directory does not exist: %s", dir) + } + } + } + + // Validate CSV input path if provided + if toolCfg.CSVInput != "" { + if _, err := os.Stat(toolCfg.CSVInput); os.IsNotExist(err) { + return fmt.Errorf("input CSV file does not exist: %s", toolCfg.CSVInput) + } + if !strings.HasSuffix(strings.ToLower(toolCfg.CSVInput), ".csv") { + return fmt.Errorf("input file must have .csv extension: %s", toolCfg.CSVInput) + } + } + + return nil +} + +// validateFilterFlags validates filter configuration flags. +func validateFilterFlags() error { + // Check for region conflicts + if err := validateNoConflicts(toolCfg.IncludeRegions, toolCfg.ExcludeRegions, "region"); err != nil { + return err + } + + // Check for instance type conflicts + if err := validateNoConflicts(toolCfg.IncludeInstanceTypes, toolCfg.ExcludeInstanceTypes, "instance type"); err != nil { + return err + } + + // Check for engine conflicts + if err := validateNoConflicts(toolCfg.IncludeEngines, toolCfg.ExcludeEngines, "engine"); err != nil { + return err + } + + // Validate instance types format + if err := validateInstanceTypes(toolCfg.IncludeInstanceTypes); err != nil { + return fmt.Errorf("invalid include-instance-types: %w", err) + } + if err := validateInstanceTypes(toolCfg.ExcludeInstanceTypes); err != nil { + return fmt.Errorf("invalid exclude-instance-types: %w", err) + } + + return nil +} + +// validateNoConflicts checks that include and exclude lists don't overlap. +func validateNoConflicts(include, exclude []string, itemType string) error { + if len(include) == 0 || len(exclude) == 0 { + return nil + } + + for _, inc := range include { + for _, exc := range exclude { + if inc == exc { + return fmt.Errorf("%s '%s' cannot be both included and excluded", itemType, inc) + } + } + } + + return nil +} + +// validateInstanceTypes performs basic validation on instance type names. +func validateInstanceTypes(instanceTypes []string) error { + if len(instanceTypes) == 0 { + return nil + } + + for _, t := range instanceTypes { + if t == "" { + return fmt.Errorf("empty instance type") + } + if !strings.Contains(t, ".") { + return fmt.Errorf("invalid instance type format '%s': expected format like 'db.t3.micro'", t) + } + } + + return nil +} diff --git a/cmd/validators_test.go b/cmd/validators_test.go new file mode 100644 index 000000000..201e7bbd1 --- /dev/null +++ b/cmd/validators_test.go @@ -0,0 +1,688 @@ +package main + +import ( + "os" + "path/filepath" + "strings" + "testing" + + "github.com/LeanerCloud/CUDly/pkg/common" + "github.com/spf13/cobra" +) + +func TestValidateNumericRanges(t *testing.T) { + tests := []struct { + setupFunc func() + name string + errMsg string + wantErr bool + }{ + { + name: "valid coverage percentage", + setupFunc: func() { + toolCfg.Coverage = 80.0 + toolCfg.MaxInstances = 100 + toolCfg.OverrideCount = 10 + toolCfg.CoverageLookbackDays = 30 + }, + wantErr: false, + }, + { + name: "coverage below zero", + setupFunc: func() { + toolCfg.Coverage = -1.0 + toolCfg.CoverageLookbackDays = 30 + }, + wantErr: true, + errMsg: "coverage percentage must be between 0 and 100", + }, + { + name: "coverage above 100", + setupFunc: func() { + toolCfg.Coverage = 101.0 + toolCfg.CoverageLookbackDays = 30 + }, + wantErr: true, + errMsg: "coverage percentage must be between 0 and 100", + }, + { + name: "negative max instances", + setupFunc: func() { + toolCfg.Coverage = 80.0 + toolCfg.MaxInstances = -1 + toolCfg.CoverageLookbackDays = 30 + }, + wantErr: true, + errMsg: "max-instances must be 0", + }, + { + name: "max instances exceeds limit", + setupFunc: func() { + toolCfg.Coverage = 80.0 + toolCfg.MaxInstances = MaxReasonableInstances + 1 + toolCfg.CoverageLookbackDays = 30 + }, + wantErr: true, + errMsg: "max-instances", + }, + { + name: "negative override count", + setupFunc: func() { + toolCfg.Coverage = 80.0 + toolCfg.MaxInstances = 100 + toolCfg.OverrideCount = -1 + toolCfg.CoverageLookbackDays = 30 + }, + wantErr: true, + errMsg: "override-count must be 0", + }, + { + name: "override count exceeds limit", + setupFunc: func() { + toolCfg.Coverage = 80.0 + toolCfg.MaxInstances = 100 + toolCfg.OverrideCount = MaxReasonableInstances + 1 + toolCfg.CoverageLookbackDays = 30 + }, + wantErr: true, + errMsg: "override-count", + }, + { + name: "min-pool-size negative", + setupFunc: func() { + toolCfg.Coverage = 80.0 + toolCfg.MaxInstances = 0 + toolCfg.OverrideCount = 0 + toolCfg.CoverageLookbackDays = 30 + toolCfg.MinPoolSize = -1.0 + }, + wantErr: true, + errMsg: "min-pool-size must be 0 (disabled) or a positive whole number", + }, + { + name: "min-pool-size fractional rejected", + setupFunc: func() { + toolCfg.Coverage = 80.0 + toolCfg.MaxInstances = 0 + toolCfg.OverrideCount = 0 + toolCfg.CoverageLookbackDays = 30 + toolCfg.MinPoolSize = 1.5 + }, + wantErr: true, + errMsg: "min-pool-size must be a whole number", + }, + { + name: "min-pool-size disabled (zero)", + setupFunc: func() { + toolCfg.Coverage = 80.0 + toolCfg.MaxInstances = 0 + toolCfg.OverrideCount = 0 + toolCfg.CoverageLookbackDays = 30 + toolCfg.MinPoolSize = 0 + }, + wantErr: false, + }, + { + name: "min-pool-size whole number accepted", + setupFunc: func() { + toolCfg.Coverage = 80.0 + toolCfg.MaxInstances = 0 + toolCfg.OverrideCount = 0 + toolCfg.CoverageLookbackDays = 30 + toolCfg.MinPoolSize = 2.0 + }, + wantErr: false, + }, + } + + for _, tt := range tests { + t.Run(tt.name, func(t *testing.T) { + // Setup + toolCfg = Config{} + tt.setupFunc() + + // Execute (nil cmd — these tests only exercise the numeric + // bounds checks; the flag-source-detection branch is covered + // by TestValidateTargetCoverage below). + err := validateNumericRanges(nil) + + // Verify + if tt.wantErr { + if err == nil { + t.Errorf("validateNumericRanges() expected error containing %q, got nil", tt.errMsg) + } else if tt.errMsg != "" && !strings.Contains(err.Error(), tt.errMsg) { + t.Errorf("validateNumericRanges() error = %v, want error containing %q", err, tt.errMsg) + } + } else { + if err != nil { + t.Errorf("validateNumericRanges() unexpected error = %v", err) + } + } + }) + } +} + +func TestValidatePaymentAndTerm(t *testing.T) { + tests := []struct { + setupFunc func() + name string + errMsg string + wantErr bool + }{ + { + name: "valid payment option - no-upfront", + setupFunc: func() { + toolCfg.PaymentOption = "no-upfront" + toolCfg.TermYears = 3 + toolCfg.Services = []string{"elasticache"} + }, + wantErr: false, + }, + { + name: "valid payment option - all-upfront", + setupFunc: func() { + toolCfg.PaymentOption = "all-upfront" + toolCfg.TermYears = 1 + }, + wantErr: false, + }, + { + name: "valid payment option - partial-upfront", + setupFunc: func() { + toolCfg.PaymentOption = "partial-upfront" + toolCfg.TermYears = 3 + }, + wantErr: false, + }, + { + name: "invalid payment option", + setupFunc: func() { + toolCfg.PaymentOption = "invalid-option" + toolCfg.TermYears = 3 + }, + wantErr: true, + errMsg: "invalid payment option", + }, + { + name: "empty payment option", + setupFunc: func() { + toolCfg.PaymentOption = "" + toolCfg.TermYears = 1 + }, + wantErr: true, + errMsg: "invalid payment option", + }, + { + name: "invalid term - 2 years", + setupFunc: func() { + toolCfg.PaymentOption = "no-upfront" + toolCfg.TermYears = 2 + }, + wantErr: true, + errMsg: "invalid term", + }, + { + name: "invalid term - 0 years", + setupFunc: func() { + toolCfg.PaymentOption = "no-upfront" + toolCfg.TermYears = 0 + }, + wantErr: true, + errMsg: "invalid term", + }, + } + + for _, tt := range tests { + t.Run(tt.name, func(t *testing.T) { + // Setup + toolCfg = Config{} + tt.setupFunc() + + // Execute + err := validatePaymentAndTerm() + + // Verify + if tt.wantErr { + if err == nil { + t.Errorf("validatePaymentAndTerm() expected error containing %q, got nil", tt.errMsg) + } else if tt.errMsg != "" && !strings.Contains(err.Error(), tt.errMsg) { + t.Errorf("validatePaymentAndTerm() error = %v, want error containing %q", err, tt.errMsg) + } + } else { + if err != nil { + t.Errorf("validatePaymentAndTerm() unexpected error = %v", err) + } + } + }) + } +} + +func TestContainsService(t *testing.T) { + tests := []struct { + name string + service common.ServiceType + services []common.ServiceType + want bool + }{ + { + name: "service found in list", + services: []common.ServiceType{common.ServiceRDS, common.ServiceEC2, common.ServiceElastiCache}, + service: common.ServiceRDS, + want: true, + }, + { + name: "service not found in list", + services: []common.ServiceType{common.ServiceRDS, common.ServiceEC2}, + service: common.ServiceElastiCache, + want: false, + }, + { + name: "empty list", + services: []common.ServiceType{}, + service: common.ServiceRDS, + want: false, + }, + { + name: "single service list - match", + services: []common.ServiceType{common.ServiceRDS}, + service: common.ServiceRDS, + want: true, + }, + } + + for _, tt := range tests { + t.Run(tt.name, func(t *testing.T) { + got := containsService(tt.services, tt.service) + if got != tt.want { + t.Errorf("containsService() = %v, want %v", got, tt.want) + } + }) + } +} + +func TestValidateFilePaths(t *testing.T) { + // Create a temporary directory for testing + tmpDir := t.TempDir() + + tests := []struct { + setupFunc func() func() + name string + errMsg string + wantErr bool + }{ + { + name: "valid CSV output path", + setupFunc: func() func() { + toolCfg.CSVOutput = filepath.Join(tmpDir, "output.csv") + toolCfg.CSVInput = "" + return func() {} + }, + wantErr: false, + }, + { + name: "valid CSV input path", + setupFunc: func() func() { + // Create a test CSV file + inputPath := filepath.Join(tmpDir, "input.csv") + if err := os.WriteFile(inputPath, []byte("test"), 0644); err != nil { + t.Fatalf("Failed to create test file: %v", err) + } + toolCfg.CSVInput = inputPath + toolCfg.CSVOutput = "" + return func() { + os.Remove(inputPath) + } + }, + wantErr: false, + }, + { + name: "CSV output directory does not exist", + setupFunc: func() func() { + toolCfg.CSVOutput = "/nonexistent/directory/output.csv" + toolCfg.CSVInput = "" + return func() {} + }, + wantErr: true, + errMsg: "output directory does not exist", + }, + { + name: "CSV input file does not exist", + setupFunc: func() func() { + toolCfg.CSVInput = filepath.Join(tmpDir, "nonexistent.csv") + toolCfg.CSVOutput = "" + return func() {} + }, + wantErr: true, + errMsg: "input CSV file does not exist", + }, + { + name: "CSV input file wrong extension", + setupFunc: func() func() { + // Create a test file with wrong extension + inputPath := filepath.Join(tmpDir, "input.txt") + if err := os.WriteFile(inputPath, []byte("test"), 0644); err != nil { + t.Fatalf("Failed to create test file: %v", err) + } + toolCfg.CSVInput = inputPath + toolCfg.CSVOutput = "" + return func() { + os.Remove(inputPath) + } + }, + wantErr: true, + errMsg: "input file must have .csv extension", + }, + } + + for _, tt := range tests { + t.Run(tt.name, func(t *testing.T) { + // Setup + toolCfg = Config{} + cleanup := tt.setupFunc() + defer cleanup() + + // Execute + err := validateFilePaths() + + // Verify + if tt.wantErr { + if err == nil { + t.Errorf("validateFilePaths() expected error containing %q, got nil", tt.errMsg) + } else if tt.errMsg != "" && !strings.Contains(err.Error(), tt.errMsg) { + t.Errorf("validateFilePaths() error = %v, want error containing %q", err, tt.errMsg) + } + } else { + if err != nil { + t.Errorf("validateFilePaths() unexpected error = %v", err) + } + } + }) + } +} + +func TestValidateNoConflicts(t *testing.T) { + tests := []struct { + name string + itemType string + errMsg string + include []string + exclude []string + wantErr bool + }{ + { + name: "no conflicts", + include: []string{"us-east-1", "us-west-2"}, + exclude: []string{"eu-west-1", "ap-south-1"}, + itemType: "region", + wantErr: false, + }, + { + name: "conflict found", + include: []string{"us-east-1", "us-west-2"}, + exclude: []string{"us-west-2", "eu-west-1"}, + itemType: "region", + wantErr: true, + errMsg: "region 'us-west-2' cannot be both included and excluded", + }, + { + name: "empty include list", + include: []string{}, + exclude: []string{"us-west-2"}, + itemType: "region", + wantErr: false, + }, + { + name: "empty exclude list", + include: []string{"us-east-1"}, + exclude: []string{}, + itemType: "region", + wantErr: false, + }, + { + name: "both lists empty", + include: []string{}, + exclude: []string{}, + itemType: "region", + wantErr: false, + }, + } + + for _, tt := range tests { + t.Run(tt.name, func(t *testing.T) { + err := validateNoConflicts(tt.include, tt.exclude, tt.itemType) + + if tt.wantErr { + if err == nil { + t.Errorf("validateNoConflicts() expected error containing %q, got nil", tt.errMsg) + } else if tt.errMsg != "" && !strings.Contains(err.Error(), tt.errMsg) { + t.Errorf("validateNoConflicts() error = %v, want error containing %v", err, tt.errMsg) + } + } else { + if err != nil { + t.Errorf("validateNoConflicts() unexpected error = %v", err) + } + } + }) + } +} + +// Note: TestValidateInstanceTypes and TestValidateFlags already exist in main_test.go + +// TestValidateTargetCoverage covers the --target-coverage range check +// and the "both flags explicitly set" info-log gate. Range cases pass nil +// cmd (the explicit-flag detection is irrelevant for them); the "both set" +// case constructs a real cobra command and marks --coverage as explicitly +// set so the Changed("coverage") branch actually fires. The log line +// itself isn't asserted (log.Printf goes to stderr and capturing it from +// this package is more friction than value). +func TestValidateTargetCoverage(t *testing.T) { + tests := []struct { + name string + errSubstr string + target float64 + coverage float64 + wantErr bool + useCobraCmd bool + }{ + {name: "disabled (zero) is valid", target: 0, coverage: 80, wantErr: false}, + {name: "min boundary valid", target: 0.0001, coverage: 80, wantErr: false}, + {name: "max boundary valid", target: 100, coverage: 80, wantErr: false}, + {name: "mid-range valid", target: 95, coverage: 80, wantErr: false}, + { + name: "negative target rejected", target: -0.5, coverage: 80, wantErr: true, + errSubstr: "target-coverage percentage must be between 0 and 100", + }, + { + name: "above 100 rejected", target: 100.01, coverage: 80, wantErr: true, + errSubstr: "target-coverage percentage must be between 0 and 100", + }, + { + // Both flags explicitly set — the precedence info-log fires; + // validation still passes. useCobraCmd builds a real command so + // Changed("coverage") returns true; without this the branch is + // effectively dead code in the test. + name: "target + coverage both set, both valid", + target: 95, + coverage: 50, + wantErr: false, + useCobraCmd: true, + }, + } + + for _, tt := range tests { + t.Run(tt.name, func(t *testing.T) { + toolCfg = Config{Coverage: tt.coverage, TargetCoverage: tt.target, CoverageLookbackDays: 30} + var cmd *cobra.Command + if tt.useCobraCmd { + cmd = &cobra.Command{Use: "test"} + // Default doesn't matter — the Set call below marks the flag + // as Changed, which is what validateTargetCoverage checks. + cmd.Flags().Float64("coverage", 80, "") + if err := cmd.Flags().Set("coverage", "50"); err != nil { + t.Fatalf("failed to mark coverage flag as Changed: %v", err) + } + } + err := validateNumericRanges(cmd) + if tt.wantErr { + if err == nil { + t.Errorf("validateNumericRanges() expected error containing %q, got nil", tt.errSubstr) + } else if tt.errSubstr != "" && !strings.Contains(err.Error(), tt.errSubstr) { + t.Errorf("validateNumericRanges() error = %v, want substring %q", err, tt.errSubstr) + } + } else if err != nil { + t.Errorf("validateNumericRanges() unexpected error = %v", err) + } + }) + } +} + +// TestValidateCoverageLookbackDays verifies that validateNumericRanges rejects +// non-positive --coverage-lookback-days and accepts positive values. +func TestValidateCoverageLookbackDays(t *testing.T) { + tests := []struct { + name string + errSubstr string + days int + wantErr bool + }{ + {name: "default 30 is valid", days: 30, wantErr: false}, + {name: "1 day is valid", days: 1, wantErr: false}, + {name: "90 days is valid", days: 90, wantErr: false}, + { + name: "zero rejected", + days: 0, + wantErr: true, + errSubstr: "coverage-lookback-days must be >= 1", + }, + { + name: "negative rejected", + days: -1, + wantErr: true, + errSubstr: "coverage-lookback-days must be >= 1", + }, + } + + for _, tt := range tests { + t.Run(tt.name, func(t *testing.T) { + origCfg := toolCfg + defer func() { toolCfg = origCfg }() + toolCfg = Config{ + Coverage: 80, + CoverageLookbackDays: tt.days, + } + err := validateNumericRanges(nil) + if tt.wantErr { + if err == nil { + t.Errorf("validateNumericRanges() expected error containing %q, got nil", tt.errSubstr) + } else if tt.errSubstr != "" && !strings.Contains(err.Error(), tt.errSubstr) { + t.Errorf("validateNumericRanges() error = %v, want substring %q", err, tt.errSubstr) + } + } else if err != nil { + t.Errorf("validateNumericRanges() unexpected error = %v", err) + } + }) + } +} + +// TestValidateRecLookbackPeriod verifies that validateRecLookbackPeriod accepts +// the three valid values and rejects anything else, including empty string. +func TestValidateRecLookbackPeriod(t *testing.T) { + tests := []struct { + name string + period string + wantErr bool + errSubstr string + }{ + {name: "7d valid", period: "7d", wantErr: false}, + {name: "30d valid", period: "30d", wantErr: false}, + {name: "60d valid", period: "60d", wantErr: false}, + {name: "empty rejected", period: "", wantErr: true, errSubstr: "invalid rec-lookback-period"}, + {name: "14d rejected", period: "14d", wantErr: true, errSubstr: "invalid rec-lookback-period"}, + {name: "90d rejected", period: "90d", wantErr: true, errSubstr: "invalid rec-lookback-period"}, + {name: "SEVEN_DAYS rejected", period: "SEVEN_DAYS", wantErr: true, errSubstr: "invalid rec-lookback-period"}, + } + for _, tt := range tests { + t.Run(tt.name, func(t *testing.T) { + origCfg := toolCfg + defer func() { toolCfg = origCfg }() + toolCfg.RecLookbackPeriod = tt.period + err := validateRecLookbackPeriod() + if tt.wantErr { + if err == nil { + t.Errorf("validateRecLookbackPeriod() expected error containing %q, got nil", tt.errSubstr) + } else if tt.errSubstr != "" && !strings.Contains(err.Error(), tt.errSubstr) { + t.Errorf("validateRecLookbackPeriod() error = %v, want substring %q", err, tt.errSubstr) + } + } else if err != nil { + t.Errorf("validateRecLookbackPeriod() unexpected error = %v", err) + } + }) + } +} + +// TestValidateCSVModeFilterFlags pins that --input-csv refuses the two scorer +// thresholds it cannot enforce, instead of accepting them and buying as if +// they were never set. +// +// A recommendations CSV carries neither a savings-percentage column nor a +// break-even column, so both fields load as zero on every row and any gate on +// them is either vacuous or rejects the whole file. #1741's complaint is +// exactly that silent partial enforcement of a spend guard is worse than not +// offering the guard. +func TestValidateCSVModeFilterFlags(t *testing.T) { + tests := []struct { + name string + csvInput string + errSubstr string + minSavingsPct float64 + maxBreakEvenMonths int + }{ + { + name: "min-savings-pct refused in CSV mode", + csvInput: "recs.csv", + minSavingsPct: 20, + errSubstr: "--min-savings-pct cannot be applied to --input-csv", + }, + { + name: "max-break-even-months refused in CSV mode", + csvInput: "recs.csv", + maxBreakEvenMonths: 12, + errSubstr: "--max-break-even-months cannot be applied to --input-csv", + }, + { + // Both flags default to 0, so an ordinary CSV run is unaffected. + name: "CSV mode without the thresholds is accepted", + csvInput: "recs.csv", + }, + { + // The thresholds are enforced normally off the CSV path. + name: "thresholds accepted without --input-csv", + minSavingsPct: 20, + maxBreakEvenMonths: 12, + }, + } + + for _, tt := range tests { + t.Run(tt.name, func(t *testing.T) { + origCfg := toolCfg + defer func() { toolCfg = origCfg }() + toolCfg.CSVInput = tt.csvInput + toolCfg.MinSavingsPct = tt.minSavingsPct + toolCfg.MaxBreakEvenMonths = tt.maxBreakEvenMonths + + err := validateCSVModeFilterFlags() + if tt.errSubstr == "" { + if err != nil { + t.Errorf("validateCSVModeFilterFlags() unexpected error = %v", err) + } + return + } + if err == nil { + t.Fatalf("validateCSVModeFilterFlags() expected error containing %q, got nil", tt.errSubstr) + } + if !strings.Contains(err.Error(), tt.errSubstr) { + t.Errorf("validateCSVModeFilterFlags() error = %v, want substring %q", err, tt.errSubstr) + } + }) + } +} diff --git a/docker-compose.test.yml b/docker-compose.test.yml new file mode 100644 index 000000000..a63a7d89e --- /dev/null +++ b/docker-compose.test.yml @@ -0,0 +1,130 @@ +# Docker Compose for Testing CUDly +# Use this for local E2E testing and integration tests + +services: + # PostgreSQL Database + postgres: + # Digest-pinned for the same reason as the ci.yml service container and the + # linter images: a floating tag lets the database change under an unchanged + # repo. This file is what ci.yml's e2e-tests job runs, so leaving it float + # would have pinned the integration-tests path and left e2e open. + image: postgres:16-alpine@sha256:cf78e76683b9ca8c5733cbbdce6c9262b45b6767934dd0a95e671f9a0fc20685 + container_name: cudly-test-postgres + environment: + POSTGRES_DB: cudly_test + POSTGRES_USER: cudly_test + POSTGRES_PASSWORD: test_password + ports: + - "5432:5432" + volumes: + - postgres_test_data:/var/lib/postgresql/data + healthcheck: + test: ["CMD-SHELL", "pg_isready -U cudly_test"] + interval: 5s + timeout: 3s + retries: 5 + networks: + - cudly-test + + # CUDly Application (HTTP mode) + cudly-app: + build: + context: . + dockerfile: Dockerfile + container_name: cudly-test-app + depends_on: + postgres: + condition: service_healthy + environment: + # Runtime configuration + RUNTIME_MODE: http + PORT: 8080 + ENVIRONMENT: test + VERSION: test + + # Database configuration + DB_HOST: postgres + DB_PORT: 5432 + DB_NAME: cudly_test + DB_USER: cudly_test + DB_PASSWORD: test_password + DB_SSL_MODE: disable + DB_AUTO_MIGRATE: "true" + DB_MIGRATIONS_PATH: /app/migrations + + # Secret provider (use env for testing). With the env provider, + # *_SECRET variables name the environment variable holding the value. + SECRET_PROVIDER: env + ADMIN_EMAIL: admin@test.example + ADMIN_PASSWORD_SECRET: TEST_ADMIN_PASSWORD + TEST_ADMIN_PASSWORD: e2e-local-test-only-password + # Allow the all-zero dev encryption key; this stack never holds real + # cloud credentials. + CREDENTIAL_ENCRYPTION_ALLOW_DEV_KEY: "1" + + # Scheduled-task auth: disabled for the local/CI compose stack (no + # cloud OIDC issuer or bearer secret available) + SCHEDULED_TASK_AUTH_MODE: disabled + + # DynamoDB tables (for backward compatibility testing) + CONFIG_TABLE: test-config + PLANS_TABLE: test-plans + HISTORY_TABLE: test-history + USERS_TABLE: test-users + GROUPS_TABLE: test-groups + SESSIONS_TABLE: test-sessions + + # Email configuration: disabled, the app falls back to a no-op sender + # (SECRET_PROVIDER=env is not a supported email backend) + EMAIL_ENABLED: "false" + EMAIL_ADDRESS: test@example.com + + # Feature flags + ENABLE_DASHBOARD: "true" + + # CORS + CORS_ALLOWED_ORIGIN: http://localhost:3000 + + ports: + - "8080:8080" + networks: + - cudly-test + healthcheck: + test: ["CMD", "curl", "-f", "http://localhost:8080/health"] + interval: 10s + timeout: 3s + retries: 3 + start_period: 10s + + # Test runner: black-box E2E suite (tests/e2e) against cudly-app over HTTP. + test-runner: + build: + context: . + dockerfile: Dockerfile.test + container_name: cudly-test-runner + depends_on: + postgres: + condition: service_healthy + cudly-app: + condition: service_healthy + environment: + # Point to test services + API_URL: http://cudly-app:8080 + DB_HOST: postgres + DB_PORT: 5432 + DB_NAME: cudly_test + DB_USER: cudly_test + DB_PASSWORD: test_password + DB_SSL_MODE: disable + networks: + - cudly-test + command: ["go", "test", "-v", "-tags=e2e", "./..."] + profiles: + - test # Only start with: docker compose --profile test up + +networks: + cudly-test: + driver: bridge + +volumes: + postgres_test_data: diff --git a/docker-compose.yml b/docker-compose.yml new file mode 100644 index 000000000..d9e45a85f --- /dev/null +++ b/docker-compose.yml @@ -0,0 +1,137 @@ +services: + # PostgreSQL database + postgres: + # Same digest as docker-compose.test.yml and ci.yml, so local development + # and both CI database paths run identical PostgreSQL. + image: postgres:16-alpine@sha256:cf78e76683b9ca8c5733cbbdce6c9262b45b6767934dd0a95e671f9a0fc20685 + container_name: cudly-postgres + environment: + POSTGRES_DB: cudly + POSTGRES_USER: cudly + POSTGRES_PASSWORD: cudly_local_dev + POSTGRES_INITDB_ARGS: "-E UTF8 --locale=C" + ports: + - "5432:5432" + volumes: + - postgres_data:/var/lib/postgresql/data + # Note: Migrations are handled by the app via golang-migrate (DB_AUTO_MIGRATE=true) + # Do NOT mount to docker-entrypoint-initdb.d as it conflicts with golang-migrate + healthcheck: + test: ["CMD-SHELL", "pg_isready -U cudly"] + interval: 5s + timeout: 5s + retries: 5 + networks: + - cudly-network + + # CUDly application (development mode) + app: + build: + context: . + dockerfile: Dockerfile.dev + container_name: cudly-app + depends_on: + postgres: + condition: service_healthy + environment: + # Database configuration + DB_HOST: postgres + DB_PORT: 5432 + DB_NAME: cudly + DB_USER: cudly + DB_PASSWORD: cudly_local_dev + DB_SSL_MODE: disable + DB_AUTO_MIGRATE: "true" + DB_MIGRATIONS_PATH: /app/internal/database/postgres/migrations + + # Secret provider — "env" reads secrets from env vars (local-dev only). + # Production uses "aws" / "azure" / "gcp" with real Secrets Manager. + SECRET_PROVIDER: env + + # Admin password secret: the EnvResolver looks up the env var whose + # NAME equals ADMIN_PASSWORD_SECRET. Here it points at ADMIN_PASSWORD_DEV. + ADMIN_PASSWORD_SECRET: ADMIN_PASSWORD_DEV + ADMIN_PASSWORD_DEV: "LocalDev!Pass123" + ADMIN_EMAIL: admin@cudly.local + + # API-key secret: same pattern — name → env var that holds the value. + # Frontend asks for this in the admin-setup modal. + API_KEY_SECRET_ARN: ADMIN_API_KEY_DEV + ADMIN_API_KEY_DEV: "cudly-local-dev-api-key-not-for-prod" + + # Credential encryption: dev key (all-zero) is gated behind this flag. + CREDENTIAL_ENCRYPTION_ALLOW_DEV_KEY: "1" + + # Email disabled for local development (no SNS topic / SES creds) + EMAIL_ENABLED: "false" + # Scheduled-task auth: OIDC for prod, disabled for local dev so the + # internal /api/scheduled/* endpoints don't require Google ID tokens. + SCHEDULED_TASK_AUTH_MODE: disabled + + # Application settings + ENVIRONMENT: development + LOG_LEVEL: debug + CORS_ALLOWED_ORIGIN: http://localhost:3001 + + # Optional: set initial admin password (skips password reset) + # ADMIN_PASSWORD: "YourSecureP@ss1" + + # AWS configuration (if needed for testing) + AWS_REGION: us-east-1 + # AWS_PROFILE: default # Uncomment if using AWS profiles + + ports: + - "8080:8080" + volumes: + # Mount source code for hot reload + - .:/app + # Cache Go modules + - go_modules:/go/pkg/mod + networks: + - cudly-network + command: air # Use Air for hot reload in development + + # Frontend (nginx with reverse proxy to backend) + frontend: + image: nginx:alpine + container_name: cudly-frontend + depends_on: + - app + ports: + - "3001:80" + volumes: + - ./frontend/dist:/usr/share/nginx/html:ro + - ./scripts/nginx.conf:/etc/nginx/conf.d/default.conf:ro + networks: + - cudly-network + + # pgAdmin for database management (local development only — do not expose in production) + pgadmin: + image: dpage/pgadmin4:9.2 + container_name: cudly-pgadmin + environment: + PGADMIN_DEFAULT_EMAIL: ${PGADMIN_EMAIL} + PGADMIN_DEFAULT_PASSWORD: ${PGADMIN_PASSWORD} + PGADMIN_CONFIG_SERVER_MODE: 'False' + ports: + - "5050:80" + volumes: + - pgadmin_data:/var/lib/pgadmin + networks: + - cudly-network + depends_on: + - postgres + profiles: + - tools # Only start with: docker-compose --profile tools up + +volumes: + postgres_data: + driver: local + pgadmin_data: + driver: local + go_modules: + driver: local + +networks: + cudly-network: + driver: bridge diff --git a/docs/DEPLOYMENT.md b/docs/DEPLOYMENT.md new file mode 100644 index 000000000..af8a8dfd2 --- /dev/null +++ b/docs/DEPLOYMENT.md @@ -0,0 +1,733 @@ +# CUDly Deployment Guide + +CUDly supports deployment via Terraform across three cloud providers (AWS, Azure, GCP). A helper script (`scripts/tf-deploy.sh`) simplifies common operations. + +## Table of Contents + +- [Quick Start](#quick-start) +- [Platform Comparison](#platform-comparison) +- [Terraform Deployment](#terraform-deployment) +- [AWS Deployment Details](#aws-deployment-details) +- [Azure Deployment Details](#azure-deployment-details) +- [GCP Deployment Details](#gcp-deployment-details) +- [Deploying Code Updates](#deploying-code-updates) +- [Fargate (Side-by-Side with Lambda)](#fargate-side-by-side-with-lambda) +- [CDN Architecture](#cdn-architecture) +- [Cost Estimates](#cost-estimates) +- [Monitoring](#monitoring) +- [Maintenance](#maintenance) +- [Troubleshooting](#troubleshooting) + +--- + +## Quick Start + +### Using the Deploy Script + +```bash +# Deploy to AWS dev environment +./scripts/tf-deploy.sh aws dev + +# Plan only (dry run) +./scripts/tf-deploy.sh aws dev plan + +# Deploy to other providers/environments +./scripts/tf-deploy.sh azure dev +./scripts/tf-deploy.sh gcp dev +./scripts/tf-deploy.sh aws prod +``` + +The script uses profile-based tfvars from `terraform/profiles//.tfvars`. + +### Manual Terraform + +```bash +cd terraform/environments/aws +cp dev.tfvars.example dev.tfvars # edit with your values + +terraform init -backend-config=backends/dev.tfbackend +terraform plan -var-file=dev.tfvars +terraform apply -var-file=dev.tfvars +``` + +Terraform automatically handles: Docker image build/push (via build module), frontend build/deploy to CDN, CDN cache invalidation, and admin user creation. + +### Prerequisites + +- Docker with buildx support +- Terraform >= 1.6.0 +- Go 1.26.6+ +- Cloud CLI configured: `aws`, `az`, or `gcloud` + +--- + +## Platform Comparison + +| Provider | Serverless | Containers/Kubernetes | Status | +| -------- | --------- | --------------------- | ------ | +| **AWS** | Lambda | Fargate (ECS) | Fully implemented | +| **Azure** | Container Apps | AKS | Container Apps implemented | +| **GCP** | Cloud Run | GKE | Cloud Run implemented | + +| | Lambda | Fargate | Container Apps | Cloud Run | +| --- | --- | --- | --- | --- | +| **Timeout** | 15 min max | Unlimited | Unlimited | 60 min max | +| **Memory** | 128-10240 MB | 512-30720 MB | 0.5-4 GB | 128 MB-32 GB | +| **CPU** | Tied to memory | 256-4096 units | 0.25-2 vCPU | 1-8 vCPU | +| **Scaling** | Auto (1000 concurrent) | Auto (1-N tasks) | Auto (0-N) | Auto (0-1000) | +| **Cold Start** | ~1-2s | No (always warm) | ~1-2s | ~0.5-1s | +| **Cost (idle)** | $0 | ~$30/mo (1 task) | $0 (scale to zero) | $0 (scale to zero) | +| **Load Balancer** | Not needed | Required (ALB) | Built-in | Built-in | + +--- + +## Terraform Deployment + +### Directory Structure + +```text +terraform/ +├── environments/ +│ ├── aws/ # main.tf, variables.tf, outputs.tf, backend.tf, +│ │ # networking.tf, database.tf, compute.tf, frontend.tf, +│ │ # secrets.tf, build.tf, ses.tf, route53.tf, acm.tf, +│ │ # dev.tfvars.example, backends/ +│ ├── azure/ # similar structure +│ └── gcp/ # similar structure +├── modules/ +│ ├── build/ # Docker build (docker-build.tf) +│ ├── compute/ +│ │ ├── aws/lambda/ +│ │ ├── aws/fargate/ +│ │ ├── aws/cleanup-lambda/ +│ │ ├── azure/container-apps/ +│ │ └── gcp/cloud-run/ +│ ├── database/ # aws/ (Aurora), azure/ (Flexible Server), gcp/ (Cloud SQL) +│ ├── frontend/ # aws/ (CloudFront+S3), azure/ (CDN+Blob), gcp/ (Cloud CDN+GCS) +│ ├── monitoring/ # aws/, azure/, gcp/ +│ ├── networking/ # aws/ (VPC), azure/ (VNet), gcp/ (VPC) +│ ├── registry/ # aws/ (ECR), azure/ (ACR), gcp/ (Artifact Registry) +│ └── secrets/ # aws/ (Secrets Manager), azure/ (Key Vault), gcp/ (Secret Manager) +└── profiles/ + ├── aws/ # dev.tfvars, prod.tfvars, fargate-dev.tfvars, *.example + ├── azure/ # dev.tfvars.example + └── gcp/ # dev.tfvars.example +``` + +### Manual Terraform Operations + +```bash +cd terraform/environments/aws + +terraform init -backend-config=backends/dev.tfbackend +terraform plan -var-file=../../profiles/aws/dev.tfvars +terraform apply -var-file=../../profiles/aws/dev.tfvars +terraform output +terraform destroy -var-file=../../profiles/aws/dev.tfvars +``` + +### Example dev.tfvars (AWS) + +See `terraform/environments/aws/dev.tfvars.example` for a complete reference. Key variables: + +```hcl +project_name = "cudly" +environment = "dev" +stack_name = "cudly-dev" +region = "us-east-1" +aws_profile = "default" + +# Compute platform: "lambda" or "fargate" +compute_platform = "lambda" + +# Lambda configuration +lambda_architecture = "arm64" +lambda_memory_size = 512 +lambda_timeout = 60 + +# Database configuration (Aurora Serverless v2) +database_engine_version = "16.4" +database_name = "cudly" +database_username = "cudly" +database_min_capacity = 0.5 +database_max_capacity = 2.0 +database_backup_retention_days = 7 + +# Admin user +admin_email = "admin@example.com" + +# Frontend (optional) +enable_frontend_build = true +frontend_price_class = "PriceClass_100" +``` + +### State Management + +**Local (dev):** State stored in `terraform.tfstate` (default). + +**Remote (production):** Configure backend using `backends/*.tfbackend` files: + +```bash +# Initialize with a specific backend config +terraform init -backend-config=backends/prod.tfbackend +``` + +Example backend config (`backends/prod.tfbackend`): + +```hcl +bucket = "cudly-terraform-state-prod" +key = "prod/terraform.tfstate" +region = "us-east-1" +encrypt = true +use_lockfile = true +``` + +--- + +## AWS Deployment Details + +### Infrastructure Created + +- **Compute:** Lambda (ARM64, container image) with Function URL, or Fargate (ECS) with ALB +- **Database:** Aurora Serverless v2 PostgreSQL 16.4 (0.5-2.0 ACU) +- **Network:** VPC (10.0.0.0/16) with IPv6 dual-stack, private/public subnets, no NAT Gateway +- **Proxy:** RDS Proxy for Lambda connection pooling +- **Frontend:** CloudFront (dual-origin: S3 for static, Lambda/ALB for API) + S3 +- **Secrets:** Secrets Manager (DB password, JWT secret, session secret) +- **Monitoring:** CloudWatch log groups, alarms, EventBridge scheduled tasks + +### Deploy with Script + +```bash +# Deploy to dev +./scripts/tf-deploy.sh aws dev + +# Plan only +./scripts/tf-deploy.sh aws dev plan + +# Deploy to staging/prod +./scripts/tf-deploy.sh aws prod +``` + +### Verify Deployment + +```bash +cd terraform/environments/aws + +FUNCTION_URL=$(terraform output -raw lambda_function_url) +curl "$FUNCTION_URL/health" +# Expected: {"status":"healthy","version":"...","timestamp":"...","checks":{"config_store":{"status":"healthy"},"auth_store":{"status":"healthy"}}} + +# Monitor logs +aws logs tail /aws/lambda/cudly-dev-api --follow +aws logs tail /aws/lambda/cudly-dev-api --filter-pattern "ERROR" +``` + +### Deploy Frontend + +```bash +cd terraform/environments/aws +S3_BUCKET=$(terraform output -raw frontend_bucket) +CF_DIST_ID=$(terraform output -raw cloudfront_distribution_id) + +# Build and upload +cd ../../../../frontend && npm install && npm run build + +aws s3 sync dist/ "s3://${S3_BUCKET}/" --delete \ + --cache-control "public,max-age=3600" + +aws cloudfront create-invalidation --distribution-id "$CF_DIST_ID" \ + --paths "/*" +``` + +--- + +## Azure Deployment Details + +### Infrastructure Created + +- **Compute:** Azure Container Apps (serverless containers) +- **Database:** Azure PostgreSQL Flexible Server +- **Frontend:** Azure Blob Storage (static website) + Azure CDN +- **Secrets:** Azure Key Vault +- **Monitoring:** Azure Monitor alerts + +```bash +# Deploy with script +./scripts/tf-deploy.sh azure dev +``` + +### Deploy Frontend + +```bash +az storage blob upload-batch \ + --account-name cudlyfrontendprod \ + --destination '$web' --source dist/ --overwrite + +az cdn endpoint purge \ + --resource-group cudly-prod-rg \ + --profile-name cudly-cdn-profile \ + --name cudly-cdn-endpoint --content-paths "/*" +``` + +--- + +## GCP Deployment Details + +### Infrastructure Created + +- **Compute:** Cloud Run service (serverless containers) +- **Database:** Cloud SQL PostgreSQL +- **Frontend:** Cloud Storage + Global HTTPS Load Balancer + Cloud CDN +- **Secrets:** Secret Manager +- **Monitoring:** Cloud Monitoring alerts +- **Optional:** Cloud Armor (WAF/DDoS protection) + +```bash +# Deploy with script +./scripts/tf-deploy.sh gcp dev +``` + +### Deploy Frontend + +```bash +gsutil -m rsync -r -d -x ".*\.map$" dist/ gs://cudly-frontend-prod/ + +gsutil -m setmeta -h "Cache-Control:public, max-age=31536000, immutable" \ + gs://cudly-frontend-prod/js/** + +gcloud compute url-maps invalidate-cdn-cache cudly-url-map --path "/*" +``` + +--- + +## Deploying Code Updates + +When infrastructure already exists and you only need to update application code: + +### Using the Deploy Script + +```bash +./scripts/tf-deploy.sh aws dev # dev +./scripts/tf-deploy.sh aws prod # production +./scripts/tf-deploy.sh aws dev plan # dry run +``` + +The script: initializes Terraform if needed, runs `terraform apply` with the profile-specific tfvars, and shows outputs on success. + +### Manual Steps + +```bash +# 1. Build image +GIT_COMMIT=$(git rev-parse --short HEAD) +AWS_ACCOUNT_ID=$(aws sts get-caller-identity --query Account --output text) +IMAGE_URI="${AWS_ACCOUNT_ID}.dkr.ecr.us-east-1.amazonaws.com/cudly:${GIT_COMMIT}" + +docker build --platform linux/arm64 --build-arg VERSION="$GIT_COMMIT" -t "$IMAGE_URI" . + +# 2. Push to ECR +aws ecr get-login-password --region us-east-1 | \ + docker login --username AWS --password-stdin ${AWS_ACCOUNT_ID}.dkr.ecr.us-east-1.amazonaws.com +docker push "$IMAGE_URI" + +# 3. Force replace Lambda +cd terraform/environments/aws +terraform apply -replace="module.compute_lambda[0].aws_lambda_function.main" -auto-approve +``` + +### CI/CD Integration + +```yaml +# .github/workflows/deploy.yml +name: Deploy to AWS +on: + push: + branches: [main] +jobs: + deploy: + runs-on: ubuntu-latest + steps: + - uses: actions/checkout@v4 + - uses: aws-actions/configure-aws-credentials@v4 + with: + aws-access-key-id: ${{ secrets.AWS_ACCESS_KEY_ID }} + aws-secret-access-key: ${{ secrets.AWS_SECRET_ACCESS_KEY }} + aws-region: us-east-1 + - run: ./scripts/tf-deploy.sh aws dev +``` + +--- + +## Fargate (Side-by-Side with Lambda) + +You can run Fargate alongside an existing Lambda deployment for comparison testing. Both connect to the same Aurora database. The terraform setup uses a single directory with different `.tfvars` and `.tfbackend` files per environment. + +### Architecture + +| | Lambda (`dev`) | Fargate (`fargate-dev`) | +| --- | --- | --- | +| **VPC** | 10.0.0.0/16 | 10.1.0.0/16 (separate) | +| **Compute** | Lambda + Function URL | ECS Fargate + ALB | +| **Database** | Aurora via RDS Proxy | Aurora via direct endpoint | +| **Frontend** | (none yet) | CloudFront + S3 | + +### Deploy Fargate + +```bash +cd terraform/environments/aws + +# Initialize with fargate-dev backend +terraform init -backend-config=backends/fargate-dev.tfbackend + +# Plan and apply with fargate-dev vars +terraform plan -var-file=fargate-dev.tfvars +terraform apply -var-file=fargate-dev.tfvars # ~10-15 minutes + +# Get URLs +terraform output fargate_api_url +terraform output frontend_url +``` + +### Compare Performance + +```bash +cd terraform/environments/aws + +# Lambda (1-2s cold start, then fast) +time curl $(terraform output -raw lambda_function_url)/health + +# Fargate (consistent ~100-200ms) +time curl $(terraform output -raw fargate_api_url)/health +``` + +### Fargate Configuration (tfvars) + +```hcl +compute_platform = "fargate" + +fargate_cpu = 512 # 256, 512, 1024, 2048, 4096 +fargate_memory = 1024 +fargate_desired_count = 2 +fargate_min_capacity = 1 +fargate_max_capacity = 10 +fargate_enable_https = false +fargate_enable_execute_command = false # ECS Exec for debugging +``` + +### Switching Between Platforms + +Change `compute_platform` in your tfvars and redeploy. The frontend module automatically adapts the API endpoint (Function URL vs ALB DNS). Update custom domain DNS if applicable. + +### Cleanup + +```bash +# Destroy only Fargate (leaves Lambda and database intact) +cd terraform/environments/aws +terraform destroy -var-file=fargate-dev.tfvars +``` + +--- + +## CDN Architecture + +The frontend is served through CDN with dual-origin routing: + +```text +User Request + | +CDN (CloudFront / Azure CDN / Cloud CDN) + | + +-- /api/* --> Backend (Lambda / Container Apps / Cloud Run) + +-- /* --> Static Files (S3 / Blob Storage / Cloud Storage) +``` + +**Static assets** (JS, CSS, images): Cached 1 year (content-hashed filenames). Compression enabled. +**HTML files**: No cache (`no-cache, no-store, must-revalidate`). +**API requests** (`/api/*`): No caching. All headers, cookies, and query strings forwarded. + +The frontend uses relative paths (`/api`) by default. Since the CDN proxies `/api/*` to the backend, requests are same-origin and CORS is not needed. + +### Per-Provider CDN Features + +**AWS CloudFront:** Origin Access Control (OAC) for S3, CloudFront Function for security headers (HSTS, X-Frame-Options), custom error pages for SPA routing, optional WAF, `X-CloudFront-Secret` header for origin verification. + +**Azure CDN:** Static website hosting via Blob Storage, Standard CDN or Front Door, managed SSL certificates, delivery rules for URL rewriting. + +**GCP Cloud CDN:** Global HTTPS Load Balancer, Cloud Armor for WAF/DDoS protection, automatic managed SSL certificates. + +### Verify CloudFront + +```bash +CF_DIST_ID=$(terraform output -raw cloudfront_distribution_id) + +# Check origins (should show S3 + Lambda/ALB) +aws cloudfront get-distribution --id "$CF_DIST_ID" \ + --query 'Distribution.DistributionConfig.Origins' --output json + +# Test path routing +CF_URL=$(terraform output -raw frontend_url) +curl -I "$CF_URL/" # X-Cache: Hit from cloudfront (static) +curl "$CF_URL/api/health" # X-Cache: Miss from cloudfront (API) +``` + +--- + +## Cost Estimates + +### AWS Lambda (low traffic, ~100K requests/month) + +| Resource | Monthly Cost | +| -------- | ------------ | +| Lambda | ~$0.20 | +| Aurora Serverless v2 (0.5 ACU) | ~$43.80 | +| RDS Proxy | ~$10.95 | +| CloudFront + S3 | ~$1 | +| Secrets Manager | ~$1 | +| **Total** | **~$57/month** | + +### AWS Fargate (2 tasks, 24/7) + +| Resource | Monthly Cost | +| -------- | ------------ | +| Fargate (0.25 vCPU, 0.5GB x 2) | ~$21.90 | +| ALB | ~$16.20 | +| Aurora Serverless v2 (0.5 ACU) | ~$43.80 | +| CloudFront + S3 | ~$1 | +| **Total** | **~$83/month** | + +### Multi-Cloud Comparison (Serverless, Dev) + +| Platform | Estimated Monthly Cost | +| -------- | --------------------- | +| AWS Lambda | ~$57 | +| GCP Cloud Run | ~$13-27 | +| Azure Container Apps | ~$18-32 | + +### Cost Optimization + +- ARM64 Lambda/Fargate: 20% cheaper than x86 +- Aurora scales to 0.5 ACU when idle +- Lambda/Cloud Run/Container Apps scale to zero +- IPv6 dual-stack eliminates NAT Gateway costs +- CloudFront PriceClass_100: US/Europe only for lower costs +- Fargate Spot: up to 70% discount for non-critical workloads + +--- + +## Monitoring + +### AWS Lambda + +```bash +aws logs tail /aws/lambda/cudly-dev-api --follow +aws logs tail /aws/lambda/cudly-dev-api --filter-pattern "ERROR" --since 10m +``` + +### AWS Fargate + +```bash +# Service status +aws ecs describe-services --cluster cudly-dev-fargate --services cudly-dev-fargate \ + --query 'services[0].{Status:status,Running:runningCount,Desired:desiredCount}' + +# Logs +aws logs tail /ecs/cudly-dev-fargate --follow + +# ECS Exec (if enabled) +aws ecs execute-command --cluster cudly-dev-fargate --task \ + --container app --interactive --command "/bin/sh" +``` + +### Azure Container Apps + +```bash +az containerapp logs show --name cudly-dev --resource-group cudly-rg +``` + +### GCP Cloud Run + +```bash +gcloud run services logs read cudly-dev --region us-central1 +``` + +--- + +## Maintenance + +### Container Registry Cleanup + +Each cloud provider has lifecycle policies configured via Terraform (`terraform/modules/registry/{aws,azure,gcp}/`): + +- **AWS ECR:** Keep last 10 tagged images, delete untagged after 7 days, vulnerability scanning on push +- **GCP Artifact Registry:** Cleanup policies for untagged and old images +- **Azure ACR:** ACR tasks for automated cleanup + +### Database Cleanup Jobs + +Automated cleanup for expired sessions and completed purchase executions, running on a daily schedule: + +- **AWS:** Lambda function triggered by EventBridge (`terraform/modules/compute/aws/cleanup-lambda/`) +- **Azure:** Azure Function App with timer trigger +- **GCP:** Cloud Function with Cloud Scheduler + +Alternatively, use pg_cron for database-native scheduling. + +--- + +## Troubleshooting + +### Docker Build Fails + +```bash +docker ps # is Docker running? +docker system df # disk space +go build ./cmd/server # syntax errors? +go mod tidy # dependency issues? +``` + +### Lambda Function Not Updating + +Terraform may show "No changes" if the image tag didn't change. Force replace: + +```bash +terraform apply -replace="module.compute_lambda[0].aws_lambda_function.main" -auto-approve +``` + +### Database Connection Failed + +```bash +# Check Lambda is in VPC +aws lambda get-function-configuration --function-name cudly-dev-api --query 'VpcConfig' + +# Check RDS Proxy +aws rds describe-db-proxies --db-proxy-name cudly-dev-proxy +# Check database cluster +aws rds describe-db-clusters --db-cluster-identifier cudly-dev-postgres +``` + +### Multi-Account Cross-Account Access + +Declare every AWS account the deployment will call `sts:AssumeRole` against. Nothing is reachable +cross-account until you do. + +| Deployment shape | Setting | +| ---------------- | ------- | +| Terraform (`terraform/environments/aws`) | `cross_account_target_account_ids = ["111111111111", ...]` in your tfvars | +| CloudFormation (`cloudformation/stacks/CUDly`) | `CrossAccountTargetAccountIds` stack parameter, comma-separated | + +Both render an `aws:ResourceAccount` condition onto the grant. An account that is not listed is +denied by IAM, not merely by the app's account selection. Leaving the setting empty creates **no +cross-account grant at all**, which is the intended fail-closed default rather than an oversight. + +For accounts using `bastion` auth mode, list the **bastion's** account ID rather than the target's. +Only the first hop runs on the deployment's own identity; the bastion assumes into the target on its +own identity policy, which this setting does not govern. + +> **Upgrading an existing multi-account deployment**: the grant used to be scoped by role name only +> (`arn:aws:iam::*:role/CUDly*`), which matched a CUDly role in *every* AWS account rather than in +> yours (#1636). List your linked accounts **in the same change that picks up this version**. On +> Terraform the apply removes `aws_iam_role_policy.cross_account_sts` and exits 0; on CloudFormation +> the stack update drops the statement. Either way the first symptom otherwise is a runtime +> `AccessDenied` during collection, not a failed deploy. + +### Multi-Account Credential Encryption + +Multi-account support requires an AES-256-GCM encryption key for stored cloud account credentials. Terraform creates the key secret automatically (see `specs/multi-account-execution/iac.md`). + +**New environment variables (AWS Lambda):** + +| Variable | Source | Purpose | +| -------- | ------ | ------- | +| `CREDENTIAL_ENCRYPTION_KEY_SECRET_ARN` | Terraform → Lambda env | ARN of Secrets Manager secret holding the 32-byte AES key | +| `CREDENTIAL_ENCRYPTION_KEY` | Direct (local dev only) | 64-char hex key; bypasses Secrets Manager | +| `CUDLY_MAX_ACCOUNT_PARALLELISM` | Terraform → Lambda env | Fan-out goroutine cap for parallel account execution (default: `10`) | + +**Migration 000011** (`000011_cloud_accounts`) adds: + +- `cloud_accounts` — central registry for all managed accounts (AWS/Azure/GCP) +- `account_credentials` — encrypted credential material +- `account_service_overrides` — sparse per-account service config overrides +- `plan_accounts` — M2M join: which accounts a purchase plan targets +- `cloud_account_id` FK column on `purchase_executions`, `purchase_history`, `savings_snapshots`, `ri_exchange_history` + +**Migration 000012** (`000012_global_config_fields`) adds `auto_collect`, `collection_schedule`, and `notification_days_before` columns to the `global_config` table. + +**Migration 000013** (`000013_add_running_status`) adds a `running` status value to the `purchase_executions` status enum. + +**Migration 000014** (`000014_azure_auth_mode`) adds `azure_auth_mode` column to `cloud_accounts`. Supported values: `client_secret`, `managed_identity`, `workload_identity_federation`. + +**Migration 000015** (`000015_gcp_auth_mode`) adds `gcp_auth_mode` column to `cloud_accounts`. Supported values: `service_account_key`, `application_default`, `workload_identity_federation`. + +**Migration 000016** (`000016_aws_wif`) adds `aws_web_identity_token_file` column to `cloud_accounts`, used for AWS Workload Identity Federation (token file path on the host platform). + +If `DB_AUTO_MIGRATE=true` (the default), migration 000011 runs automatically on Lambda cold start. To run manually: + +> The URL scheme is `pgx5://`, not `postgres://`. golang-migrate selects its +> driver by scheme, and `migrate` is built here with `-tags pgx5` so it does not +> link `lib/pq` (issue #1849). A `postgres://` URL fails with +> `unknown driver postgres`. + +```bash +DB_PASSWORD=$(aws secretsmanager get-secret-value --secret-id cudly-dev-db-password-* --query SecretString --output text | jq -r .password) +RDS_ENDPOINT=$(cd terraform/environments/aws && terraform output -raw database_proxy_endpoint) + +migrate -path internal/database/postgres/migrations \ + -database "pgx5://cudly:${DB_PASSWORD}@${RDS_ENDPOINT}:5432/cudly?sslmode=require" up +``` + +--- + +### Migrations Not Running + +Check: `DB_AUTO_MIGRATE=true`, `DB_MIGRATIONS_PATH=/app/internal/database/postgres/migrations`, correct credentials. Run manually: + +```bash +DB_PASSWORD=$(aws secretsmanager get-secret-value --secret-id cudly-dev-db-password-* --query SecretString --output text) +RDS_ENDPOINT=$(cd terraform/environments/aws && terraform output -raw database_proxy_endpoint) + +migrate -path internal/database/postgres/migrations \ + -database "pgx5://cudly:${DB_PASSWORD}@${RDS_ENDPOINT}:5432/cudly?sslmode=require" up +``` + +### Terraform State Lock + +```bash +terraform force-unlock +``` + +### CloudFront Returns 403 + +S3 bucket is empty. Deploy frontend or add a placeholder: + +```bash +S3_BUCKET=$(terraform output -raw frontend_bucket) +echo "

CUDly

" | aws s3 cp - "s3://${S3_BUCKET}/index.html" +``` + +### Fargate Tasks Not Starting + +```bash +aws ecs describe-services --cluster "$CLUSTER" --services "$SERVICE" --query 'services[0].events[:5]' +# Common: image pull errors, resource limits, health check failures +``` + +### ALB Returns 502/504 Through CloudFront + +Test ALB directly to isolate: + +```bash +ALB_DNS=$(terraform output -raw fargate_alb_dns_name) +curl "http://${ALB_DNS}/health" +# If ALB works but CloudFront doesn't: check origin settings, custom header, security groups +``` + +### Rollback + +Deploy the previous git commit: + +```bash +git log --oneline -5 +git checkout +./scripts/tf-deploy.sh aws dev +git checkout main +``` diff --git a/docs/DEVELOPMENT.md b/docs/DEVELOPMENT.md new file mode 100644 index 000000000..92a77ec4c --- /dev/null +++ b/docs/DEVELOPMENT.md @@ -0,0 +1,437 @@ +# CUDly Development Guide + +## Prerequisites + +- Docker and Docker Compose +- Go 1.26.6+ +- Node.js and npm (for frontend development) +- Make (optional, for convenience commands) + +## Quick Start + +### 1. Start Local Environment + +```bash +# Start PostgreSQL and CUDly application +docker-compose up -d + +# View logs +docker-compose logs -f app + +# Stop environment +docker-compose down +``` + +### 2. Access Services + +- **CUDly API**: +- **PostgreSQL**: localhost:5432 + - Database: `cudly` + - User: `cudly` + - Password: `cudly_local_dev` +- **pgAdmin** (optional): + - Email: `admin@cudly.local` + - Password: `admin` + - Start with: `docker-compose --profile tools up` + +### 3. Database Access + +> Migration URLs use the `pgx5://` scheme, not `postgres://`. golang-migrate +> picks its database driver by URL scheme, and this repo builds `migrate` with +> `-tags pgx5` so the binary does not link `lib/pq`, which carries advisories +> with no published fix (issue #1849). A `postgres://` URL fails against it with +> `unknown driver postgres`. + +```bash +# Connect to PostgreSQL using psql +docker-compose exec postgres psql -U cudly -d cudly + +# Run migrations manually +docker-compose exec app migrate -path /app/internal/database/postgres/migrations \ + -database "pgx5://cudly:cudly_local_dev@postgres:5432/cudly?sslmode=disable" up + +# Check migration status +docker-compose exec app migrate -path /app/internal/database/postgres/migrations \ + -database "pgx5://cudly:cudly_local_dev@postgres:5432/cudly?sslmode=disable" version +``` + +## Development Workflow + +### Hot Reload + +The development environment uses **Air** for hot reload. Any changes to `.go` files automatically trigger a rebuild and restart. + +```bash +# Air automatically detects changes and reloads +docker-compose logs -f app +``` + +### Database Migrations + +```bash +# Create a new migration +migrate create -ext sql -dir internal/database/postgres/migrations -seq add_new_feature + +# Run migrations +docker-compose exec app migrate -path /app/internal/database/postgres/migrations \ + -database "pgx5://cudly:cudly_local_dev@postgres:5432/cudly?sslmode=disable" up + +# Rollback last migration +docker-compose exec app migrate -path /app/internal/database/postgres/migrations \ + -database "pgx5://cudly:cudly_local_dev@postgres:5432/cudly?sslmode=disable" down 1 +``` + +## Environment Variables + +Configured in docker-compose.yml: + +### Database + +- `DB_HOST`: PostgreSQL hostname (docker-compose: `postgres`) +- `DB_PORT`: PostgreSQL port (default: `5432`) +- `DB_NAME`: Database name (default: `cudly`) +- `DB_USER`: Database user (default: `cudly`) +- `DB_PASSWORD`: Database password (docker-compose: `cudly_local_dev`) +- `DB_SSL_MODE`: SSL mode (app default: `require`; docker-compose overrides to `disable`) +- `DB_AUTO_MIGRATE`: Auto-run migrations on startup (default: `true`) + +### Secrets + +- `SECRET_PROVIDER`: Secret manager provider (default: `env`) + - `env`: Use environment variables (suitable for local dev) + - `aws`: AWS Secrets Manager + - `gcp`: GCP Secret Manager + - `azure`: Azure Key Vault + +### Multi-Account Credential Encryption + +CUDly encrypts stored cloud account credentials with AES-256-GCM. The encryption key must be provided via one of: + +- **Local dev** — set `CREDENTIAL_ENCRYPTION_KEY` to a 64-character hex string (32 bytes): + + ```bash + export CREDENTIAL_ENCRYPTION_KEY=$(openssl rand -hex 32) + ``` + + Add this to your `docker-compose.yml` or `.env` file for local use. + +- **Production (AWS Lambda)** — set `CREDENTIAL_ENCRYPTION_KEY_SECRET_ARN` to the ARN of an AWS Secrets Manager secret whose value is the 64-char hex key. Terraform creates this secret automatically when `create_credential_encryption_key = true` is set in `secrets.tf`. See [Deployment Guide](DEPLOYMENT.md#multi-account-credential-encryption) and `specs/multi-account-execution/iac.md` for details. + +If neither variable is set, the application falls back to an insecure dev key and logs a warning. **Never use the dev key in production.** + +### Application + +- `ENVIRONMENT`: Environment name (default: `development`) +- `LOG_LEVEL`: Logging level (default: `debug`) + +--- + +## Testing + +### Testing Levels + +**Unit tests** verify individual functions in isolation. Located in `*_test.go` files, run by default. Coverage target: >85%. + +```bash +make test-unit +# or +go test -v -race -short ./... +``` + +**Integration tests** verify component interactions with real dependencies (databases via testcontainers). Located in `*_test.go` files with `//go:build integration` tag. Coverage target: >80%. + +```bash +make test-integration +# or +go test -v -race -tags=integration ./... +``` + +**E2E tests** verify the complete application flow using docker-compose. + +```bash +make docker-compose-test +# or +docker-compose -f docker-compose.test.yml up --abort-on-container-exit --exit-code-from test-runner +``` + +### Running Tests + +```bash +# Quick test (unit only) +make test + +# Full test suite (unit + integration + coverage) +make full-test + +# Coverage report +make test-coverage +# Generates coverage.out and coverage.html +open coverage.html + +# Specific package +go test -v ./internal/server/... + +# Specific function +go test -v -run TestHandleScheduledTask ./internal/server/ + +# Integration tests (requires PostgreSQL) +docker-compose up -d postgres +DB_HOST=localhost DB_PASSWORD=cudly_local_dev go test -tags=integration ./internal/database/... +``` + +### Writing Tests + +Use the AAA (Arrange-Act-Assert) pattern: + +```go +func TestMyFunction(t *testing.T) { + // Arrange + ctx := testutil.TestContext(t) + input := "test data" + + // Act + result, err := MyFunction(ctx, input) + + // Assert + testutil.AssertNoError(t, err) + testutil.AssertEqual(t, "expected", result) +} +``` + +Table-driven tests for multiple scenarios: + +```go +func TestMultipleScenarios(t *testing.T) { + tests := []struct { + name string + input string + expected string + expectError bool + }{ + {name: "valid input", input: "test", expected: "TEST"}, + {name: "invalid input", input: "", expectError: true}, + } + + for _, tt := range tests { + t.Run(tt.name, func(t *testing.T) { + result, err := Transform(tt.input) + if tt.expectError { + testutil.AssertError(t, err) + } else { + testutil.AssertNoError(t, err) + testutil.AssertEqual(t, tt.expected, result) + } + }) + } +} +``` + +### Test Helpers (`internal/testutil`) + +```go +// Context with timeout +ctx := testutil.TestContext(t) + +// Environment variables +testutil.SetEnv(t, "DB_HOST", "localhost") + +// Skip conditions +testutil.SkipIfShort(t) +testutil.SkipCI(t) + +// Assertions +testutil.AssertNoError(t, err) +testutil.AssertEqual(t, expected, actual) +testutil.AssertTrue(t, condition, "message") +testutil.AssertContains(t, haystack, needle) + +// Wait for condition +testutil.WaitFor(t, func() bool { + return server.IsReady() +}, 5*time.Second, "server to be ready") +``` + +### Mocking Dependencies + +```go +mockScheduler := &testutil.MockScheduler{ + CollectRecommendationsFunc: func(ctx context.Context) (*scheduler.CollectResult, error) { + return &scheduler.CollectResult{Count: 10}, nil + }, +} + +app := &Application{Scheduler: mockScheduler} +``` + +### Integration Test Setup + +```go +//go:build integration + +func TestWithPostgres(t *testing.T) { + if testing.Short() { + t.Skip("Skipping integration test") + } + + ctx := testutil.TestContext(t) + pg, err := testutil.SetupPostgresContainer(ctx, t) + testutil.AssertNoError(t, err) + + for k, v := range pg.Config() { + testutil.SetEnv(t, k, v) + } + + // Test with real database... +} +``` + +### Coverage + +**Targets:** Unit >85%, Integration >80%, Critical paths 100% + +```bash +make test-coverage +go tool cover -func=coverage.out # by package +go tool cover -html=coverage.out # in browser +``` + +**Exceptions**: Generated code, trivial getters/setters, unreachable panic handlers. + +### CI/CD + +Tests run on PRs, pushes to main, and release tags via `.github/workflows/ci.yml`. + +```bash +# Run full CI pipeline locally +make ci # formatting, vet, complexity check, unit tests, security scanning, terraform validation + +# Pre-commit hook +make pre-commit # formatting, vet, complexity check, unit tests +``` + +### Security Testing + +```bash +# All security scans +make security-scan + +# Individual scans +make security-scan-go # gosec +make security-scan-docker # trivy (container + filesystem) +make security-scan-terraform # tfsec +``` + +### Terraform & Docker Testing + +```bash +make terraform-validate +make terraform-fmt-check +make terraform-fmt + +make docker-build # build Docker image +make docker-test # build and test image +make docker-compose-test # E2E tests with docker-compose +``` + +--- + +## Frontend Development + +The frontend is a TypeScript application in `frontend/src/`, built with webpack. + +```bash +cd frontend + +# Install dependencies +npm install + +# Development build (with watch) +npm run dev + +# Production build +npm run build + +# Run tests +npx jest + +# Run tests with coverage +npx jest --coverage +``` + +The frontend builds to `frontend/dist/` and is deployed to CDN (CloudFront/Azure CDN/Cloud CDN) as static files. + +--- + +## Troubleshooting + +### Database Connection Issues + +```bash +docker-compose ps postgres +docker-compose logs postgres +docker-compose exec app pg_isready -h postgres -U cudly +``` + +### Migration Issues + +```bash +# Check current migration version +docker-compose exec postgres psql -U cudly -d cudly -c "SELECT * FROM schema_migrations;" + +# Force migration version (use with caution) +migrate -path internal/database/postgres/migrations \ + -database "pgx5://cudly:cudly_local_dev@localhost:5432/cudly?sslmode=disable" \ + force +``` + +### Clean Restart + +```bash +docker-compose down -v +docker-compose up --build -d +``` + +### Test Issues + +- **"context deadline exceeded"**: Increase timeout in `testutil.TestContext()` or use a longer context +- **Docker not available**: Integration tests require Docker: +- **testcontainers fails**: Ensure Docker daemon is running (`docker ps`) +- **Race condition detected**: Run with `go test -race ./...` +- **Coverage too low**: Find uncovered code with `go tool cover -func=coverage.out | grep -v "100.0%"` + +--- + +## Testing with Real AWS Credentials + +```bash +# Option 1: Mount AWS credentials +# In docker-compose.yml: +# app: +# environment: +# SECRET_PROVIDER: aws +# volumes: +# - ~/.aws:/root/.aws:ro + +# Option 2: Environment variables +export AWS_ACCESS_KEY_ID=xxx +export AWS_SECRET_ACCESS_KEY=xxx +export AWS_REGION=us-east-1 + +docker-compose restart app +``` + +## Production-Like Testing + +```bash +# Build production image +docker build -t cudly:latest . + +# Run with PostgreSQL +docker run --rm \ + --network cudly_cudly-network \ + -e DB_HOST=postgres \ + -e DB_PASSWORD=cudly_local_dev \ + -e RUNTIME_MODE=http \ + -p 8080:8080 \ + cudly:latest +``` diff --git a/docs/audits/codebase-audit-2026-09-02.md b/docs/audits/codebase-audit-2026-09-02.md new file mode 100644 index 000000000..49ce984dc --- /dev/null +++ b/docs/audits/codebase-audit-2026-09-02.md @@ -0,0 +1,10352 @@ +# CUDly codebase audit, 2026-09-02 + +| | | +|---|---| +| Audited commit | `3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd` | +| Commit is | the tip of `origin/main` at audit start, not the working tree | +| Audit window | 2026-09-01T23:29Z to 2026-09-02T20:41Z | +| Review shards | 19, all completed | +| Findings written | 545 | +| Verification | every finding checked by an independent verifier that did not write it | + +The audit read the pinned commit only. Uncommitted working-tree changes were out of scope. +Each shard was reviewed by one agent and then verified by a second agent that had not seen the +first agent's reasoning. Verdicts below are the verifier's, not the reviewer's. + +## How to read this report + +**CONFIRMED** means the verifier reproduced the reviewer's claim against the pinned commit, by +reading the cited code, running a test, or both. **PLAUSIBLE** means the mechanism holds but the +verifier could not close the last step, usually reachability or a runtime value it could not +observe. **REJECTED** means the verifier found the claim wrong: the code does not do what the +finding says, or a compensating control the reviewer missed makes the scenario impossible. + +Rejected findings are not deleted. They are moved to the end of this report with the verifier's +reasoning intact, so that the same claim is not re-raised later by someone who has not seen why +it fails. Nothing in the rejected section should be actioned. + +A `severity-adjusted` line means the verifier disagreed with the reviewer's rating and changed +it. Where that line is present, the adjusted value governs: this report sorts, groups and counts +the finding by the adjusted severity, and the original `severity` line is left in place so the +disagreement stays visible. + +Some findings sit on code paths that no live caller reaches. The verifier says so in its verdict, +usually as the reason for a downgrade. A latent defect on an unreachable path is real and worth +fixing, but it is not shipping a failure today, so it ranks below a reachable finding of the same +severity. + +## Shard coverage + +Counts are parsed from the shard files, not quoted from any summary. + +| Shard | Scope | Findings | Confirmed | Plausible | Rejected | Not read | +|---|---|---:|---:|---:|---:|---| +| A01 | Money-path API handlers: purchases, RI exchange, plans, plan health, request validation | 22 | 22 | 0 | 0 | internal/api/types_apikeys.go unread; most test files read by excerpt only, deferred to A16. | +| A02 | The rest of internal/api: routing, middleware, auth, accounts, users, groups, API keys, recommendations, history, dashboard, config, analytics, rate limiters, OpenAPI spec | 26 | 25 | 1 | 0 | OpenAPI component schemas spot-checked only. Tests and mocks grepped for specific questions, not read end to end. | +| A03 | Authentication and secrets: internal/auth, internal/oidc, internal/credentials, internal/secrets | 25 | 21 | 1 | 3 | Most test files in these four packages unread, folded into A16. | +| A04 | internal/config, internal/database, SQL migrations | 20 | 17 | 2 | 1 | Most internal/config and internal/database test files unread, folded into A16. | +| A05 | Purchase execution and scheduling: internal/purchase, internal/execution, internal/scheduler, internal/commitmentopts | 19 | 14 | 4 | 1 | All non-test files read. About 22 test files grepped only, deferred to A16. | +| A06 | internal/server, email, analytics, reporter, accounts, iacfiles, runtime | 30 | 23 | 3 | 4 | Tests not read as tests; deferred to A16. | +| A07 | providers/aws | 33 | 28 | 1 | 4 | All 27 non-test Go files read in full. 32 test files deferred to A16. providers/aws/services/AUDIT.md not read. | +| A08 | providers/azure and providers/gcp, first pass | 28 | 18 | 8 | 2 | Left unread: azure httpclient.go, six Azure service clients, compute/exchange.go, displayname.go, mocks; GCP cloudsql, cloudstorage, memorystore clients and most of computeengine. Split into A08b. All Azure and GCP tests, about 23k lines, deferred to A16. | +| A08b | The Azure and GCP provider clients A08 left unread, plus pkg/httpclient and Azure internal pricing | 45 | 40 | 5 | 0 | computeengine lines 560-1188 cited by grep rather than full read in findings 019, 029 and 032. | +| A09 | pkg/ shared libraries: httpclient, exchange, recfilter, common, provider, ladder, scorer, retry, logging, config | 32 | 26 | 3 | 3 | pkg/config/load.go and types.go bodies unread; the claim that pkg/config has no importers was left to the verifier. | +| A10 | cmd/: CLI, MCP server, Lambda handlers, rekey, cleanup, gen-permissions | 30 | 28 | 2 | 0 | Tests deferred to A16. | +| A11 | frontend/src part 1: auth, state, settings, users, groups, permissions, API client | 25 | 15 | 5 | 5 | Frontend tests not read as tests. | +| A12 | frontend/src part 2, money-facing: recommendations, plans, purchase modal, RI exchange, history, dashboard, ladder | 76 | 64 | 4 | 8 | recommendations.ts read to about 3100 of 5481 lines; setup and column-filter helpers unread. riexchange.ts and history.ts were delegated to sub-reviewers, with criticals and highs re-verified by the shard reviewer. | +| A13 | terraform, iac, cloudformation, arm, Docker, scanner suppressions, first pass | 24 | 22 | 2 | 0 | Covered only a fraction of scope. Most of terraform/modules, terraform/environments, iac/federation and both cloudformation/stacks templates unread. Split into A13b and A13c. | +| A13b | Customer-facing federation trust boundary: aws-target, aws-cross-account, azure-target, gcp-sa-impersonation, CloudFormation stacks, ARM templates | 16 | 14 | 1 | 1 | Scope covered as assigned. | +| A13c | Terraform provisioning CUDly's own infrastructure | 26 | 20 | 2 | 4 | Stopped in priority 5. Unread: monitoring modules for all three clouds, frontend/azure and frontend/gcp, build/outputs.tf and build/scripts, policy_guard_test.go, test-iam-role.sh, ci-cd-permissions/policy_*.tf, and the AKS, GKE and cleanup-function compute modules. | +| A14 | .github/workflows, scripts, Makefiles, pre-commit and linter configs, secret allowlists | 43 | 37 | 2 | 4 | Scope covered as assigned. | +| A15 | Cross-cutting: dependency and vulnerability scanning, npm audit, govulncheck, currency handling, CI tool health | 11 | 9 | 0 | 2 | A sampling pass rather than a file-by-file read; it examined what crosses package boundaries. | +| A16 | Test files judged as tests, using go test coverage profiles rather than greps | 14 | 11 | 0 | 3 | Stopped partway through priority 2. Untouched: internal/credentials and internal/secrets tests, three OIDC cloud signers, most of internal/api and all internal/config tests, and all tests under providers, pkg, cmd and frontend. | +| **Total** | | **545** | **454** | **46** | **45** | | + +Shards A08b, A13b, A13c and A16 exist because an earlier reviewer left part of its assigned scope +unread. A08 deferred most Azure and GCP service clients, A13 covered only a fraction of the +infrastructure tree, and every shard that deferred its test files pushed them to A16. Each +follow-up shard was given the unread files as its own scope and was verified the same way. + +## Summary + +Actionable means CONFIRMED or PLAUSIBLE. Rejected findings are excluded from every table in this +section. + +### Actionable findings by severity + +| Severity | Findings | +|---|---:| +| critical | 10 | +| high | 82 | +| medium | 218 | +| low | 190 | +| **Total** | **500** | + +### Actionable findings by category + +| Category | Findings | +|---|---:| +| security | 78 | +| money-path | 86 | +| silent-fallback | 65 | +| correctness | 118 | +| concurrency | 12 | +| performance | 9 | +| test-gap | 29 | +| duplication | 13 | +| dead-code | 31 | +| over-engineering | 3 | +| ops | 37 | +| hygiene | 14 | +| bug | 1 | +| bugs | 4 | +| **Total** | **500** | + +### Severity against category + +| Category | critical | high | medium | low | Total | +|---|---:|---:|---:|---:|---:| +| security | 2 | 21 | 31 | 24 | 78 | +| money-path | 6 | 32 | 40 | 8 | 86 | +| silent-fallback | 0 | 3 | 35 | 27 | 65 | +| correctness | 2 | 12 | 59 | 45 | 118 | +| concurrency | 0 | 2 | 7 | 3 | 12 | +| performance | 0 | 0 | 5 | 4 | 9 | +| test-gap | 0 | 8 | 13 | 8 | 29 | +| duplication | 0 | 0 | 4 | 9 | 13 | +| dead-code | 0 | 0 | 8 | 23 | 31 | +| over-engineering | 0 | 0 | 0 | 3 | 3 | +| ops | 0 | 4 | 15 | 18 | 37 | +| hygiene | 0 | 0 | 1 | 13 | 14 | +| bug | 0 | 0 | 0 | 1 | 1 | +| bugs | 0 | 0 | 0 | 4 | 4 | +| **Total** | **10** | **82** | **218** | **190** | **500** | + +## Critical and high findings + +Every actionable finding rated critical or high after verification, in one line each. The full +block for each is in the findings section under its category. + +| ID | Severity | Category | Location | Summary | +|---|---|---|---|---| +| A03-001 | critical | security | `internal/auth/group_ceiling.go:85` | Resource wildcard `execute:*` bypasses the admin money-verb carve-out at both the grant ceiling and enforcement | +| A05-001 | critical | money-path | `internal/purchase/execution.go:371` | Direct-execute spanning two cloud accounts silently buys in the ambient host account | +| A08-001 | critical | money-path | `providers/gcp/services/computeengine/client.go:1392` | GCP CUD purchase reads memory from a value-typed Details, but the purchase-execution path supplies a pointer | +| A08b-001 | critical | security | `pkg/httpclient/httpclient.go:41` | IMDS blocklist matches literal host text, so IPv6-mapped, alternate-radix and DNS forms of 169.254.169.254 pass through | +| A08b-005 | critical | money-path | `providers/azure/services/cache/client.go:207` | `ReservationsDetails` is a daily usage API, so `GetExistingCommitments` emits one Commitment per reservation per day | +| A09-004 | critical | money-path | `pkg/exchange/exchange.go:305` | An absent PaymentDue is treated as $0, disabling the spend cap on an irreversible exchange | +| A11-001 | critical | correctness | `frontend/src/groups/handlers.ts:17` | The group create/edit form's submit handler is never wired in production | +| A12-001 | critical | money-path | `frontend/src/recommendations.ts:5412` | Purchase modal's Term and Payment selects mutate the submitted rec without re-pricing it | +| A12-002 | critical | money-path | `frontend/src/recommendations.ts:4336` | Fan-out modal says an incompatible bucket "will be skipped", then submits it | +| A12-003 | critical | correctness | `frontend/src/riexchange.ts:1809` | RI-exchange execute never sends `region`, which the backend rejects with 400 | +| A01-001 | high | money-path | `internal/api/handler_purchases.go:2255` | MaxPurchaseAmount cap is enforced against client-asserted dollar amounts, not the priced commitment | +| A01-002 | high | security | `internal/api/handler_purchases.go:685` | Session-authed approve, cancel, retry and revoke never check the session's allowed_accounts scope | +| A01-003 | high | correctness | `internal/api/handler_purchases.go:320` | "Run now" strands the execution in `running`; nothing executes it and the reaper marks it failed | +| A01-004 | high | money-path | `internal/api/handler_plans.go:281` | PUT /api/plans/{id} rebuilds the ramp schedule from scratch, resetting CurrentStep/StartDate and re-arming the ramp | +| A02-001 | high | security | `internal/api/handler_dashboard.go:38` | Dashboard commitment KPIs skip allowed_accounts scoping when the caller supplies an explicit account filter | +| A02-002 | high | correctness | `internal/api/handler_auth.go:238` | setupAdmin stores the base64-encoded password, so the bootstrap admin cannot log in with the password they typed | +| A03-002 | high | security | `internal/auth/service_apikeys.go:128` | An admin can mint a user API key carrying `execute:*` in a single request and spend with it | +| A03-003 | high | security | `internal/auth/service_user.go:551` | Self-membership guard checks carved-out pairs exactly, so an admin can join a wildcard group they created | +| A03-004 | high | security | `internal/auth/service_user.go:396` | User-membership writes have no grant ceiling: any create:users or update:users holder can mint an Administrator or Purchaser | +| A03-005 | high | security | `internal/auth/service.go:316` | Deactivating a user does not end their sessions, and session validation never re-checks the user | +| A03-006 | high | security | `internal/auth/service_password.go:355` | A deactivated account can reactivate itself through the forgot-password flow | +| A05-002 | high | test-gap | `internal/purchase/execution_test.go:1628` | `singleCloudAccountIDFromRecs` is unit-tested in isolation, so the ambient fallback above stays green | +| A06-001 | high | concurrency | `internal/server/handler.go:112` | Advisory lock is released with the request context, so a canceled/expired scheduled run strands the lock on a pooled connection | +| A06-003 | high | security | `internal/server/http.go:303` | SourceIP carries the TCP port (or the proxy's IP), defeating the login brute-force rate limit | +| A07-001 | high | money-path | `providers/aws/services/savingsplans/client.go:425` | CE-supplied OfferingID short-circuits every Savings Plans purchase validation | +| A07-006 | high | correctness | `providers/aws/services/elasticache/client.go:311` | ElastiCache offering lookup sends the raw CE engine string as ProductDescription, unvalidated | +| A07-007 | high | money-path | `providers/aws/services/elasticache/client.go:92` | ElastiCache and MemoryDB commitments carry no Engine, so the duplicate-purchase guard never matches | +| A07-010 | high | correctness | `providers/aws/recommendations/usage_history.go:85` | Daily-sparkline coverage call sets Granularity together with GroupBy, which the API rejects | +| A07-012 | high | money-path | `providers/aws/ladder/layer_states.go:307` | EC2 ladder coverage blends in ElastiCache, OpenSearch, Redshift and MemoryDB pools | +| A08-004 | high | correctness | `providers/azure/services/compute/client.go:914` | Azure VM SKU enrichment is dead on the recommendations path and burns a full SKU-catalogue walk per subscription | +| A08-005 | high | money-path | `providers/azure/services/compute/client.go:765` | Azure retail-price extraction is last-item-wins across Spot/Windows/Linux SKUs and silently mixes currencies | +| A08-006 | high | money-path | `providers/azure/internal/recommendations/converter.go:423` | ExpandPaymentVariants fabricates 100% savings when the provider omitted the commitment cost | +| A08-007 | high | money-path | `providers/azure/services/savingsplans/client.go:261` | Azure Savings Plan purchases hardcode CurrencyCode "USD" on the commitment body | +| A08-008 | high | money-path | `providers/azure/internal/recommendations/converter.go:360` | Modern (MCA) recommendation extraction discards the currency Azure reported | +| A08-009 | high | silent-fallback | `providers/gcp/services/computeengine/client.go:876` | GCP offering details silently bill an unrecognized payment option as all-upfront | +| A08b-002 | high | security | `pkg/httpclient/httpclient.go:31` | IMDS denylist covers two addresses; Azure WireServer and the rest of link-local are reachable | +| A08b-004 | high | security | `providers/azure/services/cache/client.go:103` | Four Azure clients accept a nil HTTP client in `NewClientWithHTTP` with no hardened fallback | +| A08b-006 | high | money-path | `providers/azure/services/cache/client.go:251` | Azure commitments fabricate `State: "active"` and `Region` from fields the SDK response does not carry | +| A08b-007 | high | money-path | `providers/azure/services/cache/client.go:260` | Azure commitments never populate Count, StartDate, EndDate or Cost | +| A08b-008 | high | money-path | `providers/azure/services/cosmosdb/client.go:532` | Cosmos DB pricing ignores the SKU it was asked to price | +| A08b-009 | high | money-path | `providers/azure/services/search/client.go:447` | Azure Search pricing ignores the SKU it was asked to price | +| A08b-010 | high | money-path | `providers/azure/services/cache/client.go:599` | Retail-price extraction is last-item-wins with no SKU or unit-of-measure check | +| A08b-014 | high | money-path | `providers/gcp/services/cloudstorage/client.go:356` | Cloud Storage prices a per-GiB-month SKU as if it were per-hour, inflating cost by 730x | +| A08b-016 | high | silent-fallback | `providers/gcp/services/cloudsql/client.go:507` | GCP pricing failures are logged and the recommendation is emitted with zero cost alongside non-zero savings | +| A08b-017 | high | correctness | `providers/gcp/services/memorystore/client.go:456` | GCP `ResourceType` is a resource instance name, then used as a pricing tier and validated against tier constants | +| A08b-022 | high | silent-fallback | `providers/azure/services/cache/client.go:185` | Four Azure clients report "no existing commitments" when the reservations pager cannot be built | +| A08b-024 | high | money-path | `providers/azure/services/managedredis/client.go:41` | Azure Cache for Redis is enumerated by two clients, so its reservations and recommendations are counted twice | +| A08b-029 | high | correctness | `providers/gcp/services/computeengine/client.go:524` | Compute Engine existing commitments report `ResourceType` as "VCPU" instead of a machine type | +| A08b-030 | high | money-path | `providers/gcp/services/computeengine/client.go:1296` | The vCPU amount is selected by "not memory", so an accelerator or local-SSD amount is read as the vCPU count | +| A08b-032 | high | money-path | `providers/gcp/services/computeengine/client.go:1415` | An empty idempotency token silently produces a non-idempotent, second-granularity commitment name | +| A09-001 | high | security | `pkg/httpclient/httpclient.go:41` | IMDS blocklist is bypassed by any hostname that resolves to the metadata address | +| A09-002 | high | security | `pkg/httpclient/httpclient.go:31` | IMDS blocklist misses the ECS/EKS credential endpoints and the rest of 169.254.0.0/16 | +| A09-005 | high | money-path | `pkg/exchange/exchange.go:313` | The USD spend cap is compared against a quote whose currency is never checked | +| A09-008 | high | money-path | `pkg/exchange/auto.go:319` | The per-exchange cap is skipped entirely when the quote reports no PaymentDue | +| A10-001 | high | money-path | `cmd/multi_service_engine_versions.go:486` | Extended-support exclusion cuts Count without scaling the row's money fields | +| A10-002 | high | money-path | `cmd/multi_service_helpers.go:589` | A failed duplicate check falls back to purchasing the full, un-deduplicated counts | +| A10-003 | high | money-path | `cmd/multi_service.go:63` | `--target-coverage` sizes as if nothing is owned when the Cost Explorer coverage fetch fails | +| A10-004 | high | money-path | `cmd/helpers.go:91` | A failed account lookup substitutes the account ID, silently defeating `--exclude-accounts` | +| A10-018 | high | security | `cmd/configure_gcp.go:686` | The GCP setup wizard grants `roles/compute.admin` at project scope | +| A11-002 | high | security | `frontend/src/groups/groupModals.ts:112` | Removing every permission from a group reports success while the backend keeps the old permissions | +| A12-004 | high | money-path | `frontend/src/recommendations.ts:3919` | Capacity-% scaling is count-based, so every Savings Plans rec is silently dropped below 100% | +| A12-005 | high | money-path | `frontend/src/history.ts:463` | `scheduled` executions render the green "Completed" badge | +| A12-006 | high | money-path | `frontend/src/history.ts:1463` | Marketplace consent dialog computes the list price with `Math.round` while the backend uses `math.Floor` | +| A12-008 | high | correctness | `frontend/src/plans.ts:2326` | The AWS "Savings Plans" optgroup is hidden and disabled whenever a provider is selected | +| A12-009 | high | money-path | `frontend/src/recommendations.ts:5013` | "Execute Now" warning shows an upfront total that goes stale when rows are toggled | +| A12-010 | high | money-path | `frontend/src/riexchange.ts:1609` | Changing the exchange targets after a quote does not invalidate it | +| A12-011 | high | concurrency | `frontend/src/riexchange.ts:318` | A pending AWS utilization response re-renders the shared container over the Azure/GCP table | +| A12-074 | high | money-path | `frontend/src/recommendations.ts:3115` | The Savings Plans group row scales its savings by the cost period twice | +| A13-001 | high | money-path | `terraform/modules/compute/aws/lambda/main.tf:381` | `ce:GetCostAndUsage` is granted in no IaC flavor, but the ladder baseline calls it | +| A13-002 | high | money-path | `terraform/modules/compute/aws/lambda/main.tf:352` | RI Marketplace listing actions are granted nowhere, so the sell flow 403s | +| A13-004 | high | security | `terraform/environments/aws/ci-cd-permissions/policy_networking.tf:166` | The deploy role holds unconditioned account-wide KMS Decrypt, Encrypt and CreateGrant | +| A13-005 | high | security | `terraform/environments/azure/build.tf:24` | The ACR admin password is de-sensitized into a non-sensitive module variable | +| A13b-001 | high | correctness | `iac/federation/gcp-sa-impersonation/terraform/main.tf:27` | GCP SA-impersonation module grants two roles that cannot bind at project scope and never grants the CUD-purchase permission | +| A13b-002 | high | correctness | `iac/federation/azure-target/bicep/azure-wif.bicep:24` | Azure Bicep/ARM assigns the built-in Reservation Purchaser role that the repo documents as insufficient for purchases | +| A13b-004 | high | correctness | `cloudformation/stacks/CUDly-CrossAccount/template.yaml:64` | Cross-account trust principal omits the `/lambda/` path the hub role actually carries | +| A13c-002 | high | ops | `terraform/modules/compute/aws/lambda/main.tf:506` | Lambda log group uses `name_prefix`, so Lambda's real log group is unmanaged — retention never applies and the migration alarm can never fire | +| A13c-005 | high | security | `terraform/environments/azure/registry.tf:10` | Azure ACR admin account is enabled and its credentials are written into the Container App and Terraform state, despite an AcrPull grant existing alongside | +| A14-002 | high | ops | `.github/workflows/rollback.yml:49` | The rollback workflow pins a Terraform version the modules reject | +| A14-003 | high | ops | `.github/workflows/database-migration.yml:296` | database-migration.yml runs `terraform init` with no backend config, so it can never read a DB endpoint | +| A14-004 | high | security | `.github/workflows/cleanup-staging.yml:401` | GCP staging cleanup deletes a Cloud SQL instance chosen by a substring filter, then orphans it in state | +| A14-005 | high | correctness | `scripts/entrypoint.sh:72` | `set -e` makes the entrypoint's migration exit-code handling unreachable | +| A14-006 | high | security | `.github/workflows/README.md:446` | The workflows README tells operators to provision long-lived cloud credentials the workflows never use | +| A14-008 | high | security | `.pre-commit-config.yaml:196` | The only gating Trivy IaC scan skips the AWS Terraform environment | +| A14-010 | high | security | `scripts/setup-git-secrets.sh:104` | git-secrets allowed patterns are content regexes, so `resource `, `var.`, `data.` whitelist most of the repo | +| A15-001 | high | ops | `frontend/package-lock.json:5754` | npm audit fails on the pinned commit: fast-uri 3.1.5 carries four high-severity advisories | +| A16-001 | high | test-gap | `internal/purchase/approvals.go:284` | 4-eyes: the "creator's account could not be resolved" fail-closed branch has no test | +| A16-004 | high | test-gap | `internal/purchase/scheduled_fire_test.go:157` | The scheduled-purchase fire success path is behind a t.Skip, and its stand-in guard only checks a signature | +| A16-005 | high | test-gap | `internal/server/handler_test.go:142` | FinalizeInFlightRevocations has zero coverage; its only "test" asserts a stub's own return value | +| A16-006 | high | test-gap | `internal/purchase/execution.go:269` | The AUDIT LOSS branch in the per-account fan-out is untested, and it is the one that decides SQS ack vs redelivery | +| A16-007 | high | test-gap | `internal/scheduler/scheduler.go:912` | The partial-sweep eviction guard is proven for Azure only; the GCP per-account collector duplicating it has no test | +| A16-008 | high | test-gap | `internal/api/handler_purchases.go:147` | getPlannedPurchases' account-scope filter is never exercised; the only test runs as an unrestricted admin | +| A16-009 | high | test-gap | `internal/api/handler_purchases_revoke.go:320` | authorizeSessionRevokeExecution's allow and deny boundaries are both untested | + +92 of the 500 actionable findings are critical or high. + +## Findings + +Grouped by category, then by verified severity, then by shard. Each block is reproduced exactly +as the reviewer wrote it and the verifier annotated it. The `- issue:` line is the only addition. + +### Category: security + +78 findings: 2 critical, 21 high, 31 medium, 24 low. + +### A03-001 Resource wildcard `execute:*` bypasses the admin money-verb carve-out at both the grant ceiling and enforcement +- category: security +- severity: critical +- location: internal/auth/group_ceiling.go:85 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: The carve-out set `adminCarvedOuts` is keyed on exact (action, resource) pairs, but enforcement (`checkPermissionMatch`, service_group.go:375-383; `AuthContext.HasPermission`, types.go:159-166) treats a stored `Resource == "*"` as matching every resource. An admin (holding only `admin:*`) sends `POST /api/groups` with permissions `[{execute,*},{approve-any,*},{retry-any,*}]`: the ceiling looks up `adminCarvedOuts[{execute,"*"}]` (false), then `grantCeilingAllows` returns true because the admin branch only refuses carved-out pairs. The group is written. Any member now passes `permissionsAllow(execute, purchases)`, `(approve-any, purchases)`, `(retry-any, purchases)` and `(execute, ri-exchange)`. Reproduced: `CreateGroupAPI(execute:*, approve-any:*, retry-any:*) by admin -> err=`; `HasPermission(execute,purchases)=true`, `HasPermission(execute,ri-exchange)=true` with `[{admin,*},{execute,*}]`, while the baseline `admin:*` alone gives false. Issues #923/#1550/#1644 separation of duties is void for every admin. +- evidence: + ```go + for _, req := range requested { + if adminCarvedOuts[[2]string{req.Action, req.Resource}] { + if permissionCoveredBy(existing, req) { + continue + } + return fmt.Errorf( + "%w: %s:%s is reserved for separation of duties (issue #923) and cannot be granted through the API", + ErrPermissionNotGrantable, req.Action, req.Resource) + } + if !grantCeilingAllows(actorPerms, req) { + ``` +- suggested fix: Make the carve-out test resource-aware in one shared helper (`isCarvedOut(action, resource)` returning true when the action is a carved-out action and the resource is the carved-out resource OR `ResourceAll`), and use it in `checkGrantCeiling`, `grantCeilingAllows`, `permissionsAllow`, `AuthContext.HasPermission` and `firstUnheldCarvedOut`; additionally refuse `Resource == "*"` for the carved-out actions in `validateRequestedPermissions` so a wildcard can never be stored for them. +- verdict: CONFIRMED — traced POST /api/groups (internal/api/handler_groups.go:32-53, gated only on create:groups) into checkGrantCeiling, where the exact-pair lookup at internal/auth/group_ceiling.go:85 misses {execute,"*"}, grantCeilingAllows:348-356 then returns true on the admin branch, validateRequestedPermissions:128-145 only rejects blanks, and at enforcement checkPermissionMatch (service_group.go:379) and AuthContext.HasPermission (types.go:162) accept a stored "*" resource for execute:purchases and execute:ri-exchange; the session execute path (handler_purchases.go:2128 -> HasPermissionForConstraintsAPI -> permissionsAllow) has no second check. +- issue: (pending cross-reference) + +### A08b-001 IMDS blocklist matches literal host text, so IPv6-mapped, alternate-radix and DNS forms of 169.254.169.254 pass through +- category: security +- severity: critical +- location: pkg/httpclient/httpclient.go:41 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd (entry point `providers/azure/internal/httpclient/httpclient.go:18`) +- failure scenario: `blockIMDSDialer.DialContext` receives the address string built from the URL host, before name resolution, and compares it to a two-entry map. A URL of `http://[::ffff:169.254.169.254]/metadata/identity/oauth2/token` yields host `::ffff:169.254.169.254`, which is not a map key, so the dial proceeds and connects to the IPv4 metadata endpoint. The same holds for `http://2852039166/`, `http://0xa9fea9fe/`, `http://0251.0376.0251.0376/` (getaddrinfo accepts all three on the cgo resolver), and for any attacker-controlled DNS name whose A record is 169.254.169.254, including a name that resolves benignly on the first lookup and to the metadata IP on the second (DNS rebinding). Every Azure client reaches this dialer with a managed-identity-bearing process, so a redirect or attacker-influenced pricing URL exfiltrates the subscription's credentials. +- evidence: + ```go + func (d *blockIMDSDialer) DialContext(ctx context.Context, network, addr string) (net.Conn, error) { + host, _, err := net.SplitHostPort(addr) + if err != nil { + host = addr + } + if imdsAddresses[host] { + return nil, fmt.Errorf("connection to metadata endpoint %s is blocked", host) + } + return d.inner.DialContext(ctx, network, addr) + } + ``` +- suggested fix: resolve the host first (`net.DefaultResolver.LookupIPAddr`), then reject if any resolved `netip.Addr` (after `Unmap()`) falls in a denied set, and dial only the vetted IPs so the resolution that was checked is the resolution that is used. +- verdict: CONFIRMED — `DialContext` compares the pre-resolution host string from the URL against a two-key map (pkg/httpclient/httpclient.go:41-49), so `::ffff:169.254.169.254` and any DNS name whose A record is the metadata IP dial straight through; the Azure package delegates to this exact function (providers/azure/internal/httpclient/httpclient.go:17-19). +- issue: (pending cross-reference) + +### A01-002 Session-authed approve, cancel, retry and revoke never check the session's allowed_accounts scope +- category: security +- severity: high +- location: internal/api/handler_purchases.go:685 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: A user in a group with `approve-any:purchases` (or `retry-any`, `cancel-any`, `revoke-any`) and `allowed_accounts=[acct-A]` calls `POST /api/purchases/approve/{id}` for a pending execution whose recs target acct-B. `approvePurchaseViaSession` runs only `authorizeSessionApprove` (verb + creator match) and `requireDifferentApprover`; `requireExecutionAccess`/`validatePurchaseRecommendationScope` are called only by pause/resume/run/delete (lines 263, 287, 311, 411). The purchase in acct-B executes. Same gap in `loadAndValidateRetryRequest` (1699, mints a new approvable execution for the out-of-scope account), `cancelPurchaseViaSession` (1148), `tryRevokeViaSession` (1329) and `revokeScheduledExecution` (handler_purchases_revoke.go:249). The per-permission `AccountIDs` constraint does not cover this axis (see the rationale on `requireAzureSubscriptionScope`, handler_ri_exchange.go:779-788). No test in `handler_purchases_test.go` registers `GetAllowedAccountsAPI` for these paths, so the suite passes with the gap present. +- evidence: + ```go + if execution.Status != "pending" && execution.Status != "notified" { + return nil, NewClientError(409, ...) + } + + if err := h.authorizeSessionApprove(ctx, session, execution); err != nil { + return nil, err + } + if err := h.requireDifferentApprover(ctx, session, execution); err != nil { + return nil, err + } + ``` +- suggested fix: Call `h.requireExecutionAccess(ctx, session, execution.ExecutionID)` (or a scope check over `execution.Recommendations` for plan-less rows) right after RBAC in `approvePurchaseViaSession`, `cancelPurchaseViaSession`, `loadAndValidateRetryRequest`, `tryRevokeViaSession` and `revokeScheduledExecution`, and add one scoped-session test per path asserting the mutation is not called. +- verdict: CONFIRMED — requireExecutionAccess is called only at internal/api/handler_purchases.go:263,287,311,411 (pause/resume/run/delete); approvePurchaseViaSession:665-735, cancelPurchaseViaSession:1131-1183, loadAndValidateRetryRequest:1672-1719, tryRevokeViaSession:1324-1346 and revokeScheduledExecution (internal/api/handler_purchases_revoke.go:240-300) run only verb/creator RBAC (authorizeSession*), and the only tests registering GetAllowedAccountsAPI in handler_purchases_test.go are for getPlannedPurchases/pause/execute/deletePlanned (lines 1881, 2433, 3684-4354), none for approve/cancel/retry/revoke. +- issue: (pending cross-reference) + +### A02-001 Dashboard commitment KPIs skip allowed_accounts scoping when the caller supplies an explicit account filter +- category: security +- severity: high +- location: internal/api/handler_dashboard.go:38 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: A session whose group allowed_accounts is ["Production"] calls `GET /api/dashboard/summary?account_id=123456789012` where 123456789012 is the AWS account number of the out-of-scope "Finance" account (or `account_ids=`). `resolveSingleAccountFilterIDs` (scoping.go:277-279) maps the unknown value to an external-id filter under the "" provider key, the non-empty result skips the `resolveAllowedAccountScope` fallback, and `calculateCommitmentMetrics` -> `fetchCommitmentPurchases` -> `GetActivePurchaseHistory` (store_postgres.go:2112-2114, `account_id = ANY(...)`) aggregates Finance's rows. The response returns Finance's `active_commitments`, `committed_monthly`, `ytd_savings` and per-service `current_savings`. Only the recommendations half is scope-filtered (`filterDashboardRecommendations`); the purchase-history half has no equivalent, unlike `/api/inventory/*` which runs `filterPurchaseHistoryByAllowedAccounts` after the same fetch (handler_inventory.go:47, 197) and `/api/history/analytics` which validates the requested account against the scope first (handler_analytics.go:233-249). +- evidence: + ```go + accountUUIDs, accountExternalIDsByProvider, err := h.resolveDashboardAccountScope(ctx, params) + ... + if len(accountUUIDs) == 0 && len(accountExternalIDsByProvider) == 0 { + accountUUIDs, accountExternalIDsByProvider, err = h.resolveAllowedAccountScope(ctx, session) + ... + } + ... + activeCommitments, committedMonthly, ytdSavings, currentSavingsByService := h.calculateCommitmentMetrics(ctx, params["provider"], accountUUIDs, accountExternalIDsByProvider) + ``` +- suggested fix: When the session is scoped, validate every explicitly requested account (UUID or external id) with `validateAnalyticsAccountScope`-style `AccountScope.Allows` before `calculateCommitmentMetrics`, or intersect the explicit filter with `resolveAllowedAccountScope` and short-circuit to zeroed KPIs on an empty intersection. Add a test with `grantScoped` + an out-of-scope `account_id`. +- verdict: CONFIRMED — handler_dashboard.go:38-42 skips resolveAllowedAccountScope as soon as resolveSingleAccountFilterIDs (scoping.go:277-279) returns a non-empty external-id set for the unknown value, and calculateCommitmentMetrics/fetchCommitmentPurchases (handler_dashboard.go:625-680) pass it straight to GetActivePurchaseHistory (store_postgres.go:2050-2058) with no AccountScope.Allows check; every getDashboardSummary test (handler_dashboard_test.go:79,142,186,1212-1396) omits account_id/account_ids. +- issue: (pending cross-reference) + +### A03-002 An admin can mint a user API key carrying `execute:*` in a single request and spend with it +- category: security +- severity: high +- location: internal/auth/service_apikeys.go:128 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: `validateAPIKeyPermissions` checks each requested key permission with `authCtx.HasPermission(perm.Action, perm.Resource)`. For an admin, `HasPermission("execute", "*")` returns true (the carve-out lookup is `adminCarvedOuts[{execute,"*"}]`, false). The key is stored with `[{execute,*}]`; `computeEffectivePermissionsFromAuthCtx` keeps it for the same reason, and `HasAPIKeyPermissionAPI` then answers `HasPermission(execute, purchases) == true`. No group edit, no second actor, one request. Reproduced: `CreateAPIKey(execute:*) -> err=`; `effective=[{execute *}] HasPermission(execute,purchases)=true`. Same root cause as A03-001 through a different guard. +- evidence: + ```go + for _, perm := range permissions { + if !authCtx.HasPermission(perm.Action, perm.Resource) { + return fmt.Errorf("user does not have permission for action=%s resource=%s", perm.Action, perm.Resource) + } + } + ``` +- suggested fix: Route key-permission validation through the same carve-out-aware helper as A03-001, and reject `Resource == "*"` on carved-out actions when a key is created; a key should never be able to hold a money verb the owner does not hold explicitly. +- verdict: CONFIRMED — validateAPIKeyPermissions (service_apikeys.go:128-131) accepts {execute,"*"} for an admin because AuthContext.HasPermission (types.go:150-155) only carves out the exact pair, computeEffectivePermissionsFromAuthCtx:462-473 keeps it for the same reason, and HasAPIKeyPermissionAPI (service_apikeys_api.go:363-364) then grants execute:purchases; the direct-execute path is partly narrowed because HasAPIKeyPermissionForConstraintsAPI:411 also requires the OWNER's group permissions to allow execute:purchases, but runPlannedPurchase (handler_purchases.go:307), approve-any and retry-any gates call requirePermission alone, so the key still spends. +- issue: (pending cross-reference) + +### A03-003 Self-membership guard checks carved-out pairs exactly, so an admin can join a wildcard group they created +- category: security +- severity: high +- location: internal/auth/service_user.go:551 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: After A03-001 stores a group with `{execute,*}`, the admin edits their own membership to add it. `guardSelfEscalation` passes (update:users held), `guardSelfCarvedOutGrant` calls `firstUnheldCarvedOut`, which skips every permission whose exact pair is not in `adminCarvedOuts`; `{execute,*}` is skipped, the change is written, and `HasPermissionAPI(admin, execute, purchases)` becomes true. Reproduced: `self-add to execute:* group -> err=`; `HasPermissionAPI(admin, execute, purchases) after self-add = true`. Two requests from one compromised admin account drain commitments, which is exactly what types.go:123 says cannot happen. +- evidence: + ```go + for i, perm := range group.Permissions { + if !adminCarvedOuts[[2]string{perm.Action, perm.Resource}] { + continue + } + if permissionsAllow(held, perm.Action, perm.Resource, nil) { + continue + } + return &group.Permissions[i] + } + ``` +- suggested fix: Use the shared carve-out-aware helper from A03-001 here as well. +- verdict: CONFIRMED — PUT /api/users/{id} (handler_users.go:123-137) reaches guardSelfEscalation (service_user.go:440-452), which passes update:users for admin:* and then guardSelfCarvedOutGrant -> firstUnheldCarvedOut, whose exact-pair test at service_user.go:551 skips a group permission {execute,"*"} and returns nil, so the self-add is written and permissionsAllow later matches the wildcard resource (service_group.go:379). +- issue: (pending cross-reference) + +### A03-004 User-membership writes have no grant ceiling: any create:users or update:users holder can mint an Administrator or Purchaser +- category: security +- severity: high +- location: internal/auth/service_user.go:396 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: `guardGroupChange` returns nil for every non-self edit, and `validateCreateUserRequest` (service_user.go:122-148) never sees an actor at all; the API layer gates `POST /api/users` on `create:users` and `PUT /api/users/{id}` on `update:users` only (internal/api/handler_users.go:32, 123). (a) An admin, who by policy cannot add themself to Purchaser, creates a second user with `group_ids=[DefaultPurchaserGroupID]` and a password they chose, logs in as it, and executes purchases: the two-person control in `guardSelfCarvedOutGrant` is defeated by a puppet account. (b) A non-admin custom group holding only `update:users` (grantable by any admin through the ceiling) can move any user, including a second account it controls, into the Administrators group; the group-permission ceiling (#1550) is bypassed via membership. Group *permission* writes are ceilinged; group *membership* writes are not. +- evidence: + ```go + // Internal callers (actorUserID == "") are already trusted and skip it. + ... + if actorUserID == "" || actorUserID != targetUserID { + return nil + } + return s.guardSelfEscalation(ctx, prior, next) + ``` +- suggested fix: Apply the same ceiling to membership grants as to permission grants: for a non-self edit and for create, resolve the target groups' permissions and refuse any group that carries a permission the actor does not hold (reusing `grantCeilingAllows`), and refuse assignment of any group carrying a carved-out verb through the API unless the actor already holds that verb. Thread `actorUserID` into `CreateUser`/`CreateUserAPI`. +- verdict: CONFIRMED — guardGroupChange returns nil for every non-self edit (service_user.go:396-398), CreateUserAPI passes no actor (service_api.go:202-212) and validateCreateUserRequest:122-148 checks only email/groups/password, the handlers gate on create:users / update:users alone (handler_users.go:32, :123), update:users is not in adminCarvedOuts (types.go:124-134) so an admin can grant it through the ceiling, and nothing compares the target groups' permissions to the actor's; the comment at service_user.go:514-516 explicitly accepts the non-self Purchaser add as design, but that design does not cover a puppet account or an update:users-only actor moving accounts into Administrators. +- issue: (pending cross-reference) + +### A03-005 Deactivating a user does not end their sessions, and session validation never re-checks the user +- category: security +- severity: high +- location: internal/auth/service.go:316 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: An admin sets `active=false` on a user via `UpdateUser`; `guardDeactivation` only protects the last admin and nothing calls `DeleteUserSessions` (service_user.go:319-333, the only session purges are in DeleteUser, password change and reset). `ValidateSession` checks only the sessions row (`WHERE token = $1 AND expires_at > NOW()`, store_postgres.go:670), `requireSessionPermission` (internal/api/handler.go:301-314) then resolves permissions through `GetUserPermissions`, which never reads `user.Active`, and `/usr/bin/grep -rn "\.Active" internal/api` (non-test) returns nothing. The deactivated user keeps full group-derived access through every session-authenticated endpoint, including purchase execution if they hold the verbs, for the rest of `sessionDuration` (24h default). Contrast: the API-key path (`lookupAPIKeyUser`, service_apikeys.go:272) and `CreateAPIKey` (service_apikeys.go:54) do check Active, so the two credential types behave differently on the same input. +- evidence: + ```go + session, err := s.store.GetSession(ctx, hashedToken) + if err != nil { + return nil, err + } + if session == nil { + return nil, fmt.Errorf("session not found") + } + // Constant-time comparison after fetch ... + ``` +- suggested fix: In `UpdateUser`, when `priorActive && !user.Active`, call `DeleteUserSessions` after the row write; and have the session path fail closed like the API-key path by loading the user in `ValidateSession` (or in `requireSessionPermission`) and refusing when missing or inactive. +- verdict: CONFIRMED — UpdateUser (service_user.go:300-345) never calls DeleteUserSessions (the only callers are DeleteUser:636, UpdateUserProfile:692 and the two password paths in service_password.go:255,359), ValidateSession (service.go:316-345) only reads the sessions row via GetSession (store_postgres.go:670), requireSessionPermission (handler.go:301-314) resolves permissions through HasPermissionAPI with no user load, and `/usr/bin/grep -rn "\.Active\b" internal/api` (non-test) returns nothing, while lookupAPIKeyUser (service_apikeys.go:272) does refuse inactive owners. +- issue: (pending cross-reference) + +### A03-006 A deactivated account can reactivate itself through the forgot-password flow +- category: security +- severity: high +- location: internal/auth/service_password.go:355 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: `RequestPasswordReset` has no `Active` check (service_password.go:267-333), so a reset email is sent to a deactivated user. `ConfirmPasswordReset` then flips `Active = true` unconditionally, because it cannot tell an invited user (never activated) from an admin-deactivated one. A user an admin disabled (offboarding, suspected compromise) clicks "Forgot password" and is back in, with all prior group memberships. Reproduced: `reset email sent to deactivated account: 1`; `after ConfirmPasswordReset: deactivated user Active=true`. +- evidence: + ```go + // Activate user on first password set (admin bootstrap flow) + if !user.Active { + user.Active = true + } + ``` +- suggested fix: Distinguish "invited" from "deactivated" explicitly (the invite path already sets `PasswordResetToken` at create time; record an invite marker, or treat "never had a real password / `LastLoginAt == nil`" as the invite signal) and only activate on that path; refuse `RequestPasswordReset` for deactivated, non-invited users while keeping the anti-enumeration response. +- verdict: CONFIRMED — RequestPasswordReset (service_password.go:267-333) loads the user and issues a token with no Active check, ConfirmPasswordReset:354-357 sets Active = true for any inactive user, and CreateUser (service_user.go:203-224) records nothing that distinguishes an invite (Active=false + setup token) from an admin deactivation via applyUpdateUserRequest:611-613, so the two states are indistinguishable at confirm time. +- issue: (pending cross-reference) + +### A06-003 SourceIP carries the TCP port (or the proxy's IP), defeating the login brute-force rate limit +- category: security +- severity: high +- location: internal/server/http.go:303 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: `SourceIP` is the sole rate-limit key for the credential endpoints — `checkRateLimitStrict` builds `"IP#" + req.RequestContext.HTTP.SourceIP` (internal/api/middleware.go:557, internal/api/db_rate_limiter.go:212) and login has no email-keyed limiter. On any container deployment that is not behind an `X-Forwarded-For`-setting proxy (Fargate behind an NLB, direct container, local), `r.RemoteAddr` is `"10.0.0.7:54321"`, so every new TCP connection produces a distinct key: an attacker gets unlimited password guesses and one `rate_limits` row per attempt. In the other direction, on Cloud Run the header is `client, google-lb`, so the rightmost entry is the load balancer and every client on the deployment shares a single bucket — one noisy user locks everyone out of login. The port is never stripped (no `net.SplitHostPort`), and the only test, `TestHttpToLambdaRequest_XForwardedFor` (internal/server/app_test.go:91), exercises only the header-present case. +- evidence: + ```go + sourceIP := r.RemoteAddr + if xff := r.Header.Get("X-Forwarded-For"); xff != "" { + parts := strings.Split(xff, ",") + sourceIP = strings.TrimSpace(parts[len(parts)-1]) + } + ``` +- suggested fix: strip the port with `net.SplitHostPort(r.RemoteAddr)` for the no-XFF case, and select the client entry by trusted-proxy-hop count rather than always taking the rightmost element. +- verdict: CONFIRMED — `sourceIP := r.RemoteAddr` with no `net.SplitHostPort` anywhere on the path (internal/server/http.go:303-307,315), and that value is the sole rate-limit key for the credential endpoints: `checkRateLimitStrict` passes `req.RequestContext.HTTP.SourceIP` to `AllowWithIP`, which formats `"IP#%s"` (internal/api/middleware.go:557-559, internal/api/db_rate_limiter.go:211-215); login has no email-keyed companion (internal/api/handler_auth.go:24 vs the only `AllowWithEmail` caller at handler_auth.go:268 for forgot_password). +- issue: (pending cross-reference) + +### A08b-002 IMDS denylist covers two addresses; Azure WireServer and the rest of link-local are reachable +- category: security +- severity: high +- location: pkg/httpclient/httpclient.go:31 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: this is the Azure provider's hardened client, yet the map omits `168.63.129.16`, Azure's WireServer / host-agent endpoint that serves instance configuration and extension settings, and omits the rest of `169.254.0.0/16` and `fe80::/10`. An SSRF into `http://168.63.129.16/machine?comp=goalstate` is not blocked. `metadata.google.internal` (a name, so also missed by A08b-001) and Alibaba's `100.100.100.200` are likewise reachable. +- evidence: + ```go + var imdsAddresses = map[string]bool{ + "169.254.169.254": true, // AWS/Azure/GCP link-local IMDS (IPv4) + "fd00:ec2::254": true, // AWS IMDS (IPv6) + } + ``` +- suggested fix: replace the map with a CIDR denylist (`169.254.0.0/16`, `fe80::/10`, `fd00:ec2::/64`, plus loopback and the RFC1918 ranges if outbound is meant to be internet-only) evaluated against resolved addresses. +- verdict: CONFIRMED — `imdsAddresses` at pkg/httpclient/httpclient.go:31-34 holds exactly the two literal strings quoted; there is no CIDR test anywhere in the file and no entry for 168.63.129.16, 100.100.100.200 or `metadata.google.internal`. +- issue: (pending cross-reference) + +### A08b-004 Four Azure clients accept a nil HTTP client in `NewClientWithHTTP` with no hardened fallback +- category: security +- severity: high +- location: providers/azure/services/cache/client.go:103 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd (same shape at `cosmosdb/client.go:101`, `database/client.go:125`, `search/client.go:80`) +- failure scenario: `managedredis/client.go:95` and `synapse/client.go:77` guard the nil case and fall back to `httpclient.New()`; these four do not. A caller passing nil stores a nil `HTTPClient` interface, and the first pricing or purchase call panics on `c.httpClient.Do`. Any caller that instead passes its own bare `&http.Client{}` silently loses the IMDS dialer on a path that carries an ARM bearer token (`DoIdempotentPurchaseTwoStep` at `cache/client.go:345`). +- evidence: + ```go + func NewClientWithHTTP(cred azcore.TokenCredential, subscriptionID, region string, httpClient HTTPClient) *CacheClient { + return &CacheClient{ + cred: cred, + subscriptionID: subscriptionID, + region: region, + httpClient: httpClient, + } + } + ``` +- suggested fix: add the same `if httpClient == nil { httpClient = httpclient.New() }` guard the two fixed siblings already carry, and add the regression test that asserts the nil branch blocks 169.254.169.254. +- verdict: CONFIRMED — cache/client.go:103-110, cosmosdb/client.go:101-108, database/client.go:125-132 and search/client.go:80-87 assign the parameter straight into the struct, while synapse/client.go:77-79 and managedredis/client.go:95-97 carry the nil guard the finding names. +- issue: (pending cross-reference) + +### A09-001 IMDS blocklist is bypassed by any hostname that resolves to the metadata address +- category: security +- severity: high +- location: pkg/httpclient/httpclient.go:41 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: `http.Transport` hands `DialContext` the URL host, not a resolved IP; `net.Dialer.DialContext` does the DNS lookup afterwards. With a JWKS URL of `http://metadata.attacker.example/latest/meta-data/iam/security-credentials/` whose A record is `169.254.169.254`, `net.SplitHostPort` yields `metadata.attacker.example`, the map lookup misses, and the connection to IMDS proceeds. The only control point every provider delegates to therefore blocks literal IPs and nothing else. The same hole is reachable through a 302 redirect, since redirects are followed by default and re-dial through the identical path. +- evidence: + ```go + func (d *blockIMDSDialer) DialContext(ctx context.Context, network, addr string) (net.Conn, error) { + host, _, err := net.SplitHostPort(addr) + if err != nil { + host = addr + } + if imdsAddresses[host] { + return nil, fmt.Errorf("connection to metadata endpoint %s is blocked", host) + } + return d.inner.DialContext(ctx, network, addr) + } + ``` +- suggested fix: dial with `d.inner.DialContext` then inspect `conn.RemoteAddr()` (or resolve first via `net.DefaultResolver.LookupIPAddr` and dial the vetted IP), rejecting and closing when the peer IP is link-local/metadata rather than matching on the pre-resolution host string. +- verdict: CONFIRMED — `blockIMDSDialer.DialContext` matches `imdsAddresses[host]` on the host string `http.Transport` passes it before `net.Dialer` resolves it (pkg/httpclient/httpclient.go:41-48), so a hostname with an A record of 169.254.169.254 misses the map and is dialed; no post-dial `RemoteAddr` check exists anywhere in the file. +- issue: (pending cross-reference) + +### A09-002 IMDS blocklist misses the ECS/EKS credential endpoints and the rest of 169.254.0.0/16 +- category: security +- severity: high +- location: pkg/httpclient/httpclient.go:31 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: the map holds exactly two addresses. A request to `http://169.254.170.2/v2/credentials/` (the ECS task credential provider) or `http://169.254.170.23/v1/credentials` (EKS Pod Identity) is dialed normally and returns live IAM credentials. `http://169.254.169.254./latest/...` (trailing-dot FQDN form) and `http://[fd00:ec2:0:0:0:0:0:254]/` (expanded IPv6, which `net.SplitHostPort` returns verbatim rather than in canonical form) also miss the map. The package doc claims it "blocks connections to the cloud Instance Metadata Service (IMDS) endpoints", which is stronger than what the two entries deliver. +- evidence: + ```go + var imdsAddresses = map[string]bool{ + "169.254.169.254": true, // AWS/Azure/GCP link-local IMDS (IPv4) + "fd00:ec2::254": true, // AWS IMDS (IPv6) + } + ``` +- suggested fix: replace the string map with `net.ParseIP` plus CIDR checks over `169.254.0.0/16`, `fe80::/10` and `fd00:ec2::/32`, applied to the resolved peer address (see A09-001) so every link-local credential endpoint is covered by construction. +- verdict: CONFIRMED — the map at pkg/httpclient/httpclient.go:31-34 holds exactly the two literals quoted; there is no CIDR check, no `net.ParseIP` normalization and no trailing-dot handling in the file, so 169.254.170.2 and 169.254.170.23 pass straight to `d.inner.DialContext` at line 47. +- issue: (pending cross-reference) + +### A10-018 The GCP setup wizard grants `roles/compute.admin` at project scope +- category: security +- severity: high +- location: cmd/configure_gcp.go:686 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: Step 4 of `cudly configure-gcp` grants the CUDly service account `roles/compute.admin` on the operator's project, then Step 5 mints a long-lived JSON key for it and Step 6 uploads that key to AWS Secrets Manager. `compute.admin` confers full control of every Compute Engine resource: creating and deleting VMs, disks, images, firewall rules and networks. CUDly needs to read usage and purchase committed use discounts. Anyone who obtains the stored key can delete the project's production infrastructure. The project's own IAM policy in CLAUDE.md names this exact role as the anti-pattern ("Prefer custom roles ... over broad predefined roles like `roles/compute.admin`"), and the prompt defaults to Run on empty input. +- evidence: + ```go + member := fmt.Sprintf("serviceAccount:%s", saEmail) + role := "roles/compute.admin" + ... + fmt.Printf("[R]un, [S]kip? (grants %s to %s on project %s via SDK) ", role, saEmail, projectID) + ``` +- suggested fix: Grant the narrowest predefined pair the CUD flow needs (`roles/compute.viewer` plus the commitment-purchase permissions) or provision a `google_project_iam_custom_role` holding only `compute.commitments.*` and the usage reads, matching the runtime-permissions rule in CLAUDE.md. +- verdict: CONFIRMED — gcpStepGrantRole hardcodes roles/compute.admin at project scope (cmd/configure_gcp.go:684-710) and its `case "r", "run", "":` arm at :700 makes empty input Run, and the identity it is granted to gets a JSON key minted at :722 and pushed to Secrets Manager at cmd/configure_gcp.go:162. +- issue: (pending cross-reference) + +### A11-002 Removing every permission from a group reports success while the backend keeps the old permissions +- category: security +- severity: high +- location: frontend/src/groups/groupModals.ts:112 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: `collectPermissions()` skips any row whose action or resource is empty (groupModals.ts:418), so an operator who clicks Remove on every permission row and saves sends `permissions: []`. The backend's `applyUpdateGroupRequest` only assigns when the list is non-empty (`internal/auth/service_api.go:330 — if len(perms) > 0`), so the stored permissions are left untouched. The UI then shows "Group updated successfully" and reloads a group that still carries, say, `admin:*`. An admin who believes they have just stripped a group's privileges has not. The same shape applies to a group whose only remaining row is left on the "Select Action" placeholder. +- evidence: + ```typescript + // groups/groupModals.ts:107-117 + const permissions = collectPermissions(); + try { + if (currentEditingGroup) { + await api.updateGroup(currentEditingGroup.id, { name, description, permissions }); + showSuccess('Group updated successfully'); + ``` +- suggested fix: refuse to submit an empty permission list from the form with an explicit message ("a group must grant at least one permission; delete the group instead"), since the API cannot express "clear all permissions". +- verdict: CONFIRMED — collectPermissions skips rows with an empty action or resource (frontend/src/groups/groupModals.ts:417-418) and saveGroup sends the result unconditionally (groupModals.ts:107-117), while the backend's `if len(perms) > 0` at internal/auth/service_api.go:330 documents empty as "not sent" and leaves `group.Permissions` untouched, so the success toast at groupModals.ts:113 reports a strip that never happened; note the UI path is currently masked by A11-001, which never binds the submit handler. +- issue: (pending cross-reference) + +### A13-004 The deploy role holds unconditioned account-wide KMS Decrypt, Encrypt and CreateGrant +- category: security +- severity: high +- location: terraform/environments/aws/ci-cd-permissions/policy_networking.tf:166 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: A leaked GitHub Actions OIDC token assumes `cudly-terraform-deploy` and calls `kms:Decrypt` against any CMK in the account, including keys belonging to unrelated workloads that share it, and `kms:CreateGrant` to hand `Decrypt` on any CMK to an arbitrary grantee principal it controls, which survives the deploy role being revoked. This is the one statement in the whole `ci-cd-permissions/` directory with no comment justifying `Resource = "*"`, and it directly contradicts the sibling policies: `KMSReadTaggedOnly` (policy_compute_b.tf:160) gates the far weaker `kms:GetKeyPolicy` on `Project=CUDly` explicitly to stop "account-wide key-policy reconnaissance". All four actions support `aws:ResourceTag`. +- evidence: + ```hcl + Sid = "KMS" + Effect = "Allow" + Action = [ + "kms:CreateGrant", + "kms:Decrypt", + "kms:DescribeKey", + "kms:Encrypt", + "kms:GenerateDataKey", + ] + Resource = "*" + ``` +- suggested fix: Add the same `StringEqualsIgnoreCase` on `aws:ResourceTag/Project` used by `KMSMutateTaggedOnly`, keeping only `kms:DescribeKey` unconditioned if a plan-time lookup needs it. +- verdict: CONFIRMED — policy_networking.tf:164-175 is the `KMS` Sid with `Resource = "*"` and no `Condition` block, while every other KMS grant in the directory is gated: `KMSAliasMutate` on an ARN prefix (policy_compute_b.tf:135-141), `KMSReadTaggedOnly` and `KMSTagOnCreate` on `aws:ResourceTag`/`aws:RequestTag` Project (policy_compute_b.tf:160-190), and `KMSMutateTaggedOnly` in policy_compute.tf:347-375. Since AWS CMKs carry the default key policy that delegates to IAM, an unconditioned `kms:Decrypt`/`kms:CreateGrant` here really does reach unrelated keys in the account. The finding's aside that this is the only `Resource = "*"` statement lacking a comment is inaccurate (ACM at :139 and Route53 at :152 are also uncommented `"*"`), but that does not affect the defect. +- issue: (pending cross-reference) + +### A13-005 The ACR admin password is de-sensitized into a non-sensitive module variable +- category: security +- severity: high +- location: terraform/environments/azure/build.tf:24 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: `nonsensitive()` strips the provider's sensitivity marker from `azurerm_container_registry.main.admin_password`, and the receiving variable `registry_login_command` (terraform/modules/build/variables.tf:43) carries no `sensitive = true`. The registry admin password, a long-lived push credential for the image the production Container App runs, is therefore stored unredacted in the module input and rendered in plain text wherever Terraform prints that value, including CI plan output. The adjacent `terraform/environments/azure/secrets.tf:42` carries an explicit "do NOT wrap this merge in nonsensitive()" warning for exactly this reason. +- evidence: + ```hcl + registry_login_command = "echo '${nonsensitive(azurerm_container_registry.main.admin_password)}' | docker login ${azurerm_container_registry.main.login_server} -u ${azurerm_container_registry.main.admin_username} --password-stdin" + ``` +- suggested fix: Mark `registry_login_command` `sensitive = true` in `terraform/modules/build/variables.tf` and drop the `nonsensitive()` wrapper; better, use `az acr login` with the deploy identity so no admin credential is materialized at all. +- verdict: CONFIRMED — terraform/environments/azure/build.tf:24 wraps `azurerm_container_registry.main.admin_password` in `nonsensitive()` and terraform/modules/build/variables.tf:43-46 declares the receiving variable with no `sensitive = true`, so the marker is gone for the whole downstream path. The value is interpolated into the `local-exec` command body at terraform/modules/build/main.tf:74, which terraform echoes at apply time and persists in state, and the registry is `admin_enabled = true` (terraform/environments/azure/registry.tf:10) so this is a live long-lived push credential. The contrast the finding draws is real: terraform/environments/azure/secrets.tf:42-44 carries an explicit "do NOT wrap this merge in nonsensitive()" warning for the same class of value. +- issue: (pending cross-reference) + +### A13c-005 Azure ACR admin account is enabled and its credentials are written into the Container App and Terraform state, despite an AcrPull grant existing alongside +- category: security +- severity: high +- location: terraform/environments/azure/registry.tf:10 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: `admin_enabled = true` activates a shared registry account with push as well + as pull. `environments/azure/compute.tf:102-103` reads `admin_username` / `admin_password` off + the resource and hands them to the container-apps module, which materialises them as a + `secret` block on `azurerm_container_app.main` (container-apps/main.tf:224-230). The password + therefore lands in Terraform state and in the ARM resource definition, readable by anyone with + Reader on the resource group. Line 16 of the same file already grants the container-app managed + identity `AcrPull`, so the workload can pull without any credential at all — the admin path is + redundant and strictly wider. The reusable `modules/registry/azure` defaults + `enable_admin_user = false`; the environment declares its own registry resource and overrides. +- evidence: + ```hcl + resource "azurerm_container_registry" "main" { + name = local.acr_name + sku = "Basic" + admin_enabled = true # Enables username/password login for docker push + } + ``` +- suggested fix: set `admin_enabled = false`, drop the three `registry_*` inputs at + compute.tf:101-103, and give the Container App `registry { identity = }` so the pull + runs on the existing AcrPull assignment. +- verdict: CONFIRMED — `admin_enabled = true` is set directly on the resource, not merely defaulted + (environments/azure/registry.tf:10), so the `admin_username`/`admin_password` attributes are + populated; compute.tf:102-103 reads them into the module, which emits them as a `secret` block on + `azurerm_container_app.main` (container-apps/main.tf:224-230), putting the password in state on + both the registry resource and the container app. The redundant AcrPull grant to the same identity + is at registry.tf:16-21, and modules/registry/azure/variables.tf:37 does default the flag false. +- issue: (pending cross-reference) + +### A14-004 GCP staging cleanup deletes a Cloud SQL instance chosen by a substring filter, then orphans it in state +- category: security +- severity: high +- location: .github/workflows/cleanup-staging.yml:401 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: `gcloud sql instances list --filter="name:cudly-staging"` uses gcloud's `:` operator, which is a contains match, not equality. With `cudly-staging-prod-mirror` and `cudly-staging-5e4d3c2b` both present, `head -1` picks whichever sorts first and `gcloud sql instances delete --quiet` destroys it. This is exactly the over-match that `scripts/select-owned-name.sh` was written to remove for ECR (#1592/#1820) and RDS (#1821); the GCP path never got the guard, and the sweeps in `test-ecr-delete-selection.sh` / `test-rds-deletion-protection-scope.sh` key on `aws ecr delete-repository` and `aws rds modify-db-instance`, so they never look at `gcloud sql instances delete`. The `|| true` on line 405 then hides a failed delete, and lines 419-421 `terraform state rm` the instance regardless, leaving a running, billing, untracked Cloud SQL instance. The step validates `PROJECT` carefully (line 396) precisely because a silent skip is unacceptable, but applies no equivalent check to `INSTANCE`. +- evidence: + ```bash + INSTANCE=$(gcloud sql instances list --project="$PROJECT" \ + --filter="name:cudly-staging" --format="value(name)" 2>/dev/null | head -1) + if [ -n "$INSTANCE" ]; then + gcloud sql instances delete "$INSTANCE" --project="$PROJECT" --quiet || true + ``` + ```bash + terraform state rm google_sql_database_instance.main 2>/dev/null || true + ``` +- suggested fix: Read the owned instance name from `terraform output -json` and pipe the `gcloud sql instances list` output through `scripts/select-owned-name.sh`, matching the ECR/RDS pattern; drop the `|| true` on the delete. +- verdict: CONFIRMED — cleanup-staging.yml:401-405 selects by `--filter="name:cudly-staging" | head -1` and deletes with `|| true`, and lines 419-421 `terraform state rm` unconditionally outside the `if [ -n "$INSTANCE" ]` block; scripts/select-owned-name.sh is wired only into force-delete-owned-ecr-repo.sh:80 and disable-owned-rds-deletion-protection.sh:143, and the two sweeps key on `aws ecr delete-repository` (test-ecr-delete-selection.sh:278) and `aws rds modify-db-instance` (test-rds-deletion-protection-scope.sh:35), never on gcloud. +- issue: (pending cross-reference) + +### A14-006 The workflows README tells operators to provision long-lived cloud credentials the workflows never use +- category: security +- severity: high +- location: .github/workflows/README.md:446 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: An operator following the Setup Guide creates and stores three sets of static credentials: AWS access key + secret (446-448), a downloaded GCP service-account JSON key (463-467), and an Azure service principal client secret via `az ad sp create-for-rbac --sdk-auth` (476-479). None of them is read by any workflow: AWS uses OIDC via `vars.AWS_ROLE_TO_ASSUME`, GCP uses Workload Identity Federation via `vars.GCP_WORKLOAD_IDENTITY_PROVIDER`, Azure uses `azure/login` with client/tenant/subscription IDs and no secret. Grep for `AWS_SECRET_ACCESS_KEY`, `GCP_SA_KEY` and `AZURE_CREDENTIALS` across `.github/workflows/` returns nothing outside this README. Following the guide therefore creates exactly the long-lived, exfiltratable credentials the OIDC design exists to eliminate, and the GCP step also grants project-wide `roles/run.admin` (line 460), contradicting the narrow-scope rule in the project CLAUDE.md. +- evidence: + ```bash + gh secret set AWS_ACCESS_KEY_ID + gh secret set AWS_SECRET_ACCESS_KEY + ... + gcloud iam service-accounts keys create key.json \ + --iam-account=cudly-cicd@.iam.gserviceaccount.com + gh secret set GCP_SA_KEY < key.json + ... + az ad sp create-for-rbac --name cudly-cicd --sdk-auth > azure-credentials.json + gh secret set AZURE_CREDENTIALS < azure-credentials.json + ``` +- suggested fix: Replace the Setup Guide's credential sections with the OIDC/WIF variables the workflows actually read, and state explicitly that no static cloud credentials are stored. +- verdict: CONFIRMED — `/usr/bin/grep` over `.github/workflows/` returns `AWS_SECRET_ACCESS_KEY`, `GCP_SA_KEY` and `AZURE_CREDENTIALS` only inside README.md (lines 92, 93, 172, 213, 446, 447, 467, 479), while every workflow authenticates via `vars.AWS_ROLE_TO_ASSUME` (deploy-aws-lambda.yml:246), `vars.GCP_WORKLOAD_IDENTITY_PROVIDER` (deploy-gcp.yml:149) and azure/login client-id (deploy-azure.yml:168). +- issue: (pending cross-reference) + +### A14-008 The only gating Trivy IaC scan skips the AWS Terraform environment +- category: security +- severity: high +- location: .pre-commit-config.yaml:196 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: The `trivy-config` hook is the only Trivy invocation in the repo that uses `--exit-code 1`. It passes `--skip-dirs terraform/environments/aws`, so a HIGH/CRITICAL misconfiguration introduced anywhere under the primary cloud's Terraform root (public S3 bucket, open security group, unencrypted RDS) is never gated. ci.yml's `Run Trivy IaC misconfiguration scanner` (line 779) scans `terraform/` but deliberately runs at the default exit code 0 and only uploads SARIF (its comment at 770-778 says so), so nothing else catches it. The hook's long comment justifies the `**/.terraform` and `.claude` skips in detail and says nothing about the AWS environment skip. +- evidence: + ```yaml + entry: bash -c 'trivy config --severity HIGH,CRITICAL --exit-code 1 --skip-dirs terraform/environments/aws --skip-dirs "**/.terraform" --skip-dirs .claude .' + ``` +- suggested fix: Remove `--skip-dirs terraform/environments/aws`, fix or explicitly `#trivy:ignore` the findings it surfaces, and document each remaining suppression. +- verdict: CONFIRMED — .pre-commit-config.yaml:196 is the only Trivy invocation in the repo using `--exit-code 1` and it skips `terraform/environments/aws`, while ci.yml:780-797's IaC step carries a comment at 770-778 stating it "uses the default exit-code 0 so misconfig findings are reported to the Security tab without gating the job". +- issue: (pending cross-reference) + +### A14-010 git-secrets allowed patterns are content regexes, so `resource `, `var.`, `data.` whitelist most of the repo +- category: security +- severity: high +- location: scripts/setup-git-secrets.sh:104 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: `git secrets --add --allowed ` suppresses any scanned **line** matching the regex; it is not a path filter. `'resource\s'` therefore whitelists every Terraform resource-block line and any line containing the word "resource" followed by whitespace; `'var\.'`, `'local\.'`, `'data\.'` and `'module\.'` whitelist any line containing those substrings. A real AWS account ID or access key placed on such a line — `resource "aws_iam_access_key" "x" { secret = "AKIA…" }` — passes the local gate silently. The neighbouring `'_test\.go'` and `'testdata/'` entries show the intent was to exclude paths, which this mechanism cannot do. +- evidence: + ```bash + git secrets --add --allowed 'var\.' + git secrets --add --allowed 'data\.' + ... + git secrets --add --allowed 'resource\s' + git secrets --add --allowed 'module\.' + ``` +- suggested fix: Delete the substring allowlist entries and rely on `.gitallowed`, which is already scoped to the specific placeholder values; if path exclusions are needed, filter the file list before invoking `git secrets --scan`. +- verdict: CONFIRMED — reproduced in a throwaway repo: with `git secrets --add --allowed 'resource\s'` (scripts/setup-git-secrets.sh:104), the line `resource "aws_iam_access_key" "x" { key = "AKIA-EXAMPLE-KEY-REDACTED" }` scanned clean while the same key on a plain line exited 1, because git-secrets matches the allowed regex against whole output lines. +- issue: (pending cross-reference) + +### A01-006 AWS RI exchange execute/quote/target-offerings ignore the session's allowed_accounts scope +- category: security +- severity: medium +- location: internal/api/handler_ri_exchange.go:1806 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: A user scoped to `allowed_accounts=[azure-sub-1]` who holds an unconstrained `execute:ri-exchange` calls `POST /api/ri-exchange/execute` naming the deployment AWS account's convertible RIs. `executeExchange` checks the verb, the permission Constraints (AccountIDs on the permission only) and the idempotency claim, then exchanges the RIs. The three read siblings (`listConvertibleRIs`, `getRIUtilization`, `getReshapeRecommendations`) and the Azure execute path all gate on `reshapeCloudAccountInScope`/`requireAzureSubscriptionScope`; `executeExchange`, `getExchangeQuote` (1685) and `listTargetOfferings` (117, enumerates the account's RIs) do not. `TestExecuteExchange_*` never register `GetAllowedAccountsAPI`. +- evidence: + ```go + err = h.requirePermissionConstraints(ctx, session, "ri-exchange", []auth.PermissionConstraints{{ + AccountIDs: []string{cloudAccountID}, + Providers: []string{string(common.ProviderAWS)}, + Services: []string{string(common.ServiceEC2)}, + Regions: []string{region}, + MaxPurchaseAmount: maxPayment, + }}) + if err != nil { + return nil, err + } + ``` +- suggested fix: Call `reshapeCloudAccountInScope` in `executeExchange`, `getExchangeQuote` and `listTargetOfferings` and refuse (errNotFound) when it returns false, mirroring the read endpoints. +- verdict: CONFIRMED — reshapeCloudAccountInScope is called only at internal/api/handler_ri_exchange.go:1351,1386,1539 (listConvertibleRIs/getRIUtilization/getReshapeRecommendations); executeExchange:1765-1846 runs the verb, the permission Constraints and the idempotency claim only, getExchangeQuote:1685-1727 and listTargetOfferings:117-140 check only view:purchases before calling AWS on the deployment account. +- issue: (pending cross-reference) + +### A01-009 Planned-purchase rows are minted with approval tokens that never expire +- category: security +- severity: medium +- location: internal/api/handler_plans.go:539 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: `createPurchaseExecutionsTx` writes up to 52 rows with `ApprovalToken` set and `ApprovalTokenExpiresAt` nil. `validateApprovalToken` (internal/purchase/approvals.go:124) skips the TTL check when the field is nil, and the scheduler's `getOrCreateExecution` reuses the existing row (and its token) for the notification email. Every other creation site stamps `config.ApprovalTokenTTL` (`newPendingExecution` 2575, `persistRetryExecution` 1837, notifications.go:142). A forwarded or leaked approval link for a scheduled row therefore stays valid until the row leaves pending/notified, which for a step months out can be months. +- evidence: + ```go + execution := &config.PurchaseExecution{ + PlanID: planID, + ExecutionID: uuid.New().String(), + Status: "pending", + StepNumber: plan.RampSchedule.CurrentStep + i + 1, + ScheduledDate: scheduledDate, + ApprovalToken: approvalToken, + CreatedByUserID: creator, + } + ``` +- suggested fix: Do not mint the token at scheduling time; let the notification step (`getOrCreateExecution`) generate it with `ApprovalTokenExpiresAt` when the email is actually sent, or stamp `ApprovalTokenExpiresAt` relative to `scheduledDate` here. +- verdict: CONFIRMED — createPurchaseExecutionsTx (internal/api/handler_plans.go:539-547) sets ApprovalToken with no ApprovalTokenExpiresAt while newPendingExecution (handler_purchases.go:2575), persistRetryExecution (:1837) and getOrCreateExecution (internal/purchase/notifications.go:127) all stamp config.ApprovalTokenTTL; validateApprovalToken (internal/purchase/approvals.go:124) skips the TTL when nil, and getOrCreateExecution:115-119 returns the existing plan+date row whose token buildNotificationData:158 embeds in the email. +- issue: (pending cross-reference) + +### A01-012 GET /api/plans lists every plan regardless of the session's allowed_accounts +- category: security +- severity: medium +- location: internal/api/handler_plans.go:35 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: A Read-Only user scoped to acct-A calls `GET /api/plans`. `listPlans` applies only the caller-supplied `account_ids` query filter and returns every plan (names, services, ramp position, health score) including plans whose accounts are all outside the user's scope. `getPlan`, `updatePlan`, `patchPlan`, `deletePlan` and `getPlannedPurchases` all enforce `requirePlanAccess`/`isPlanAllowedCached`; the list endpoint is the only unscoped read, and it is what feeds the Plans page. `TestHandler_listPlans` uses an admin session only. +- evidence: + ```go + filter := config.PurchasePlanFilter{AccountIDs: accountIDs} + plans, err := h.config.ListPurchasePlans(ctx, filter) + if err != nil { + return nil, err + } + ``` +- suggested fix: After loading, drop plans for which `isPlanAllowedCached` returns false (unrestricted sessions short-circuit), the same way `getPlannedPurchases` does. +- verdict: CONFIRMED — listPlans (internal/api/handler_plans.go:21-49) applies only the caller-supplied account_ids filter and never calls getAccountScope or requirePlanAccess, whereas getPlan:218, updatePlan:258, patchPlan:617, deletePlan:316 and getPlannedPurchases (handler_purchases.go:142 via isPlanAllowedCached) all scope per plan. +- issue: (pending cross-reference) + +### A02-008 Behind CloudFront every IP-keyed rate limit uses the edge server address, not the client +- category: security +- severity: medium +- location: internal/api/middleware.go:530 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: With `enable_cdn` the Lambda Function URL is called by CloudFront via OAC (terraform/environments/aws/compute.tf:25, 269-277). The function-URL event's `requestContext.http.sourceIp` is then the CloudFront edge IP and the viewer IP is only in `X-Forwarded-For`; internal/server/lambda.go:99 passes the event through unchanged (the plain HTTP server does parse XFF, internal/server/http.go:300-315). `login` (5/15 min), `setup_admin`, `reset_password`, `change_password`, `register` and `approve_cancel_public` are all keyed on this value. Five wrong passwords from anyone routed through the same edge lock every user on that edge out of login for 15 minutes, and an attacker gets a fresh budget per edge POP. +- evidence: + ```go + clientIP := req.RequestContext.HTTP.SourceIP + allowed, err := h.rateLimiter.AllowWithIP(ctx, clientIP, endpoint) + ``` +- suggested fix: In the Lambda transport, when the function URL is OAC-protected, replace `SourceIP` with the rightmost trusted entry of `X-Forwarded-For` (mirroring http.go), so the limiter keys on the viewer; keep `SourceIP` as-is for direct (auth_type NONE) invocations. +- verdict: CONFIRMED — checkRateLimit and checkRateLimitStrict (middleware.go:530,557) key on RequestContext.HTTP.SourceIP, internal/server/lambda.go has no X-Forwarded-For handling (only http.go:300-306 parses it), and terraform/environments/aws/compute.tf:25-26,269-277 puts CloudFront+OAC in front of the Function URL when enable_cdn=true; enable_cdn defaults to false (variables.tf:398-402), so only CDN deployments are affected. +- issue: (pending cross-reference) + +### A03-007 MFA can be replaced with only a session and the password, bypassing the proof-of-possession that MFADisable requires +- category: security +- severity: medium +- location: internal/auth/service_mfa.go:285 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: `MFADisable` demands password plus a TOTP or recovery code ("a stolen session alone shouldn't disable MFA", service_mfa.go:427-431). But `MFASetup` never checks `user.MFAEnabled`; it only re-verifies the password, writes a new pending secret, and `MFAEnable` promotes it over the existing `MFASecret` and replaces all recovery codes. An attacker with a live session and the password (the exact threat MFADisable models) calls setup then enable with a code from their own authenticator: the victim's authenticator stops working, the attacker's is enrolled, and the victim's recovery codes are gone. The re-enrol path is strictly weaker than the disable path it is equivalent to. +- evidence: + ```go + if !s.verifyPassword(password, user.PasswordHash) { + return nil, fmt.Errorf("%w", ErrMFAInvalidPassword) + } + secret, err := generateMFASecret() + ... + user.MFAPendingSecret = secret + user.MFAPendingSecretExpiresAt = &expiresAt + ``` +- suggested fix: When `user.MFAEnabled` is true, require a current TOTP or recovery code in `MFASetup` (same check as `MFADisable`) before writing a new pending secret; or refuse setup while enabled and require disable first. +- verdict: CONFIRMED — mfaSetup (handler_auth.go:542-556) requires only a session, MFASetup (service_mfa.go:285-316) checks the password and never reads MFAEnabled, and MFAEnable:393-397 overwrites MFASecret and MFARecoveryCodes from the pending secret with no check that MFA was already on, so a session plus password replaces the authenticator without the proof-of-possession MFADisable:436 demands. +- issue: (pending cross-reference) + +### A04-004 pgx error/warn logs carry bound query arguments (session tokens, bcrypt hashes, approval tokens) +- category: security +- severity: medium +- location: internal/database/connection.go:399-405 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: pgx v5 `tracelog.TraceQueryEnd` logs `{"sql", "args", "err"}` at `LogLevelError` whenever a query fails. `sanitizeLogData` strips the `args` key only when `level == LogLevelDebug`. A transient failure of `INSERT INTO sessions (token, ...)`, `UPDATE users SET password_hash = $1`, or `UPDATE purchase_executions SET approval_token = $7` therefore writes the raw session token, bcrypt hash or approval token to CloudWatch (pgx truncates strings only past 64 bytes; a 64-hex token and a 60-char bcrypt hash are logged whole). `security_test.go:184-198` pins this as intended ("args kept at warn/error level"). +- evidence: + ```go + for k, v := range data { + if isSensitiveKey(k) { + continue + } + if k == "args" && level == tracelog.LogLevelDebug && !bindParams { + continue + } + safe[k] = v + } + ``` +- suggested fix: drop the `level == LogLevelDebug` condition so `args` is stripped at every level unless `DB_LOG_BIND_PARAMETERS=true`, and flip the two security_test cases to expect stripping. +- verdict: CONFIRMED — `sanitizeLogData` gates the `args` strip on `level == tracelog.LogLevelDebug` (internal/database/connection.go:399), `isSensitiveKey` matches only the literal keys password/secret/token and never the args payload (internal/database/connection.go:384-386), the tracer is installed on every pooled connection (internal/database/connection.go:226), `stdLogger.Log` prints `safeData` at warn and error (internal/database/connection.go:417-420), and internal/database/security_test.go:183-198 pins "args kept at warn level"/"args kept at error level". +- issue: (pending cross-reference) + +### A06-010 Unauthenticated /health echoes raw internal error strings +- category: security +- severity: medium +- location: internal/server/health.go:92 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: `/health` is registered with no auth (internal/server/http.go:31) and always returns 200 with a JSON body. A migration failure puts `err.Error()` in the body verbatim — golang-migrate errors carry the failing migration number and the raw Postgres error including relation and column names; a DB health-check failure (health.go:123) and an auth-store ping failure (health.go:170) put pgx connection errors in the body, which name the host, port and user from the DSN. Any unauthenticated caller can poll the endpoint to map the schema and the database topology. +- evidence: + ```go + case err != nil: + return CheckResult{Status: "failed", Message: err.Error()} + ``` +- suggested fix: return a fixed status string in the public body and log the detail server-side, or gate the detailed body behind the scheduled-task/admin credential. +- verdict: CONFIRMED — `/health` is registered outside every auth wrapper (internal/server/http.go:31, contrast the scheduledauth wrap at http.go:39) and always writes 200 with the full JSON body (internal/server/health.go:58-70); all three checks put the raw error text in `Message` (health.go:92, 123, 170), and the golang-migrate and pgx errors reaching them carry migration numbers and DSN host/user detail respectively. +- issue: (pending cross-reference) + +### A06-012 SMTP send paths log raw recipient addresses while the sibling approval path redacts them +- category: security +- severity: medium +- location: internal/email/smtp_sender.go:182 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: `SendToEmailWithCC` (lines 182/184) and `SendToEmailWithCCMultipart` (lines 151/153) interpolate `toEmail` verbatim, so every password-reset, invite, welcome, scheduled-purchase and executed-purchase send on a GCP (SendGrid) or Azure (ACS) deployment writes a customer email address into the log aggregator. The same file redacts in `sendMultipartWithUnsubscribe` (lines 591/593) and the whole SES path uses `redactEmail`, so this is an internal inconsistency, and it is the case the project memory `feedback_pii_in_logs` describes ("Emails: log only the domain part or mask the local part"). Redacted value withheld here; no live address was observed, only the format string. +- evidence: + ```go + if len(sanitizedCC) > 0 { + logging.Debugf("Sent email via SMTP to %s (cc %d): %s", toEmail, len(sanitizedCC), subject) + } else { + logging.Debugf("Sent email via SMTP to %s: %s", toEmail, subject) + } + ``` +- suggested fix: wrap both call sites in `redactEmail(toEmail)` as the approval path already does. +- verdict: CONFIRMED — raw `toEmail` is interpolated at internal/email/smtp_sender.go:151,153 and 182,184, while the sibling `sendMultipartWithUnsubscribe` uses `redactEmail(toEmail)` on the identical log lines (smtp_sender.go:591,593) and every SES log does the same (internal/email/sender.go:291,293,325,327); both SMTP methods are the live delivery path on GCP (SendGrid) and Azure (ACS) per internal/email/factory.go:140,196. +- issue: (pending cross-reference) + +### A06-013 Azure bicep/ARM deploy script is rendered from unescaped data while every sibling shell template is shell-escaped +- category: security +- severity: medium +- location: internal/iacfiles/templates/azure-wif-bicep-deploy.sh.tmpl:64 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: the package doc for `iacfiles` states the renderer must escape user-controlled fields before `template.Execute`. `renderSingleFile` does so for CLI scripts (internal/api/handler_federation.go:274) and `writeCFNFiles` does so for the CFN deploy script (:640), but `writeAzureTemplateFiles` (:532) renders `azure-wif-bicep-deploy.sh.tmpl` with the raw `data` and writes the result into the zip with mode 0755. `.ContactEmail` (line 64) and `.CUDlyAPIURL` (line 83) land inside double-quoted bash. Addresses are validated only with `mail.ParseAddress`, which accepts a backtick in the local part (RFC 5322 atext), so a user whose account email contains one downloads a deploy script with a live command substitution that executes when they run it. Impact is bounded because `ContactEmail` is always the downloader's own session email (handler_federation.go:166), making this self-inflicted rather than cross-user, but the guard is simply absent on one of three shell-template paths. +- evidence: + ```sh + CONTACT_EMAIL="${CUDLY_CONTACT_EMAIL:-{{.ContactEmail}}}" + ... + -X POST "{{.CUDlyAPIURL}}/api/register" \ + ``` +- suggested fix: pass `shellEscapeData(data)` when rendering the deploy script in `writeAzureTemplateFiles`, exactly as `writeCFNFiles` does. +- verdict: CONFIRMED — `writeAzureTemplateFiles` renders the deploy script with the raw `data` (internal/api/handler_federation.go:532) and writes it with `addExecBytesToZip` (handler_federation.go:557), while the two sibling shell paths both escape first (`renderSingleFile` at handler_federation.go:273-274, `writeCFNFiles` at handler_federation.go:638-640); `.ContactEmail` and `.CUDlyAPIURL` land inside double-quoted bash and a heredoc (internal/iacfiles/templates/azure-wif-bicep-deploy.sh.tmpl:64,83), and I confirmed by running `mail.ParseAddress` that both `` a`id`b@example.com `` and `a${x}b@example.com` parse cleanly, so the only validator (internal/api/validation.go:134-150) admits them. +- issue: (pending cross-reference) + +### A10-005 The CLI still registers a `--yes` flag that skips the purchase confirmation +- category: security +- severity: medium +- location: cmd/main.go:124 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: `cudly --purchase --yes` reaches `ConfirmPurchase(..., skipConfirmation=true)`, which returns true immediately (cmd/helpers.go:208-210) without printing the "About to purchase N instances" line or reading stdin. The only remaining human gate on a non-reversible multi-thousand-dollar RI purchase is removed by a single flag, and the flag is discoverable in `--help`. The project's own standing rule is that this confirmation must run on every invocation, dry-run included. There is a `TestDryRunFlagRemoved` guard against reintroducing `--dry-run` but no equivalent guard for `--yes`. +- evidence: + ```go + rootCmd.Flags().BoolVar(&toolCfg.SkipConfirmation, "yes", false, "Skip confirmation prompt for purchases (use with caution)") + ``` +- suggested fix: Remove the flag and the `SkipConfirmation` field, drop the parameter from `ConfirmPurchase`, and add a `rootCmd.Flags().Lookup("yes") != nil` regression test alongside `TestDryRunFlagRemoved`. +- verdict: CONFIRMED — the flag is bound at cmd/main.go:124 and reaches ConfirmPurchase, which returns true ahead of both the TTY check and the prompt (cmd/helpers.go:207-215). It can reach exactly two call sites, the main purchase path (cmd/multi_service.go:158) and the CSV per-region prompt (cmd/multi_service.go:701); no dry-run path is affected because both sit behind `!isDryRun` (:156) / the `else` of `if isDryRun` (:689-705), and dry runs never prompt at all. No test in cmd/ references the flag — TestDryRunFlagRemoved at cmd/effective_dry_run_test.go:38 has no --yes counterpart. +- issue: (pending cross-reference) + +### A10-019 `ListSecrets` errors are discarded and `arns[0]` is written to unvalidated in both configure paths +- category: security +- severity: medium +- location: cmd/configure_azure.go:148 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: The returned `err` is consulted only inside `if err == nil && len(arns) > 0` and is then overwritten by the `json.Marshal` assignment, so an AccessDenied or throttling failure from `secretsmanager:ListSecrets` is never reported. The code proceeds with the bare secret name; if a secret of that exact name does not exist, `UpdateSecret` fails with a resource-not-found error that gives the operator no hint the listing was the real problem. Conversely, the Secrets Manager `name` filter is a prefix match, so with secrets `cudly-AzureCredentials` and `cudly-AzureCredentialsOld` both present, `arns[0]` is whichever AWS returns first and the credentials can be written into the wrong secret. Identical code at cmd/configure_gcp.go:119-125. +- evidence: + ```go + arns, err := store.ListSecrets(ctx, secretName) + // Use the ARN if found, otherwise use the name (will fail if secret doesn't exist) + secretID := secretName + if err == nil && len(arns) > 0 { + secretID = arns[0] + } + ``` +- suggested fix: Return the `ListSecrets` error to the caller, and when more than one ARN comes back, require an exact name match on the returned entry (or error naming the candidates) instead of taking index 0. +- verdict: CONFIRMED — cmd/configure_azure.go:148-154 and cmd/configure_gcp.go:119-125 consult err only inside the guard and then reassign it on the next `:=`, and ListSecrets hands the name to a FilterNameStringTypeName filter and returns every ARN with no exact-name check before the caller takes arns[0] (cmd/secrets_store.go:31-55). +- issue: (pending cross-reference) + +### A10-021 The minted GCP service-account private key is left on disk after being uploaded +- category: security +- severity: medium +- location: cmd/configure_gcp.go:722 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: Step 5 writes a fresh, never-expiring GCP service-account private key to `~/cudly-gcp-key.json`, `runConfigureGCP` reads it and stores it in AWS Secrets Manager, then prints "GCP configuration complete!" and exits. The plaintext key stays in the operator's home directory indefinitely, where it is picked up by backup tools, Dropbox/iCloud sync and `find ~ -name '*.json'`. Nothing in the wizard deletes it, warns about it, or tells the operator to rotate it. The Azure path deliberately never persists its secret to disk, so the two flows disagree on the same hazard. +- evidence: + ```go + keyFile := filepath.Join(home, "cudly-gcp-key.json") + ... + fmt.Printf("Key file written to: %s\n", keyFile) + ``` +- suggested fix: After `storeGCPCredentials` succeeds on a wizard-minted key, delete the local file and say so, or print an explicit instruction to remove and rotate it. +- verdict: CONFIRMED — the key is written to ~/cudly-gcp-key.json at cmd/configure_gcp.go:722-745 and runConfigureGCP ends at printGCPConfigurationSuccess (cmd/configure_gcp.go:162-167, :256-262) with no unlink and no warning about the file. +- issue: (pending cross-reference) + +### A10-022 `clear-rate-limit` interpolates an env var into a `LIKE` pattern for an unguarded DELETE +- category: security +- severity: medium +- location: cmd/lambda/clear-rate-limit/main.go:41 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: The domain is a query parameter (no SQL injection) but it lands inside a `LIKE` pattern, where `%` and `_` are metacharacters. With `RATE_LIMIT_DOMAIN=%` the pattern becomes `EMAIL#%@%#ENDPOINT#forgot_password` and the function deletes the forgot-password rate limits for every domain in the table, disabling brute-force protection on the password-reset endpoint tenant-wide. A single `_` in a legitimate domain broadens the match by one character position. The handler has no dry-run mode and no confirmation, unlike its sibling `cmd/cleanup-lambda`, which does have one. `defaultDomain` is additionally a hardcoded `leanercloud.com`, so a misconfigured deployment silently targets a specific tenant. +- evidence: + ```go + tag, err := db.Exec(ctx, + "DELETE FROM rate_limits WHERE id LIKE $1", + "EMAIL#%@"+domain+"#ENDPOINT#forgot_password") + ``` +- suggested fix: Validate the domain against a hostname regex and reject `%`/`_`, or escape them and add `ESCAPE '\'`; add a `dryRun` field to the event mirroring `CleanupEvent`. +- verdict: CONFIRMED — cmd/lambda/clear-rate-limit/main.go:39-42 concatenates getDomain() straight into the LIKE pattern with no hostname validation and no metacharacter escaping, the handler takes no event at all (:30, :65) so there is no dry-run switch, and the sibling does carry one (cmd/cleanup-lambda/main.go:15, :44). Triggering it requires control of RATE_LIMIT_DOMAIN at deploy time, which is the stated scenario. +- issue: (pending cross-reference) + +### A11-008 A failing /api/info offers the first-admin setup screen +- category: security +- severity: medium +- location: frontend/src/api/auth.ts:415 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: `getPublicInfo` swallows every non-OK response and returns `{ version: '', admin_exists: false }`. `init()` reads that value and, seeing `!publicInfo.admin_exists`, renders the "No admin account exists yet. Set up the first admin to get started." modal (app.ts:49-54). So during any backend outage, 5xx, or misrouted deploy, an unauthenticated visitor is told the installation has no admin and is invited to claim it. The API key requirement is what stops the takeover, not this code path; the UI is nonetheless asserting a security-relevant fact it did not verify, and a user who reaches this screen on a healthy deployment has no signal that the check failed. +- evidence: + ```typescript + // api/auth.ts:412-419 + export async function getPublicInfo(): Promise { + const response = await fetch(`${API_BASE}/info`); + if (response.ok) { + return response.json() as Promise; + } + return { version: '', admin_exists: false }; + } + ``` +- suggested fix: throw on a non-OK response so `init()`'s existing catch falls through to the login modal, which is the safe default for an unknown bootstrap state. +- verdict: CONFIRMED — the non-OK branch returns `admin_exists: false` rather than throwing (frontend/src/api/auth.ts:414-418), and `init()` branches straight into `showAdminSetupModal` on that value (frontend/src/app.ts:47-53); its `catch` only covers a network-level failure, so any 4xx/5xx from `/info` renders the first-admin screen to an unauthenticated visitor. +- issue: (pending cross-reference) + +### A11-011 The permission matrix hides 13 of the 20 actions, including every -any verb +- category: security +- severity: medium +- location: frontend/src/users/permissionMatrix.ts:11 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: `ACTIONS` is a hand-written list of seven verbs. `permissions.ts` exports `ALL_ACTIONS`, a compile-time-exhaustive list of all twenty, and the group-edit form was already migrated onto it for exactly this reason (see the drift comment at groupModals.ts:166-188). The Permission Overview table therefore renders no row at all for `approve-any`, `retry-any`, `cancel-any`, `update-any`, `execute-any`, `sell-any`, `revoke-any` and the -own variants. An admin auditing which groups can approve or retry a purchase reads a matrix that shows those capabilities nowhere, and concludes a group holding `approve-any:purchases` grants nothing beyond `view`. The only test touching this file is an accessibility assertion (a11y.test.ts:75). +- evidence: + ```typescript + // users/permissionMatrix.ts:11 + const ACTIONS = ['view', 'create', 'update', 'delete', 'execute', 'approve', 'admin'] as const; + ``` +- suggested fix: import `ALL_ACTIONS` from `../permissions` and render from it, as `buildActionOptions` already does. +- verdict: CONFIRMED — the hand-written seven-verb list at frontend/src/users/permissionMatrix.ts:11 drives the row set (permissionMatrix.ts:46, matching on `p.action === action`), while `ACTION_EXHAUSTIVENESS_CHECK` enumerates twenty actions including every `-any` and `-own` verb (frontend/src/permissions.ts:100-122); the matrix is live, rendered from userActions.ts:98, so a group holding only `approve-any:purchases` shows dashes in every row, and a11y.test.ts:75 is indeed the only test that touches the file. +- issue: (pending cross-reference) + +### A13-008 `fargate_certificate_arn` never enables HTTPS, so the ALB serves the dashboard in cleartext +- category: security +- severity: medium +- location: terraform/environments/aws/compute.tf:177 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: `enable_https` is derived only from `frontend_domain_names` and `subdomain_zone_name`; `var.fargate_certificate_arn` is passed as `certificate_arn` on the next line but is not part of the condition. An operator following `terraform/profiles/aws/fargate-dev.tfvars.example:21` ("Only needed if not using subdomain_zone_name") supplies a pre-issued ACM certificate and no zone, gets `enable_https = false`, and the module creates only the port-80 listener, whose `default_action` forwards to the target group rather than redirecting (terraform/modules/compute/aws/fargate/main.tf:482). The admin login form, session cookie and API traffic all travel unencrypted, and the supplied certificate is never attached to anything. The shipped fargate profile sets none of the three, so this is its default state. +- evidence: + ```hcl + enable_https = length(var.frontend_domain_names) > 0 && var.subdomain_zone_name != "" + certificate_arn = ( + length(aws_acm_certificate.frontend) > 0 + ? aws_acm_certificate_validation.frontend[0].certificate_arn + : var.fargate_certificate_arn + ) + ``` +- suggested fix: Make the condition `(length(var.frontend_domain_names) > 0 && var.subdomain_zone_name != "") || var.fargate_certificate_arn != ""`, and add a precondition that refuses a public Fargate ALB with neither. +- verdict: CONFIRMED — terraform/environments/aws/compute.tf:177 derives `enable_https` from `frontend_domain_names` and `subdomain_zone_name` only, and `var.fargate_certificate_arn` (variables.tf:382-386, default `""`) appears solely as the fallback on the `certificate_arn` line below it. With `enable_https = false` the module creates no 443 listener (fargate/main.tf:498-499 is `count = var.enable_https ? 1 : 0`) and the port-80 listener's default action forwards to the target group rather than redirecting (fargate/main.tf:481-493). terraform/profiles/aws/fargate-dev.tfvars.example:21 tells the operator the certificate is "Only needed if not using subdomain_zone_name", which is exactly the combination that yields cleartext. +- issue: (pending cross-reference) + +### A13-010 The ACR AcrPull role assignment is dead because the Container App authenticates with admin credentials +- category: security +- severity: medium +- location: terraform/environments/azure/registry.tf:10 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: `admin_enabled = true` turns on a static registry username/password, and compute.tf:102-103 feeds those to the Container App's `registry` block (terraform/modules/compute/azure/container-apps/main.tf:92-94), so image pulls authenticate as the admin account. The `azurerm_role_assignment.acr_pull` created directly below therefore grants a managed identity that is never used for pulls, and the long-lived admin credential remains live in Key Vault, in state and in the container app secret. Removing the role assignment would change nothing observable, which is the definition of a guard on an unreachable state. +- evidence: + ```hcl + admin_enabled = true # Enables username/password login for docker push + ... + resource "azurerm_role_assignment" "acr_pull" { + scope = azurerm_container_registry.main.id + role_definition_name = "AcrPull" + principal_id = module.compute_container_apps[0].managed_identity_principal_id + ``` +- suggested fix: Drop `registry_username`/`registry_password` and set the Container App registry block to use the managed identity, then set `admin_enabled = false`; add `depends_on = [azurerm_role_assignment.acr_pull]` on the container app so the first pull does not race RBAC propagation. +- verdict: CONFIRMED — terraform/environments/azure/compute.tf:101-103 passes `registry_server`/`registry_username`/`registry_password` from `azurerm_container_registry.main` (`admin_enabled = true`, registry.tf:10), and the module's `registry` block authenticates with `username` plus `password_secret_name = "registry-password"` and never sets `identity` (terraform/modules/compute/azure/container-apps/main.tf:88-96). The `AcrPull` assignment to `managed_identity_principal_id` (registry.tf:15-21) therefore governs no pull; deleting it changes nothing observable. +- issue: (pending cross-reference) + +### A13-013 `.trivyignore` AVD-AWS-0013 states a justification the code contradicts +- category: security +- severity: medium +- location: .trivyignore:17 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: The suppression claims TLS 1.0 "was tightened in a prior PR" and that the finding only trips because the value is "inherited rather than set explicitly". `terraform/modules/frontend/aws/main.tf:113` sets `minimum_protocol_version` explicitly to `"TLSv1"` whenever `acm_certificate_arn` is null, which is the default for every environment. The finding is live, not an artifact, so the suppression hides a real TLS 1.0 viewer policy rather than a false positive. It is not currently exploitable only because `enable_cdn = false` in every tfvars, which is a different reason from the one written down. +- evidence: + ```hcl + minimum_protocol_version = var.acm_certificate_arn != null ? "TLSv1.2_2021" : "TLSv1" + ``` +- suggested fix: Either drop the no-certificate branch (a CloudFront default certificate forces TLSv1 regardless, so the distribution should require a real certificate) or rewrite the suppression to state the actual reason, which is that the distribution is never created. +- verdict: CONFIRMED — .trivyignore:17-22 says the value "still trips when the value is inherited rather than set explicitly", but terraform/modules/frontend/aws/main.tf:113 sets `minimum_protocol_version` explicitly, to `"TLSv1"` on the `acm_certificate_arn == null` branch. The mitigating fact the suppression does not mention also checks out: the distribution is behind `count = var.enable_cdn ? 1 : 0` (terraform/environments/aws/frontend.tf:9) and every shipped tfvars sets `enable_cdn = false` (github-{dev,staging,prod}.tfvars). So the written justification is false and the real one is unwritten. +- issue: (pending cross-reference) + +### A13-015 `setup.sh` prints the Azure client secret to stdout while the WIF branch deliberately uses stderr +- category: security +- severity: medium +- location: arm/CUDly-CrossSubscription/setup.sh:135 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: The `wif` branch writes the private key to stderr with an explicit warning that stdout redirects would capture it (lines 86-94). The default `client_secret` branch prints a two-year Azure AD application secret to stdout with no warning at all, so `./setup.sh > onboarding.txt` or any wrapper that captures stdout persists the credential to disk, and a CI invocation captures it in the run log. The two branches emit the same class of secret through opposite channels. +- evidence: + ```bash + if [[ "$MODE" == "client_secret" ]]; then + echo " client_secret : ${CLIENT_SECRET}" + echo "" + echo " Save the client_secret now — it will not be shown again." + ``` +- suggested fix: Route the client-secret block to stderr with the same warning the WIF branch carries, so both credential modes behave identically under redirection. +- verdict: CONFIRMED — arm/CUDly-CrossSubscription/setup.sh:86-88 carries the warning "Do NOT run this script in CI/CD or any environment that captures stdout. Both blocks are written to stderr so stdout redirects do not capture them", and lines 91-95 duly `>&2` the key and certificate. The `client_secret` branch mints a two-year secret at lines 107-112 and the summary block prints it with a bare `echo` at line 139, inside a `[[ "$MODE" == "client_secret" ]]` guard, with no redirection and no warning. +- issue: (pending cross-reference) + +### A13b-003 Azure Bicep role assignment takes the role definition ID as an unconstrained parameter +- category: security +- severity: medium +- location: iac/federation/azure-target/bicep/azure-wif.bicep:24 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: `roleDefinitionId` is a plain string parameter with no `@allowed` list, and the documented deployment flow (lines 12-16 of the same file) tells the customer to run `az deployment sub create --parameters @target-azure-wif-bicep-params.json`. Anyone who supplies or edits that parameter file — a support engineer, a doc page, a tampered download — can set it to the Owner role GUID `8e3af657-a8ff-443c-a75c-2fe8c4bcb635` or Contributor. `subscriptionResourceId` resolves whatever GUID it is handed at subscription scope, so the deployment grants CUDly's service principal full control of the customer's subscription while the template, its description text and its assignment description all still say "Reservation Purchaser". Nothing in the template compares the value against the role it claims to assign. Same defect in the generated ARM at `iac/federation/azure-target/bicep/azure-wif.arm.json:20`. +- evidence: + ```bicep + param roleDefinitionId string = 'f7b75c60-3036-4b75-91c3-6b41c27c1689' + ... + roleDefinitionId: subscriptionResourceId('Microsoft.Authorization/roleDefinitions', roleDefinitionId) + description: 'CUDly Reservation Purchaser — assigned via CUDly federation setup.' + ``` +- suggested fix: Drop the parameter and inline the role definition the template intends to assign, or constrain it with `@allowed([...])` naming only the acceptable GUIDs. A customer-deployed template should never let its caller choose which privilege it grants. +- verdict: CONFIRMED — `iac/federation/azure-target/bicep/azure-wif.bicep:20-37` has no `@allowed` on `roleDefinitionId` and no comparison against the role it names in its own `description` at line 35, and `subscriptionResourceId` at line 34 resolves any built-in GUID at subscription scope; the generated ARM repeats it verbatim at `iac/federation/azure-target/bicep/azure-wif.arm.json:18-24,41`, and the documented flow at lines 12-16 passes the value through a params file the customer does not author. +- issue: (pending cross-reference) + +### A13b-008 CloudFormation cross-account role trusts the whole source account with no way to pin the execution role +- category: security +- severity: medium +- location: iac/federation/aws-cross-account/cloudformation/template.yaml:152 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: The CloudFormation template hardcodes `:root` as the trusted principal, so every IAM principal in CUDly's source account — a CI role, a developer user, a compromised low-privilege function — can assume the customer's role and place irreversible multi-year commitment purchases, subject only to knowing the external ID. The Terraform module for the identical access offers `cudly_execution_role_arn` (`iac/federation/aws-cross-account/terraform/variables.tf:24-33`) and pins the trust to that single principal when set. Customers who deploy the CloudFormation variant, which is the copy-paste path, silently get the wider trust with no parameter to narrow it. +- evidence: + ```yaml + AssumeRolePolicyDocument: + Version: "2012-10-17" + Statement: + - Effect: Allow + Principal: + AWS: !Sub "arn:aws:iam::${SourceAccountID}:root" + Action: sts:AssumeRole + ``` +- suggested fix: Add an optional `CUDlyExecutionRoleARN` parameter and an `Fn::If` that selects it over `:root`, mirroring the Terraform local. +- verdict: CONFIRMED — `iac/federation/aws-cross-account/cloudformation/template.yaml:150-152` hardcodes `:root` and the template's Parameters block (lines 7-49) offers no execution-role input at all, while `iac/federation/aws-cross-account/terraform/main.tf:23-27` selects `var.cudly_execution_role_arn` over the account root; the CloudFormation side is the wrong one, though the external-ID condition at lines 153-155 still narrows the trust to a caller holding that secret. +- issue: (pending cross-reference) + +### A13b-009 External ID is an operator-chosen 8-character string in both CloudFormation templates while Terraform generates a UUID +- category: security +- severity: medium +- location: iac/federation/aws-cross-account/cloudformation/template.yaml:13 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: `MinLength: 8` with no `AllowedPattern` accepts `password`, `cudly123` or the customer's own account number as the external ID, and `cloudformation/stacks/CUDly-CrossAccount/template.yaml:31` has the same shape. The Terraform module for the same access auto-generates a `random_uuid` when the value is empty (`iac/federation/aws-cross-account/terraform/main.tf:19,27`), so the two paths produce secrets of wildly different strength. The external ID is the sole control that stops a confused-deputy registration: the role name defaults to a published constant (`CUDly-CrossAccount` here, `CUDly` in the stacks template) and the account ID is often discoverable, so an attacker who registers the victim's role ARN with a guessed external ID gets CUDly to assume into the victim account on their behalf. A short, human-chosen value makes that guess practical. +- evidence: + ```yaml + ExternalID: + Type: String + Description: > + Random string used as the sts:ExternalId condition to prevent confused-deputy + attacks. Generate a UUID and share it securely with the source account. + MinLength: 8 + NoEcho: true + ``` +- suggested fix: Raise `MinLength` to 32 and add an `AllowedPattern` requiring a hex or UUID shape, in both CloudFormation templates. Better still, have CUDly issue the external ID at registration time rather than accepting one the deploying party chooses. +- verdict: CONFIRMED — `iac/federation/aws-cross-account/cloudformation/template.yaml:13-19` and `cloudformation/stacks/CUDly-CrossAccount/template.yaml:30-37` both carry `MinLength: 8` with no `AllowedPattern`, against `iac/federation/aws-cross-account/terraform/main.tf:19-27` which substitutes a `random_uuid` when the value is empty, so the two paths for the same trust boundary produce different secret strength; the confused-deputy end of the scenario is weaker than stated, since `internal/api/handler_registrations.go:76-78` lands every registration as `pending` and `approveRegistration` at line 252 requires a CUDly admin before the ARN is ever assumed. +- issue: (pending cross-reference) + +### A13b-010 Registration endpoint URL is unvalidated, so `http://` sends the external ID and GCP credential config in cleartext +- category: security +- severity: medium +- location: iac/federation/aws-cross-account/terraform/registration.tf:21 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: `var.cudly_api_url` has no validation in any of the five modules and is interpolated straight into `data.http`. A customer handed a mistyped or downgraded `http://` URL POSTs `aws_external_id` (`registration.tf:15`, the confused-deputy secret) over cleartext; anyone on the path recovers it and, with the predictable role ARN from A13b-009, can drive purchases in that account. The GCP variant is worse in payload: `iac/federation/gcp-target/terraform/registration.tf:45` sends the full external-account credential config JSON. The CloudFormation Lambda has the same gap at `iac/federation/aws-cross-account/cloudformation/template.yaml:206`, where `props["CUDlyAPIURL"]` is concatenated with no scheme check. +- evidence: + ```hcl + data "http" "cudly_registration" { + count = local.do_register ? 1 : 0 + url = "${var.cudly_api_url}/api/register" + method = "POST" + ``` +- suggested fix: Add a `validation` block requiring `startswith(var.cudly_api_url, "https://")` to all five modules' `cudly_api_url`, and an `AllowedPattern: "^$|^https://.+"` to the CloudFormation `CUDlyAPIURL` parameter. +- verdict: CONFIRMED — the `cudly_api_url` variable block is byte-identical and validation-free in all five modules (`iac/federation/{aws-target,aws-cross-account,azure-target,gcp-sa-impersonation,gcp-target}/terraform/variables.tf`) and is interpolated directly at each `registration.tf` `url` line; the AWS cross-account payload carries `aws_external_id` (`iac/federation/aws-cross-account/terraform/registration.tf:15`), the GCP WIF payload carries the full `credential_payload` config JSON (`iac/federation/gcp-target/terraform/registration.tf:45`), and the CloudFormation Lambda concatenates the parameter with no scheme check (`iac/federation/aws-cross-account/cloudformation/template.yaml:206`) behind a `CUDlyAPIURL` parameter that has no `AllowedPattern` (lines 26-29). +- issue: (pending cross-reference) + +### A13b-012 Azure federated identity credential accepts any issuer URL with no validation +- category: security +- severity: medium +- location: iac/federation/azure-target/terraform/variables.tf:19 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: `cudly_issuer_url` is a bare required string with no validation block, and it becomes the `issuer` of the federated identity credential at `main.tf:57`. Azure AD then fetches JWKS from whatever host it names and mints tokens for CUDly's app registration for any JWT that host signs with `sub` equal to `cudly_federated_subject` (default `cudly-controller`, a constant shared by every CUDly customer) and `aud` `api://AzureADTokenExchange`. A customer given a wrong or attacker-controlled issuer URL — a typosquat in a docs page, a stale value in a params file — hands whoever controls that host the ability to obtain tokens for the app and, via the subscription role assignment at `main.tf:78`, to purchase reservations in the customer's subscription. The AWS WIF sibling validates its issuer URL hard (`iac/federation/aws-target/terraform/variables.tf:11-14`) and rejects `$` and `*` in both audience and subject; the Azure module applies none of that to any of its three federation inputs. +- evidence: + ```hcl + variable "cudly_issuer_url" { + description = "CUDly OIDC issuer URL (e.g. https://cudly.example.com/oidc). Azure AD fetches JWKS from this issuer to verify client assertion JWTs." + type = string + } + ``` +- suggested fix: Add a validation requiring `https://` and no whitespace on `cudly_issuer_url`, and a non-empty validation on `cudly_federated_subject` so it cannot be blanked. Consider making the subject per-subscription rather than one global constant, so a single leaked assertion does not replay across every customer tenant. +- verdict: CONFIRMED — `iac/federation/azure-target/terraform/variables.tf:19-33` declares `cudly_issuer_url`, `cudly_federated_subject` and `cudly_federated_audience` with no `validation` block on any of the three, and all three feed the federated credential unchecked at `iac/federation/azure-target/terraform/main.tf:56-58`; the AWS sibling guards the equivalent inputs hard (`iac/federation/aws-target/terraform/variables.tf:11-15` on the issuer, `:36-39` on the audience, `:59-62` and the `$`/`*` rejection on the subject), so this is a missing guard rather than a difference in threat model, and the default subject `cudly-controller` is one constant across every customer tenant. +- issue: (pending cross-reference) + +### A13c-001 Lambda secrets policy renders `Resource = "*"` when the DB secret ARN is empty, while its three siblings guard against exactly that +- category: security +- severity: medium +- location: terraform/modules/compute/aws/lambda/main.tf:245 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: `compact()` drops empty strings, so each of the admin / credential-encryption / + scheduled-task entries is written as `X != "" ? "${X}*" : ""` and disappears when unset. The + database entry on line 247 has no such guard: with `database_password_secret_arn = ""` the + element renders as the literal `"*"`, which `compact()` keeps, and the Lambda execution role + gains `secretsmanager:GetSecretValue` on **every secret in the account**. The Fargate twin + writes `"${arn}-*"` (fails closed to the unmatchable `-*`), so the same input produces opposite + polarity on the two compute paths. +- evidence: + ```hcl + Resource = compact([ + var.database_password_secret_arn, + "${var.database_password_secret_arn}*", + var.admin_password_secret_arn, + var.admin_password_secret_arn != "" ? "${var.admin_password_secret_arn}*" : "", + var.credential_encryption_key_secret_arn, + var.credential_encryption_key_secret_arn != "" ? "${var.credential_encryption_key_secret_arn}*" : "", + var.scheduled_task_secret_arn, + var.scheduled_task_secret_arn != "" ? "${var.scheduled_task_secret_arn}*" : "", + ]) + ``` +- suggested fix: guard the database entry like its siblings, or add a `validation` block on + `database_password_secret_arn` requiring a non-empty `arn:aws:secretsmanager:` prefix. +- verdict: PLAUSIBLE — the asymmetry is real and the variable has no default and no validation + (lambda/variables.tf:72-75), so `""` renders the literal `"*"` that `compact()` keeps, and the + Fargate twin's `-*` polarity is confirmed (fargate/main.tf:109); but the scenario needs a caller + that passes the empty string, and the sole in-tree caller passes a real ARN + (`module.database.password_secret_arn` at environments/aws/compute.tf:47, sourced from + `module.secrets.database_password_secret_arn` via database.tf:15 and database/aws/main.tf:64). +- issue: (pending cross-reference) + +### A13c-009 Fargate's two original EventBridge roles lack the `ecs:cluster` and `iam:PassedToService` conditions its two newer ones carry +- category: security +- severity: medium +- location: terraform/modules/compute/aws/fargate/main.tf:896 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: four EventBridge invoker roles exist in this file with identical purpose. + `eventbridge_ladder_run` (:1103) and `eventbridge_fire_scheduled_purchases` (:1213) constrain + `ecs:RunTask` with `ArnEquals ecs:cluster` and `iam:PassRole` with `StringEquals + iam:PassedToService = ecs-tasks.amazonaws.com`. The recommendations role (:896) and the + RI-exchange role (:997) have neither, so each can pass the task role — which holds the SP/RI + purchase grants — to any AWS service that accepts a passed role, and can run the task + definition in any cluster in the account. Same input, same intent, two of four guarded. +- evidence: + ```hcl + { + Effect = "Allow" + Action = ["iam:PassRole"] + Resource = [ + aws_iam_role.task_execution.arn, + aws_iam_role.task.arn + ] + } + ``` +- suggested fix: copy the two `Condition` blocks from `eventbridge_ladder_run` onto the + recommendations and RI-exchange policies; better, collapse all four into one `for_each` role so + the guards cannot diverge again. +- verdict: CONFIRMED — fargate/main.tf:896-923 and :997-1024 each carry a bare `ecs:RunTask` on the + task-definition ARN and a bare `iam:PassRole` on the execution and task roles, with no `Condition` + block; the two newer roles do constrain both (`ecs:cluster` at :1120 and :1230, + `iam:PassedToService` at :1135 and :1245). Reach is bounded by the roles' `events.amazonaws.com` + trust policy, so exploitation needs an actor able to create EventBridge targets. +- issue: (pending cross-reference) + +### A14-009 The git-secrets AWS secret-key pattern is inverted and can never match a key +- category: security +- severity: high +- location: scripts/setup-git-secrets.sh:52 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: An AWS secret access key is 40 characters drawn from `[A-Za-z0-9/+=]`. The registered pattern is the negated class, so it matches 41 consecutive characters that are **not** in that set. A real secret key committed to the repo is not matched by this pattern at all; the only things that can match are runs of punctuation such as the `━━━━` separator lines this repo's own scripts print. The pattern is registered by `make setup-git-secrets` and becomes part of every developer's local pre-commit gate, where its comment claims AWS-secret coverage it does not provide. +- evidence: + ```bash + git secrets --add '[^A-Za-z0-9/+=]{40}[^A-Za-z0-9/+=]' # AWS Secret Access Key + ``` +- suggested fix: Replace with a positive class anchored on non-key boundaries, e.g. `(^|[^A-Za-z0-9/+=])[A-Za-z0-9/+=]{40}([^A-Za-z0-9/+=]|$)`, and add a fixture proving it fires on a synthetic 40-char key. +- verdict: CONFIRMED — scripts/setup-git-secrets.sh:52 registers the negated class, and tested with `/usr/bin/grep -E` it matched only a run of U+2501 box-drawing characters, never a synthetic 40-character base64 key. +- severity-adjusted: medium — `git secrets --register-aws` (scripts/setup-git-secrets.sh:43) plus the pattern on line 53 still block an assignment-shaped key (verified: `secret_key = "<40 chars>"` exits 1), so only a bare unassigned key slips through. +- issue: (pending cross-reference) + +### A14-024 The AWS IAM parity extractor cannot see a wildcard action +- category: security +- severity: medium +- location: scripts/check-aws-iam-parity.sh:79 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: The extraction regex requires an uppercase letter after the colon, so `ec2:*` and `Action: "*"` produce no token. If the CloudFormation stack were changed to `"ec2:*"` while the Terraform module keeps its explicit list, both sides extract the same set of remaining actions, `compare_pair` finds no diff and the guard reports "OK: AWS IAM action lists are in parity" for two policies that are not equivalent. The Azure sibling covers this class explicitly (`scripts/testdata/role-parity/dataactions-wildcard-arm.json`, cases 13-15 of `test-azure-role-parity.sh`); the AWS suite has no wildcard fixture. No live wildcard exists in the compared namespaces today, so this is a latent gap rather than an active drift. +- evidence: + ```bash + actions=$(grep -oE "(^|[^A-Za-z])(${ACTION_PREFIXES}):[A-Z][A-Za-z]+" "$file" \ + | sed 's/^[^a-zA-Z]//' \ + | sort -u) + ``` +- suggested fix: Extend the pattern to capture `:\*` and a bare `"*"` action, and add a fixture to `scripts/test-aws-iam-parity.sh` asserting that a wildcard on one side fails parity. +- verdict: CONFIRMED — the regex at scripts/check-aws-iam-parity.sh:79 requires `[A-Z]` after the colon, and on synthetic input `ec2:*`, `rds:*` and `Action: "*"` produced zero tokens while `ec2:DescribeInstances` matched, so a wildcard on one side leaves both lists identical and compare_pair (line 160) reports parity; the Azure sibling fixture scripts/testdata/role-parity/dataactions-wildcard-arm.json is exercised at scripts/test-azure-role-parity.sh:240 while the AWS suite has no wildcard case. +- issue: (pending cross-reference) + +### A14-025 `--source` is never validated, so a typo renders an AWS trust policy with an empty issuer +- category: security +- severity: medium +- location: scripts/generate-federation-iac.go:518 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: `populateData` allowlists `--target` against `validTargets` but passes `source` through unchecked. With `--target aws --source Azure` (or any typo), `subjectClaimModeFor` returns `subjectClaimRequired` so a subject claim is accepted, `awsOIDCIssuer` falls through its `default:` to `""`, and `singleFileTmpl` selects `aws-wif.tfvars.tmpl`. The generated file carries `oidc_issuer_url = ""` alongside a populated `oidc_subject_claim`, so the operator applies a WIF trust policy whose issuer was silently dropped. `--target aws --source AWS` is worse: it produces a WIF artifact where the cross-account external-ID artifact was intended. `TestGenerator_InvalidTargetReportedAsTarget` covers the target axis; there is no equivalent source case. +- evidence: + ```go + if !validTargets[target] { + return fmt.Errorf("--target must be aws, azure, or gcp (got %q)", target) + } + ``` + ```go + func awsOIDCIssuer(source, tenantID string) string { + switch source { + case "azure": ... + case "gcp": ... + default: + return "" + ``` +- suggested fix: Add a `validSources` allowlist checked alongside `validTargets`, and make `awsOIDCIssuer`'s default arm an error rather than an empty string. +- verdict: CONFIRMED — populateData validates only `--target` (scripts/generate-federation-iac.go:518), and running `--target aws --source Azure` rendered `oidc_issuer_url = ""` beside a populated `oidc_subject_claim`, because subjectClaimModeFor:427 keys on `source != "aws"`, awsOIDCIssuer:154 falls through to `return ""`, and singleFileTmpl:314 selects aws-wif.tfvars.tmpl. +- issue: (pending cross-reference) + +### A14-027 Only the subject claim is validated; six sibling fields reach the same Bash/JSON/HCL sinks raw +- category: security +- severity: medium +- location: scripts/generate-federation-iac.go:596 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: The file's own comment (355-376) explains that `--oidc-subject-claim` is allowlisted because it is interpolated verbatim into three grammars. `AccountName`, `AccountExternalID`, `TenantID`, `ProjectID`, `ServiceAccountEmail`, `CUDlyAPIURL` and `ContactEmail` reach the same templates with no validation and no escaping: `aws-wif.tfvars.tmpl` renders `account_name = "{{.AccountName}}"` (an unescaped HCL string) and `aws-cfn-deploy.sh.tmpl:78` renders `-X POST "{{.CUDlyAPIURL}}/api/register"` inside a Bash script, where a value containing `$(…)` executes when the operator runs the generated deploy script. The server path escapes all of them: `internal/api/handler_federation.go:285-299` calls `shellEscape` on every field. Same templates, two callers, one of which escapes nothing. +- evidence: + ```go + data := iacData{ + AccountName: *accountName, + AccountExternalID: *accountID, + AccountSlug: slug, + Source: *source, + ContactEmail: *contactEmail, + CUDlyAPIURL: *cudlyAPIURL, + } + ``` +- suggested fix: Apply the same charset validation (or the server's `shellEscape`) to every field written into `iacData`, not only `OIDCSubjectClaim`. +- verdict: CONFIRMED — only OIDCSubjectClaim is validated (validateOIDCSubjectClaim at scripts/generate-federation-iac.go:441); `--account-name 'Acme"\nevil_var = "pwned'` rendered a second HCL attribute into the tfvars output and `--cudly-api-url 'https://x$(id)'` passed through verbatim, against internal/api/handler_federation.go:285-299 which shellEscapes every field. +- severity-adjusted: medium — unchanged: the Bash sink at aws-cfn-deploy.sh.tmpl:78 is unreachable today because bundle mode dies at A14-026, but the HCL sink is reachable and injectable. +- issue: (pending cross-reference) + +### A14-036 `detect-private-key` skips every Go test file, and `.gitleaksignore` uses a form gitleaks does not honour +- category: security +- severity: medium +- location: .pre-commit-config.yaml:61 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: A PEM private key committed in any `*_test.go` file passes the pre-commit gate, and test fixtures are exactly where real keys are most often pasted. The neighbouring comment explains that the `internal/credentials/resolver.go` exclusion was removed to tighten the gate, but the far broader `_test\.go` exclusion was kept without justification. Separately, `.gitleaksignore` contains a bare file path; gitleaks expects finding fingerprints (`commit:file:rule:line`), so the entry matches nothing — and no workflow or hook in this shard invokes gitleaks at all, so the file is inert either way while reading as an active suppression. +- evidence: + ```yaml + - id: detect-private-key + name: Detect private keys + exclude: '(_test\.go|frontend/src/index\.html)$' + ``` + ``` + internal/secrets/aws_resolver_httptest_test.go + ``` +- suggested fix: Narrow the exclusion to the specific fixture files that need it; either wire gitleaks into CI and convert `.gitleaksignore` to real fingerprints, or delete the file. +- verdict: CONFIRMED — .pre-commit-config.yaml:61 excludes `_test\.go$` from detect-private-key with no accompanying justification, and a repo-wide search for gitleaks returns only .github/runbooks/credential-compromise.md:128 ("Enable gitleaks pre-commit hook"), so no tool ever consumes `.gitleaksignore`'s bare path entry. +- issue: (pending cross-reference) + +### A01-017 AuthPublic revoke/approve/cancel routes disclose execution existence and status before any authentication +- category: security +- severity: low +- location: internal/api/handler_purchases.go:1258 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: An unauthenticated caller POSTs `/api/purchases/revoke/{uuid}` with no token and no session. `revokeViaEmailToken` loads the row and returns 404 (no such execution), 409 "still pending", or 409 "cannot be revoked (status=)" before `tryRevokeViaSession`/`authorizeApprovalAction` run, so the response is a per-UUID existence and status oracle. `cancelPurchase` (1029-1035) and `loadApproveExecution` (524-535, including orphan details) return the same pre-auth 404/409s. The scoping helpers elsewhere deliberately return 404 to avoid exactly this enumeration signal (scoping.go:16-21). +- evidence: + ```go + execution, err := h.config.GetExecutionByID(ctx, execID) + if err != nil { + return nil, fmt.Errorf("failed to get execution: %w", err) + } + if execution == nil { + return nil, NewClientError(404, "execution not found") + } + if statusErr := checkRevokableStatus(execution); statusErr != nil { + return nil, statusErr + } + ``` +- suggested fix: Resolve the principal (session or token) first and collapse pre-auth failures into a single 401/404, running the status checks only after the caller is authorized. +- verdict: CONFIRMED — internal/api/router.go:163-172 registers approve/cancel/revoke as AuthPublic; revokeViaEmailToken (handler_purchases.go:1248-1260) returns 404 or a status-bearing 409 before tryRevokeViaSession/authorizeApprovalAction run, cancelPurchase:1029-1035 404s pre-auth, and loadApproveExecution:523-537 returns 404/orphan-409 before either auth branch; requireAccountAccess (scoping.go:16-21) documents the enumeration rationale these public routes do not follow. +- issue: (pending cross-reference) + +### A02-015 PATCH is registered as a mutating route but excluded from CSRF validation and the CORS method list +- category: security +- severity: low +- location: internal/api/middleware.go:274 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: router.go:150 serves `PATCH /api/plans/{id}` (a plan mutation). `requiresCSRFValidation` returns false for any method other than POST/PUT/DELETE, so the CSRF header is never checked on PATCH; `buildResponseHeaders` (handler.go:720) advertises `GET, POST, PUT, DELETE, OPTIONS`, so a cross-origin deployment's preflight for PATCH fails. Exploitability is low because the session credential is a bearer header rather than a cookie, but the middleware's own contract ("state-changing requests need CSRF protection") is not met for this verb and nothing tests it. +- evidence: + ```go + if method != "POST" && method != "PUT" && method != "DELETE" { + return false + } + ``` +- suggested fix: Add PATCH to the CSRF method set and to `Access-Control-Allow-Methods`, with a middleware test for `PATCH /api/plans/x`. +- verdict: CONFIRMED — router.go:150 registers PATCH /api/plans/ as AuthUser, requiresCSRFValidation (middleware.go:274-276) returns false for any method outside POST/PUT/DELETE, the only handler-level validateCSRF calls are in handler_purchases.go:672,1135,1335 and handler_ri_exchange.go:2183 (none in handler_plans.go), and buildResponseHeaders (handler.go:720) advertises only GET/POST/PUT/DELETE/OPTIONS; low severity stands because the session credential is a bearer header. +- issue: (pending cross-reference) + +### A02-016 The public prefix "/api/info" also matches the authenticated /api/info/deployment route +- category: security +- severity: low +- location: internal/api/middleware.go:22 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: `isPublicEndpoint("/api/info/deployment")` is true, so `validateSecurityContext` skips authentication and API-key usage booking for a route that returns the Secrets Manager URL and the host AWS account id (router.go:363). Only the router-level `requireAuth` in `Route` keeps it protected; a future change that trusts the middleware (as the comment at router.go:16-25 anticipates) would expose it, and `TestHandler_isPublicEndpoint` (middleware_test.go:15-44) has no case for the path. The same file explicitly exact-matches `/version` and `/api/register` to avoid exactly this overlap. +- evidence: + ```go + publicPrefixEndpoints := []string{ + "/health", // Root health endpoint (no /api prefix) + "/api/health", // API health endpoint + "/api/info", + ``` +- suggested fix: Move `/api/info` to the exact-match switch and add `{"/api/info/deployment", false}` to the test table. +- verdict: CONFIRMED — "/api/info" sits in the prefix list (middleware.go:22) so isPublicEndpoint("/api/info/deployment") is true and validateSecurityContext (handler.go:762) skips authenticatePrincipal and the API-key usage booking; the route is protected only by its AuthUser level (router.go:363) enforced in Router.Route, which the router.go:16-25 comment describes as defense-in-depth rather than the primary gate. +- issue: (pending cross-reference) + +### A02-020 Registration list and detail expose reference_token, which reject deliberately strips +- category: security +- severity: low +- location: internal/api/handler_registrations.go:156 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: `config.AccountRegistration.ReferenceToken` is tagged `json:"reference_token"` (internal/config/types.go:1078). `listRegistrations` and `getRegistration` return the struct whole, so every admin (and any log/export of the response) sees the registrant's private status token, while `rejectRegistration` (lines 373-381) builds a filtered map precisely so the token is "not exposed to admin". The token is the only credential for `GET /api/register/{token}`, so the two endpoints disagree about the same secret. +- evidence: + ```go + regs, err := h.config.ListAccountRegistrations(ctx, filter) + ... + return regs, nil + ``` +- suggested fix: Tag `ReferenceToken` with `json:"-"` and return it only from `submitRegistration`'s explicit map. +- verdict: CONFIRMED — registrationColumns (internal/config/store_postgres_registrations.go:225) selects reference_token for the list and get queries, the field is tagged json:"reference_token" (internal/config/types.go:1078), and listRegistrations/getRegistration (handler_registrations.go:153-175) return the structs whole, while rejectRegistration (handler_registrations.go:373-381) builds a filtered map explicitly to hide it. +- issue: (pending cross-reference) + +### A03-015 Account scope matches on display name, so two accounts sharing a name share scope +- category: security +- severity: low +- location: internal/auth/types.go:207 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: `MatchesAccount` and `AccountScope.Allows` (account_scope.go:112) treat a scope entry equal to `accountName` as a match. Display names are free text and not unique. A group scoped to `["prod"]` (by name) gains access to every account later registered with the display name "prod", including one belonging to another team, with no change to the group. Renaming an account also silently changes who can see it. +- evidence: + ```go + for _, a := range allowed { + if a == accountID { + return true + } + if accountName != "" && a == accountName { + return true + } + } + ``` +- suggested fix: Match on account ID only; migrate existing name-based entries to IDs once and drop the name parameter. +- verdict: CONFIRMED — MatchesAccount (types.go:203-209) and AccountScope.Allows (account_scope.go:108-114) both match on accountName, production callers pass real display names (handler_ri_exchange.go:431, :809; handler_ladder.go:60; handler_analytics.go:245; handler_history.go:1015), and the group write path compares allowed_accounts entries only against the actor's own scope (group_ceiling.go:322-338), never against the account table, so a name entry is storable and matches any later account with that name. +- issue: (pending cross-reference) + +### A03-016 TOTP codes are replayable within the acceptance window +- category: security +- severity: low +- location: internal/auth/service_mfa.go:77 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: `verifyTOTP` accepts the current step and one on each side with no record of the last accepted counter. A code observed once (shoulder-surfed, phished, or captured from a request log) remains valid for up to 90 seconds and can be used for a second login, an `MFADisable`, or a recovery-code regeneration in that window. RFC 6238 section 5.2 requires rejecting a previously used code. +- evidence: + ```go + valid := 0 + for _, offset := range []int64{-1, 0, 1} { + counter := (currentTime / timeStep) + offset + expected := generateTOTP(secret, counter) + if subtle.ConstantTimeCompare([]byte(expected), []byte(code)) == 1 { + valid = 1 + } + } + ``` +- suggested fix: Persist the last accepted counter on the user row and refuse any counter less than or equal to it. +- verdict: CONFIRMED — verifyTOTP (service_mfa.go:63-87) accepts counters -1..+1 with no state, the User row carries no last-accepted counter (UpdateUser column list store_postgres.go:193-212) and a grep for any counter tracking in internal/auth returns nothing, so a captured code is accepted again within the skew window by Login, MFADisable and the recovery-code regeneration path. +- issue: (pending cross-reference) + +### A05-003 The "universal" 4-eyes gate is not on the SQS / cron execute path +- category: security +- severity: medium +- location: internal/purchase/approvals.go:167 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: `enforceFourEyesPolicy` documents itself as the single choke point every approve/execute entry point inherits, and it runs only inside `ApproveAndExecute`. `claimAndExecute` (manager.go:193) goes straight to `executeAndFinalize`, so both of its callers — the SQS `execute_purchase` worker (messages.go:136) and the cron sweep `processOneExecution` (manager.go:678) — commit money without ever reaching the gate. With `RequireDifferentApprover` on and a plan carrying `AutoPurchase=true`, user A's own pending row is executed by the next cron tick with no second person involved, while the same row approved through the dashboard would be denied. The four-eyes tests cover `ApproveAndExecute` and `ProcessMessage`'s approve branch only; none covers `handleExecutePurchase` or `ProcessScheduledPurchases`. +- evidence: + ```go + // enforceFourEyesPolicy is the UNIVERSAL 4-eyes approval gate (issue #1005). + // It is the single choke point ApproveAndExecute runs before mutating any + // execution state, so every approve/execute entry point inherits the policy + // regardless of which caller reaches ApproveAndExecute: + ``` +- suggested fix: decide whether dual control is meant to bind AutoPurchase and say so at the gate. If it is, move the check into `executeAndFinalize` alongside `armedRedriveRefusal`, which is already positioned as the funnel every executor reaches money through; if it is not, correct the comment and name the AutoPurchase carve-out explicitly. +- verdict: PLAUSIBLE — the structural claim holds: claimAndExecute (manager.go:177-193) calls executeAndFinalize directly with no 4-eyes call, and the test list in approvals_test.go covers only ApproveAndExecute/ApproveExecution (:347-:558), never handleExecutePurchase or ProcessScheduledPurchases. But the stated scenario needs a pending/notified row that is BOTH non-web-sourced and carries a creator, and I could not find a writer that produces one: executableByScheduler rejects `common.PurchaseSourceWeb` (manager.go:635), and the only two writers of CreatedByUserID (handler_purchases.go:1901 and :2585) both stamp Source=cudly-web. The reachable divergence is narrower — a system-created row (nil creator, empty Source, AutoPurchase plan) executes on the cron tick where checkDifferentApprover (approvals.go:255) would have denied it fail-closed. +- severity-adjusted: low — no human can self-approve through this path today; the only rows that reach it have no creator to compare against. +- issue: (pending cross-reference) + +### A06-029 CUDLY_FEDERATED_SUBJECT is validated on the AWS and GCP onboarding scripts but not on either Azure one +- category: security +- severity: low +- location: internal/iacfiles/templates/azure-wif-cli.sh.tmpl:26 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: the same knob is guarded on the two sibling paths — `aws-wif-cli.sh.tmpl:22-41` refuses an empty or `$`/`*`/whitespace-bearing `OIDC_SUBJECT_CLAIM` because IAM expands policy variables inside Condition values, and `gcp-wif-cli.sh.tmpl:50-60` pins a charset plus a 127-character cap because the value lands inside a CEL string literal. The Azure templates (also azure-wif-deploy.sh.tmpl:61) interpolate the override straight into the federated-credential JSON heredoc with no check, so a value containing `"` closes the `subject` string and injects sibling keys into the request body (for example widening `audiences`). Operator-supplied rather than attacker-supplied, so this is defense in depth, but the guard exists on two of three paths for a reason that applies to the third. +- evidence: + ```sh + CUDLY_FEDERATED_SUBJECT="${CUDLY_FEDERATED_SUBJECT:-cudly-controller}" + ... + "subject": "${CUDLY_FEDERATED_SUBJECT}", + ``` +- suggested fix: apply the GCP script's charset and length check to `CUDLY_FEDERATED_SUBJECT` in both Azure templates. +- verdict: CONFIRMED — both Azure templates take the override and interpolate it straight into the federated-credential JSON heredoc with no validation of any kind (internal/iacfiles/templates/azure-wif-cli.sh.tmpl:26,47-56 and azure-wif-deploy.sh.tmpl:61,66-75), while the two siblings do guard: the GCP script applies a charset regex plus a 127-character cap (gcp-wif-cli.sh.tmpl:50-61) and the AWS script rejects whitespace, `$` and `*` (aws-wif-cli.sh.tmpl:31-41). A `"` in the value closes the `subject` string and lets sibling keys such as `audiences` be injected into the request body. +- issue: (pending cross-reference) + +### A06-030 tfvars templates interpolate values into HCL string literals with no escaping +- category: security +- severity: low +- location: internal/iacfiles/templates/aws-wif.tfvars.tmpl:25 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: the bundle path renders the `.auto.tfvars` files with the raw `data` (internal/api/handler_federation.go:592) — only the shell templates get `shellEscapeData`. Inside an HCL double-quoted string, `${...}` is a template interpolation, and `mail.ParseAddress` accepts `{`, `}` and `$` as RFC 5322 atext, so a user whose account email is `a${x}b@example.com` downloads a bundle whose `contact_email = "a${x}b@example.com"` line makes `terraform apply` fail with an unresolved-variable error, or evaluate an HCL expression. Reach is narrow today because `buildGenericIaCData` leaves `AccountName`, `AccountExternalID` and `ProjectID` empty and `ContactEmail` is the downloader's own address, but the tfvars path has no escaper at all while the two shell paths each have one. +- evidence: + ```hcl + cudly_api_url = "{{.CUDlyAPIURL}}" + contact_email = "{{.ContactEmail}}" + account_name = "{{.AccountName}}" + ``` +- suggested fix: add an `hclEscape` helper (escape `\`, `"`, and `$`/`%` before `{`) and apply it to the tfvars render the way `shellEscapeData` is applied to the script renders. +- verdict: CONFIRMED — `addBundleTerraform` renders the tfvars template with the raw `data` and no escaper of any kind (internal/api/handler_federation.go:591-593), and `shellEscapeData` is the only escaper in the file, applied solely to the two shell paths (handler_federation.go:273-274, 638-640); the values land unquoted-inside-quotes at internal/iacfiles/templates/aws-wif.tfvars.tmpl:24-26. I confirmed by running `mail.ParseAddress` that `a${x}b@example.com` parses cleanly (so the only validator, internal/api/validation.go:134-150, admits it) while a literal `"` does not, which bounds the impact to HCL interpolation rather than string-breakout. +- issue: (pending cross-reference) + +### A08-003 Five Azure service clients drop the SSRF-hardened HTTP client on the nil branch of NewClientWithHTTP +- category: security +- severity: high +- location: providers/azure/services/compute/client.go:134 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: `NewClientWithHTTP(cred, sub, region, nil)` stores a nil `HTTPClient`. Every outbound call (`c.httpClient.Do(req)` in `fetchCapacityProviderState`, and the purchase two-step) nil-derefs and panics. The three clients that were fixed (managedredis:95, savingsplans:87, synapse:77) fall back to `httpclient.New()`, which installs the `blockIMDSDialer` that refuses `169.254.169.254`; compute, cache (client.go:103), cosmosdb (client.go:101), database (client.go:125) and search (client.go:80) have neither the guard nor the SSRF defence on that branch. +- evidence: + ```go + func NewClientWithHTTP(cred azcore.TokenCredential, subscriptionID, region string, httpClient HTTPClient) *ComputeClient { + return &ComputeClient{ + cred: cred, + subscriptionID: subscriptionID, + region: region, + httpClient: httpClient, + } + } + ``` +- suggested fix: add `if httpClient == nil { httpClient = httpclient.New() }` to the five constructors, matching managedredis/savingsplans/synapse. +- verdict: PLAUSIBLE — the nil guard is genuinely absent in the five constructors (compute/client.go:134, cache/client.go:103, cosmosdb/client.go:101, database/client.go:125, search/client.go:80) and present in the three siblings (managedredis/client.go:87, savingsplans/client.go:86, synapse/client.go:76), but no non-test caller of `NewClientWithHTTP` exists anywhere in the tree, so nothing passes nil today. +- severity-adjusted: low — reaching the nil deref requires a caller that does not exist; this is a constructor-consistency gap, not a live panic or SSRF exposure. +- issue: (pending cross-reference) + +### A08-016 The accounts-cache generation guard protects the cache write but not the value handed to in-flight callers +- category: security +- severity: medium +- location: providers/azure/accounts_cache.go:123 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: `SetCredential` bumps `accountsGen` so a fetch that started under the old credential cannot publish. That fetch still returns its list to every caller that joined the singleflight. `GetRecommendationsClient` then reads the NEW credential (provider.go:659) and builds a fan-out client over the OLD credential's subscription list — precisely the hazard `SetCredential`'s own doc names: "Serving it to the new credential would report subscriptions this principal may have no access to, and — via GetRecommendationsClient's fan-out — fan out across them." +- evidence: + ```go + p.accountsMu.Lock() + if p.accountsGen == gen { + p.cachedAccounts = accounts + } + p.accountsMu.Unlock() + return accounts, nil + ``` +- suggested fix: when `p.accountsGen != gen`, return an error (or re-fetch under the new generation) instead of returning the stale-credential list, so the guard covers the returned value as well as the cache. +- verdict: PLAUSIBLE — `fetchAccountsShared` (accounts_cache.go:122-132) does return `accounts` whether or not the generation matched, so the pre-invalidation list reaches every joined caller; but SetCredential's own contract records that "every caller installs the credential before the first accounts fetch" (provider.go:223-224) and GetRecommendationsClient documents the same window as the accepted guarantee (provider.go:651-657), so the race needs a concurrent SetCredential no production caller performs. +- severity-adjusted: low — no live caller mutates the credential during an in-flight accounts fetch. +- issue: (pending cross-reference) + +### A08b-026 Recommendation SKU and region are interpolated unescaped into the Retail Prices OData `$filter` +- category: security +- severity: medium +- location: providers/azure/services/database/client.go:550 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd (same at `cache/client.go:573`, `managedredis/client.go:468`, `synapse/client.go:417`) +- failure scenario: `rec.ResourceType` reaches `GetOfferingDetails` from a request body on the purchase/quote paths and is pasted between single quotes with no escaping. A value containing `'` terminates the literal: `X' or armSkuName ne '` turns the SKU predicate into a tautology, so the pricing walk returns every SKU in the region and, given the last-item-wins extractor in A08b-010, an attacker-chosen price is what the quote reports. A crafted value can also inject a `contains(...)` clause that makes the request expensive enough to exhaust the 50-page walk. +- evidence: + ```go + filter := fmt.Sprintf("serviceName eq 'SQL Database' and armRegionName eq '%s' and armSkuName eq '%s'", + region, sku) + ``` +- suggested fix: escape embedded single quotes by doubling them, and validate the SKU against `[A-Za-z0-9_.-]+` at the boundary before it reaches the filter builder. +- verdict: PLAUSIBLE — the unescaped interpolation into a single-quoted OData literal is confirmed at database:550-551, cache:573-574, managedredis:468-469 and synapse:417-418, and no `ResourceType` validation exists at the API boundary; what I could not establish is the delivery half, since the only route into these builders is `GetOfferingDetails`, which no code in this repo calls (grep outside `providers/` and tests yields only pkg/provider/interface.go:50). +- severity-adjusted: low — the injection primitive is real but currently has no reachable attacker-controlled feed. +- issue: (pending cross-reference) + +### A09-006 ExpectedAccount is silently ignored on the dependency-injected exchange path +- category: security +- severity: high +- location: pkg/exchange/exchange.go:343 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: `assertAccount` is called only by the package-level `GetExchangeQuote` (line 222) and `ExecuteExchange` (line 295). The methods `ExchangeClient.GetQuote` and `ExchangeClient.Execute` — the path `internal/server/handler_ri_exchange.go:61` and `internal/server/ladder_write.go:75` construct and use — go straight to `getQuoteWithAPI`/`executeWithAPI`, neither of which calls it. `executeWithAPI` even copies `req.ExpectedAccount` into `quoteReq` at line 350, where it is also ignored. A caller that sets `ExpectedAccount` on the DI path believes it has a cross-account guard on an irreversible purchase and has none; the field is silently inert. +- evidence: + ```go + func executeWithAPI(ctx context.Context, client EC2ExchangeAPI, req ExchangeExecuteRequest) (string, *ExchangeQuoteSummary, error) { + if req.MaxPaymentDueUSD == nil { + return "", nil, fmt.Errorf("refusing to execute without max-payment-due-usd guardrail") + } + quoteReq := ExchangeQuoteRequest{ + Region: req.Region, + ExpectedAccount: req.ExpectedAccount, + ``` +- suggested fix: either drop `ExpectedAccount` from the two request structs so no caller can believe it is enforced, or move the STS check into `executeWithAPI`/`getQuoteWithAPI` behind an injected identity resolver so both entry paths honour it. +- verdict: CONFIRMED — `git grep -n assertAccount` returns only pkg/exchange/exchange.go:202 (definition), :222 and :295 (the two package-level wrappers); `GetQuote`/`Execute` at exchange.go:185 and :191 call `getQuoteWithAPI`/`executeWithAPI` directly, and the copy at line 349 is never read. +- severity-adjusted: low — no live caller sets `ExpectedAccount` on the DI path: `git grep -n ExpectedAccount` shows the only setters are ci_cd_sanity_tests/cmd/ri-exchange/main.go:98 and :153, both of which go through the package-level `GetExchangeQuote`/`ExecuteExchange` that do call `assertAccount`; internal/server/handler_ri_exchange.go:61 and ladder_write.go:75 never set the field. It is a latent trap, not an active bypass. +- issue: (pending cross-reference) + +### A10-028 The Azure service-principal client secret is printed to stdout unconditionally +- category: security +- severity: low +- location: cmd/configure_azure.go:531 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: `printAzureSPResult` writes the freshly minted client secret to stdout with no TTY check. Run under `script`, `tee`, a CI job, a tmux logging pane, or simply left in terminal scrollback, a live Azure credential with the "Reservations Administrator" role is captured in plaintext. The choice mirrors `az ad sp create-for-rbac` and is deliberate, but that tool is not the one that then stores the same secret in Secrets Manager two prompts later, so CUDly could avoid the round trip through the screen entirely. +- evidence: + ```go + fmt.Printf(" appId (Client ID): %s\n", result.AppID) + fmt.Printf(" password (Client Secret): %s\n", result.ClientSecret) + fmt.Printf(" tenant (Tenant ID): %s\n", result.TenantID) + ``` +- suggested fix: Carry the wizard's result straight into `collectAzureCredentials` so the secret never reaches stdout; if it must be shown, gate the print on `term.IsTerminal(int(os.Stdout.Fd()))`. +- verdict: CONFIRMED — printAzureSPResult writes the secret with a bare fmt.Printf, no TTY test and no redaction (cmd/configure_azure.go:526-540), and it is reached unconditionally after a successful create at :519; the wizard then asks the operator to retype the same value in the next step, so the exposure is structural rather than incidental. +- issue: (pending cross-reference) + +### A12-015 `planId` is interpolated unescaped into a hidden input's `value` attribute +- category: security +- severity: medium +- location: frontend/src/plans.ts:2150 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: `openAddPurchasesModal` builds its markup with `innerHTML`. `planName`, two lines above, goes through `escapeHtml`; `planId` does not. A plan id containing `" onfocus=alert(1) autofocus x="` breaks out of the attribute and executes in the Plans tab. The id reaches this function from `btn.dataset['id']`, which came from an API-supplied `plan.id`, so the trust boundary is the API response, not the DOM. +- evidence: + ```html +

Schedule additional purchases for ${escapeHtml(planName)}

+
+ + ``` +- suggested fix: Wrap it as `value="${escapeHtmlAttr(planId)}"`, matching the `planName` line directly above. +- verdict: PLAUSIBLE — planId really is interpolated unescaped into `value="${planId}"` one line below an escaped planName (frontend/src/plans.ts:2148-2150), but the only supplier is purchase_plans.id, a UUID primary key (internal/database/postgres/migrations/000001_initial_schema.up.sql:51), so a quote-bearing id is a runtime condition I could not establish. +- severity-adjusted: low — the only supplier of planId is a UUID primary key, so the missing escape is defence in depth rather than a reachable injection +- issue: (pending cross-reference) + +### A12-047 Server-controlled keys are assigned onto a plain object, allowing `__proto__` assignment +- category: security +- severity: low +- location: frontend/src/commitmentOptions.ts:288 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: `fetchAndPopulateCommitmentOptions` iterates `Object.entries(body.aws)` and writes `awsConfigs[service] = {...}` with no key validation. A `/api/commitment-options` response containing a `"__proto__"` key invokes the `__proto__` setter and replaces the prototype of the module's AWS config object with the attacker-supplied value, changing what every unlisted service lookup resolves to. The lookup on line 110 has the mirror-image problem: `commitmentConfigs[provider.toLowerCase()]` resolves prototype members, so `getCommitmentConfig('constructor', 'name')` returns the string `"Object"` as a `CommitmentConfig` and `getValidPaymentOptions` then throws on `config.payments.filter`. +- evidence: + ```ts + const awsConfigs = commitmentConfigs.aws ?? (commitmentConfigs.aws = {}); + for (const [service, supportedCombos] of Object.entries(body.aws)) { + // ... + awsConfigs[service] = { terms: STANDARD_TERMS, payments: AWS_PAYMENTS, invalidCombinations: ... }; + } + ``` +- suggested fix: Skip keys that are not own-enumerable safe names (reject `__proto__`, `constructor`, `prototype`), and read the tables with `Object.prototype.hasOwnProperty.call`. +- verdict: PLAUSIBLE — The missing own-property guard is real, with keys assigned straight from Object.entries(body.aws) (frontend/src/commitmentOptions.ts:288) and raw-string indexing at :110, but reaching it needs the server to emit a `__proto__` or `constructor` key and those come from the DB-persisted AWS probe (internal/api/handler_commitment_options.go:53-58), a condition I could not establish. +- issue: (pending cross-reference) + +### A12-053 `canRevokeCompletedRow` omits the creator match its three sibling predicates enforce +- category: security +- severity: low +- location: frontend/src/history.ts:678 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: `canCancelPendingRow`, `rbacAllowsApprove` and `canRetryFailedRow` all require `p.created_by_user_id === user.id` for the `-own` verb and deny on an absent creator. `canRevokeCompletedRow` returns `canAccess('revoke-own', 'purchases')` unconditionally, so any `revoke-own` holder sees a Revoke button on every in-window Azure row in the tenant and gets a 403 on click. The backend does enforce ownership, so this is the UX-versus-RBAC drift the function's own comment cites PR #995 for, in the one predicate that did not adopt the fix. +- evidence: + ```ts + if (canAccess('admin', '*') || canAccess('revoke-any', 'purchases')) return true; + return canAccess('revoke-own', 'purchases'); + ``` +- suggested fix: Mirror the siblings: `return canAccess('revoke-own','purchases') && !!p.created_by_user_id && p.created_by_user_id === user.id;`. +- verdict: CONFIRMED — canRevokeCompletedRow returns canAccess('revoke-own','purchases') with no creator comparison (frontend/src/history.ts:677-678) unlike its three siblings (frontend/src/history.ts:543-545, :575-578, :636-638), and the backend denies revoke-own on another user's execution (internal/api/handler_purchases_revoke_test.go:743-760), so the button renders and 403s. +- issue: (pending cross-reference) + +### A12-054 `_fourEyesMode` initialises to `false`, contradicting the documented fail-closed default +- category: security +- severity: low +- location: frontend/src/history.ts:47 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: The doc comment at lines 50-62 states that an unreadable config "fails closed, to dual-control ON", and the catch path does set `true`. The module-level initialiser is `false`, so any render reaching `renderPendingActionButtons` before the first `refreshFourEyesMode()` resolves treats dual control as off and offers the creator an Approve button on their own row. All current paths await a `Promise.all` first, so this is latent, but the safe default costs nothing. +- evidence: + ```ts + let _fourEyesMode = false; + ``` +- suggested fix: Initialise to `true`. +- verdict: CONFIRMED — The module-level initialiser is `let _fourEyesMode = false` (frontend/src/history.ts:47) while the docstring directly above states an unreadable config fails closed to dual-control ON and the catch path does set true (frontend/src/history.ts:57-67). +- issue: (pending cross-reference) + +### A12-056 A handful of API-sourced numbers bypass escaping into `innerHTML` +- category: security +- severity: low +- location: frontend/src/history.ts:1154 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: `api.getHistory` is cast with `as unknown as Promise` and never validated at runtime, so the TypeScript `number` types are assumptions the JSON does not guarantee. `p.count` (lines 1154, 1799) and `p.retry_attempt_n` (line 881, inside a `title="…"` attribute) are the only API values in the file that skip `escapeHtml`; the `title` occurrence would break out of the attribute. `riexchange.ts:1970` has the same shape for `config.max_payment_per_exchange_usd` and the other numeric settings, and `plans.ts:737` for `purchase.count`. +- evidence: + ```ts + ${p.count} + // line 881 + lineage.push(`↻ Retry #${p.retry_attempt_n}`); + ``` +- suggested fix: Coerce at the boundary, `escapeHtml(String(p.count ?? ''))` and `Number(p.retry_attempt_n) || 0`, so no unescaped API value reaches an `innerHTML` template. +- verdict: PLAUSIBLE — p.count and p.retry_attempt_n really are the only API values in the file that skip escaping, one of them inside a title attribute (frontend/src/history.ts:1154, :881, :1799), and the response is cast with no runtime validation (frontend/src/history.ts:346), but a break-out needs the API to emit a string where the type says number, which I could not establish. +- issue: (pending cross-reference) + +### A13-014 GCP CI impersonation is granted to the whole repository on a generically named shared pool +- category: security +- severity: medium +- location: terraform/environments/gcp/ci-cd-permissions/github_oidc.tf:46 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: The `attribute_condition` on the provider pins repo and ref, but the `workloadIdentityUser` grant is `principalSet://.../attribute.repository/` with no ref component, and the pool is named `github-actions` rather than something CUDly-specific. Principal identifiers are pool-scoped, not provider-scoped, so any second provider added to that pool that can mint `attribute.repository` for this repo satisfies the grant and impersonates a service account carrying `roles/resourcemanager.projectIamAdmin`, `roles/iam.roleAdmin`, `roles/cloudkms.admin` and `roles/storage.admin`. The sibling module `iac/federation/gcp-target/terraform/main.tf:203-205` documents exactly this hazard and tells the reader to keep the pool dedicated; the CI pool does the opposite. +- evidence: + ```hcl + member = "principalSet://iam.googleapis.com/${google_iam_workload_identity_pool.github[0].name}/attribute.repository/${var.github_repo}" + ``` +- suggested fix: Map `attribute.ref` into the grant as well (`.../attribute.repository/` becomes a two-attribute condition or the member pins `attribute.ref`), and rename the pool to something CUDly-scoped so an unrelated provider cannot be added to it. +- verdict: PLAUSIBLE — every code fact checks out: the pool id is the generic `"github-actions"` (github_oidc.tf:4), the ref pin lives only in the provider's `attribute_condition` (:37), the `workloadIdentityUser` member is the ref-less `attribute.repository` principalSet (:46), the impersonated SA holds `roles/resourcemanager.projectIamAdmin`, `roles/iam.roleAdmin`, `roles/cloudkms.admin` and `roles/storage.admin` (service_account.tf:12,21,28,35), and iac/federation/gcp-target/terraform/main.tf:203-205 documents the pool-scoping hazard verbatim. The escalation itself needs a second provider to be added to that pool, which no code in the repo does and which I cannot establish from source; today the single provider's condition still pins repo and ref. +- severity-adjusted: low — with one provider in the pool this is a defense-in-depth gap, not a reachable escalation. +- issue: (pending cross-reference) + +### A13c-021 Rotation Lambda's RDS grant wildcards the account segment +- category: security +- severity: low +- location: terraform/modules/secrets/aws/main.tf:390 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: the ARN's account field is `*`, so the statement authorises + `rds:ModifyDBInstance` against any DB instance in any account that would accept the principal. + IAM identity policies do not cross account boundaries on their own, so the practical reach is + today's account, but the pattern also fails to name the one instance being rotated: any + instance in the account, including unrelated stacks', is in scope. The module already knows the + target (`var.rds_cluster_id`, which gates this very resource at line 376) and does not use it. +- evidence: + ```hcl + Action = [ + "rds:ModifyDBInstance", + "rds:DescribeDBInstances" + ] + Resource = "arn:aws:rds:${var.region}:*:db:*" + ``` +- suggested fix: build the ARN from `data.aws_caller_identity.current.account_id` and + `var.rds_cluster_id`. +- verdict: CONFIRMED — `Resource = "arn:aws:rds:${var.region}:*:db:*"` at secrets/aws/main.tf:390 + wildcards both the account and the instance, and the same resource's own count at :376 already + holds the specific `var.rds_cluster_id`. Latent in practice: the rotation Lambda this role serves + cannot be applied at all (see A13c-003), so nothing assumes the role today. +- issue: (pending cross-reference) + +### A13c-022 GCP Cloud Run service account holds project-wide `roles/compute.viewer` +- category: security +- severity: low +- location: terraform/modules/compute/gcp/cloud-run/main.tf:431 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: the stated need is `compute.commitments.list` and `compute.machineTypes.list`. + `roles/compute.viewer` grants read on every Compute Engine resource in the project — instance + metadata (including startup scripts, a common credential-leak surface), disks, images, + snapshots, firewall rules and network topology. The module already demonstrates the narrower + pattern immediately below by defining a two-permission custom role for the write side, so the + read side is the outlier rather than a constraint of the platform. +- evidence: + ```hcl + resource "google_project_iam_member" "compute_viewer" { + project = var.project_id + role = "roles/compute.viewer" + member = "serviceAccount:${google_service_account.cloud_run.email}" + } + ``` +- suggested fix: extend `cudlyCommitmentWriter` (or add a sibling custom role) with + `compute.commitments.list/get` and `compute.machineTypes.list/get`, and drop the predefined role. +- verdict: CONFIRMED — the ungated project-wide grant is at cloud-run/main.tf:431-435 and its own + comment names only `compute.commitments.list` and `compute.machineTypes.list` as the need, while + the two-permission custom role for the write side sits immediately below at :454-463, proving the + narrow pattern is available in this module. +- issue: (pending cross-reference) + +### A13c-023 Container App identity holds `Reader` over the entire host subscription +- category: security +- severity: low +- location: terraform/modules/compute/azure/container-apps/main.tf:269 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: the grant is scoped to `/subscriptions/`, which is the same subscription + that holds CUDly's own Key Vault, PostgreSQL server, ACR and Container App environment. + A compromised runtime can therefore enumerate the whole control plane of its own deployment — + resource names, network layout, diagnostic settings, key vault metadata — not only the customer + workloads it is meant to inventory. The stated need ("host account ingested as a Self account") + is real, but it is an opt-in mode; the grant is unconditional. +- evidence: + ```hcl + resource "azurerm_role_assignment" "subscription_reader" { + scope = local.subscription_resource_id + role_definition_name = "Reader" + principal_id = azurerm_user_assigned_identity.container_app.principal_id + } + ``` +- suggested fix: gate it behind an `enable_self_account_ingestion` input defaulting false, so a + deployment that only manages customer subscriptions does not carry it. +- verdict: CONFIRMED — `azurerm_role_assignment.subscription_reader` (container-apps/main.tf:269) + has no `count` and is scoped to `local.subscription_resource_id` (:254), the same subscription + that holds the Key Vault, PostgreSQL server, ACR and Container App environment; the comment at + :265-268 states the Self-account rationale without gating on it. +- issue: (pending cross-reference) + +### A14-023 A migrate error is written unquoted into `$GITHUB_ENV` +- category: security +- severity: low +- location: .github/workflows/database-migration.yml:352 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: `2>&1` folds stderr into `VERSION`, and golang-migrate's failure output is multi-line. Writing a multi-line value as a single `NAME=value` line to `$GITHUB_ENV` lets the second and subsequent lines be parsed as further environment assignments for later steps in the job. It is only bounded today because this is the job's last step. The error text can also carry the connection string, so it is echoed to the log as well. +- evidence: + ```bash + VERSION=$(migrate -path "$MIGRATIONS_PATH" -database "$DB_URL" version 2>&1 || echo "unknown") + echo "Current migration version: $VERSION" + echo "MIGRATION_VERSION=$VERSION" >> "$GITHUB_ENV" + ``` +- suggested fix: Capture stdout only, or use the heredoc delimiter form (`{name}< 0 { + return int64(cd.MemoryGB * 1024), nil + } + return 0, fmt.Errorf("memoryMBFromDetails: MEMORY resource amount absent from recommendation Details (no MEMORY op in Recommender payload); cannot build CUD insert without explicit memory") + } + ``` +- suggested fix: accept both forms in the assertion (`case common.ComputeDetails` / `case *common.ComputeDetails` in a type switch, guarding the typed-nil pointer), and add a regression test that drives `PurchaseCommitment` with `Details` set to `&common.ComputeDetails{MemoryGB: 16}`. +- verdict: CONFIRMED — pkg/common/service_details_codec.go:127 returns `&ComputeDetails{}` for ServiceCompute and internal/purchase/execution.go:1068 assigns that pointer to `recommendation.Details`, so the value-type assertion at providers/gcp/services/computeengine/client.go:1392 is always false on the executed purchase path. +- issue: (pending cross-reference) + +### A08b-005 `ReservationsDetails` is a daily usage API, so `GetExistingCommitments` emits one Commitment per reservation per day +- category: money-path +- severity: critical +- location: providers/azure/services/cache/client.go:207 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd (identical in `cosmosdb/client.go:209`, `database/client.go:239`, `search/client.go:154`, `synapse/client.go:182`, `managedredis/client.go:209`) +- failure scenario: `armconsumption.ReservationDetailProperties` carries `UsageDate`, `ReservedHours` and `UsedHours` — it is one row per reservation per usage day, not a reservation inventory. With no `Filter` on `properties/UsageDate`, the pager returns the whole default window, so a single Redis reservation held for 30 days produces 30 `common.Commitment` values that all share the same `CommitmentID`. Every consumer that sums existing commitments to compute coverage, or that dedupes recommendations against existing commitments, sees roughly 30x the real committed capacity and suppresses purchases that should happen. +- evidence: + ```go + scope := fmt.Sprintf("subscriptions/%s", c.subscriptionID) + return client.NewListPager(scope, &armconsumption.ReservationsDetailsClientListOptions{}), nil + ``` +- suggested fix: use `armreservations.ReservationClient` / `ReservationOrderClient` (an inventory API, as `compute/exchange.go` already does) for existing commitments, or at minimum dedupe by `ReservationID` and pass a bounded `Filter` on `properties/UsageDate`. +- verdict: CONFIRMED — all six pagers pass an empty `ReservationsDetailsClientListOptions` (cache:208, cosmosdb:210, database:240, search:155, synapse:183, managedredis:210) and the SDK type is per-usage-day, carrying `UsageDate`, `ReservedHours` and `UsedHours` (armconsumption@v1.1.0 models.go:2229-2267); no converter dedupes on `ReservationID`. +- issue: (pending cross-reference) + +### A09-004 An absent PaymentDue is treated as $0, disabling the spend cap on an irreversible exchange +- category: money-path +- severity: critical +- location: pkg/exchange/exchange.go:305 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: `getQuoteWithAPI` leaves `PaymentDueUSD` nil whenever `out.PaymentDue` is nil or empty (line 261 gates the parse on a non-empty raw string). `resolvePaymentDue` then substitutes a fresh zero `big.Rat`, so `checkInitialQuote` and `checkReQuote` compare `0` against `MaxPaymentDueUSD` and always pass. If AWS omits `PaymentDue` on a genuinely non-zero exchange (an API change, a partial response, or a shape the SDK does not populate), `executeWithAPI` proceeds to `AcceptReservedInstancesExchangeQuote` with no effective cap and the charge is irreversible. The doc comment asserts the only cause is a zero-cost exchange; nothing verifies that. +- evidence: + ```go + func resolvePaymentDue(q *ExchangeQuoteSummary) *big.Rat { + if q.PaymentDueUSD != nil { + return q.PaymentDueUSD + } + return new(big.Rat) + } + ``` +- suggested fix: make an absent `PaymentDue` an error in `checkInitialQuote`/`checkReQuote` rather than a zero, or require `IsValidExchange && PaymentDueRaw == "0"`-style positive evidence of a zero-cost exchange before proceeding. +- verdict: CONFIRMED — traced end to end: pkg/exchange/exchange.go:260 gates the parse on `s.PaymentDueRaw != ""` so `PaymentDueUSD` stays nil, `resolvePaymentDue` (exchange.go:303-308) substitutes `new(big.Rat)`, and both `checkInitialQuote` (line 313) and `checkReQuote` (line 330) compare that zero against the cap, so `executeWithAPI` reaches `AcceptReservedInstancesExchangeQuote` at line 372 with no effective ceiling. +- issue: (pending cross-reference) + +### A12-001 Purchase modal's Term and Payment selects mutate the submitted rec without re-pricing it +- category: money-path +- severity: critical +- location: frontend/src/recommendations.ts:5412 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: A 3yr all-upfront rec (`upfront_cost` $36,000, `monthly_cost` 0) is opened in the purchase modal. The user changes the row's Term select to 1yr. `live.term` becomes 1, `rebuildPaymentOptions` re-derives the payment list, and nothing else changes. The Upfront / Monthly Cost / Eff. Savings / Eff. % cells are static text nodes built once in `renderPurchaseModalRow` and are never re-rendered; `updatePurchaseModalTotals` reads the same untouched `rec.upfront_cost` / `rec.savings`. The POST therefore carries `term: 1` with the 3-year price, and the totals row, the approval email and the stored execution all record a price belonging to a term the user did not buy. The Payment select (line 5430) has the same shape: flipping `no-upfront` to `all-upfront` leaves `upfront_cost` at the no-upfront value, so the modal shows "$0 upfront" for an all-upfront purchase. `recTotalCommitment` in `internal/api/handler_purchases.go:2247` multiplies the submitted `MonthlyCost` by the submitted `Term`, so the `MaxPurchaseAmount` permission constraint is also evaluated against a mismatched pair. +- evidence: + ```ts + termSelect.addEventListener('change', () => { + const live = currentPurchaseRecommendations[idx]; + if (!live) return; + const newTerm = parseInt(termSelect.value, 10) === 3 ? 3 : 1; + live.term = newTerm; + rebuildPaymentOptions(paymentSelect, live.provider as CompatProvider, + live.service, newTerm, (live.payment ?? '') as CompatPayment); + live.payment = paymentSelect.value; + }); + ``` +- suggested fix: On a term/payment change, look up the matching variant from the loaded recommendation set (the API already fans out per `(term, payment)` cell) and swap the whole rec in, then re-render the row and the totals; if no matching variant exists, disable that option rather than keeping stale prices. +- verdict: CONFIRMED — The term/payment handlers mutate only `live.term`/`live.payment` (frontend/src/recommendations.ts:5412-5435) while the Upfront/Monthly/Eff. cells are one-shot text nodes built in renderPurchaseModalRow (frontend/src/recommendations.ts:5339-5364), and app.ts:385-391 spreads the same rec into the POST, so the submitted term reaches recTotalCommitment (internal/api/handler_purchases.go:2255) paired with the other term's price. +- issue: (pending cross-reference) + +### A12-002 Fan-out modal says an incompatible bucket "will be skipped", then submits it +- category: money-path +- severity: critical +- location: frontend/src/recommendations.ts:4336 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: A bulk selection produces two buckets, one of which fails `isBucketPaymentCompatible` (for example AWS RDS 3yr + no-upfront). `renderFanOutBucketSection` renders "Invalid combo: … This bucket will be skipped." `handleFanOutExecute` in `frontend/src/app.ts:534` maps over `buckets` with no compatibility filter, so both buckets are POSTed. The backend only warns for that combination (`cmd/validators.go:warnRDS3YearNoUpfront`), so an approval email is minted for a commitment the UI just told the user would not be submitted. The confirm dialog and the "Total upfront / Total savings" header (`computeFanOutTotals`, line 4295) also count the skipped bucket's money. +- evidence: + ```ts + status.textContent = compat + ? `${b.capacityPercent}% capacity · ${b.term}yr · ${b.payment}` + : `Invalid combo: ${b.provider} / ${serviceLabel} doesn't support ${b.term}yr + ${b.payment}. This bucket will be skipped.`; + // app.ts:534 — no filter: + const promises = buckets.map((b) => api.executePurchase(b.recs.map(...), b.capacityPercent)); + ``` +- suggested fix: Filter incompatible buckets out of `currentFanOutBuckets` before `handleFanOutExecute` runs (or exclude them from `getFanOutBuckets`), and exclude them from `computeFanOutTotals` and the email count. +- verdict: CONFIRMED — handleFanOutExecute maps over every bucket with no compatibility predicate (frontend/src/app.ts:534-544) even though handleBulkPurchaseClick deliberately routes incompatible buckets into the modal (frontend/src/recommendations.ts:3976-3988) and renderFanOutBucketSection promises they are skipped (frontend/src/recommendations.ts:4334-4337); no rejection exists on the API side, only the CLI-flag warning at cmd/validators.go:150. +- issue: (pending cross-reference) + +### A01-001 MaxPurchaseAmount cap is enforced against client-asserted dollar amounts, not the priced commitment +- category: money-path +- severity: high +- location: internal/api/handler_purchases.go:2255 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: A user whose execute:purchases permission carries MaxPurchaseAmount=1000 and execute-own (or execute-any) POSTs `/api/purchases/execute` with `execute_mode:"direct"` and a rec `{provider:aws, service:ec2, resource_type:"m5.24xlarge", count:100, term:3, payment:"all-upfront", upfront_cost:1, monthly_cost:null}`. `recTotalCommitment` sums the request's own `upfront_cost`/`monthly_cost` (1.0), `requireNonZeroCommitment` only rejects an exact 0, the constraint check passes (1 <= 1000), `directExecutePurchase` runs `ApproveAndExecute`, and the provider buys 100 three-year RIs at the real list price. Nothing in `validateExecutePurchaseRequest` re-derives the amount from the stored recommendation (`rec.ID` is never looked up) or from provider pricing. The retry path inherits the same self-reported totals. +- evidence: + ```go + func recTotalCommitment(rec *config.RecommendationRecord) float64 { + total := rec.UpfrontCost + if rec.Term > 0 && rec.MonthlyCost != nil { + total += *rec.MonthlyCost * float64(rec.Term*12) + } + return total + } + // ... + func requireNonZeroCommitment(sets []auth.PermissionConstraints) error { + if len(sets) > 0 && sets[0].MaxPurchaseAmount == 0 { + ``` +- suggested fix: Before the constraint check, resolve each rec by `rec.ID` against the cached recommendations store (the scheduler's `ListRecommendations`/`GetRecommendationByID`) and take upfront/monthly cost from the stored row scaled by the requested count, refusing recs that do not resolve; treat the client-sent costs as untrusted input. +- verdict: CONFIRMED — validateExecutePurchaseRequest (internal/api/handler_purchases.go:2127-2185) never resolves rec.ID against any store (no GetRecommendationByID caller in internal/api or internal/purchase), purchaseConstraintSets:2221-2242 caps on recTotalCommitment of the request's own UpfrontCost/MonthlyCost, and the AWS purchase at providers/aws/services/ec2/client.go:155-161 sends only offering ID + count with no LimitPrice, so nothing downstream re-prices the batch. +- issue: (pending cross-reference) + +### A01-004 PUT /api/plans/{id} rebuilds the ramp schedule from scratch, resetting CurrentStep/StartDate and re-arming the ramp +- category: money-path +- severity: high +- location: internal/api/handler_plans.go:281 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: A weekly-25pct plan is on `CurrentStep=3` (three steps bought). The operator renames it through the Edit modal, which issues PUT (frontend/src/plans.ts:1420). `updatePlan` builds `plan := req.toPurchasePlan()`, whose `buildRampSchedule` returns the preset with `CurrentStep=0`, `StartDate=now` (types.go:797-801) and `NextExecutionDate=now+7d`; `LastExecutionDate`/`LastNotificationSent` are left nil. `UpdatePurchasePlan` writes all of it (`ramp_schedule = $7`, `next_execution_date = $9`, `last_execution_date = $10`, store_postgres.go:763-774). The scheduler's `getOrCreateExecution` then finds no row for the new date and mints a `StepNumber=1` execution (notifications.go:139), so steps 1..4 are notified and bought again. `refuseOccupiedRampSteps` guards only the manual create endpoint. `TestHandler_updatePlan` (handler_plans_test.go:551) never asserts ramp preservation. +- evidence: + ```go + // Create new plan from request + plan := req.toPurchasePlan() + plan.ID = planID + + // Preserve timestamps from existing plan + plan.CreatedAt = existingPlan.CreatedAt + plan.UpdatedAt = time.Now() + ``` +- suggested fix: Carry `existingPlan.RampSchedule.CurrentStep`, `StartDate`, `NextExecutionDate`, `LastExecutionDate` and `LastNotificationSent` onto the rebuilt plan (or only rebuild the ramp when the requested ramp type/params actually changed), and add a test asserting a rename leaves `CurrentStep` untouched. +- verdict: CONFIRMED — frontend/src/plans.ts:1420 calls api.updatePlan which is PUT (frontend/src/api/plans.ts:45-50); updatePlan (internal/api/handler_plans.go:281-300) rebuilds via req.toPurchasePlan(), whose buildRampSchedule (internal/api/types.go:797-801) copies the preset (weekly-25pct carries no CurrentStep, internal/config/types.go:241-246) with StartDate=now and leaves LastExecutionDate/LastNotificationSent nil; UpdatePurchasePlanTx (internal/config/store_postgres.go:763-786) overwrites ramp_schedule, next_execution_date, last_execution_date and last_notification_sent, and getOrCreateExecution stamps StepNumber=CurrentStep+1 (internal/purchase/notifications.go:139); TestHandler_updatePlan (handler_plans_test.go:551-590) asserts only ID and Name. +- issue: (pending cross-reference) + +### A07-001 CE-supplied OfferingID short-circuits every Savings Plans purchase validation +- category: money-path +- severity: high +- location: providers/aws/services/savingsplans/client.go:425 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: `findOfferingID` returns `spDetails.OfferingID` before `resolveSPPlanType`, `convertTermToSeconds` and `convertPaymentOption` run. `extractEC2SPFields` (parser_sp.go:268) populates `OfferingID` on every EC2Instance SP rec, so for those recs the scoped client's "reject mismatches to prevent buying the wrong product" check at client.go:279 never executes, and neither does term or payment-option validation. With a persisted rec whose `Term` was later changed from `1yr` to `3yr` by config (or a rec routed to a client scoped to a different plan type), `PurchaseCommitment` sends the stale CE offering ID to `CreateSavingsPlan` and buys the term/payment/plan-type the offering encodes, not the one the recommendation now says. +- evidence: + ```go + if spDetails.OfferingID != "" { + log.Printf("purchase[%s]: SavingsPlans findOfferingID: using CE-provided OfferingID %s (skipping DescribeSavingsPlansOfferings)", tag, spDetails.OfferingID) + return spDetails.OfferingID, nil + } + planType, err := c.resolveSPPlanType(spDetails.PlanType) + ``` +- suggested fix: run `resolveSPPlanType`, `convertTermToSeconds` and `convertPaymentOption` before the short-circuit so a malformed or mis-scoped rec still fails loud, and keep the fast path only for the offering lookup itself. +- verdict: CONFIRMED — providers/aws/services/savingsplans/client.go:425-428 returns before the three validators at :430-441, and providers/aws/recommendations/parser_sp.go:268 + :397 populate `OfferingID` on every EC2Instance SP rec, so the scoped-client mismatch check at client.go:279 is bypassed on exactly those recs. +- issue: (pending cross-reference) + +### A07-007 ElastiCache and MemoryDB commitments carry no Engine, so the duplicate-purchase guard never matches +- category: money-path +- severity: high +- location: providers/aws/services/elasticache/client.go:92 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: `recfilter.dedupeKey` is built from `(resourceType, region, engine, deployment)`. For a cache recommendation the engine comes from `common.EngineFromDetails(rec.Details)` and is `"redis"`; the commitment side reads `c.Engine`, which neither `elasticache.GetExistingCommitments` nor `memorydb.GetExistingCommitments` ever sets. The two keys can never be equal, so `AdjustRecommendationsForExisting` never suppresses a cache recommendation and a reserved cache node bought minutes ago is recommended and bought again on the next run. RDS sets `Engine` from `instance.ProductDescription` (rds/client.go:110) and does not have this problem; `types.ReservedCacheNode.ProductDescription` is available and simply unread. +- evidence: + ```go + commitment := common.Commitment{ + Provider: common.ProviderAWS, + CommitmentID: aws.ToString(node.ReservedCacheNodeId), + CommitmentType: common.CommitmentReservedInstance, + Service: common.ServiceCache, + Region: c.region, + ResourceType: aws.ToString(node.CacheNodeType), + Count: int(aws.ToInt32(node.CacheNodeCount)), + ``` +- suggested fix: set `Engine: aws.ToString(node.ProductDescription)` in the ElastiCache mapping and the equivalent field in `memorydb.GetExistingCommitments`, and add a dedupe test that fails when `Engine` is blank. +- verdict: CONFIRMED — neither providers/aws/services/elasticache/client.go:92-102 nor providers/aws/services/memorydb/client.go:88-99 sets `Engine`, while pkg/recfilter/dedupe.go:105 keys the existing map on `NormalizeEngineName(c.Engine)` (empty) and :135-137 keys the rec on `EngineFromDetails(rec.Details)` (`"redis"`); `types.ReservedCacheNode.ProductDescription` does exist in the SDK (elasticache@v1.50.3/types/types.go:1758) and is never read. +- issue: (pending cross-reference) + +### A07-012 EC2 ladder coverage blends in ElastiCache, OpenSearch, Redshift and MemoryDB pools +- category: money-path +- severity: high +- location: providers/aws/ladder/layer_states.go:307 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: `computeEC2CoveragePct` selects map entries by "one colon in the key". But `coverageServiceFilters` (coverage.go:28) populates that same map with ElastiCache, OpenSearch, Redshift and MemoryDB pools, all keyed `region:instance_type` with exactly one colon. An account with a fully covered Redshift or ElastiCache fleet in the region produces a high blended `CoveragePct` on `LayerConvertibleRI`, which the ladder engine reads as "the EC2 convertible-RI buffer is covered" and uses to gate buy and reshape decisions. This is the same cross-service blend PR #1361 fixed for utilization (see `utilsForConvertibleRIs` 30 lines below), left unfixed on the coverage side. +- evidence: + ```go + for key, cov := range coverageMap { + if !strings.HasPrefix(key, prefix) { + continue + } + // Exclude RDS keys (contain extra ":" segments for engine:deployment). + // EC2 pool keys are exactly "region:instance_type" (one colon). + if strings.Count(key, ":") != 1 { + continue + } + ``` +- suggested fix: have `GetRICoverageMap` record the CE service alongside each pool (or key non-RDS entries by `region:service:instance_type`) so the EC2 aggregate can select only EC2 pools. +- verdict: CONFIRMED — `coverageServiceFilters` (providers/aws/recommendations/coverage.go:28-33) lists ElastiCache, OpenSearch, Redshift and MemoryDB alongside EC2, `GetRICoverageMap` :193-196 loops all five, and `fetchCoverageForServiceRegion` :250 writes every one of them as `poolKey(region, instType)` — exactly one colon — which `computeEC2CoveragePct` (providers/aws/ladder/layer_states.go:296-307) then accepts into the `LayerConvertibleRI` aggregate. +- issue: (pending cross-reference) + +### A08-005 Azure retail-price extraction is last-item-wins across Spot/Windows/Linux SKUs and silently mixes currencies +- category: money-path +- severity: high +- location: providers/azure/services/compute/client.go:765 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: the `armSkuName eq ''` filter returns every meter for the size, including Spot, Low Priority, Windows and Linux variants. The loop keeps the LAST matching item for both `onDemand` and `reservation` with no disambiguation on `SKUName`/`MeterName`/`ProductName` (all present on `pricing.RetailPriceItem`), so `GetOfferingDetails` can report a Spot on-demand rate against a standard reservation price, and `savingsPercentage` (client.go:719) is computed from that pair. `currency` is likewise the last item's code and is not required to match the item the prices came from; when no item carries a code it stays hardcoded `"USD"`. The identical shape is in database/client.go:617, cache/client.go:599, cosmosdb/client.go:597, search/client.go:510, synapse/client.go:456 and managedredis/client.go:521. +- evidence: + ```go + currency = "USD" + for _, item := range items { + if item.CurrencyCode != "" { currency = item.CurrencyCode } + if item.ReservationTerm == termStr { + reservation = item.RetailPrice + } else if item.Type == "Consumption" { + onDemand = item.UnitPrice + } + } + ``` +- suggested fix: reject Spot/Low-Priority meters and require the on-demand and reservation items to share a currency and SKU name; error rather than defaulting `currency` to `"USD"` when no item reports one. +- verdict: CONFIRMED — `extractVMPricing` (compute/client.go:765) is last-item-wins across the whole `$filter` result with no SKUName/MeterName/ProductName disambiguation even though those fields exist on the item (internal/pricing/types.go:16-20), the filter itself is only serviceName+armRegionName+armSkuName (client.go:692), and the same loop shape is at database:617, cache:599, cosmosdb:597, search:510, synapse:456 and managedredis:521. +- issue: (pending cross-reference) + +### A08-006 ExpandPaymentVariants fabricates 100% savings when the provider omitted the commitment cost +- category: money-path +- severity: high +- location: providers/azure/internal/recommendations/converter.go:423 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: a Legacy payload with `CostWithNoReservedInstances = 4000` and no `TotalCostWithReservedInstances` yields `CommitmentCost == 0`. `ExpandPaymentVariants` then computes `savings = 4000 - 0 = 4000` and `savingsPct = 100`, overwriting the provider-reported `NetSavings` the extractor had populated, and emits both variants as a 100%-savings recommendation. Every Azure service converter calls this unconditionally (e.g. compute/client.go:218), so nothing upstream validates it. The function's own doc claims the opposite: "if CommitmentCost is zero both variants are still emitted with zero costs". +- evidence: + ```go + var savingsPct float64 + var savings float64 + if totalOnDemand != 0 { + savings = totalOnDemand - totalReservation + savingsPct = savings / totalOnDemand * 100 + } + ``` +- suggested fix: when `totalReservation == 0`, keep `base.EstimatedSavings` (the provider's `NetSavings`) and leave `SavingsPercentage` unset, or drop the recommendation; either way stop deriving savings from an absent commitment cost. +- verdict: CONFIRMED — `extractLegacy` (converter.go:156) leaves CommitmentCost at 0 when TotalCostWithReservedInstances is nil (an absence the repo itself treats as expected — `deriveCoveredMonthlyCost`, converter.go:90, exists for it) and `ExpandPaymentVariants` (converter.go:415-425) then overwrites the provider's NetSavings with `totalOnDemand - 0` and a 100% SavingsPercentage on both variants. +- issue: (pending cross-reference) + +### A08-007 Azure Savings Plan purchases hardcode CurrencyCode "USD" on the commitment body +- category: money-path +- severity: high +- location: providers/azure/services/savingsplans/client.go:261 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: `PurchaseCommitment` sends `spDetails.HourlyCommitment` with `CurrencyCode: "USD"` regardless of the billing account's actual billing currency. For a tenant billed in EUR/GBP/KWD, a recommendation whose hourly commitment was derived in the billing currency is committed as that many US dollars. The same literal is on the validate body (client.go:355) so `ValidateOffering` cannot catch the mismatch, and `GetOfferingDetails` reports `Currency: "USD"` (client.go:445) so the UI corroborates the wrong denomination. `auth.PermissionConstraints.MaxPurchaseAmount` is USD-denominated with no currency field, so a non-USD amount also compares against the spend cap in the wrong units. +- evidence: + ```go + Commitment: &armbillingbenefits.Commitment{ + Amount: &hourlyAmount, + CurrencyCode: toPtr("USD"), + Grain: &grain, + }, + ``` +- suggested fix: carry the currency on `common.SavingsPlanDetails` and reject the purchase when it is absent or not USD, rather than stamping USD onto whatever amount arrived. +- verdict: CONFIRMED — the literal is stamped on the purchase body (savingsplans/client.go:261), the validate body (client.go:355) and the reported offering currency (client.go:445), and `common.SavingsPlanDetails` (pkg/common/types.go:612-634) carries no currency field for the purchase to check against. +- issue: (pending cross-reference) + +### A08-008 Modern (MCA) recommendation extraction discards the currency Azure reported +- category: money-path +- severity: high +- location: providers/azure/internal/recommendations/converter.go:360 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: `armconsumption.Amount` carries `{Currency, Value}`. `amountValue` unwraps `.Value` and throws `.Currency` away, so an MCA billing account billed in EUR produces `OnDemandCost`, `CommitmentCost` and `EstimatedSavings` in EUR that are stored, ranked, summed and compared against USD-denominated spend caps as if they were dollars. Nothing downstream can detect it because `common.Recommendation` has no currency field on this path. The comment states the assumption ("downstream consumers assume a single-currency view per subscription") but there is no check that enforces it. +- evidence: + ```go + func amountValue(a *armconsumption.Amount) float64 { + if a == nil || a.Value == nil { + return 0 + } + return *a.Value + } + ``` +- suggested fix: read `a.Currency` and return an error (or drop the recommendation with a loud log) when it is set to anything other than USD, so a non-USD tenant fails closed instead of silently mis-denominating a money path. +- verdict: CONFIRMED — `amountValue` (converter.go:360) reads only `.Value`, and `extractModern` (converter.go:206-215) feeds it straight into OnDemandCost, CommitmentCost and EstimatedSavings with no currency check anywhere on the path; `amountValuePtr` (converter.go:369) discards it too. +- issue: (pending cross-reference) + +### A08b-006 Azure commitments fabricate `State: "active"` and `Region` from fields the SDK response does not carry +- category: money-path +- severity: high +- location: providers/azure/services/cache/client.go:251 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd (identical in `cosmosdb/client.go:252`, `database/client.go:282`, `search/client.go:197`, `synapse/client.go:223`, `managedredis/client.go:223`) +- failure scenario: `ReservationDetailProperties` has no state and no region field, so both values are invented. The reservation listing is subscription-wide and unfiltered by region, so a `westeurope` reservation returned to a client constructed for `eastus` is stamped `Region: "eastus"` and matched against `eastus` recommendations that it does not discount. `State: "active"` is unconditional, so an expired or cancelled reservation still counts as live coverage and suppresses a needed repurchase. +- evidence: + ```go + commitment := &common.Commitment{ + Provider: common.ProviderAzure, + Account: c.subscriptionID, + CommitmentType: common.CommitmentReservedInstance, + Service: common.ServiceCache, + Region: c.region, + State: "active", + } + ``` +- suggested fix: source region and state from an inventory API that reports them (`armreservations.ReservationResponse` has `Location` and `ProvisioningState`), and leave them empty rather than guessing when unavailable. +- verdict: CONFIRMED — `ReservationDetailProperties` (armconsumption@v1.1.0 models.go:2229-2267) carries neither a state nor a region field, the pager scope is the whole subscription with no region filter (cache:207-208), and the converter hardcodes `Region: c.region` and `State: "active"` (cache:172-179, managedredis:223-230, synapse:223-230). +- issue: (pending cross-reference) + +### A08b-007 Azure commitments never populate Count, StartDate, EndDate or Cost +- category: money-path +- severity: high +- location: providers/azure/services/cache/client.go:260 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd (identical in the five sibling converters) +- failure scenario: `common.Commitment` has `Count`, `StartDate`, `EndDate` and `Cost` (`pkg/common/types.go:436-440`); the converters set only `CommitmentID` and `ResourceType`. `TotalReservedQuantity` is present on the SDK response and dropped, so every Azure commitment reports `Count: 0` — coverage arithmetic that multiplies count by a rate yields zero committed capacity. `EndDate` stays at the zero `time.Time`, so any "is this commitment still active" filter comparing against `time.Now()` classifies every Azure commitment as long expired. +- evidence: + ```go + if props.ReservationID != nil { + commitment.CommitmentID = *props.ReservationID + } + if props.SKUName != nil { + commitment.ResourceType = *props.SKUName + } + return commitment + ``` +- suggested fix: populate `Count` from `TotalReservedQuantity`, and take start/end/cost from a reservation-order lookup; do not emit a Commitment at all when the required money fields cannot be filled. +- verdict: CONFIRMED — the converters write only `CommitmentID` and `ResourceType` (cache/client.go:181-188, managedredis:231-234, synapse:231-234); `Count`, `StartDate`, `EndDate` and `Cost` exist on the struct at pkg/common/types.go:436-441, and `TotalReservedQuantity` is present on the SDK response (models.go:2260) but never read. +- issue: (pending cross-reference) + +### A08b-008 Cosmos DB pricing ignores the SKU it was asked to price +- category: money-path +- severity: critical +- location: providers/azure/services/cosmosdb/client.go:532 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: `getCosmosPricing(ctx, sku, region, termYears)` builds a filter on service name and region only; the `sku` parameter is never read. The Retail Prices response therefore contains every Cosmos DB meter in the region (provisioned RU/s, autoscale RU/s, transactional storage, analytical storage, backup, restore), and `extractCosmosPricing` keeps the last matching item of each kind. `GetOfferingDetails` then reports that arbitrary meter's price as `TotalCost`/`EffectiveHourlyRate` for the specific reservation being priced — for instance a backup-storage price presented as the cost of a 100 RU/s three-year reservation. +- evidence: + ```go + func (c *CosmosDBClient) getCosmosPricing(ctx context.Context, sku, region string, termYears int) (*CosmosPricing, error) { + filter := fmt.Sprintf("serviceName eq 'Azure Cosmos DB' and armRegionName eq '%s'", region) + priceData, err := c.fetchAzurePricing(ctx, filter) + ``` +- suggested fix: add an exact `armSkuName eq` (or `skuName eq`) clause for the SKU, and reject the lookup when more than one distinct reservation price survives the filter rather than taking the last one. +- verdict: CONFIRMED — `sku` appears nowhere in the body of `getCosmosPricing` (cosmosdb/client.go:531-567); the filter is service plus region only and `extractCosmosPricing` (597-614) overwrites on every match, so the last item wins. +- severity-adjusted: high — no in-repo caller reaches `GetOfferingDetails`; grepping the whole tree outside `providers/` and tests returns only the interface declaration at pkg/provider/interface.go:50, so the wrong quote is latent rather than shipping today. +- issue: (pending cross-reference) + +### A08b-009 Azure Search pricing ignores the SKU it was asked to price +- category: money-path +- severity: critical +- location: providers/azure/services/search/client.go:447 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: identical shape to A08b-008. `sku` is accepted and used only inside an error string at line 470; the filter selects every Azure Cognitive Search meter in the region. Pricing a `standard3` reservation returns whichever `basic` or `storage_optimized_l2` item happened to sort last in the response, so `TotalCost` can be off by an order of magnitude in either direction on a path that feeds purchase previews. +- evidence: + ```go + func (c *SearchClient) getSearchPricing(ctx context.Context, sku, region string, termYears int) (*SearchPricing, error) { + filter := fmt.Sprintf("serviceName eq 'Azure Cognitive Search' and armRegionName eq '%s'", region) + ``` +- suggested fix: filter on the SKU, and fail loud when the filtered result set contains more than one candidate reservation price. +- verdict: CONFIRMED — the filter at search/client.go:447 carries service and region only, `sku` is read solely inside the error string at line 470, and `extractSearchPricing` (510-527) is last-item-wins. +- severity-adjusted: high — same reachability caveat as A08b-008: `GetOfferingDetails` has no caller outside the providers' own tests (pkg/provider/interface.go:50). +- issue: (pending cross-reference) + +### A08b-010 Retail-price extraction is last-item-wins with no SKU or unit-of-measure check +- category: money-path +- severity: high +- location: providers/azure/services/cache/client.go:599 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd (same in `cosmosdb/client.go:597`, `database/client.go:617`, `search/client.go:510`, `managedredis/client.go:520`, `synapse/client.go:456`) +- failure scenario: the loop overwrites `reservation` and `onDemand` on every match, so the final values come from whichever item the API returned last. In cache and managedredis the query uses `contains(armSkuName, 'Premium_P1')`, which also matches `Premium_P10` through `Premium_P15`, so the price attributed to a P1 reservation can be a P15 price. `UnitOfMeasure` is never inspected, so a "100 Hours" meter and a "1 Hour" meter are treated as interchangeable. On-demand is read from `UnitPrice` while the reservation is read from `RetailPrice`, two different fields of the same record, and the resulting savings percentage compares them directly. +- evidence: + ```go + for _, item := range items { + if item.CurrencyCode != "" { + currency = item.CurrencyCode + } + if item.ReservationTerm == termStr { + reservation = item.RetailPrice + } else if item.Type == "Consumption" { + onDemand = item.UnitPrice + } + } + ``` +- suggested fix: require an exact `ArmSKUName` match against the requested SKU, assert a single `UnitOfMeasure`, and error when two candidate items disagree instead of letting the last one win. +- verdict: CONFIRMED — the loops overwrite unconditionally at cache:603-613, cosmosdb:601-611, search:514-524, managedredis:523-532 and synapse:463-475; `contains(armSkuName, '%s')` is the filter at cache:573 and managedredis:468; `UnitOfMeasure` exists on the shared item (providers/azure/internal/pricing/types.go:21) and is read nowhere; on-demand comes from `UnitPrice` while the reservation comes from `RetailPrice`. +- issue: (pending cross-reference) + +### A08b-014 Cloud Storage prices a per-GiB-month SKU as if it were per-hour, inflating cost by 730x +- category: money-path +- severity: high +- location: providers/gcp/services/cloudstorage/client.go:356 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: GCS storage SKUs are denominated per GiB-month (`unitOfMeasure` "GiBy.mo"); none are per-hour. Multiplying the catalog unit price by `8760 * termYears` therefore converts a monthly rate into a fictitious term total 730x too large per year. `CommitmentPrice` and `OnDemandPrice` both flow into `common.OfferingDetails.TotalCost` and into `rec.CommitmentCost` via `fillStoragePricing`, and `HourlyRate` labels a per-GiB-month figure as an hourly rate. +- evidence: + ```go + hoursInTerm := 8760.0 * float64(termYears) + commitmentPriceTerm := commitmentPrice * hoursInTerm + savingsPercentage := calculateStorageSavingsPercentage(onDemandPrice, hoursInTerm, commitmentPriceTerm) + ``` +- suggested fix: read `PricingExpression.UsageUnit` and scale by the unit the SKU actually uses (months for GiBy.mo), erroring when the unit is unrecognized rather than assuming hours. +- verdict: CONFIRMED — cloudstorage/client.go:356-368 multiplies the catalog unit price by `8760 * termYears` and returns the raw per-unit price as `HourlyRate`; `extractStoragePriceFromSKU` (415-432) reads only `TieredRates[0].UnitPrice` and never touches `PricingExpression.UsageUnit`, so no unit check exists anywhere on the path. Unlike the sibling clients this one is live: `fillStoragePricing` (497-512) runs inside `convertGCPRecommendation`. +- issue: (pending cross-reference) + +### A08b-024 Azure Cache for Redis is enumerated by two clients, so its reservations and recommendations are counted twice +- category: money-path +- severity: high +- location: providers/azure/services/managedredis/client.go:41 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd (twin at `cache/client.go:157`; both registered at `providers/azure/provider.go:521-522` and dispatched at `provider.go:586-591`) +- failure scenario: both clients issue the identical Consumption filter `properties/resourceType eq 'RedisCache'` and both classify existing reservations with `strings.Contains(strings.ToLower(*props.SKUName), "redis")`. `SupportedServices()` lists `ServiceCache` and `ServiceMemoryDB`, so a caller iterating supported services receives every Redis reservation twice under one `CommitmentID` and every Redis recommendation twice under two `ServiceType` values. Acting on both recommendations buys the same reserved capacity twice. +- evidence: + ```go + func (c *ManagedRedisClient) recommendationsListArgs() (string, *armconsumption.ReservationRecommendationsClientListOptions) { + scope := fmt.Sprintf("/subscriptions/%s", c.subscriptionID) + filter := "properties/scope eq 'Shared' and properties/resourceType eq 'RedisCache'" + return scope, &armconsumption.ReservationRecommendationsClientListOptions{Filter: &filter} + } + ``` +- suggested fix: pick one client as the owner of Azure Cache for Redis and have the other return empty, or split the filter so each covers a disjoint SKU family. +- verdict: CONFIRMED — the recommendation filter is byte-identical at cache:157 and managedredis:41, the reservation classifier is `strings.Contains(strings.ToLower(*props.SKUName), "redis")` at cache:247 and managedredis:220, and both `ServiceCache` and `ServiceMemoryDB` are advertised (provider.go:521-522) and dispatched to the two separate clients (provider.go:586-587 and 590-591). +- issue: (pending cross-reference) + +### A08b-030 The vCPU amount is selected by "not memory", so an accelerator or local-SSD amount is read as the vCPU count +- category: money-path +- severity: high +- location: providers/gcp/services/computeengine/client.go:1296 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: `memoryMBFromOperationGroups` selects positively (`isMemoryAmountOp` true), but `vcpuCountFromOperationGroups` selects by exclusion: any commitment operation whose path contains "amount" and whose path filter is not MEMORY is returned as the vCPU count. A recommendation carrying an `ACCELERATOR` or `LOCAL_SSD` amount op — and accelerator families are exactly the ones `familyRequiredResource` exists to describe — returns that amount as `rec.Count`. The count is what `GroupCommitments` sums into the purchased vCPU total, so the resulting CUD commits to the wrong number of vCPUs. Ordering decides which wins, since the first match returns. +- evidence: + ```go + if !strings.Contains(strings.ToLower(op.GetPath()), "amount") { + continue + } + if isMemoryAmountOp(op) { + continue + } + if v := op.GetValue(); v != nil { + ``` +- suggested fix: give the vCPU selector its own positive path-filter test for `VCPU`, mirroring `isMemoryAmountOp`, and error when the payload carries an amount op of an unhandled resource type. +- verdict: CONFIRMED — `vcpuCountFromOperationGroups` (computeengine:1287-1307) accepts any commitment op whose path contains "amount" and is not `isMemoryAmountOp`, returning on the first match, while `isMemoryAmountOp` (1269-1281) and its memory sibling select positively; the resulting `rec.Count` is summed into `agg.vcpus` at GroupCommitments:582 and becomes the purchased VCPU amount at 595. +- issue: (pending cross-reference) + +### A08b-032 An empty idempotency token silently produces a non-idempotent, second-granularity commitment name +- category: money-path +- severity: high +- location: providers/gcp/services/computeengine/client.go:1415 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd (consumer at `client.go:787`) +- failure scenario: the whole dedupe mechanism is the name collision — the same token yields the same name and GCP rejects the duplicate. With an empty token the name becomes `cud-`, so a retried purchase a minute later gets a different name and creates a second commitment for the same capacity, with no error. In the other direction, two distinct purchases inside the same second derive the same name and the second fails with ALREADY_EXISTS, reported as a duplicate when it is a genuinely new commitment. Both failures are silent at the call site, which passes `opts.IdempotencyToken` through without checking it is set. +- evidence: + ```go + func idempotentCommitmentName(token string) string { + if token == "" { + return fmt.Sprintf("cud-%d", time.Now().Unix()) + } + ``` +- suggested fix: return an error for an empty token on the purchase path so a caller that forgot to set one fails loudly instead of losing idempotency. +- verdict: CONFIRMED — computeengine:1414-1417 returns `cud-` for an empty token, the call site at 787 passes `opts.IdempotencyToken` through with no check, and the function's own doc at 1409-1410 records that the CLI path reaches the empty branch by design, so both the lost-idempotency and the same-second-collision cases are live. +- issue: (pending cross-reference) + +### A09-005 The USD spend cap is compared against a quote whose currency is never checked +- category: money-path +- severity: high +- location: pkg/exchange/exchange.go:313 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: `ExchangeQuoteSummary.CurrencyCode` is captured from the AWS response at line 253 and then read by nothing. `MaxPaymentDueUSD` and the field name `PaymentDueUSD` both assert USD, but for an account billed in another currency AWS returns `PaymentDue` in that currency. With `CurrencyCode == "JPY"` and `MaxPaymentDueUSD == 500`, a quote of `PaymentDue = "480"` (≈ $3) passes, and equally a cap of 500 would pass a 500-unit quote worth far more than $500 in a stronger currency. Nothing in the package fails closed on a non-USD quote. +- evidence: + ```go + func checkInitialQuote(q *ExchangeQuoteSummary, maxPayment *big.Rat) error { + if !q.IsValidExchange { + return fmt.Errorf("exchange is not valid: %s", q.ValidationFailureReason) + } + paymentDue := resolvePaymentDue(q) + if paymentDue.Cmp(maxPayment) == 1 { + ``` +- suggested fix: reject any quote whose `CurrencyCode` is not `"USD"` in `checkInitialQuote` and `checkReQuote` before the cap comparison, so non-USD accounts fail loud instead of comparing unlike units. +- verdict: CONFIRMED — `git grep -n CurrencyCode -- 'pkg/exchange/*.go'` over non-test files returns only the field declaration (pkg/exchange/exchange.go:23) and the assignment (line 252); the reshape.go hits are `OfferingOption.CurrencyCode`, a different type. No read of `ExchangeQuoteSummary.CurrencyCode` exists on the quote/cap path. +- issue: (pending cross-reference) + +### A09-008 The per-exchange cap is skipped entirely when the quote reports no PaymentDue +- category: money-path +- severity: high +- location: pkg/exchange/auto.go:319 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: the guard is conditioned on `PaymentDueUSD != nil`, so a quote with an empty `PaymentDue` bypasses `perExchangeCap` rather than being rejected. `processRecommendation` then sets `paymentDueStr = "0"` (line 244), the daily-cap arithmetic adds zero, and `chooseEffectiveCap` hands `Execute` a cap that A09-004 also lets through. Every layer of the cap stack collapses on the same nil, so an unpriced quote reaches `AcceptReservedInstancesExchangeQuote` completely uncapped. +- evidence: + ```go + if quote.PaymentDueUSD != nil && quote.PaymentDueUSD.Cmp(perExchangeCap) > 0 { + return nil, &SkippedRecommendation{ + SourceRIID: rec.SourceRIID, + Reason: fmt.Sprintf("exceeds per-exchange cap: payment $%s > cap $%.2f", + ``` +- suggested fix: skip the recommendation with an explicit "quote returned no payment amount" reason when `PaymentDueUSD` is nil, instead of falling through the cap check. +- verdict: CONFIRMED — `getValidatedQuote` guards the per-exchange cap behind `quote.PaymentDueUSD != nil` (pkg/exchange/auto.go:319), so a nil skips the check and returns the quote; `processRecommendation` then sets `paymentDueStr = "0"` (auto.go:243-246) and the daily-cap addition at auto.go:526 adds zero. One correction: "completely uncapped" overstates it — `processAutoExchange` still passes `chooseEffectiveCap`'s value as `MaxPaymentDueUSD` (auto.go:540-548), so the full bypass additionally requires Execute's fresh re-quote to omit PaymentDue as well, which is A09-004's scenario. +- issue: (pending cross-reference) + +### A10-001 Extended-support exclusion cuts Count without scaling the row's money fields +- category: money-path +- severity: high +- location: cmd/multi_service_engine_versions.go:486 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: An RDS rec for `db.r5.large` in `us-east-1` with `Count=10`, `EstimatedSavings=$1000/mo`, `CommitmentCost=$12000` passes filters while 3 running instances of that class sit on an extended-support version. `adjustRecommendationForExcludedVersions` sets `Count=7` but leaves EstimatedSavings at $1000 and CommitmentCost at $12000. The confirmation prompt (`sumPassedRecs`), the CSV report's UpfrontPayment/EstimatedSavings columns and the TOTAL row all describe a 10-instance purchase the run will never make, overstating both spend and benefit by 30%. Every other count-reducing path in the repo (`ApplyInstanceLimit` cmd/helpers.go:189, `ApplyCountOverride` cmd/helpers_count_override.go:56, `applyTargetCoverageRI` pkg/recfilter/sizing.go:303) routes through `common.ScaleRecommendationCosts`; this one does not. +- evidence: + ```go + if excludedCount > 0 { + originalCount := rec.Count + newCount := max(0, rec.Count-excludedCount) + if newCount != originalCount { + log.Printf("📉 Adjusting recommendation ...") + rec.Count = newCount // money fields untouched + } + } + return rec + ``` +- suggested fix: Replace the bare `rec.Count = newCount` with `rec = common.ScaleRecommendationCosts(rec, float64(newCount)/float64(originalCount)); rec.Count = newCount`, guarding `originalCount > 0` exactly as ApplyInstanceLimit does. +- verdict: CONFIRMED — cmd/multi_service_engine_versions.go:479-488 assigns rec.Count with no money scaling while the three sibling count-reducing paths all route through common.ScaleRecommendationCosts (cmd/helpers.go:189, cmd/helpers_count_override.go:56, pkg/recfilter/sizing.go:303), and the adjusted rec flows into the purchase set via cmd/multi_service_filters.go:73. +- issue: (pending cross-reference) + +### A10-002 A failed duplicate check falls back to purchasing the full, un-deduplicated counts +- category: money-path +- severity: high +- location: cmd/multi_service_helpers.go:589 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: With `--purchase`, if `AdjustRecommendationsForExistingRIs` fails (throttling, `DescribeReservedDBInstances` AccessDenied, a transient 5xx) the run logs a warning and proceeds with `filteredRecs` unchanged. The recs still carry the full pre-dedup counts, so the run buys reserved capacity the account already owns. The same fallback exists on the `--input-csv` path at cmd/multi_service.go:546-549 (`adjustedRecs = recs // Continue with original recommendations if check fails`) and inside cmd/multi_service_helpers.go:200-201. This is the one guard standing between a re-run and a double purchase, and it fails open on a non-reversible money path. +- evidence: + ```go + adjustedRecs, dedupedOut, err := duplicateChecker.AdjustRecommendationsForExistingRIs(ctx, filteredRecs, serviceClient) + if err != nil { + AppLogger.Printf(" ⚠️ Warning: Could not check for existing RIs: %v\n", err) + // Continue with original filteredRecs on error; adjustedRecs is not used in this branch. + } else { + ... + filteredRecs = adjustedRecs + } + ``` +- suggested fix: Propagate the error and abort the purchase for that service/region when `!isDryRun` (dry runs may continue with a loud banner), mirroring the "refuse to spend rather than purchase uncapped" stance already taken for `--max-instances` at cmd/multi_service_helpers.go:412. +- verdict: CONFIRMED — checkDuplicates (cmd/multi_service_helpers.go:587-591) is live on the purchase path via cmd/multi_service_helpers.go:666, keeps the pre-dedup counts on error, and executePurchasePipeline (cmd/multi_service.go:391-410) applies no second idempotency guard before executePurchase, so the dedup call really is the only one. +- issue: (pending cross-reference) + +### A10-003 `--target-coverage` sizes as if nothing is owned when the Cost Explorer coverage fetch fails +- category: money-path +- severity: high +- location: cmd/multi_service.go:63 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: `--target-coverage 80` on an account already at 75% RI coverage. If `GetRICoverageMap` errors (CE throttling, missing `ce:GetReservationCoverage`), `fetchExistingCoverage` returns nil, every rec keeps `ExistingCoveragePct == 0`, and `applyTargetCoverageRI` computes `gapPct := targetPct - rec.ExistingCoveragePct` = 80 (pkg/recfilter/sizing.go:256). The run sizes to buy 80% coverage on top of the 75% already owned, landing near 155% and paying for idle commitments. The `Recommendation.ExistingCoverageKnown` flag exists precisely to separate "unknown" from "zero" (the CSV writer renders `n/a` vs `0.0`, cmd/multi_service_csv.go:443) but the sizing formula never reads it. +- evidence: + ```go + cov, err := adapter.GetRICoverageMap(ctx, lookbackDays, regions) + if err != nil { + AppLogger.Printf(" ⚠️ Could not fetch existing-RI coverage (%v); sizing will assume zero existing coverage\n", err) + return nil + } + ``` +- suggested fix: When `cfg.TargetCoverage > 0` and the coverage fetch fails, abort a `--purchase` run rather than returning nil; at minimum have `applyTargetCoverageRI` drop any rec whose `ExistingCoverageKnown` is false instead of treating unknown as 0. +- verdict: CONFIRMED — fetchExistingCoverage returns nil on error (cmd/multi_service.go:62-66), AverageInstancesUsedPerHour is still populated by the rec parser (providers/aws/recommendations/parser_ri.go:106) so avg > 0 keeps the rec on the sizing path rather than the no-signal one, and gapPct = targetPct - ExistingCoveragePct at pkg/recfilter/sizing.go:256 never reads ExistingCoverageKnown, whose only reader is the CSV writer at cmd/multi_service_csv.go:443. +- issue: (pending cross-reference) + +### A10-004 A failed account lookup substitutes the account ID, silently defeating `--exclude-accounts` +- category: money-path +- severity: high +- location: cmd/helpers.go:91 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: `cudly --purchase --exclude-accounts production`. `organizations:DescribeAccount` fails for account `123456789012` (the CLI is running from a member account, or Organizations is throttling). `GetAccountAlias` caches and returns the numeric ID as the name. `shouldIncludeAccount` then matches `"123456789012"` against the filter `"production"`, finds no substring match, and the account is **not** excluded, so RIs are purchased in the account the operator explicitly ruled out. The bad value is cached for the process lifetime, so one transient failure poisons every rec for that account. The same fallback also silently narrows an `--include-accounts` run in the opposite direction. +- evidence: + ```go + result, err := c.orgClient.DescribeAccount(ctx, &organizations.DescribeAccountInput{ + AccountId: aws.String(accountID), + }) + if err != nil { + c.cache[accountID] = accountID // Use ID as fallback + return accountID + } + ``` +- suggested fix: Have `GetAccountAlias` return `(string, error)`; when an account filter is in force and the alias cannot be resolved, fail the run rather than filtering on a stand-in value. Do not cache the failure. +- verdict: CONFIRMED — cmd/helpers.go:91-94 caches the numeric ID as the alias, and cmd/multi_service_filters.go:105 filters on exactly that name via shouldIncludeAccount (:156-175), so an unresolvable account slips past --exclude-accounts and is dropped by --include-accounts. +- issue: (pending cross-reference) + +### A12-004 Capacity-% scaling is count-based, so every Savings Plans rec is silently dropped below 100% +- category: money-path +- severity: high +- location: frontend/src/recommendations.ts:3919 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: AWS Savings Plans recommendations are created with `Count: 1` (`providers/aws/services/savingsplans/client.go:191`, "Savings Plans don't have a count"). With the Capacity % toolbar set to anything in 1..99, `Math.floor(1 * capacity / 100)` is 0 and the rec hits `continue`, so it vanishes from the purchase with no per-row notice; an SP-only selection aborts with "Try a higher %". The Go sizing code special-cases exactly this: `pkg/recfilter/sizing.go:55` scales `SavingsPlanDetails.HourlyCommitment` for SPs and never touches Count. The frontend also scales `upfront_cost` / `monthly_cost` / `savings` by the count ratio while leaving the SP hourly commitment in `details` untouched, so any SP rec that did survive would display a scaled dollar figure against an unscaled committed quantity. +- evidence: + ```ts + for (const r of recommendations) { + const newCount = Math.floor((r.count * tb.capacity) / 100); + if (newCount <= 0) continue; + const ratio = r.count > 0 ? newCount / r.count : 1; + scaled.push({ ...r, count: newCount, recommended_count: r.count, + upfront_cost: r.upfront_cost * ratio, + monthly_cost: r.monthly_cost != null ? r.monthly_cost * ratio : null, + savings: r.savings * ratio }); + } + ``` +- suggested fix: Branch on `isSavingsPlanService(r.service)` and scale the SP hourly commitment in `details` by `capacity / 100` while leaving `count` at 1, mirroring `ApplyCoverage`'s SP branch. +- verdict: CONFIRMED — SP recs carry Count 1 from the parser (providers/aws/recommendations/parser_sp.go:382) and pkg/recfilter/sizing.go:45-55 scales SavingsPlanDetails.HourlyCommitment rather than Count, while the frontend's count-ratio branch computes floor(1*capacity/100)==0 and `continue`s for every capacity 1..99 (frontend/src/recommendations.ts:3919-3928). +- issue: (pending cross-reference) + +### A12-005 `scheduled` executions render the green "Completed" badge +- category: money-path +- severity: high +- location: frontend/src/history.ts:463 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: `historyExecutionStatuses` in `internal/api/handler_history.go:124` deliberately includes `"scheduled"` so the Revoke button is reachable. `statusBadgeHTML` has cases for pending/notified/approved/running/paused/canceled/partial/failed/expired but not `scheduled`, so a delayed purchase whose provider call has not fired yet falls to `default` and renders `badge-success` "Completed". That is exactly what the `isInFlightStatus` comment at lines 77-85 says must never happen. The row then reads "Completed" beside a Revoke button, and the backend's `summarizePurchaseHistory` likewise lets `scheduled` fall into `TotalCompleted` and `TotalUpfront`, so the "Total Upfront Spent" card counts money not yet spent. +- evidence: + ```ts + case 'expired': + return 'Expired'; + default: + return 'Completed'; + ``` +- suggested fix: Add `case 'scheduled':` returning a warning/muted "Scheduled" badge, and include it in `isInFlightStatus` so it buckets under Pending. +- verdict: CONFIRMED — statusBadgeHTML has no `scheduled` case so it hits the green default (frontend/src/history.ts:461-464), `scheduled` is deliberately in historyExecutionStatuses (internal/api/handler_history.go:124), and it also falls through summarizePurchaseHistory's switch into TotalCompleted and TotalUpfront (internal/api/handler_history.go:1042-1084). +- issue: (pending cross-reference) + +### A12-006 Marketplace consent dialog computes the list price with `Math.round` while the backend uses `math.Floor` +- category: money-path +- severity: high +- location: frontend/src/history.ts:1463 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: `api.createMarketplaceListing(id)` sends no price; the backend recomputes it with `computeRemainingMonths` = `int(math.Floor(remaining))`, clamped to a minimum of 1 (`internal/api/handler_marketplace.go:447`). The dialog rounds instead, and clamps to a minimum of 0. For a 1-year, $1,200-upfront, count-1 RI purchased 5.4 months ago: remaining is 6.6, so the dialog shows 7 months and a list price of $665 with net proceeds $585, while the backend floors to 6 and lists at $570 for $501.60 net. The user authorises one amount and a different one is submitted, under a comment claiming the two mirror each other "EXACTLY". +- evidence: + ```ts + const remainingMonths = Math.max(0, Math.round(termMonths - elapsedMonths)); + const perUnitResidual = termMonths > 0 && upfront > 0 + ? (upfront * (remainingMonths / termMonths)) / count + : 0; + const listPricePerUnit = perUnitResidual * AWS_MARKETPLACE_BUYER_DISCOUNT; + ``` +- suggested fix: Use `Math.max(1, Math.floor(termMonths - elapsedMonths))` to match `computeRemainingMonths`, or add a backend preview endpoint that returns the resolved price so the dialog cannot drift. +- verdict: CONFIRMED — The dialog uses `Math.max(0, Math.round(...))` (frontend/src/history.ts:1463) while computeRemainingMonths uses `int(math.Floor(remaining))` clamped to 1 (internal/api/handler_marketplace.go:447-459) and the listing recomputes the price server-side from it (internal/api/handler_marketplace.go:157-163), so any fractional remainder >= 0.5 authorises one price and submits another. +- issue: (pending cross-reference) + +### A12-009 "Execute Now" warning shows an upfront total that goes stale when rows are toggled +- category: money-path +- severity: high +- location: frontend/src/recommendations.ts:5013 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: `updateExecuteMode` computes `totalUpfront` over `checkedPurchaseIndices` and is bound only to the two radio `change` events. A user who selects "Execute Now" first and then unchecks rows (or uses the header select-all) sees the original amount: the include-checkbox handler at line 5402 calls only `updatePurchaseModalTotals`, which never touches `.direct-execute-warning`. The dialog that says "This will charge $X upfront immediately" therefore states an amount that does not match what the button submits, on the one path that bypasses approval entirely. +- evidence: + ```ts + let totalUpfront = 0; + for (const idx of checkedPurchaseIndices) { + const r = currentPurchaseRecommendations[idx]; + if (r) totalUpfront += r.upfront_cost; + } + // ... includeCb.addEventListener('change', ...) calls only updatePurchaseModalTotals + ``` +- suggested fix: Move the warning-text rebuild into `updatePurchaseModalTotals` (or expose a module-level `refreshDirectExecuteWarning()` that it calls) so the amount tracks the checked set. +- verdict: CONFIRMED — updateExecuteMode is bound only to the two radio change events (frontend/src/recommendations.ts:5041-5042) while the include-checkbox and select-all handlers call updatePurchaseModalTotals alone (frontend/src/recommendations.ts:5409, :5118), and `.direct-execute-warning` is written nowhere else in the file. +- issue: (pending cross-reference) + +### A12-010 Changing the exchange targets after a quote does not invalidate it +- category: money-path +- severity: high +- location: frontend/src/riexchange.ts:1609 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: The user quotes 2x `m5.large` (payment due $40), then switches the picker to `m5.4xlarge` and raises Count to 8. The chip and the running total refresh to the new figure, but the quote block, the visible Execute button and `modalQuoteReq` are untouched. Clicking Execute buys the previously quoted 2x `m5.large`, not what the form and running total display. Removing a target row (line 1590) has the same effect, and nothing disables the Quote button, so a double-click fires overlapping quote requests. +- evidence: + ```ts + pickerSelect.addEventListener('change', () => { + offeringInput.value = pickerSelect.value; + updateRowChip(pickerSelect.value, chipEl); + updateRunningTotal(); + }); + countInput.addEventListener('input', updateRunningTotal); + ``` +- suggested fix: In the change / input / remove handlers clear `modalQuote` and `modalQuoteReq`, re-hide `executeBtn`, and clear the quote result so a fresh quote is mandatory. +- verdict: CONFIRMED — The picker `change`, count `input` and row-remove handlers refresh only the chip and running total (frontend/src/riexchange.ts:1608-1614, :1590-1597) and never clear modalQuote/modalQuoteReq or re-hide executeBtn, so submitModalExecute still posts the previously quoted targets (frontend/src/riexchange.ts:1809-1812); the Quote button is likewise never disabled (frontend/src/riexchange.ts:1726-1728). +- issue: (pending cross-reference) + +### A12-074 The Savings Plans group row scales its savings by the cost period twice +- category: money-path +- severity: high +- location: frontend/src/recommendations.ts:3115 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: `scaleCost` multiplies by `PERIOD_FACTOR[period]`, and `formatCostForPeriod` calls `scaleCost` again internally (line 1502). The SP group parent row passes an already-scaled value into `formatCostForPeriod`, so the factor is applied twice: with the cost-period selector on "yearly" a $1,000/mo Savings Plans group renders $144,000/yr instead of $12,000/yr, and on "hourly" it renders roughly 1/720th of the correct figure. Every sibling site avoids this by feeding raw monthly values into `formatCostForPeriod` (lines 3154, 3221) or by using `formatScaledRange`, which does not re-scale (lines 3155, 3222, 801). The default period is monthly, whose factor is 1, which is why the bug is invisible until a user changes the selector. +- evidence: + ```ts + const scaledSpSavings = scaleCost(spSavingsTotal, period) ?? spSavingsTotal; + const spSavingsText = `${formatCostForPeriod(scaledSpSavings, period)}${sfxLabel}`; + ``` +- suggested fix: Pass the raw `spSavingsTotal` to `formatCostForPeriod` and drop the outer `scaleCost` call. +- verdict: CONFIRMED — formatCostForPeriod calls scaleCost internally (frontend/src/recommendations.ts:1502), so passing an already-scaled value at frontend/src/recommendations.ts:3114-3115 applies PERIOD_FACTOR twice, whereas the siblings pass raw monthly values (:3152, :3220) or use formatScaledRange, which does not re-scale (:3153, :3221); the only test runs at the default monthly factor of 1 (frontend/src/__tests__/recommendations.test.ts:8113-8128). +- issue: (pending cross-reference) + +### A13-001 `ce:GetCostAndUsage` is granted in no IaC flavor, but the ladder baseline calls it +- category: money-path +- severity: high +- location: terraform/modules/compute/aws/lambda/main.tf:381 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: A scheduled ladder run reaches `awsladder.NewFromAWSConfig` -> `baseline.GetUsageBaseline` -> `recommendations.Client.GetOnDemandSeries`, which issues `costexplorer.GetCostAndUsage` (providers/aws/recommendations/ondemand_series.go:182) under the ambient Lambda/Fargate execution role. `ce:GetCostAndUsage` appears in none of the seven IaC files that encode CUDly IAM (`/usr/bin/grep -rn GetCostAndUsage terraform iac cloudformation arm internal/iacfiles` returns nothing). Every ladder run therefore AccessDenies on its usage baseline and is recorded as Errored, producing no plan. `scripts/check-aws-iam-parity.sh` cannot catch this: it compares the templates only against each other, never against the calls the Go code makes, so a uniform omission passes. +- evidence: + ```hcl + Action = [ + "ce:GetReservationUtilization", + "ce:GetReservationPurchaseRecommendation", + "ce:GetReservationCoverage", + "ce:GetSavingsPlansPurchaseRecommendation", + "ce:GetSavingsPlansUtilization", + "ce:GetSavingsPlansCoverage", + ] + ``` +- suggested fix: Add `ce:GetCostAndUsage` to the runtime CE statement in `terraform/modules/compute/aws/{lambda,fargate}/main.tf` and `cloudformation/stacks/CUDly/template.yaml`, and extend `scripts/check-aws-iam-parity.sh` (or a Go guard test) to assert the union of SDK calls in `providers/aws` is covered. +- verdict: CONFIRMED — `/usr/bin/grep -rn 'ce:Get' terraform iac cloudformation arm` returns only the six Reservation/SavingsPlans actions in every flavor (terraform/modules/compute/aws/lambda/main.tf:381-386, fargate/main.tf:373-378, cloudformation/stacks/CUDly/template.yaml:423-428) and `GetCostAndUsage` appears nowhere; the only `ce:*` hit is the permissions-boundary ceiling (policy_boundary.tf:163), which caps but never grants. The path is live, not stubbed: internal/server/app.go:570 assigns `awsladder.NewFromAWSConfig`, which wires `&onDemandSeriesAdapter{client: recoClient}` (providers/aws/ladder/factory.go:85) over ambient `LoadDefaultConfig` credentials (factory.go:57), and handler_ladder.go:306 calls `GetUsageBaseline`, whose only data source is `GetOnDemandSeries` -> `costExplorerClient.GetCostAndUsage` (providers/aws/recommendations/ondemand_series.go:182). +- issue: (pending cross-reference) + +### A13-002 RI Marketplace listing actions are granted nowhere, so the sell flow 403s +- category: money-path +- severity: high +- location: terraform/modules/compute/aws/lambda/main.tf:352 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: `POST` to the marketplace-list endpoint runs `internal/api/handler_marketplace.go:248` `ec2Client.CreateMarketplaceListing`, which calls `ec2:CreateReservedInstancesListing` (providers/aws/services/ec2/client.go:1048) against the ambient runtime credentials from `loadAWSConfigWithRegion` -> `getBaseAWSConfig` -> `awsconfig.LoadDefaultConfig`. None of `ec2:CreateReservedInstancesListing`, `ec2:DescribeReservedInstancesListings` or `ec2:CancelReservedInstancesListing` appears in any terraform module, CloudFormation template or federation bundle. Worse, `reserveAndCreateListing` claims the listing slot in the DB before the AWS call, so on the AccessDenied the row transits pending and depends on `releaseMarketplaceClaim` to recover; the cancel path is equally unauthorized, so the operator cannot unwind the listing through CUDly either. +- evidence: + ```hcl + "ec2:DescribeReservedInstances", + "ec2:DescribeReservedInstancesOfferings", + "ec2:GetReservedInstancesExchangeQuote", + "ec2:AcceptReservedInstancesExchangeQuote", + "ec2:PurchaseReservedInstancesOffering", + "ec2:DescribeInstanceTypeOfferings", + "ec2:DescribeRegions", + ``` +- suggested fix: Add the three marketplace actions to the runtime EC2 statement in the Lambda module, the Fargate module and `cloudformation/stacks/CUDly/template.yaml`. +- verdict: CONFIRMED — `/usr/bin/grep -rn ReservedInstancesListing terraform iac cloudformation arm` exits 1 (no hits), while the three SDK calls are declared at providers/aws/services/ec2/client.go:31-33 and issued at client.go:1048, 1070 and 1111. Both routes are registered unconditionally (internal/api/router.go:194-195) and the client is built from ambient credentials: `loadAWSConfigWithRegion` -> `getBaseAWSConfig` -> `awsconfig.LoadDefaultConfig` (internal/api/handler_ri_exchange.go:1259-1275), so the runtime role's `ri_exchange` statement (terraform/modules/compute/aws/lambda/main.tf:349-390) is the effective policy and grants none of the three. +- issue: (pending cross-reference) + +### A01-007 RI exchange approval bypasses the 4-eyes policy +- category: money-path +- severity: medium +- location: internal/api/handler_ri_exchange.go:2195 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: `GlobalConfig.RequireDifferentApprover=true`. The user who submitted a manual RI exchange (record `CreatedByUserID` = self) holds `approve-own:purchases` and clicks approve. `approveRIExchangeViaSession` -> `fetchAndAuthorizeRIExchange` -> `authorizeSessionApproveRIExchange` explicitly allows the creator, and neither this path nor `approveRIExchangeViaToken` calls `requireDifferentApprover` (callers: handler_purchases.go:600 and :693 only). The irreversible exchange executes with the same person as requester and approver. `TestApproveRIExchange_SessionApproveOwn` pins the self-approval as the expected behaviour. +- evidence: + ```go + record, err := h.fetchAndAuthorizeRIExchange(ctx, session, id) + if err != nil { + return nil, err + } + + // Session-authed approval: stamp the session user as the actor. + transitioned, err := h.config.TransitionRIExchangeStatus(ctx, id, "pending", "processing", resolveCreatorUserID(session)) + ``` +- suggested fix: After `fetchAndAuthorizeRIExchange`, apply the same dual-control check against `record.CreatedByUserID` (generalise `requireDifferentApprover` to take the creator pointer), and fail closed on the token path when the mode is on and no session resolves. +- verdict: CONFIRMED — approveRIExchangeViaSession (internal/api/handler_ri_exchange.go:2176-2223) goes fetchAndAuthorizeRIExchange → authorizeSessionApproveRIExchange:2287-2320, which grants the creator under approve-own; requireDifferentApprover's callers are handler_purchases.go:600,693 only, enforceFourEyesPolicy runs inside purchase.Manager.ApproveAndExecute (internal/purchase/approvals.go:323) which executeApprovedExchange:2430-2487 never calls, and TestApproveRIExchange_SessionApproveOwn (handler_ri_exchange_test.go:422-447) asserts the creator's self-approval succeeds. +- issue: (pending cross-reference) + +### A01-008 Approving never re-evaluates the approver's own permission constraints +- category: money-path +- severity: medium +- location: internal/api/handler_purchases.go:716 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: The approving user's group grants `approve-any:purchases` with `MaxPurchaseAmount=5000` and `AccountIDs=[acct-A]`. A different submitter creates a $200k execution for acct-B. `approvePurchaseViaSession` and `approveViaToken` run RBAC, 4-eyes and delay checks, then `ApproveAndExecute`; `requirePermissionConstraints` is hard-wired to the `execute` action (handler.go:475) and is never called on approval, so constraints configured on approve permissions are dead and the approver commits spend their own permission was meant to cap. The retry path was explicitly given this re-check for the identical rationale (lines 1622-1632). +- evidence: + ```go + if globalCfg.GetPurchaseDelay() > 0 { + return h.approveWithDelay(ctx, execution, globalCfg.GetPurchaseDelay(), session.Email, actor) + } + + if err := h.purchase.ApproveAndExecute(ctx, execution.ExecutionID, fourEyesActorIdentity(session), actor); err != nil { + ``` +- suggested fix: In both approve branches call `h.enforcePurchaseConstraints`-style evaluation of the approver's session against `execution.Recommendations` for the approve action (add the action parameter back to `requirePermissionConstraints`), or document and enforce that approve permissions may not carry constraints. +- verdict: CONFIRMED — requirePermissionConstraintsAction is the constant "execute" (internal/api/handler.go:475) and its only purchases-side callers are enforcePurchaseConstraints at handler_purchases.go:1630 (retry) and :2181 (execute); approvePurchaseViaSession:685-716 and approveViaToken:589-619 gate only on HasPermissionAPI, which passes nil constraints (internal/auth/service_api.go:402-404), before ApproveAndExecute. +- issue: (pending cross-reference) + +### A01-010 Azure revoke executes without the user's consented refund amount when the body omits it or Azure returns no amount +- category: money-path +- severity: medium +- location: internal/api/handler_purchases_revoke.go:661 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: `POST /api/purchases/{id}/revoke` with an empty body (or `{}`) for an Azure reservation. `revokeConfirmBody.ExpectedRefundAmount` is documented as "Required when the purchase has an Azure revocation window" but nothing enforces it; `callAzureReturn` skips the TOCTOU/consent check whenever `expectedRefundAmount == nil` OR Azure's `BillingRefundAmount` is nil, and submits the Return with whatever `sessionID` CalculateRefund returned (possibly ""). The two-step quote-then-confirm UX is bypassable by any client that just does not send the field. `TestRevokePurchase_AzureSuccess` (handler_purchases_revoke_test.go:247) passes `nil` and asserts success. +- evidence: + ```go + if expectedRefundAmount != nil && calcRefundAmount != nil { + if math.Abs(*expectedRefundAmount-*calcRefundAmount) > revokeQuoteEpsilon { + return nil, NewClientError(422, ...) + } + } + ``` +- suggested fix: Return 400 when `expectedRefundAmount` is nil for the Azure path, 422 when Azure returns no `BillingRefundAmount` or an empty `SessionID`, and keep the divergence check for the remaining case. +- verdict: CONFIRMED — revokeConfirmBody (internal/api/handler_purchases_revoke.go:70-73) documents the field as required but loadAndRevokePurchaseHistory:217-224 only unmarshals it; callAzureReturn:672-678 runs the divergence check only when both pointers are non-nil, azureCalculateRefund:718-735 returns sessionID "" with a nil error when the response omits it, and the Return POST at :687-697 fires regardless; TestRevokePurchase_AzureSuccess (handler_purchases_revoke_test.go:247-275) passes nil with no BillingRefundAmount and asserts "revoked". +- issue: (pending cross-reference) + +### A04-001 Deleting a cloud account silently converts a scoped plan into an ambient-credential plan +- category: money-path +- severity: high +- location: internal/config/store_postgres.go:3299 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd (and internal/database/postgres/migrations/000011_cloud_accounts.up.sql:113-117) +- failure scenario: plan P is enabled + auto_purchase and targets exactly one cloud account A (say an Azure subscription); its next_execution_date is next month, so no purchase_executions row references A yet and the 000053 RESTRICT FK does not fire. An admin deletes A. `plan_accounts.account_id` is `ON DELETE CASCADE`, so P's only plan_accounts row vanishes and P keeps running with zero accounts. The notification sweep (`internal/purchase/notifications.go:111-146`) creates a pending execution with nil CloudAccountID, and `executePurchase` (`internal/purchase/execution.go:56-66`) sees `len(accounts) == 0` and falls through to `executeSingleAccount` using CUDly's own ambient credentials. Migration 000060 deleted exactly this "universal plan" shape because it "could trigger unintended purchases"; `DeleteCloudAccount` recreates it with no check. +- evidence: + ```go + tag, err := tx.Exec(ctx, `DELETE FROM cloud_accounts WHERE id = $1`, id) + if err != nil { + return fmt.Errorf("failed to delete cloud account: %w", err) + } + if tag.RowsAffected() == 0 { + return fmt.Errorf("cloud account not found: %s", id) + } + if err = tx.Commit(ctx); err != nil { + ``` + ```sql + CREATE TABLE plan_accounts ( + plan_id UUID NOT NULL REFERENCES purchase_plans(id) ON DELETE CASCADE, + account_id UUID NOT NULL REFERENCES cloud_accounts(id) ON DELETE CASCADE, + ``` +- suggested fix: inside the `DeleteCloudAccount` transaction, refuse (or disable, `enabled = false`) every plan whose only plan_accounts row is the account being deleted, and make `executePurchase` refuse a plan-scoped execution whose plan has zero accounts instead of falling back to ambient credentials. +- verdict: PLAUSIBLE — the unscoped-plan resurrection is confirmed (`plan_accounts.account_id` is ON DELETE CASCADE at internal/database/postgres/migrations/000011_cloud_accounts.up.sql:115, `DeleteCloudAccount` at internal/config/store_postgres.go:3274-3310 checks no plan, and internal/purchase/execution.go:66-68 falls through on `len(accounts) == 0`), but the ambient *purchase* needs an execution carrying recommendations and neither writer populates them — `getOrCreateExecution` (internal/purchase/notifications.go:126-149) nor `createPurchaseExecutionsTx` (internal/api/handler_plans.go:539-547) sets `Recommendations`, so `resolveSingleAccountProvider` short-circuits at internal/purchase/execution.go:367-369 and nothing is bought. +- severity-adjusted: medium — the plan is silently left in the shape migration 000060 deleted, but no purchase can result on the traced path. +- issue: (pending cross-reference) + +### A04-002 Notification sweep's full-row UpdatePurchasePlan can roll a ramp position back and freeze the plan +- category: money-path +- severity: medium +- location: internal/config/store_postgres.go:733-738 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd (writer at internal/purchase/notifications.go:100-104) +- failure scenario: `UpdatePurchasePlan` is a blind full-row UPDATE of `ramp_schedule`, `next_execution_date` and `last_execution_date` with no version/CAS check. The scheduled notification sweep reads every plan via `ListPurchasePlans` (current_step=1, next date D1), sends the email, then calls `UpdatePurchasePlan(plan)` only to stamp `last_notification_sent`. If, between that read and that write, a user approval in the API Lambda completes step 2 and `CompletePlanStep` commits current_step=2 / next date D2 under its row lock, the sweep's write lands afterwards and restores current_step=1 and D1. From then on `getOrCreateExecution` keeps finding the already-completed D1 execution, no D2 execution is ever created, and a later completion of step 3 through the API create path is refused by `advanceRampStep` as "step(s) 2-2 never completed": the ramp silently stops buying (the #1669 under-buy shape) while the plan reports step 1. The same overwrite happens from any PUT /api/plans edit that races a completion. This is the #1071 lost-update class reintroduced through a secondary writer; the scheduled-task advisory lock does not serialize against API approvals. +- evidence: + ```go + func (s *PostgresStore) UpdatePurchasePlan(ctx context.Context, plan *PurchasePlan) error { + plan.UpdatedAt = time.Now() + return s.WithTx(ctx, func(tx pgx.Tx) error { + return s.UpdatePurchasePlanTx(ctx, tx, plan) + }) + } + ... + UPDATE purchase_plans SET + name = $2, ... ramp_schedule = $7, updated_at = $8, + next_execution_date = $9, last_execution_date = $10, last_notification_sent = $11 + ``` +- suggested fix: add a narrow `StampPlanNotificationSent(ctx, planID, at)` UPDATE for the sweep, and make `UpdatePurchasePlanTx` optimistic (`WHERE id = $1 AND updated_at = $prevUpdatedAt`, returning a conflict error) so no caller can overwrite a ramp position it did not read under `LockPurchasePlanTx`. +- verdict: CONFIRMED — `UpdatePurchasePlanTx` (internal/config/store_postgres.go:762-786) is a blind full-row UPDATE of `ramp_schedule`/`next_execution_date`/`last_execution_date` with no version predicate, the sweep's write at internal/purchase/notifications.go:100-104 uses a plan read outside `LockPurchasePlanTx` (internal/config/store_postgres.go:667-676, the only lock `CompletePlanStep` holds at 636-654), and `GetExecutionByPlanAndDate` (internal/config/store_postgres.go:1549-1573) has no status filter, so the restored D1 row is returned to `getOrCreateExecution` forever. +- issue: (pending cross-reference) + +### A04-005 GetPendingExecutionsTx duplicate-detection read is a capped page, not a predicate +- category: money-path +- severity: medium +- location: internal/config/store_postgres.go:1493-1508 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd (caller internal/api/handler_purchases.go:2457-2463) +- failure scenario: `persistExecutionAndSuppressions` relies on `GetPendingExecutionsTx` to find an existing pending row for the same creator + idempotency key before inserting (#643). The query is `ORDER BY scheduled_date ASC LIMIT 1000 FOR UPDATE`. An estate with more than 1000 pending/notified rows (100 accounts x a 12-step ramp is 1200 rows created up front) silently truncates the newest scheduled rows out of the page; a double-submitted direct purchase scheduled today lands past the cut once older-dated plan rows fill the page, `matchDuplicateInList` finds nothing, and a second execution is inserted and later approved. The FOR UPDATE also row-locks up to 1000 unrelated pending rows on every submit. +- evidence: + ```go + const query = ` + SELECT plan_id, execution_id, status, ... + FROM purchase_executions + WHERE status IN ('pending', 'notified') + AND (expires_at IS NULL OR expires_at > NOW()) + ORDER BY scheduled_date ASC + LIMIT 1000 + FOR UPDATE + ` + ``` +- suggested fix: push the duplicate predicate into SQL (`WHERE status IN (...) AND created_by_user_id = $1 AND idempotency_key = $2 AND created_at >= $3 FOR UPDATE`) so the read is bounded by matching rows, or enforce the dedupe with a partial unique index on `(idempotency_key) WHERE status IN ('pending','notified')`. +- verdict: PLAUSIBLE — the read really is an unfiltered capped page (`ORDER BY scheduled_date ASC LIMIT 1000 FOR UPDATE`, internal/config/store_postgres.go:1502-1507) and `matchDuplicateInList` scans only what that page returned (internal/api/handler_purchases.go:2419-2434, called at 2457-2463), so correctness rests on a LIMIT rather than a predicate; the actual double-insert additionally needs more than 1000 live pending/notified rows scheduled earlier than the resubmit, a data-volume and date-distribution condition I could not establish from source. +- issue: (pending cross-reference) + +### A04-007 ri_exchange_history schema cannot record non-AWS exchanges, so the daily-spend ledger is AWS-only +- category: money-path +- severity: medium +- location: internal/database/postgres/migrations/000009_ri_exchange_history.up.sql:3 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd (writer internal/config/store_postgres.go:2639-2651; ledger internal/config/store_postgres.go:2893-2900) +- failure scenario: `account_id VARCHAR(20) NOT NULL CHECK (account_id ~ '^\d{12}$')` was never widened (000095 widened purchase_history, 000067 savings_snapshots, but no migration touched this table). Any `SaveRIExchangeRecord` carrying an Azure subscription GUID fails with SQLSTATE 23514/22001, which is why `executeAzureExchange` (internal/api/handler_ri_exchange.go:1174) performs no store write at all. Consequently `GetRIExchangeDailySpend`, the only input to `RIExchangeMaxDailyUSD`, sums AWS rows only: an operator who set a daily cap has no cap on Azure exchanges, and Azure exchanges leave no audit row for the History page. +- evidence: + ```sql + CREATE TABLE ri_exchange_history ( + id UUID PRIMARY KEY DEFAULT gen_random_uuid(), + account_id VARCHAR(20) NOT NULL CHECK (account_id ~ '^\d{12}$'), + ``` + ```go + SELECT COALESCE(SUM(payment_due), 0)::text + FROM ri_exchange_history + WHERE status IN ('completed', 'processing') + ``` +- suggested fix: migration dropping the `^\d{12}$` CHECK and widening `account_id` to VARCHAR(255) (mirror 000095's guarded probe), then have the Azure execute path write a ledger row and consult `GetRIExchangeDailySpend`. +- verdict: CONFIRMED — the `CHECK (account_id ~ '^\d{12}$')` on VARCHAR(20) still stands at internal/database/postgres/migrations/000009_ri_exchange_history.up.sql:3 and no later migration touching the table (000011, 000026, 000054, 000077, 000080, 000089) alters `account_id`; `executeAzureExchange` (internal/api/handler_ri_exchange.go:1173-1242) performs no store write, and its only money gate `checkAzureExchangeMoneyGuardrails` (internal/api/handler_ri_exchange.go:1110-1135) checks the per-request cap only, never `GetRIExchangeDailySpend` (internal/config/store_postgres.go:2893-2900). +- issue: (pending cross-reference) + +### A04-009 FailRIExchange and CompleteRIExchange have no status guard, so an accepted exchange can drop out of the daily-cap ledger +- category: money-path +- severity: medium +- location: internal/config/store_postgres.go:2865-2882 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd (caller internal/api/handler_ri_exchange.go:2350-2355) +- failure scenario: `executeApprovedExchange` transitions pending to processing with a CAS, calls `AcceptReservedInstancesExchangeQuote`, and on any `execErr` (including a client timeout after AWS accepted the quote) calls `failExchange`, which runs `UPDATE ri_exchange_history SET status='failed' WHERE id=$1` unconditionally. `GetRIExchangeDailySpend` sums only `completed`/`processing`, so the exchange that AWS actually executed disappears from the daily ledger and the next approval's cap check under-counts by its `payment_due`. Nothing in the store distinguishes "never submitted" from "outcome unknown after submission". `CompleteRIExchange` / `CompleteRIExchangeWithPayment` are equally unguarded, so a duplicate completion callback can overwrite a `failed` audit row. +- evidence: + ```go + query := ` + UPDATE ri_exchange_history + SET status = 'failed', error = $2 + WHERE id = $1 + ` + ``` +- suggested fix: make the terminal writers CAS on the expected source status (`WHERE id = $1 AND status = 'processing'`) and add a distinct `ambiguous`/`unknown` terminal status that the ledger keeps counting until reconciled. +- verdict: CONFIRMED — `FailRIExchange` is an unconditional `UPDATE ... SET status='failed' WHERE id=$1` (internal/config/store_postgres.go:2863-2882), and `executeApprovedExchange` routes every `execErr` — including the post-`AcceptReservedInstancesExchangeQuote` ambiguous ones — into `failExchange` (internal/api/handler_ri_exchange.go:2465-2467, helper at 2349-2355); `GetRIExchangeDailySpend` sums only `completed`/`processing` (internal/config/store_postgres.go:2893-2900), so the row leaves the ledger. `CompleteRIExchange` (2798-2816) and `CompleteRIExchangeWithPayment` (2822-2840) are equally unguarded. +- issue: (pending cross-reference) + +### A05-014 Confirmation email reports the estimated upfront, purchase_history stores the actual +- category: money-path +- severity: low +- location: internal/purchase/execution.go:660 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: The aggregator accumulates `totalUpfront` from `rec.UpfrontCost`, the recommendation's estimate captured at collection time, while `savePurchaseHistory` writes `UpfrontCost: result.Cost`, the figure the provider returned for the commitment it actually created (execution.go:906). When AWS prices the offering differently from the Cost Explorer estimate (a stale rec, a repriced offering), the "Purchase confirmed" email quotes the estimate and the History row quotes the real charge, with no note that they differ. `totalSavings` has the same shape. +- evidence: + ```go + exec.Recommendations[i].Purchased = true + exec.Recommendations[i].PurchaseID = v.purchase.CommitmentID + totalSavings += rec.Savings + totalUpfront += rec.UpfrontCost + ``` +- suggested fix: accumulate `v.purchase.Cost` when the provider returned one and fall back to `rec.UpfrontCost` only when it did not, so the confirmation reports what was charged. +- verdict: CONFIRMED, and worse than described — the divergence is at execution.go:665 (`totalUpfront += rec.UpfrontCost`, fed to sendPurchaseNotification at :117 and into NotificationData.TotalUpfrontCost at :944) versus execution.go:906 (`UpfrontCost: result.Cost`). Only four service clients ever set result.Cost — rds/client.go:184, elasticache/client.go:185, memorydb/client.go:180, redshift/client.go:204 — and `/usr/bin/grep -n Cost providers/aws/services/ec2/client.go` shows the EC2 RI client never assigns it, so every EC2 RI writes UpfrontCost=0 to purchase_history while the email quotes the full estimate. +- severity-adjusted: medium — this is not a rounding difference on repriced offerings; EC2 Reserved Instances, the most common commitment, record $0 upfront in the History table on every purchase. +- issue: (pending cross-reference) + +### A07-002 Savings Plan hourly commitment is rounded to two decimals before purchase +- category: money-path +- severity: medium +- location: providers/aws/services/savingsplans/client.go:232 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: `HourlyCommitment` comes from CE's `HourlyCommitmentToPurchase` and is parsed as a full-precision float (parser_sp.go:301). `%.2f` truncates it to cents-per-hour. With `HourlyCommitment = 0.004` the request commits `"0.00"`; with `0.1250001` it commits `"0.13"`, a 4% over-commitment for the whole term. Neither case errors — AWS accepts the rounded string and the recorded purchase silently differs from what was sized and shown to the operator. +- evidence: + ```go + input := &savingsplans.CreateSavingsPlanInput{ + SavingsPlanOfferingId: aws.String(offeringID), + Commitment: aws.String(fmt.Sprintf("%.2f", spDetails.HourlyCommitment)), + UpfrontPaymentAmount: nil, + ``` +- suggested fix: format with the precision AWS accepts (`strconv.FormatFloat(v, 'f', -1, 64)`), and reject a commitment that rounds to zero rather than sending `"0.00"`. +- verdict: CONFIRMED — providers/aws/services/savingsplans/client.go:232 formats `%.2f` while providers/aws/recommendations/parser_sp.go:301 parses `HourlyCommitmentToPurchase` as a full-precision float64 into the same field (:393), so the value sent to `CreateSavingsPlan` differs from the sized one with no error path. +- issue: (pending cross-reference) + +### A07-014 Expiry adjustment divides an org-wide expiring count by a per-rec share of demand +- category: money-path +- severity: medium +- location: providers/aws/recommendations/expiry.go:83 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: `ApplyCoverageMapToRecommendations` (coverage.go:447) rescales each rec's `AverageInstancesUsedPerHour` so the recs in a pool *sum* to the org-wide average. `expiringByPool` is keyed by the same org-wide pool key and holds the org-wide expiring count. With three linked-account recs sharing one pool, each rec's avg is roughly one third of the pool demand while `expCount` is the full pool's expiring count, so `expiringPct` comes out about three times too large, clamps `ExistingCoveragePct` to zero, and `--target-coverage` sizes each of the three recs as if the pool were entirely uncovered. +- evidence: + ```go + expCount, ok := expiringByPool[lookupPoolKey(recs[i])] + if !ok || expCount == 0 { + continue + } + expiringPct := float64(expCount) / recs[i].AverageInstancesUsedPerHour * 100.0 + ``` +- suggested fix: compute `expiringPct` once per pool against the pool's total average (sum the recs' avgs, or carry the coverage map's `AvgInstancesPerHour`), then apply that single percentage to every rec in the pool. +- verdict: CONFIRMED — `ApplyCoverageMapToRecommendations` (providers/aws/recommendations/coverage.go:447) rescales each rec's avg to its proportional share of `cov.AvgInstancesPerHour`, while `expiringCountsByPool` (expiry.go:59) sums the whole pool's `Count` under the same `lookupPoolKey`, so the ratio at expiry.go:83 grows with the number of recs sharing the pool and the clamp at :84-85 then zeroes `ExistingCoveragePct`. +- issue: (pending cross-reference) + +### A08-002 Five Azure clients return an empty commitment inventory with a nil error when the pager cannot be built +- category: money-path +- severity: high +- location: providers/azure/services/compute/client.go:230 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: with an expired credential or a 403 on `armconsumption.NewReservationsDetailsClient`, `GetExistingCommitments` logs a WARNING and returns `([]common.Commitment{}, nil)`. A caller that uses the commitment inventory to decide whether a reservation already exists sees "no existing commitments" and proceeds. The very next function in the same file states the invariant this breaks: "A partial commitment list is unsafe for the purchase flow — it could trigger duplicate purchases for reservations that exist but weren't loaded." The same shape is in cache/client.go:189, cosmosdb/client.go:191, database/client.go:221 and search/client.go:136. +- evidence: + ```go + pager, err := c.createReservationsPager() + if err != nil { + log.Printf("WARNING: failed to create VM reservations pager: %v", err) + return []common.Commitment{}, nil + } + return c.collectVMReservations(ctx, pager) + ``` +- suggested fix: return the wrapped error from all five clients, matching `collectVMReservations`'s own contract; drop the log-and-swallow branch. +- verdict: PLAUSIBLE — the log-and-swallow really is in all five files (compute/client.go:229, cache/client.go:188, cosmosdb/client.go:190, database/client.go:220, search/client.go:134), but the only production consumer of GetExistingCommitments is the AWS-only CLI dedupe path (cmd/multi_service_helpers.go:652 → pkg/recfilter/dedupe.go:40, whose client comes from the AWS-only cmd/main.go:243 switch), and `armconsumption.NewReservationsDetailsClient` performs no auth call, so the stated "expired credential or 403" trigger is not established from source. +- severity-adjusted: medium — no Azure caller reaches the swallowed branch today, and cmd/multi_service_helpers.go:653 continues with unadjusted recommendations on an error anyway. +- issue: (pending cross-reference) + +### A08-010 Azure exchange purchases force BillingPlan=Upfront regardless of the caller's request +- category: money-path +- severity: high +- location: providers/azure/services/compute/exchange_operations.go:457 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: `ExchangeTarget` has no billing-plan field and `buildCalculateExchangeRequest` unconditionally sets `ReservationBillingPlanUpfront` on every target. A customer exchanging into a monthly-billed reservation is priced and committed for the full term today. The file's own contract is that nothing is coerced: "There is no coercion of invalid values (no clamping quantity to 1, no defaulting an unrecognized term) — a caller mistake here is a validation error, not a silently different exchange than the one requested" (exchange_operations.go:227). `AppliedScopeType` is optional and threaded through; `BillingPlan` is not. +- evidence: + ```go + Properties: &armreservations.PurchaseRequestProperties{ + AppliedScopeType: to.Ptr(scopeType), + BillingPlan: to.Ptr(armreservations.ReservationBillingPlanUpfront), + BillingScopeID: to.Ptr(tgt.BillingScopeID), + ``` +- suggested fix: add a required `BillingPlan` field on `ExchangeTarget`, validate it against `armreservations.PossibleReservationBillingPlanValues()` in `validateExchangeTargets`, and thread it through. +- verdict: PLAUSIBLE — `ExchangeTarget` (exchange_operations.go:111-137) has no billing-plan field and `buildCalculateExchangeRequest` hardcodes `ReservationBillingPlanUpfront` (exchange_operations.go:457), but the HTTP boundary that builds those targets carries no billing-plan input either (internal/api/handler_ri_exchange.go:662-668), so no caller request is coerced; the impact needs a customer who wants a monthly-billed exchange target. +- severity-adjusted: medium — an upfront-only exchange feature rather than a silent override of a plan the caller asked for. +- issue: (pending cross-reference) + +### A08-013 The terminal-failed reservation-order set uses bare literals and is a denylist that misses BillingFailed +- category: money-path +- severity: medium +- location: providers/azure/services/internal/reservations/purchase.go:449 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: `matchReservationOrderInPage` short-circuits a purchase for any existing order whose state is not in this three-entry set. `armreservations.ProvisioningState` (armreservations@v1.1.0/constants.go:369) also defines `BillingFailed`. A prior attempt that reached `BillingFailed` therefore suppresses the re-drive: the recommendation is reported as already purchased and the order ID of a dead order is returned as the commitment ID, so the customer never gets the reservation and the execution is recorded as successful. The three entries are also bare strings, not the SDK constants, so a rename or casing change in a future API version breaks the guard with no compile error. +- evidence: + ```go + var reservationOrderTerminalFailedStates = map[string]struct{}{ + "Cancelled": {}, + "Failed": {}, + "Expired": {}, + } + ``` +- suggested fix: build the set from `armreservations.ProvisioningState*` constants and invert it into an allowlist of states that legitimately suppress a purchase (Succeeded, Created, PendingBilling, ConfirmedBilling, …), so an unknown or newly added state fails safe. +- verdict: CONFIRMED — armreservations@v1.1.0/constants.go:369 defines `ProvisioningStateBillingFailed`, absent from the three bare-string entries at reservations/purchase.go:449-453, so `matchReservationOrderInPage` (purchase.go:525-538) returns a BillingFailed order and `DoIdempotentPurchaseTwoStep` (purchase.go:566-570) short-circuits the re-drive with that dead order's ID. +- issue: (pending cross-reference) + +### A08-017 Advisor recommendations hardcode PaymentOption "upfront" and swallow an unparseable savings amount +- category: money-path +- severity: medium +- location: providers/azure/recommendations.go:427 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: `convertAdvisorRecommendation` stamps `PaymentOption: "upfront"` on every Advisor recommendation. Advisor never reports a payment option, and `populateFromExtendedProperties` never overrides it, so an Advisor-sourced recommendation reaches `BillingPlanForPaymentOption` as Upfront and charges the whole commitment immediately. `Term` is defaulted to `"1yr"` and then overwritten by `ext["term"]` verbatim (recommendations.go:444), which Azure reports as "P1Y"/"P3Y", so the same field carries two vocabularies. Separately `extFloat` returns 0 for a missing or unparseable `annualSavingsAmount` and the caller divides it by a bare `12` (line 442), so a malformed Advisor payload becomes a zero-savings recommendation rather than an error. +- evidence: + ```go + rec := &common.Recommendation{ + Provider: common.ProviderAzure, + Service: common.ServiceType(service), + Account: r.subscriptionID, + CommitmentType: common.CommitmentReservedInstance, + Term: "1yr", + PaymentOption: "upfront", + } + ``` +- suggested fix: run the Advisor term through `normaliseTerm` and drop the recommendation (or leave PaymentOption empty so the purchase path refuses it) rather than asserting Upfront; have `extFloat` distinguish absent from zero on the savings field. +- verdict: CONFIRMED — `convertAdvisorRecommendation` (recommendations.go:420-433) hardcodes `PaymentOption: "upfront"` and `populateFromExtendedProperties` (recommendations.go:437-446) never writes that field, overwrites Term with the raw `ext["term"]` without passing it through `normaliseTerm`, and divides `extFloat`'s absent-or-unparseable 0 (recommendations.go:455-462) by a bare 12. +- issue: (pending cross-reference) + +### A08-018 Azure Savings Plan offering details invent a 50/50 split for a payment option Azure cannot express +- category: money-path +- severity: medium +- location: providers/azure/services/savingsplans/client.go:427 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: `GetOfferingDetails` accepts "partial-upfront" and reports half the total as due today and half spread hourly. Azure savings plans have exactly two billing plans and no partial-upfront — the same package's `BillingPlanForPaymentOption` rejects it explicitly: "partial-upfront is a real payment option elsewhere in CUDly (AWS offers it), so the caller needs to know Azure specifically has no equivalent" (reservations/purchase.go:120). The 0.5 is a hardcoded ratio with no source in the pricing payload, so the UI shows a cost breakdown for a purchase that can never be made. +- evidence: + ```go + case "Partial Upfront", "partial-upfront": + upfrontCost = totalCost * 0.5 + recurringCost = (totalCost * 0.5) / hoursInTerm + ``` +- suggested fix: delete the partial-upfront case so it falls into the existing `default:` error branch, matching `BillingPlanForPaymentOption`. +- verdict: CONFIRMED — savingsplans/client.go:426-428 accepts "Partial Upfront"/"partial-upfront" and splits the total on a hardcoded 0.5 with no pricing-payload source, while `BillingPlanForPaymentOption` rejects the same value as having no Azure equivalent (reservations/purchase.go:119-127). +- issue: (pending cross-reference) + +### A08-019 An unrecognized term is amortized over 12 months +- category: money-path +- severity: medium +- location: providers/azure/internal/recommendations/converter.go:388 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: `normaliseTerm` passes an unrecognized Azure term through verbatim with a warning (converter.go:350), so a "P5Y" recommendation keeps `Term == "P5Y"`. `termToMonths("P5Y")` then returns 12, and `ExpandPaymentVariants` sets the monthly variant's `RecurringMonthlyCost` to `CommitmentCost / 12` — five times the real monthly charge for a five-year commitment. A nil or empty term is likewise defaulted to "1yr" (converter.go:342), so a payload that omitted the term is presented as a one-year purchase. +- evidence: + ```go + func termToMonths(term string) int { + switch term { + case "3yr": + return 36 + default: + return 12 + } + } + ``` +- suggested fix: return `(int, error)` and have `ExpandPaymentVariants` leave `RecurringMonthlyCost` nil for a term it cannot map, rather than amortizing over an assumed year. +- verdict: CONFIRMED — `normaliseTerm` (converter.go:341-353) passes an unrecognised term through verbatim after a warning and defaults nil/empty to "1yr", and `termToMonths` (converter.go:388-395) returns 12 for anything but "3yr", which `ExpandPaymentVariants` (converter.go:427, 436) divides CommitmentCost by for the monthly variant. +- issue: (pending cross-reference) + +### A08-023 A GCP SKU with no ServiceRegions matches every region +- category: money-path +- severity: medium +- location: providers/gcp/services/computeengine/client.go:1062 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: `skuMatchesMachineType` returns true for any SKU whose `ServiceRegions` is nil, so a global or region-less catalogue entry is accepted as the price for whichever region the client is scoped to. Combined with the last-wins assignment in `extractComputePricingFromSKUs` (client.go:1018-1022), the region-less entry can be the one that ends up as `onDemand` or `commitment`, and `savingsPercentage`, `BreakEvenMonths` and `RecurringMonthlyCost` are all computed from it. The same `nil ServiceRegions` fallthrough is in cloudsql/client.go, memorystore/client.go and cloudstorage/client.go. +- evidence: + ```go + if sku.ServiceRegions != nil { + for _, serviceRegion := range sku.ServiceRegions { + if strings.EqualFold(serviceRegion, region) { return true } + } + return false + } + return true + ``` +- suggested fix: return false when `ServiceRegions` is empty; a SKU that does not declare the region should not price that region. +- verdict: PLAUSIBLE — the fallthrough is real and reaches money (computeengine/client.go:1062, cloudsql:456, cloudstorage:448, memorystore:442; last-wins at client.go:1018-1022 feeds `getComputePricing` client.go:966-981, whose CommitmentPrice/SavingsPercentage become CommitmentCost, BreakEvenMonths and RecurringMonthlyCost at client.go:1160-1175 and 1140), and a `["global"]` SKU would not match either, so only a literal nil triggers it; nothing in the tree establishes that the Cloud Billing catalog ever returns a SKU with nil `ServiceRegions`, and the behaviour is asserted deliberately at client_test.go:243-252. +- issue: (pending cross-reference) + +### A08b-011 Currency is defaulted to USD, then silently overwritten by whichever item came last, and never validated +- category: money-path +- severity: high +- location: providers/azure/services/cache/client.go:600 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd (same pattern in the five sibling extractors and in `gcp/services/{cloudsql,cloudstorage,memorystore}/client.go`) +- failure scenario: `currency = "USD"` is a fabricated default when the API reports none, and each subsequent item overwrites it, so the reported currency can belong to a different item than the reported price. `GetOfferingDetails` copies it straight into `common.OfferingDetails.Currency` while `TotalCost`/`UpfrontCost` are compared downstream against a USD-denominated `MaxPurchaseAmount` spend cap that has no currency field. A tenant priced in a weaker currency clears a cap it should not, in exactly the shape recorded for PR #1515. +- evidence: + ```go + currency = "USD" + termStr := azureTermString(termYears) + for _, item := range items { + if item.CurrencyCode != "" { + currency = item.CurrencyCode + } + ``` +- suggested fix: return an error when the price items carry no currency or disagree on one, and have the offering path fail closed on any non-USD currency reaching a spend-cap comparison. +- verdict: CONFIRMED — the fabricated `currency = "USD"` seed followed by an unconditional per-item overwrite is at cache:600-606, cosmosdb:598-604, search:511-517, managedredis:521-526, synapse:457-466, and on the GCP side at cloudstorage:388/400-402 with identical copies in cloudsql and memorystore. +- severity-adjusted: medium — the spend-cap half of the scenario is not established: `OfferingDetails.Currency` has no consumer at all (only pkg/common/types.go:444 and pkg/provider/interface.go:50 mention the type outside providers), so nothing carries it to a `MaxPurchaseAmount` comparison. +- issue: (pending cross-reference) + +### A08b-012 `managedredis.GetOfferingDetails` silently bills an unrecognized payment option as all-upfront +- category: money-path +- severity: high +- location: providers/azure/services/managedredis/client.go:355 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: cache, cosmosdb, database, search and synapse all return an error in this `default` branch and their comments cite managedredis as the model; managedredis itself was never fixed. A recommendation carrying `PaymentOption: "partial-upfront"` (or any typo) is quoted with the entire term cost as `UpfrontCost` and `RecurringCost: 0`, so the user is shown, and approves, a one-time charge for a commitment that will actually bill monthly. +- evidence: + ```go + case "monthly", "no-upfront": + upfrontCost = 0 + recurringCost = totalCost / (float64(termYears) * 12) + default: + upfrontCost = totalCost + } + ``` +- suggested fix: return the same explicit error the five sibling clients return for an unsupported payment option. +- verdict: CONFIRMED — managedredis/client.go:355-357 has a bare `default: upfrontCost = totalCost` while the five siblings return an explicit error in the same branch (cache:396-399, cosmosdb:397-400, database:427-430, search:347-350, synapse:361-364), each with the "no silent fallbacks on money-affecting fields" comment. +- severity-adjusted: medium — the branch is only reachable through `GetOfferingDetails`, which has no in-repo caller (pkg/provider/interface.go:50). +- issue: (pending cross-reference) + +### A08b-013 All three GCP service clients silently bill an unrecognized payment option as all-upfront and an unrecognized term as one year +- category: money-path +- severity: high +- location: providers/gcp/services/cloudsql/client.go:280 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd (same at `cloudstorage/client.go:295`, `memorystore/client.go:291`; term default at `cloudsql/client.go:260`, `cloudstorage/client.go:275`, `memorystore/client.go:271`) +- failure scenario: `termYears` starts at 1 and is only raised for the literals `"3yr"`/`"3"`, so a term of `"36mo"` or `"P3Y"` is priced as a one-year commitment — a third of the real term cost — with no warning. The payment `default` branch then charges the whole (already wrong) total upfront. `computeengine.termPlan` at line 48 errors on exactly this input, so the two paths disagree about whether an unknown term is fatal. +- evidence: + ```go + termYears := 1 + if rec.Term == "3yr" || rec.Term == "3" { + termYears = 3 + } + ``` +- suggested fix: reuse `termPlan`-style parsing that errors on an unrecognized term, and return an error from the payment `default` branch. +- verdict: CONFIRMED — the two-literal term test and the all-upfront `default` are at cloudsql:260-263 and 280-282, cloudstorage:275-278 and 295-297, memorystore:271-274 and 291-293; the divergence with `termPlan` is sharper than stated, since `termPlan` (computeengine:48-57) accepts `"36mo"` as three years while these three price it as one. +- severity-adjusted: medium — reachable only via `GetOfferingDetails`, which has no in-repo caller (pkg/provider/interface.go:50). +- issue: (pending cross-reference) + +### A08b-031 vCPU count falls back to an untyped `numericValue` from the recommendation overview blob +- category: money-path +- severity: medium +- location: providers/gcp/services/computeengine/client.go:1333 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: when no VCPU operation is found, the count is taken from `Content.Overview["numericValue"]`, a free-form struct field whose meaning is not contractually the vCPU count — for a cost-oriented recommendation it can be a dollar amount or a percentage. That value becomes `rec.Count` and, through `GroupCommitments`, the number of vCPUs actually committed for one or three years. The doc comment at line 1321 acknowledges the fallback is for "older recommender versions" but nothing checks which version produced the payload. +- evidence: + ```go + if gcpRec.Content.GetOverview() != nil { + if nv := gcpRec.Content.GetOverview().GetFields()["numericValue"]; nv != nil { + if count := nv.GetNumberValue(); count > 0 { + rec.Count = int(count) + } + } + } + ``` +- suggested fix: drop the fallback and return an error when no VCPU operation is present, consistent with `memoryMBFromDetails`, which already refuses to guess the memory amount. +- verdict: CONFIRMED — computeengine:1333-1339 writes `rec.Count` from `Overview["numericValue"]` with no type, unit or recommender-version check, and the doc at 1318-1321 confirms it is a bare fallback; the contrast holds, `memoryMBFromDetails` returns an explicit error rather than guessing (1395). +- issue: (pending cross-reference) + +### A08b-035 Azure Search sends an unverified `reservedResourceType` literal on the purchase path +- category: money-path +- severity: medium +- location: providers/azure/services/search/client.go:262 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: every sibling client sends an `armreservations.ReservedResourceType*` constant; search sends the string `"SearchService"`, which the comment states has no counterpart in the SDK enum at v1.1.0 or v2.0.0 and has not been checked against the live catalog. If Azure's accepted value differs, `calculatePrice` returns a price for a different reserved-resource type or the purchase is rejected mid-flow, after the two-step exchange has already begun. Nothing in the code path degrades gracefully on a wrong enum value. +- evidence: + ```go + // "SearchService" has no counterpart in the armreservations + // ReservedResourceType enum (checked v1.1.0 and v2.0.0), so it + // cannot be expressed as an SDK constant like the other service + // clients do; the literal is kept until verified against the + // live reservation catalog (see issue #1189). + "reservedResourceType": "SearchService", + ``` +- suggested fix: verify the value against the live `reservationOrders/calculatePrice` catalog and, until then, have `PurchaseCommitment` return a not-supported error rather than issuing a request built on an unverified enum. +- verdict: PLAUSIBLE — the bare literal and its self-admitted unverified status are confirmed in the purchase body at search:256-262, and every sibling does send an SDK constant, but whether Azure accepts, rejects or silently reinterprets `"SearchService"` is a live-API fact no reading of this tree can settle. +- issue: (pending cross-reference) + +### A08b-041 Synapse reads on-demand price from a different field and matches the term by substring +- category: money-path +- severity: medium +- location: providers/azure/services/synapse/client.go:456 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: every other Azure client takes on-demand from `item.UnitPrice`; synapse takes it from `item.RetailPrice`. The two fields differ for meters whose unit of measure is a multiple (a "100 Hours" meter reports the block price in one and the per-unit price in the other), so the same catalog produces different savings percentages depending on which client asked. Synapse also re-implements the term string inline and compares with `strings.Contains` rather than equality, so `ReservationTerm: "13 Years"` — or any future term string containing "3 Years" — matches a three-year request. +- evidence: + ```go + switch { + case strings.Contains(item.ReservationTerm, termStr): + if item.RetailPrice > 0 { + reservation = item.RetailPrice + } + case item.Type == "Consumption" && item.RetailPrice > 0: + onDemand = item.RetailPrice + } + ``` +- suggested fix: use the shared `azureTermString` with exact equality and read on-demand from `UnitPrice`, matching the five sibling clients, after confirming which field the API actually documents as the per-unit rate. +- verdict: CONFIRMED — synapse:472-473 reads on-demand from `RetailPrice` where cache:611, cosmosdb:609, search:522 and managedredis:530 all use `UnitPrice`, and synapse:468 matches the term with `strings.Contains` against a term string it rebuilds inline at 458-461 instead of calling the `azureTermString` helper it also declares. +- issue: (pending cross-reference) + +### A08b-043 `skuMatches*` treats a nil region list as "available everywhere" and a substring description match as identity +- category: money-path +- severity: medium +- location: providers/gcp/services/memorystore/client.go:435 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd (same at `cloudsql/client.go:449`, `cloudstorage/client.go:441`) +- failure scenario: two independent weaknesses in one guard. A SKU whose `ServiceRegions` is nil passes the region test for every region, so an `asia-east1` price can be attributed to a `europe-west4` commitment; an empty but non-nil slice fails for all regions, so the nil/empty distinction silently flips the guard's polarity. The description test is a substring match, so tier `STANDARD` matches every SKU whose description mentions "standard", and Cloud SQL tier `db-n1-standard-1` matches `db-n1-standard-16`. Combined with the last-wins loop in `extractPricingFromSKUs`, the price finally used is whichever loose match came last. +- evidence: + ```go + if !strings.Contains(strings.ToLower(sku.Description), strings.ToLower(tier)) { + return false + } + if sku.ServiceRegions != nil { + for _, serviceRegion := range sku.ServiceRegions { + if strings.EqualFold(serviceRegion, region) { + return true + } + } + ``` +- suggested fix: require an explicit region membership (treat both nil and empty as "not available here" unless the SKU is marked global) and match the tier on the SKU's structured category/description fields rather than a substring. +- verdict: CONFIRMED — the guard falls through to `return true` when `ServiceRegions` is nil and returns false after an empty loop when it is non-nil but empty (memorystore:441-451, cloudsql:455-465, cloudstorage:447-457), so the nil/empty distinction does flip the polarity; the tier test above it is a plain `strings.Contains` on the description, and `extractStoragePricingFromSKUs` (cloudstorage:390-409) keeps the last match rather than rejecting an ambiguous set. +- issue: (pending cross-reference) + +### A09-007 Any ladder mode string other than the literal "manual" auto-executes real exchanges +- category: money-path +- severity: high +- location: pkg/exchange/auto.go:248 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: `RIExchangeConfig.Mode` is an untyped `string` and the dispatch compares it to a bare literal, even though `ExchangeMode` typed constants exist 350 lines below at line 606. With `Config.Mode` set to `"Manual"`, `"MANUAL"`, `""`, or any typo, the manual branch is skipped and `processAutoExchange` runs the irreversible `AcceptReservedInstancesExchangeQuote`. The failure direction is the dangerous one: an unrecognized mode buys rather than holds. `internal/api/handler.go:554` documents that the API layer does forward `""`, `"Auto"` and arbitrary typos to the scheduler, so the unvalidated value does reach this comparison. +- evidence: + ```go + if params.Config.Mode == "manual" { + outcome := processManualExchange(ctx, params, rec, offeringID, paymentDueStr) + ``` +- suggested fix: type `RIExchangeConfig.Mode` as `ExchangeMode`, validate it once at the top of `RunAutoExchange`, and dispatch on `ExchangeModeAuto` explicitly so an unknown mode errors instead of falling through to auto-purchase. +- verdict: CONFIRMED — `RIExchangeConfig.Mode` is a bare `string` (pkg/exchange/auto.go:72), the dispatch compares it to the literal `"manual"` (auto.go:248) while the typed constants sit unused at auto.go:606-607, and the fall-through calls `processAutoExchange`. internal/api/handler.go:551-559 states in the code that `GlobalConfig.Validate` never constrains `RIExchangeMode`, so any non-"manual" string reaches this comparison via PUT /api/config. +- severity-adjusted: medium — a compensating gate exists one layer up: `riExchangeArmState.armed()` (internal/api/handler.go:562) is deliberately defined as `mode != "manual"`, so `requireRIExchangeAutoModeGrant` demands the execute:ri-exchange verb before any typo'd mode can be written. The residual risk is an already-authorized operator typing "Manual" and getting auto-execution, not an unprivileged escalation. +- issue: (pending cross-reference) + +### A09-016 A failed pending-exchange cancellation only warns, and the run then creates a second pending set +- category: money-path +- severity: medium +- location: pkg/exchange/auto.go:173 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: a transient database error makes `CancelPendingExchangesByOrigin` fail. The run logs a warning and continues to `processManualExchange`, which issues fresh approval tokens and saves new pending records for the same underutilized RIs. The user's inbox now holds two live approval links per RI: the previous run's (still `pending`, valid for its 24h `ExpiresAt`) and the new one. Approving both executes two exchanges against the same source RI, and the second failure is only caught downstream by AWS. +- evidence: + ```go + cancelled, err := params.Store.CancelPendingExchangesByOrigin(ctx, origin) + if err != nil { + logging.Warnf("failed to cancel pending exchanges: %v", err) + } else if cancelled > 0 { + ``` +- suggested fix: return the error from `RunAutoExchange` when cancellation fails, so a run that cannot clear stale approvals does not issue overlapping ones. +- verdict: CONFIRMED — the cancellation error is logged and swallowed (pkg/exchange/auto.go:171-174) with no early return, so control falls through to the recommendation loop at auto.go:200 and `processManualExchange` mints a fresh `GenerateApprovalToken` and saves a new `pending` record with a 24h `ExpiresAt` (auto.go:344, :371, :383-386). Note the blast radius is bounded: auto.go:467-470 records that AWS replaces the source RI atomically, so a second approval on the same RI fails at the provider rather than double-buying. +- issue: (pending cross-reference) + +### A09-021 A zero Count fabricates a scaling ratio equal to the instance count +- category: money-path +- severity: medium +- location: pkg/recfilter/sizing.go:301 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: for a malformed recommendation with `Count == 0` but non-zero costs — reachable for any parser that populates `AverageInstancesUsedPerHour` and cost fields before setting `Count` — the fallback sets `ratio = float64(nTarget)`. With `nTarget = 3` and `CommitmentCost = 1000`, `ScaleRecommendationCosts` multiplies the commitment to `3000` and `EstimatedSavings` likewise, then `adjusted.Count = 3`. The in-code justification ("so a zero-cost rec stays zero-cost rather than NaN") holds only when every cost field happens to be zero; when it is not, the rec is inflated by an invented factor rather than rejected. +- evidence: + ```go + var ratio float64 + if rec.Count > 0 { + ratio = float64(nTarget) / float64(rec.Count) + } else { + ratio = float64(nTarget) + } + adjusted := common.ScaleRecommendationCosts(rec, ratio) + ``` +- suggested fix: drop the recommendation (with a named drop reason) when `rec.Count <= 0`, since a rec with no count cannot have its per-unit costs derived. +- verdict: PLAUSIBLE — the arithmetic is exactly as described: `ratio = float64(nTarget)` when `rec.Count <= 0` (pkg/recfilter/sizing.go:295-301), then `ScaleRecommendationCosts` multiplies every cost field by that invented factor and `adjusted.Count = nTarget` (sizing.go:302-303). The runtime condition I could not establish from source is a producer: reaching this branch needs `AverageInstancesUsedPerHour > 0` (sizing.go:233) and `gapPct > 0` together with `Count == 0` and non-zero costs, and I found no parser that emits that shape — the in-code comment at sizing.go:288-292 calls it a malformed rec. +- issue: (pending cross-reference) + +### A09-023 The duplicate checker misses AWS RIs in the "queued" state, permitting a double purchase +- category: money-path +- severity: medium +- location: pkg/recfilter/dedupe.go:82 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: `Commitment.State` is an untyped string compared against two bare literals. The AWS EC2 `ReservedInstanceState` enum also carries `queued` (a scheduled future RI purchase), which `providers/aws/services/ec2/client.go:102` passes through verbatim as `string(ri.State)`. An RI queued twenty minutes ago has a future `StartDate` (so `StartDate.After(cutoffTime)` is true) but is filtered out by the state check, so `existingMap` never records it and the next run recommends and purchases the same RI again. The literals also carry no link to the SDK enum, so a future state addition repeats the gap. +- evidence: + ```go + func isRecentActiveCommitment(c common.Commitment, cutoffTime time.Time) bool { + return (c.State == "active" || c.State == "payment-pending") && c.StartDate.After(cutoffTime) + } + ``` +- suggested fix: add `"queued"` to the accepted set and define the accepted states as named constants in `pkg/common` next to `Commitment`, so each provider maps its SDK enum onto a typed vocabulary rather than a raw string. +- verdict: CONFIRMED — `isRecentActiveCommitment` compares `c.State` to the two bare literals (pkg/recfilter/dedupe.go:82-84) and `filterRecentCommitments` drops everything else before `buildExistingCommitmentsMap` ever sees it (dedupe.go:72-75), so a queued RI is not recorded and the next run re-recommends it. One correction to the mechanism: the queued RI never reaches dedupe in the first place, because `GetExistingCommitments` filters the AWS call itself with `state IN ("active","payment-pending")` at providers/aws/services/ec2/client.go:78-84. The suggested fix is therefore incomplete — adding "queued" to dedupe.go:83 alone changes nothing until the upstream API filter is widened too. +- issue: (pending cross-reference) + +### A10-008 `determineCSVCoverage` infers "the operator did not set --coverage" from the literal value 80.0 +- category: money-path +- severity: medium +- location: cmd/multi_service_csv.go:21 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: `cudly --input-csv recs.csv --coverage 80 --purchase`. The operator explicitly asked for 80% of each row's count. Because the value happens to equal the flag default, `determineCSVCoverage` returns 100.0 and the run buys the full count on every row, 25% more than requested. `validateTargetCoverage` (cmd/validators.go:121) already solves exactly this problem correctly with `cmd.Flags().Changed("coverage")`; this call site compares against the magic literal instead. `TestDetermineCSVCoverage` codifies the current behaviour ("Default coverage (80) changed to 100 for CSV"), so the suite stays green with the bug present. +- evidence: + ```go + func determineCSVCoverage(cfg Config) float64 { + if cfg.Coverage == 80.0 { + // User didn't override the default, so use 100% for CSV mode + return 100.0 + } + return cfg.Coverage + } + ``` +- suggested fix: Thread `cmd.Flags().Changed("coverage")` into Config (or pass the `*cobra.Command`) and branch on that, exactly as `validateTargetCoverage` does. +- verdict: CONFIRMED — cmd/multi_service_csv.go:21 branches on the literal 80.0 while validateTargetCoverage solves the same problem with cmd.Flags().Changed("coverage") at cmd/validators.go:121, and cmd/multi_service_csv_test.go:23-29 pins the buggy mapping as expected behaviour. +- issue: (pending cross-reference) + +### A10-016 Account filters match on substring, so a scoping flag reaches accounts the operator did not name +- category: money-path +- severity: medium +- location: cmd/multi_service_filters.go:205 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: `accountMatchesFilter` returns true on any substring hit. In an org with accounts `dev` and `dev-prod-mirror`, `--include-accounts dev --purchase` buys commitments in both. In an org with `prod` and `preprod`, `--exclude-accounts prod` removes both, silently under-buying. There is no anchored or exact-match mode and no warning when a filter matches more than one account, so neither direction is visible in the output. +- evidence: + ```go + func accountMatchesFilter(accountLower, filter string) bool { + filterLower := strings.ToLower(filter) + return filterLower == accountLower || strings.Contains(accountLower, filterLower) + } + ``` +- suggested fix: Default to exact match and require an explicit wildcard (`*dev*`) for substring behaviour; at minimum log every account name a filter expanded to before the confirmation prompt. +- verdict: CONFIRMED — accountMatchesFilter returns strings.Contains (cmd/multi_service_filters.go:203-206) and both list checks call it (:184 and :195), so a filter widens in the include direction and in the exclude direction with no anchored mode and no expansion log. +- issue: (pending cross-reference) + +### A11-005 A saved default_coverage of 0 is rewritten to 80 on the next load-and-save cycle +- category: money-path +- severity: high +- location: frontend/src/settings.ts:3184 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: `loadGlobalSettings` populates the target-coverage input with `String(data.global.default_coverage || 80)`. A stored 0 (buy no commitments by default) is falsy, so the input shows 80. `saveGlobalSettings` then reads that input and persists `default_coverage: 80` (settings.ts:3482-3507), so opening Settings and clicking Save silently raises the organisation-wide purchase target from 0% to 80%. The same `||` is applied to `cachedGlobalDefaults.coverage` (settings.ts:3192), which drives the "Inherit (currently: X)" labels in the override modal, and to `notification_days_before || 3` (settings.ts:3196), where 0 means "notify on the day". The file already knows the rule: `populateGraceInput` documents that "An explicit 0 must round-trip as 0" and uses `?? 7` (settings.ts:196-200), and the per-SP-card coverage on line 3265 correctly uses `??`. +- evidence: + ```typescript + // settings.ts:3183-3193 + const coverageInput = document.getElementById('setting-default-coverage') as HTMLInputElement | null; + if (coverageInput) coverageInput.value = String(data.global.default_coverage || 80); + cachedGlobalDefaults = { + term: data.global.default_term || 3, + payment: data.global.default_payment || 'all-upfront', + coverage: data.global.default_coverage || 80, + }; + ``` +- suggested fix: switch the coverage and notification-days reads to `??`, matching `populateGraceInput` and the SP-card path on line 3265. +- verdict: CONFIRMED — `default_coverage` is a plain `json:"default_coverage"` with no omitempty (internal/config/types.go:22), so a stored 0 reaches the form as 0, renders as 80 through the `||` at frontend/src/settings.ts:3184, and is written back as 80 by the save at settings.ts:3482/3507, while the sibling SP-card read on settings.ts:3265 uses `??` and `populateGraceInput` (settings.ts:196-200) documents the opposite rule. The `notification_days_before || 3` half of the claim does not hold: the save validator requires 1..30 (settings.ts:3476), so 0 is not a storable value there. +- severity-adjusted: medium — 0 is not unambiguously "buy nothing" on the backend either (`handler_dashboard.go:394` treats `DefaultCoverage > 0` as the only set value), so this is a silent state rewrite rather than a live purchase-widening. +- issue: (pending cross-reference) + +### A12-014 "Avg Monthly Savings" divides by the number of non-empty buckets, not the number of periods +- category: money-path +- severity: medium +- location: frontend/src/modules/savings-history.ts:289 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: `QueryHistory` groups `purchase_history` by `date_trunc(interval, timestamp)` and returns only buckets that contain rows, so a 7-day hourly window with purchases in 3 hours yields `dataPoints.length === 3`. The tile labelled "Avg Monthly Savings" then reads `total / 3` rather than an average over the 168 periods in the window, and the value changes with how sparse the data is rather than with the savings rate. The same figure carries a `/mo` suffix, so the reader cannot tell the denominator is data-dependent. +- evidence: + ```ts + const avgPerPeriod = dataPoints.length > 0 ? totalSavings / dataPoints.length : 0; + ``` +- suggested fix: Divide by the number of buckets in the requested window (derivable from `start`, `end` and `interval`), or relabel the tile to say it averages over periods with activity. +- verdict: CONFIRMED — The query GROUPs by date_trunc and folds only returned rows into the bucket index (internal/api/analytics_postgres.go:174-195), so empty periods produce no data point and `totalSavings / dataPoints.length` (frontend/src/modules/savings-history.ts:289) divides by the count of periods that happened to contain purchases. +- issue: (pending cross-reference) + +### A12-018 The approval-details modal falls back to a no-amount body and still offers Approve +- category: money-path +- severity: medium +- location: frontend/src/approval-details.ts:364 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: `buildApprovalDetailsBody` wraps `api.getPurchaseDetails` in a try/catch. On a 403, a 404 or a network blip it returns `buildApprovalDetailsFallback()`, a single sentence with no upfront, no savings, no per-rec table. The confirm dialog still opens with a working "Approve purchase" button (`frontend/src/purchases-deeplink.ts:91`), so the informed-consent guarantee the module exists for silently disappears exactly when the data cannot be read. The same fallback is used when `details.recommendations` is empty (line 64). +- evidence: + ```ts + } catch (err) { + console.error('Failed to load purchase details for approval modal:', err); + return buildApprovalDetailsFallback(); + } + ``` +- suggested fix: Return a marker with the fallback and have `handlePurchaseDeeplink` refuse to render the confirm button when details could not be loaded, pointing the user at History instead. +- verdict: CONFIRMED — Any getPurchaseDetails rejection returns the amount-free single-sentence fallback (frontend/src/approval-details.ts:364-372, :379-384) and handlePurchaseDeeplink passes that body straight into confirmDialog with a live `Approve purchase` button and no guard (frontend/src/purchases-deeplink.ts:85-95). +- issue: (pending cross-reference) + +### A12-020 The Capacity % input silently applies 100% when it holds 0 or is blank +- category: money-path +- severity: medium +- location: frontend/src/recommendations.ts:3712 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: The persist handler is `Math.max(1, Math.min(100, parseInt(capacityInput.value, 10) || 100))`. Typing `0` gives `parseInt('0') === 0`, which `||` replaces with 100; clearing the field gives `NaN`, likewise replaced with 100. The input still displays `0` or blank while `handleBulkPurchaseClick` reads the persisted 100 from `loadBulkPurchaseState()`, so the user believes they scoped the buy down and gets a full-size purchase. +- evidence: + ```ts + const persist = (): void => { + saveBulkPurchaseState({ + payment: tbState.payment, + capacity: Math.max(1, Math.min(100, parseInt(capacityInput.value, 10) || 100)), + }); + }; + ``` +- suggested fix: Reject a blank or out-of-range capacity with an inline error and disable the Purchase button, rather than substituting 100. +- verdict: CONFIRMED — The persist handler collapses both `0` and NaN to 100 via `|| 100` (frontend/src/recommendations.ts:3712) while the input keeps displaying the typed value, and handleBulkPurchaseClick reads the persisted state through loadBulkPurchaseState (frontend/src/recommendations.ts:3910). +- issue: (pending cross-reference) + +### A12-035 Marketplace consent copy prints "5.000000000000004%" from a float subtraction +- category: money-path +- severity: medium +- location: frontend/src/history.ts:1518 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: `1 - 0.95` is `0.050000000000000044` in IEEE-754, so `(1 - AWS_MARKETPLACE_BUYER_DISCOUNT) * 100` evaluates to `5.000000000000004`. Every user who opens the Sell-on-Marketplace dialog reads "prices the listing at 5.000000000000004% below remaining value" in the sentence that justifies the amount they are about to accept. +- evidence: + ```ts + noteEl.textContent = `AWS charges a ${AWS_MARKETPLACE_FEE_PERCENT}% transaction fee on proceeds. The default schedule prices the listing at ${(1 - AWS_MARKETPLACE_BUYER_DISCOUNT) * 100}% below remaining value. ...`; + ``` +- suggested fix: Introduce `AWS_MARKETPLACE_BUYER_DISCOUNT_PERCENT = 5` and derive the multiplier from it, interpolating the integer. +- verdict: CONFIRMED — AWS_MARKETPLACE_BUYER_DISCOUNT is 0.95 (frontend/src/history.ts:36) and `(1 - 0.95) * 100` evaluates to 5.000000000000004, which is interpolated straight into the consent copy (frontend/src/history.ts:1518). +- issue: (pending cross-reference) + +### A12-038 Cost chips, picker labels and the running total hardcode `$`, discarding `currency_code` +- category: money-path +- severity: medium +- location: frontend/src/riexchange.ts:1390 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: `pkg/exchange.OfferingOption` carries `currency_code`, described as the ISO-4217 currency the `EffectiveMonthlyCost` is denominated in, but the frontend `OfferingOption` type omits the field and all four display sites pass a literal `'$'` (lines 1390, 1524, 1632, 1688). A €52.00 offering renders as `$52.00/mo` and the operator compares it directly against the USD figures elsewhere on the page. The exchange-history payment column (line 2216) does the same with a raw `'$' + rec.payment_due`, and an empty `payment_due` renders as a bare `$`. +- evidence: + ```ts + return alternatives + .map((alt) => `${escapeHtml(alt.instance_type)} ${formatCurrency(alt.effective_monthly_cost, '$', 2)}/mo`) + .join(' '); + ``` +- suggested fix: Add `currency_code` to the TypeScript `OfferingOption` and pass it (or its symbol) into `formatCurrency`; render `—` for an empty amount instead of a lone symbol. +- verdict: CONFIRMED — The Go OfferingOption carries `currency_code` as the ISO-4217 denomination (pkg/exchange/reshape.go:70-74) while the TS interface omits it entirely (frontend/src/api/types.ts:730-734) and all four display sites pass a literal '$' (frontend/src/riexchange.ts:1390, :1524, :1632, :1688); the history cell concatenates a bare '$' onto payment_due (frontend/src/riexchange.ts:2215). +- issue: (pending cross-reference) + +### A12-039 `max_payment_due_usd` is filled with the quote amount without checking its currency +- category: money-path +- severity: medium +- location: frontend/src/riexchange.ts:1814 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: The client copies `quote.PaymentDueRaw`, an amount denominated in `quote.CurrencyCode` (which the modal displays as a separate row precisely because it is not always USD), into the field named `max_payment_due_usd`. The backend parses it as a USD decimal and uses it both as the `pkg/exchange` cap and as the `MaxPurchaseAmount` for a USD-denominated permission constraint (`handler_ri_exchange.go:1781`). Unlike the Azure path's `checkAzureExchangeMoneyGuardrails`, the AWS path never compares currencies, so a non-USD quote is evaluated against a cap in different units. +- evidence: + ```ts + max_payment_due_usd: modalQuote.PaymentDueRaw, + ``` +- suggested fix: Refuse to enable Execute (or send the currency for the backend to reject) unless `quote.CurrencyCode` is USD. +- verdict: CONFIRMED — The client copies PaymentDueRaw into max_payment_due_usd while displaying CurrencyCode as a separate row (frontend/src/riexchange.ts:1814, :1861-1862), and the AWS handler parses it as a plain decimal and feeds it to the USD-denominated MaxPurchaseAmount constraint with no currency comparison (internal/api/handler_ri_exchange.go:1781-1811), unlike checkAzureExchangeMoneyGuardrails which rejects a currency mismatch 422 (internal/api/handler_ri_exchange.go:1123-1124). +- issue: (pending cross-reference) + +### A12-040 RI exchange Execute has no confirmation step for an irreversible payment +- category: money-path +- severity: medium +- location: frontend/src/riexchange.ts:1730 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: A single click on "Execute Exchange" submits the AWS accept call, which the backend's own comment describes as financially irreversible. The far less consequential Approve action in the same file goes through `confirmDialog` (line 2247) and so does enabling auto mode (line 2085). A misclick on the primary button sitting next to "Get Quote" spends money with no interstitial and no restated amount. The Approve dialog itself (line 2247) also carries no amount even though the row has `payment_due`, source/target types and counts. +- evidence: + ```ts + executeBtn.addEventListener('click', () => { + void submitModalExecute(); + }); + ``` +- suggested fix: Gate `submitModalExecute` behind `confirmDialog` with `destructive: true`, quoting `modalQuote.PaymentDueRaw` plus `CurrencyCode` and the targets actually being submitted; add the same amounts to the Approve dialog. +- verdict: CONFIRMED — The Execute click goes straight to submitModalExecute (frontend/src/riexchange.ts:1730-1732) while the sibling Approve action does gate on confirmDialog (frontend/src/riexchange.ts:2247-2251), and that dialog's body carries no payment_due even though the record has one (frontend/src/api/types.ts:841). +- issue: (pending cross-reference) + +### A15-009 Azure retail pricing labels prices with a currency taken from unrelated items +- category: money-path +- severity: medium +- location: providers/azure/services/cache/client.go:600 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: In all seven `extract*Pricing` functions the loop assigns `currency` from every item that carries a non-empty `CurrencyCode`, while `onDemand` and `reservation` are assigned only from items matching the term or `Type == "Consumption"`. The returned currency is therefore whatever the last item in the response happened to carry, not the currency of the two prices actually returned. With a response whose Consumption item is USD and whose later reservation item is EUR, the function returns `onDemand` in USD, `reservation` in EUR, and labels the pair EUR; `getRedisPricing` then computes `savingsPercentage` by mixing the two denominations and reports a fabricated saving. No caller pins `currencyCode` in the request (`params.Add` only ever sets `$filter` and `api-version` across all seven services), so the code depends entirely on the Retail Prices API's undocumented default of USD to keep the item set single-currency. Same shape at compute:766, cosmosdb:598, database:618, managedredis:521, search:511, synapse:457. +- evidence: + ```go + currency = "USD" + termStr := azureTermString(termYears) + + for _, item := range items { + if item.CurrencyCode != "" { + currency = item.CurrencyCode + } + if item.ReservationTerm == termStr { + reservation = item.RetailPrice + } else if item.Type == "Consumption" { + onDemand = item.UnitPrice + } + } + ``` +- suggested fix: Pin `currencyCode=USD` in the request params, and take `currency` only from the items the prices were read from, returning an error when those two disagree rather than labelling a mixed pair. +- verdict: CONFIRMED — I opened all seven sites and the mechanism is identical in each: an unconditional `if item.CurrencyCode != "" { currency = item.CurrencyCode }` that last-writer-wins over every item, while `onDemand`/`reservation` are gated on `ReservationTerm`/`Type` (cache/client.go:604-612, compute:770-778, cosmosdb:602-610, database:622-630, managedredis:523-529, search:515-523, synapse:465-475). The count of seven is right; the only naming slip is that the managedredis copy is `parsePriceItems` (managedredis/client.go:520), not an `extract*Pricing`. The request-side claim holds too: `params.Add` across all seven sets only `$filter` and `api-version` (e.g. cache/client.go:577-578), never `currencyCode`. And the mixing really reaches money: cache/client.go:558 computes `savingsPercentage` from `onDemandPrice` and `reservationPrice` and stamps the loop's `currency` onto the result at :564. +- issue: (pending cross-reference) + +### A06-006 Convertible-RI monthly cost silently becomes $0 when the term duration is missing +- category: money-path +- severity: medium +- location: internal/server/handler_ri_exchange.go:234 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: `MonthlyCost` feeds the exchange engine's cross-family dollar-units pre-filter (`exchange.RIInfo.MonthlyCost`). If `DescribeReservedInstances` returns an RI with `Duration == 0` (unset field, an offering type the parser does not populate), this returns `0` instead of an error, so the source RI is presented to the filter as costing nothing per month. A comparison that expects "target must not cost more than source" then admits every candidate target, and in `automatic` mode the exchange is executed with real money. The zero is indistinguishable from a genuine $0 RI. +- evidence: + ```go + func monthlyCostFromConvertibleRI(ri ec2svc.ConvertibleRI) float64 { + if ri.Duration <= 0 { + return 0 + } + hoursPerTerm := float64(ri.Duration) / 3600 + if hoursPerTerm <= 0 { + return 0 + } + ``` +- suggested fix: return `(float64, error)` (or `*float64`) and have `convertForAutoExchange` fail the run for that RI rather than emitting a fabricated $0. +- verdict: PLAUSIBLE — the bare `return 0` is there (internal/server/handler_ri_exchange.go:234-241) and `MonthlyCost <= 0` makes `pricingGatePasses` return true unconditionally, admitting every alternative (pkg/exchange/reshape.go:679-680), but reaching it requires `DescribeReservedInstances` to omit `Duration` (providers/aws/services/ec2/client.go:757 is a nil-safe `aws.ToInt64`), a runtime condition I could not establish from source; the money escalation is also blunted because every auto exchange is still gated on AWS's own `quote.IsValidExchange` and the per-exchange cap (pkg/exchange/auto.go:310-325). +- severity-adjusted: low — the fabricated $0 only widens a local pre-filter that the code documents as a skip-on-zero approximation (pkg/exchange/reshape.go:19-26,144-151); AWS re-validates each exchange before any payment is made. +- issue: (pending cross-reference) + +### A09-020 A short or malformed idempotency token silently drops Azure purchase idempotency +- category: money-path +- severity: medium +- location: pkg/common/tokens.go:115 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: `IdempotencyGUID` returns `""` for any token under 32 characters or whose first 32 characters are not hex, and `ReservationOrderID` then hands back `fallback` — the caller's prior non-idempotent, typically random or timestamp-derived, order ID. A truncated or corrupted `IdempotencyToken` (from a shortened DB column, a partial write, or a future caller passing a non-hex value) therefore silently reverts to the non-idempotent path on the exact re-drive scenario the token exists to protect, and a re-driven Azure reservation purchase creates a second reservation. There is no signal at either call site distinguishing "no token supplied" from "token supplied but unusable". +- evidence: + ```go + func ReservationOrderID(token, fallback string) string { + if guid := IdempotencyGUID(token); guid != "" { + return guid + } + return fallback + } + ``` +- suggested fix: return an error (or a distinguishable second return value) when `token != ""` but does not yield a GUID, so a malformed token aborts the purchase instead of downgrading it to non-idempotent. +- verdict: PLAUSIBLE — the fail-open shape is exactly as described (pkg/common/tokens.go:98-107 returns "" for a short or non-hex token; :115-120 then returns `fallback` with no way to tell "no token" from "unusable token"). The runtime condition I could not establish is a malformed token: the only production producer is `DeriveIdempotencyToken` (tokens.go:64-67), which always yields a 64-char sha256 hex string, and it is computed at call time in internal/purchase/execution.go:606 rather than read from a DB column, so the "shortened column / partial write" path does not exist today. Two corrections to the scenario: the sole `ReservationOrderID` caller is providers/azure/services/savingsplans/client.go:282 (a Savings Plans alias name, not a reservation order), and providers/azure/services/internal/reservations/purchase.go:21 records that the reservations path deliberately does not use this mechanism at all. +- severity-adjusted: low — reachable only via a future caller that supplies a non-derived token. +- issue: (pending cross-reference) + +### A09-022 A Savings Plan rec with unexpected Details survives sizing at its full unscaled commitment +- category: money-path +- severity: medium +- location: pkg/recfilter/sizing.go:45 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: when `Details` is not a non-nil `*SavingsPlanDetails`, `ApplyCoverage` logs a warning and appends `adjusted` — which is still `rec`, unscaled — to the result. An operator running `--coverage 50` on a rec whose Details failed to decode (see A09-018) gets a plan carrying the full 100% `HourlyCommitment`, and the purchase commits twice the dollars requested. `logf` is nil-safe and callers may pass nil, so the warning can be invisible. The `applyTargetCoverageSP` sibling at line 340 has the same shape. +- evidence: + ```go + if common.IsSavingsPlan(rec.Service) { + if details, ok := rec.Details.(*common.SavingsPlanDetails); ok && details != nil { + adjusted = common.ScaleRecommendationCosts(adjusted, ratio) + } else { + logf.printf("WARNING: SP recommendation for service %q has missing or unexpected Details (%T); passing through unscaled\n", rec.Service, rec.Details) + } + result = append(result, adjusted) + ``` +- suggested fix: drop the recommendation with a named drop reason rather than passing it through, so a sizing flag never silently fails open on the dollar-denominated field it exists to shrink. +- verdict: PLAUSIBLE — the fail-open branch is real: `ApplyCoverage` appends the still-unscaled `adjusted` after logging (pkg/recfilter/sizing.go:45-53), and `applyTargetCoverageSP` returns `(rec, true)` unchanged on the same condition (sizing.go:339-343). What I could not establish is a producer of an SP rec with missing or wrong Details on this path: the AWS SP parser always sets a non-nil `&common.SavingsPlanDetails{}` (providers/aws/recommendations/parser_sp.go:391), and the cited decode failure (A09-018) does not feed here — `ApplyCoverage`'s only callers are the CLI wrappers at cmd/helpers.go:118 and :124, which never route through `DecodeServiceDetailsFor`. +- severity-adjusted: low — no reachable producer today; the branch is defensive. +- issue: (pending cross-reference) + +### A11-006 Per-Savings-Plan coverage is persisted with no range or integer check +- category: money-path +- severity: medium +- location: frontend/src/settings.ts:3562 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: the global `setting-default-coverage` is guarded at save time with an explicit `Number.isInteger` + 0..100 check whose comment calls it "the defensive guard for clients that bypass" the inline validator (settings.ts:3469-3488). The four per-plan-type SP coverage inputs go through no such guard: any finite number is accepted, so `-40`, `500` or `12.7` reaches `updateServiceConfig` as the service's target coverage. The inline validator wired at settings.ts:2695-2702 only paints a red hint; it does not block Save. +- evidence: + ```typescript + // settings.ts:3562-3566 + if ('coverageId' in field && field.coverageId) { + const rawCov = byId(field.coverageId)?.value ?? ''; + const parsed = Number(rawCov); + if (rawCov !== '' && Number.isFinite(parsed)) coverage = parsed; + } + ``` +- suggested fix: reuse the same `Number.isInteger(parsed) && parsed >= 0 && parsed <= 100` test the global field uses and abort the save with the same toast when it fails. +- verdict: CONFIRMED — the save path applies no integer or range test (frontend/src/settings.ts:3562-3566), and native validation cannot cover for it because the Purchasing panel's inputs sit outside `#global-settings-form` (index.html:343-423 vs the SP cards at 587-611) and the Save button dispatches a synthetic `new Event('submit')` (app.ts:296-297), which skips constraint validation entirely. The out-of-range half of the claim is blocked one layer down: `ServiceConfig.Validate` rejects coverage outside 0..100 (internal/config/validation.go:517-520) from the update handler at internal/api/handler_config.go:352, so -40 and 500 fail the PUT loudly. +- severity-adjusted: low — only a fractional coverage such as 12.7 survives end to end; the range violations surface as a failed save rather than a bad stored target. +- issue: (pending cross-reference) + +### A12-012 Two different hours-per-month constants convert the same monthly figure to $/hr +- category: money-path +- severity: medium +- location: frontend/src/recommendations.ts:1471 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: The Opportunities cost-period selector divides by 720 while the Purchases savings-history unit selector divides by 730 (`frontend/src/modules/savings-history.ts:25`). The same $730/month savings figure therefore reads `$1.0139/hr` on Opportunities and `$1.00/hr` on Purchases, a 1.4% divergence between two screens the user compares directly. The daily factor diverges the same way (`1/30` versus the 30.4375 days/month the marketplace and analytics code use). +- evidence: + ```ts + const PERIOD_FACTOR: Record = { + hourly: 1 / 720, // 24 × 30 hrs/mo + daily: 1 / 30, + monthly: 1, + yearly: 12, + }; + ``` +- suggested fix: Export one `HOURS_PER_MONTH` / `DAYS_PER_MONTH` pair from `utils.ts` and have both modules derive their factors from it. +- verdict: CONFIRMED — frontend/src/recommendations.ts:1471 divides by 720 while frontend/src/modules/savings-history.ts:25 divides by HOURS_PER_MONTH = 730, two live constants for the same monthly-to-hourly conversion on two screens the user compares. +- severity-adjusted: low — display-only divergence of about 1.4 percent; neither factor reaches a submitted amount, a permission cap or a stored record +- issue: (pending cross-reference) + +### A12-044 The plans "Run now" confirmation authorises an immediate purchase without an amount +- category: money-path +- severity: medium +- location: frontend/src/plans.ts:775 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: Clicking the row's ▶ button opens a dialog whose entire body is "This will immediately execute the purchase." No upfront cost, no savings, no service, no count, no plan name. `renderApprovalDetailsBody` in `approval-details.ts` exists to solve exactly this for the email deep-link path (issue #374) and is not reused here, so the in-app immediate-execution path has strictly weaker informed consent than the email path. +- evidence: + ```ts + const runOk = await confirmDialog({ + title: 'Run purchase now?', + body: 'This will immediately execute the purchase.', + confirmLabel: 'Run now', + destructive: true, + }); + ``` +- suggested fix: Pass the `PlannedPurchase` row into the handler and render the upfront, monthly savings, count and term in the dialog body. +- verdict: CONFIRMED — The Run-now dialog body is the bare sentence with no cost, count, term or plan name (frontend/src/plans.ts:773-779) while renderApprovalDetailsBody exists for exactly this purpose on the email path (frontend/src/approval-details.ts:53) and is not reused here. +- severity-adjusted: low — the originating row already renders plan name, step, count, term, upfront and monthly savings beside the button (frontend/src/plans.ts:727-741), so the amount is on screen even though the dialog omits it +- issue: (pending cross-reference) + +### A12-058 The three marketplace money lines are rounded independently and do not add up +- category: money-path +- severity: low +- location: frontend/src/history.ts:1502 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: `formatCurrency` defaults to zero fraction digits. With `listPriceTotal = 100.5` the dialog shows "Default list price $101", "AWS fee (12%) $12" and "Estimated net proceeds $88": 12 plus 88 is 100, not 101. The three money lines in one consent dialog are mutually inconsistent. +- evidence: + ```ts + addRow('Default list price', count > 1 ? ... : formatCurrency(listPriceTotal)); + addRow(`AWS fee (${AWS_MARKETPLACE_FEE_PERCENT}%)`, formatCurrency(listPriceTotal * (AWS_MARKETPLACE_FEE_PERCENT / 100))); + addRow('Estimated net proceeds', formatCurrency(netProceedsTotal)); + ``` +- suggested fix: Pass `digits: 2` to all three, as the cents-precision summary cards elsewhere already do. +- verdict: CONFIRMED — All three rows call formatCurrency with the default CURRENCY_DEFAULT_DIGITS = 0 (frontend/src/utils.ts:14, frontend/src/history.ts:1500-1504), so list price, fee and net proceeds are rounded independently and need not sum. +- issue: (pending cross-reference) + +### A12-059 With amortize on, the upfront appears both as a lump sum and folded into Monthly Cost +- category: money-path +- severity: low +- location: frontend/src/history.ts:1156 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: For a 1-year row with `upfront_cost` 1200, `monthly_cost` 100 and `estimated_savings` 300, the toggle produces `Upfront Cost $1,200 | Monthly Cost (amortized) $200 | Monthly Savings $300`. Only the Monthly Cost header discloses the amortization, so `Upfront + 12 × Monthly` double-counts the upfront, and cost is shown on an amortized basis in the same row where savings is shown on a cash basis. +- evidence: + ```ts + ${formatCurrency(p.upfront_cost)} + ${monthlyCostCell} + ${formatCurrency(p.estimated_savings)} + ``` +- suggested fix: When amortize is on, mute the Upfront Cell with an "(included in monthly)" title so the two columns cannot be read as additive. +- verdict: CONFIRMED — The row renders the raw upfront cell beside the amortized monthly cell and the cash-basis savings cell (frontend/src/history.ts:1153-1156), and only the column header discloses the amortization (frontend/src/history.ts:1163). +- issue: (pending cross-reference) + +### Category: silent-fallback + +65 findings: 3 high, 35 medium, 27 low. + +### A08-009 GCP offering details silently bill an unrecognized payment option as all-upfront +- category: silent-fallback +- severity: high +- location: providers/gcp/services/computeengine/client.go:876 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: a recommendation with `PaymentOption` of "" (the pre-migration-000032 empty default), "partial-upfront", or any typo falls into `default:` and is reported with the whole commitment charged today and `recurringCost` left at 0. The sibling Azure path errors on exactly this input with the comment "Fail loud on an unrecognised payment option rather than silently billing it as all-upfront (owner policy: no silent fallbacks on money-affecting fields)" (providers/azure/services/compute/client.go:573). All four GCP clients share the bug: cloudsql/client.go:280, memorystore/client.go:291, cloudstorage/client.go:295. +- evidence: + ```go + case "monthly", "no-upfront": + upfrontCost = 0 + recurringCost = totalCost / (float64(termYears) * 12) + default: + upfrontCost = totalCost + } + ``` +- suggested fix: return `fmt.Errorf("unsupported payment option ...")` in the default branch of all four GCP clients, matching the Azure compute client. +- verdict: CONFIRMED — the default branch bills the whole commitment today in all four GCP clients (computeengine/client.go:875, cloudsql/client.go:279, memorystore/client.go:290, cloudstorage/client.go:294) while the Azure sibling returns an error at compute/client.go:574-577, and the empty payment option is a documented reachable state (migration 000032, reservations/purchase.go:103-108). +- issue: (pending cross-reference) + +### A08b-016 GCP pricing failures are logged and the recommendation is emitted with zero cost alongside non-zero savings +- category: silent-fallback +- severity: high +- location: providers/gcp/services/cloudsql/client.go:507 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd (same at `cloudstorage/client.go:499`, `memorystore/client.go:493`) +- failure scenario: when the catalog lookup fails for any reason — including the guaranteed pagination failure in A08b-015 — `fillSQLPricing` returns without touching the recommendation. `EstimatedSavings` has already been set from the Recommender payload at line 558, so the recommendation reaches the scorer with a real savings number, `CommitmentCost: 0`, `OnDemandCost: 0` and `SavingsPercentage: 0`. Any ranking that divides by cost, or any "savings per dollar committed" screen, sees a free commitment with positive savings and ranks it first. +- evidence: + ```go + pricing, err := c.getSQLPricing(ctx, rec.ResourceType, c.region, termYears) + if err != nil { + log.Printf("cloudsql: pricing unavailable for %s in %s (issue #1020): %v", rec.ResourceType, c.region, err) + return + } + ``` +- suggested fix: drop the recommendation (or propagate the error) when pricing cannot be established, rather than emitting a half-populated money row. +- verdict: CONFIRMED — `fillSQLPricing` logs and returns without writing any cost field (cloudsql:505-510), and `convertGCPRecommendation` has already set `EstimatedSavings` at line 558 before calling it at 566, so the emitted row carries real savings with zero cost; the same shape is at cloudstorage:497-502 and memorystore:491-496. +- issue: (pending cross-reference) + +### A08b-022 Four Azure clients report "no existing commitments" when the reservations pager cannot be built +- category: silent-fallback +- severity: high +- location: providers/azure/services/cache/client.go:185 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd (same at `cosmosdb/client.go:187`, `database/client.go:217`, `search/client.go:132`) +- failure scenario: a credential, permission or client-construction failure is logged to stdout and converted into an empty slice with a nil error. A caller comparing recommendations against existing commitments concludes the subscription holds no reservations and recommends (or executes) a purchase that duplicates an existing one. `synapse/client.go:168` and `managedredis/client.go:179` return the error on this same path, so the behaviour is inconsistent across the provider. +- evidence: + ```go + pager, err := c.createReservationsPager() + if err != nil { + log.Printf("WARNING: failed to create Redis reservations pager: %v", err) + return []common.Commitment{}, nil + } + ``` +- suggested fix: return the wrapped error, matching synapse and managedredis. +- verdict: CONFIRMED — cache:185-190, cosmosdb:187-192, database:217-222 and search:132-137 log and return `[]common.Commitment{}, nil`, while synapse:165-169 and managedredis:174-180 return the wrapped error on the same path. One narrowing: the discarded error comes only from `NewReservationsDetailsClient` construction (cache:202-205), so a permission failure surfaces later at `NextPage`, not here. +- issue: (pending cross-reference) + +### A01-014 executeApprovedExchange permanently fails an approved exchange on transient store errors and reports HTTP 200 +- category: silent-fallback +- severity: medium +- location: internal/api/handler_ri_exchange.go:2431 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: The approver clicks the email link; `TransitionRIExchangeStatus` moves pending->processing, then `GetRIExchangeDailySpend` (or `GetGlobalConfig`) returns a transient DB error. `failExchange` marks the record `failed` for good (the CAS cannot be re-approved) and returns `{"status":"failed"}` with a nil error, i.e. HTTP 200. If `FailRIExchange` itself fails, the row is stranded in `processing` and the caller still gets 200 "failed". On the session path `approveRIExchangeViaSession` then stamps `approved_by` because `execErr == nil` (2213-2216), attributing a failed exchange to the approver. +- evidence: + ```go + dailySpendStr, err := h.config.GetRIExchangeDailySpend(ctx, time.Now()) + if err != nil { + return h.failExchange(ctx, id, "daily spending cap check failed") + } + + globalCfg, err := h.config.GetGlobalConfig(ctx) + if err != nil { + return h.failExchange(ctx, id, "config load failed") + } + ``` +- suggested fix: For pre-commit store failures, revert the record to `pending` (CAS processing->pending) and return a 5xx so the approval can be retried; reserve `failExchange` for provider-side failures, and make it return an error when `FailRIExchange` fails. +- verdict: CONFIRMED — executeApprovedExchange (internal/api/handler_ri_exchange.go:2431-2439) routes GetRIExchangeDailySpend/GetGlobalConfig errors into failExchange:2350-2356, which logs a FailRIExchange failure and returns (map{"status":"failed"}, nil) either way; TransitionRIExchangeStatus (internal/config/store_postgres.go:2751-2755) requires status=pending so a failed row cannot be re-approved, and approveRIExchangeViaSession:2213-2216 stamps approved_by whenever execErr == nil. +- issue: (pending cross-reference) + +### A02-005 History merges a hard-capped, unfiltered page of executions, so older matching rows disappear once 100 non-clean executions exist +- category: silent-fallback +- severity: medium +- location: internal/api/handler_history.go:149 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: `fetchExecutionsAsHistory` reads `GetExecutionsByStatuses(..., config.DefaultListLimit)` = 100 rows ordered by `scheduled_date DESC` (store_postgres.go:1280-1283) across every terminal status (failed, expired, canceled, partially_completed) plus future-dated pending ramp steps. Provider/account/date filters and the allowed_accounts scope are applied in Go afterwards. With a few multi-step plans plus accumulated failed/expired rows the cap is exceeded, and a request for `account_ids=&start=2026-06-01` returns none of X's failed executions from June even though they exist; the stale-approval sweep (`expireStaleExecutions`) likewise never reaches pending rows beyond the first 100. No warning is emitted. Same shape as the #1140 truncation fixed for `/api/inventory`. +- evidence: + ```go + executions, err := h.config.GetExecutionsByStatuses(ctx, historyExecutionStatuses, config.DefaultListLimit) + ... + for _rvc := range executions { + exec := executions[_rvc] + ... + if !filters.matchesExecution(exec) { + continue + } + ``` +- suggested fix: Push the provider/account/date predicates (and ideally the scope's account set) into the store query and page by `filters.Limit`, or at minimum drop the fixed cap for the filtered path and query stale pending/notified rows separately for the sweep. +- verdict: CONFIRMED — fetchExecutionsAsHistory (handler_history.go:149) passes config.DefaultListLimit=100 (internal/config/constants.go:9) to GetExecutionsByStatuses, whose SQL applies `ORDER BY scheduled_date DESC LIMIT $2` before any filter (store_postgres.go:1280-1282); historyExecutionStatuses (handler_history.go:124) also includes "completed" rows that are then skipped in Go (line 166), so the page fills even faster than the finding states, and matchesExecution (line 801) plus the stale-sweep collection (line 172) only ever see that page. +- issue: (pending cross-reference) + +### A02-010 Coverage breakdown swallows a recommendations read error and reports 100% coverage +- category: silent-fallback +- severity: medium +- location: internal/api/handler_inventory.go:224 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: `ListRecommendations` fails (DB blip, or `account_id` is a raw external number that is not a UUID: `buildCoverageRecFilter` at lines 255-258 forwards it unvalidated into `AccountIDs`, which the store compares against a uuid column). `recs` becomes nil, `onDemandByKey` is empty, and `coveragePct(covered, 0)` returns 100 for every service with an active commitment. The page shows "100% covered" with HTTP 200 and nothing in the response marks the number as degraded. +- evidence: + ```go + recs, err := h.scheduler.ListRecommendations(ctx, buildCoverageRecFilter(params)) + if err != nil { + // Non-fatal: recommendations are best-effort for coverage display. + recs = nil + } + ``` +- suggested fix: Return the error (500) like `getDashboardSummary` does at line 48-50, and resolve `account_id` through `resolveSingleAccountFilterIDs` before passing UUIDs to the recommendations filter. +- verdict: CONFIRMED — getCoverageBreakdown (handler_inventory.go:224-229) sets recs=nil on a ListRecommendations error, aggregateOnDemandByKey then yields no entries, and coveragePct (handler_inventory.go:390-396) returns covered/(covered+0)*100 = 100 for every service with an active commitment while the handler still returns 200; buildCoverageRecFilter (lines 255-258) forwards params["account_id"] into AccountIDs unvalidated. +- issue: (pending cross-reference) + +### A03-009 Forgot-password is silently rate-limited for the whole invite window because the age is derived from the wrong expiry constant +- category: silent-fallback +- severity: medium +- location: internal/auth/service_password.go:288 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: `tokenAge := PasswordResetExpiry - time.Until(expiry)` assumes every stored token was issued with a 1h expiry. Invites are issued with `PasswordSetupExpiry` (7d), so for an invited user `tokenAge` is negative for 7d - 59m and the request returns nil without issuing a token or sending mail. `CreateUserResult` (service_user.go:150-157) and `sendInviteEmail` tell the admin to "re-mail the setup link via the Forgot Password flow" when the invite send failed; that recovery path does nothing for a week and reports success. Reproduced with an invite 1h old: `UpdateUser calls=0 email sends=0 err=`. +- evidence: + ```go + if user.PasswordResetExpiry != nil { + tokenAge := PasswordResetExpiry - time.Until(*user.PasswordResetExpiry) + if tokenAge < PasswordResetRateLimit { + logging.Debugf("Password reset rate-limited for %s (token age %s < %s)", + redactEmail(email), tokenAge.Round(time.Second), PasswordResetRateLimit) + return nil + } + } + ``` +- suggested fix: Store the issue time (or compute age from `UpdatedAt`), or pick the window by flow (`PasswordSetupExpiry` when `!user.Active`), so the one-minute limit is measured against the real issue time. +- verdict: CONFIRMED — CreateUser sets the invite expiry to now + PasswordSetupExpiry (7d, service_user.go:220-222, service.go:28) while RequestPasswordReset computes age as PasswordResetExpiry (1h, service.go:22) minus time.Until(expiry) at service_password.go:288, which is negative for the first 6d23h of an invite and therefore below PasswordResetRateLimit (1m, service_password.go:45), returning nil before the token write and the send; CreateUserResult's comment at service_user.go:154-156 still directs admins to that flow. +- issue: (pending cross-reference) + +### A03-011 Bastion mode silently falls back to the caller's ambient STS identity in two production callers +- category: silent-fallback +- severity: medium +- location: internal/credentials/resolver.go:222 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: When `opts.AccountLookup` or `opts.STSClientFactory` is nil, a `bastion` account is resolved as plain `role_arn` using the caller-supplied STS client, i.e. the CUDly host's own credentials, and the bastion account's stored credentials are never touched. The no-opts wrapper `ResolveAWSCredentialProvider` is what `internal/scheduler/scheduler.go:761` (recommendation collection) and `internal/api/handler_accounts.go:1528` (org discovery) call, so every bastion account on those paths assumes the target role from the wrong principal. If the customer's trust policy also trusts the CUDly role it "works" with the wrong identity and the wrong audit trail; if it trusts only the bastion, collection fails with an AccessDenied that points nowhere near the cause. +- evidence: + ```go + if opts.AccountLookup == nil || opts.STSClientFactory == nil { + // Legacy fallback: trust caller's stsClient. Tracked in known_issues/03. + return resolveRoleARNProvider(ctx, account, stsClient, nil) + } + ``` +- suggested fix: Return an error for `bastion` when the options are missing, and wire `AWSResolveOptions` into the scheduler and org-discovery callers (the purchase path already does). +- verdict: CONFIRMED — ResolveAWSCredentialProvider passes AWSResolveOptions{} (resolver.go:95-102), resolveBastionProvider:222-225 then assumes the target role with the caller's stsClient and never loads the bastion account, and the only two non-test callers of that wrapper are scheduler.go:761 (s.assumeRoleSTS) and handler_accounts.go:1528 (sts.NewFromConfig(baseCfg)), both host-credentialed; the WithOpts callers at server/app.go:825 and purchase/execution.go:449 set only AmbientProvider, so no production path supplies AccountLookup/STSClientFactory at all. +- issue: (pending cross-reference) + +### A06-008 AWS sender is constructed with no FROM_EMAIL, so password-reset / invite / welcome mail silently succeeds and is never sent +- category: silent-fallback +- severity: medium +- location: internal/email/factory.go:76 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: the AWS branch validates nothing (the SMTP constructors, by contrast, reject an empty `FromEmail` at internal/email/smtp_sender.go:73). With `FROM_EMAIL` unset and `SNS_TOPIC_ARN` unset, a user clicks "forgot password": `SendPasswordResetEmail` → `sendMultipartVia` → `SendToEmailWithCCMultipart` → `SendToEmailWithCC`, which logs at debug level and returns `nil` (internal/email/sender.go:306). The API reports success, the user is told to check their inbox, and nothing was ever sent. The same silent nil covers welcome, invite and scheduled-purchase notifications; only `SendPurchaseApprovalRequest` and `SendPurchaseExecutedNotification` check `isValidFromEmail` and return `ErrNoFromEmail`. +- evidence: + ```go + case ProviderAWS: + return NewSender(SenderConfig{ + TopicARN: os.Getenv("SNS_TOPIC_ARN"), + FromEmail: os.Getenv("FROM_EMAIL"), + EmailAddress: os.Getenv("EMAIL_ADDRESS"), + }) + ``` +- suggested fix: apply the existing `isValidFromEmail` guard in `SendToEmailWithCC`/`SendToEmailWithCCMultipart` and return `ErrNoFromEmail` instead of `nil`. +- verdict: CONFIRMED — the AWS branch validates nothing (internal/email/factory.go:75-81) while `NewSMTPSender` rejects an empty `FromEmail` (internal/email/smtp_sender.go:73-75); `SendPasswordResetEmail` → `sendMultipartVia` → `SendToEmailWithCCMultipart` (internal/email/templates.go:481-486,780-796) and both SES entry points return a bare `nil` on `s.fromEmail == ""` (internal/email/sender.go:271-274 and 306-309), whereas only `SendPurchaseApprovalRequest` and `SendPurchaseExecutedNotification` apply `isValidFromEmail` (templates.go:819-820,1070-1071). +- issue: (pending cross-reference) + +### A06-014 SMTP transport redirects token-bearing approval mail to the static notify inbox where SES returns ErrNoRecipient +- category: silent-fallback +- severity: medium +- location: internal/email/smtp_sender.go:527 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: `(*Sender).SendPurchaseApprovalRequest` returns `ErrNoRecipient` when `data.RecipientEmail == ""`, and internal/api/handler_purchases.go:2889 branches on that sentinel to tell the operator which side is unconfigured. The SMTP implementations of `SendPurchaseApprovalRequest` (:527), `SendPurchaseExecutedNotification` (:634) and `SendRIExchangePendingApproval` (:485) instead fall back to `s.notifyEmail`, so on every Azure/GCP deployment an unresolved recipient is masked: the approval link and its one-time token are delivered to a shared operations inbox, and the handler's precise error branch is unreachable. The two transports therefore behave differently on the same money path. +- evidence: + ```go + recipient := data.RecipientEmail + if recipient == "" { + recipient = s.notifyEmail + } + if recipient == "" { + return ErrNoRecipient + } + ``` +- suggested fix: return `ErrNoRecipient` when `data.RecipientEmail` is empty on the token-bearing methods, matching the SES path; keep the `notifyEmail` fallback only for the tokenless broadcast-style sends. +- verdict: CONFIRMED — the three SMTP methods fall back to `s.notifyEmail` (internal/email/smtp_sender.go:484-490, 526-533, 633-640) while the SES equivalents return `ErrNoRecipient` on the empty field before anything else (internal/email/templates.go:810-812, 1066-1068), and internal/api/handler_purchases.go:2886-2893 branches on that sentinel. Stronger than stated: `notifyEmail` defaults to `cfg.FromEmail` when unset (smtp_sender.go:85-88) and `FromEmail` is mandatory (smtp_sender.go:73-75), so on GCP/Azure the fallback always resolves and `ErrNoRecipient` is unreachable from these methods. +- issue: (pending cross-reference) + +### A06-015 Analytics pipeline keeps writing after partition creation fails, into a partition retention can never drop +- category: silent-fallback +- severity: medium +- location: internal/server/analytics_collect.go:142 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: step 1 exists specifically so the snapshot "lands in a real monthly partition rather than the catch-all default (M3)". When `CreateFuturePartitions` fails (lock timeout, the 5-minute DDL deadline, missing privilege) the code logs a warning, marks the result "partial", and proceeds to `Collect`, which COPYs the batch into the default partition. `DropOldPartitions` only detaches month partitions, so those rows are never reclaimed and the default partition grows forever; every later month with the same failure adds more. The scheduled task still returns HTTP 200 with `status: partial`, which no alert consumes. +- evidence: + ```go + if err := withDDLTimeout(ctx, app.Analytics.CreateFuturePartitions, cfg.PartitionsAhead); err != nil { + log.Printf("Warning: failed to ensure future partitions: %v", err) + result["status"] = "partial" + } else { + result["partitions_ensured"] = true + } + ``` +- suggested fix: return the error and skip the collect step when partition creation fails, since the write is the thing the partition guarantees. +- verdict: CONFIRMED — the partition step only logs and marks `partial` before falling through to `Collect` (internal/server/analytics_collect.go:140-159), and retention explicitly excludes the catch-all: `drop_old_savings_partitions` filters `tablename != 'savings_snapshots_default'` (internal/database/postgres/migrations/000067_analytics_snapshot_correctness.up.sql:207-212), so rows that land there are never reclaimed. The task still returns `(result, nil)` and the HTTP wrapper reports `"status":"success"` around it (internal/server/http.go:257-265). +- issue: (pending cross-reference) + +### A06-016 DEFAULT_TERM and DEFAULT_COVERAGE silently fall back to built-in defaults on an unparseable value +- category: silent-fallback +- severity: medium +- location: internal/server/app.go:947 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: `LoadApplicationConfig` reads `DEFAULT_TERM` and `DEFAULT_COVERAGE` through these helpers and both feed the purchase manager as system-wide money defaults (`purchase.ManagerConfig.DefaultTerm` / `DefaultCoverage`, app.go:441-443). A typo such as `DEFAULT_TERM=3y` logs one warning line and silently commits every subsequent purchase at 3 years instead of the intended value. The two sibling knobs on the same struct are validated and fail startup (`validateAppConfigEnvDefaults` at app.go:288, added for issue #1026), and the analytics knobs deliberately use a non-defaulting parser for the same reason (`loadAnalyticsInt`, analytics_collect.go:60), so the term/coverage pair are the outliers. +- evidence: + ```go + result, err := strconv.Atoi(val) + if err != nil { + log.Printf("WARNING: %s=%q is not a valid integer; using default %d", key, val, defaultVal) + return defaultVal + } + ``` +- suggested fix: extend `validateAppConfigEnvDefaults` to reject a set-but-unparseable `DEFAULT_TERM` / `DEFAULT_COVERAGE` at startup, using the `loadAnalyticsInt` sentinel pattern. +- verdict: CONFIRMED — `getEnvInt` and `getEnvFloat` both log a warning and return the built-in default (internal/server/app.go:945-955, 962-972); they are the readers for `DEFAULT_TERM` (default 3) and `DEFAULT_COVERAGE` (default 80) at app.go:263,265, both of which flow into `purchase.ManagerConfig` at app.go:441-443 and again at app.go:790-792. `validateAppConfigEnvDefaults` covers only `DEFAULT_PAYMENT_OPTION` and `DEFAULT_RAMP_SCHEDULE` (app.go:288-297), and `loadAnalyticsInt` uses the fail-fast sentinel instead (internal/server/analytics_collect.go:60-70), so the term/coverage pair are indeed the outliers. +- issue: (pending cross-reference) + +### A06-028 Azure federated-credential creation swallows every error and still reports success +- category: silent-fallback +- severity: medium +- location: internal/iacfiles/templates/azure-wif-deploy.sh.tmpl:79 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: `|| echo "(federated credential may already exist — continuing)"` catches every failure of `az ad app federated-credential create`, not just the name conflict it names. If a credential named `cudly` already exists on the app registration bound to a *different* issuer, subject or audience — an earlier deployment pointing at a different CUDly instance, or a hand-edited one — the create fails, the script prints "continuing", proceeds to assign the Reservation Purchaser role at subscription scope (step 4), and prints "=== Done ===". The operator is told federation is configured while the credential actually in place trusts a different issuer, so either CUDly cannot authenticate or a stale issuer can. The GCP script goes to considerable length to avoid exactly this (gcp-wif-cli.sh.tmpl:177-233 re-reads an existing provider and aborts unless issuer, attribute condition, mapping, state and audience all match). The same swallow is in azure-wif-cli.sh.tmpl:60. +- evidence: + ```sh + az ad app federated-credential create \ + --id "${APP_ID}" \ + --parameters "${FEDCRED_PARAMS}" \ + --output none 2>&1 || echo " (federated credential may already exist — continuing)" + ``` +- suggested fix: on failure, `az ad app federated-credential show --federated-credential-id cudly` and compare issuer/subject/audiences against the intended values; abort when they differ instead of continuing to the role assignment. +- verdict: CONFIRMED — the `|| echo` swallows every non-zero exit and defeats the script's own `set -euo pipefail` (internal/iacfiles/templates/azure-wif-deploy.sh.tmpl:14, 77-81), after which step 4 assigns the Reservation Purchaser role and the script prints `=== Done ===` (azure-wif-deploy.sh.tmpl:83-97); the same swallow is at azure-wif-cli.sh.tmpl:58-61, and the GCP path does the compare-and-abort the finding cites (gcp-wif-cli.sh.tmpl). One correction: each run creates a fresh app registration (`az ad app create`, azure-wif-deploy.sh.tmpl:49), so the "credential bound to a different issuer" variant is hard to reach; a Graph permission denial or a throttle produces the same false success with no credential in place at all. +- issue: (pending cross-reference) + +### A07-005 Savings Plan start/end dates are dropped when unparseable +- category: silent-fallback +- severity: medium +- location: providers/aws/services/savingsplans/client.go:195 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: a `Start`/`End` string that does not parse as RFC3339 leaves `commitment.StartDate` / `EndDate` at the zero time with no log line. `recfilter.isRecentActiveCommitment` gates on `c.StartDate.After(cutoff)`, so a zero `StartDate` makes an active SP invisible to the duplicate check and the next run re-recommends it. `ladder/adapters.go:195` fails loud on exactly this input for the same field, so the two readers of the same API disagree. +- evidence: + ```go + if sp.Start != nil { + if startTime, err := time.Parse(time.RFC3339, *sp.Start); err == nil { + commitment.StartDate = startTime + } + } + ``` +- suggested fix: return an error (or at minimum log) on a parse failure, matching `parseSPDate` in providers/aws/ladder/adapters.go. +- verdict: CONFIRMED — providers/aws/services/savingsplans/client.go:195-204 drops both parse failures with no log; pkg/recfilter/dedupe.go:83 gates on `c.StartDate.After(cutoffTime)` so a zero StartDate makes the SP invisible to the dedupe map built at :105, while providers/aws/ladder/adapters.go:195-204 errors on the identical field. +- issue: (pending cross-reference) + +### A07-008 MemoryDB recommendations hardcode the engine as "redis" +- category: silent-fallback +- severity: medium +- location: providers/aws/recommendations/parser_services.go:255 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: MemoryDB supports Valkey as well as Redis. Every MemoryDB recommendation is stamped `Engine: "redis"` regardless of what the cluster runs, and the same function errors loudly when `NodeType` is absent rather than guessing — the engine gets the opposite treatment. The fabricated value flows into `EngineFromDetails` and the dedupe key, and into any consumer that displays or filters on engine. +- evidence: + ```go + rec.Details = &common.CacheDetails{ + Engine: "redis", + NodeType: *mdbDetails.NodeType, + } + ``` +- suggested fix: read the engine from `MemoryDBInstanceDetails` if CE exposes it, otherwise leave it empty so downstream code can tell "unknown" from "redis". +- verdict: CONFIRMED — providers/aws/recommendations/parser_services.go:255 stamps `"redis"` unconditionally while :248-250 errors on a missing NodeType, and costexplorer@v1.63.1/types/types.go:1492-1509 shows `MemoryDBInstanceDetails` carries no engine field at all, so the value is fabricated rather than read; only the "leave it empty" half of the suggested fix is available. +- issue: (pending cross-reference) + +### A07-009 Reserved-node term is silently assumed to be 12 months for any non-3yr duration +- category: silent-fallback +- severity: medium +- location: providers/aws/services/rds/client.go:88 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: `instance.Duration` is a `*int32`; a nil pointer (or any value that is not exactly 94608000) yields `termMonths = 12`. `EndDate` is then `StartTime + 12 months`. A 3-year RI whose `Duration` came back nil is reported as expiring two years early, which makes `AdjustExistingCoverageForExpiringCommitments` treat its pool as uncovered and recommend a replacement purchase that is not needed. ElastiCache has the identical construct at elasticache/client.go:86, and MemoryDB/OpenSearch/Redshift use `getTermMonthsFromDuration`, which likewise returns 12 for a zero duration. +- evidence: + ```go + duration := aws.ToInt32(instance.Duration) + termMonths := 12 + if duration == ThreeYearSeconds { + termMonths = 36 + } + ``` +- suggested fix: treat an absent or unrecognised duration as unknown — leave `EndDate` zero (which `expiringCountsByPool` already skips) rather than fabricating a one-year term. +- verdict: CONFIRMED — providers/aws/services/rds/client.go:87-91 and providers/aws/services/elasticache/client.go:86-89 map every duration other than `ThreeYearSeconds` (nil included, via `aws.ToInt32`) to 12 months and set `EndDate = StartTime + termMonths`; `getTermMonthsFromDuration` does the same for a zero duration at memorydb/client.go:498, opensearch/client.go:617 and redshift/client.go:684, and providers/aws/recommendations/expiry.go:52 does skip a zero `EndDate` as the fix assumes. +- issue: (pending cross-reference) + +### A07-022 Exchange target lookup swallows an invalid payment option and silently widens the query +- category: silent-fallback +- severity: medium +- location: providers/aws/services/ec2/client.go:874 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: `normalizeTargetOfferingsParams` calls `convertEC2PaymentOption` and discards the error with `if ... err == nil`. A caller passing an unrecognised payment option (for example the display form `"All Upfront"` rather than `"all-upfront"`) leaves `offeringType` at its zero value, which the SDK omits, so AWS returns offerings for *every* payment option. `ListTargetOfferings` then presents partial-upfront and all-upfront exchange targets for a request that asked for one specific option, and the caller has no signal that its input was rejected. +- evidence: + ```go + if p.OfferingType != "" { + if ot, err := convertEC2PaymentOption(p.OfferingType); err == nil { + offeringType = ot + } + } + return + ``` +- suggested fix: return the error from `normalizeTargetOfferingsParams` and propagate it out of `ListTargetOfferings`; keep the "all options" behaviour only for a genuinely empty input. +- verdict: CONFIRMED — providers/aws/services/ec2/client.go:874-878 keeps `offeringType` at its zero value when `convertEC2PaymentOption` errors and returns no error to the caller, and the path is live from internal/api/handler_ri_exchange.go:154. +- issue: (pending cross-reference) + +### A07-026 Mid-pagination AccessDenied returns a partial account list as if it were complete +- category: silent-fallback +- severity: medium +- location: providers/aws/provider.go:292 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: the comment above `orgListAccountsSilentErrorCodes` states that returning a silently truncated list is unsafe for the purchase flow, but the silent-code check is not gated on how many pages have already succeeded. If page one of `ListAccounts` succeeds and page two fails with `AccessDeniedException` (an SCP change, a permission boundary applied mid-run), `appendOrgAccounts` returns the accounts collected so far with a nil error and the caller cannot distinguish it from a complete organization listing. +- evidence: + ```go + var apiErr smithy.APIError + if errors.As(err, &apiErr) { + if _, silent := orgListAccountsSilentErrorCodes[apiErr.ErrorCode()]; silent { + return accounts, nil + } + } + ``` +- suggested fix: treat the silent codes as expected only on the first page (no member accounts collected yet); once pagination has produced results, any error including AccessDenied is a real mid-run failure and must propagate. +- verdict: CONFIRMED — providers/aws/provider.go:287-293 tests only the error code and returns `accounts, nil` regardless of how many pages already appended at :298-308, directly contradicting the "returning a silently-truncated list is unsafe for the purchase flow" rationale stated at :255-261; the non-silent branch immediately below (:295-297) shows the intended propagation. +- issue: (pending cross-reference) + +### A08-014 pricing.FetchAll silently truncates when the page cap is hit +- category: silent-fallback +- severity: medium +- location: providers/azure/internal/pricing/retail_prices.go:72 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: when a query needs more than `maxPages` pages the loop exits with a non-empty `nextURL` and returns `(all, nil)`. The caller (`getVMPricing`, compute/client.go:700) cannot distinguish "all prices for this SKU" from "the first 5000 of them", so a reservation or on-demand meter on a later page is read as missing and the run either errors with a wrong message or, where the meter type differs, prices against a partial set. The self-referential-link case in the same loop returns an explicit error; the cap does not. +- evidence: + ```go + for pageIdx := 0; pageIdx < maxPages && nextURL != ""; pageIdx++ { + ... + nextURL = page.NextPageLink + } + return all, nil + ``` +- suggested fix: after the loop, return an error when `nextURL != ""`, naming the cap — same treatment the self-referential-link guard already gets. +- verdict: CONFIRMED — the loop at internal/pricing/retail_prices.go:72-86 exits on `pageIdx == maxPages` with `nextURL` still set and returns `(all, nil)`, while the self-referential-link branch nine lines above returns an error; `getVMPricing` (compute/client.go:700-717) then reads a missing meter as absent rather than truncated. +- issue: (pending cross-reference) + +### A08-015 Subscriptions with a nil DisplayName are silently dropped from the org-wide account list +- category: silent-fallback +- severity: medium +- location: providers/azure/accounts_cache.go:219 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: `fetchAccounts` skips any `armsubscriptions` entry whose `DisplayName` is nil, with no log and no error. The result feeds `NewMultiSubscriptionRecommendationsClient` (provider.go:681), which fans out only over the accounts it was given, so the dropped subscription is never queried and never counted in `PartialSubscriptionFailureError.Attempted` — the sweep reports complete while missing it. It also feeds `accountsContain`, so `validateConfiguredSubscription` rejects an explicitly configured subscription that exists but has no display name. A missing display name has nothing to do with whether the subscription is usable. +- evidence: + ```go + for _, sub := range page.Value { + if sub.SubscriptionID == nil || sub.DisplayName == nil { + continue + } + accounts = append(accounts, common.Account{ ... Name: *sub.DisplayName, ...}) + ``` +- suggested fix: only skip when `SubscriptionID` is nil; fall back to the subscription ID for `Name`/`DisplayName` and log the substitution. +- verdict: PLAUSIBLE — `fetchAccounts` (accounts_cache.go:230-232) does drop a nil-DisplayName subscription with no log, and the downstream claims hold (the list feeds the fan-out at provider.go:679 and `accountsContain` at accounts_cache.go:288 via validateConfiguredSubscription, provider.go:415), but nothing in the tree establishes that ARM ever returns a subscription whose `DisplayName` is nil. +- issue: (pending cross-reference) + +### A08-020 GCP treats any 403-shaped error as a permission gap and returns an empty, nil-error sweep +- category: silent-fallback +- severity: medium +- location: providers/gcp/recommendations.go:126 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: when `getRegions` fails and `isPermissionError` is true, `GetRecommendations` returns `([]common.Recommendation{}, nil)`. That is exactly the outcome `mergeRegionResults` in the same file exists to prevent: "Returning (recs, nil) on a total failure makes a broken run indistinguishable from 'no savings available': the scheduler would count the account as succeeded, evict its previously collected rows, and clear last_collection_error (COR-03)". The classifier is also loose — its string fallback (recommendations.go:412) matches any error message containing both "403" and "permission", so an unrelated failure can take the silent-empty branch. +- evidence: + ```go + if isPermissionError(err) { + logging.Warnf("GCP account %s: skipping recommendations — insufficient Compute permission to list regions (grant roles/compute.viewer): %v", r.projectID, err) + return []common.Recommendation{}, nil + } + ``` +- suggested fix: return a typed permission error that the scheduler can log at WARN without treating the sweep as a successful zero-result collection, so previously collected rows are not evicted. +- verdict: CONFIRMED — `GetRecommendations` (recommendations.go:125-128) returns `([]common.Recommendation{}, nil)` on a permission-classified `getRegions` failure, exactly the outcome `mergeRegionResults` documents as unsafe for COR-03 (recommendations.go:178-189, guard at :207), and `isPermissionError` (recommendations.go:411-412) falls back to a "403"+"permission" substring test on any error string. +- issue: (pending cross-reference) + +### A08b-021 `pricing.FetchAll` truncates at the page cap and returns partial results with a nil error +- category: silent-fallback +- severity: medium +- location: providers/azure/internal/pricing/retail_prices.go:72 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: the loop exits when `pageIdx == maxPages` even though `nextURL` is still non-empty, and returns `(all, nil)`. Every Azure client calls it with `DefaultMaxPages = 50` and, for cosmosdb and search, a filter that is region-wide rather than SKU-scoped (A08b-008, A08b-009), so 5000 items is reachable. The caller cannot distinguish "this SKU has no reservation price" from "the reservation price was on page 51", and reports the former. +- evidence: + ```go + for pageIdx := 0; pageIdx < maxPages && nextURL != ""; pageIdx++ { + ... + nextURL = page.NextPageLink + } + return all, nil + ``` +- suggested fix: return an explicit error when the loop exits with `nextURL != ""`, so a truncated price set can never be read as a complete one. +- verdict: CONFIRMED — the loop condition at providers/azure/internal/pricing/retail_prices.go:72 exits on `pageIdx == maxPages` regardless of `nextURL`, and line 87 returns `all, nil` with no post-loop check; every Azure client passes `pricing.DefaultMaxPages` (50, declared at line 49). +- issue: (pending cross-reference) + +### A08b-023 `GetValidResourceTypes` falls back to a hardcoded SKU list when the live listing fails, including on context cancellation +- category: silent-fallback +- severity: medium +- location: providers/azure/services/managedredis/client.go:374 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd (same shape at `cache/client.go:417`, `cosmosdb/client.go:418`, `search/client.go:368`) +- failure scenario: `collectSKUsFromPager` correctly returns the pager error, and the caller discards it — including `context.Canceled` and `context.DeadlineExceeded` — and substitutes a hand-maintained list of Redis SKU names. `ValidateOffering` then approves a SKU that may not exist in this region (the curated list is region-blind), or rejects a valid new SKU that Azure has added since the list was written. The cache and cosmosdb variants additionally `break` out of the page loop on error at `cache/client.go:470` and `cosmosdb/client.go:470`, silently accepting a partial SKU set. +- evidence: + ```go + skuSet, err := collectSKUsFromPager(ctx, pager) + if err != nil { + // Discard any partial results and fall back to the curated SKU list + // rather than risk false validation failures for valid SKUs. + return c.commonSKUs(), nil + } + ``` +- suggested fix: propagate the error (and always propagate context errors), reserving the curated list for an explicitly-flagged offline mode. +- verdict: CONFIRMED at the cited location, with a correction to the "same shape" claim — managedredis:380-385 does discard `collectSKUsFromPager`'s error, context errors included, but cache:424-427, cosmosdb:424-427 and search:374-377 all propagate it; what those three actually share is the curated fallback when the pager cannot be constructed (cache:419-421, cosmosdb:420-421, search:370-371) plus the silent `break` on a page error (cache:467-471, cosmosdb:467-471). +- issue: (pending cross-reference) + +### A08b-028 `walkManagedInstances` derives a "dominant" AZ configuration from counts collected before an error +- category: silent-fallback +- severity: medium +- location: providers/gcp/../azure/services/database/client.go:829 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: on a page error or context cancellation the walk returns the counts accumulated so far. `fetchServerInfo` at line 794 then compares `zoneRedundantCount == total` on that truncated sample: if page one held two zone-redundant instances and page two (which failed) held fifty non-redundant ones, the client reports `AZConfig: "zoneRedundant"` for the whole subscription. The partial-result signal is indistinguishable from a complete one. +- evidence: + ```go + page, err := pager.NextPage(ctx) + if err != nil { + if ctx.Err() != nil { + return zoneRedundant, nonZoneRedundant, total + } + logging.Warnf("azure database: managed instances page fetch failed: %v; AZConfig/Deployment signal unavailable", err) + return zoneRedundant, nonZoneRedundant, total + } + ``` +- suggested fix: return a completeness flag (or zeroed counts) on any error so the caller emits an empty `AZConfig` rather than a conclusion drawn from a partial walk. +- verdict: CONFIRMED — both error branches at database:829-833 return the counts accumulated so far, and `fetchServerInfo` at 790-800 derives `AZConfig` from `zoneRedundantCount == total` on that sample with no completeness signal; the pager-construction branch at 822-824 does return zeroes, which is the only path that degrades safely. +- issue: (pending cross-reference) + +### A08b-034 `termPlan` errors on an unknown term while `termYearsFromTerm` silently answers "one year" +- category: silent-fallback +- severity: medium +- location: providers/gcp/services/computeengine/client.go:200 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd (consumers at `client.go:1139` and `client.go:1157`) +- failure scenario: the two functions parse the same `rec.Term` string with opposite policies. `termPlan` refuses `"P3Y"` with an explicit error precisely because "a silent mis-default can purchase the wrong term and waste money"; `termYearsFromTerm`, fed the same value, returns 1 and is used to compute the savings and cost fields shown to the user. The recommendation therefore displays one-year economics for a three-year term, right up until the purchase fails at `termPlan`. +- evidence: + ```go + func termYearsFromTerm(term string) int { + switch strings.ToLower(strings.TrimSpace(term)) { + case "3yr", "3", "36mo": + return 3 + default: + return 1 + } + } + ``` +- suggested fix: have `termYearsFromTerm` return `(int, error)` sharing `termPlan`'s switch, so an unrecognized term fails in one place. +- verdict: CONFIRMED — `termPlan` errors on an unrecognized term with that exact money rationale in its doc (computeengine:46-56) while `termYearsFromTerm` silently returns 1 (200-207); both read `rec.Term`, the lenient one at 1157 for the pricing lookup and 1139 for `RecurringMonthlyCost`, the strict one at 568 inside `GroupCommitments` and at 773's sibling on the purchase path. +- issue: (pending cross-reference) + +### A09-009 AnalyzeReshapingWithRecs discards the lookup error without logging it +- category: silent-fallback +- severity: medium +- location: pkg/exchange/reshape.go:550 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: a database outage or a permission regression in the `PurchaseRecLookup` closure makes every call return an error. The reshape page renders with zero cross-family alternatives on every request, indistinguishable from "AWS has not recommended anything for this region", and `err` is dropped on the floor rather than logged. An operator sees a silently degraded page with no signal anywhere that the lookup is broken. +- evidence: + ```go + offerings, err := lookup(ctx, region, currencyCode) + if err != nil || len(offerings) == 0 { + // Fall through to base recs — losing alternatives is strictly + // less bad than losing the whole reshape page. + return recs + } + ``` +- suggested fix: keep the fall-through but split the branches and emit `logging.Warnf("reshape alternatives lookup failed: %v", err)` on the error arm so a persistent failure is visible. +- verdict: CONFIRMED — pkg/exchange/reshape.go:550-555 collapses `err != nil` and `len(offerings) == 0` into one branch that returns `recs`; `err` is never logged or wrapped anywhere in `AnalyzeReshapingWithRecs`, so a persistently failing lookup is indistinguishable from an empty region. +- issue: (pending cross-reference) + +### A09-019 An unrecognized term string is recorded as 0 months in the purchase audit log +- category: silent-fallback +- severity: medium +- location: pkg/common/audit.go:84 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: `Recommendation.Term` is an untyped string and `termMonths` matches two bare literals. A provider that emits `"1yr "` (trailing space), `"P1Y"`, `"12mo"` or `"5yr"` produces `AuditRecord.Term = 0` on a record that says a real commitment was purchased. The JSONL audit log is the artifact reconciled against `purchase_history`, so the reconciliation sees a zero-month commitment and the operator has no way to recover the real term from the record. The warning goes to `log.Printf`, not to the audit sink. +- evidence: + ```go + func termMonths(t string) int { + switch t { + case "1yr": + return 12 + case "3yr": + return 36 + default: + if t != "" { + log.Printf("warn: unrecognized term string %q, using 0 months", t) + } + return 0 + ``` +- suggested fix: reuse `ladder.ParseTerm`-style validation (or make `termMonths` return an error) so `NewAuditRecord` refuses to write a record whose term could not be resolved. +- verdict: CONFIRMED — `termMonths` matches only the two literals and returns 0 for everything else with a `log.Printf` warning (pkg/common/audit.go:84-99), and `NewAuditRecord` writes that 0 into `AuditRecord.Term` unconditionally (audit.go:68). A non-canonical term reaches it in practice: `normaliseTerm` in providers/azure/internal/recommendations/converter.go:350-352 explicitly passes any Azure term other than P1Y/P3Y through verbatim, and that value lands on `Recommendation.Term` at converter.go:148 and :202. The single production caller is cmd/multi_service.go:401. +- issue: (pending cross-reference) + +### A09-026 Provider factory failures are swallowed, so credential detection reports "no credentials found" +- category: silent-fallback +- severity: medium +- location: pkg/provider/registry.go:110 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: `GetAllProviders` logs a factory error through the stdlib `log` package (not `pkg/logging`) and drops the provider from the slice. `DetectAvailableProviders` iterates only what it got back, so a provider whose factory failed never contributes to its `errors` slice. When all three factories fail, `available` and `errors` are both empty and the caller receives "no cloud credentials found. Please configure AWS, Azure, or GCP credentials" — pointing the operator at credentials when the real cause was, say, the GCP factory's `Projects.List()` call failing on a permission error. +- evidence: + ```go + for name, factory := range factories { + provider, err := factory(&ProviderConfig{Name: name}) + if err != nil { + log.Printf("provider %q factory error: %v", name, err) + continue + } + ``` +- suggested fix: give `GetAllProviders` a second return value carrying the per-provider factory errors and have `DetectAvailableProviders` fold them into the error it reports, so a construction failure is never reported as absent credentials. +- verdict: CONFIRMED — `GetAllProviders` logs the factory error through the stdlib `log` package and `continue`s, returning only successfully constructed providers (pkg/provider/registry.go:109-116). `DetectAvailableProviders` iterates that slice alone and builds its `errors` slice only from `ValidateCredentials` failures (pkg/provider/credentials.go:26-41), so with all factories failing both `available` and `errors` are empty and line 49 returns "no cloud credentials found. Please configure AWS, Azure, or GCP credentials". +- issue: (pending cross-reference) + +### A10-007 CSV `Service`, `Term` and `PaymentOption` are cast into typed fields with no validation +- category: silent-fallback +- severity: medium +- location: cmd/multi_service_csv.go:106 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: `parseCSVRecord` casts an arbitrary cell to `common.ServiceType` and copies `Term` / `PaymentOption` verbatim. A hand-edited CSV with `Service=rds ` (trailing space) or `elasticsearch` produces a ServiceType that `createServiceClient` does not recognise, so the row is dropped at cmd/multi_service.go:538 with "Service client not yet implemented" rather than an error naming the bad cell; the operator sees a short run and no failure. Worse, `PaymentOption=All Upfront` (instead of `all-upfront`) is never checked against the same `validPaymentOptions` map `validatePaymentAndTerm` enforces for the flag, and is handed straight to `PurchaseCommitment`. External input on a money path is not parsed into the enum at the boundary. +- evidence: + ```go + rec.Service = common.ServiceType(getCSVField(record, colIdx, "Service")) + ... + rec.Term = getCSVField(record, colIdx, "Term") + rec.PaymentOption = getCSVField(record, colIdx, "PaymentOption") + ``` +- suggested fix: Parse each of the three through a `parseServiceType` / `parsePaymentOption` / `parseTerm` helper that errors on an unrecognised value, reusing the `serviceMap` in cmd/main.go:190 and the `validPaymentOptions` map in cmd/validators.go:131. +- verdict: CONFIRMED on the Service cast — cmd/multi_service_csv.go:106 bypasses the serviceMap at cmd/main.go:190, so even a valid flag alias like `elasticsearch` becomes a ServiceType that createServiceClient's switch cannot match (cmd/main.go:244-268) and the row is dropped with the "not yet implemented" message at cmd/multi_service.go:538-541. The PaymentOption half is real at the boundary but does not reach a purchase: every provider converts and errors on an unknown value first (providers/aws/services/rds/client.go:550-560, plus the matching converters in ec2, elasticache, memorydb and savingsplans). +- issue: (pending cross-reference) + +### A10-009 A short CSV write is reported as success; a zero-result run claims a file that was never created +- category: silent-fallback +- severity: medium +- location: cmd/multi_service_csv.go:193 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: `csv.Writer` buffers; `writer.Flush()` runs in a defer and its error is never read via `writer.Error()`. On a full disk or a failing NFS mount the header and rows are lost, `writeMultiServiceCSVReport` returns nil, and the caller prints "📋 CSV report written to: ri-helper-purchase-….csv" over a truncated or empty file that is the only record of what a `--purchase` run bought. Separately, `len(results) == 0` returns nil before `os.Create`, so a run with nothing to purchase prints the same "written to" line for a file that does not exist. +- evidence: + ```go + writer := csv.NewWriter(file) + defer writer.Flush() // error never inspected + ... + if len(results) == 0 { + return nil // no file created, caller still prints "written to" + } + ``` +- suggested fix: Call `writer.Flush()` explicitly before returning and return `writer.Error()`; return a sentinel (or have the caller check `len(allResults)`) so the "written to" line is only printed when a file was written. +- verdict: CONFIRMED — cmd/multi_service_csv.go:194-196 returns nil before os.Create and :208-209 defers Flush without ever consulting writer.Error(), while both callers print "CSV report written to" on any nil return (cmd/multi_service.go:175-179 and :579-583). +- issue: (pending cross-reference) + +### A10-012 An unknown `--services` value is warned about and skipped instead of rejected +- category: silent-fallback +- severity: medium +- location: cmd/main.go:219 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: `cudly --services rds,elasticahe --purchase` (typo). `parseServices` logs one `log.Printf` warning to stderr, drops ElastiCache, and the run proceeds against RDS alone. Amid the wizard's heavy emoji-decorated stdout the warning is easy to miss, and the operator believes both services were covered. Only a run where *every* name is bad reaches the `log.Fatalf("No valid services specified")` guard. External input is not parsed into the enum at the boundary with an error on unknown. +- evidence: + ```go + if service, ok := serviceMap[key]; ok { + add(service) + } else { + log.Printf("Warning: Unknown service '%s', skipping", name) + } + ``` +- suggested fix: Return an error from `parseServices` naming the unrecognised value and the valid set, and call it from `validateFlags` so the run never starts. +- verdict: CONFIRMED — cmd/main.go:217-220 adds the match or logs a warning and drops the name, and only a fully empty result reaches log.Fatalf("No valid services specified") at cmd/multi_service.go:94-96. +- issue: (pending cross-reference) + +### A10-013 Engine-version query failures silently disable the extended-support exclusion +- category: silent-fallback +- severity: medium +- location: cmd/multi_service_helpers.go:322 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: The default is to exclude instances on extended-support engine versions. If `queryRunningInstanceEngineVersions` or `queryMajorEngineVersions` fails, both helpers return an empty map and the run continues; `isInExtendedSupport` then returns false for every version ("If we don't have info, assume not in extended support", cmd/multi_service_engine_versions.go:395) and the filter becomes a no-op, so the run buys 3-year RIs for instances the operator wanted excluded. The failure need not be visible: `queryRDSInstancesInRegions` logs per-region errors and still returns `(map, nil)` (line 128), and `queryMajorEngineVersionsWithClient` logs per-engine errors and returns `(map, nil)` (line 226), so a total failure across all regions and all four engines is reported to the operator as "✅ Found 0 instance types". +- evidence: + ```go + instanceVersions, err := queryRunningInstanceEngineVersions(ctx, cfg) + if err != nil { + AppLogger.Printf("⚠️ Warning: Failed to query running instances ...: %v\n", err) + AppLogger.Printf(" Continuing without engine version filtering\n") + return make(map[string][]InstanceEngineVersion) + } + ``` +- suggested fix: Have both query helpers return an error when every region or every engine failed, and abort a `--purchase` run (not a dry run) when `!cfg.IncludeExtendedSupport` and the signal is unavailable. +- verdict: CONFIRMED — cmd/multi_service_helpers.go:318-329 swallows the error into an empty map, and neither query can report total failure: queryRDSInstancesInRegions returns (map, nil) unconditionally after per-region logging (cmd/multi_service_engine_versions.go:127-128, :140-143) and queryMajorEngineVersionsWithClient does the same per engine (:220-226), so isInExtendedSupport's "no info means not in extended support" default at :393-397 turns the filter into a no-op. +- issue: (pending cross-reference) + +### A10-015 Savings Plan type is matched against bare string literals with no default arm +- category: silent-fallback +- severity: medium +- location: cmd/multi_service_stats_helpers.go:27 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: `SavingsPlanDetails.PlanType` is a plain `string` (pkg/common/types.go:613) and both `categorizeSPRecommendations` and `collectSPSavings` (line 107) switch on the literals `"Compute"`, `"EC2Instance"`, `"SageMaker"`, `"Database"` with no default. If AWS returns or a future parser emits any other spelling, the rec's savings fall into no bucket and vanish from the SAVINGS PLANS section and from the RI-vs-SP comparison, so the recommended purchasing strategy is computed from a subset of the data without any indication that rows were dropped. The same four literals are duplicated across the two functions, so they can drift apart. +- evidence: + ```go + switch details.PlanType { + case "Compute": + breakdown.ComputeSavings += rec.EstimatedSavings + breakdown.ComputeCount++ + case "EC2Instance": + ... + } // no default: unrecognised plan types are silently dropped + ``` +- suggested fix: Define a `common.SavingsPlanType` const set, use it in both switches, and add a default arm that logs the unrecognised value rather than dropping the row. +- verdict: PLAUSIBLE — the missing default arm and the duplicated literals are real (cmd/multi_service_stats_helpers.go:27-40 and :107-116, the second switch omitting SageMaker entirely), but reaching the drop needs the runtime condition of Cost Explorer returning a plan type outside the four modelled SDK members, which spPlanTypeDisplayString passes through verbatim at providers/aws/recommendations/parser_sp.go:286; costexplorer v1.63.1 models exactly the four the switches handle. +- issue: (pending cross-reference) + +### A10-020 Skipping GCP service-account creation still returns a fabricated account email +- category: silent-fallback +- severity: medium +- location: cmd/configure_gcp.go:678 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: An operator answers `s` at Step 3 because they already have a differently-named service account. `gcpStepCreateServiceAccount` returns the *constructed* string `cudly-service-account@.iam.gserviceaccount.com` rather than signalling that nothing was created. Step 4 then prints "grants roles/compute.admin to cudly-service-account@… on project …" for a principal that may not exist, and Step 5 tries to mint a key for it. The wizard states a fact about an identity it never verified, and the failure surfaces several steps later as an opaque IAM error. +- evidence: + ```go + saName := "cudly-service-account" + saEmail := fmt.Sprintf("%s@%s.iam.gserviceaccount.com", saName, projectID) + ... + case "s", "skip": + fmt.Println("Skipping Create Service Account") + } + return saEmail, nil + ``` +- suggested fix: Return an empty string on skip and prompt for the existing service-account email, the way `gcpStepCreateKey` already returns "" on skip so the caller asks for an existing credentials file. +- verdict: CONFIRMED — the skip and default arms at cmd/configure_gcp.go:673-677 fall through to `return saEmail, nil` at :678 with the address constructed at :651-652, unlike gcpStepCreateKey which returns "" on skip (:744-750). +- issue: (pending cross-reference) + +### A10-023 `rekey` counts any zero-key decrypt failure as "already re-keyed" and exits successfully +- category: silent-fallback +- severity: medium +- location: cmd/rekey/main.go:168 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: `rekeyOne` treats *every* `credentials.Decrypt` failure as proof the row is already encrypted under the real key. A row whose `encrypted_blob` is truncated, base64-corrupt, empty, or written by a third key version fails the same way and lands in `skippedAlreadyReal`. The run then reports `errored=0`, `run` returns nil, and the migration is declared complete while an unreadable credential row survives untouched. The only signal is a count in one log line that the operator has no baseline to compare against. +- evidence: + ```go + plaintext, err := credentials.Decrypt(zeroKey, blob) + if err != nil { + // Decrypt with zero key failed — assume already real-key encrypted. + return outcomeSkipped + } + ``` +- suggested fix: Attempt a real-key decrypt in the skip branch and only count the row as `skippedAlreadyReal` when that succeeds; anything failing under both keys is an error with its id logged. +- verdict: CONFIRMED — rekeyOne returns outcomeSkipped for any Decrypt error whatsoever (cmd/rekey/main.go:167-172), that outcome only ever increments skippedAlreadyReal (:146-151), and run returns nil whenever cs.errored == 0 (cmd/rekey/main.go:81-87), so a corrupt or third-key row leaves the migration reporting success. +- issue: (pending cross-reference) + +### A11-009 A truncated 200 response is turned into `null` for every endpoint +- category: silent-fallback +- severity: medium +- location: frontend/src/api/client.ts:284 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: the catch is documented as covering 204s and body-stripping proxies, but it is unconditional: any 200 whose body is truncated or malformed (a proxy cutting a large recommendations payload, a partial gateway write) resolves as `null` instead of rejecting. Callers that spread the result then see "no data" rather than an error — for example `listActiveCommitments` does `resp.commitments ?? []` and would throw on null, while `getHistory` returns null into a renderer that treats it as an empty list. A user sees an empty table where a failure occurred. +- evidence: + ```typescript + // api/client.ts:284-288 + try { + return await response.json() as T; + } catch { + return null as T; + } + ``` +- suggested fix: return null only when the response has no body to parse (204, or `content-length: 0`), and rethrow the parse error otherwise. +- verdict: CONFIRMED — the catch at frontend/src/api/client.ts:283-287 is unconditional and inspects neither the status nor `content-length`, so every 2xx whose body fails to parse resolves as `null` for every endpoint; the sibling catch on the error path (client.ts:271-273) is deliberately narrow by comparison, and nothing downstream distinguishes "empty by design" from "truncated". +- issue: (pending cross-reference) + +### A11-016 A failed account fetch tells the operator no accounts are configured, then clones an unscoped group +- category: silent-fallback +- severity: medium +- location: frontend/src/groups/groupModals.ts:623 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: `openDuplicateGroupModal` catches any `listAccounts` failure and continues with `accounts = []`. `renderDuplicateAccountsList` then renders "No cloud accounts configured yet. Duplicating without scope clones the full source group — add accounts first if you want to restrict." (groupModals.ts:566). On a 403 or a transient 5xx the operator is told a factual untruth about their deployment and is nudged to create the duplicate with no account scoping, inheriting the source group's `allowed_accounts` unchanged. That is the widest available outcome on an authorization-scoping path. +- evidence: + ```typescript + // groups/groupModals.ts:620-627 + let accounts: api.CloudAccount[] = []; + try { + accounts = await api.listAccounts(); + } catch (err) { + console.error('Failed to list accounts for duplicate modal:', err); + accounts = []; + } + ``` +- suggested fix: distinguish the two states — on a fetch failure render an error line and disable the duplicate submit, keeping the "none configured" copy for a genuinely empty list. +- verdict: CONFIRMED — the catch swallows every `listAccounts` failure into `accounts = []` (frontend/src/groups/groupModals.ts:620-626), the empty branch of `renderDuplicateAccountsList` prints the "No cloud accounts configured yet" copy unconditionally (groupModals.ts:562-568), and the modal still opens (groupModals.ts:630) with the submit path enabled, whose own doc comment confirms that ticking nothing inherits the source group's `allowed_accounts` as-is (groupModals.ts:646-650). +- issue: (pending cross-reference) + +### A13c-012 Two secrets are seeded with hardcoded placeholder literals that the runtime cannot distinguish from a real credential +- category: silent-fallback +- severity: medium +- location: terraform/modules/secrets/gcp/main.tf:172 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: with `create_sendgrid_secret = true` and no `sendgrid_api_key`, the secret is + created holding the string `PLACEHOLDER_REPLACE_ME`. The Azure module does the same for both + SMTP credentials (`secrets/azure/main.tf:190` and `:207`, value + `PLACEHOLDER_GENERATE_IN_AZURE_PORTAL`). The application resolves a secret that exists and is + non-empty, so nothing errors at startup; the first real send fails at the provider with an + authentication error, at whatever hour the first approval email goes out. A secret that is + absent fails loudly at resolve time; a secret holding a placeholder does not. +- evidence: + ```hcl + resource "google_secret_manager_secret_version" "sendgrid_api_key" { + count = var.sendgrid_api_key != null || var.create_sendgrid_secret ? 1 : 0 + + secret = google_secret_manager_secret.sendgrid_api_key[0].id + secret_data = var.sendgrid_api_key != null ? var.sendgrid_api_key : "PLACEHOLDER_REPLACE_ME" + } + ``` +- suggested fix: create the empty secret container without a version when no value is supplied, + so the runtime's "no version" resolve error names the missing manual step directly. +- verdict: CONFIRMED — the three placeholder literals are as cited (secrets/gcp/main.tf:172, + secrets/azure/main.tf:190 and :207), and `/usr/bin/grep -rn PLACEHOLDER internal/ providers/ cmd/` + returns nothing, so no runtime code distinguishes a placeholder from a real credential; the + resolvers only fail on absent or unreadable secrets. +- issue: (pending cross-reference) + +### A01-016 A GetGlobalConfig failure silently reverts purchase suppressions to the default grace period +- category: silent-fallback +- severity: low +- location: internal/api/handler_purchases.go:2628 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: An operator has set `grace_period_days` to 0 for GCP (feature off) or to 30 for AWS. During a DB blip `GetGlobalConfig` errors; `executePurchase` (and `persistRetryExecution`, 1913) pass `nil` config to `buildSuppressions`, which substitutes `config.DefaultGracePeriodDays` for every provider. Suppression rows are written for the disabled provider and with the wrong expiry for the others; the error is not even logged. Every other config read on these paths fails closed (`approveViaToken`, 608-611). +- evidence: + ```go + var gracePeriodCfg *config.GlobalConfig + g, getConfigErr := h.config.GetGlobalConfig(ctx) + if getConfigErr == nil { + gracePeriodCfg = g + } + + suppressions := buildSuppressions(execReq.Recommendations, executionID, gracePeriodCfg, time.Now()) + ``` +- suggested fix: Return the config error (500) from both call sites instead of defaulting; `buildSuppressions` can then require a non-nil config. +- verdict: CONFIRMED — executePurchase (internal/api/handler_purchases.go:2627-2633) and persistRetryExecution:1912-1916 pass nil on a GetGlobalConfig error without logging it, and buildSuppressions:71-73 substitutes config.DefaultGracePeriodDays whenever cfg is nil, so a provider configured to 0 still gets suppression rows; approveViaToken:608-611 and approvePurchaseViaSession:708-711 return the same error instead. +- issue: (pending cross-reference) + +### A02-013 Account update and create treat an omitted `enabled` (and `aws_is_org_root`) as false +- category: silent-fallback +- severity: low +- location: internal/api/handler_accounts.go:515 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: `Enabled *bool` signals an optional field, but `cloudAccountFromRequest` leaves `a.Enabled` at the zero value when the pointer is nil and `updateAccount` (lines 568-576) builds the stored row from the request alone, copying only CreatedAt/CreatedBy from `existing`. An API client (or Terraform) sending `PUT /api/accounts/{id}` with name and credentials but no `enabled` key silently disables the account and clears `aws_is_org_root`; the scheduler then skips it with no error. The frontend happens to always send `enabled` (settings.ts:2453), which hides this from UI users only. +- evidence: + ```go + if req.Enabled != nil { + a.Enabled = *req.Enabled + } + ... + account := cloudAccountFromRequest(req) + account.ID = id + account.CreatedAt = existing.CreatedAt + account.CreatedBy = existing.CreatedBy + ``` +- suggested fix: In `updateAccount`, fall back to `existing.Enabled` when `req.Enabled == nil` (and treat `aws_is_org_root` the same way with a `*bool`), or require `enabled` and return 400 when absent. +- verdict: CONFIRMED — CloudAccountRequest has Enabled *bool but AWSIsOrgRoot bool (types.go:29,48); cloudAccountFromRequest (handler_accounts.go:504-516) leaves Enabled false when the pointer is nil and copies AWSIsOrgRoot's zero value, and updateAccount (handler_accounts.go:568-576) copies only ID/CreatedAt/CreatedBy from `existing` before UpdateCloudAccount, so an omitted key persists as false. +- issue: (pending cross-reference) + +### A02-022 Commitment-option validation errors are swallowed, allowing the save +- category: silent-fallback +- severity: low +- location: internal/api/handler_config.go:177 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: `checkCommitmentOptionCombo` is the server-side guard that a (term, payment) pair is actually sold for the service. When `commitmentOpts.Validate` errors (DB blip, timeout) the handler logs and returns nil, so `PUT /api/config/service/aws/rds` with an unsupported combination persists and the scheduler later fails the purchase. The comment defends this as "the frontend's hardcoded rules are the primary gate", which makes the backend check advisory on precisely the request it exists for. +- evidence: + ```go + ok, err := h.commitmentOpts.Validate(ctx, cfg.Provider, cfg.Service, cfg.Term, cfg.Payment) + if err != nil { + logging.Warnf("commitment-option validation error (allowing save): %v", err) + return nil + } + ``` +- suggested fix: Return a 503 (or 500) on a validation error so the operator retries, keeping only `ErrNoData` as the permissive case. +- verdict: CONFIRMED — checkCommitmentOptionCombo (handler_config.go:177-181) logs and returns nil on a Validate error and updateServiceConfig (handler_config.go:356-362) proceeds to SaveServiceConfig; the swallow is documented as intentional in the function comment (handler_config.go:166-171), so this is a design choice the finding disputes rather than an accidental gap, and low severity stands. +- issue: (pending cross-reference) + +### A03-012 A missing role ARN resolves to the host's ambient credentials instead of an error +- category: silent-fallback +- severity: medium +- location: internal/credentials/resolver.go:175 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: `role_arn` with an empty `AWSRoleARN` returns `ambient` (the Lambda/instance role); `ResolveGCPTokenSourceWithOpts` returns `(nil, nil)` for `application_default` (resolver.go:390-392) and Azure `managed_identity` returns the host identity (resolver.go:346-347). The API validator permits an empty ARN in `role_arn` mode (internal/api/handler_accounts.go:412-426), so an account row whose ARN is blanked by an edit, or an account registered by any `create:accounts` holder, quietly runs collection and purchases as the CUDly deployment identity rather than failing. The "Self account" shape is intentional, but the resolver cannot distinguish "operator chose Self" from "customer ARN went missing", and the choice is not gated to admins anywhere in scope. +- evidence: + ```go + if account.AWSRoleARN == "" { + // Self-account: auth_mode=role_arn with no role ARN means "use the + // CUDly Lambda's own credentials to access this account." ... + if ambient != nil { + return ambient, nil + } + return nil, fmt.Errorf("credentials: aws_role_arn is empty and no ambient credentials available (account %s)", account.ID) + } + ``` +- suggested fix: Make the ambient shape an explicit auth mode value (e.g. `self`) validated at the boundary and creatable only by admins, and have `role_arn` with an empty ARN error; return an explicit ambient sentinel instead of `(nil, nil)` on the GCP path. +- verdict: PLAUSIBLE — the empty-ARN-means-host-identity shape is documented and deliberate in three places (resolver.go:165-168 and :175-183, AWSResolveOptions comment :80-83, scheduler.go:753-759), validateAWSRoleARN accepts "" (validation.go:90-93) and createAccount is gated on create:accounts only (handler_accounts.go:307), so the fallback is real but the "silent" and privilege claims need a deployment that delegates create:accounts/update:accounts to non-admins, which the source cannot establish; on the no-opts callers an empty ARN errors rather than falls back (resolver.go:182). +- severity-adjusted: low — a documented feature reachable only by account-management holders, not an unintended fallback on the money path. +- issue: (pending cross-reference) + +### A04-006 saveGlobalConfigWith silently rewrites an RI-exchange utilization threshold of 0 to 95 +- category: silent-fallback +- severity: medium +- location: internal/config/store_postgres.go:265-268 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: the RI exchange config handler accepts `utilization_threshold` in `[0, 100]` (internal/api/handler_ri_exchange.go:2576) and writes it through `UpdateGlobalConfigAtomic` without a whole-config Validate. An operator with auto-exchange enabled sets the threshold to 0 to stop exchanges from triggering (no RI has utilization below 0). The store replaces 0 with 95.0 before the UPSERT, the config round-trips as 95, and the auto-exchange task now considers every RI under 95 percent for an irreversible exchange. The same block rewrites `ri_exchange_lookback_days` 0 to 30 and `recommendations_lookback_days` 0 to 7. +- evidence: + ```go + riExchangeUtilizationThreshold := config.RIExchangeUtilizationThreshold + if riExchangeUtilizationThreshold == 0 { + riExchangeUtilizationThreshold = 95.0 + } + ``` +- suggested fix: delete the zero-to-default rewrites in `saveGlobalConfigWith`; reject or explicitly define 0 in the handler's `validate` and persist what was validated. +- verdict: CONFIRMED — the handler accepts 0 (`UtilizationThreshold < 0 || > 100`, internal/api/handler_ri_exchange.go:2576-2578) and `saveGlobalConfigWith` rewrites it to 95.0 before the UPSERT (internal/config/store_postgres.go:265-268), so the persisted value is not the validated one; the sibling rewrites of `ri_exchange_lookback_days` and `recommendations_lookback_days` are at 260-264. +- severity-adjusted: low — no backend consumer reads `RIExchangeUtilizationThreshold`; every reference is the config read/write plus the two GET responses (internal/api/handler_ri_exchange.go:1997, internal/server/handler_ri_exchange.go:103), so no exchange is selected by the rewritten value and the harm is a lied-about setting rather than an unintended exchange. +- issue: (pending cross-reference) + +### A04-016 database env parsing falls back to defaults on malformed values +- category: silent-fallback +- severity: low +- location: internal/database/config.go:179-204 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: `DB_MAX_CONNECTIONS=2O` (letter O) yields 25, `DB_CONNECT_TIMEOUT=10` (no unit, which `time.ParseDuration` rejects) yields 10s, `DB_AUTO_MIGRATE=yes` yields false. The operator's setting is discarded without any log line and `Validate` sees only the defaults, so a pool sized for a Lambda concurrency limit or a deliberately disabled auto-migrate silently reverts. +- evidence: + ```go + func getEnvInt(key string, defaultValue int) int { + if value := os.Getenv(key); value != "" { + if intVal, err := strconv.Atoi(value); err == nil { + return intVal + } + } + return defaultValue + } + ``` +- suggested fix: return an error from `LoadFromEnv` when a set variable fails to parse (mirroring `maybeForceMigrationVersion`'s non-numeric rejection). +- verdict: CONFIRMED — `getEnvInt`, `getEnvBool` and `getEnvDuration` each discard the parse error and fall through to the default with no log line (internal/database/config.go:179-203), and the three cited defaults are exactly the values claimed: `DB_MAX_CONNECTIONS` 25 (internal/database/config.go:47), `DB_CONNECT_TIMEOUT` 10s (config.go:52), `DB_AUTO_MIGRATE` false (config.go:55). +- issue: (pending cross-reference) + +### A05-012 Azure probe treats a missing `Valid` flag as "combo is available" +- category: silent-fallback +- severity: low +- location: internal/commitmentopts/probe_azure.go:187 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: `probeCombo` returns true unless some benefit explicitly reports `Valid == false`. A response with an empty `Benefits` slice, or one whose `Valid` pointer is nil (a field the service omitted, or an SDK shape change), is read as "this (term, payment) is sold". The 5-year candidate combos exist precisely so the API can reject what it does not sell, so a nil flag would persist `P5Y` combos as live and the frontend would offer a 5-year Azure savings plan that cannot be bought. The code comments the empty-slice case as "can't happen in practice" but does not guard the nil-pointer case at all. +- evidence: + ```go + for _, b := range resp.Benefits { + if b != nil && b.Valid != nil && !*b.Valid { + return false, nil + } + } + return true, nil + ``` +- suggested fix: require positive confirmation — return true only when at least one benefit reports `Valid == true`, and treat an empty slice or a nil flag as not-offered. +- verdict: PLAUSIBLE — the code reads exactly as claimed (probe_azure.go:188-196: the loop can only return false, and a nil `b.Valid` or an empty Benefits slice falls through to `return true`), and probe_azure_test.go:138 and :150 pin that permissive behaviour as intended. The named consequence needs two conditions I could not establish: that the Azure service ever omits the Valid flag, and that the prober runs at all — A05-009 shows ProbeAzure has no non-test caller, so no P5Y combo can currently reach the store or the frontend. +- issue: (pending cross-reference) + +### A06-017 ANALYTICS_COLLECTION_ENABLED silently stays enabled on an unparseable value +- category: silent-fallback +- severity: low +- location: internal/server/analytics_collect.go:96 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: `strconv.ParseBool` rejects `no`, `off`, `False `, `disabled`. An operator who sets `ANALYTICS_COLLECTION_ENABLED=no` to stop the collector (for example while a partition problem is being fixed) gets the default `true` with no warning at all, and the collector keeps writing. The immediately preceding helper in the same file, `loadAnalyticsInt`, deliberately returns an out-of-range sentinel so `Validate` fails startup rather than "running with a default the operator never asked for" — this function contradicts that rationale. +- evidence: + ```go + func getEnvBool(key string, defaultVal bool) bool { + if val := os.Getenv(key); val != "" { + if result, err := strconv.ParseBool(val); err == nil { + return result + } + } + return defaultVal + } + ``` +- suggested fix: make a set-but-unparseable value a startup error via `AnalyticsConfig.Validate`, or at minimum log a warning as the sibling env parsers do. +- verdict: CONFIRMED — `getEnvBool` discards the `ParseBool` error entirely and returns `defaultVal` with no log line at all (internal/server/analytics_collect.go:96-103), unlike `getEnvInt`/`getEnvFloat` which at least warn (internal/server/app.go:948,965); it is the reader for `ANALYTICS_COLLECTION_ENABLED` with `defaultVal=true` (analytics_collect.go:49), and `strconv.ParseBool` rejects `no`/`off`/`disabled`, so `cfg.Enabled` stays true and the gate at analytics_collect.go:129-133 never trips. `Validate` checks only the two int knobs (analytics_collect.go:75-83). +- issue: (pending cross-reference) + +### A06-018 analytics_refresh reports success with fabricated zero counters when the analytics store is absent +- category: silent-fallback +- severity: low +- location: internal/server/handler.go:369 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: when `app.Analytics` is nil the task logs "skipping" and returns `{"status":"success","views_refreshed":0,"partitions_created":0,"partitions_dropped":0}`. The HTTP scheduled-task path (`handleScheduledHTTP`) wraps that in `{"status":"success"}` too, so a monitor watching the endpoint sees a healthy refresh that never happened. `partitions_created` and `partitions_dropped` are hardcoded zeros this function never writes to at all. The neighbouring `handleCleanupExpiredRecords` has the same defect: `result["sessions_deleted"]` is initialised to 0 and never assigned, then logged as "Cleanup complete: 0 sessions". +- evidence: + ```go + } else { + log.Println("Analytics store not available, skipping materialized view refresh") + } + log.Printf("Analytics refresh complete") + return result, nil + ``` +- suggested fix: return `status: "skipped"` when the store is nil (as `handleCollectAnalytics` already does) and drop the counters that are never populated. +- verdict: CONFIRMED — the nil-store branch only logs and falls through to `return result, nil` with `status: "success"` still set (internal/server/handler.go:372-397), `partitions_created` and `partitions_dropped` are initialised at handler.go:375-376 and never written anywhere in the function, and the HTTP wrapper adds its own `"status":"success"` envelope (internal/server/http.go:257-265). `handleCollectAnalytics` does use `"skipped"` for the same condition (internal/server/analytics_collect.go:134-138). The sibling defect also reproduces: `sessions_deleted` is set to 0 at handler.go:275, never assigned (`CleanupExpiredSessions` returns only an error, handler.go:280-286), then logged as a count at handler.go:303. +- issue: (pending cross-reference) + +### A06-024 Request body read errors are discarded and oversize bodies are silently truncated +- category: silent-fallback +- severity: low +- location: internal/server/http.go:270 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: `err` from `io.ReadAll` is dropped, so a client that disconnects mid-upload or a body that fails to read yields `body = ""` and the request continues into the API router as if it had been sent with an empty body. Separately the 10 MB `LimitReader` truncates rather than rejecting: an oversize POST reaches the handler as a body cut mid-JSON, producing a confusing 400 parse error instead of a 413. +- evidence: + ```go + limited := io.LimitReader(r.Body, maxBodySize) + bodyBytes, err := io.ReadAll(limited) + if err == nil && len(bodyBytes) > 0 { + body = string(bodyBytes) + } + ``` +- suggested fix: propagate the read error as a 400 and use `http.MaxBytesReader` so an oversize body is rejected with 413 instead of silently cut. +- verdict: CONFIRMED — `httpToLambdaRequest` has no error return and the `err` from `io.ReadAll` is only used as a guard on assigning `body`, never propagated (internal/server/http.go:271-280), so a mid-upload disconnect or read fault yields `Body: ""` on the request handed to `app.API.HandleRequest` (http.go:169-172); `io.LimitReader` truncates at 10 MB rather than erroring, and no `http.MaxBytesReader` appears anywhere in the package. +- issue: (pending cross-reference) + +### A06-025 STS failure yields an "unknown" account id that is persisted on exchange audit rows +- category: silent-fallback +- severity: low +- location: internal/server/handler_ri_exchange.go:130 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: on a transient `GetCallerIdentity` failure this returns the literal `"unknown"`, which flows into `RunAutoExchangeParams.AccountID` and is written to `ri_exchange_records.account_id` for every record created by that run. The audit trail for a money action then carries a fabricated identifier that cannot be distinguished from a genuine one, and the sibling ladder path deliberately refuses to do this (`defaultLadderAccountResolver`, internal/server/handler_ladder.go:150, "a fabricated/'unknown' value is unsafe"). +- evidence: + ```go + identity, err := stsClient.GetCallerIdentity(ctx, &sts.GetCallerIdentityInput{}) + if err != nil { + log.Printf("Warning: failed to get AWS account ID via STS: %v (using 'unknown')", err) + return "unknown" + } + ``` +- suggested fix: return an error and abort the reshape run, as the ladder resolver does; an audit row is not worth writing with a fabricated subject. +- verdict: CONFIRMED — `resolveAccountID` returns the `"unknown"` literal on both the STS error and the nil-`Account` path (internal/server/handler_ri_exchange.go:130-141), it is assigned to `clients.accountID` (handler_ri_exchange.go:71) and passed as `RunAutoExchangeParams.AccountID` (handler_ri_exchange.go:108), which is written onto every record the run creates (pkg/exchange/auto.go:374,566,615) and persisted into the `account_id` column (internal/config/store_postgres.go:2641). The sibling ladder resolver refuses exactly this and returns an error instead (internal/server/handler_ladder.go:144-156). +- issue: (pending cross-reference) + +### A07-003 Partial-upfront breakdown uses a hardcoded 50% split and silently treats unknown payment options as all-upfront +- category: silent-fallback +- severity: medium +- location: providers/aws/services/savingsplans/client.go:663 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: `GetOfferingDetails` reports `UpfrontCost` / `RecurringCost` to the caller. For a partial-upfront SP the split is fixed at exactly 0.5 rather than read from the offering, and any payment-option string outside the six recognised spellings (for example the CE display form `"Partial Upfront"` arriving from a persisted rec) falls into `default` and is reported as 100% upfront with zero recurring — the most expensive shape, presented as fact. +- evidence: + ```go + case "Partial Upfront", "partial-upfront": + return totalCost * 0.5, (totalCost * 0.5) / hoursInTerm + case "No Upfront", "no-upfront": + return 0, totalCost / hoursInTerm + default: + return totalCost, 0 + } + ``` +- suggested fix: return an error for unrecognised payment options (matching `convertPaymentOption` two functions above), and derive the upfront share from the offering rates rather than a fixed 0.5. +- verdict: CONFIRMED — the fixed 0.5 at providers/aws/services/savingsplans/client.go:668 and the `default: return totalCost, 0` at :671-672 are exactly as described, though the cited `"Partial Upfront"` example is in fact handled at :667 so the unknown value must be some other string. +- severity-adjusted: low — the only consumer is `GetOfferingDetails` (client.go:607), which has no production caller in the repo; `GetOfferingDetails` appears only as an interface declaration at pkg/provider/interface.go:50, so no operator sees these numbers today. +- issue: (pending cross-reference) + +### A07-004 Term helpers silently fall back to one year on any unrecognised term +- category: silent-fallback +- severity: medium +- location: providers/aws/services/savingsplans/client.go:655 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: `convertTermToSeconds` in the same file errors on an unknown term, but `calculateHoursInTerm` and `normalizeTermString` on the `GetOfferingDetails` path silently return the 1-year value. A rec carrying `Term: "5yr"` or an empty term produces `TotalCost` computed over 8760 hours and a `Term: "1yr"` label, which the caller displays as the offering's real terms. +- evidence: + ```go + func calculateHoursInTerm(term string) float64 { + if term == "3yr" || term == "3" { + return 3 * 365 * 24 + } + return 365 * 24 + } + ``` +- suggested fix: route both helpers through `convertTermToSeconds` and propagate its error so `GetOfferingDetails` fails rather than labelling an unknown term as one year. +- verdict: CONFIRMED — `calculateHoursInTerm` (providers/aws/services/savingsplans/client.go:655) and `normalizeTermString` (:677) both fall through to the 1-year value, and the path is reachable only through the CE-OfferingID short-circuit at :425 because otherwise `convertTermToSeconds` (:536) errors first inside `findOfferingID`. +- severity-adjusted: low — same reachability limit as A07-003: `GetOfferingDetails` has no production caller, only the pkg/provider/interface.go:50 declaration. +- issue: (pending cross-reference) + +### A07-028 Unrecognised OpenSearch payment options are passed through unchanged +- category: silent-fallback +- severity: low +- location: providers/aws/services/opensearch/client.go:477 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: `normalizeOpenSearchPaymentOption` returns the input verbatim in its `default` arm. A rec carrying `"All Upfront"` produces `wantPayment = "All Upfront"`, which never equals the SDK enum string, so the first offering matching instance type and duration triggers the "has payment option X, want Y" error and aborts the whole offering search. The operator sees an API-mismatch error rather than "your recommendation's payment option is not one of the three supported values". +- evidence: + ```go + case "no-upfront": + return string(types.ReservedInstancePaymentOptionNoUpfront) + default: + return option + } + ``` +- suggested fix: return an error for unrecognised values and surface it from `findOfferingID`, matching `convertEC2PaymentOption` and the ElastiCache/MemoryDB equivalents. +- verdict: CONFIRMED — providers/aws/services/opensearch/client.go:478-479 returns the input verbatim in the `default` arm, and `scanOpenSearchOfferingPage` :448/:457-460 then compares it against the SDK enum string and aborts the whole search with an "has payment option X, want Y" error rather than naming the unsupported input. +- issue: (pending cross-reference) + +### A08-024 GCP offering details parse the term with bare literals and silently fall back to one year +- category: silent-fallback +- severity: medium +- location: providers/gcp/services/computeengine/client.go:856 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: `GetOfferingDetails` only recognizes "3yr" and "3". A recommendation carrying "36mo" — a form `termPlan` (client.go:50) explicitly accepts on the purchase path — is priced as a one-year commitment, so `recurringCost` is three times the real monthly charge and `hoursInTerm` inside `getComputePricing` is a third of the real term. The purchase path and the pricing path therefore disagree about the same recommendation. Repeated verbatim in cloudsql/client.go:260, memorystore/client.go:271 and cloudstorage/client.go:275. +- evidence: + ```go + termYears := 1 + if rec.Term == "3yr" || rec.Term == "3" { + termYears = 3 + } + ``` +- suggested fix: call the existing `termYearsFromTerm` helper (client.go:200) — or better, a variant that returns an error — in all four clients so the pricing path accepts exactly what the purchase path accepts. +- verdict: PLAUSIBLE — the bare, case-sensitive literals are in all four `GetOfferingDetails` (computeengine/client.go:856-859, cloudsql:260-263, memorystore:271-274, cloudstorage:275-278) while `termPlan` (client.go:48-56) accepts "36mo" and `convertGCPRecommendation` (client.go:1094-1102) can therefore emit `Term == "36mo"`, but the live recommendations path already uses `termYearsFromTerm` (`enrichRecWithPricing`, client.go:1157) and `GetOfferingDetails` has no non-test caller anywhere in the tree — it is only a `pkg/provider/interface.go:50` member — so the mispricing needs an out-of-tree consumer of that exported interface. +- severity-adjusted: low — no in-tree caller reaches the divergent branch; the in-tree pricing path uses the correct helper. +- issue: (pending cross-reference) + +### A09-018 A non-empty Details payload for an unrecognized service is discarded without error +- category: silent-fallback +- severity: medium +- location: pkg/common/service_details_codec.go:91 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: `newDetailsForService` has no entry for `memorydb`, and none for any service slug a newer writer introduces. A row persisted by a newer deployment with `service = ""` and a fully populated `Details` blob decodes to `(nil, nil)` on an older reader: the payload is dropped and the purchase proceeds with nil details rather than failing. The comment calls this "suspicious but tolerated", which is the tolerance that let #453 hide for a release. +- evidence: + ```go + target, ok := newDetailsForService(service) + if !ok { + // Service has no *Details type — nothing to decode. Empty raw + // payload is fine; a non-empty payload on a service we don't + // recognize is suspicious but tolerated + return nil, nil + } + ``` +- suggested fix: return `(nil, nil)` only when `raw` is empty; when a payload exists for a service with no typed shape, return an error naming the service so the version skew surfaces instead of silently dropping purchase parameters. +- verdict: CONFIRMED — `newDetailsForService` (pkg/common/service_details_codec.go:121-167) has no case for memorydb or any unlisted slug, and `DecodeServiceDetailsFor` returns `(nil, nil)` at line 95 before ever looking at `raw`, so a fully populated payload is discarded without error. +- severity-adjusted: low — no current service slug loses data this way: service_details_codec.go:154-161 documents memorydb's nil details as deliberate (the MemoryDB client reads `rec.ResourceType` directly), and the version-skew case needs a slug a future writer has not introduced yet. The consumers that need details already fail loud on nil, e.g. providers/aws/services/ec2/client.go:455-457. +- issue: (pending cross-reference) + +### A11-004 A malformed Max Amount is silently dropped, widening the permission instead of refusing the save +- category: silent-fallback +- severity: medium +- location: frontend/src/groups/groupModals.ts:450 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: when the number input holds something `parseFloat` cannot turn into a finite non-negative value, the branch simply skips the assignment and the permission is saved with no `max_amount`. For a permission that previously carried a cap, the save removes it and the operator sees a success toast. The same file refuses to save at all when a list constraint cannot round-trip (`unrepresentablePermissionErrors`, groupModals.ts:361), so the money dimension is the one dimension that fails open. +- evidence: + ```typescript + // groups/groupModals.ts:450-460 + if (maxAmount) { + const parsed = parseFloat(maxAmount); + if (Number.isFinite(parsed) && parsed >= 0) { + permission.constraints.max_amount = parsed; + } + } + ``` +- suggested fix: push a message into `unrepresentablePermissionErrors` when the box is non-empty but does not parse, so `saveGroup` refuses rather than sending an uncapped permission. +- verdict: PLAUSIBLE — the skip is real and silent (frontend/src/groups/groupModals.ts:450-460), but the input is `type="number" min="0"` (groupModals.ts:269) inside `#group-form`, which is submitted by a genuine submit-button click (index.html:1093), so native constraint validation blocks the reachable malformed cases: a negative fails `min`, and non-numeric text leaves `value === ""`. Reaching the branch needs a value that passes validation yet fails `Number.isFinite`, e.g. an exponent overflow like `1e400`, which I could not establish from source. +- severity-adjusted: low — the common malformed inputs never reach the branch; only an exponent-overflow value slips past the input's own validation. +- issue: (pending cross-reference) + +### A11-018 A drifted API-keys response shape renders the reassuring "No API keys yet" empty state +- category: silent-fallback +- severity: low +- location: frontend/src/apikeys.ts:51 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: the triple fallback ends in `?? []`. If the backend ever returns a shape carrying neither `api_keys` nor a bare array (a wrapped envelope, a partial response), the list resolves to empty and `renderApiKeysList` paints "No API keys yet — Create an API key to let automation tools call CUDly programmatically". An operator reviewing which keys exist before revoking credentials is shown "none" for a deployment that has active keys. The catch branch already has a correct error rendering path that this case bypasses. +- evidence: + ```typescript + // apikeys.ts:50-55 + const response = await api.getApiKeys(); + const list = (response as { api_keys?: APIKeyInfo[] } | undefined)?.api_keys + ?? (Array.isArray(response) ? response as APIKeyInfo[] : undefined) + ?? []; + currentApiKeys = Array.isArray(list) ? list : []; + ``` +- suggested fix: treat "neither known shape" as an error and route it through `renderApiKeysListError`, keeping `[]` only for a genuine empty `api_keys` array. +- verdict: PLAUSIBLE — the fallback chain does end in `?? []` and feeds the reassuring empty state (frontend/src/apikeys.ts:50-55), and it composes with A11-009 so a body-stripped 200 arrives as `null` and renders "No API keys yet" instead of the error path at apikeys.ts:69-75; the trigger is an off-contract response from `/api-keys` (api/apikeys.ts:16-18), which I could not produce from source. +- issue: (pending cross-reference) + +### A12-055 Absent money values filter as `0` while the cell renders `--` +- category: silent-fallback +- severity: low +- location: frontend/src/history.ts:951 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: The block comment at lines 920-923 states the contract that "a user typing the visible '$123' matches the row that displays that exact value". `formatCurrency(null)` renders `--`, but `purchaseHistoryNumericCellValue` coerces null to `0`, so typing `0` into the Upfront Cost or Monthly Savings filter returns rows whose cells show `--`, conflating "not reported" with "actually free". The Approval Queue extractor gets this right for `monthly_cost` (line 1661, returns `NaN`) but not for the other two. `plans.ts:247` has the same `?? 0` shape. +- evidence: + ```ts + case 'count': return p.count ?? 0; + case 'upfront_cost': return p.upfront_cost ?? 0; + case 'savings': return p.estimated_savings ?? 0; + ``` +- suggested fix: Return `Number.NaN` for absent money values in both extractors, matching the approval queue's `monthly_cost` case. +- verdict: CONFIRMED — formatCurrency renders `--` for null (frontend/src/utils.ts:44-46) but the extractor coerces count/upfront_cost/savings to 0 (frontend/src/history.ts:951-953), against the stated contract at :920-923; the approval-queue extractor returns NaN for the analogous monthly_cost case (frontend/src/history.ts:1660). +- issue: (pending cross-reference) + +### A12-057 Marketplace dialog silently degrades to a no-amount consent when the row lookup misses +- category: silent-fallback +- severity: low +- location: frontend/src/history.ts:1451 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: The whole price summary is wrapped in `if (purchase)`. When `lastPurchases.find` returns `undefined` (a stale render, or the row scrolled out of the fetched window after A12-029), the dialog renders only the generic fee note with no RI id, no remaining term, no list price and no net proceeds, and "Confirm listing" still submits. The user authorises a marketplace sale with no amount on screen. +- evidence: + ```ts + const purchase = lastPurchases.find(p => p.purchase_id === id); + const bodyEl = document.createElement('div'); + bodyEl.className = 'marketplace-pricing-modal-body'; + if (purchase) { + // entire price summary + } + ``` +- suggested fix: Abort with an error toast when the lookup fails; never open a money-authorising dialog without an amount. +- verdict: CONFIRMED — The entire price summary is inside `if (purchase)` (frontend/src/history.ts:1451-1512) while the confirmDialog and the createMarketplaceListing call sit outside it (frontend/src/history.ts:1520-1535), so a missed lastPurchases lookup opens a money-authorising dialog carrying only the generic fee note. +- issue: (pending cross-reference) + +### A12-060 `total_completed` falls back to the total row count +- category: silent-fallback +- severity: low +- location: frontend/src/history.ts:405 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: If the API supplies `total_pending` and `total_purchases` but omits `total_completed` — the partial-deploy case the surrounding comment is written for — `completed` becomes the full row count and the line reads "10 completed · 3 pending" for a dataset with 7 completed. The lines immediately above go out of their way to render `--` rather than fabricate money values, then this one fabricates a count. +- evidence: + ```ts + const completed = summary.total_completed ?? total; + const pending = summary.total_pending ?? 0; + const detail = (total !== null && pending > 0) + ? `

${completed} completed · ${pending} pending

` + : ''; + ``` +- suggested fix: Render the detail line only when `summary.total_completed != null`. +- verdict: CONFIRMED — `const completed = summary.total_completed ?? total` fabricates the count from the row total (frontend/src/history.ts:405) immediately after the same function renders `--` rather than fabricate money values (frontend/src/history.ts:371-390). +- issue: (pending cross-reference) + +### A12-064 `collectTargets` silently coerces an empty or sub-1 exchange count to 1 +- category: silent-fallback +- severity: low +- location: frontend/src/riexchange.ts:1757 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: A user who clears the Count field intending to retype it, or types `0`, and clicks Get Quote gets a quote for a single unit instead of an inline validation error. The same coercion in `updateRunningTotal` (line 1680) makes the displayed total agree with the substitution, so nothing signals the typed value was discarded. The offering-id branch four lines above correctly returns an error for the analogous empty case. +- evidence: + ```ts + const rawCount = parseInt(row.countInput.value, 10); + const targetCount = isNaN(rawCount) || rawCount < 1 ? 1 : rawCount; + targets.push({ offering_id: offeringId, count: targetCount }); + ``` +- suggested fix: Return `{ targets: [], error: 'Target N: count must be a whole number >= 1.' }`, matching the offering-id check. +- verdict: CONFIRMED — collectTargets coerces an empty or sub-1 count to 1 (frontend/src/riexchange.ts:1756-1758) four lines after the offering-id branch returns an explicit error for the analogous empty case (:1747-1749), and updateRunningTotal repeats the same coercion so the display agrees with the substitution (frontend/src/riexchange.ts:1679-1681). +- issue: (pending cross-reference) + +### A12-065 RI utilization failures are swallowed and the column stays "..." forever +- category: silent-fallback +- severity: low +- location: frontend/src/riexchange.ts:324 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: When `getRIUtilization` fails or is throttled, the catch logs and returns, leaving `currentUtilization` empty so every row renders a permanent loading ellipsis. The operator cannot distinguish "Cost Explorer is slow" from "utilization is unavailable", and an active `utilization_pct` column filter silently matches zero rows because the extractor returns `NaN` for every row. +- evidence: + ```ts + } catch (error) { + console.error('Failed to load RI utilization:', error); + } + ``` +- suggested fix: Record the failure in module state and render "n/a" with the error in a `title`, instead of a permanent ellipsis. +- verdict: CONFIRMED — The catch logs and returns without recording the failure (frontend/src/riexchange.ts:324-326), leaving currentUtilization empty so every row renders the permanent loading ellipsis at frontend/src/riexchange.ts:378. +- issue: (pending cross-reference) + +### A12-075 The summary card's "absent savings" branch is unreachable, so missing savings render as $0 +- category: silent-fallback +- severity: medium +- location: frontend/src/recommendations.ts:800 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: `pageLevelRange` initialises `savingsMin`/`savingsMax` to `0` and only ever `+=` into them, so both are always numbers; a rec whose `savings` arrives as null contributes `0` through `cellSummary`. The `plr.savingsMin === null || plr.savingsMax === null` test therefore never fires and the `'--'` the comment above it promises is dead. Execution falls to the final branch, `formatCostForPeriod(0, period)`, so a page whose savings data is entirely absent renders "Potential Monthly Savings $0" — the fabricated zero the M-1 note says it prevents. The `?? null` on lines 795-796 is likewise a no-op for the same reason. +- evidence: + ```ts + const savingsText = hasRecs && plr.savingsMax > 0 && scaledSavingsMin !== null && scaledSavingsMax !== null + ? formatScaledRange(scaledSavingsMin, scaledSavingsMax, period) + : hasRecs && (plr.savingsMin === null || plr.savingsMax === null) + ? '--' + : formatCostForPeriod(0, period); + ``` +- suggested fix: Have `cellSummary`/`pageLevelRange` propagate `null` when no variant carried a savings figure (skipping nulls the way `approval-details.ts:104` does), so the `'--'` branch can actually fire. +- verdict: CONFIRMED — pageLevelRange initialises savingsMin/savingsMax to 0 and only ever `+=` into them (frontend/src/recommendations.ts:1073-1082), so the null test at frontend/src/recommendations.ts:802 can never fire, the `'--'` branch is dead and the `?? null` at :795-796 is a no-op. +- severity-adjusted: low — Recommendation.savings is non-nullable in the API contract (frontend/src/api/types.ts:148), so the fabricated zero needs the backend to violate its own schema; with conforming data only the dead branch remains +- issue: (pending cross-reference) + +### A13b-011 Registration gate ignores `contact_email` in all four Terraform modules despite the message claiming otherwise +- category: silent-fallback +- severity: low +- location: iac/federation/aws-target/terraform/registration.tf:2 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: `local.do_register` tests only `var.cudly_api_url`, but the payload always carries `contact_email`, and the CloudFormation equivalent gates on both (`iac/federation/aws-cross-account/cloudformation/template.yaml:53-55`, `DoRegister` is an `Fn::And` over URL and email). A customer who sets the URL and forgets the email registers an account with an empty contact, so CUDly has no one to notify about a pending purchase, and the skip message printed by every `registration_response` output ("Skipped (cudly_api_url or contact_email not set)") states a condition the code never evaluates. Identical in the `aws-cross-account`, `azure-target`, `gcp-sa-impersonation` and `gcp-target` registration files. +- evidence: + ```hcl + locals { + do_register = var.cudly_api_url != "" + ``` +- suggested fix: Make the local `var.cudly_api_url != "" && var.contact_email != ""` in all five files, or drop the email clause from the output messages so the two agree. +- verdict: CONFIRMED — `do_register = var.cudly_api_url != ""` appears verbatim in all five registration files (`aws-target:2`, `aws-cross-account:4`, `azure-target:2`, `gcp-sa-impersonation:2`, `gcp-target:2`) while every payload still sets `contact_email` and every `registration_response` output prints "Skipped (cudly_api_url or contact_email not set)"; the CloudFormation counterpart really does gate on both (`iac/federation/aws-cross-account/cloudformation/template.yaml:52-55`, `DoRegister` as an `Fn::And`), so the Terraform side is the one that disagrees with its own message. +- issue: (pending cross-reference) + +### A13c-019 `secrets/aws` accepts an empty-string database password and stores it, where the admin password in the same file rejects it +- category: silent-fallback +- severity: low +- location: terraform/modules/secrets/aws/main.tf:51 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: `random_password.database` is gated on `var.database_password == null`, and + the version body picks the variable whenever it is `!= null`. Passing `""` therefore skips + generation *and* selects the empty string, writing `{"username":"cudly","password":""}` into + Secrets Manager and, through `database/aws`, into `aws_db_instance.password`. The admin password + eight lines below handles precisely this (`var.admin_password == null || var.admin_password == + ""` on the generator, and the matching two-part test on the version), so the asymmetry is inside + one file. +- evidence: + ```hcl + resource "random_password" "database" { + count = var.database_password == null ? 1 : 0 + } + + resource "aws_secretsmanager_secret_version" "database_password" { + secret_string = jsonencode({ + password = var.database_password != null ? var.database_password : random_password.database[0].result + }) + } + ``` +- suggested fix: use the same `== null || == ""` test on both the generator count and the version + value, matching `admin_password`. +- verdict: PLAUSIBLE — the asymmetry is exactly as described: `random_password.database` gates on + `== null` alone (secrets/aws/main.tf:25) and the version body on `!= null` (:51), while + `random_password.admin_password` (:60) and its version (:85) both carry the `|| == ""` test, and + `database_password` has no validation (variables.tf:34-39). The scenario needs a caller passing + `""`; the sole in-tree caller passes `null` (environments/aws/secrets.tf:14). +- issue: (pending cross-reference) + +### A14-042 The ECR guard omits the empty-string check its RDS sibling documents as the decisive one +- category: silent-fallback +- severity: low +- location: scripts/force-delete-owned-ecr-repo.sh:75 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: `jq -er` accepts an empty string (it is neither `null` nor `false`), so a state publishing `"ecr_repository_name": {"value": ""}` yields `OWNED_REPO=""`. The script then echoes "This state owns ECR repository ''" and runs `aws ecr describe-repositories` before `select-owned-name.sh` refuses with exit 2. `disable-owned-rds-deletion-protection.sh:120-137` adds an explicit check for exactly this row of its own measured table, with a comment explaining that the selector's refusal comes too late and names the wrong problem. The ECR script, which shares the selector and the same failure mode, has no equivalent branch. +- evidence: + ```bash + OWNED_REPO="$(jq -er '.ecr_repository_name.value' <<<"$OUTPUTS_JSON")" + echo "This state owns ECR repository '$OWNED_REPO'" + ``` +- suggested fix: Add the same `case "$OWNED_REPO" in '' | *[![:graph:]]*)` guard the RDS script uses, before the first AWS call. +- verdict: CONFIRMED — scripts/force-delete-owned-ecr-repo.sh:75-78 takes `jq -er '.ecr_repository_name.value'`, echoes it and calls `aws ecr describe-repositories` with no empty or whitespace check, while disable-owned-rds-deletion-protection.sh:126-137 carries exactly the `case "$OWNED_INSTANCE" in '' | *[![:graph:]]*)` guard before its first AWS call. +- issue: (pending cross-reference) + +### Category: correctness + +118 findings: 2 critical, 12 high, 59 medium, 45 low. + +### A11-001 The group create/edit form's submit handler is never wired in production +- category: correctness +- severity: critical +- location: frontend/src/groups/handlers.ts:17 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: `setupGroupHandlers()` is the only code that binds `saveGroup` to `#group-form`, and nothing calls it at runtime. Its sole caller is `users.ts:setupHandlers()` (frontend/src/users.ts:40), which has no callers anywhere outside tests. `app.ts:setupEventListeners()` calls `setupUserHandlers()` (frontend/src/app.ts:176) but never the group equivalent, and `index.ts`'s DOMContentLoaded block wires `create-group-btn`, `close-group-modal-btn`, `add-permission-btn` and `group-duplicate-form` but not `group-form` (frontend/src/index.ts:43-50). `#group-form` in index.html:1076 carries `` and no inline `onsubmit`. So clicking "Save Group" in the Create Group / Edit Group modal runs a native form submission against the current URL instead of `saveGroup`: the page navigates/reloads and the group is never created or updated. Every permission edit made through the Groups panel is silently lost. `groups.test.ts` does not catch this because it calls `groupModals.saveGroup(event)` directly (groups.test.ts:560) and calls `groupHandlers.setupGroupHandlers()` itself (groups.test.ts:933) — the suite proves the handler works, never that the app installs it. +- evidence: + ```typescript + // groups/handlers.ts:10-21 — the only binder for #group-form + export function setupGroupHandlers(): void { + (window as any).openCreateGroupModal = openCreateGroupModal; + (window as any).closeGroupModal = closeGroupModal; + (window as any).addPermission = () => addPermission(); + const groupForm = document.getElementById('group-form'); + if (groupForm) { + groupForm.addEventListener('submit', (e) => void saveGroup(e)); + } + } + ``` +- suggested fix: call `setupGroupHandlers()` from `setupEventListeners()` in app.ts next to `setupUserHandlers()`, and add a test that runs the real init path and asserts a `submit` on `#group-form` reaches `api.updateGroup`. +- verdict: CONFIRMED — `setupGroupHandlers` has exactly three non-test references (definition at frontend/src/groups/handlers.ts:10, barrel re-export at frontend/src/groups/index.ts:27, dynamic import inside the uncalled `setupHandlers` at frontend/src/users.ts:43), `#group-form` appears in only handlers.ts:17 and groupModals.ts:26/53 with no delegated `submit` listener anywhere (the only non-test `addEventListener('submit')` sites are ladder.ts:395, plans.ts:2190, settings.ts:2716, app.ts:155/160, apikeys.ts:394, users/handlers.ts:27, riexchange.ts:2003, index.ts:50, auth.ts:136/470/600/774/1060) and no `onsubmit` in index.html, and groups.test.ts:917-976 installs the binding itself so nothing asserts the app does. +- issue: (pending cross-reference) + +### A12-003 RI-exchange execute never sends `region`, which the backend rejects with 400 +- category: correctness +- severity: critical +- location: frontend/src/riexchange.ts:1809 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: The user gets a valid quote (the quote handler resolves the region from the SDK chain, so it succeeds) and clicks "Execute Exchange". The POST body omits `region`; `validateExecuteExchangeBody` at `internal/api/handler_ri_exchange.go:1756` returns `400 "region is required for execute; omitting it risks exchanging RIs in the wrong region"`. Every UI-initiated exchange fails at the last step, and the failure reads as a backend fault. `ExchangeExecuteRequest.region` is optional in the TypeScript type, so there is no compile-time signal. +- evidence: + ```ts + const result = await api.executeExchange({ + ri_ids: modalQuoteReq.ri_ids, + targets: modalQuoteReq.targets, + target_offering_id: modalQuoteReq.target_offering_id, + target_count: modalQuoteReq.target_count, + max_payment_due_usd: modalQuote.PaymentDueRaw, + }); + ``` +- suggested fix: Carry a region into the modal (from the source RI's availability zone, or have the quote response echo the region the backend resolved) and send the same value on quote and execute. +- verdict: CONFIRMED — submitModalExecute posts exactly five fields with no region (frontend/src/riexchange.ts:1808-1814), executeExchange forwards the body verbatim (frontend/src/api/riexchange.ts:91-96), and validateExecuteExchangeBody returns 400 on an empty Region (internal/api/handler_ri_exchange.go:1756-1758) before executeExchange reads it at :1793. +- issue: (pending cross-reference) + +### A01-003 "Run now" strands the execution in `running`; nothing executes it and the reaper marks it failed +- category: correctness +- severity: high +- location: internal/api/handler_purchases.go:320 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: Operator clicks Run now on a planned purchase. `runPlannedPurchase` CAS-transitions pending->running and returns "Purchase execution initiated" (frontend toasts "Purchase executed successfully", frontend/src/plans.ts:782-783). No executor consumes `running` rows: the scheduler fires only `GetPendingExecutions` (status IN pending,notified; store_postgres.go:1479), `claimAndExecute` claims from approved/pending/notified (internal/purchase/manager.go:179), and `purchase.Reaper` flips `running` rows older than 10 minutes to `failed` (reaper.go:25). The row also drops out of the Planned list (`plannedListStatuses` = pending/notified/paused, line 95). Net effect: the operator is told the purchase ran, nothing is bought, and 10 minutes later the row reads "failed". `TestHandler_runPlannedPurchase` (handler_purchases_test.go:1245) only asserts the status flip. +- evidence: + ```go + if _, err := h.config.TransitionExecutionStatus(ctx, executionID, []string{"pending", "paused"}, "running", resolveCreatorUserID(session)); err != nil { + return nil, NewClientError(409, fmt.Sprintf("execution %s cannot be started: %v", executionID, err)) + } + + return map[string]any{ + "execution_id": executionID, + "status": "running", + "message": "Purchase execution initiated", + }, nil + ``` +- suggested fix: Either hand the claimed row to the purchase manager synchronously (the same `ApproveAndExecute`/`claimAndExecute` funnel the approve path uses, with the 4-eyes and constraint gates) or transition to `approved` and enqueue the execute message; do not report success from a bare status flip. +- verdict: CONFIRMED — runPlannedPurchase (internal/api/handler_purchases.go:320-328) CAS-flips to running and returns; the only executors are ProcessScheduledPurchases → GetPendingExecutions (WHERE status IN pending,notified; internal/config/store_postgres.go:1479), FireScheduledDelayedPurchases → GetScheduledExecutionsDue (internal/purchase/scheduled_fire.go:45), and the reaper whose stuckStatuses include running (internal/purchase/reaper.go:25); frontend/src/plans.ts:782 toasts "Purchase executed successfully" on the bare 200. +- issue: (pending cross-reference) + +### A02-002 setupAdmin stores the base64-encoded password, so the bootstrap admin cannot log in with the password they typed +- category: correctness +- severity: high +- location: internal/api/handler_auth.go:238 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: The frontend base64-encodes the setup-admin password (frontend/src/api/auth.ts:285, `password: base64Encode(password)`; `base64Encode` is `btoa`, client.ts:191). The handler forwards `setupReq` untouched, the adapter copies `req.Password` verbatim (internal/server/app.go:1023-1027), and `Service.SetupAdmin` hashes it (internal/auth/service_user.go:33-38). `login` (handler_auth.go:34-38) decodes base64 before `auth.Login`, so bcrypt compares the plaintext against a hash of the base64 text and every login of the first admin fails. base64_password_guard_test.go:58-63 exempts `setupAdmin` with the justification that "the auth service handles hashing internally", which does not address the encoding mismatch. Verified by reading the three layers; not executed. If the deployed flow works, something outside these files must be re-encoding, and I could not find it. +- evidence: + ```go + var setupReq SetupAdminRequest + if err := json.Unmarshal([]byte(req.Body), &setupReq); err != nil { + return nil, NewClientError(400, "invalid request body") + } + response, err := h.auth.SetupAdmin(ctx, setupReq) + ``` +- suggested fix: Decode with `decodeBase64Password` in `setupAdmin` exactly as `login`/`resetPassword` do, drop the `knownExemptFunctions` entry, and add an end-to-end test that creates the admin through the handler and then logs in through `login` with the same frontend-encoded body. +- verdict: CONFIRMED — frontend/src/auth.ts:526 calls api.setupAdmin which sends base64Encode(password) (frontend/src/api/auth.ts:285); setupAdmin (handler_auth.go:238-249) unmarshals and forwards without decodeBase64Password, the adapter copies req.Password verbatim (internal/server/app.go:1023-1027), SetupAdmin hashes it (internal/auth/service_user.go:33-38), and login (handler_auth.go:34-38) decodes before auth.Login, so the plaintext can never match; no decode exists anywhere in internal/auth (grep base64 finds only API-key encoding) and no e2e test covers setup-then-login. +- issue: (pending cross-reference) + +### A07-006 ElastiCache offering lookup sends the raw CE engine string as ProductDescription, unvalidated +- category: correctness +- severity: high +- location: providers/aws/services/elasticache/client.go:311 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: `parseElastiCacheDetails` (parser_services.go:70) copies CE's `ProductDescription` verbatim into `CacheDetails.Engine`. `resolveEC2Tenancy` in that same file documents that CE returns these values title-cased, so a real rec carries `"Redis"`, while the ElastiCache offering filter matches on the lowercase product description. The filter matches nothing and the purchase fails as "no offerings found" rather than naming the real cause. When CE omits the field entirely, `details.Engine` is `""` and the code sends an explicit empty-string filter — a positive filter for an empty product description — instead of leaving it unset or erroring. RDS handles the identical problem with `normalizeEngineName` (rds/client.go:573); ElastiCache has no equivalent. +- evidence: + ```go + input := &elasticache.DescribeReservedCacheNodesOfferingsInput{ + CacheNodeType: aws.String(rec.ResourceType), + ProductDescription: aws.String(details.Engine), + Duration: aws.String(duration), + OfferingType: aws.String(offeringType), + ``` +- suggested fix: normalise `details.Engine` to the lowercase `redis` / `memcached` values and return an explicit error when it is empty or unrecognised, mirroring `normalizeEngineName`. +- verdict: PLAUSIBLE — the raw pass-through is real (providers/aws/recommendations/parser_services.go:69-71 copies CE's `ProductDescription` verbatim, providers/aws/services/elasticache/client.go:311 sends it unchanged, and that file contains no `strings.ToLower` or normalise helper at all, unlike rds/client.go:573), but the failure needs two runtime facts I could not establish from source: that CE actually returns `"Redis"` title-cased for ElastiCache (the title-case evidence at parser_services.go:83-85 is about the EC2 tenancy field), and that AWS treats an empty `ProductDescription` as a match-nothing filter rather than ignoring it. +- issue: (pending cross-reference) + +### A07-010 Daily-sparkline coverage call sets Granularity together with GroupBy, which the API rejects +- category: correctness +- severity: high +- location: providers/aws/recommendations/usage_history.go:85 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: the `GetReservationCoverage` SDK operation doc states "If GroupBy is set, Granularity can't be set." `fetchDailyCoverage` sets both, so every real call returns a ValidationException. `AttachDailyUsageHistory` logs the error at WARN and continues, so the entire per-row usage-history feature is dead in production while the collection looks healthy. No test asserts `Granularity` on this input (the three `GranularityDaily` assertions in the package are in ondemand_series_test.go and sp_coverage_test.go, whose operations do accept it), which is why mock-backed tests stay green. The sibling path in coverage.go correctly omits `Granularity` when it sets `GroupBy`. +- evidence: + ```go + input := &costexplorer.GetReservationCoverageInput{ + TimePeriod: &types.DateInterval{...}, + Granularity: types.GranularityDaily, + GroupBy: []types.GroupDefinition{ + {Type: types.GroupDefinitionTypeDimension, Key: aws.String(string(types.DimensionInstanceType))}, + }, + Filter: dailyUsageFilter(serviceFilter, region), + Metrics: []string{"Hour"}, + } + ``` +- suggested fix: drop the `Granularity` field (the response is already per-day when the time period is daily-resolvable), and assert its absence in the test so the pair cannot be reintroduced. +- verdict: CONFIRMED — providers/aws/recommendations/usage_history.go:85-91 sets both, and costexplorer@v1.63.1/api_op_GetReservationCoverage.go:118 states verbatim "If GroupBy is set, Granularity can't be set"; the sibling inputs at coverage.go:216 and :241 omit it, usage_history_test.go asserts nothing about `Granularity`, and the live path runs on every refresh via client.go:339. +- issue: (pending cross-reference) + +### A08-004 Azure VM SKU enrichment is dead on the recommendations path and burns a full SKU-catalogue walk per subscription +- category: correctness +- severity: high +- location: providers/azure/services/compute/client.go:914 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: `RecommendationsClientAdapter` constructs the compute client with an empty region (`newComputeClientFn(r.cred, r.subscriptionID, "")`, providers/azure/recommendations.go:158). `populateVMSKUMapFromPage` then filters every SKU through `isAvailableInRegion(sku, "")`, whose `strings.EqualFold(*location, "")` is false for every real location, so the catalogue map is always empty and `cachedSKULookup` always returns ok=false. Every VM recommendation ships with `Details.VCPU == 0` and `Details.MemoryGB == 0`, while the client still pages up to `maxSKUPages` (20) pages of `ResourceSKUs` per subscription and discards 100% of them. With the new org-wide fan-out this repeats once per subscription per sweep. +- evidence: + ```go + if !c.isAvailableInRegion(sku, c.region) { + continue + } + ``` +- suggested fix: skip the region filter when `c.region == ""` (the subscription-wide recommendations call is region-agnostic by design, see the GetRecommendations doc comment), or skip the catalogue fetch entirely on that path so the wasted pagination goes away. +- verdict: CONFIRMED — providers/azure/recommendations.go:158 constructs the client with region "", `isAvailableInRegion` (compute/client.go:667) EqualFolds every real location against "" and returns false, so `fetchSKUCatalogue` (client.go:871) pages the catalogue and returns an empty map that `cachedSKULookup` (client.go:846) can never hit, leaving Details.VCPU/MemoryGB at 0 in convertAzureVMRecommendation (client.go:813). +- issue: (pending cross-reference) + +### A08b-017 GCP `ResourceType` is a resource instance name, then used as a pricing tier and validated against tier constants +- category: correctness +- severity: high +- location: providers/gcp/services/memorystore/client.go:456 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd (same helper at `cloudsql/client.go:470`, `cloudstorage/client.go:462`) +- failure scenario: `extractGCPResourceType` returns the last path segment of the first operation resource — for Memorystore that is `projects/p/locations/l/instances/my-redis`, so `ResourceType` becomes `my-redis`; for Cloud Storage it becomes the bucket name. That value is then passed to `skuMatchesTier` as the tier to substring-match against SKU descriptions (never matches, so pricing always fails) and to `ValidateOffering`, which compares it against the constant list `{"BASIC","STANDARD_HA"}` at line 311 and always returns `invalid Memorystore tier: my-redis`. `computeengine` fixed exactly this class at `client.go:1197` by requiring a `/machineTypes/` path; the three other clients did not. +- evidence: + ```go + for _, op := range opGroup.Operations { + if op.Resource == "" { + continue + } + parts := strings.Split(op.Resource, "/") + if len(parts) > 0 { + return parts[len(parts)-1] + } + } + ``` +- suggested fix: select the operation whose resource path names the type being priced (mirroring `machineTypeFromResourcePath`) and return an error when none is present, instead of taking the first segment available. +- verdict: CONFIRMED — `extractGCPResourceType` returns the last path segment of the first non-empty operation resource (memorystore:456-472, cloudsql:470-486, cloudstorage:462-478); that value is passed to `skuMatchesTier` (memorystore:435-452) through `fillRedisPricing` (491-496) on the live conversion path, and to `ValidateOffering` (254-266) against the two-entry list at 311-314. `computeengine` requires a `/machineTypes/` segment at client.go:1197-1228, exactly the pattern the finding cites. +- issue: (pending cross-reference) + +### A08b-029 Compute Engine existing commitments report `ResourceType` as "VCPU" instead of a machine type +- category: correctness +- severity: high +- location: providers/gcp/services/computeengine/client.go:524 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: `commitment.Resources` is the list of `ResourceCommitment` entries a CUD is composed of, and each entry's `Type` is the enum `VCPU`, `MEMORY`, `LOCAL_SSD` or `ACCELERATOR`. Taking index zero writes `com.ResourceType = "VCPU"` for every commitment. Recommendations set `ResourceType` to a machine type such as `n2-standard-8` (`extractResourceTypeFromRecommendation`, line 1206), so no existing commitment can ever match a recommendation by resource type — dedupe and coverage silently see zero overlap and re-recommend capacity the project already owns. The commitment's own `Type` field (`GENERAL_PURPOSE_N2`), which does identify the covered family, is discarded. +- evidence: + ```go + if len(commitment.Resources) > 0 { + resource := commitment.Resources[0] + if resource.Type != nil { + com.ResourceType = *resource.Type + } + } + ``` +- suggested fix: map `commitment.Type` (the `Commitment_Type` enum) to the machine family it discounts, the inverse of `machineFamilyCommitmentType`, and carry the vCPU/memory amounts in `Count` and the details struct. +- verdict: CONFIRMED — computeengine:524-529 assigns `Resources[0].Type`, which the repo's own `ResourceCommitment` doc at 535-538 states is the `"VCPU"` or `"MEMORY"` enum, while the recommendation side sets `ResourceType` to a machine type via `extractResourceTypeFromRecommendation` (1197-1222); `commitment.Type` is never read in the converter. +- issue: (pending cross-reference) + +### A12-008 The AWS "Savings Plans" optgroup is hidden and disabled whenever a provider is selected +- category: correctness +- severity: high +- location: frontend/src/plans.ts:2326 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: `#plan-provider` has no empty option, so its default value is `aws` and `setupPlanHandlers()` runs `updateServiceDropdownForProvider('aws')` at app init. The match is `optgroup.label.toLowerCase().includes(provider.toLowerCase())`; the shipped markup labels the SP group `"Savings Plans"` (`frontend/src/index.html:853`), which does not contain the substring `aws`. The group therefore gets `hidden` plus `disabled`, and all four AWS Savings Plans options become unselectable in the Plan modal for the entire session. The sibling optgroup carries `id="plan-service-aws-sp"`, which the code ignores. +- evidence: + ```ts + optgroups.forEach(optgroup => { + const optgroupLabel = optgroup.label.toLowerCase(); + const shouldShow = optgroupLabel.includes(provider.toLowerCase()); + optgroup.classList.toggle('hidden', !shouldShow); + optgroup.disabled = !shouldShow; + ``` +- suggested fix: Match on a `data-provider` attribute (or the existing `id` prefix) instead of a substring of the human-readable label, and add the SP group to the fixture used by the `provider scopes service dropdown` tests. +- verdict: CONFIRMED — index.html labels the group `Savings Plans` (frontend/src/index.html:853), the match is a substring test against the provider slug (frontend/src/plans.ts:2326-2330), and setupPlanHandlers runs it at init against `#plan-provider`'s default `aws` (frontend/src/plans.ts:2299), so all four SP options are hidden and disabled for the session. +- issue: (pending cross-reference) + +### A13b-001 GCP SA-impersonation module grants two roles that cannot bind at project scope and never grants the CUD-purchase permission +- category: correctness +- severity: high +- location: iac/federation/gcp-sa-impersonation/terraform/main.tf:27 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: A customer who follows the SA-impersonation path grants CUDly's target service account `roles/commerceorgpolicy.commitmentAdmin` and `roles/billing.viewer` through `google_project_iam_member`. Both are organization- / billing-account-scoped roles, so the bindings either 400 at apply or confer nothing at project scope. Neither role carries `compute.commitments.create`, `compute.commitments.update` or `compute.regions.list`, which are the permissions the code actually needs: `providers/gcp/services/computeengine/client.go:708` calls `RegionCommitments.Insert` and `providers/gcp/recommendations.go:127` warns that region listing needs `roles/compute.viewer`. The onboarding reports success and every commitment purchase 403s. The sibling module states the rule in its own docs at `iac/federation/gcp-target/terraform/variables.tf:198` ("GCP's built-in commitment/billing roles (e.g. roles/commerceorgpolicy.commitmentAdmin, roles/billing.viewer) are organization- or billing-account-scoped and will 400 if granted at project scope") and grants `roles/compute.viewer` plus a custom role holding `compute.commitments.create`/`.update` instead. +- evidence: + ```hcl + resource "google_project_iam_member" "cudly_commitment_admin" { + project = var.project_id + role = "roles/commerceorgpolicy.commitmentAdmin" + member = "serviceAccount:${var.service_account_email}" + } + + resource "google_project_iam_member" "cudly_billing_viewer" { + project = var.project_id + role = "roles/billing.viewer" + member = "serviceAccount:${var.service_account_email}" + } + ``` +- suggested fix: Mirror `gcp-target`: bind `roles/compute.viewer` plus a project-scoped custom role carrying `compute.commitments.create` and `compute.commitments.update`, and move `roles/billing.viewer` to a `google_billing_account_iam_member` gated on a `billing_account_id` variable, exactly as `terraform/modules/compute/gcp/cloud-run/main.tf:484` already does. The same two roles appear in `internal/iacfiles/templates/gcp-sa-impersonation-cli.sh.tmpl:28` and need the same correction. +- verdict: CONFIRMED — dead grant by missing action: `iac/federation/gcp-sa-impersonation/terraform/main.tf:27,33` binds only `roles/commerceorgpolicy.commitmentAdmin` and `roles/billing.viewer` at project scope, neither of which carries `compute.commitments.create` or `compute.regions.list`, while `providers/gcp/services/computeengine/client.go:708` calls `RegionCommitments.Insert` and `providers/gcp/recommendations.go:127` swallows the 403 from region listing as a Warn and returns an empty slice; the sibling module's own note at `iac/federation/gcp-target/terraform/variables.tf:198-203` states both roles 400 at project scope, and the CLI twin at `internal/iacfiles/templates/gcp-sa-impersonation-cli.sh.tmpl:28-35` hides even that with `|| echo "(role may already be bound or unavailable)"`, so onboarding reports success either way. +- issue: (pending cross-reference) + +### A13b-002 Azure Bicep/ARM assigns the built-in Reservation Purchaser role that the repo documents as insufficient for purchases +- category: correctness +- severity: high +- location: iac/federation/azure-target/bicep/azure-wif.bicep:24 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: A customer who onboards through the Bicep/ARM path grants CUDly's service principal only the built-in Reservation Purchaser role at subscription scope. The Terraform equivalent for the same onboarding (`iac/federation/azure-target/terraform/main.tf:69`) instead creates a custom role, and its module documents why at `terraform/modules/iam/azure/cudly-reservation-role/main.tf:17`: the built-in role "lacks reservationOrders/write and calculatePrice/action, plus every Microsoft.BillingBenefits action, which is what caused 403s on production reservation purchases". The Bicep customer therefore reaches a subscription that reads as onboarded, passes deployment, and 403s on every reservation and savings-plan purchase. The two IaC paths for the same access disagree about scope; the Bicep side is the wrong one. +- evidence: + ```bicep + @description('Built-in role definition ID for Reservation Purchaser. Default is the well-known built-in role ID.') + param roleDefinitionId string = 'f7b75c60-3036-4b75-91c3-6b41c27c1689' + ``` +- suggested fix: Have the Bicep template create the same custom role definition (the eleven `Microsoft.Capacity` / `Microsoft.BillingBenefits` actions in `terraform/modules/iam/azure/cudly-reservation-role/main.tf:44-54`) and assign that, so the Bicep, Terraform and `arm/CUDly-CrossSubscription/template.json` definitions stay in lockstep. Regenerate `azure-wif.arm.json` from the corrected Bicep. +- verdict: CONFIRMED — dead grant by missing action, and the Bicep is the wrong side of a three-way drift: `iac/federation/azure-target/bicep/azure-wif.bicep:24,34` assigns only the built-in GUID, while both other expressions of the same access create the eleven-action custom role (`iac/federation/azure-target/terraform/main.tf:69-83` via `terraform/modules/iam/azure/cudly-reservation-role/main.tf:44-54`, and `arm/CUDly-CrossSubscription/template.json:37-57,78`); the Go purchase path posts to `/providers/Microsoft.Capacity/calculatePrice` then `/providers/Microsoft.Capacity/reservationOrders/{id}/purchase` (`providers/azure/services/database/client.go:376`, `providers/azure/services/cache/client.go:345`), needing `calculatePrice/action` and `reservationOrders/write`, the two actions the module's own comment at `terraform/modules/iam/azure/cudly-reservation-role/main.tf:16-20` says the built-in role lacks. +- issue: (pending cross-reference) + +### A13b-004 Cross-account trust principal omits the `/lambda/` path the hub role actually carries +- category: correctness +- severity: high +- location: cloudformation/stacks/CUDly-CrossAccount/template.yaml:64 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: The hub stack creates its Lambda execution role with `Path: /lambda/` (`cloudformation/stacks/CUDly/template.yaml:397`) and `RoleName: !Sub "${AWS::StackName}-LambdaRole"` (line 389), so its real ARN is `arn:aws:iam:::role/lambda/CUDly-LambdaRole`. The cross-account template builds the trusted principal without the path, producing `arn:aws:iam:::role/CUDly-LambdaRole`, an ARN that names no existing role. IAM rejects a trust policy naming a non-existent IAM principal, so the customer's stack fails with `MalformedPolicyDocument: Invalid principal in policy`; if it is created against a pre-existing decoy role of that name, the hub Lambda can never assume it. The parameter's own description ("set this to the full role name (without path prefix)") tells the operator to reproduce the wrong ARN. +- evidence: + ```yaml + AssumeRolePolicyDocument: + Version: "2012-10-17" + Statement: + - Effect: Allow + Principal: + AWS: !Sub "arn:aws:iam::${PrimaryAccountId}:role/${PrimaryRoleName}" + Action: sts:AssumeRole + ``` +- suggested fix: Add a `PrimaryRolePath` parameter defaulting to `/lambda/` and build the principal as `arn:aws:iam::${PrimaryAccountId}:role${PrimaryRolePath}${PrimaryRoleName}`, or drop `Path: /lambda/` from the hub role so the two agree. Either way the two templates must be changed together. +- verdict: CONFIRMED — dead grant by unmatchable ARN: `cloudformation/stacks/CUDly/template.yaml:386-397` is the only IAM role in the hub stack and carries `RoleName: !Sub "${AWS::StackName}-LambdaRole"` with `Path: /lambda/`, and it is the principal that performs the cross-account assume (`cloudformation/stacks/CUDly/template.yaml:524-528`, `Resource: arn:aws:iam::*:role/CUDly*`), so its real ARN is `.../role/lambda/CUDly-LambdaRole` while `cloudformation/stacks/CUDly-CrossAccount/template.yaml:64` builds `.../role/${PrimaryRoleName}` with no path segment, and the parameter description at lines 21-29 instructs the operator to supply the bare name. +- issue: (pending cross-reference) + +### A14-005 `set -e` makes the entrypoint's migration exit-code handling unreachable +- category: correctness +- severity: high +- location: scripts/entrypoint.sh:72 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: The script opens with `set -eu` (line 2). `migrate ... up 2>&1` is a simple command in no errexit-exempt position, so any non-zero exit terminates the shell immediately and `MIGRATE_EXIT_CODE=$?` never runs. Verified: `sh -c 'set -eu; false_cmd; RC=$?; echo RC=$RC'` exits 1 without printing. The entire branch below it, including the documented "exit code 1 means no change, which is okay" tolerance, is dead code. `DB_AUTO_MIGRATE=true` is the Dockerfile default (Dockerfile:176), so on every container start where golang-migrate returns non-zero for a benign reason the container dies with the raw migrate exit code and none of the diagnostics. +- evidence: + ```sh + migrate -path "$DB_MIGRATIONS_PATH" -database "$DB_URL" up 2>&1 + MIGRATE_EXIT_CODE=$? + if [ $MIGRATE_EXIT_CODE -eq 0 ]; then + echo " ✅ Migrations completed successfully" + elif [ $MIGRATE_EXIT_CODE -eq 1 ]; then + echo " ℹ️ No new migrations to apply" + ``` +- suggested fix: `MIGRATE_EXIT_CODE=0; migrate ... || MIGRATE_EXIT_CODE=$?` so the branch below actually receives the status. +- verdict: CONFIRMED — scripts/entrypoint.sh:2 sets `set -eu` and line 72's `migrate ... up` is a simple command in an errexit-live position, so line 73's `MIGRATE_EXIT_CODE=$?` is unreachable on failure (`sh -c 'set -eu; false; RC=$?; echo RC=$RC'` prints nothing and exits 1), and Dockerfile:176 defaults `DB_AUTO_MIGRATE=true`. +- issue: (pending cross-reference) + +### A01-011 Session-authed revoke skips the revocation window and token expiry the token path enforces +- category: correctness +- severity: medium +- location: internal/api/handler_purchases.go:1338 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: A user with `cancel-any` (or the creator with `cancel-own`) POSTs `/api/purchases/revoke/{id}` with a session and no token for an execution completed 30 days ago. `tryRevokeViaSession` goes straight to `revokeViaSession`; `checkRevocationWindow` (config.RevocationWindow after CompletedAt) and the expiry check live only in `validateRevokeToken`, which runs only on the token branch (1288). The row flips to `revocation_requested` and the History UI shows a revocation in flight for a commitment whose provider window closed weeks ago. The session tests (`TestHandler_revokePurchase_SessionAdminCancelAny`, 4911) use `buildCompletedExec` with no timestamps, so they cannot observe the missing check. +- evidence: + ```go + case sessErr == nil: + if csrfErr := h.validateCSRF(ctx, req); csrfErr != nil { + return nil, true, NewClientError(403, "CSRF validation failed") + } + res, revokeErr := h.revokeViaSession(ctx, execution, session.Email) + return res, true, revokeErr + ``` +- suggested fix: Call `checkRevocationWindow(execution)` in `revokeViaSession` (shared by both branches) so the window is enforced regardless of how the caller authenticated. +- verdict: CONFIRMED — checkRevocationWindow's only caller is validateRevokeToken (internal/api/handler_purchases.go:1383), which is reached only from the token branch at :1288; tryRevokeViaSession:1338 calls revokeViaSession:1446-1478 directly, which CAS-flips completed/partially_completed → revocation_requested with no CompletedAt/ExecutedAt check. +- issue: (pending cross-reference) + +### A02-003 sell-own marketplace authorization compares the scope against the account UUID with an empty name, but groups store account names +- category: correctness +- severity: medium +- location: internal/api/handler_marketplace.go:438 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: The group editor writes account NAMES into `allowed_accounts` (frontend/src/groups/groupModals.ts:578 `cb.value = acct.name`). `AccountScope.Allows(id, name)` matches either, but `authorizeAllowedAccount` passes `""` for the name, so a sell-own user scoped to "Production" is refused 403 on every RI in Production. Every other scope check in the package resolves the name first (`requireAccountAccess`, `filterPurchaseHistoryByAllowedAccounts`). The only tests (handler_marketplace_test.go:164-165, 200-201) put UUID-like values in the scope, so they pass with the bug present. +- evidence: + ```go + scope, err := h.getAccountScope(ctx, session) + ... + if scope.Allows(cloudAccountID, "") { + return nil + } + return NewClientError(403, "permission denied: purchase is in a cloud account not covered by your session's allowed accounts") + ``` +- suggested fix: Replace the body with `_, err := h.requireAccountAccess(ctx, session, cloudAccountID)` (it fetches the account and passes `account.Name`), and change the tests to grant the account by name. +- verdict: CONFIRMED — authorizeAllowedAccount (handler_marketplace.go:438) passes "" as the name so AccountScope.Allows (internal/auth/account_scope.go:104-116) can only match the UUID; the only UI writer of allowed_accounts is the duplicate-group modal (frontend/src/groups/groupModals.ts:558-578, whose comment says "names are what the backend matcher accepts") and it stores acct.name, while requireAccountAccess (scoping.go:26-41) resolves account.Name first; the marketplace tests (handler_marketplace_test.go:164-165,200-201) grant "acct-1"/"acct-other" and never exercise the name branch. +- issue: (pending cross-reference) + +### A02-004 Marketplace list/cancel run against the host's ambient AWS credentials regardless of the row's cloud account +- category: correctness +- severity: medium +- location: internal/api/handler_marketplace.go:169 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: A purchase_history row with `CloudAccountID` pointing at a federated member account (role_arn / bastion / WIF) is authorized by `authorizeSessionSell` against that account, but `loadAWSConfigWithRegion` (handler_ri_exchange.go:1267-1276) only sets the region on `getBaseAWSConfig`, the Lambda's own credentials. `CreateReservedInstancesListing` / `CancelReservedInstancesListing` are issued in the host account, where the RI id does not exist, so every multi-account deployment gets `InvalidReservedInstancesId.NotFound` (mapped to 400) and the feature only works for the self-account. The per-account credential resolver already exists (`credentials.ResolveAWSCredentialProvider`, used by `buildOrgRootAWSConfig` at handler_accounts.go:1528). +- evidence: + ```go + cfg, err := h.loadAWSConfigWithRegion(ctx, row.Region) + if err != nil { + return nil, fmt.Errorf("failed to load AWS config: %w", err) + } + ec2Client := h.buildMarketplaceEC2Client(cfg) + ``` +- suggested fix: Resolve the row's `CloudAccountID` to a `config.CloudAccount` and build the config through `credentials.ResolveAWSCredentialProvider` (as `buildOrgRootAWSConfig` does); fail with 400 when the row has no `CloudAccountID` and the deployment is multi-account. +- verdict: CONFIRMED — loadAWSConfigWithRegion (handler_ri_exchange.go:1267-1276) only overrides Region on the ambient getBaseAWSConfig and marketplaceList/Cancel (handler_marketplace.go:169,343) never consult row.CloudAccountID for credentials, while purchases in federated accounts are executed with per-account credentials (internal/purchase/execution.go:448 via ResolveAWSCredentialProviderWithOpts), so the RI exists only in the member account the host credentials cannot see; the only api-package caller of the per-account resolver is buildOrgRootAWSConfig (handler_accounts.go:1528). +- issue: (pending cross-reference) + +### A02-009 A global-defaults PUT overwrites term/payment/coverage/ramp on every service config, including fields the request did not send +- category: correctness +- severity: medium +- location: internal/api/handler_config.go:154 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: Operator has `aws/ec2` at term 3 / all-upfront via `PUT /api/config/service/aws/ec2`. Later `PUT /api/config {"default_coverage": 70}` is sent. `anyKeyPresent` is true, `propagateGlobalDefaults` rewrites `svc.Term = cfg.DefaultTerm` (1) and `svc.Payment = cfg.DefaultPayment` on ec2 and every other service, so the next scheduled purchase buys a 1-year commitment instead of the configured 3-year one. The presence guard at line 126 only decides whether to propagate; it does not restrict propagation to the keys actually sent, and failures are logged rather than returned (lines 160-162). +- evidence: + ```go + for i := range services { + svc := &services[i] + svc.Term = cfg.DefaultTerm + svc.Payment = cfg.DefaultPayment + svc.Coverage = cfg.DefaultCoverage + svc.RampSchedule = cfg.DefaultRampSchedule + if saveErr := h.config.SaveServiceConfig(ctx, svc); saveErr != nil { + logging.Warnf(...) + ``` +- suggested fix: Pass the `present` map into `propagateGlobalDefaults` and overlay only the defaults whose key was in the body (the same present-key pattern `mergeServiceConfig` already uses), and return an error if any save fails. +- verdict: CONFIRMED — updateGlobalConfig (handler_config.go:122-128) calls propagateGlobalDefaults whenever anyKeyPresent sees any one of the four keys, and propagateGlobalDefaults (handler_config.go:144-163) overwrites Term/Payment/Coverage/RampSchedule on every ServiceConfig from the merged global cfg regardless of which key was sent, logging (not returning) save failures; the guard's own comment (lines 122-125) states the intent this violates. +- issue: (pending cross-reference) + +### A03-008 A consumed recovery code stays valid when the persist fails, and the comment claims the opposite +- category: correctness +- severity: medium +- location: internal/auth/service.go:255 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: `consumeRecoveryCode` removes the hash from the in-memory slice; if `UpdateUser` fails the login is still granted and the store row is unchanged. The next login re-reads the row from the store, so the same code matches again. The comment says "repeated use of the same code on the next login will fail because the slice is stale", which is false: the slice is rebuilt from the store on every login. Reproduced with `UpdateUser` returning an error: `login#1 with recovery code (persist fails): token=true`; `login#2 SAME code: token=true`. Single-use recovery codes become multi-use exactly during a DB outage, which is also when an attacker who obtained one code gets unlimited retries. +- evidence: + ```go + if s.consumeRecoveryCode(user, req.MFACode) { + if err := s.store.UpdateUser(ctx, user); err != nil { + logging.Warnf("Failed to persist recovery-code consumption for user %s: %v", user.ID, err) + // The recovery code already verified; still allow + // login but warn -- repeated use of the same code on + // the next login will fail because the slice is + // stale, which is the safe failure mode. + } + return nil + } + ``` +- suggested fix: Fail the login when the consumption write fails (return `ErrInvalidMFACode` or a 5xx), so a recovery code is only accepted once it is provably burned; delete the incorrect comment. +- verdict: CONFIRMED — verifyPasswordAndMFA (service.go:254-262) returns nil after a failed UpdateUser, Login:158 re-reads the row through getUserAndValidateStatus -> GetUserByEmail (store_postgres.go:60-75) on every attempt, so the in-memory slice the comment relies on is discarded and the unburned hash matches again on the next login. +- issue: (pending cross-reference) + +### A03-014 KMS signers cache the first resolution error forever +- category: correctness +- severity: medium +- location: internal/oidc/aws_signer.go:86 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: `resolveOnce` uses `sync.Once`; a transient `GetPublicKey` failure, a throttle, or simply a caller whose `ctx` was already cancelled poisons `s.err`, and every later `KeyID`/`PublicKey`/`Mint` on that signer fails for the process lifetime. The signer is constructed once per process (`NewSignerFromEnv` at startup) and used by every Azure and GCP federated credential exchange, so one bad first call takes down all federated purchases and collection until the Lambda instance is recycled. Same pattern in azure_signer.go:103 and gcp_signer.go:107. +- evidence: + ```go + func (s *AWSKMSSigner) resolveOnce(ctx context.Context) { + s.once.Do(func() { + out, err := s.client.GetPublicKey(ctx, &kms.GetPublicKeyInput{KeyId: &s.keyID}) + if err != nil { + s.err = fmt.Errorf("oidc: kms:GetPublicKey: %w", err) + return + } + ``` +- suggested fix: Only memoize success: guard with a mutex, retry the fetch when `pubKey == nil`, and return the error without recording it (or resolve eagerly in the constructor and fail startup). +- verdict: CONFIRMED — resolveOnce stores the error inside sync.Once.Do (aws_signer.go:86-115, azure_signer.go:103-109, gcp_signer.go:107-113) and PublicKey/KeyID return s.err unconditionally afterwards (:75-84); NewSignerFromEnv is called exactly once at startup (server/app.go:409) with no eager resolve, so the first caller's ctx and the first transient failure decide the signer's state for the process lifetime. +- issue: (pending cross-reference) + +### A04-003 ON DELETE RESTRICT plus a pending-only preflight makes accounts with any terminal execution undeletable +- category: correctness +- severity: medium +- location: internal/config/store_postgres.go:1600-1611 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd (constraint at internal/database/postgres/migrations/000053_executions_account_fk_restrict.up.sql:26-30; handler at internal/api/handler_accounts.go:609-655) +- failure scenario: account A has one execution in status `completed`, `failed` or `canceled`. `CountPendingExecutionsForAccount` counts only `pending`/`notified`, so the preflight passes, `DELETE FROM cloud_accounts` hits the RESTRICT FK (which applies to every referencing row regardless of status), and the handler maps SQLSTATE 23503 to a 409 saying "pending purchase(s) must be canceled first" with no IDs. The frontend's Cancel-All-Then-Delete flow only cancels (rows keep `cloud_account_id`), so it can never succeed. Failed rows are never purged by `CleanupOldExecutions`, so an account with one failed purchase is permanently undeletable, and the operator is told to cancel purchases that do not exist. The 000053 test's comment ("the API layer's Cancel-All-And-Delete flow does exactly that", 000053_executions_account_fk_restrict_test.go:84) claims a row deletion the API never performs. +- evidence: + ```go + err := s.db.QueryRow(ctx, ` + SELECT COUNT(*) FROM purchase_executions + WHERE cloud_account_id = $1 + AND status IN ('pending', 'notified') + `, accountID).Scan(&n) + ``` + ```sql + ALTER TABLE purchase_executions + ADD CONSTRAINT purchase_executions_cloud_account_id_fkey + FOREIGN KEY (cloud_account_id) REFERENCES cloud_accounts(id) + ON DELETE RESTRICT; + ``` +- suggested fix: count all referencing rows split by live vs terminal in the preflight, and have `DeleteCloudAccount` detach terminal rows (`UPDATE purchase_executions SET cloud_account_id = NULL WHERE cloud_account_id = $1 AND status NOT IN (live set)`) inside its transaction before the DELETE, so only genuinely live purchases block deletion. +- verdict: CONFIRMED — `CountPendingExecutionsForAccount` counts only `pending`/`notified` (internal/config/store_postgres.go:1600-1611) while the 000053 FK restricts every referencing row regardless of status (internal/database/postgres/migrations/000053_executions_account_fk_restrict.up.sql:23-30); `CancelExecutionAtomic` leaves `cloud_account_id` set (internal/config/store_postgres.go:1158-1166) and `CleanupOldExecutions` excludes `HealthScoredExecutionStatuses` — failed and both canceled spellings — from its only expiry branch (internal/config/store_postgres.go:1873-1890, internal/config/types.go:485-489), so a failed row blocks the delete permanently. The 000053 test comment at line 84 is verified as claiming a row deletion the API never performs. +- issue: (pending cross-reference) + +### A04-008 ClaimMarketplaceListingSlot and the other purchase_id-keyed writers assume purchase_id is unique +- category: correctness +- severity: medium +- location: internal/config/store_postgres.go:1985-1998 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd (also 1948-1962, 2356-2365, 2424-2442) +- failure scenario: `purchase_history.purchase_id` has only a non-unique index (000002). `savePurchaseHistory` (internal/purchase/execution.go:893-930) writes one row per successful rec with `PurchaseID = result.CommitmentID`; a stranded execution re-driven under the same idempotency lineage (#1012) gets the existing commitment ID back from the provider dedupe and writes a second row with the same purchase_id. `ClaimMarketplaceListingSlot` then updates both rows to `listing_state='pending'` but returns `RowsAffected() == 1` false, the handler answers 409 and never releases the slot, and every later attempt sees `pending` and 409s forever. `GetPurchaseHistoryByPurchaseID` (`LIMIT 1`, no ORDER BY) and `MarkPurchaseRevoked` pick or stamp an arbitrary one of the duplicates. +- evidence: + ```go + tag, err := s.db.Exec(ctx, query, + ListingStatePending, purchaseID, ListingStateActive, ListingStatePending) + if err != nil { + return false, fmt.Errorf("failed to claim marketplace listing slot for purchase %s: %w", purchaseID, err) + } + return tag.RowsAffected() == 1, nil + ``` +- suggested fix: add a unique index on `purchase_history(purchase_id)` after a dedupe migration (or make `SavePurchaseHistory` an upsert on purchase_id), and return `RowsAffected() > 0` in the claim. +- verdict: CONFIRMED — `purchase_id` carries only the non-unique `idx_purchase_history_purchase_id` (internal/database/postgres/migrations/000002_add_indexes.up.sql:28) and `SavePurchaseHistory` is a plain INSERT with no ON CONFLICT (internal/config/store_postgres.go:2004-2011); a re-drive gets the same commitment id back from the provider dedupe (providers/aws/services/ec2/client.go:129-142, providers/aws/services/savingsplans/client.go:238-256) and `savePurchaseHistory` writes it again (internal/purchase/execution.go:893-930), after which `ClaimMarketplaceListingSlot`'s `RowsAffected() == 1` (internal/config/store_postgres.go:1990-1997) is false forever while both rows sit at `pending`. `GetPurchaseHistoryByPurchaseID`'s `LIMIT 1` with no ORDER BY is at 2355-2365. +- issue: (pending cross-reference) + +### A05-004 Scheduler-created executions have no creator, so 4-eyes makes them permanently unapprovable +- category: correctness +- severity: medium +- location: internal/purchase/notifications.go:130 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: `getOrCreateExecution` builds the row with no `CreatedByUserID` (the column stays NULL). The plan-notification email it triggers offers exactly one action, the token deep link, which lands in `ApproveExecution` -> `ApproveAndExecute` -> `checkDifferentApprover`. That function denies any row whose `CreatedByUserID` is nil with "this execution predates the dual-control feature" (approvals.go:255). So with `RequireDifferentApprover` on, every execution the notification sweep creates is un-approvable by any operator, and the only stated remedy in the error text is to turn dual control off globally. The row is not legacy — it is minted by current code on every tick that finds no existing execution for the plan+date. +- evidence: + ```go + execution := &config.PurchaseExecution{ + PlanID: plan.ID, + ExecutionID: uuid.New().String(), + Status: "pending", + StepNumber: plan.RampSchedule.CurrentStep + 1, + ScheduledDate: *plan.NextExecutionDate, + ApprovalToken: approvalToken, + ApprovalTokenExpiresAt: &tokenExpiresAt, + } + ``` +- suggested fix: stamp the plan's owner onto `CreatedByUserID` here so the identity comparison has something to compare against, or treat a system-created row (no creator, no human requester) as satisfying dual control rather than failing it, and test the notification-created row against the 4-eyes gate. +- verdict: CONFIRMED — the struct literal at notifications.go:129-146 sets no CreatedByUserID, and the email's only action is the ApprovalToken deep link, which lands in ApproveExecution (approvals.go:27) -> ApproveAndExecute (approvals.go:313) -> enforceFourEyesPolicy (approvals.go:216) -> checkDifferentApprover, whose first statement (approvals.go:253-256) denies any row with a nil CreatedByUserID and names disabling 4-eyes as the only remedy; the session path denies the same row independently at handler_purchases.go:958. +- issue: (pending cross-reference) + +### A05-005 The recommendation ID's account component is empty for every AWS reservation rec, so IDs collide across accounts +- category: correctness +- severity: medium +- location: internal/scheduler/scheduler.go:1492 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: The composite ID names `rec.Account` as the field that "separates per-account/per-subscription recs sharing the same provider+SKU+region+payment". AWS reservation recs only set it when Cost Explorer returns `AccountId` (providers/aws/recommendations/parser_ri.go:72), which is absent for payer-scope recommendations; `parser_services.go` never sets it at all. With two registered AWS accounts both surfacing an m5.large 1yr all-upfront EC2 RI, both rows get the identical id `aws||ec2|us-east-1|m5.large||1|all-upfront`. The DB unique key is `(account_key, provider, service, region, resource_type, engine, term, payment_option)` (migration 000043) and keys on the CUDly account UUID, not `rec.Account`, so both rows persist and `GetRecommendationByID` (scheduler.go:1123) returns `recs[0]` — a deep link to one account's recommendation renders the other account's. It is also the collision the comment cites #187/#188 for on the frontend selection set. +- evidence: + ```go + recordID := fmt.Sprintf("%s|%s|%s|%s|%s|%s|%d|%s", + providerName, rec.Account, rec.Service, rec.Region, + rec.ResourceType, engine, term, rec.PaymentOption) + ``` +- suggested fix: key the ID on the same account dimension the DB unique index uses — the record's `CloudAccountID` (empty for the ambient path) — instead of the provider-reported `rec.Account`, and add a test with two accounts whose recs are otherwise identical asserting distinct IDs. +- verdict: CONFIRMED — `rec.Account` is set only under `if details.AccountId != nil` (providers/aws/recommendations/parser_ri.go:72-73) and `/usr/bin/grep -n Account providers/aws/recommendations/parser_services.go` returns nothing; the ID is built at scheduler.go:1492 BEFORE tagAccount stamps CloudAccountID (fetchAndConvert, scheduler.go:1027-1029, tagAccount at :1035), so two accounts collected via fanOutPerAccount (scheduler.go:495) yield byte-identical IDs. The upsert key at store_postgres_recommendations.go:224 includes account_key so both rows persist, and GetRecommendationByID returns `&recs[0]` (scheduler.go:1138) after a filter on ID alone. +- issue: (pending cross-reference) + +### A05-010 `findAWSAccount` ignores the enabled flag despite its name and doc +- category: correctness +- severity: medium +- location: internal/commitmentopts/service.go:130 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: The doc says "returns the first enabled AWS account", but the filter sets only `Provider` — `Enabled` is left nil, so `ListCloudAccounts` returns disabled rows too and the loop takes the first one. An operator who disables a decommissioned AWS account but leaves the row in place has that account's (now-revoked) credentials used for the probe. `buildConfig` or the probe calls then fail, `probeAndPersist` returns `ErrNoData`, and `Validate` falls permissive-true forever even though a healthy enabled account sits later in the list. Compare `Scheduler.enabledAccounts` (scheduler.go:917), which sets both filter fields. +- evidence: + ```go + provider := "aws" + accounts, err := s.accounts.ListCloudAccounts(ctx, config.CloudAccountFilter{Provider: &provider}) + if err != nil { + return nil, fmt.Errorf("list cloud accounts: %w", err) + } + for i := range accounts { + if accounts[i].Provider == "aws" { + return &accounts[i], nil + ``` +- suggested fix: pass `Enabled: &enabled` in the filter as `enabledAccounts` does; the redundant in-loop `Provider == "aws"` re-check can then go too. +- verdict: CONFIRMED — service.go:127 builds `config.CloudAccountFilter{Provider: &provider}` and leaves the `Enabled *bool` field (internal/config/types.go:1046) nil, so disabled rows are returned and the loop at :132-136 takes the first by index; the sibling Scheduler.enabledAccounts sets both fields (scheduler.go:916-920). The permissive-forever consequence follows from probeAndPersist returning ErrNoData on a buildConfig failure (service.go:86) and Validate's ErrNoData branch (service.go:148). +- issue: (pending cross-reference) + +### A06-005 Two plain-text email bodies are rendered with html/template and come out HTML-escaped +- category: correctness +- severity: medium +- location: internal/email/templates.go:1031 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: `RenderPurchaseExecutedNotificationEmail` and `RenderPurchaseScheduledDelayEmail` (templates.go:895) call `renderTemplate`, whose own doc comment says "Never use it for plain-text bodies" (internal/email/template_renderers.go:26). Rendering `purchaseExecutedNotificationTemplate` with `RequestedByName="O'Brien & Co"`, `RequestedByEmail="a&b@example.com"` produces, verified by running the same template through html/template with the same FuncMap: `Requested by: O'Brien & Co <a&b@example.com>`. The recipient of a post-execution notice sees entity-encoded names and cannot copy-paste the address. The 07-H1 regression suite that fixed exactly this class (internal/email/template_renderers_test.go:504) covers PasswordReset, UserInvite, RegistrationDecision and PurchaseApprovalRequest — it omits the two renderers that were never converted, so the gap is invisible to CI. +- evidence: + ```go + func RenderPurchaseExecutedNotificationEmail(data NotificationData) (string, error) { + return renderTemplate("purchase-executed-notification", purchaseExecutedNotificationTemplate, data) + } + ``` +- suggested fix: switch both to `renderTextTemplate` and extend `TestPlainTextTemplates_NoHTMLEscaping` to cover them. +- verdict: CONFIRMED — both renderers call `renderTemplate` (internal/email/templates.go:895,1031) against that function's own "Never use it for plain-text bodies" contract (internal/email/template_renderers.go:24-28), while every sibling plain-text renderer uses `renderTextTemplate` (template_renderers.go:62-180); executing the exact line at templates.go:931-932 through html/template yields `Requested by: O'Brien & Co <a&b@example.com>`, and the regression suite covers only PasswordReset/UserInvite/RegistrationDecision/PurchaseApprovalRequest (internal/email/template_renderers_test.go:504-560). +- issue: (pending cross-reference) + +### A07-013 Compute SP layer mixes a region-scoped coverage figure with a global utilization figure +- category: correctness +- severity: medium +- location: providers/aws/ladder/layer_states.go:184 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: `fetchSPUtilizationPct` deliberately passes `region = ""` for Compute SPs because they are global, but `fetchSPCoveragePct` (line 156) is called once with `a.cfg.Region` and its result is assigned to both SP layers. The Compute SP `LayerState` therefore reports coverage measured over one region and utilization measured over all regions. In a multi-region account where the ladder runs in a low-spend region, coverage reads near zero while utilization reads near 100, and the engine sizes a Compute SP purchase against a denominator that excludes most of the commitment it is already paying for. +- evidence: + ```go + region := a.cfg.Region + if planType == spPlanTypeCompute { + region = "" // "" = all regions in the CE GetSavingsPlansUtilization API + } + summary, err := a.spUtil.GetSPUtilization(ctx, cePlanType, region, a.cfg.lookbackDays()) + ``` +- suggested fix: fetch SP coverage per layer with the same region convention utilization uses (`""` for Compute, `cfg.Region` for EC2Instance) instead of sharing one region-scoped value. +- verdict: CONFIRMED — providers/aws/ladder/layer_states.go:65 fetches coverage once through `fetchSPCoveragePct`, which passes `a.cfg.Region` at :156, and :69-70 hand that same pointer to both SP layers, while `fetchSPUtilizationPct` :183-186 substitutes `""` for Compute. +- issue: (pending cross-reference) + +### A07-021 Region display-name map is stale for the renamed EU regions, and the fallback branch is dead +- category: correctness +- severity: medium +- location: providers/aws/recommendations/converters.go:137 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: the map keys five regions under the old `"EU (...)"` display form while keying the newer ones under `"Europe (...)"`. AWS renamed those to `Europe (Ireland)`, `Europe (Frankfurt)`, `Europe (London)`, `Europe (Paris)` and `Europe (Stockholm)`. When CE returns the current name, `normalizeRegionName` falls through and returns the display string verbatim, so `rec.Region` becomes `"Europe (Ireland)"`. `filterByIncludedRegions` then drops that rec for an `--include-regions eu-west-1` filter, and any rec that survives carries a region string no service client can use. Separately, the `strings.HasPrefix` block below returns exactly the same value as the final statement, so it is dead code that reads as a guard. +- evidence: + ```go + if strings.HasPrefix(region, "us-") || strings.HasPrefix(region, "eu-") || + strings.HasPrefix(region, "ap-") || strings.HasPrefix(region, "sa-") || + strings.HasPrefix(region, "ca-") || strings.HasPrefix(region, "me-") || + strings.HasPrefix(region, "af-") || strings.HasPrefix(region, "il-") { + return region + } + return region + ``` +- suggested fix: add the `Europe (...)` aliases for the five renamed regions, delete the dead prefix branch, and log once when a display-form value falls through unmapped so future renames surface. +- verdict: CONFIRMED — the map at providers/aws/recommendations/converters.go:136-166 keys eu-west-1/2/3, eu-central-1 and eu-north-1 as `"EU (...)"` while eu-south-1, eu-south-2 and eu-central-2 use `"Europe (...)"` in the same literal, so the two spellings are inconsistent within one map and the `Europe (...)` form for the five older regions falls through; the prefix block at :173-179 returns `region`, identical to the `return region` immediately below it, so it is dead as claimed. +- issue: (pending cross-reference) + +### A07-023 FindConvertibleOffering returns the first result without paginating or checking the payment option +- category: correctness +- severity: medium +- location: providers/aws/services/ec2/client.go:794 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: this function uses the `Filters[]`-heavy request shape that the comment on `findOfferingID` (line 480) says AWS answers with empty pages plus a `NextToken` on sparse offering sets — and it does not paginate at all, so a first empty page yields "no convertible offering found" for an instance type that has one. It also caps at `MaxResults: 20`, never sets `OfferingType`, and returns `ReservedInstancesOfferings[0]`, so the offering it hands back for an exchange target may carry a different payment option than the source RI. +- evidence: + ```go + input := &ec2.DescribeReservedInstancesOfferingsInput{ + Filters: filters, + IncludeMarketplace: aws.Bool(false), + MaxResults: aws.Int32(20), + } + result, err := c.client.DescribeReservedInstancesOfferings(ctx, input) + ``` +- suggested fix: build the request from typed fields the way `describeInputFromQuery` does, paginate with the existing `isLastEC2Page` helper and page cap, and match the payment option before returning. +- verdict: CONFIRMED — providers/aws/services/ec2/client.go:794-818 builds a six-entry `Filters[]` request with `MaxResults: 20`, no `OfferingType`, a single `DescribeReservedInstancesOfferings` call and `ReservedInstancesOfferings[0]`; the comment at :481-487 documents that this exact request shape returns empty pages plus a `NextToken`, and the function has live callers at internal/server/ladder_write.go:64 and internal/server/handler_ri_exchange.go:63. +- issue: (pending cross-reference) + +### A08b-015 GCP billing SKU listing never follows the catalog page token +- category: correctness +- severity: high +- location: providers/gcp/services/cloudsql/client.go:106 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd (same at `cloudstorage/client.go:148`, `memorystore/client.go:118`, `computeengine/client.go:331`) +- failure scenario: `Services.Skus.List(serviceID).Do()` returns one page (default 50 SKUs) and a `NextPageToken` that is discarded; the `BillingService` interface has no page-token parameter, so the omission is not fixable at the call site. The Cloud SQL catalog alone has several hundred SKUs, so the tier being priced is almost never in the first page and `getSQLPricing` returns "no pricing found" — which `fillSQLPricing` then swallows (A08b-016). `/usr/bin/grep -rn "PageToken" providers/gcp/` returns no matches at this commit. +- evidence: + ```go + func (r *realBillingService) ListSKUs(serviceID string) (*cloudbilling.ListSkusResponse, error) { + return r.service.Services.Skus.List(serviceID).Do() + } + ``` +- suggested fix: widen the interface to accept a page token (or return an iterator) and loop until `NextPageToken` is empty, with an explicit cap that errors rather than truncating. +- verdict: PLAUSIBLE — the dropped token is real (cloudsql:105-107, cloudstorage:147-149, memorystore:117-119, computeengine:330-332, and `/usr/bin/grep -rn PageToken providers/gcp/` returns nothing), but the "default 50 SKUs" premise is wrong: `services.skus.list` defaults to 5000 items per page (google.golang.org/api@v0.274.0 cloudbilling/v1, `ServicesSkusListCall.PageSize` doc), so whether any given service catalog overflows one page is a runtime fact I could not establish from source. +- severity-adjusted: medium — silent truncation is certain, a guaranteed miss for Cloud SQL is not. +- issue: (pending cross-reference) + +### A08b-018 Cosmos DB `ValidateOffering` compares reservation SKUs against account capability names, so it can never pass +- category: correctness +- severity: high +- location: providers/azure/services/cosmosdb/client.go:510 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: `GetValidResourceTypes` returns either capability names harvested from existing accounts (`collectCapabilitiesFromAccounts` returns `capability.Name` values such as `EnableMongo`) or the hardcoded `getCommonSKUs` list of the same shape. A Cosmos reservation SKU is a throughput tier — `detailsFromCosmosSKU` at line 780 documents the real form as `"100RU"`. `ValidateOffering` does `strings.EqualFold("100RU", "EnableMongo")` for every entry and always returns `invalid Azure Cosmos DB SKU`, so any pre-purchase validation gate blocks every legitimate Cosmos recommendation. +- evidence: + ```go + func (c *CosmosDBClient) getCommonSKUs() []string { + return []string{ + "EnableCassandra", + "EnableMongo", + "EnableGremlin", + "EnableTable", + "EnableServerless", + } + } + ``` +- suggested fix: validate against the throughput-SKU grammar the purchase body actually sends, or drop the check rather than shipping a validator that cannot succeed. +- verdict: CONFIRMED — `GetValidResourceTypes` returns either `capability.Name` values (cosmosdb:456-498) or the `EnableMongo`-shaped `getCommonSKUs` list (510-519), and `ValidateOffering` (359-372) `EqualFold`s those against `rec.ResourceType`, which the converter takes from the reservation SKU and `detailsFromCosmosSKU` (770-797) documents as `"100RU"`. The two vocabularies cannot intersect. +- severity-adjusted: medium — `ValidateOffering` and `GetValidResourceTypes` have no caller in this repo outside tests; only pkg/provider/interface.go:49 and :53 name them, so no live gate is blocked today. +- issue: (pending cross-reference) + +### A08b-019 `maxMachineTypeItems = 20` makes GCP `GetValidResourceTypes` error in every real region +- category: correctness +- severity: high +- location: providers/gcp/services/computeengine/client.go:37 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd (consumer at `client.go:920`) +- failure scenario: the constant caps a per-item iteration (one machine type per `Next()`), and the consumer returns a hard error rather than stopping: `return nil, fmt.Errorf("computeengine: GetValidResourceTypes iteration cap (%d items) reached", maxMachineTypeItems)`. A GCP zone publishes well over a hundred machine types, so the 21st item always fires the cap and `GetValidResourceTypes` never returns a list — every Compute Engine CUD `ValidateOffering` fails with `failed to get valid tiers`. +- evidence: + ```go + // maxMachineTypeItems caps GCP machine types iteration (one item per Next() call). + const maxMachineTypeItems = 20 + ``` +- suggested fix: raise the cap to a value above the real machine-type count for a zone (or page properly with a filter), and keep the hard error only as a runaway guard. +- verdict: CONFIRMED — the constant and its "one item per Next() call" comment are at computeengine:36-37, and the consumer increments `itemIdx` once per `it.Next()` and returns a hard error at 920-922, so the 21st machine type aborts the call. One wording correction: `ValidateOffering` wraps it as "failed to get valid machine types" (839-842), not "failed to get valid tiers". +- severity-adjusted: medium — `GetValidResourceTypes`/`ValidateOffering` have no caller in this repo outside tests (pkg/provider/interface.go:49,53). +- issue: (pending cross-reference) + +### A08b-020 GCP iteration caps are per-item, named "Pages", and abort the whole call +- category: correctness +- severity: medium +- location: providers/gcp/services/cloudsql/client.go:181 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd (same at `cloudstorage/client.go:201`, `memorystore/client.go:194`, `computeengine/client.go:406` and `client.go:478`) +- failure scenario: each loop increments `pageIdx` once per `it.Next()`, so `maxRecsPages = 20` means twenty recommendations, not twenty pages, and non-ACTIVE recommendations consume the budget before being skipped. A project with 21 Recommender rows returns zero recommendations and an error. `maxCommitmentsPages = 50` does the same for existing CUDs, so a project holding 51 commitments fails coverage collection entirely. +- evidence: + ```go + for pageIdx := 0; ; pageIdx++ { + if pageIdx >= maxRecsPages { + return nil, fmt.Errorf("cloudsql: GetRecommendations iteration cap (%d items) reached", maxRecsPages) + } + ``` +- suggested fix: rename the constants to reflect items and raise them to values a real account cannot legitimately exceed, so the guard fires only on a runaway iterator. +- verdict: CONFIRMED — `pageIdx` advances once per `it.Next()` at cloudsql:177-183, computeengine:402-408 and computeengine:474-480; the non-ACTIVE skip happens after the increment (cloudsql:197-199) so it consumes budget; the constants are `maxRecsPages = 20` (cloudsql:22, cloudstorage:23, memorystore:24, computeengine:31) and `maxCommitmentsPages = 50` (computeengine:34), and hitting either returns an error rather than stopping. +- issue: (pending cross-reference) + +### A08b-025 Reservation classification by SKU-name substring cross-assigns reservations between services +- category: correctness +- severity: medium +- location: providers/azure/services/database/client.go:278 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd (same shape at `cache/client.go:247`, `cosmosdb/client.go:248`, `search/client.go:193`, `synapse/client.go:216`) +- failure scenario: each client claims any reservation whose lowercased SKU name contains its own keyword. Azure SQL Data Warehouse / Synapse SKUs are `SqlDW`-family names that contain "sql", so the database client claims every Synapse reservation as `ServiceRelationalDB` while the synapse client also claims it via its `"dw"` prefix — the same reservation is reported under two services with two different `ServiceType` values. The synapse `"dw"` prefix test is likewise unanchored to a family. +- evidence: + ```go + props := detail.Properties + if props.SKUName == nil || !strings.Contains(strings.ToLower(*props.SKUName), "sql") { + return nil + } + ``` +- suggested fix: classify on the reservation's declared reserved-resource type from an inventory API rather than substring-matching the SKU string. +- verdict: PLAUSIBLE — the unanchored substring tests are exactly as described (database:278 on "sql", cache:247 on "redis", cosmosdb:248 on "cosmos", search:193 on "search", synapse:217-219 on a `dw`/`scu` prefix or a "synapse" substring), but the specific database/synapse overlap is not demonstrable from source: a Synapse `SKUName` of the `DW1000c` form matches synapse's `dw` prefix and does not contain "sql", so which client double-claims depends on the live SKU string Azure returns. +- issue: (pending cross-reference) + +### A08b-027 Six pager loops have neither a page cap nor a context check, unlike every sibling loop in the same files +- category: correctness +- severity: medium +- location: providers/azure/services/cache/client.go:697 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd (same at `cosmosdb/client.go:691`, `database/client.go:826` and `client.go:870`, `managedredis/client.go:182`, `synapse/client.go:142` and `client.go:189`, `compute/exchange.go:140`) +- failure scenario: `collectRedisReservations` in the same file checks `ctx.Err()` and enforces `maxReservationsPages`; `fetchSKUCatalogue` does neither. A cancelled request keeps issuing `NextPage` calls until the SDK's own transport notices, and a server returning a non-terminating page chain spins forever. `compute/exchange.go:140` is the tenant-wide reservation listing that feeds the exchange ownership gate, so an unbounded walk there stalls an authorization check. +- evidence: + ```go + out := make(map[string]redisSKUEntry) + for pager.More() { + page, err := pager.NextPage(ctx) + if err != nil { + logging.Warnf("azure cache: SKU catalogue page fetch failed for region %s: %v ...", c.region, err, len(out)) + return nil + } + ``` +- suggested fix: apply the same `pageIdx >= maxPages` cap and `ctx.Err()` precheck the sibling loops in each file already use. +- verdict: CONFIRMED — all eight cited loops are bare `for pager.More()` with no index and no precheck (cache:697, cosmosdb:691, database:826 and 870, managedredis:182, synapse:142 and 189, compute/exchange.go:140), while `collectRedisReservations` in the same cache file enforces both at 217-224. Narrowing on the cancellation half: database:829 and 873 do test `ctx.Err()` after a page error, and every one of these loops returns on any page error, so the non-terminating scenario needs a server emitting an endless chain of successful pages. +- issue: (pending cross-reference) + +### A09-010 findBestFit can emit a target instance type that does not exist for the family +- category: correctness +- severity: medium +- location: pkg/exchange/reshape.go:694 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: the loop keeps the LAST index in `sizeOrder` whose factor fits, and `"metal"` is the last entry with the same factor (192) as `"24xlarge"`. An m5 convertible RI with `normalizedUsed = 200` picks `bestIdx` at `"metal"` and `analyzeRI` composes `family + "." + targetSize` = `"m5.metal"`, count 2. The same path produces `"t3.9xlarge"`, `"r5.3xlarge"` and other size/family pairs AWS does not offer. `resolveOffering` in auto.go then fails the offering lookup and the recommendation is silently skipped, while the reshape dashboard shows a target no operator can act on. +- evidence: + ```go + bestIdx := -1 + for i, s := range sizeOrder { + nf := normalizationFactors[s] + if nf > 0 && nf <= normalizedUsed { + bestIdx = i + } + } + ``` +- suggested fix: drop `"metal"` from `sizeOrder` (keep it in `normalizationFactors` for parsing existing RIs) and prefer the smaller-indexed size on a factor tie, so the suggested target is always a standard size. +- verdict: CONFIRMED — `sizeOrder` ends with `"metal"` (pkg/exchange/reshape.go:382-387) and `normalizationFactors["metal"] == normalizationFactors["24xlarge"] == 192` (reshape.go:378-379), so the loop's last-wins assignment at reshape.go:694-699 selects index 17 ("metal") for any `normalizedUsed >= 192`; `analyzeRI` then composes `family + "." + targetSize` at reshape.go:486 with no per-family existence check. +- issue: (pending cross-reference) + +### A09-011 The "metal" normalization factor is a single hardcoded value for a family-dependent quantity +- category: correctness +- severity: medium +- location: pkg/exchange/reshape.go:379 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: AWS metal sizes map to different normalization factors per family: `m5.metal` is 24xlarge-equivalent (192) but `i3.metal` is 16xlarge-equivalent (128) and `m5zn.metal` is 12xlarge-equivalent (96). Any `.metal` source RI with an unpopulated `RIInfo.NormalizationFactor` falls through `resolveNormFactor` to this table and is credited with 192 units. For an `i3.metal` at 50% utilization the computed `normalizedUsed` is 96 instead of 64, and `findBestFit` sizes the exchange target 50% too large. +- evidence: + ```go + "24xlarge": 192, + "metal": 192, + } + ``` +- suggested fix: remove the `"metal"` entry so `resolveNormFactor` returns 0 and `analyzeRI` skips the RI (fail loud) unless the caller supplied the real per-family factor from `ec2:DescribeInstanceTypes`. +- verdict: CONFIRMED — `normalizationFactors["metal"] = 192` is a single family-independent entry (pkg/exchange/reshape.go:379), and `resolveNormFactor` falls back to that table whenever `ri.NormalizationFactor == 0` (reshape.go:432-437), feeding `normalizedPurchased` at reshape.go:461 and `findBestFit` at :481 with no family discrimination. +- issue: (pending cross-reference) + +### A10-006 CSV numeric fields are parsed with `fmt.Sscanf`, which truncates and accepts trailing garbage +- category: correctness +- severity: medium +- location: cmd/multi_service_csv.go:173 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: A row whose `Count` cell reads `3.7` parses as `3` with `err == nil` (`Sscanf` stops at the `.` and reports one successful conversion). `12 units` parses as `12`. `-5` parses as `-5` and flows into `savingsPerInstance`, `ApplyInstanceLimit` and the purchase loop as a negative count. `EstimatedSavings` of `1000 USD` silently becomes `1000`. Every one of these is a money quantity read from an operator-editable file with no boundary rejection, contrary to the project's strict-integer-parsing rule. +- evidence: + ```go + func parseCSVInt(record []string, colIdx map[string]int, fieldName string, target *int) error { + value := getCSVField(record, colIdx, fieldName) + if value == "" { return nil } + if _, err := fmt.Sscanf(value, "%d", target); err != nil { + return fmt.Errorf("invalid %s value '%s': %w", fieldName, value, err) + } + return nil + } + ``` +- suggested fix: Use `strconv.Atoi` / `strconv.ParseFloat` on the whole trimmed cell, which reject trailing characters, and add an explicit `< 0` rejection for `Count`. +- verdict: CONFIRMED — ran the parses against the pinned toolchain: `Sscanf("%d")` returns err=nil with 3 for "3.7", 12 for "12 units" and -5 for "-5", and `Sscanf("%f")` returns 1000 for "1000 USD", so cmd/multi_service_csv.go:167-190 admits every value the finding names. +- issue: (pending cross-reference) + +### A10-011 The CSV TOTAL row sums normalized units for rows whose own NU cell is blank +- category: correctness +- severity: medium +- location: cmd/multi_service_csv.go:314 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: A mixed `--all-services` run includes ElastiCache recs with `ResourceType=cache.r7g.large`. `formatNormalizedUnitsOrBlank` blanks the per-row NormalizedUnits cell because the service is not RDS, but `buildTotalRow` calls `RDSInstanceNUFromType` unconditionally, and that function only requires three dot-separated parts before looking up the size suffix (`providers/aws/recommendations/family_nu.go:62-67`), so `cache.r7g.large` resolves to the `large` NU value. The TOTAL cell therefore exceeds the sum of the visible cells above it, and an operator reconciling family-NU bundling by hand cannot make the column add up. +- evidence: + ```go + for i := range results { + r := results[i] + totalCount += r.Recommendation.Count + totalNU += float64(r.Recommendation.Count) * recommendations.RDSInstanceNUFromType(r.Recommendation.ResourceType) + ``` +- suggested fix: Gate the `totalNU` accumulation on the same `rec.Service != ServiceRDS && != ServiceRelationalDB` test `formatNormalizedUnitsOrBlank` uses, ideally by summing the per-row helper's own value. +- verdict: CONFIRMED — buildTotalRow calls RDSInstanceNUFromType for every row (cmd/multi_service_csv.go:314) while formatNormalizedUnitsOrBlank blanks non-RDS rows (cmd/multi_service_csv.go:370-373), and "cache.r7g.large" splits into three parts so the lookup returns the "large" entry, 4 (providers/aws/recommendations/family_nu.go:61-68, map at :23-38). +- issue: (pending cross-reference) + +### A10-014 The dry-run footer tells operators Savings Plans purchasing is not implemented, but it is +- category: correctness +- severity: medium +- location: cmd/multi_service_stats.go:184 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: A dry run over `--all-services` prints "To actually purchase these RIs, run with --purchase flag / Note: Savings Plans purchasing not yet implemented". An operator who read that line reruns with `--purchase` expecting SP rows to be skipped. They are not: `createServiceClient` returns a real `savingsplans.NewClient` for all four SP slugs (cmd/main.go:257-265) and `savingsplans.Client.PurchaseCommitment` (providers/aws/services/savingsplans/client.go:210) issues a real `CreateSavingsPlan`. The message actively misleads on a money path. +- evidence: + ```go + if isDryRun { + AppLogger.Println("\n💡 To actually purchase these RIs, run with --purchase flag") + AppLogger.Println(" Note: Savings Plans purchasing not yet implemented") + } + ``` +- suggested fix: Delete the stale second line, or replace it with the count of SP recommendations that `--purchase` will buy. +- verdict: CONFIRMED — printFinalMessage emits the line on every dry run (cmd/multi_service_stats.go:181-185), while createServiceClient returns a real savingsplans client for all four SP slugs (cmd/main.go:258-266) and PurchaseCommitment issues CreateSavingsPlan (providers/aws/services/savingsplans/client.go:210-235). +- issue: (pending cross-reference) + +### A10-017 Recommendations with no resolvable account name are dropped whenever any account filter is set +- category: correctness +- severity: medium +- location: cmd/multi_service_filters.go:158 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: A rec whose `Account` field is empty (Savings Plans rows and any provider that does not populate it — `populateAccountNames` only sets `AccountName` when `Account != ""`, cmd/multi_service_helpers.go:190) gets `AccountName == ""`. With `--exclude-accounts sandbox` set, `shouldIncludeAccount` returns false for that rec because the exclude list is non-empty, so it is dropped even though it does not match `sandbox` at all. `processRecommendation` classifies this as an expected dimension mismatch and records no drop reason (cmd/multi_service_filters.go:63-69), so the end-of-run drop summary never mentions it and the operator sees an unexplained shortfall. +- evidence: + ```go + if accountName == "" { + return len(cfg.IncludeAccounts) == 0 && len(cfg.ExcludeAccounts) == 0 + } + ``` +- suggested fix: With only an exclude list in force, an unattributed rec cannot match it and should be kept. Keep the drop for an include list, but surface it with a dedicated drop reason so the count appears in the summary. +- verdict: CONFIRMED — cmd/multi_service_filters.go:158-160 returns false as soon as either list is non-empty, populateAccountNames leaves AccountName empty when Account is empty (cmd/multi_service_helpers.go:188-193), and processRecommendation returns an empty dropReason on the dimension-filter branch (cmd/multi_service_filters.go:63-69) so the drop never reaches the summary. +- issue: (pending cross-reference) + +### A10-024 The CSV purchase path confirms per region rather than once for the run +- category: correctness +- severity: medium +- location: cmd/multi_service.go:693 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: `processPurchaseLoop` prompts only when `j == 0`, and it is called once per (service, region) from `runToolFromCSV`. A CSV spanning RDS in three regions and EC2 in two therefore prompts five separate times, and each prompt reports only that region's totals ("About to purchase 4 instances with estimated monthly savings: $210"). The operator never sees the run-wide instance count or dollar figure before the first purchase fires, and answering "no" at prompt four does not undo the twelve instances already bought under prompts one through three. The non-CSV path gets this right, confirming once against `sumPassedRecs` over the whole run (cmd/multi_service.go:157). +- evidence: + ```go + if j == 0 { + totalInstances := CalculateTotalInstances(recs) // this region only + ... + if !ConfirmPurchase(totalInstances, totalSavings, cfg.SkipConfirmation) { + return createCancelledResults(recs, region, cfg) + } + } + ``` +- suggested fix: Hoist the confirmation into `runToolFromCSV` ahead of the service/region loop, summing over the full post-dedup set, and drop the per-region prompt. +- verdict: CONFIRMED — processPurchaseLoop prompts under `if j == 0` over this region's recs only (cmd/multi_service.go:693-705) and runToolFromCSV calls it once per (service, region) inside the nested loops at :521-564, while the non-CSV path confirms once against sumPassedRecs for the whole run (:156-163). +- issue: (pending cross-reference) + +### A11-012 The synthetic "current user" row exposes Edit and Delete bound to a fake id +- category: correctness +- severity: medium +- location: frontend/src/users/userActions.ts:55 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: when the users list omits the signed-in user, `withCurrentUser` prepends a row with `id: 'current'` and a hardcoded `mfa_enabled: false`. `renderUsers` gives every row an Edit and a Delete button keyed on that id (userList.ts:122-123), so clicking Edit issues `GET /api/users/current` and Delete issues `DELETE /api/users/current`, neither of which addresses a real user. The row is also selectable, so a bulk delete includes `current` and fails the whole `Promise.all`. Separately the hardcoded `mfa_enabled: false` is counted by the "MFA Enabled" stat card (userList.ts:40), understating the count for an admin who has MFA on. +- evidence: + ```typescript + // users/userActions.ts:59-65 + const synthetic: api.APIUser = { + id: 'current', + email: current.email, + groups: Array.isArray(current.groups) ? current.groups : [], + mfa_enabled: false, + }; + return [synthetic, ...users]; + ``` +- suggested fix: mark the synthetic row read-only (skip the Edit/Delete/select controls when `user.id === 'current'`) and carry the real `mfa_enabled` from the session user, or drop the synthetic row entirely now that the list endpoint returns the caller. +- verdict: PLAUSIBLE — the row really does get the full control set, since `renderUsers` keys Edit and Delete on `user.id` with no `current` exemption (frontend/src/users/userList.ts:121-122, whose only special case is the "You" badge at userList.ts:105) and the hardcoded `mfa_enabled: false` feeds the stat card at userList.ts:41. Reaching it needs `listUsers` to omit the caller, which I could not establish: `GET /api/users` returns the unscoped list (internal/api/handler_users.go:17-27), so the `u.email === current.email` test at userActions.ts:57 normally matches and the synthetic row is never prepended; an email-case mismatch or an API-key session would be the trigger. +- issue: (pending cross-reference) + +### A11-013 A partly-failed bulk user delete leaves deleted users on screen and names none of the failures +- category: correctness +- severity: medium +- location: frontend/src/users/userActions.ts:173 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: `Promise.all` rejects on the first failing delete. The `catch` shows "Failed to delete some users" and neither clears the selection nor reloads the list, so users that were successfully deleted remain rendered and selected until a manual refresh; a second click on Bulk Delete then re-issues deletes for rows that no longer exist. The operator is never told which users survived. `handleFanOutExecute` in app.ts:545 already demonstrates the `allSettled` + per-item reporting pattern this path needs. +- evidence: + ```typescript + // users/userActions.ts:172-183 + await Promise.all( + Array.from(selectedUserIds).map(userId => api.deleteUser(userId)) + ); + clearSelectedUserIds(); + await loadUsers(); + ``` +- suggested fix: use `Promise.allSettled`, deselect only the ids that resolved, reload the list in both branches, and name the failures in the error toast. +- verdict: CONFIRMED — `Promise.all` short-circuits on the first rejection and both `clearSelectedUserIds()` and `loadUsers()` sit inside the `try` after it (frontend/src/users/userActions.ts:172-178), so the catch at userActions.ts:179-182 shows one generic string and leaves the stale, still-selected rows on screen; deletes already issued are not undone, so a second Bulk Delete re-issues them. +- issue: (pending cross-reference) + +### A11-014 A multi-account savings filter is silently narrowed to the first account +- category: correctness +- severity: medium +- location: frontend/src/api/history.ts:45 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: when the topbar filter selects three accounts, `getSavingsAnalytics` sends both `account_ids=a,b,c` and `account_id=a`. The comment states the handler honours only the singular form, so the chart renders account A's savings while the UI chip row says three accounts are selected. The user reads an under-reported savings figure with no indication that two accounts were dropped. The sibling `getHistory` (history.ts:23) sends only the plural form, so the two panels on the same page disagree for the same filter. +- evidence: + ```typescript + // api/history.ts:39-46 + if (filters.account_ids && filters.account_ids.length > 0) { + params.set('account_ids', filters.account_ids.join(',')); + if (filters.account_ids[0]) params.set('account_id', filters.account_ids[0]); + } + ``` +- suggested fix: have the caller surface the limitation (disable or annotate the chart when more than one account is selected) rather than quietly charting a subset. +- verdict: CONFIRMED — the analytics handlers read only the singular param (`accountID := params["account_id"]` at internal/api/handler_analytics.go:40, 131, 197, then `resolveSingleAccountFilterIDs`) and `account_ids` is parsed nowhere under internal/api for these routes, so the extra plural param at frontend/src/api/history.ts:43 is inert and the chart is scoped to the first account; `getHistory` (history.ts:22) sends only the plural form, which internal/api/handler_history.go:678 does honour, so the two panels genuinely disagree. +- issue: (pending cross-reference) + +### A12-013 The savings-analytics summary keys the frontend reads do not exist on the wire +- category: correctness +- severity: medium +- location: frontend/src/modules/savings-history.ts:293 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: `GET /history/analytics` returns `Summary *HistorySummary` (`internal/api/handler_analytics.go:99`), whose JSON keys are `total_purchases`, `total_upfront`, `total_monthly_savings`, `total_annual_savings`. The frontend's `SavingsAnalyticsSummary` (`frontend/src/api/types.ts:650`) declares `total_period_savings`, `average_savings_per_period` and `peak_savings`, none of which are ever sent, so all three `??` fallbacks are taken on every response and the declared contract is a fiction. The Go struct that does carry those keys, `HistorySummaryAnalytics` (`internal/api/types.go:112`), is never populated anywhere — it is dead. The numbers happen to agree today only because the client-side sum reproduces `TotalMonthlySavings`; a change on either side breaks silently rather than loudly. +- evidence: + ```ts + const monthlyTotal = summary?.total_period_savings ?? totalSavings; + const monthlyAvg = summary?.average_savings_per_period ?? avgPerPeriod; + const monthlyPeak = summary?.peak_savings ?? peakSavings; + ``` +- suggested fix: Align the TypeScript interface with `HistorySummary`'s actual keys and delete the unused `HistorySummaryAnalytics` struct, or populate it and have the handler return it. +- verdict: CONFIRMED — The handler returns Summary *HistorySummary (internal/api/handler_analytics.go:99) whose JSON keys are total_purchases/total_upfront/total_monthly_savings/total_annual_savings (internal/api/types.go:910-929), and HistorySummaryAnalytics (internal/api/types.go:112) is never constructed anywhere, so all three keys read at frontend/src/modules/savings-history.ts:293-295 are absent on every response. +- issue: (pending cross-reference) + +### A12-017 Cloning the ladder form after populating it resets its three `` (line 1561); the free-text fallback the copy describes no longer exists. Opened from the Convertible RIs table there are also no Cost Explorer alternatives, so the select contains nothing but the placeholder and the user is at a dead end following misleading instructions. +- evidence: + ```ts + if (offeringsError) { + placeholder.textContent = 'Could not load offerings -- type a UUID'; + ``` +- suggested fix: Change the copy to say the offerings failed to load and offer a retry, or restore a visible text field for the manual path. +- verdict: CONFIRMED — The placeholder tells the user to type a UUID (frontend/src/riexchange.ts:1490-1491) but the only UUID-accepting field is `offeringInput`, created as `type = 'hidden'` (frontend/src/riexchange.ts:1559-1561), and the visible control is a select whose only other option group is the CE alternatives. +- issue: (pending cross-reference) + +### A12-067 `escapeHtml` output is assigned to `textContent`, double-escaping option labels +- category: correctness +- severity: low +- location: frontend/src/riexchange.ts:1507 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: `option.textContent` escapes on its own, so pre-escaping means any `&`, `<` or quote in an AWS `instance_type` or `offering_type` renders literally as `&` in the dropdown. Line 1525 does the same for the Cost Explorer alternatives. It also misleads a reader into treating the option list as an HTML sink. +- evidence: + ```ts + const label = escapeHtml(o.instance_type) + (o.offering_type ? ' -- ' + escapeHtml(o.offering_type) : ''); + opt.textContent = label; + ``` +- suggested fix: Assign the raw strings to `textContent` and drop the `escapeHtml` calls at 1507 and 1525. +- verdict: CONFIRMED — escapeHtml output is assigned to option.textContent, which escapes again (frontend/src/riexchange.ts:1507-1508 and :1525), so an ampersand or angle bracket in an instance_type or offering_type renders as its entity in the dropdown. +- issue: (pending cross-reference) + +### A12-069 A failed exchange's `error` field is never surfaced in the history table +- category: correctness +- severity: low +- location: frontend/src/riexchange.ts:2211 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: When an exchange fails at the AWS accept step, `RIExchangeHistoryRecord.error` carries the reason but the row renders only a "failed" status badge. The operator has no in-product way to see why money was not spent, or whether it partially was, and must read backend logs. +- evidence: + ```ts + + '' + escapeHtml(rec.status) + '' + ``` +- suggested fix: Put the escaped `rec.error` in the status cell's `title`, or render it in a detail row for failed statuses. +- verdict: CONFIRMED — RIExchangeHistoryRecord declares `error?: string` (frontend/src/api/types.ts:844) but the status cell renders only the badge and no cell or title carries it (frontend/src/riexchange.ts:2216). +- issue: (pending cross-reference) + +### A12-073 The "YTD Savings" tile's sparkline plots the trend-range window, not year-to-date +- category: correctness +- severity: low +- location: frontend/src/dashboard.ts:1057 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: The tile's value comes from `data.ytd_savings`, but the sparkline drawn beneath it comes from `loadSavingsTrendChart`'s data points, whose window is whatever the trend range toggle is set to (7, 30 or 90 days, or a rolling 365). Clicking the "7" range button changes the shape of the line under a label that says YTD, so the number and the line describe different periods. +- evidence: + ```ts + attachSparkline('ytd', points.map((p: SavingsDataPoint) => p.cumulative_savings || 0)); + ``` +- suggested fix: Fetch a separate year-to-date series for the tile, or relabel the sparkline so it is not read as YTD. +- verdict: CONFIRMED — The tile value is data.ytd_savings (frontend/src/dashboard.ts:308) while the sparkline plots the trend-range data points whose window comes from savingsTrendRange (frontend/src/dashboard.ts:1022-1031, :1057), which the range buttons mutate at :1174-1177. +- issue: (pending cross-reference) + +### A13-019 The `policy_arns` output claims to list all attached policies and lists three of five +- category: correctness +- severity: low +- location: terraform/environments/aws/ci-cd-permissions/outputs.tf:11 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: `role.tf:158-181` attaches five managed policies (`networking`, `compute`, `compute_b`, `data`, `iam`). The output describes itself as "ARNs of all attached managed policies" but omits `compute_b` and `iam`. An operator scripting a teardown or an audit from `terraform output -json policy_arns` silently misses the two policies that carry the whole #1705 boundary-delegation half, leaving them orphaned or unreviewed. +- evidence: + ```hcl + description = "ARNs of all attached managed policies" + value = { + networking = aws_iam_policy.networking.arn + compute = aws_iam_policy.compute.arn + data = aws_iam_policy.data.arn + } + ``` +- suggested fix: Add `compute_b` and `iam` (and `workload_boundary`, if the boundary belongs in the same inventory). +- verdict: CONFIRMED — role.tf:158-181 declares five `aws_iam_role_policy_attachment` resources (networking, compute, compute_b, data, iam) against `aws_iam_role.cudly_deploy`, while outputs.tf:11-18 describes itself as "ARNs of all attached managed policies" and emits only networking, compute and data. +- issue: (pending cross-reference) + +### A13c-016 AWS profile examples pin a full minor engine version the module documents as an incident cause +- category: correctness +- severity: low +- location: terraform/profiles/aws/prod.tfvars.example:33 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: `modules/database/aws/variables.tf:58` documents that pinning a full minor + (`"16.6"`) "causes downgrade failures if RDS auto-upgrades beyond it (incident: 2026-07-16, + #1372)" and defaults to the major-only `"16"`. All three `profiles/aws/*.tfvars.example` files + (dev:33, fargate-dev:32, prod:33) publish `"16.6"`, so anyone starting from a profile inherits + the exact configuration the incident was about. `environments/aws/dev.tfvars.example:50` + correctly uses `"16"`, which shows the fix landed in one place and not the other. +- evidence: + ```hcl + database_engine_version = "16.6" + ``` +- suggested fix: change all three profile examples to `"16"`. +- verdict: CONFIRMED — the incident text and the major-only default are at + database/aws/variables.tf:57-61, and all three profile examples publish `"16.6"` + (dev:33, fargate-dev:32, prod:33) while every file under `environments/aws/` uses `"16"` + (dev.tfvars.example:50, github-dev:49, github-prod:47, github-staging:46). +- issue: (pending cross-reference) + +### A13c-018 `database/aws` output guards on a different condition than the data source it indexes +- category: correctness +- severity: low +- location: terraform/modules/database/aws/outputs.tf:39 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: `data.aws_secretsmanager_secret.existing_password` is created when + `var.create_password` is false (main.tf:53), but this output selects between the two branches on + `var.master_password_secret_arn != null`. With `create_password = true` and an ARN also + supplied, the output indexes `[0]` of a zero-count data source and the plan fails with an + invalid-index error. The mirrored case — `create_password = false` with a null ARN — passes + `arn = null` to the data source and fails there instead. The in-tree caller + (`environments/aws/database.tf:15-16`) happens to set the one combination that works, so neither + path is exercised. +- evidence: + ```hcl + output "password_secret_name" { + value = var.master_password_secret_arn != null ? data.aws_secretsmanager_secret.existing_password[0].name : aws_secretsmanager_secret.db_password[0].name + } + ``` +- suggested fix: switch the ternary to `var.create_password` so both sides use the same predicate + as the resources, and add a variable precondition that the two inputs are mutually consistent. +- verdict: CONFIRMED — the data source's count is `var.create_password ? 0 : 1` + (database/aws/main.tf:53) while the output ternary keys off `var.master_password_secret_arn != + null` (outputs.tf:39); note `local.db_password_secret_arn` at main.tf:64 uses the correct + predicate, so the two outputs disagree with each other. The single in-tree caller sets + `create_password = false` with a non-null ARN (environments/aws/database.tf:15-16), the one + combination both branches survive. +- issue: (pending cross-reference) + +### A14-030 A failed health request is reported as a slow response +- category: correctness +- severity: low +- location: scripts/test-deployment.sh:266 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: `curl … || true` leaves `time_total` empty when the request fails outright. `echo " < 1.0" | bc -l` is then a syntax error, `|| echo 0` makes both comparisons false, and control falls to the `else`, printing `response-time: /health in s (very slow, >3s)` and counting a FAIL. An operator reading the report sees a latency problem where the endpoint was actually unreachable. +- evidence: + ```bash + time_total=$($CURL_BIN -s $CURL_K -o /dev/null -w "%{time_total}" --connect-timeout 10 "${URL}${health_path}" 2>/dev/null || true) + if (( $(echo "$time_total < 1.0" | bc -l 2>/dev/null || echo 0) )); then + ``` +- suggested fix: Fail with a distinct message when `time_total` is empty or non-numeric, before comparing it. +- verdict: PLAUSIBLE — scripts/test-deployment.sh:266-272 never checks curl's exit status, but the stated mechanism needs `time_total` to come back empty, which requires a missing curl binary or a malformed URL: measured, a refused connection still emits a number (0.002, logging PASS) and a 10s connect timeout emits ~10, logging the "very slow" FAIL the finding describes. +- issue: (pending cross-reference) + +### Category: concurrency + +12 findings: 2 high, 7 medium, 3 low. + +### A06-001 Advisory lock is released with the request context, so a canceled/expired scheduled run strands the lock on a pooled connection +- category: concurrency +- severity: high +- location: internal/server/handler.go:112 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: Cloud Scheduler POSTs `/api/scheduled/process_scheduled_purchases`; `handleScheduledHTTP` passes `r.Context()` straight through (internal/server/http.go:225). The scheduler's attempt deadline (180s by default) expires, the client disconnects, `r.Context()` is canceled, and the deferred `ReleaseAdvisoryLock(ctx, ...)` runs `SELECT pg_advisory_unlock($1)` on an already-canceled context. `internal/database/connection.go:357` logs a warning and `defer conn.Release()` returns the connection to the pool while the *session-level* advisory lock is still held. Every later invocation of that task type gets `pg_try_advisory_lock` = false from a different pooled connection and returns `{"status":"skipped","reason":"already_running"}` until that connection is recycled by the pool. The same happens on Lambda when a task runs past the function deadline. Note the file already knows this hazard: `releaseSkippedCollectionMarker` (handler.go:170) deliberately builds a detached `context.Background()` with its own timeout for exactly this reason. +- evidence: + ```go + acquired, err := locker.TryAdvisoryLock(ctx, lockID) + ... + defer locker.ReleaseAdvisoryLock(ctx, lockID) + } + return app.dispatchTask(ctx, taskType, params) + ``` +- suggested fix: release under a short detached context (`context.WithTimeout(context.Background(), 5*time.Second)`), mirroring `releaseSkippedCollectionMarker`. +- verdict: CONFIRMED — `handleScheduledHTTP` takes `ctx := r.Context()` and passes it straight to `HandleScheduledTask` (internal/server/http.go:225,249), whose deferred release runs on that same ctx (internal/server/handler.go:112); `ReleaseAdvisoryLock` only logs a warning when the unlock query fails and still runs `defer conn.Release()`, returning the still-locked session to the pool (internal/database/connection.go:354-361), while the sibling `releaseSkippedCollectionMarker` detaches deliberately (internal/server/handler.go:170). +- issue: (pending cross-reference) + +### A12-011 A pending AWS utilization response re-renders the shared container over the Azure/GCP table +- category: concurrency +- severity: high +- location: frontend/src/riexchange.ts:318 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: With the provider chip on AWS, `loadConvertibleRIs` starts `loadUtilization(gen)` against Cost Explorer. The user switches to Azure before it resolves. `loadExchangeableAzureRIs` resets `currentRIs = []` but never bumps `utilizationGeneration`, so the stale response passes the guard and `renderRIsTable` replaces the freshly rendered Azure reservations with the AWS empty state. `renderGCPEmptyStates` has the same hole. The convertible-RI list load itself (line 270) has no generation guard at all, so a slow AWS list can paint AWS rows, complete with Exchange buttons, into a panel the user believes is scoped to Azure. +- evidence: + ```ts + const utilization = await api.getRIUtilization(); + if (generation !== utilizationGeneration) return; + currentUtilization = new Map(utilization.map(u => [u.reserved_instance_id, u])); + const container = document.getElementById('ri-exchange-instances-list'); + if (container) renderRIsTable(container, currentRIAccountID); + ``` +- suggested fix: Bump `utilizationGeneration` in `loadExchangeableAzureRIs` and `renderGCPEmptyStates`, and apply the same generation check to the list load. +- verdict: CONFIRMED — utilizationGeneration is incremented only inside loadConvertibleRIs (frontend/src/riexchange.ts:273), so neither loadExchangeableAzureRIs (frontend/src/riexchange.ts:289-311) nor renderGCPEmptyStates (:120-136) invalidates an in-flight loadUtilization, whose guard therefore passes and calls renderRIsTable over the other provider's table (frontend/src/riexchange.ts:316-323); the list fetch at :270 has no guard at all. +- issue: (pending cross-reference) + +### A01-005 retryPurchase has no atomic claim on the failed row, so concurrent retries mint two successors and the full-row upsert clobbers the predecessor +- category: concurrency +- severity: medium +- location: internal/api/handler_purchases.go:1924 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: Two operators (or a double click) POST `/api/purchases/retry/{id}` for the same failed row. Both `loadAndValidateRetryRequest` reads see `RetryExecutionID == nil` (line 1708) and pass; both `persistRetryExecution` calls insert their own pending successor and then upsert `originalUpdated` (a stale in-memory copy of the failed row) with their own `retry_execution_id`. Result: two approvable successors with two approval emails, the already-retried guard is defeated, the first linkage is overwritten, and any column another writer changed on the failed row in between is reverted by the second upsert (`ON CONFLICT ... DO UPDATE SET` writes every column, store_postgres.go:991-1012). Only provider-side token dedupe stands between the two approvals and a double purchase. +- evidence: + ```go + originalUpdated := *failedExec + originalUpdated.RetryExecutionID = &newExecutionID + // ... + if err := h.config.WithTx(ctx, func(tx pgx.Tx) error { + if err := h.config.SavePurchaseExecutionTx(ctx, tx, newExecution); err != nil { + return err + } + if err := h.config.SavePurchaseExecutionTx(ctx, tx, &originalUpdated); err != nil { + return err + } + ``` +- suggested fix: Inside the tx, replace the full-row upsert of the predecessor with a conditional `UPDATE purchase_executions SET retry_execution_id=$2 WHERE execution_id=$1 AND status='failed' AND retry_execution_id IS NULL` and return 409 when zero rows are affected, before inserting the successor. +- verdict: CONFIRMED — the already-retried guard at internal/api/handler_purchases.go:1708 reads RetryExecutionID outside any tx, persistRetryExecution:1909-1937 upserts a stale value copy of the failed row, and the ON CONFLICT clause at internal/config/store_postgres.go:991-1012 rewrites status, retry_execution_id and every other mutable column unconditionally, so nothing serializes two concurrent retries. +- issue: (pending cross-reference) + +### A01-013 Pausing a `running` execution lets a claimed in-flight purchase be resumed and re-claimed +- category: concurrency +- severity: medium +- location: internal/api/handler_purchases.go:271 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: The scheduler's `claimAndExecute` has CAS-claimed a row to `running` and is mid-purchase. An operator clicks Pause (allowed from `running`), then Resume (`paused` -> `pending`). The next scheduler tick sees a pending row and `claimAndExecute` claims it again (pending is in its from-set), running the purchase a second time in parallel; the first run's terminal `SavePurchaseExecution` then overwrites whatever the second wrote. Only provider-side token dedupe prevents a second commitment (none for Azure savings plans). `running` appears in the pause from-set only to support the Run-now flow of A01-003, which itself never executes anything. +- evidence: + ```go + // Atomically transition to paused + if _, err := h.config.TransitionExecutionStatus(ctx, executionID, []string{"pending", "running"}, "paused", resolveCreatorUserID(session)); err != nil { + ``` +- suggested fix: Restrict pause to `{"pending"}` (and `notified` if desired); an execution that is actually running cannot be paused. +- verdict: CONFIRMED — pausePlannedPurchase (internal/api/handler_purchases.go:271) accepts from {pending,running} and resume:295 flips paused→pending; ProcessScheduledPurchases (internal/purchase/manager.go:715-742) re-selects due pending rows and claimAndExecute:179 claims from approved/pending/notified, while the first run's terminal SavePurchaseExecution (manager.go:234) is a full-row upsert; the double run additionally requires pause, resume and a scheduler tick all inside the first run's in-flight window, but every step on that path is unguarded. +- issue: (pending cross-reference) + +### A03-010 Failed-login bookkeeping does a whole-row read-modify-write from an unauthenticated path +- category: concurrency +- severity: medium +- location: internal/auth/service_user.go:767 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: `recordFailedLogin` writes the entire user row it loaded at the start of `Login` (`UpdateUser` sets all 18 columns, store_postgres.go:193-240). Anyone who knows the email can trigger it at will. Interleave it with a legitimate write: the victim changes password (`ChangePassword` writes the new hash), an in-flight failed login for the same email then writes its stale snapshot with the old hash, old MFA state or old `group_ids`, silently reverting the change; the same race lets an admin's deactivation or group removal be undone by the stale write. It also loses lockout increments (two concurrent failures both write `n+1`), so the 5-attempt lockout is soft under parallel guessing. +- evidence: + ```go + func (s *Service) recordFailedLogin(ctx context.Context, user *User) { + user.FailedLoginAttempts++ + now := time.Now() + user.UpdatedAt = now + ... + if err := s.store.UpdateUser(ctx, user); err != nil { + ``` +- suggested fix: Give the store a targeted atomic statement for this path (`UPDATE users SET failed_login_attempts = failed_login_attempts + 1, locked_until = CASE ... WHERE id = $1`) and use it from `recordFailedLogin` and `completeSuccessfulLogin` instead of the full-row write. +- verdict: CONFIRMED — recordFailedLogin (service_user.go:755-770) writes the *User loaded at Login:158 through PostgresStore.UpdateUser, which sets all 18 columns unconditionally (store_postgres.go:193-240, no version or WHERE guard), and it is reachable unauthenticated from any wrong-password attempt (service.go:230-232), so the lost-update and stale-overwrite interleavings need only two concurrent requests. +- issue: (pending cross-reference) + +### A05-006 Revocation-finalize retry loop sleeps without honouring the context +- category: concurrency +- severity: medium +- location: internal/purchase/finalize_revocations.go:62 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: `FinalizeInFlightRevocations` retries `MarkPurchaseRevoked` with `time.Sleep(2s)` then `time.Sleep(6s)`, and neither the retry loop nor the outer row loop looks at `ctx`. With 30 in-flight rows whose store writes are failing (the DB outage that produced the in-flight backlog in the first place), the sweep blocks for ~4 minutes past a canceled Lambda context, doing three doomed writes per row against a dead connection, instead of returning immediately. This is the exact bare-sleep-with-ctx-in-scope shape the project's own feedback memory records CodeRabbit catching repeatedly. +- evidence: + ```go + for attempt, backoff := range finalizeRevocationBackoffs { + if markErr == nil { + break + } + logging.Warnf("finalize_revocations: MarkPurchaseRevoked attempt %d for %s failed: %v (retrying in %s)", + attempt+1, record.PurchaseID, markErr, backoff) + time.Sleep(backoff) + markErr = m.config.MarkPurchaseRevoked(ctx, record.PurchaseID, now, "direct-api", "", nil, "") + } + ``` +- suggested fix: replace the sleep with `select { case <-time.After(backoff): case <-ctx.Done(): return result, ctx.Err() }` and add a `ctx.Err()` check at the top of the per-row loop so a canceled sweep stops instead of grinding through the remaining rows. +- verdict: CONFIRMED — finalize_revocations.go:53-73 is the whole loop: `ctx` appears only as the first argument to MarkPurchaseRevoked, the retry loop's only pause is the bare `time.Sleep(backoff)` at :62 over `finalizeRevocationBackoffs = {2s, 6s}` (:24-27), and neither the per-row `for _, record := range rows` at :54 nor the retry loop consults `ctx.Done()` / `ctx.Err()`, so 8s of doomed sleeping per row survives cancellation. +- issue: (pending cross-reference) + +### A06-004 DB rate-limiter cleanup worker is started with a request-scoped context and never ticks +- category: concurrency +- severity: medium +- location: internal/server/app.go:754 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: `reinitializeAfterConnect` runs inside `ensureDB`, which is called from `handleHTTPRequest` with `ctx, cancel := context.WithTimeout(r.Context(), 30*time.Second)` (internal/server/http.go:158) or from the Lambda invocation context. `StartCleanupWorker(ctx)` returns when `ctx.Done()` fires (internal/api/db_rate_limiter.go:69) and its ticker period is 10 minutes (`dbRateLimiterScheduledCleanupInterval`), so the goroutine is always canceled 30s (or one invocation) after the first DB connect and never runs a single cleanup. The documented 02-M2 mitigation — evicting perpetually-denied keys whose count never resets — is dead in production and `rate_limits` grows without bound under sustained abuse. Secondarily, a failure later in `reinitializeAfterConnect` (e.g. the `awsconfig.LoadDefaultConfig` at app.go:778) starts a fresh worker goroutine on every retry. +- evidence: + ```go + if app.appConfig.IsLambda { + dbRL := api.NewDBRateLimiter(dbConn.Pool()) + dbRL.StartCleanupWorker(ctx) + app.RateLimiter = dbRL + ``` +- suggested fix: start the worker with a process-lifetime context stored on `Application` (canceled in `Close`), not the caller's request context. +- verdict: CONFIRMED — `StartCleanupWorker(ctx)` receives the ctx `ensureDB` was called with (internal/server/app.go:750-755, reached from internal/server/app.go:633), which is the 30s request context built at internal/server/http.go:158 or the per-invocation Lambda context, while the ticker is 10 minutes (internal/api/db_rate_limiter.go:35,65), so the `select` at db_rate_limiter.go:68-73 always takes `ctx.Done()` before its first tick and `cleanup()` never runs. +- issue: (pending cross-reference) + +### A12-030 Two overlapping `loadHistory()` fetches have no sequence guard +- category: concurrency +- severity: medium +- location: frontend/src/history.ts:159 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: `setupHistoryHandlers` subscribes `loadHistory` to both the provider and the account slice. Changing the provider chip clears the accounts and sets the provider synchronously (the issue #185 ordering rule), firing both subscribers, so two `api.getHistory` calls with different filters are in flight. Neither is sequenced, so the slower first response can land last and repaint the Approval Queue with rows for accounts the user just filtered out; `lastPurchases`, which the marketplace Sell dialog reads for its price lookup, is then also stale. +- evidence: + ```ts + state.subscribeProvider(() => void loadHistory()); + state.subscribeAccount(() => void loadHistory()); + ``` +- suggested fix: Keep a monotonically increasing request id, capture it before the fetch, and skip all renders when it is no longer the latest — the pattern `topbar-filters.ts:92` already uses. +- verdict: CONFIRMED — Both slices are subscribed to loadHistory (frontend/src/history.ts:159-160) and the provider chip's onChange calls setCurrentAccountIDs([]) then setCurrentProvider synchronously (frontend/src/topbar-filters.ts:165-166), so two getHistory calls with different filters are in flight with no request-id guard on either. +- issue: (pending cross-reference) + +### A12-031 Money-mutation buttons are disabled only after the confirm dialog resolves +- category: concurrency +- severity: medium +- location: frontend/src/history.ts:1242 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: The Approve handler awaits a network `buildApprovalDetailsBody(id)` and then `confirmDialog` before it touches `btn.disabled`. `confirmDialog` appends a fresh backdrop per call with no singleton guard (`confirmDialog.ts:81`), so a double-click during that round trip stacks two modals and confirming both fires `api.approvePurchase(id)` twice. Retry (1336), Revoke (1404) and Sell (1440) have the same shape, and Retry mints a new execution. +- evidence: + ```ts + const detailsBody = await buildApprovalDetailsBody(id); + const ok = await confirmDialog({ title: 'Approve this pending purchase?', ... }); + if (!ok) return; + const rowActions = sameRowActions(btn); + rowActions.forEach((b) => { b.disabled = true; }); + ``` +- suggested fix: Disable the row's action buttons synchronously on click, re-enabling on cancel or failure. +- verdict: CONFIRMED — The Approve handler awaits buildApprovalDetailsBody and confirmDialog before touching btn.disabled (frontend/src/history.ts:1242-1258), Retry (:1340-1348), Revoke and Sell (:1440-1533) have the same shape, and confirmDialog appends a fresh backdrop per call with no singleton guard (frontend/src/confirmDialog.ts:81). +- issue: (pending cross-reference) + +### A01-018 persistAzureRevocation sleeps with a bare time.Sleep after money moved, ignoring ctx +- category: concurrency +- severity: low +- location: internal/api/handler_purchases_revoke.go:776 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: Azure has already returned the reservation and `MarkPurchaseRevoked` fails. The retry loop blocks 1s+3s+9s with `time.Sleep` while the Lambda request context may already be cancelled; the subsequent store calls then fail on the dead ctx, the 13 seconds are wasted, and the 207 RECONCILE_PENDING response arrives late or not at all. +- evidence: + ```go + for attempt, backoff := range revokeMarkRetryBackoffs { + if markErr == nil { + break + } + logging.Warnf(...) + time.Sleep(backoff) + markErr = h.config.MarkPurchaseRevoked(ctx, record.PurchaseID, now, "direct-api", "", calcRefundAmount, calcRefundCurrency) + } + ``` +- suggested fix: Use the ctx-aware `select { case <-time.After(backoff): case <-ctx.Done(): break }` form and stop retrying once ctx is done. +- verdict: CONFIRMED — persistAzureRevocation (internal/api/handler_purchases_revoke.go:767-780) loops revokeMarkRetryBackoffs (1s/3s/9s, :126-130) with a bare time.Sleep and never checks ctx.Done() before re-calling MarkPurchaseRevoked with the same ctx. +- issue: (pending cross-reference) + +### A02-017 A request-scoped context seals the process-wide AWS config cache +- category: concurrency +- severity: low +- location: internal/api/handler.go:1062 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: The first request that reaches `resolveAWSCallerIdentity` or `getLambdaInvoker` (handler_recommendations_refresh.go:225-227) runs `awsconfig.LoadDefaultConfig(ctx)` inside `awsCfgOnce` with that request's context. If the request is cancelled or hits its deadline during the load (IMDS/SSO/credential-process resolution on ECS or dev hosts), `awsCfgErr` holds `context canceled` for the container's lifetime, so every later reshape scope resolution, deployment-info call and async refresh fails until a cold start. The comment at handler.go:903-912 describes this exact hazard and avoids it for `loadAPIKey`, but the request path still has it. +- evidence: + ```go + h.awsCfgOnce.Do(func() { + h.awsCfg, h.awsCfgErr = awsconfig.LoadDefaultConfig(ctx) + }) + ``` +- suggested fix: Load with `context.Background()` plus a bounded timeout inside the `Once`, or do not cache the error (reset the `Once` on a context error) so the next request retries. +- verdict: PLAUSIBLE — resolveAWSCallerIdentity (handler.go:1062-1064) and getLambdaInvoker (handler_recommendations_refresh.go:225-227) both seal awsCfgOnce with the request ctx and the comment at handler.go:903-912 records this exact failure mode for loadAPIKey; but LoadDefaultConfig does little network I/O (an IMDS region probe only when no region is configured; credentials resolve lazily at first Retrieve), so a cancellation landing inside the Once needs a host with no AWS_REGION and a slow IMDS, which I could not establish from source. +- issue: (pending cross-reference) + +### A09-025 logging.SetOutput mutates shared logger state without synchronization +- category: concurrency +- severity: low +- location: pkg/logging/logger.go:272 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: the package doc explains that `level` is an `atomic.Int32` specifically because logging happens "from many goroutines" during the fan-out, but `output` is a plain field. `SetOutput` reads and writes it unguarded while `Logger.With` (line 128) concurrently reads `l.output` to build a derived logger. Under `-race`, a test that calls `SetOutput` to capture output while a background collector goroutine logs reports a data race on `defaultLogger.output`, and the derived logger can capture a torn or stale writer. +- evidence: + ```go + func SetOutput(w io.Writer) io.Writer { + prev := defaultLogger.output + defaultLogger.output = w + defaultLogger.logger.SetOutput(w) + return prev + } + ``` +- suggested fix: store the writer in an `atomic.Value` (or guard it with the same discipline as `level`) so `SetOutput` and `With` do not race, matching the reasoning already applied to the level field. +- verdict: CONFIRMED — `Logger.output` is a plain `io.Writer` field alongside the `atomic.Int32` level (pkg/logging/logger.go:36-42); `SetOutput` reads and writes it with no synchronization (logger.go:272-277) while `Logger.With` reads `l.output` to build the derived logger (logger.go:125-130). Narrowing note: the embedded `*log.Logger` has its own internal mutex, so the actual emit path is safe — the race is confined to the `output` field and the derived logger it seeds. +- issue: (pending cross-reference) + +### Category: performance + +9 findings: 5 medium, 4 low. + +### A07-011 Sparkline fetch issues one full region-wide coverage query per instance type and discards all but one row +- category: performance +- severity: medium +- location: providers/aws/recommendations/usage_history.go:79 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: `dailyUsageFilter` scopes only to `(service, region)`; the instance type is matched client-side in `applyPeriodsToDayMap`. `AttachDailyUsageHistory` calls `GetDailyUsagePcts` once per unique `(service, region, resourceType)` tuple, so 60 distinct EC2 SKUs in one region produce 60 identical CE queries that each return the whole region's coverage and throw away 59/60 of it. At CE's per-request charge that is 60 billed calls where one would do, on every recommendation refresh. +- evidence: + ```go + for _, k := range order { + if ctx.Err() != nil { + return + } + pcts, err := c.GetDailyUsagePcts(ctx, k.service, k.resourceType, k.region) + ``` +- suggested fix: fetch once per `(service, region)`, build a map from instance type to the daily series, and fan the result out to every rec sharing that pair. +- verdict: CONFIRMED — `dailyUsageFilter` (providers/aws/recommendations/usage_history.go:197-210) scopes only SERVICE and REGION, `applyPeriodsToDayMap` :126 discards non-matching instance types client-side, and `AttachDailyUsageHistory` :176-180 issues one `GetDailyUsagePcts` per unique `(service, region, resourceType)` from `groupRecsByTuple` :155, on the live refresh path at client.go:339. +- issue: (pending cross-reference) + +### A07-015 OpenSearch idempotency lookup pages the whole reservation inventory with no cap, on the purchase path +- category: performance +- severity: medium +- location: providers/aws/services/opensearch/client.go:242 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: `DescribeReservedInstances` has no name filter, so `findReservationByName` walks every page of every reservation in the account before each purchase, with no page ceiling. In an account with thousands of historical reservations this consumes the Lambda budget that `purchasecfg` was introduced to protect, and the purchase fails with a deadline error attributed to `PurchaseCommitment` rather than to the guard. Every other pagination loop on this shard's purchase paths carries a `maxOfferingPages` or `maxCommitmentPages` guard. +- evidence: + ```go + func (c *Client) findReservationByName(ctx context.Context, name string) (string, bool, error) { + var nextToken *string + for { + response, err := c.client.DescribeReservedInstances(ctx, &opensearch.DescribeReservedInstancesInput{ + NextToken: nextToken, + MaxResults: 100, + }) + ``` +- suggested fix: add a page cap and a `ctx.Err()` check at the top of the loop, and fail loud when the cap is hit (a truncated guard must not report "not found" and let the purchase proceed). +- verdict: CONFIRMED — `findReservationByName` (providers/aws/services/opensearch/client.go:242-268) has neither a page counter nor a `ctx.Err()` check, unlike the offering loop in the same file at :403 which guards with `maxOfferingPages` (:378), and `idempotencyGuard` :275-285 calls it before every tokened purchase. +- issue: (pending cross-reference) + +### A07-016 Redshift idempotency guard makes one DescribeTags call per reserved node, unbounded +- category: performance +- severity: medium +- location: providers/aws/services/redshift/client.go:260 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: `findNodeByIdempotencyToken` pages all reserved nodes with no cap and calls `nodeHasIdempotencyTag` (a `DescribeTags` round trip) for each active node until it finds a match. An account with 200 active reserved nodes issues up to 200 sequential `DescribeTags` calls before every Redshift purchase, at 15s HTTP timeout each in the worst case. A slow or throttled `DescribeTags` mid-scan surfaces as a lookup failure and blocks a legitimate first-time purchase. +- evidence: + ```go + arn := fmt.Sprintf("arn:aws:redshift:%s:%s:reservednode:%s", c.region, accountID, nodeID) + tagged, err := c.nodeHasIdempotencyTag(ctx, arn, token) + if err != nil { + return "", false, fmt.Errorf("failed to read tags for reserved node %s: %w", nodeID, err) + } + ``` +- suggested fix: query `DescribeTags` once with `ResourceType: "reservednode"` and the token as a tag-value filter, rather than per node, and add a page cap to the outer loop. +- verdict: CONFIRMED — `scanNodesForToken` (providers/aws/services/redshift/client.go:259-278) issues one `DescribeTags` per active node and the outer `DescribeReservedNodes` loop at :237-252 has no page cap; `purchasecfg.HTTPTimeout` is 15s (providers/aws/internal/purchasecfg/config.go:32) and the error return at :271-273 aborts the guard, which `findNodeByIdempotencyToken` :226-230 turns into a blocked purchase. +- issue: (pending cross-reference) + +### A07-018 Savings Plans coverage and utilization calls bypass the shared concurrency semaphore +- category: performance +- severity: medium +- location: providers/aws/recommendations/sp_coverage.go:361 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: `fetchCoveragePage`, `fetchUtilizationPage` and `fetchRIPageWithRetry` all wrap their SDK call in `concurrency.Acquire` / `concurrency.Release` so CE traffic stays under `CUDLY_MAX_PARALLELISM`. `fetchSPCoveragePage` and `fetchSPUtilizationPage` (line 450) do not. When the ladder runs concurrently with a recommendation sweep, the SP calls are invisible to the cap and push total CE concurrency above the configured limit, which is what the cap exists to prevent. +- evidence: + ```go + rateLimiter := c.rateLimiter.newOperation() + for { + if waitErr := rateLimiter.Wait(ctx); waitErr != nil { + return nil, fmt.Errorf("rate limiter wait failed: %w", waitErr) + } + result, err := c.costExplorerClient.GetSavingsPlansCoverage(ctx, input) + if !rateLimiter.ShouldRetry(err) { + ``` +- suggested fix: wrap both SDK calls in `concurrency.Acquire`/`Release` exactly as `fetchCoveragePage` does. +- verdict: CONFIRMED — `fetchSPCoveragePage` (providers/aws/recommendations/sp_coverage.go:361-374) and `fetchSPUtilizationPage` (:450-464) hold only the rate limiter, while `fetchCoveragePage` wraps its SDK call in `concurrency.Acquire`/`Release` at coverage.go:331-336 and documents the shared `CUDLY_MAX_PARALLELISM` cap at :322-324. +- issue: (pending cross-reference) + +### A12-022 `initRegistrations` stacks a filter listener on every visit to the Accounts tab +- category: performance +- severity: medium +- location: frontend/src/modules/registrations.ts:256 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: `#registrations-status-filter` is static markup in `index.html:714` and is never replaced. `navigation.ts:300` calls `loadAccountsTab()` on each activation of the Accounts tab, which calls `initRegistrations()` (`settings.ts:3346`). Each call adds another `change` listener to the same element, so after N visits a single dropdown change fires N concurrent `loadRegistrations()` calls and N `GET /api/registrations` requests, whose responses race to render the same container. +- evidence: + ```ts + export function initRegistrations(): void { + const filterEl = document.getElementById('registrations-status-filter'); + filterEl?.addEventListener('change', () => void loadRegistrations()); + void loadRegistrations(); + } + ``` +- suggested fix: Guard with a module-level `wired` flag or a `dataset` marker on the element, matching the `wireRefreshButton` pattern in `inventory.ts:141`. +- verdict: CONFIRMED — initRegistrations unconditionally adds a `change` listener to the static `#registrations-status-filter` (frontend/src/modules/registrations.ts:254-257), which lives outside the container loadRegistrations rewrites (frontend/src/index.html:714), and switchSettingsSubTab re-runs loadAccountsTab on every Accounts activation with no self-switch guard (frontend/src/navigation.ts:269-301 into frontend/src/settings.ts:3346). +- issue: (pending cross-reference) + +### A01-019 RI exchange history is a capped 500-row fetch filtered in Go, so scoped users silently lose rows +- category: performance +- severity: low +- location: internal/api/handler_ri_exchange.go:2064 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: A deployment with more than 500 exchange records in the last year. `GetRIExchangeHistory(ctx, since, 500)` returns the newest 500 across all accounts; the allowed_accounts filter then runs in memory, so a scoped user whose account's exchanges are older than the newest 500 sees an empty or truncated history with no indication of truncation. +- evidence: + ```go + since := time.Now().AddDate(-1, 0, 0) + records, err := h.config.GetRIExchangeHistory(ctx, since, 500) + // ... + if !allowed.AllowsAll() { + nameByID := h.resolveAccountNamesByID(ctx) + filtered := records[:0] + ``` +- suggested fix: Push the account predicate into the store query (pass the allowed account IDs) and apply the limit after filtering. +- verdict: CONFIRMED — getRIExchangeHistory (internal/api/handler_ri_exchange.go:2063-2085) calls GetRIExchangeHistory(ctx, since, 500), whose SQL (internal/config/store_postgres.go:2731-2745) is ORDER BY created_at DESC LIMIT $2 with no account predicate, then filters by allowed_accounts in memory with no truncation indicator. +- issue: (pending cross-reference) + +### A05-017 OpenSearch and Redshift probes filter client-side under a five-page cap +- category: performance +- severity: low +- location: internal/commitmentopts/probe.go:283 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: Both APIs have no instance-type filter, so the probe fetches unfiltered pages and discards non-matching rows in Go, while `walkPaginated` stops after `maxPages` (5) × `pageSize` (100). If the region's `t3.small.search` (or `dc2.large`) offerings sort past offering 500, the probe yields zero combos for that service. `Save` still writes the singleton probe-run row, `HasData` becomes true, and nothing ever re-probes — the cache is permanently warm and permanently missing that service. The cap's stated purpose is bounding API spend when pagination detection breaks, which the server-side-filtered probers do not need it for; it is only load-bearing here, where it truncates real data. +- evidence: + ```go + for _, o := range out.ReservedInstanceOfferings { + if string(o.InstanceType) != probeTargetOpenSearch { + continue + } + ``` +- suggested fix: for the two client-side-filtered probers, stop early once all six (term, payment) combos have been seen and otherwise let pagination run to completion, or raise the cap for those two and log when it is hit so the truncation is visible rather than silent. +- verdict: PLAUSIBLE — every mechanical link checks out: walkPaginated hard-stops at `page < maxPages` with maxPages=5 and pageSize=100 (probe.go:23, :27, :86) and never logs when it hits the cap; OpenSearch (probe.go:283) and Redshift (:338) discard non-matching rows in the closure while the SP prober filters server-side via PlanTypes (:540); and the cache is permanently warm because Save inserts the singleton probe-run row (store_postgres.go:110) that HasData (:80-83) reads as "data exists". The unestablished condition is external: whether AWS actually returns `t3.small.search` / `dc2.large` offerings past position 500 in a given region, which I cannot determine from source. +- issue: (pending cross-reference) + +### A12-052 `parseNumericFilter` is re-parsed for every row of every numeric filter +- category: performance +- severity: low +- location: frontend/src/lib/column-filters.ts:119 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: The parse call sits inside the per-row predicate. Recommendations, Plans, History and RI Exchange all route through this helper, so a 500-row Opportunities table with three active numeric filters performs 1,500 regex-parse plus closure-allocation cycles on every render, and every popover commit re-renders. The parse result is loop-invariant. +- evidence: + ```ts + return rows.filter((row) => { + for (const [col, filter] of entries) { + ... + } else { + const parsed = parseNumericFilter(filter.expr); + if (!parsed.ok) continue; + ``` +- suggested fix: Hoist the parse above `rows.filter`, building a `[col, predicate][]` array once. +- verdict: CONFIRMED — parseNumericFilter is called inside the per-row predicate while `entries` is hoisted once above it (frontend/src/lib/column-filters.ts:110, :119-121), so the parse is provably loop-invariant work repeated per row per numeric filter. +- issue: (pending cross-reference) + +### A12-071 Per-render `indexOf` inside the RI recommendation row map is O(n²) +- category: performance +- severity: low +- location: frontend/src/riexchange.ts:1317 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: Each `renderRecommendations` call performs `currentRecommendations.indexOf(rec)` once per visible row. Combined with A12-052's per-row filter re-parse, a few hundred recommendations with three active numeric filters cost thousands of redundant scans and regex compiles per filter interaction and per utilization re-render. +- evidence: + ```ts + const visibleWithIdx = visible.map((rec) => ({ + rec, + idx: currentRecommendations.indexOf(rec), + })); + ``` +- suggested fix: Carry the index through the filter by mapping to `{rec, idx}` before filtering. +- verdict: CONFIRMED — currentRecommendations.indexOf(rec) runs once per visible row inside the render map (frontend/src/riexchange.ts:1313-1317), making the row build quadratic in the recommendation count on every filter interaction and every utilization re-render. +- issue: (pending cross-reference) + +### Category: test-gap + +29 findings: 8 high, 13 medium, 8 low. + +### A05-002 `singleCloudAccountIDFromRecs` is unit-tested in isolation, so the ambient fallback above stays green +- category: test-gap +- severity: high +- location: internal/purchase/execution_test.go:1628 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: The only test of the two-distinct-accounts case asserts the helper returns nil and labels that "multi-account, not this path". There is no test that drives `executeSingleAccount` (or `ApproveAndExecute` / `directExecutePurchase`) with a mixed-account rec set and asserts the purchase is refused, so A05-001 passes CI. `TestMultiAccountSeedsStablePerAccountKey` and the other money-path regressions all exercise the plan fan-out, which reaches `executeMultiAccount`, never this branch. +- evidence: + ```go + { + name: "two distinct account IDs returns nil (multi-account, not this path)", + recs: []config.RecommendationRecord{ + {CloudAccountID: &aid1}, + {CloudAccountID: &aid2}, + }, + want: nil, + }, + ``` +- suggested fix: add a regression test that calls `executeSingleAccount` with recs naming two accounts and asserts `CreateAndValidateProvider` is never called (register the expectation first so the assertion is not vacuous). +- verdict: CONFIRMED — I opened the test files rather than grepping: `git grep -n "singleCloudAccountIDFromRecs\|resolveSingleAccountProvider\|executeSingleAccount" -- 'internal/**/*_test.go'` returns only execution_test.go:1279, :1294 and :1641, and the two-distinct-IDs case at execution_test.go:1628-1634 asserts the helper in isolation; the eight tests in money_path_regression_test.go (:86 through :540) all drive the plan fan-out or the idempotency key, none reaches the mixed-account single-account branch. +- issue: (pending cross-reference) + +### A16-001 4-eyes: the "creator's account could not be resolved" fail-closed branch has no test +- category: test-gap +- severity: high +- location: internal/purchase/approvals.go:284 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: `checkDifferentApprover`'s tier-2 (email) path is the one every token/SQS + approver takes. Blocks 281 and 285-287 are both uncovered: no test makes `GetUserEmailByID` return + an error, and none makes it return `("", nil)`. Delete the emptiness guard and + `strings.EqualFold("", actorEmail)` is false for every real actor, so 4-eyes silently *passes* for + any execution whose creator row is missing, deleted, or unresolvable — exactly the deleted-creator + case dual control exists for. `("", nil)` from a user lookup is an established shape in this repo, + so this is not a hypothetical input. Every existing four-eyes test + (internal/purchase/approvals_test.go:337-560) stubs `GetUserEmailByID` with a real address or + asserts it is never called. +- evidence: + ```go + creatorEmail, err := m.config.GetUserEmailByID(ctx, *execution.CreatedByUserID) + if err != nil { + return fmt.Errorf("4-eyes policy check: failed to resolve creator identity: %w", err) + } + creatorEmail = strings.TrimSpace(creatorEmail) + if creatorEmail == "" { + logging.Warnf("purchase[%s]: 4-eyes mode on; creator account %s not found, denying (fail-closed)", + executionID, *execution.CreatedByUserID) + return fmt.Errorf("4-eyes approval mode is enabled but the creator's account could not be resolved; an admin must investigate before approving") + } + ``` +- suggested fix: add two subtests to `internal/purchase/approvals_test.go` on the + `ApproveExecution` (tier-2) shape: `GetUserEmailByID` returning `("", nil)`, and returning an + error. Assert the error text and `AssertNotCalled` on `TransitionExecutionStatus`. +- verdict: CONFIRMED — a merged cross-package profile (`internal/api`, `internal/server`, `internal/scheduler` and `internal/purchase` together, `-coverpkg` over all four) still shows approvals.go:281.3-282.1 and 285.3-288.1 at 0, and the scenario is not hypothetical: internal/config/store_postgres.go:1587-1588 returns `("", nil)` on `pgx.ErrNoRows`, so a deleted creator row lands on exactly this branch, while every stub in internal/purchase/approvals_test.go:515,550 and coverage_extra_test.go:422,476 returns a real address. +- issue: (pending cross-reference) + +### A16-004 The scheduled-purchase fire success path is behind a t.Skip, and its stand-in guard only checks a signature +- category: test-gap +- severity: high +- location: internal/purchase/scheduled_fire_test.go:157 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: `result.Fired++` (internal/purchase/scheduled_fire.go:60) and `fireOneDue`'s + success return (scheduled_fire.go:105) are both uncovered. The four live tests in this file cover + only no-rows, list-error, CAS-race-lost and hard-DB-error; the one test that would exercise a + purchase actually firing is skipped, and its own comment calls it "the end-to-end smoke test that + verifies the fire-tick path does not silently no-op the pre-fire delay branch (CRITICAL)". So the + delayed-purchase feature can stop spending entirely — `fireOneDue` returning `(false, false)` on + every row, or `executeAndFinalize` erroring for all of them — and the suite is green with + `Fired == 0`, which is what three of the four live tests already assert. The compensating test + named `TestFireScheduledDelayedPurchases_DelayPathNotSilentNoOp` (line 175) is a compile-time + method-signature assertion; it cannot observe a no-op despite its name. +- evidence: + ```go + func TestFireScheduledDelayedPurchases_EndToEnd(t *testing.T) { + t.Skip("placeholder until full provider-stub wiring is available; " + + "the CAS and audit-stamp paths are covered by the unit tests above") + // When un-skipped, the test scenario is: + // 1. Create an execution with purchase_delay_hours > 0, Status="scheduled", + // ScheduledExecutionAt = time.Now().Add(-1h). + // 2. Call FireScheduledDelayedPurchases(ctx). + // 3. Assert result.Fired == 1, result.RaceLost == 0, result.Errored == 0. + ``` +- suggested fix: un-skip it using the provider/email wiring `money_path_regression_test.go` already + builds (`awsAccessKeyCredStore` plus the `MockProviderFactory` chain), and assert `Fired == 1` + plus exactly one `PurchaseCommitment` call. Separately assert that a `SavePurchaseExecution` + failure after a winning CAS (the uncovered AUDIT GAP branch at scheduled_fire.go:97) still fires. +- verdict: CONFIRMED — scheduled_fire.go:60.4-61.1 (`result.Fired++`) and 105.2-105.20 (the success return) are both 0 in the merged cross-package profile, so nothing in `internal/server` or `internal/scheduler` reaches them either. One correction that does not change the verdict: a fifth live caller exists outside the named file, armed_redrive_test.go:181, but it asserts `result.Fired == 0` and `result.Errored == 1`, so it joins rather than closes the set of tests a total no-op would satisfy. +- issue: (pending cross-reference) + +### A16-005 FinalizeInFlightRevocations has zero coverage; its only "test" asserts a stub's own return value +- category: test-gap +- severity: high +- location: internal/server/handler_test.go:142 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: every statement of `internal/purchase/finalize_revocations.go` (lines 46-75) is + uncovered. The only exercise of the feature is the scheduled-task dispatch test above, which + installs a fake returning `FinalizeResult{Found: 1, Finalized: 1}` and then asserts only that + `HandleScheduledTask` returned no error — an assertion on the fake's own return value that holds + whatever the real sweep does. Consequences that stay green: the `Found`/`Finalized`/`Errored` + accounting can be wrong in any direction; the retry loop's `time.Sleep(backoff)` + (finalize_revocations.go:62) is not ctx-aware and the loop never checks `ctx.Err()`, so a Lambda + whose context is already cancelled still burns up to 8s per row and can be killed mid-sweep; and a + row that never finalizes is counted rather than surfaced. These rows are the audit record for + Azure reservations already returned to the provider, so a silent failure leaves the DB claiming + money was never refunded. +- evidence: + ```go + setupMocks: func(s *testutil.MockScheduler, p *testutil.MockPurchaseManager) { + p.FinalizeInFlightRevocationsFunc = func(ctx context.Context) (*purchase.FinalizeResult, error) { + return &purchase.FinalizeResult{Found: 1, Finalized: 1}, nil + } + }, + expectError: false, + ``` +- suggested fix: add `internal/purchase/finalize_revocations_test.go` driving the real method against + `MockConfigStore`: three in-flight rows where one succeeds first try, one succeeds on retry, one + fails all three attempts; assert `Found=3, Finalized=2, Errored=1` and that the sweep did not abort. + Add a cancelled-context case once the sleep is made ctx-aware. +- verdict: CONFIRMED — no `finalize_revocations_test.go` exists, `FinalizeInFlightRevocations` appears in no test outside the two fakes at internal/server/handler_test.go:142,152, and nothing under tests/, pkg/, cmd/ or mcp/ names it or `GetPurchaseHistoryInFlight`; every statement block in finalize_revocations.go is 0 in the merged profile. The ctx-blindness is in the source as described: the loop at finalize_revocations.go:56-64 calls `time.Sleep(backoff)` with no `ctx.Err()` check, and the two backoffs sum to 8s per stuck row. +- issue: (pending cross-reference) + +### A16-006 The AUDIT LOSS branch in the per-account fan-out is untested, and it is the one that decides SQS ack vs redelivery +- category: test-gap +- severity: high +- location: internal/purchase/execution.go:269 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: block 269-270 is uncovered. This is the only place where `executeForAccount` + returns `committed = true` *together with* an error. `TestMultiAccountPartialSuccessIsAcked` + (internal/purchase/money_path_regression_test.go:283) proves the >=1-committed run is ACKed, but + only via the clean save path — its `SavePurchaseExecutionFn` always returns nil. Change line 269 to + `return false, ...` (the natural-looking "we failed, report failure" edit) and a run whose cloud + purchase succeeded but whose row failed to persist is reported as fully failed, the SQS message is + redelivered, and the redelivery re-buys the commitment. Nothing in the suite fails, because no test + ever makes `SavePurchaseExecution` fail on a committed account. +- evidence: + ```go + partial, committed := applyAccountOutcome(&acctExec, purchaseErrors) + + if saveErr := m.config.SavePurchaseExecution(ctx, &acctExec); saveErr != nil { + return committed, fmt.Errorf("AUDIT LOSS: failed to save execution record for account %s: %w", account.ID, saveErr) + } + ``` +- suggested fix: clone `TestMultiAccountPartialSuccessIsAcked` with a `SavePurchaseExecutionFn` that + errors for the per-account row after `PurchaseCommitment` succeeded, and assert + `handleExecutePurchase` still returns nil (ack) and that no second `PurchaseCommitment` occurs. +- verdict: CONFIRMED — execution.go:269.3-270.1 is 0 in the merged profile. Every `SavePurchaseExecutionFn` in the package either returns nil unconditionally or records and returns nil (execution_test.go:869,1038,1208,1826; money_path_regression_test.go:43,167,249,326,409,567; armed_redrive_test.go:101,245); the one stub that returns an error, execution_test.go:923, drives the credential-failure path and its assertion at line 958 lands on the other AUDIT LOSS site, execution.go:300, not on 269. +- issue: (pending cross-reference) + +### A16-007 The partial-sweep eviction guard is proven for Azure only; the GCP per-account collector duplicating it has no test +- category: test-gap +- severity: high +- location: internal/scheduler/scheduler.go:912 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: `internal/scheduler/partial_sweep_eviction_test.go` is a careful guard for the + invariant "an incomplete sweep must not authorize stale-row eviction", but every case drives Azure: + `collectAzureRecommendations`, `fetchAndConvert(..., "azure", ...)` (scheduler_test.go:756, 782) and + `tolerateIncompleteSweep("azure", ...)` (scheduler_test.go:800-814). `collectGCPForAccount` carries + its own inlined copy of the same plumbing and is entirely uncovered (lines 884-912), as is + `collectAWSForAccount` (752-770). Hardcode `true` in place of `complete` at line 912 and a partial + GCP sweep enters `outcome.SucceededAccountIDs`, which authorizes the + `DELETE ... WHERE collected_at < $1 AND (provider, account_key) IN (…)` that + `UpsertRecommendations` runs — deleting the previous-cycle recommendations for every project the + sweep never queried. The whole Azure suite stays green. +- evidence: + ```go + recs, err := recClient.GetAllRecommendations(ctx) + complete, err := tolerateIncompleteSweep("gcp", err) + if err != nil { + return nil, false, fmt.Errorf("get recommendations: %w", err) + } + return s.tagAccount(s.convertRecommendations(recs, "gcp"), acct.ID), complete, nil + ``` +- suggested fix: parameterise `newPartialSweepScheduler` over the provider name and run the same + three cases (partial, complete, hard error) through `collectGCPForAccount` and + `collectAWSForAccount`, asserting the returned `complete` bool directly. +- verdict: CONFIRMED — scheduler.go:912.2-912.84 and the whole body of `collectGCPForAccount` from 883 are 0 in the merged profile, as is `collectAWSForAccount`'s tail at 770. partial_sweep_eviction_test.go names only Azure: its provider stub, its three `collectAzureRecommendations` cases at lines 106, 128 and 174, and its `fanOutPerAccount` case at 148 all pass `"azure"`. The shared helper `tolerateIncompleteSweep` is directly tested at scheduler_test.go:800-814, so what is untested is the per-provider wiring of its `complete` return, which is exactly what the finding claims. +- issue: (pending cross-reference) + +### A16-008 getPlannedPurchases' account-scope filter is never exercised; the only test runs as an unrestricted admin +- category: test-gap +- severity: high +- location: internal/api/handler_purchases.go:147 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: blocks 140 (`plan == nil`), 144 (scope-check error) and 147 (`!ok`, the scope + exclusion) are all uncovered. `TestHandler_getPlannedPurchases` + (internal/api/handler_purchases_test.go:910) is the sole test: it calls `mockAuth.grantAdmin()`, + supplies one plan and one execution, and asserts `Len(result.Purchases, 1)`. Because the fixture + contains no plan the session should be denied, the loop's exclusion branch is not merely + unasserted, it is unreachable — the negative guard is satisfied by an empty set. Make + `isPlanAllowedCached` return `true` unconditionally, or delete lines 146-148, and a per-account + scoped user reading `GET /api/purchases/planned` receives every tenant's plan names, step + schedules, estimated savings and upfront costs. The mutating twin + (`pausePlannedPurchase`) is covered by internal/api/handler_per_account_perms_test.go:770, so this + is a listing-vs-mutation asymmetry, not a wholly missing concern. +- evidence: + ```go + for _rvc := range executions { + exec := executions[_rvc] + plan := planMap[exec.PlanID] + if plan == nil { + continue + } + ok, err := h.isPlanAllowedCached(ctx, session, exec.PlanID, allowedPlan) + if err != nil { + return nil, err + } + if !ok { + continue + } + ``` +- suggested fix: add a scoped-session variant seeded with two plans on two different cloud accounts, + the session allowed only one, and assert the response contains exactly the allowed plan's ID. Assert + a non-zero expected count first so the test cannot pass on an empty result. +- verdict: CONFIRMED — handler_purchases.go:140.4, 144.4-145.1 and 147.4 are all 0 in the merged profile. One correction that strengthens rather than weakens the finding: line 910 is not the sole test, there are six callers of `getPlannedPurchases` (handler_purchases_test.go:968, 1009, 1067, 1141, 1865, 1930), and none of them can reach the exclusion — four run `mockAuth.grantAdmin()`, one errors before the loop, and the read-only one at 1930 stubs `GetAllowedAccountsAPI` to return nil, which `isPlanAllowedCached` reads as unrestricted. Every fixture also puts every execution's PlanID in the plan map, so `plan == nil` is unreachable too. +- issue: (pending cross-reference) + +### A16-009 authorizeSessionRevokeExecution's allow and deny boundaries are both untested +- category: test-gap +- severity: high +- location: internal/api/handler_purchases_revoke.go:320 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: within this function, only the admin-API-key short-circuit (303) and the + revoke-own creator check (326) are covered. The `hasAny` allow (312), the `!hasOwn` 403 (320) and + both permission-lookup error paths (309, 317) are uncovered. So no test proves that a session + holding *neither* `revoke-any` nor `revoke-own` is refused: delete lines 319-321 and any + authenticated user reaches the creator comparison, and one who happens to be the creator revokes a + scheduled purchase with no revoke permission at all. The sibling function for purchase-history + revoke, `authorizeSessionRevoke` (line 352), has both of these branches covered — same guard, two + entry points, only one of them tested. +- evidence: + ```go + hasAny, err := h.auth.HasPermissionAPI(ctx, session.UserID, auth.ActionRevokeAny, auth.ResourcePurchases) + if err != nil { + return fmt.Errorf("permission check failed: %w", err) + } + if hasAny { + return nil + } + hasOwn, err := h.auth.HasPermissionAPI(ctx, session.UserID, auth.ActionRevokeOwn, auth.ResourcePurchases) + if err != nil { + return fmt.Errorf("permission check failed: %w", err) + } + if !hasOwn { + return NewClientError(403, "permission denied: requires revoke-any or revoke-own on purchases") + } + ``` +- suggested fix: table-drive `authorizeSessionRevokeExecution` over (revoke-any granted → allow), + (neither granted → 403), (revoke-own granted + non-creator → 403), (revoke-own + creator → allow), + mirroring the coverage `authorizeSessionRevoke` already has. +- verdict: CONFIRMED — handler_purchases_revoke.go blocks at 309, 312, 317 and 320 are all 0 in the merged profile while 303 and 326 are 1, matching the finding exactly. Only three tests reach the function (handler_purchases_revoke_test.go:766 via `revokePurchase`, 782, 801) and every one of them stubs `revoke-any` false and `revoke-own` true, so the neither-granted 403 at line 320 is never exercised on a path that is plainly reachable in production for any authenticated non-admin session. +- issue: (pending cross-reference) + +### A02-026 No test exercises a scoped session with an explicit account filter on the dashboard, nor name-based scopes on marketplace sell-own +- category: test-gap +- severity: medium +- location: internal/api/handler_dashboard_test.go:636 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: The only scoped dashboard tests (`GetAllowedAccountsAPI` at lines 636, 681, 1529) cover recommendations, upcoming purchases and the slice-aliasing regression; none sends `account_id`/`account_ids` for an out-of-scope account, so A02-001 is invisible to the suite. In handler_marketplace_test.go the scope fixtures hold `"acct-1"` / `"acct-other"` (lines 164, 200), i.e. the row's UUID, whereas production scopes hold names, so `TestMarketplaceList_SellOwnAllowed` passes while A02-003 denies every real sell-own user. Both tests would stay green with the bugs present. +- evidence: + ```go + // allowed accounts cover the row's cloud account. + authSvc.On("GetAllowedAccountsAPI", mock.Anything, "user-1"). + Return([]string{"acct-1"}, nil) + ``` +- suggested fix: Add `TestGetDashboardSummary_ScopedUser_ExplicitOutOfScopeAccountIsIgnored` (expect zeroed commitment KPIs and no `GetActivePurchaseHistory` call with the foreign id) and switch the marketplace scope fixtures to the account name with a `ListCloudAccounts` stub, confirming they fail on the current code first. +- verdict: CONFIRMED — every getDashboardSummary call in handler_dashboard_test.go (lines 79,142,186,1212,1257,1302,1352,1372,1396) passes either no account params or only provider, and the scoped tests at 636/681/1529 exercise filterDashboardRecommendations/upcoming only, so no test combines GetAllowedAccountsAPI scoping with account_id/account_ids; the marketplace fixtures (handler_marketplace_test.go:164-165,200-201) grant "acct-1"/"acct-other", which is the UUID branch of Allows, so both A02-001 and A02-003 stay green. +- issue: (pending cross-reference) + +### A03-025 Tests enumerate only the exact carved-out pairs and never exercise the failure paths above +- category: test-gap +- severity: medium +- location: internal/auth/group_ceiling_permissions_test.go:24 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: Every carve-out test iterates the three exact `(verb, purchases)` pairs (group_ceiling_permissions_test.go:24-28, self_escalation_carveout_test.go:35-39, service_apikeys_test.go:1169-1188), so A03-001/002/003 pass the suite. `TestLogin_WithMFA_RecoveryCode_ConsumedOnce` (service_mfa_test.go:438-466) only asserts the in-memory slice is empty with a succeeding `UpdateUser`, so A03-008 passes. `TestService_UpdateUser/"update active status successfully"` (service_user_test.go:646-665) registers no `DeleteUserSessions` expectation, so A03-005 passes and would keep passing after the fix unless the expectation is added. `TestService_ResetTokenStatus` (service_password_test.go:542-557) asserts an inactive user with a valid token is the "invite" flow; nothing asserts a deactivated account is refused, so A03-006 passes. `TestResolveBastionProvider_LegacyFallback` (credentials/resolver_test.go:366-374) pins the A03-011 fallback as desired behaviour. No test covers `RequestPasswordReset` for an invited user (A03-009), MFA setup on an already-enabled user (A03-007), or a signer whose first KMS call fails and second succeeds (A03-014). +- evidence: + ```go + carvedOut := []APIPermission{ + {Action: ActionExecute, Resource: ResourcePurchases}, + {Action: ActionApproveAny, Resource: ResourcePurchases}, + {Action: ActionRetryAny, Resource: ResourcePurchases}, + } + ``` +- suggested fix: Add the wildcard-resource variants to every carve-out table, a persist-failure case for recovery codes, a deactivation case asserting `DeleteUserSessions` and a rejected session, a reset-on-deactivated-account refusal, an invited-user forgot-password case, an MFA re-enrol-while-enabled refusal, and a transient-then-success signer case; convert the legacy-fallback test into an error assertion. +- verdict: CONFIRMED — the cited tables hold only the exact (verb, purchases) pairs (group_ceiling_permissions_test.go:24-28, self_escalation_carveout_test.go:35-39) and a grep of internal/auth/*_test.go for an execute permission with ResourceAll or "*" returns nothing; TestLogin_WithMFA_RecoveryCode_ConsumedOnce stubs UpdateUser to succeed and asserts only the slice (service_mfa_test.go:438-466); the deactivation case registers no DeleteUserSessions expectation (service_user_test.go:646-665); no test references PasswordSetupExpiry, none of the three MFASetup calls sets MFAEnabled=true beforehand (service_mfa_test.go:221, :236, :486), the fake KMS client's GetPublicKey never fails (aws_signer_test.go:29-34), and TestResolveBastionProvider_LegacyFallback asserts NoError on the fallback (resolver_test.go:366-374). +- issue: (pending cross-reference) + +### A05-007 `FinalizeInFlightRevocations` has no tests at all +- category: test-gap +- severity: medium +- location: internal/purchase/finalize_revocations.go:45 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: `git grep FinalizeInFlightRevocations` matches only the definition — no test file references it. This is the sweep that closes the window where an Azure Return API call succeeded but the DB write failed, i.e. it is the sole mechanism that makes the revocation audit record eventually consistent for money already returned. Nothing pins the retry count, the shared `now` timestamp, the per-row error isolation, or that a failing row is counted in `Errored` rather than silently dropped, so any of those can regress unnoticed. +- evidence: + ```go + func (m *Manager) FinalizeInFlightRevocations(ctx context.Context) (*FinalizeResult, error) { + rows, err := m.config.GetPurchaseHistoryInFlight(ctx) + if err != nil { + return nil, err + } + ``` +- suggested fix: add a table test over a mocked store covering: all rows succeed on the first attempt, a row that succeeds on retry 2, and a row that never succeeds (asserting `Errored` increments and the sweep continues to the next row). +- verdict: CONFIRMED — the substance holds, with one correction to the evidence: `git grep -n FinalizeInFlightRevocations -- '*.go'` does match test files (internal/server/handler_test.go:142 and :152), but those stub the whole method through testutil.MockPurchaseManager (internal/testutil/mocks.go:109-111) and exercise the HTTP handler at internal/server/handler.go:353, not one line of finalize_revocations.go. No test in internal/purchase references it, so the retry count, the shared `now` at :52, per-row isolation and the Errored counter are all unpinned. +- issue: (pending cross-reference) + +### A07-033 No test exercises a title-cased engine or the CE-offering-ID purchase path +- category: test-gap +- severity: medium +- location: providers/aws/services/elasticache/client_test.go:446 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: every ElastiCache test constructs `CacheDetails{Engine: "redis"}` in the already-lowercase form, so A07-006 (the raw title-cased CE value reaching the offering filter) cannot fail any test. Likewise, `savingsplans/client_test.go:1530` is the single test touching a non-empty `OfferingID`, and it asserts only that the offerings API is skipped — nothing asserts that a plan-type, term or payment-option mismatch is still rejected on that path (A07-001). Both bugs live in the exact gap the suite leaves open. +- evidence: + ```go + Details: &common.CacheDetails{Engine: "redis", NodeType: "cache.m6g.large"}, + ``` +- suggested fix: add an ElastiCache offering-lookup case with `Engine: "Redis"` and an empty engine, and a Savings Plans case where a rec carrying an `OfferingID` has a plan type that does not match the scoped client; both should fail against the current code. +- verdict: CONFIRMED — every `CacheDetails` literal in providers/aws/services/elasticache/client_test.go uses lowercase `"redis"` (:251, :288, :341, :446, :560, :616, :647, :680, :714, :737, :762) with no title-cased or empty-engine case, and the only Savings Plans test with a non-empty `OfferingID` is `TestEC2InstanceSP_CEProvidedOfferingIDUsedDirectly` (savingsplans/client_test.go:1512-1537), which asserts only that the offerings API is skipped; the plan-type rejection test at :364-385 uses a rec with no `OfferingID`, so it never exercises the short-circuit path. +- issue: (pending cross-reference) + +### A09-031 No test covers the two money-path guards that fail open in pkg/exchange +- category: test-gap +- severity: medium +- location: pkg/exchange/exchange_test.go:1 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: across all six test files in `pkg/exchange` there is no case that sets `PaymentDueRaw` to the empty string on a quote destined for `checkInitialQuote`/`checkReQuote` (A09-004), no case with a `CurrencyCode` other than `"USD"` reaching the cap comparison (A09-005), and no reference at all to `assertAccount` or `ExpectedAccount` (A09-006). The suite passes with all three defects present, so the guards' teeth are unverified on the one path in the package that spends money irreversibly. +- evidence: + ```text + $ /usr/bin/grep -rn "assertAccount\|ExpectedAccount" pkg/exchange/*_test.go + (no output) + $ /usr/bin/grep -rn "CurrencyCode" pkg/exchange/*_test.go | grep -v '"USD"' + (no output) + ``` +- suggested fix: add three table cases to `exchange_test.go` — an empty `PaymentDue` quote that must be refused, a `"EUR"` quote that must be refused, and an `ExpectedAccount` mismatch that must abort before `AcceptReservedInstancesExchangeQuote` — each asserting the accept call was never made. +- verdict: CONFIRMED — I opened all six test files rather than trusting the greps. exchange_test.go holds only `TestParseDecimalRat`, `TestPaymentDueUSDStr_InJSON` and `TestSpendCapComparison` (which exercises `big.Rat.Cmp` directly, never `checkInitialQuote`). fail_loud_test.go's `seqQuoteOut` always sets a non-empty `PaymentDue` (fail_loud_test.go:47-52) and never sets `CurrencyCode` or `ExpectedAccount`; multi_target_test.go:37 is the same shape. No test constructs a quote with an empty `PaymentDue` reaching the cap check, none uses a non-USD currency there, and `assertAccount`/`ExpectedAccount` appear in no test file. Minor evidence correction: the second grep in the finding does produce output (five EUR/empty `OfferingOption.CurrencyCode` lines in reshape_crossfamily_test.go), but those exercise `passesDollarUnitsCheck`, not the cap comparison, so the substantive claim stands. +- issue: (pending cross-reference) + +### A09-032 httpclient tests assert only the literal-IP path, so the DNS bypass stays green +- category: test-gap +- severity: medium +- location: pkg/httpclient/httpclient_test.go:27 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: `TestNew_BlocksIMDS` exercises two hardcoded URLs, both containing the literal address the map holds. It therefore passes for any implementation that string-matches the pre-resolution host, including the current one, and would keep passing if the ECS credential endpoint or a resolving hostname were added as attack vectors. `TestNew_AllowsRegularEndpoints` confirms the happy path but adds no negative coverage. Nothing in the suite would fail if the blocklist were reduced to a single entry. +- evidence: + ```go + tests := []struct { + name string + url string + }{ + {name: "ipv4 link-local IMDS", url: "http://169.254.169.254/latest/meta-data/"}, + {name: "ipv6 AWS IMDS", url: "http://[fd00:ec2::254]/latest/meta-data/"}, + } + ``` +- suggested fix: add cases for `169.254.170.2` (ECS credentials) and for a hostname stubbed to resolve to `169.254.169.254` via an injected resolver, so the blocklist's actual reach is what the test measures. +- verdict: CONFIRMED — I read the whole file. The suite is three tests: `TestNew_NotDefaultClient` (identity assertions only), `TestNew_BlocksIMDS` with the two literal-IP URLs quoted, and `TestNew_AllowsRegularEndpoints` against an `httptest` server (pkg/httpclient/httpclient_test.go:11-73). Nothing exercises a resolving hostname or the ECS/EKS addresses, so the A09-001 DNS bypass and the A09-002 gaps both stay green. One correction: the closing sentence overstates — reducing the blocklist to a single entry WOULD fail whichever of the two subtests lost its address; what the suite cannot detect is any implementation that string-matches the pre-resolution host. +- issue: (pending cross-reference) + +### A10-025 No test asserts that money fields survive a count reduction on the extended-support path +- category: test-gap +- severity: medium +- location: cmd/multi_service_engine_versions_test.go:236 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: All six `adjustRecommendationForExcludedVersions` tests assert only `result.Count`, and none of the fixtures sets `EstimatedSavings` or `CommitmentCost` at all, so A10-001 is invisible to the suite and would remain invisible after a fix regressed. The sibling flags have the coverage this path lacks: `cmd/helpers_instance_limit_rescale_test.go` and `cmd/helpers_count_override_rescale_test.go` exist precisely to pin the scaling behaviour for `--max-instances` and `--override-count`. +- evidence: + ```go + result := adjustRecommendationForExcludedVersions(recommendation, instanceVersions, versionInfo) + assert.Equal(t, 8, result.Count, "Should exclude 2 instances (5.6 and 5.7 both in extended support)") + ``` +- suggested fix: Add a rescale test mirroring `helpers_instance_limit_rescale_test.go`: a rec with non-zero EstimatedSavings/CommitmentCost, two of ten instances excluded, asserting both fields land at 0.8x. +- verdict: CONFIRMED — opened every test file that drives this function: all assertions check only result.Count (cmd/multi_service_engine_versions_test.go:180, :185, :238, :255, :282 and cmd/multi_service_coverage_test.go:416) and no fixture sets EstimatedSavings or CommitmentCost, while cmd/helpers_instance_limit_rescale_test.go and cmd/helpers_count_override_rescale_test.go do exist for the sibling paths. +- issue: (pending cross-reference) + +### A12-037 The amortized Monthly Cost path in History has no test coverage +- category: test-gap +- severity: medium +- location: frontend/src/history.ts:1140 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: All eleven `history*.test.ts` files mock `getAmortizeUpfront` to return `false` and none flips it. No fixture creates `history-controls` or `purchases-approval-queue-section`, so both `mountAmortizeCheckbox` calls, `syncAmortizeCheckbox`, the `subscribeAmortizeUpfront` re-render path, the `amortizedMonthly` cell in both tables and the "(amortized)" header are never executed. A regression that added the upfront twice, or dropped it, would ship green. +- evidence: + ```ts + const displayMonthly = (rawMonthly != null && amortize) + ? amortizedMonthly(rawMonthly, p.upfront_cost, p.term) + : rawMonthly; + ``` +- suggested fix: Add a fixture with those containers and a test that flips `getAmortizeUpfront` to `true`, asserting the cell equals `monthly + upfront/(term*12)` and the header reads "Monthly Cost (amortized)". +- verdict: CONFIRMED — Every history suite pins getAmortizeUpfront to false and none flips it (frontend/src/__tests__/history.test.ts:52 and the nine siblings), and no fixture creates `history-controls` or `purchases-approval-queue-section` (only frontend/src/index.html:180, :242), so the amortized cell at frontend/src/history.ts:1138-1141 and both mount calls are never executed. +- issue: (pending cross-reference) + +### A12-043 No test covers `submitModalExecute` or `renderExchangeHistory` +- category: test-gap +- severity: medium +- location: frontend/src/__tests__/riexchange.test.ts:150 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: `executeExchange` is mocked in three test files but never asserted on; the suite asserts only the quote request shape. That is why the missing `region` (A12-003) and the `max_payment_due_usd` currency assumption (A12-039) ship green. `renderExchangeHistory`, `canApproveRIExchangeRow` and `handleRIExchangeApproveClick` have no test at all, so A12-042 and the hardcoded `$` are equally unguarded. +- evidence: + ```ts + executeExchange: jest.fn(), // mocked; no test reads mockedApi.executeExchange.mock.calls + ``` +- suggested fix: Add a test that drives quote to execute and asserts the full posted body including `region`, plus one rendering a pending history record and asserting the cells and the Approve gating. +- verdict: CONFIRMED — executeExchange is mocked in three suites (frontend/src/__tests__/riexchange.test.ts:12, riexchange-column-filters.test.ts:27, riexchange-active-ri-filters.test.ts:20) with no assertion on its calls anywhere, and renderExchangeHistory, canApproveRIExchangeRow and handleRIExchangeApproveClick appear in no test file. +- issue: (pending cross-reference) + +### A14-015 ci.yml's "Test Docker image" step asserts nothing and rebuilds the image it was given +- category: test-gap +- severity: medium +- location: .github/workflows/ci.yml:522 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: Both container runs end in `|| true`, so a binary that panics on `--version`, is built for the wrong architecture, or is missing from `/app/cudly` passes the step. The step also runs a second `docker build`, discarding the `load: true` image the preceding build-push-action step produced specifically so downstream steps would inspect the artifact this job built (comment at 512-514); the two builds can differ if the cache resolves differently. +- evidence: + ```yaml + - name: Test Docker image + run: | + docker build -t cudly:test . + docker run --rm cudly:test /app/cudly --version || true + docker run --rm cudly:test /app/cudly --help || true + ``` +- suggested fix: Drop the redundant `docker build`, run `cudly:${{ github.sha }}`, and remove the `|| true` so a broken binary fails the job. +- verdict: CONFIRMED — ci.yml:522-526 rebuilds `cudly:test` rather than exercising the `load: true` image `cudly:${{ github.sha }}` produced at ci.yml:509-517, and both `docker run` lines end in `|| true`, so no assertion can fail the step. +- issue: (pending cross-reference) + +### A14-031 The Azure sanity workflow compares the subscription against itself +- category: test-gap +- severity: medium +- location: .github/workflows/azure_sanity.yml:66 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: `--subscription-id` and `--expected-subscription` are both `${{ secrets.AZURE_SUBSCRIPTION_ID }}`. `runAccountShowCheck` fetches that subscription by ID and reads `sub.SubscriptionID` back from the response, so `validateAccountExpectations` (ci_cd_sanity_tests/pkg/sanity/azure/azure.go:84) compares a value against the value used to fetch it. The `unexpected subscription` branch is unreachable by construction, and the `azure:account:expected_checks` PASS in the uploaded report asserts nothing about the subscription axis. The tenant comparison is meaningful; the subscription one is not. +- evidence: + ```yaml + ./azure-sanity \ + --subscription-id "${AZURE_SUBSCRIPTION_ID}" \ + --expected-subscription "${AZURE_SUBSCRIPTION_ID}" \ + --expected-tenant "${AZURE_TENANT_ID}" \ + --out "${REPORT_PATH}" + ``` +- suggested fix: Drop `--expected-subscription` from this invocation, or source it from a separate repository variable so the two values can disagree. +- verdict: CONFIRMED — runAccountShowCheck fetches via `subClient.Get(ctx, subscriptionID)` and reads `info.ID = *sub.SubscriptionID` back from that response (ci_cd_sanity_tests/pkg/sanity/azure/azure.go:263-274), so validateAccountExpectations' `a.ID != opts.ExpectedSubID` test (azure.go:81) compares the fetch key against itself; a wrong ID errors at Get and skips the check entirely (azure.go:330). +- issue: (pending cross-reference) + +### A15-004 The vacuous-assertion guard silently skips packages that declare no mock of their own +- category: test-gap +- severity: medium +- location: internal/mocks/vacuous_assertion_guard_test.go:137 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: `TestNoUnfailableMockAssertions` walks the repo, then per directory does `mocks := collectMocks(files); if len(mocks) == 0 { continue }`. A package that uses a mock imported from elsewhere but declares none locally is dropped before its assertion sites are ever collected, and it is not added to the `skipped` slice the test errors on. Four packages with 16 assertion sites are therefore invisible: `internal/server` (6), `providers/azure/services/compute` (5), `providers/azure/services/managedredis` (3), `providers/azure/services/synapse` (2). The guard reports `checked 162 mock assertion site(s)` and passes, with zero skips reported. None of the 16 is vacuous today, because all pass matchers whose count equals the real `m.Called` arity, so this is a hole in the guard rather than a live unfailable assertion. It contradicts the guard's own stated contract, which says a skip that reports nothing is the defect the guard exists to catch. +- evidence: + ```go + mocks := collectMocks(files) + if len(mocks) == 0 { + continue + } + ``` +- suggested fix: Collect assertion sites before the `len(mocks) == 0` test and resolve them through the existing repo-wide `global` index, routing anything still unresolved into `skipped` rather than dropping the directory. +- verdict: CONFIRMED — I re-derived the drop set by replaying the guard's own `collectMocks` and assertion-site walk over the pinned tree, and the hole at internal/mocks/vacuous_assertion_guard_test.go:137 is real: the same four packages are dropped with no entry in `skipped`, and the guard run logs `checked 162 mock assertion site(s)` and PASSes. The count is **15**, not 16 — `providers/azure/services/compute/client_test.go` has 4 sites (924, 948, 971, 1499), not 5; internal/server has 6, managedredis 3, synapse 2. Two further qualifiers the finding omits: the 4 `internal/server/handler_test.go` sites are on `mocks.MockConfigStore`, which shadows both helpers (internal/mocks/stores.go:1564 records before `isExpected`), so they would classify as `verdictIgnore` even if the directory were reached; and every one of the 15 passes a matcher count equal to the real `m.Called` arity (`MockHTTPClient.Do` -> `m.Called(req)`, providers/azure/mocks/azure_mocks.go:130), confirming none is vacuous today. +- issue: (pending cross-reference) + +### A16-002 4-eyes and per-account approver matching are never tested with a case-differing email +- category: test-gap +- severity: medium +- location: internal/purchase/approvals.go:290 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: the self-approval comparison is `strings.EqualFold`, but every email literal in + internal/purchase/approvals_test.go and internal/purchase/coverage_extra_test.go is all-lowercase + and byte-identical on both sides (creator `creator@example.com` vs actor `creator@example.com` at + approvals_test.go:352/358, and `owner@example.com` on both sides at coverage_extra_test.go:358/382). + Replace `EqualFold` with `==` and the whole suite stays green, while a creator whose session email + is `Creator@example.com` self-approves their own purchase under 4-eyes. The same untested + case-folding sits in `matchActorAgainstApprovers`, which authorizes the SQS approve path against + per-account `contact_email`. +- evidence: + ```go + if strings.EqualFold(creatorEmail, actorEmail) { + logging.Warnf("purchase[%s]: 4-eyes mode on; creator %s attempted self-approval via actor %q, denied", + executionID, *execution.CreatedByUserID, maskActor(actorEmail)) + return fmt.Errorf("approval declined: 4-eyes mode requires a different approver than the requester") + } + ``` +- suggested fix: change one existing self-approve subtest to stub the creator as + `Creator@Example.COM` while the actor stays `creator@example.com`, and mirror it in the + approver-matching test. Both must still deny. +- verdict: CONFIRMED — a repo-wide scan for a mixed-case email literal (`git grep -nE '"[A-Za-z0-9._%+-]*[A-Z][A-Za-z0-9._%+-]*@'`) returns exactly two hits, neither in internal/purchase: handler_purchases_test.go:5098 covers `resolveExecutedNotificationRecipients` dedup and handler_registrations_recipients_test.go:23 covers recipient dedup, so no fixture reaches approvals.go:290 or messages.go:263-266 with differing case; messages.go:263-266 does lowercase both sides as the finding states. +- issue: (pending cross-reference) + +### A01-022 4-eyes coverage for revocation exists only as t.Skip placeholders +- category: test-gap +- severity: low +- location: internal/api/handler_purchases_test.go:5304 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: `TestRevoke_4EyesMode_TODO` and `TestRevokePurchase_FourEyesApproval` (handler_purchases_revoke_test.go:1311) are permanently skipped placeholders "until #1005 lands". #1005 (4-eyes) has landed for approval, and neither revoke path (`revokeViaSession`, `revokeAzurePurchase`) applies any dual-control check, so the skipped tests document an unenforced expectation and inflate the apparent coverage of the revoke money path. +- evidence: + ```go + func TestRevoke_4EyesMode_TODO(t *testing.T) { + t.Skip("placeholder until #1005 lands: verify that 4-eyes-mode approval flows interact correctly with the revocation window") + } + ``` +- suggested fix: Decide whether revocation is subject to dual control; implement and test it, or delete the placeholders. +- verdict: CONFIRMED — both placeholders are bare t.Skip (internal/api/handler_purchases_test.go:5304-5306, handler_purchases_revoke_test.go:1311-1313); #1005 is implemented for approvals via enforceFourEyesPolicy (internal/purchase/approvals.go:216-230, called from ApproveAndExecute:323), and neither revokeViaSession (handler_purchases.go:1446-1478) nor the revokePurchase/authorizeSessionRevoke* chain (handler_purchases_revoke.go:139-330) references requireDifferentApprover or enforceFourEyesPolicy. +- issue: (pending cross-reference) + +### A04-020 Integration test harness adds a second unique index on execution_id under a false claim about the schema +- category: test-gap +- severity: low +- location: internal/config/store_postgres_db_test.go:53-59 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: the comment states "the migration schema does not create a UNIQUE constraint on execution_id"; migration 000008 has created `unique_execution_id` since day one and every `ON CONFLICT (execution_id)` in production depends on it. The harness therefore tests against a schema with two unique indexes and would keep passing if 000008 were ever dropped, which is the only regression the extra DDL could have caught. It is a stale workaround, not coverage. +- evidence: + ```go + // The store code uses ON CONFLICT (execution_id) but the migration schema + // does not create a UNIQUE constraint on execution_id. Add it for tests. + _, err := container.DB.Exec(ctx, + "CREATE UNIQUE INDEX IF NOT EXISTS idx_purchase_executions_execution_id_unique ON purchase_executions(execution_id)") + ``` +- suggested fix: delete the DDL and the comment so the integration suite runs against exactly the migrated schema. +- verdict: CONFIRMED — internal/database/postgres/migrations/000008_add_execution_id_unique.up.sql:3 adds `CONSTRAINT unique_execution_id UNIQUE (execution_id)`, so the comment at internal/config/store_postgres_db_test.go:53-54 is false; the harness has already applied the full chain at store_postgres_db_test.go:49-51 before layering the redundant index at 55-56, which can only mask the loss of 000008. +- issue: (pending cross-reference) + +### A12-061 Both History tables stamp the same `data-execution-id`, and the fixtures invert the shipped DOM order +- category: test-gap +- severity: low +- location: frontend/src/history.ts:1792 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: `renderHistoryList` (line 1136) and `renderApprovalQueue` (line 1792) both stamp `data-execution-id`, so a pending purchase produces two rows with the same id and `applyExecutionDeepLink`'s unscoped `document.querySelector` returns the first in DOM order. In `index.html` the queue precedes the list, so production highlights the queue row; every test fixture appends `history-list` before `purchases-approval-queue`, so the tests exercise the opposite element. The deep-link tests use a hand-built single table and never see the duplicate. +- evidence: + ```ts + const execIdAttr = p.purchase_id ? ` data-execution-id="${escapeHtmlAttr(p.purchase_id)}"` : ''; + ``` +- suggested fix: Scope the deep-link query to `#history-list` (or give queue rows a distinct attribute) and reorder the fixtures to match `index.html`. +- verdict: CONFIRMED — Both renderers stamp data-execution-id (frontend/src/history.ts:1136, :1792) and applyExecutionDeepLink resolves it with an unscoped document.querySelector that takes the first in DOM order (frontend/src/history.ts:122-124); index.html puts the queue first (frontend/src/index.html:180 before :253) while the fixtures append history-list first (frontend/src/__tests__/history-approval-queue.test.ts:105-107). +- issue: (pending cross-reference) + +### A14-028 A stale comment keeps the cross-account generator test from asserting its exit code +- category: test-gap +- severity: low +- location: scripts/generate_federation_iac_test.go:428 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: The comment says `iacData` is missing `SourceAccountID / CUDlyAPIURL / ContactEmail / OIDCIssuerHost` and that "no tfvars path of this script renders on main today", and on that basis `TestGenerator_CrossAccountDoesNotRequireSubjectClaim` asserts only on stderr. Three of those four fields are now present (SourceAccountID at generate-federation-iac.go:119, CUDlyAPIURL and ContactEmail at 141-142) and `TestGenerateFederationIaC_TfvarsCombinations` exercises tfvars rendering, so the stated justification no longer holds. The test therefore passes whatever exit code the aws→aws path returns, including a regression that breaks it outright. +- evidence: + ```go + // It asserts only on stderr, not on the exit code, because that invocation + // currently fails for an unrelated pre-existing reason: iacData is missing the + // SourceAccountID / CUDlyAPIURL / ContactEmail / OIDCIssuerHost fields the + // tfvars and deploy-script templates reference, so no tfvars path of this + // script renders on main today. + ``` +- suggested fix: Fix the remaining `OIDCIssuerHost` gap (A14-026), then assert `res.exitCode == 0` and delete the stale rationale. +- verdict: CONFIRMED — TestGenerator_CrossAccountDoesNotRequireSubjectClaim (scripts/generate_federation_iac_test.go:434-446) asserts only `strings.Contains(res.stderr, "--oidc-subject-claim")`, while three of the four fields its rationale names now exist (SourceAccountID at generate-federation-iac.go:119, CUDlyAPIURL and ContactEmail at 141-142) and generate-federation-iac_test.go:40 renders tfvars successfully, so the stated justification is stale. +- issue: (pending cross-reference) + +### A15-006 A singleflight test orchestrates a ten-goroutine race with 450ms of sleeps +- category: test-gap +- severity: low +- location: internal/api/ri_utilization_cache_test.go:259 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: The test needs all ten background-refresh goroutines to reach `sf.Do` before the in-flight fetch is released, and sleeps 250ms to arrange it, then 200ms more to let `storePayload` finish before reading `calls.Load()`. Under `-race` on a loaded runner a straggler can still arrive late and open a second batch, making `calls` 2 and failing the test for scheduling reasons rather than a real defect; conversely a genuine singleflight regression that collapses slowly can pass. The in-file comment already documents the flakiness it is working around. +- evidence: + ```go + time.Sleep(250 * time.Millisecond) + + // Unblock the single in-flight fetch and wait for the collapsed + // batch's storePayload to complete so calls.Load() reflects the + // final state. + close(release) + time.Sleep(200 * time.Millisecond) + ``` +- suggested fix: Have the fetch function signal arrival on a buffered channel and read exactly ten arrivals before `close(release)`, then wait on a completion signal from `storePayload` instead of the trailing sleep. +- verdict: CONFIRMED — the two sleeps are exactly as quoted at internal/api/ri_utilization_cache_test.go:259 and :265, and the in-file comment at :251-258 documents the straggler race it is papering over, so the false-failure direction is real and self-admitted. One half of the claim is not supported: a genuine singleflight regression makes `calls` larger, not smaller (removing the collapse entirely would drive all ten fetchers through and fail at the `n != 1` check on line 269), so the test cannot pass through a slow-collapse regression the way the finding asserts. +- issue: (pending cross-reference) + +### A16-011 The parallelism bound test asserts only the ceiling, so full serialisation passes +- category: test-gap +- severity: low +- location: internal/scheduler/scheduler_test.go:1145 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: `assert.LessOrEqual(peak, 3)` is satisfied by `peak == 1`. If + `fanOutPerAccount` regressed to a serial loop, or `CUDLY_MAX_ACCOUNT_PARALLELISM` stopped being + parsed and defaulted to 1, this test still passes while a 20-account collection sweep takes 20x + longer — which on Lambda means timing out the whole cycle. The test also carries a + `time.Sleep(5 * time.Millisecond)` inside the worker purely to make overlap observable, which is + what makes the missing lower bound easy to add. +- evidence: + ```go + out, outcome := fanOutPerAccount(context.Background(), "Test", accounts, fn) + assert.Len(t, out, len(accounts), "all accounts contribute one record") + assert.LessOrEqual(t, peak.Load(), int32(3), "peak in-flight must not exceed CUDLY_MAX_ACCOUNT_PARALLELISM") + assert.Equal(t, len(accounts), outcome.SucceededCount) + assert.Zero(t, outcome.FailedCount) + ``` +- suggested fix: add `assert.Greater(t, peak.Load(), int32(1), "the fan-out must actually run + accounts concurrently")`. With 20 accounts, a limit of 3 and a 5ms body, a peak of 1 means the + limiter serialised everything. +- verdict: CONFIRMED — scheduler_test.go:1143-1147 asserts only `LessOrEqual(peak, 3)`, `Len(out, 20)`, `SucceededCount` and `Zero(FailedCount)`, every one of which a fully serial `fanOutPerAccount` satisfies. The 5ms body at scheduler_test.go:1137 and the peak tracker at 1123-1131 are already in place, so the missing lower bound is one line. +- issue: (pending cross-reference) + +### A16-012 A tautological assertion presented as "the property in its own terms" +- category: test-gap +- severity: low +- location: internal/auth/account_scope_fail_closed_test.go:74 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: `require.Error(t, err)` on line 69 has already established that `err != nil`, so + the `&& err == nil` conjunct is false and the whole `assert.False` can never fire, whatever + `IsUnrestrictedAccess(got)` returns. The comment above it claims this restates the security + property independently of the preceding assertions, which is the risk: a reader auditing this file + sees the invariant explicitly asserted and stops looking. The real protection comes from + `assert.Nil(t, got)` on line 70. The same construction is repeated at line 192. +- evidence: + ```go + require.Error(t, err, "an unestablishable scope must be refused, not treated as unrestricted") + assert.Nil(t, got) + assert.Contains(t, err.Error(), "could not be established") + // The property in its own terms: whatever comes back must not be + // readable as "all accounts". + assert.False(t, IsUnrestrictedAccess(got) && err == nil, + "a failed resolution must never yield an unrestricted scope") + ``` +- suggested fix: drop the `&& err == nil` conjunct so the assertion reads + `assert.False(t, IsUnrestrictedAccess(got))`, which is the claim the comment makes and can actually + fail. Same edit at line 192. +- verdict: CONFIRMED — `require.Error` at account_scope_fail_closed_test.go:69 aborts the subtest when `err == nil`, so by line 74 `err != nil` holds unconditionally, the `&& err == nil` conjunct is constant false and `assert.False` cannot fire for any value of `IsUnrestrictedAccess(got)`. The construction repeats verbatim at line 192, where the preceding `require.Error` sits at line 187, and in both places `assert.Nil(t, got)` (lines 70 and 189) is the assertion carrying the property. +- issue: (pending cross-reference) + +### A16-014 permissionCoveredBy never returns true in the whole suite +- category: test-gap +- severity: low +- location: internal/auth/group_ceiling.go:376 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: the `return true` is uncovered, so no test in the group-ceiling suite exercises + "this write KEEPS a permission the group already had at constraints no narrower than the request". + The false direction is well covered — the over-ceiling denial tests would fail if this returned + true unconditionally — but the true direction is not. A change that makes it always return false + turns every no-op resubmission of an existing group permission into a ceiling violation, locking a + group admin out of editing any other field on their own group, and the suite stays green. +- evidence: + ```go + func permissionCoveredBy(set []Permission, req Permission) bool { + for _, p := range set { + if p.Action != req.Action || p.Resource != req.Resource { + continue + } + if constraintsCover(p.Constraints, req.Constraints) { + return true + } + ``` +- suggested fix: add a case to internal/auth/group_ceiling_permissions_test.go where an actor whose + own permissions do NOT reach the ceiling resubmits a group's existing permission unchanged, and + assert the update succeeds. +- verdict: CONFIRMED — group_ceiling.go:376.4 is 0 in the merged profile, so the carry-through direction is genuinely never asserted. One scope correction, which also invalidates the suggested fix as written: `permissionCoveredBy` has a single call site, group_ceiling.go:86, inside the `adminCarvedOuts` branch, so a permission that is not carved out never reaches it and a resubmission of an ordinary permission goes through `grantCeilingAllows` instead. The consequence of an always-false return is therefore narrower than stated — not "every no-op resubmission", but any edit to a group that holds one of the separation-of-duties reserved pairs, which can then never be saved again. group_ceiling_permissions_test.go:220-235 documents this same distinction. +- issue: (pending cross-reference) + +### Category: duplication + +13 findings: 4 medium, 9 low. + +### A05-008 Third copy of the approval-token check, and it is the one missing the TTL guard +- category: duplication +- severity: medium +- location: internal/purchase/messages.go:240 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: The same non-empty + constant-time hash comparison exists three times: `validateApprovalToken` (approvals.go:115), `loadCancelableExecution` (approvals.go:494) and `loadAsyncExecutionForApproval` here. The first two then enforce `ApprovalTokenExpiresAt`; this one does not. An SQS approve message carrying an expired token therefore clears `verifyAsyncApprovalActor` and reaches the approver-list lookup, and is only rejected later by `ApproveExecution`'s own TTL check. The security property currently holds by accident of call ordering — any future caller of `loadAsyncExecutionForApproval` that does not re-check inherits an expiry-blind token check. +- evidence: + ```go + if execution.ApprovalToken == "" || msg.Token == "" { + return nil, fmt.Errorf("invalid approval token") + } + storedHash := sha256.Sum256([]byte(execution.ApprovalToken)) + userHash := sha256.Sum256([]byte(msg.Token)) + if subtle.ConstantTimeCompare(storedHash[:], userHash[:]) != 1 { + return nil, fmt.Errorf("invalid approval token") + } + return execution, nil + ``` +- suggested fix: have all three sites call the existing `validateApprovalToken(execution, token)` helper, which already covers the empty check, the constant-time compare and the TTL. +- verdict: CONFIRMED — validateApprovalToken (approvals.go:109-125) and loadCancelableExecution (approvals.go:494-507) both end with the `ApprovalTokenExpiresAt != nil && time.Now().After(...)` guard, while loadAsyncExecutionForApproval (messages.go:229-244) returns the execution immediately after the ConstantTimeCompare with no TTL check; the only thing that rejects the expired token afterwards is ApproveExecution's own validateApprovalToken call at approvals.go:40, one frame later in the same request. +- issue: (pending cross-reference) + +### A08-021 isPermissionError re-implements the package's own IsPermissionError, less correctly +- category: duplication +- severity: medium +- location: providers/gcp/recommendations.go:402 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: the same package already exports `IsPermissionError` (providers/gcp/permission_errors.go:29), which uses `errors.As` and also recognizes the gRPC `codes.PermissionDenied` surface. The local copy hand-rolls unwrapping via `isGoogleAPIError`, which only follows single-`Unwrap() error` chains and therefore misses an error joined with `errors.Join` or a multi-`%w` wrap, recognizes no gRPC status at all, and adds the loose "403"+"permission" substring fallback from A08-020. The `errorAs` shim (recommendations.go:417) exists only to support this duplicate. +- evidence: + ```go + var gapiErr *googleapi.Error + if ok := errorAs(err, &gapiErr); ok && gapiErr.Code == 403 { + return true + } + msg := err.Error() + return strings.Contains(msg, "403") && strings.Contains(msg, "permission") + ``` +- suggested fix: call the exported `IsPermissionError` and delete `isPermissionError`, `errorAs` and `isGoogleAPIError`. +- verdict: CONFIRMED — providers/gcp/permission_errors.go:29-42 already exports `IsPermissionError` using `errors.As` plus the gRPC `codes.PermissionDenied` surface, while the local copy (recommendations.go:402-412) unwraps through `isGoogleAPIError` (recommendations.go:430-441), which recurses only over single `Unwrap() error` chains and recognises no gRPC status, and `errorAs` (recommendations.go:417) has no other caller. +- issue: (pending cross-reference) + +### A08b-039 The three GCP service clients are near-verbatim copies of one another +- category: duplication +- severity: medium +- location: providers/gcp/services/memorystore/client.go:454 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd (twins at `cloudsql/client.go:468` and `cloudstorage/client.go:460`) +- failure scenario: `extractGCPResourceType`, `extractGCPSavings`, `termYearsFromLabel`, `skuMatchesTier`/`skuMatchesStorageClass`, `extractPriceFromSKU`, `calculateSavingsPercentage`, `fill*Pricing`, `getOrCreateBillingService`, `realBillingService` and `convertGCPRecommendation` are byte-identical (modulo the receiver type and one identifier) across the three packages. Every finding in this report that names one of them applies three times, and a fix landed in one package leaves the other two wrong — which is how the payment-option default in A08b-013 survives in all three while the Azure siblings were fixed. +- evidence: + ```go + func extractGCPSavings(rec *recommenderpb.Recommendation) float64 { + if rec.PrimaryImpact == nil { + return 0 + } + costProj := rec.PrimaryImpact.GetCostProjection() + if costProj == nil || costProj.Cost == nil { + return 0 + } + ``` +- suggested fix: lift the shared helpers into a `providers/gcp/internal/recommendations` package, mirroring what `providers/azure/internal/{pricing,recommendations}` already does for Azure. +- verdict: CONFIRMED — comparing cloudsql:442-529, cloudstorage:434-520 and memorystore:428-500 line by line, `extractGCPResourceType`, `extractGCPSavings`, `termYearsFromLabel`, the savings-percentage helper, the SKU/tier matcher, `extractPriceFromSKU`, `getOrCreateBillingService` and `realBillingService.ListSKUs` differ only by receiver and identifier; A08b-013 surviving identically in all three is the drift the finding predicts. +- issue: (pending cross-reference) + +### A12-026 Two different permission predicates gate the same cancel endpoint +- category: duplication +- severity: medium +- location: frontend/src/dashboard.ts:410 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: Both the Home widget's Cancel button and the Plans row's Disable button call `DELETE /api/purchases/planned/{id}`, but they compute eligibility differently. `canCancelUpcomingPurchase` treats `cancel-any` as full scope; `plans.ts:717` folds it into a conjunction with `canManageScheduledPurchase`, whose full-scope set is only `{admin, update-any}`. A user holding `cancel-any:purchases` but not `update-any:purchases`, looking at a scheduler-created row with a null `created_by_user_id`, sees a working Cancel button on Home and no Disable button on Plans for the same execution. +- evidence: + ```ts + if (canAccess('admin', '*') || canAccess('cancel-any', 'purchases') || canAccess('update-any', 'purchases')) return true; + // plans.ts:717-725 + const canManagePurchase = canManageScheduledPurchase(purchase); // admin | update-any | creator + const canDisablePlan = canManagePurchase && (canAccess('delete','purchases') || canAccess('cancel-any','purchases') || canAccess('cancel-own','purchases')); + ``` +- suggested fix: Extract one shared `canCancelScheduledExecution(row)` helper and call it from both surfaces. +- verdict: CONFIRMED — canCancelUpcomingPurchase grants on cancel-any alone (frontend/src/dashboard.ts:412-421) while the Plans row ANDs canManageScheduledPurchase, whose full-scope set is only admin/update-any (frontend/src/plans.ts:673-679, :717-725), and both buttons hit DELETE /api/purchases/planned/{id} (frontend/src/dashboard.ts:648-669, frontend/src/plans.ts:812-824). +- issue: (pending cross-reference) + +### A01-020 Azure exchange execute resolves the same CloudAccount three times per request +- category: duplication +- severity: low +- location: internal/api/handler_ri_exchange.go:1011 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: For a scoped session, `authorizeAzureExchangeExecution` calls `GetCloudAccountByExternalID(ctx, "azure", sub)` in `requireAzureSubscriptionScope` (805), again in `buildAzureExchangeClient` (214), and again in `resolveAzureExchangeAccountID` (1029). Three identical round-trips on the money path, and three places that must agree on the "no account registered" semantics (`unattributedAccountConstraint` vs 404 vs errNotFound). +- evidence: + ```go + if scopeErr := h.requireAzureSubscriptionScope(ctx, session, body.SubscriptionID); scopeErr != nil { + // ... + client, err := h.buildAzureExchangeClient(ctx, body.SubscriptionID) + // ... + accountID, err := h.resolveAzureExchangeAccountID(ctx, body.SubscriptionID) + ``` +- suggested fix: Resolve the account once at the top of `authorizeAzureExchangeExecution` and pass it to the scope check, the client builder and the constraint set. +- verdict: CONFIRMED — authorizeAzureExchangeExecution (internal/api/handler_ri_exchange.go:982-1022) calls requireAzureSubscriptionScope (GetCloudAccountByExternalID at :805, scoped sessions only), buildAzureExchangeClient (:214, whenever no test factory is injected) and resolveAzureExchangeAccountID (:1029), three lookups of the same subscription with three different no-account outcomes (errNotFound, 404, unattributedAccountConstraint). +- issue: (pending cross-reference) + +### A02-023 Two copies of the recommendation scope filter plus a third name-map builder +- category: duplication +- severity: low +- location: internal/api/handler_dashboard.go:168 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: `filterDashboardRecommendations` (handler_dashboard.go:168-189) and `filterRecommendationsByAllowedAccounts` (handler_recommendations.go:123-157) implement the same loop; the first swallows a `ListCloudAccounts` failure through `resolveAccountNamesByID` (scoping.go:293-296) while the second returns it as a 500, so the same user sees different outcomes for the same DB error on the dashboard and the recommendations page, and any future scope fix has to land three times. +- evidence: + ```go + nameByID := h.resolveAccountNamesByID(ctx) + filtered := make([]config.RecommendationRecord, 0, len(recs)) + for _rvc := range recs { + if recs[_rvc].CloudAccountID == nil { + continue + } + ``` +- suggested fix: Delete `filterDashboardRecommendations` and call `filterRecommendationsByAllowedAccounts` from the dashboard. +- verdict: CONFIRMED — filterDashboardRecommendations (handler_dashboard.go:168-189) and filterRecommendationsByAllowedAccounts (handler_recommendations.go:123-157) are the same scope/name-map/loop with divergent error handling: the first goes through resolveAccountNamesByID, which returns an empty map when ListCloudAccounts fails (scoping.go:293-296), while the second returns that failure as an error. +- issue: (pending cross-reference) + +### A02-025 Weak contact_email check on the public registration endpoint duplicates a stricter validator +- category: duplication +- severity: low +- location: internal/api/handler_registrations.go:429 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: `validateRegistrationRequest` accepts any string of length 5 containing "@" (including CR/LF, spaces and display-name syntax), stores it, and `notifyRegistrant` later hands it to the SES sender as the To address, while the same file already CR/LF-strips `AccountName` for header-injection defence and `validateEmailFormat` (validation.go:134) exists for exactly this purpose. `"a@b\r\nBcc: x@y"` passes. +- evidence: + ```go + if !strings.Contains(req.ContactEmail, "@") || len(req.ContactEmail) < 5 { + return NewClientError(400, "contact_email must be a valid email address") + } + ``` +- suggested fix: Replace the ad-hoc check with `validateEmailFormat` and store `mail.ParseAddress(...).Address`. +- verdict: CONFIRMED — validateRegistrationRequest (handler_registrations.go:429-431) accepts any 5+ character string containing "@" while validateEmailFormat (validation.go:134-146) exists in the same package and notifyRegistrant (handler_registrations.go:240-247) hands the stored value to the notifier; the header-injection impact is nil in practice because the address travels as an SES Destination field rather than a raw header, so this stays a duplication/weak-validation finding at low severity. +- issue: (pending cross-reference) + +### A03-023 Repeated helpers where one exists +- category: duplication +- severity: low +- location: internal/auth/service_user.go:279 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: `loadUser` normalises the lookup-or-not-found path, but five callers reimplement it inline with slightly different messages: `UpdateUserProfile` (service_user.go:660-669), `ChangePassword` (service_password.go:213-222), `CreateAPIKey`, `ListUserAPIKeys`, `authorizeAPIKeyAccess` and `lookupAPIKeyUser` (service_apikeys.go:44-53, 140-149, 190-199, 262-271). `DeleteUser` (service_user.go:625-633) inlines `checkLastAdminConstraint` (568-577). `resolver.go` computes `sessionSuffix` twice (185-188, 269-272). `redactEmail` exists three times across packages (auth/service_password.go:496, api/handler_auth.go:631, email/sender.go:665) with different masking rules. `MatchesAccount` (types.go:199) and `AccountScope.Allows` (account_scope.go:104) are the same loop. Divergent copies are where the A03-005 asymmetry (API-key path checks Active, session path does not) came from. +- evidence: + ```go + user, err := s.store.GetUserByID(ctx, userID) + if err != nil { + if errors.Is(err, pgx.ErrNoRows) { + return "", nil, fmt.Errorf("user not found") + } + return "", nil, fmt.Errorf("failed to get user: %w", err) + } + if user == nil { + return "", nil, fmt.Errorf("user not found") + } + ``` +- suggested fix: Route the inline lookups through `loadUser` (adding an `activeOnly` variant for the API-key path), call `checkLastAdminConstraint` from `DeleteUser`, extract `roleSessionName(accountID)`, and keep one `redactEmail` in `pkg/logging`. +- verdict: CONFIRMED — loadUser exists (service_user.go:279-291) while UpdateUserProfile:660-669, CreateAPIKey (service_apikeys.go:44-53) and lookupAPIKeyUser (:262-271) inline the same lookup, DeleteUser:625-633 repeats checkLastAdminConstraint:568-577 verbatim, the 8-char sessionSuffix is computed twice (resolver.go:185-188, :269-272), `func redactEmail` is defined in service_password.go:496, api/handler_auth.go:631 and email/sender.go:665 (plus redactEmailLocal in handler_notifications.go:132), and MatchesAccount (types.go:199-212) and AccountScope.Allows (account_scope.go:104-117) are the same loop. +- issue: (pending cross-reference) + +### A06-020 Cache-Control switch is duplicated between the HTTP and Lambda static paths +- category: duplication +- severity: low +- location: internal/server/static.go:39 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: `setCacheHeaders` (used by `spaHandler`) and `cacheControlForExt` (used by `serveStaticForLambda`, static.go:177) carry byte-identical extension lists and header values. Adding `.avif` or `.map` to one and not the other gives the same asset a different cache policy depending on whether the deployment is a container or a Lambda Function URL — the exact drift class this file's `resolveStaticFilePath` refactor (04-M6) was created to close. +- evidence: + ```go + func setCacheHeaders(w http.ResponseWriter, urlPath string) { + ext := strings.ToLower(path.Ext(urlPath)) + switch ext { + case ".html": + w.Header().Set("Cache-Control", "no-cache, no-store, must-revalidate") + ``` +- suggested fix: make `setCacheHeaders` call `cacheControlForExt(strings.ToLower(path.Ext(urlPath)))`. +- verdict: CONFIRMED — the two switches are byte-identical in case labels and header values (internal/server/static.go:38-50 vs 175-186), one used by `spaHandler` (static.go:33) and the other by `serveStaticForLambda` (static.go:198), with no shared helper between them. No behavioural divergence exists today; the finding is a verified duplication/drift hazard rather than a live bug. +- issue: (pending cross-reference) + +### A08b-040 Synapse and Search re-declare the shared pricing types the pricing package exists to centralize +- category: duplication +- severity: low +- location: providers/azure/services/synapse/client.go:110 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd (same at `search/client.go:115`) +- failure scenario: `pricing/types.go:3` documents `RetailPriceItem` as the single shape "so a field rename or addition only requires a single edit", and cache, cosmosdb, database and managedredis alias it. Synapse declares its own copy omitting `Location` and `UnitOfMeasure`, and search declares its own `AzureRetailPrice` envelope instead of aliasing `pricing.Page`. A new field added to the shared item (unit of measure is the one A08b-010 needs) silently decodes to zero in these two packages. +- evidence: + ```go + type SynapseRetailPriceItem struct { + CurrencyCode string `json:"currencyCode"` + RetailPrice float64 `json:"retailPrice"` + UnitPrice float64 `json:"unitPrice"` + ``` +- suggested fix: alias `pricing.RetailPriceItem` and `pricing.Page` as the four sibling clients do. +- verdict: CONFIRMED — `SynapseRetailPriceItem` (synapse:110-122) omits both `Location` and `UnitOfMeasure`, which the shared item carries at providers/azure/internal/pricing/types.go:15 and :21 with the single-edit rationale at lines 9-11; search re-declares the envelope at 115-119 rather than aliasing `pricing.Page`, though its `Items` field does use the shared item type. +- issue: (pending cross-reference) + +### A09-012 Confidence thresholds are duplicated between pkg/exchange and internal/api with the same magic numbers +- category: duplication +- severity: medium +- location: pkg/exchange/reshape.go:328 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: `confidenceComponent` reimplements `confidenceBucketFor` at `internal/api/handler_recommendations.go:382`, which itself mirrors `frontend/src/recommendations.ts`. The `200` / `50` / `3` cut-offs are inline literals on all three sides with no shared constant. Tuning the bucket in the API layer leaves the reshape ranking scoring the same offerings under the old thresholds, so the UI badge and the ordering it accompanies disagree, and only the API side has a test (`handler_recommendations_test.go:765`). +- evidence: + ```go + switch { + case savings >= 200 && count >= 3: + return 1.0 + case savings >= 50: + return 0.5 + default: + return 0.0 + } + ``` +- suggested fix: export the three thresholds as named constants from `pkg/exchange` (the module `internal/api` can import) and have `confidenceBucketFor` read them, leaving only the frontend as documented cross-language duplication. +- verdict: CONFIRMED — the Go-side duplication is real: the 200/50/3 literals appear inline in `confidenceComponent` (pkg/exchange/reshape.go:326-334) and again in `confidenceBucketFor` (internal/api/handler_recommendations.go:382-393), with a test only on the API side (handler_recommendations_test.go:765). One correction: the claimed third copy does not exist — `/usr/bin/grep -rn "confidence" frontend/src/recommendations.ts` returns nothing, and handler_recommendations.go:373-377 documents that the heuristic was moved off the client. The two Go copies have also already drifted (the API side clamps `count` to 1; the exchange side returns a neutral 0.5 when count is 0). +- severity-adjusted: low — drift risk only, with no failure scenario beyond a ranking/badge mismatch, and one of the three cited call sites does not exist. +- issue: (pending cross-reference) + +### A15-007 MockHTTPClient and createMockHTTPResponse are copy-pasted into four Azure service test files +- category: duplication +- severity: low +- location: providers/azure/services/cache/client_test.go:94 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: `providers/azure/mocks` already exports `MockHTTPClient` and `CreateMockHTTPResponse`, and `compute`, `managedredis` and `synapse` import them. Four other service test files redeclare a byte-identical local copy instead: `cache/client_test.go:94`, `cosmosdb/client_test.go:104`, `database/client_test.go:87`, `search/client_test.go:92`. A behaviour change to the shared double (adding call recording, or a `Do` that honours `ctx`) silently reaches three services and misses four, so the two halves of the Azure suite drift apart. +- evidence: + ```go + type MockHTTPClient struct { + mock.Mock + } + + func (m *MockHTTPClient) Do(req *http.Request) (*http.Response, error) { + args := m.Called(req) + if args.Get(0) == nil { + return nil, args.Error(1) + } + return args.Get(0).(*http.Response), args.Error(1) + } + ``` +- suggested fix: Delete the four local copies and import `providers/azure/mocks`, as the three sibling services already do. +- verdict: CONFIRMED — the split is exactly 4/3 and measurable: `type MockHTTPClient struct` is declared in providers/azure/mocks/azure_mocks.go:122 plus cache/client_test.go:94, cosmosdb/client_test.go:104, database/client_test.go:87 and search/client_test.go:92, and a count of `mocks.MockHTTPClient`/`mocks.CreateMockHTTPResponse` references gives 51 in compute, 41 in managedredis and 28 in synapse against 0 in all four shadowing packages. One correction: the copies are not byte-identical to the shared type, which also carries `ResponseBody string` and `StatusCode int` fields (azure_mocks.go:123-124); only the `Do` method and the response helper match verbatim, which if anything strengthens the drift argument. +- issue: (pending cross-reference) + +### A15-008 azureTermString is duplicated in six packages and reimplemented inline in a seventh +- category: duplication +- severity: low +- location: providers/azure/services/cache/client.go:591 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: The same four-line helper is declared byte-identically in `cache`, `compute`, `cosmosdb`, `database`, `managedredis` and `search`. `synapse` does not call it but recomputes the identical rule inline at `providers/azure/services/synapse/client.go:458-461`, which is why a name-based search finds six copies rather than seven. All seven encode the Retail Prices API's `"1 Year"` / `"N Years"` term spelling. If that spelling ever changes, a fix applied to the shared helper still leaves the synapse copy matching nothing, and `extractSynapsePricing` then silently finds no reservation price. +- evidence: + ```go + // synapse/client.go:458 — the seventh copy, written differently + termStr := fmt.Sprintf("%d Year", termYears) + if termYears > 1 { + termStr = fmt.Sprintf("%d Years", termYears) + } + ``` +- suggested fix: Move the helper to the shared `providers/azure/internal/pricing` package that four of these services already import, and have synapse call it instead of recomputing the rule. +- verdict: CONFIRMED — `func azureTermString` is declared in exactly six packages (cache/client.go:591, compute:757, cosmosdb:589, database:609, managedredis:511, search:502) and synapse recomputes the identical rule inline at providers/azure/services/synapse/client.go:458-461, so a name-based search does under-report the seventh; all seven of those service clients already import `providers/azure/internal/pricing`, so the suggested home is reachable without a new dependency. The failure scenario is conditional on the Retail Prices API changing its term spelling, so this stands as a duplication finding rather than a live defect. +- issue: (pending cross-reference) + +### Category: dead-code + +31 findings: 8 medium, 23 low. + +### A05-009 The whole Azure commitment-options prober is unreachable +- category: dead-code +- severity: medium +- location: internal/commitmentopts/probe_azure.go:107 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: `ProbeAzure` and `DefaultAzureProbers` have no non-test callers anywhere in the repo, and the wiring the doc comment points at, `Service.probeAndPersistAzure`, does not exist — `probeAndPersist` (service.go:76) only ranges over `s.probers`, which `AzureSPProber` cannot join because its `Probe` signature takes an `azcore.TokenCredential` rather than an `aws.Config`. So 212 lines of production code plus 304 lines of tests describe a feature that never runs, and `Validate` for `("azure", "savingsplans", …)` always takes the permissive-true branch. A reader auditing whether Azure savings-plan term/payment combos are validated will conclude they are. +- evidence: + ```go + // The method signature differs from the AWS Prober interface because Azure + // authentication uses azcore.TokenCredential rather than aws.Config. Callers + // use this method directly; Service.probeAndPersistAzure wires it up. + func (p *AzureSPProber) ProbeAzure(ctx context.Context, cred azcore.TokenCredential) ([]Combo, error) { + ``` +- suggested fix: either wire it up (an Azure sibling of `probeAndPersist` that resolves a credential and calls `ProbeAzure`) or delete the file and its tests. Leaving it in place with a comment naming a function that does not exist is the worst of the three. +- verdict: CONFIRMED — `git grep -n "ProbeAzure\|DefaultAzureProbers\|probeAndPersistAzure" -- '*.go'` matches only probe_azure.go and probe_azure_test.go; probeAndPersistAzure does not exist anywhere, probeAndPersist (service.go:76-121) ranges only over `s.probers` typed as the aws.Config-taking Prober, and the Azure prober cannot satisfy it. Validate for ("azure", ...) therefore hits either the ErrNoData branch (service.go:148) or the missing-provider branch (service.go:154) and returns permissive true. +- issue: (pending cross-reference) + +### A08-022 GroupCommitments is unused and would rebuild the #1538 wrong-commitment-type bug if called +- category: dead-code +- severity: medium +- location: providers/gcp/services/computeengine/client.go:555 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: `GroupCommitments` and `CommitmentRequest` have no caller outside their own tests. The function aggregates vCPUs and memory across every machine type in a (account, region, term) group and emits a request with no commitment `Type` at all — the exact defect `buildInsertRequest` guards against with `commitmentTypeForMachineType` and documents as issue #1538 ("an N2/C3/M3/... recommendation bought a commitment that applied to none of its instances: full commitment spend plus undimmed on-demand charges, booked as realized savings"). The generated `Name` is also timestamp-based, so a re-drive through it would double-buy. +- evidence: + ```go + result = append(result, CommitmentRequest{ + Name: fmt.Sprintf("cud-%s-%d-%d", k.region, ts, counter), + Plan: a.plan, + Region: k.region, + Resources: []ResourceCommitment{ + {Type: computepb.ResourceCommitment_VCPU.String(), Amount: a.vcpus}, + {Type: computepb.ResourceCommitment_MEMORY.String(), Amount: a.memoryMB}, + }, + }) + ``` +- suggested fix: delete `GroupCommitments`, `CommitmentRequest` and `ResourceCommitment` along with their tests; `buildInsertRequest` is the only supported way to construct a CUD insert. +- verdict: CONFIRMED — `GroupCommitments` (client.go:555) and `CommitmentRequest` (client.go:541) have no reference outside client.go and client_test.go, the emitted request carries no commitment `Type` field at all, unlike `buildInsertRequest` (client.go:758) which derives it via `commitmentTypeForMachineType` (client.go:190) for issue #1538, and the name is `time.Now().UnixNano()`-based (client.go:586, 591). +- issue: (pending cross-reference) + +### A08b-036 `convertAzureSearchRecommendation` is unreachable production code kept alive by its own tests +- category: dead-code +- severity: medium +- location: providers/azure/services/search/client.go:536 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: `GetRecommendations` at line 127 returns an empty slice unconditionally, so nothing in production calls the converter. `/usr/bin/grep -rn convertAzureSearchRecommendation providers/` returns only the definition and two calls in `client_test.go:675` and `client_test.go:691`. The tests pass, the coverage report counts the function as exercised, and the `ctx` parameter it accepts is never read — a reviewer scanning coverage sees a tested conversion path that cannot run. +- evidence: + ```go + func (c *SearchClient) GetRecommendations(_ context.Context, _ *common.RecommendationParams) ([]common.Recommendation, error) { + return []common.Recommendation{}, nil + } + ``` +- suggested fix: delete the converter and its tests until Azure exposes a Search recommendations resource type, so coverage reflects reachable code. +- verdict: CONFIRMED — `GetRecommendations` returns an empty slice unconditionally (search:127-129), and `/usr/bin/grep -rn convertAzureSearchRecommendation providers/` returns only the doc line, the definition at 536 and the two test calls at client_test.go:675 and :691; reading the body at 536-567 confirms `ctx` is never used. +- issue: (pending cross-reference) + +### A08b-037 Cloud Storage and Memorystore carry full resource-creation plumbing that no production path reads +- category: dead-code +- severity: medium +- location: providers/gcp/services/cloudstorage/client.go:26 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd (same at `memorystore/client.go:27`, partially at `cloudsql/client.go:25`) +- failure scenario: `PurchaseCommitment` was changed to a not-supported no-op and `GetExistingCommitments` returns nil, so `c.storageService` is written by `SetStorageService` and never read; `/usr/bin/grep -n storageService providers/gcp/services/cloudstorage/client.go` shows only the field, the setter and the type declarations. The same holds for memorystore's `redisService`, `CreateInstanceOperation` and `realRedisService`, and for `SQLAdminService.InsertInstance`/`ListInstances` in cloudsql. This keeps `cloud.google.com/go/storage` and the Redis create-instance surface linked in, and leaves an `Insert`/`Create` capability wired to a client whose documented purpose is that it must never create billable resources (issue #640). +- evidence: + ```go + type StorageService interface { + Buckets(ctx context.Context, projectID string) BucketIterator + Bucket(name string) BucketHandle + Close() error + } + ``` +- suggested fix: delete the unused interfaces, wrappers, setters and their mocks, so the create-resource capability is not reachable from a client that must not create resources. +- verdict: CONFIRMED — `PurchaseCommitment` is a not-supported no-op (cloudstorage:245-255) and `GetValidResourceTypes` is a hardcoded four-entry list (313-323), so `storageService` appears only as the field at 64 and the setter at 81; memorystore's `redisService` is likewise only field 65 and setter 82 with `CreateInstance` wired at 104; cloudsql reaches the API through `ListTiers` (312), leaving `ListInstances` and `InsertInstance` (26-27, 88-95) unreferenced. +- issue: (pending cross-reference) + +### A09-013 pkg/errors is a fully exported package with no importer anywhere in the repository +- category: dead-code +- severity: medium +- location: pkg/errors/errors.go:12 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: `git grep -l 'LeanerCloud/CUDly/pkg/errors"'` outside `pkg/errors/` returns zero files. The package ships 7 error types, 7 constructors, 7 `Is*` helpers, 7 field-insensitive `Is` methods and 6 sentinels, plus 312 lines of tests, none of which any caller reaches. Its field-insensitive `Is` contract is also a trap for a future adopter: `errors.Is(err, &NotFoundError{ID: "x"})` is true for any `*NotFoundError` regardless of ID. +- evidence: + ```go + // Package errors provides custom error types for CUDly. + // + // Type-level Is matching: every error type in this package implements Is by + // matching purely on the dynamic type of the target, ignoring the target's + // struct fields. + package errors + ``` +- suggested fix: delete the package, or adopt it at the boundaries that currently hand-roll `fmt.Errorf` sentinels; leaving an unused public error taxonomy invites divergent adoption later. +- verdict: CONFIRMED — I rebuilt the import graph myself rather than trusting the reviewer: `git grep -n "pkg/errors"` across the whole tree (all six modules named in go.work: root, pkg, providers/{aws,azure,gcp}, tests/e2e) returns only two go.sum lines for the unrelated `github.com/pkg/errors` dependency. No file imports `github.com/LeanerCloud/CUDly/pkg/errors`. Line counts check out: errors.go is 299 lines, errors_test.go 312, and the six sentinels are declared at pkg/errors/errors.go:281-298. +- issue: (pending cross-reference) + +### A09-014 pkg/config is dead and is the sole reason the pkg module depends on pflag and yaml.v3 +- category: dead-code +- severity: medium +- location: pkg/config/load.go:1 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: `git grep -l 'LeanerCloud/CUDly/pkg/config"'` outside `pkg/config/` returns zero files. The 399-line YAML/env/flag precedence loader, its 91-line type set and 267 lines of tests have no caller. `pkg/go.mod` carries `github.com/spf13/pflag v1.0.5` and `gopkg.in/yaml.v3 v3.0.1` as direct requirements purely for this package, so every consumer of the shared module inherits two dependencies for unreachable code. The package also defines a second `DefaultScorerConfig` that no caller can reconcile against the live scorer configuration. +- evidence: + ```go + import ( + "github.com/LeanerCloud/CUDly/pkg/scorer" + "github.com/spf13/pflag" + "gopkg.in/yaml.v3" + ) + ``` +- suggested fix: delete `pkg/config` and drop `pflag` and `yaml.v3` from `pkg/go.mod`, or move it to the root module next to the CLI that would actually use it. +- verdict: CONFIRMED — I established the import graph independently and read the package the reviewer skipped. `git grep -n "pkg/config"` across all six go.work modules returns exactly one hit, and it is a negative: pkg/scorer/scorer.go:2 says "must not import pkg/config". Nothing imports `github.com/LeanerCloud/CUDly/pkg/config`. Line counts are exact (load.go 399, types.go 91, config_test.go 267). The dependency claim holds: `git grep -n "spf13/pflag\|yaml.v3" -- pkg/` shows the only importers are pkg/config/load.go:12-13 and its test, while both sit in `pkg/go.mod`'s direct require block at lines 13 and 16. One nuance on the fix: yaml.v3 would become indirect rather than droppable, since testify pulls it in. +- issue: (pending cross-reference) + +### A13-009 The whole `terraform/modules/monitoring/` tree is never instantiated +- category: dead-code +- severity: medium +- location: terraform/modules/monitoring/README.md:38 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: No environment root declares `module "monitoring"`; grepping `modules/monitoring` across the repo hits only that README. About 2,500 lines of unexercised Terraform (aws 465 + azure 507 + gcp 659 lines of `main.tf`, plus outputs, variables and a 788-line README) declare SNS topics, KMS keys, metric filters, dashboards and alarms that no apply has ever created, and none of it is `terraform validate`-covered by an environment plan. `.trivyignore:49` accepts an SNS-encryption finding whose only source is this dead tree, which is how a suppression outlives the resource it describes. +- evidence: + ```hcl + # README shows the intended call site; no .tf file in the repo contains it + source = "../../modules/monitoring/aws" + ``` +- suggested fix: Either wire the module into the environment roots behind an `enable_monitoring` flag, or delete the tree and the suppressions that exist only for it. +- verdict: CONFIRMED — `/usr/bin/grep -rn modules/monitoring .` returns ten hits, all inside terraform/modules/monitoring/README.md; no `.tf` file anywhere declares the module. Line counts match the finding (aws/main.tf 465, azure/main.tf 507, gcp/main.tf 659, README.md 788). The suppression link holds too: `aws_sns_topic` is declared exactly once in the repo, at terraform/modules/monitoring/aws/main.tf:11, so the AVD-AWS-0136 entry at .trivyignore:49-52 has no other source. +- issue: (pending cross-reference) + +### A13c-003 Secret rotation is unappliable: the rotation Lambda points at a zip that does not exist in the repo +- category: dead-code +- severity: medium +- location: terraform/modules/secrets/aws/main.tf:282 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: `find . -name 'rotation_function*'` returns nothing at this commit, and + `terraform/modules/secrets/aws/` contains only `main.tf`, `variables.tf`, `outputs.tf` and the + lockfile. Setting `enable_secret_rotation = true` (which `environments/aws/secrets.tf:32` + exposes as `var.enable_secret_rotation`, and whose own comment tells production tfvars to set + it true) fails the apply on a missing file. `source_code_hash` already `fileexists`-guards + itself, so the absence was known; `filename` does not, so the whole rotation subsystem — the + Lambda, its role, three policies, the `aws_secretsmanager_secret_rotation` — is unreachable. +- evidence: + ```hcl + filename = "${path.module}/rotation_function.zip" + source_code_hash = fileexists("${path.module}/rotation_function.zip") ? filebase64sha256("${path.module}/rotation_function.zip") : null + ``` +- suggested fix: either ship the rotation function (or build it with `archive_file`), or delete + the rotation block and the `enable_secret_rotation` / `rotation_days` / `rds_cluster_id` + variables so no tfvars can request a configuration that cannot apply. +- verdict: CONFIRMED — `find . -name 'rotation_function*'` returns nothing and + `terraform/modules/secrets/aws/` holds only main.tf, variables.tf, outputs.tf and the lockfile; + `filename` at secrets/aws/main.tf:282 is unguarded while `source_code_hash` on the next line is + `fileexists`-guarded, and `environments/aws/secrets.tf:32` wires `var.enable_secret_rotation` + straight through, so flipping it aborts the apply on the missing zip. +- issue: (pending cross-reference) + +### A01-015 findDuplicatePendingExecution is dead code kept alive only by its test; revokeViaSession keeps a dead variable +- category: dead-code +- severity: low +- location: internal/api/handler_purchases.go:2521 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: `findDuplicatePendingExecution` has no non-test caller (`/usr/bin/grep -rn findDuplicatePendingExecution --include='*.go'` returns only its definition and `TestFindDuplicatePendingExecution`, handler_purchases_guards_test.go:481); the live duplicate check is `matchDuplicateInList` inside `persistExecutionAndSuppressions`. The test therefore pins behaviour of code the product never runs. In `revokeViaSession` the CAS result is discarded with `_ = updated // kept for future use` (1471). +- evidence: + ```go + func (h *Handler) findDuplicatePendingExecution(ctx context.Context, creatorID, key string, now time.Time) (*config.PurchaseExecution, error) { + pending, err := h.config.GetPendingExecutions(ctx) + // ... + _ = updated // kept for future use; CancelledBy is now persisted atomically + ``` +- suggested fix: Delete `findDuplicatePendingExecution` and its test (or point the test at `matchDuplicateInList`), and drop the `updated` return binding in `revokeViaSession`. +- verdict: CONFIRMED — /usr/bin/grep finds findDuplicatePendingExecution only at internal/api/handler_purchases.go:2513-2547 and handler_purchases_guards_test.go:509-564; the live duplicate check is matchDuplicateInList:2419 via persistExecutionAndSuppressions:2461, and `_ = updated` sits at :1471. +- issue: (pending cross-reference) + +### A02-014 Dead middleware and helper code, including an unreachable CSRF exemption branch +- category: dead-code +- severity: low +- location: internal/api/middleware.go:314 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: `requiresCSRFValidation` is only called from `validateSecurityContext` (handler.go:784), which returns at line 762 whenever `isPublicEndpoint` is true; every prefix in `csrfExemptWhenTokenOnly` is also in `isPublicEndpoint`, so lines 314-325 can never execute and the comment at 280-297 describes behaviour that lives elsewhere. Grep of non-test callers also finds none for `authenticate` / `checkUserAPIKey` / `checkBearerToken` (middleware.go:100-112, 219-239), `validateRequest` and `validateSecurity` (handler.go:729-732, 756-759), `formatNotFoundError` (router.go:889), `toAPIPermissions` (types_apikeys.go:34) and `formatTimePtr` (handler_apikeys.go:135). Each is exercised only by its own unit test, so the tests assert code the product never runs. +- evidence: + ```go + csrfExemptWhenTokenOnly := []string{ + "/api/purchases/approve/", + "/api/purchases/cancel/", + "/api/ri-exchange/approve/", + "/api/ri-exchange/reject/", + } + for _, prefix := range csrfExemptWhenTokenOnly { + if strings.HasPrefix(path, prefix) { + return h.extractBearerToken(req) != "" + ``` +- suggested fix: Delete the unreachable block and the unused helpers together with their tests; keep `authenticatePrincipal` as the single auth path. +- verdict: CONFIRMED — /usr/bin/grep over internal/ and cmd/ excluding *_test.go finds no callers of authenticate, validateRequest, validateSecurity, formatNotFoundError, toAPIPermissions or formatTimePtr, and checkUserAPIKey/checkBearerToken are reached only from the dead authenticate (middleware.go:107-111); requiresCSRFValidation's sole caller is validateSecurityContext (handler.go:784), which returns at handler.go:762 for every path in isPublicEndpoint, and all four csrfExemptWhenTokenOnly prefixes (middleware.go:314-319) are in that list (middleware.go:23-28), so lines 320-325 cannot execute. +- issue: (pending cross-reference) + +### A03-022 Dead code left over from the removed role model and superseded paths +- category: dead-code +- severity: low +- location: internal/auth/types.go:321 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: `RoleAdmin`, `RoleUser`, `RoleReadOnly` have zero references outside their declaration (Session.Role was removed in #907/#940); `Service.UpdateLastUsed` (service_apikeys.go:442) has no non-test caller; `ResolveGCPCredentials` (credentials/resolver.go:305) has no non-test caller; `EmailSenderInterface.SendWelcomeEmail` (interfaces.go:71) is never invoked by the auth service and still carries a `role` string parameter; `User.Salt` is written as `""` at every site and read by nothing. Each keeps the pre-#907 vocabulary alive for the next reader and widens the surface every mock has to implement. +- evidence: + ```go + // Predefined roles. + const ( + RoleAdmin = "admin" + RoleUser = "user" + RoleReadOnly = "readonly" + ) + ``` +- suggested fix: Delete the constants, the two unused functions, the unused interface method (and its mock implementations), and the `Salt` field once the column is dropped. +- verdict: CONFIRMED — repo-wide greps show RoleAdmin/RoleUser/RoleReadOnly referenced only at their declaration (types.go:321-325), no non-test caller of Service.UpdateLastUsed (service_apikeys.go:442) or ResolveGCPCredentials (resolver.go:305), SendWelcomeEmail present only in interface declarations and sender implementations with no invocation from internal/auth, and User.Salt assigned "" at service_user.go:744, service_password.go:251, :487 and only round-tripped by the store (store_postgres.go:113, :225, :451, :754). +- issue: (pending cross-reference) + +### A04-015 ri_exchange_history.cloud_account_id and RIExchangeRecord.CloudAccountID are never written or read +- category: dead-code +- severity: low +- location: internal/config/store_postgres.go:2639-2646 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd (column at migrations/000011_cloud_accounts.up.sql:140-141; field at internal/config/types.go:985) +- failure scenario: the column added for multi-tenant scoping is NULL on every row and absent from every SELECT, so `RIExchangeRecord.CloudAccountID` is always nil. Exchange-history scoping (`internal/api/handler_ri_exchange.go:2064-2085`) has to fall back to matching the provider account-number string, the same ambiguity purchase_history needed the dual-column predicate for (#701/#866). Anyone who adds a `cloud_account_id = ANY($1)` filter here would silently return zero rows. +- evidence: + ```go + INSERT INTO ri_exchange_history ( + id, account_id, exchange_id, region, source_ri_ids, + source_instance_type, source_count, target_offering_id, + target_instance_type, target_count, payment_due, + status, approval_token, error, mode, completed_at, expires_at, + created_at, updated_at, created_by_user_id, ladder_run_id + ) + ``` +- suggested fix: either populate and project the column (resolve via `GetCloudAccountByExternalID` at save time) or drop the column and the struct field so no caller can filter on it. +- verdict: CONFIRMED — `SaveRIExchangeRecord`'s twenty-one-column INSERT omits `cloud_account_id` (internal/config/store_postgres.go:2639-2651) and none of the four read projections include it (internal/config/store_postgres.go:2685-2691, 2710-2716, 2735-2743, 2980-2986), so `RIExchangeRecord.CloudAccountID` (internal/config/types.go:985) is always nil; exchange-history scoping consequently filters on the `AccountID` string (internal/api/handler_ri_exchange.go:2069-2085). Column added at internal/database/postgres/migrations/000011_cloud_accounts.up.sql:140-141. +- issue: (pending cross-reference) + +### A05-019 `maskActor`'s length guard is always true and its variable unused +- category: dead-code +- severity: low +- location: internal/purchase/approvals.go:95 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: `actor == ""` is rejected two lines earlier, so `len(actor)-1 >= 0` always holds, and `at` is never read inside the block. The `if` neither guards the loop nor communicates anything; it reads as though an index check were being performed on a PII-masking path, which is where a reader most wants the control flow to be obvious. +- evidence: + ```go + if at := len(actor) - 1; at >= 0 { + for i, c := range actor { + if c == '@' { + return "***" + actor[i:] + } + } + } + ``` +- suggested fix: drop the `if` and keep the loop; better still, use `strings.Index(actor, "@")` so the intent is stated once. +- verdict: CONFIRMED — approvals.go:89-91 returns early on `actor == ""`, so at approvals.go:95 `len(actor) >= 1` and `at := len(actor) - 1` is always >= 0; the block body (:96-100) ranges over `actor` and never reads `at`. The `if` is unconditionally taken and guards nothing. +- issue: (pending cross-reference) + +### A07-027 opensearch.matchesPaymentOption is dead production code kept alive by its own test +- category: dead-code +- severity: low +- location: providers/aws/services/opensearch/client.go:483 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: the only reference outside the definition is `client_test.go:333`. `scanOpenSearchOfferingPage` does its matching through `normalizeOpenSearchPaymentOption` instead. The two encode different behaviour for an unrecognised option (this one returns false, the live one passes the string through), so the test asserts a contract the purchase path does not use and a reader tracing payment-option handling finds the wrong function first. +- evidence: + ```go + func (c *Client) matchesPaymentOption(offeringOption types.ReservedInstancePaymentOption, required string) bool { + switch required { + case "all-upfront": + return offeringOption == types.ReservedInstancePaymentOptionAllUpfront + ``` +- suggested fix: delete the method and its test; the live behaviour is already covered through `scanOpenSearchOfferingPage`. +- verdict: CONFIRMED — a repo-wide grep for `matchesPaymentOption` finds the OpenSearch method only at its definition (providers/aws/services/opensearch/client.go:483) and at client_test.go:333; the live comparison uses `normalizeOpenSearchPaymentOption` (:448, compared at :457), and the two differ on an unrecognised option exactly as described (:492 returns false, :479 passes through). Redshift's identically-named free function is a separate symbol with real callers. +- issue: (pending cross-reference) + +### A07-032 describeInputFromQuery carries an unreachable offering-class default +- category: dead-code +- severity: low +- location: providers/aws/services/ec2/client.go:430 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: the only caller is `findOfferingID`, which sets `q.offeringClass` from `resolveOfferingClassType` on the line above; that function returns either a non-empty enum value or an error. `oc == ""` is therefore unreachable, and the duplicated default means a future caller that forgets to set the field silently buys convertible rather than failing the way `resolveOfferingClassType` was written to. +- evidence: + ```go + oc := q.offeringClass + if oc == "" { + oc = types.OfferingClassTypeConvertible + } + ``` +- suggested fix: delete the branch and let `resolveOfferingClassType` remain the single place that decides the default. +- verdict: CONFIRMED — the only production caller of `describeInputFromQuery` is providers/aws/services/ec2/client.go:522, reached from `findOfferingID` which sets `q.offeringClass` from `resolveOfferingClassType` at :495-499, and that function (:385-394) returns either a non-empty enum or an error, so `oc == ""` at :431 is unreachable outside client_test.go:1176-1182. +- issue: (pending cross-reference) + +### A08-026 fetchOnDemandRate is exercised only by its own tests +- category: dead-code +- severity: low +- location: providers/azure/services/savingsplans/client.go:455 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: no production code calls `fetchOnDemandRate`; the only callers are `client_test.go:533/552/571`. It is the sole reason the savings plans client holds an `httpClient` field at all, so the tests keep alive a code path and a dependency the client does not otherwise use. +- evidence: + ```go + func (c *Client) fetchOnDemandRate(ctx context.Context, planType string) (float64, error) { + ``` +- suggested fix: delete the function and its tests, or wire it into `GetOfferingDetails` where an on-demand rate is actually needed to report savings. +- verdict: CONFIRMED — `fetchOnDemandRate` (savingsplans/client.go:455) is referenced only at client_test.go:533/552/571, and the `httpClient` field's only production read is inside that function (client.go:465); every other mention is the constructor assignment (client.go:79, 87-94) or a test. +- issue: (pending cross-reference) + +### A08-027 RecommendationsClientAdapter stores a context it never reads +- category: dead-code +- severity: low +- location: providers/gcp/recommendations.go:73 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: `GetRecommendationsClient` (provider.go:491) populates the `ctx` field, but every method on the adapter takes its own `ctx` parameter and none reads `r.ctx`. The field invites a future edit to use the stored (typically `context.Background()`, provider.go:116) context for an RPC, which would silently ignore the caller's deadline and cancellation. +- evidence: + ```go + type RecommendationsClientAdapter struct { + ctx context.Context + projectID string + clientOpts []option.ClientOption + } + ``` +- suggested fix: drop the `ctx` field and the assignment in `GetRecommendationsClient`. +- verdict: CONFIRMED — the field is written once at provider.go:489 and never read by production code: the only `.ctx` reads in providers/gcp are assertions in provider_test.go:265/336 and recommendations_test.go:122, and every adapter method takes its own `ctx` (e.g. `GetRecommendations`, recommendations.go:103). The parenthetical is imprecise, though: the field stores the caller's `ctx` argument, not provider.go:116's `context.Background()`. +- issue: (pending cross-reference) + +### A08b-038 All four GCP clients store a `context.Context` in the struct and never read it +- category: dead-code +- severity: low +- location: providers/gcp/services/cloudsql/client.go:49 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd (same at `cloudstorage/client.go:60`, `memorystore/client.go:61`, `computeengine/client.go:255`) +- failure scenario: `NewClient` captures the caller's context into the struct, but every method takes its own `ctx` parameter and `/usr/bin/grep -n "c\.ctx"` returns no matches in any of the four files. The field is a live trap: a future method that reads it would silently use the construction-time context, which for a Lambda-scoped client is already cancelled by the time a later request runs. +- evidence: + ```go + type CloudSQLClient struct { + ctx context.Context + projectID string + region string + ``` +- suggested fix: drop the field and the `ctx` parameter from `NewClient`. +- verdict: CONFIRMED — the field is declared at cloudsql:49, cloudstorage:60, memorystore:61 and computeengine:255, each `NewClient` takes a `ctx` to fill it (cloudsql:59, cloudstorage:70, memorystore:71, computeengine:266), and `/usr/bin/grep -rn "c\.ctx" providers/gcp/` returns nothing at this commit. +- issue: (pending cross-reference) + +### A09-015 RIExchangeStore forces every implementer to carry an uncalled method +- category: dead-code +- severity: low +- location: pkg/exchange/auto.go:49 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: `CancelAllPendingExchanges` is in the interface, implemented at `internal/config/store_postgres.go:2914`, mocked at `internal/mocks/stores.go:694`, and tested — but no production caller invokes it. `RunAutoExchange` uses `CancelPendingExchangesByOrigin` instead, and the doc comment says so. The method is a live "cancel every pending exchange regardless of origin" primitive kept reachable only through the interface, which is the exact cross-origin behaviour the origin-scoped replacement was written to prevent. +- evidence: + ```go + // CancelAllPendingExchanges cancels every pending record regardless of origin. + // Kept for interface compatibility; RunAutoExchange now calls + // CancelPendingExchangesByOrigin instead to avoid cross-origin contamination. + CancelAllPendingExchanges(ctx context.Context) (int64, error) + ``` +- suggested fix: remove the method from `RIExchangeStore` (and from the store, if nothing else calls it) so the dangerous unscoped variant is not one method call away. +- verdict: CONFIRMED — `git grep -n CancelAllPendingExchanges` returns the interface declaration (pkg/exchange/auto.go:49), the store implementation (internal/config/store_postgres.go:2914), the adapter passthrough (internal/server/handler_ri_exchange.go:296), four mocks and six tests. No production code path invokes it; `RunAutoExchange` calls `CancelPendingExchangesByOrigin` at auto.go:171 instead. +- issue: (pending cross-reference) + +### A09-024 IsSavingsPlan ends with a comparison that can never be reached +- category: dead-code +- severity: low +- location: pkg/common/types.go:157 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: `ServiceSavingsPlansAll` is declared at line 107 with the value `"savingsplans"`, and the switch above already matches it. The trailing `return string(s) == "savingsplans"` can therefore only be evaluated for values the switch rejected, all of which differ from `"savingsplans"`. It always returns false and can never be the deciding branch. The doc comment claims it exists to catch "the dash-free frontend spelling that the API handler stores verbatim", implying a case the constant does not already cover. +- evidence: + ```go + case ServiceSavingsPlansAll, + ServiceSavingsPlansCompute, + ... + return true + } + return string(s) == "savingsplans" + ``` +- suggested fix: replace the trailing comparison with `return false`, and if a dash-form legacy spelling still needs recognising, match `"savings-plans"` explicitly instead. +- verdict: CONFIRMED — `ServiceSavingsPlansAll ServiceType = "savingsplans"` (pkg/common/types.go:107) is the first case of the switch at types.go:150-155, so any `s` reaching the trailing `return string(s) == "savingsplans"` at types.go:157 has already been rejected by that case and cannot equal the literal. The line always returns false. The doc comment at types.go:143-146 describes it as covering a spelling the constant already covers. +- issue: (pending cross-reference) + +### A10-026 `Config.Providers` is never bound to a flag or read +- category: dead-code +- severity: low +- location: cmd/main.go:47 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: The field is the only occurrence of the identifier in the whole `cmd/` tree (`/usr/bin/grep -rn "Providers" cmd/` returns one line). It is declared on the config struct that documents the CLI's surface, so a reader adding multi-provider support reasonably assumes a `--providers` flag exists and wires against it; nothing does. The main pipeline is AWS-only (`recClient := awsprovider.NewRecommendationsClient`, cmd/multi_service.go:122). +- evidence: + ```go + ExcludeAccounts []string + Providers []string + IncludeRegions []string + ``` +- suggested fix: Delete the field. +- verdict: CONFIRMED — `/usr/bin/grep -rn "Providers" cmd/` returns only cmd/main.go:47, and a repo-wide search for a `.Providers` selector finds no reader anywhere. +- issue: (pending cross-reference) + +### A10-027 `processService` / `processRegionRecommendations` / `applyCommonCoverage` are reachable only from tests +- category: dead-code +- severity: low +- location: cmd/multi_service.go:653 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: `processService` is called from six test functions and nowhere in production; its only caller of `processRegionRecommendations` is itself, and `applyCommonCoverage` is called once, from `multi_service_helpers_test.go:277`. The comments already concede this ("Used by legacy callers", "This legacy path (test-only)"). Roughly 140 lines of purchase-executing code, including a `--max-instances` refusal guard at cmd/multi_service_helpers.go:412 that protects a state no production caller can reach, are maintained and reviewed as if live. The tests exercising them report coverage for a pipeline the CLI never runs, which overstates how well the real path is tested. +- evidence: + ```go + // processService processes a single service and returns recommendations and results. + // Used by legacy callers; new code should use fetchAllRecs + executePurchasePipeline. + func processService(ctx context.Context, awsCfg aws.Config, ... + ``` +- suggested fix: Delete all three along with the tests that exist only to drive them, or move whatever behaviour is still worth pinning onto `fetchAndFilterRegionRecs` and `executePurchasePipeline`. +- verdict: CONFIRMED — processService's only callers are cmd/multi_service_coverage_test.go (3) and cmd/multi_service_test.go (6); processRegionRecommendations has exactly one caller, processService itself (cmd/multi_service.go:667); applyCommonCoverage has exactly one, cmd/multi_service_helpers_test.go:277. The --max-instances refusal at cmd/multi_service_helpers.go:412 therefore sits on a test-only path. +- issue: (pending cross-reference) + +### A11-020 The user role filter and the bulk-group prompt are dead code +- category: dead-code +- severity: low +- location: frontend/src/users/filters.ts:35 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: `index.html` contains no `#user-role-filter`, no `#user-group-filter` and no `#bulk-group-btn` (all three grep to zero occurrences). The role-filter state (`roleFilter`, `setRoleFilter`, the `case 'role'` branch in `handleFilterChange`, the `applyFilters` role block) and the `#bulk-group-btn` handler are therefore unreachable. The dead role branch also encodes a wrong rule if it is ever revived: `roleFilter !== 'admin' && isAdminUser` excludes admins for *any* non-admin filter value, so a "readonly" selection would return every non-admin rather than read-only users. The dead bulk handler is the only remaining caller of `prompt()` in the module, and `bulkAddToGroup` still uses a native `confirm()` (userActions.ts:200) where every sibling path uses `confirmDialog`. +- evidence: + ```typescript + // users/filters.ts:35-40 + if (roleFilter) { + const isAdminUser = Array.isArray(user.groups) && user.groups.includes(ADMINISTRATORS_GROUP_ID); + if (roleFilter === 'admin' && !isAdminUser) return false; + if (roleFilter !== 'admin' && isAdminUser) return false; + } + ``` +- suggested fix: delete the role-filter state and its handler branches along with the `#bulk-group-btn` block, and move `bulkAddToGroup`'s `confirm()` onto `confirmDialog`. +- verdict: CONFIRMED — `user-role-filter`, `user-group-filter` and `bulk-group-btn` appear only in frontend/src/users/filters.ts and users/handlers.ts, never in index.html, so the role state, the `case 'role'` branch, the `applyFilters` block at filters.ts:35-39 and the `prompt()` handler at handlers.ts:76-85 are all unreachable, and the surviving bulk path is the `#bulk-group-select` dropdown at handlers.ts:88; `bulkAddToGroup` does still call native `confirm()` (users/userActions.ts:200). The dead role rule is wrong as written, as claimed. +- issue: (pending cross-reference) + +### A12-068 `MODE_VALUES` is dead, kept alive by a `void` with an inaccurate comment +- category: dead-code +- severity: low +- location: frontend/src/riexchange.ts:75 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: The comment claims `MODE_VALUES` is used in `saveAutomationSettings`, but that function reads `modeInput.value` directly (line 2046) and there is no other reference. The `void` statement exists only to defeat the unused-variable lint, so a reader looking for the label-to-value inversion follows the comment to code that does not exist. +- evidence: + ```ts + // Suppress unused variable warning — MODE_VALUES is used in saveAutomationSettings + void MODE_VALUES; + ``` +- suggested fix: Delete `MODE_VALUES` and the `void` statement. +- verdict: CONFIRMED — MODE_VALUES is referenced only by its own definition and the `void` statement (frontend/src/riexchange.ts:71, :76), and saveAutomationSettings reads modeInput.value directly (frontend/src/riexchange.ts:2046), so the comment points at code that does not exist. +- issue: (pending cross-reference) + +### A12-072 The Archera offer modal's `'plan'` context is unreachable +- category: dead-code +- severity: low +- location: frontend/src/archera.ts:47 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: `ArcheraContext` declares `'purchase' | 'plan'` and `openArcheraOfferModal` branches on it for the headline, but the only two call sites (`app.ts:456` and `app.ts:645`) both pass `'purchase'`. The plan-creation surface named in the module docstring never opens the modal, so the `'plan'` headline is unreachable and a test asserts a string no user can see. +- evidence: + ```ts + export type ArcheraContext = 'purchase' | 'plan'; + // ... + title.textContent = context === 'plan' + ? 'Insure this plan with Archera?' + : 'Insure your commitments with Archera?'; + ``` +- suggested fix: Either wire the plan-creation path to call `openArcheraOfferModal('plan')`, or drop the parameter and the branch. +- verdict: CONFIRMED — ArcheraContext declares both values and the headline branches on them (frontend/src/archera.ts:47, :114) but both production call sites pass 'purchase' (frontend/src/app.ts:456, :645) and a test asserts the unreachable headline (frontend/src/__tests__/archera.test.ts:228); the finding's claim that the module docstring names a plan surface is the one inaccurate detail. +- issue: (pending cross-reference) + +### A13-020 The ARM template declares a `roleAssignmentGuidPrefix` parameter nothing reads +- category: dead-code +- severity: low +- location: arm/CUDly-CrossSubscription/template.json:17 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: The parameter defaults to `[newGuid()]` and is documented as the "Base GUID used to derive deterministic role assignment names", but no `variables` or `resources` expression references `parameters('roleAssignmentGuidPrefix')`; all four names come from `guid(...)` over the principal and subscription. A customer who sets it to pin assignment names sees no effect, and `newGuid()` in a default value is itself a non-deterministic expression that would make redeployment non-idempotent if the parameter were ever wired up. +- evidence: + ```json + "roleAssignmentGuidPrefix": { + "type": "string", + "defaultValue": "[newGuid()]", + "metadata": { + "description": "Base GUID used to derive deterministic role assignment names. Leave at default to auto-generate." + } + } + ``` +- suggested fix: Delete the parameter. +- verdict: CONFIRMED — `/usr/bin/grep -n "parameters(" arm/CUDly-CrossSubscription/template.json` returns six references, all to `servicePrincipalObjectId` (lines 73, 79, 88, 91, 100, 103); `roleAssignmentGuidPrefix` is declared at :17-23 and read nowhere, and the four generated names come from `guid(...)` over the principal, a literal and `subscription().subscriptionId` (the custom-role name at :32 plus the three assignment names). Every other repo hit is a scripts/testdata/role-parity fixture copied from this template. +- issue: (pending cross-reference) + +### A13b-013 `organizations:DescribeOrganization` is granted but never called +- category: dead-code +- severity: low +- location: iac/federation/aws-cross-account/cloudformation/template.yaml:138 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: `DescribeOrganization` has zero call sites in the Go tree (`grep -rn "DescribeOrganization" --include='*.go'` returns nothing at this commit), yet it is granted by the `EnableOrgDiscovery` statement here, by its Terraform twin at `iac/federation/aws-cross-account/terraform/main.tf:157`, and by the hub Lambda's `AccountDiscovery` statement in `cloudformation/stacks/CUDly/template.yaml`. It reads the organization's management-account ID, master email and feature set. A customer who enables org discovery grants CUDly organization metadata it does not use, widening the role past its call set for no benefit. +- evidence: + ```yaml + - Sid: OrganizationsDiscovery + Effect: Allow + Action: + - organizations:ListAccounts + - organizations:DescribeOrganization + Resource: "*" + ``` +- suggested fix: Drop `organizations:DescribeOrganization` from the three statements, keeping `organizations:ListAccounts` which `providers/aws/provider.go` and `internal/accounts/org_discovery.go` do call. +- verdict: CONFIRMED — `/usr/bin/grep -rn "DescribeOrganization" --include='*.go' .` returns nothing at this commit (exit 1), and the only non-Go hits are IaC: `iac/federation/aws-cross-account/cloudformation/template.yaml:138`, `iac/federation/aws-cross-account/terraform/main.tf:157`, `cloudformation/stacks/CUDly/template.yaml`, plus the deploy-side policies under `terraform/`; `organizations:ListAccounts` by contrast is genuinely used via the paginator at `providers/aws/provider.go:283-286`. +- issue: (pending cross-reference) + +### A13c-020 `aws_iam_policy.secret_read` is created and exported but never attached to any role +- category: dead-code +- severity: low +- location: terraform/modules/secrets/aws/main.tf:443 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: a grep for `secret_read_policy` across `terraform/**.tf` finds only the + module output and its pass-through at `environments/aws/outputs.tf:158` — no + `aws_iam_role_policy_attachment` anywhere. Both compute modules write their own inline + secrets policies instead. The policy is a permanent, unused IAM object that reads as the + authoritative secret-access grant, and its `data.aws_iam_policy_document.secret_read` resource + list (main.tf:431-438) already omits the credential-encryption-key ARN, so anyone who did attach + it would get a subtly incomplete grant. +- evidence: + ```hcl + resource "aws_iam_policy" "secret_read" { + name_prefix = "${var.stack_name}-secret-read-" + description = "Allow reading secrets for ${var.stack_name}" + policy = data.aws_iam_policy_document.secret_read.json + tags = var.tags + } + ``` +- suggested fix: delete the policy, the data source and both outputs, or attach it and delete the + duplicated inline policies in the Lambda and Fargate modules. +- verdict: CONFIRMED — a repo-wide grep for `secret_read` matches only the definition + (secrets/aws/main.tf:423, :443), the two module outputs (outputs.tf:67, :72) and the env + pass-through (environments/aws/outputs.tf:158); no attachment exists. The credential-encryption + key is its own resource (main.tf:209), not one of `aws_secretsmanager_secret.additional`, so the + document at :431-438 does omit it. Note the unattached state is deliberate and load-bearing: + environments/aws/ci-cd-permissions/policy_iam.tf:121 cites it to justify the + `iam:AttachRolePolicy` allowlist, so a fix must update that allowlist too. +- issue: (pending cross-reference) + +### A13c-025 `modules/registry/azure` is unreferenced dead code, and its safer defaults are what the environment overrides +- category: dead-code +- severity: low +- location: terraform/modules/registry/azure/main.tf:1 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: `grep -rn 'modules/registry'` over `terraform/**.tf` matches only + `environments/gcp/registry.tf` and `environments/aws/registry.tf`; the Azure environment + declares `azurerm_container_registry.main` inline instead. The unused module carries the + Premium-SKU policy blocks, the `acr purge` cleanup task, the AcrPull assignment and + `enable_admin_user = false` — every one of which the live inline resource lacks, including the + admin-user default that A13c-005 is about. A reader auditing "the ACR module" audits the wrong + file. +- evidence: + ```hcl + resource "azurerm_container_registry" "main" { + name = var.acr_name + sku = var.sku + admin_enabled = var.enable_admin_user + quarantine_policy_enabled = var.sku == "Premium" + } + ``` +- suggested fix: consume the module from `environments/azure/registry.tf` (or delete it), so + there is one Azure registry definition. +- verdict: CONFIRMED — `/usr/bin/grep -rn 'modules/registry' terraform/` matches only + environments/gcp/registry.tf:6 and environments/aws/registry.tf:6; the Azure environment declares + `azurerm_container_registry.main` inline at environments/azure/registry.tf:5. The unused module + does carry every element claimed: `enable_admin_user` defaulting false + (registry/azure/variables.tf:37), the Premium policy blocks (main.tf:14-28), the `acr purge` + task (:34) and its own AcrPull assignment (:67-74). +- issue: (pending cross-reference) + +### A14-018 `deploy_to=aws-only` never deploys Fargate; the Fargate caller job is unreachable +- category: dead-code +- severity: medium +- location: .github/workflows/deploy-all.yml:127 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: `deploy-aws-fargate` is written as the literal string `false` for every value of `deploy_to`, so the `deploy-aws-fargate` job's `if:` at line 185 is never true. An operator selecting `aws-only` or `all` for a disaster-recovery deploy gets Lambda only, while the aggregate summary prints "⏭️ AWS Fargate: Skipped" as if that were a configured choice. The README documents `all` as "Deploy to AWS, GCP, and Azure" and lists Fargate as job 3 in the fan-out (README.md:247, 263). +- evidence: + ```bash + # AWS Fargate (optional, can be enabled separately) + echo "deploy-aws-fargate=false" >> "$GITHUB_OUTPUT" + ``` +- suggested fix: Add an explicit `aws-fargate` value to the `deploy_to` choice list and drive the output from it, or delete the `deploy-aws-fargate` caller job and its README entry. +- verdict: CONFIRMED — deploy-all.yml:127 writes the literal `deploy-aws-fargate=false` for every `deploy_to` value, so the caller job's `if:` at line 185 can never be true. +- severity-adjusted: low — only the caller job is dead: deploy-aws-fargate.yml:34 remains directly dispatchable and README.md:262 already documents `aws-only` as "AWS Lambda only" with Fargate marked optional, so no deployment capability is actually lost. +- issue: (pending cross-reference) + +### A14-040 Unreachable duplicate range validation in the sanity CLI +- category: dead-code +- severity: low +- location: ci_cd_sanity_tests/cmd/sanity/main.go:32 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: `requireInt32Range` on line 30 already calls `os.Exit(2)` for any value outside `[1, MaxInt32]`, so the identical check on lines 32-35 can never be reached. It is the only consumer of the `math` import, and it duplicates the bound in two spellings (`1<<31-1` and `math.MaxInt32`) that must be kept in sync by hand. +- evidence: + ```go + requireInt32Range("--max-list", *maxList) + + if *maxList < 1 || *maxList > math.MaxInt32 { + fmt.Fprintf(os.Stderr, "ERROR: --max-list must be between 1 and %d, got %d\n", math.MaxInt32, *maxList) + os.Exit(2) + } + ``` +- suggested fix: Delete lines 32-35 and the `math` import, and use `math.MaxInt32` inside `requireInt32Range` in place of `1<<31-1`. +- verdict: CONFIRMED — `requireInt32Range("--max-list", *maxList)` (ci_cd_sanity_tests/cmd/sanity/main.go:30) already exits 2 on `n < 1 || n > (1<<31-1)` (main.go:16), the identical bound as the block at main.go:32-35, which is `math`'s only non-import consumer. +- issue: (pending cross-reference) + +### Category: over-engineering + +3 findings: 3 low. + +### A03-020 Post-fetch constant-time compares guard nothing the SQL equality did not already leak +- category: over-engineering +- severity: low +- location: internal/auth/service.go:334 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: The row is fetched by `WHERE token = $1` (and `key_hash = $1`, `password_reset_token = $1`); whatever timing the database comparison leaks has already leaked before `subtle.ConstantTimeCompare` runs, and the compare can only be false if the store returned a row for a different key, which is a store bug, not an attack. The comments at service.go:331-333, service_apikeys.go:300-302, service_password.go:423 and :457 claim the check "closes the SQL-equality timing oracle"; it does not, and a reader relying on that claim will not look for a real mitigation. Four sites, plus two tests that pin the theatre (service_test.go:194-220, service_password_test.go:624-648). +- evidence: + ```go + // Constant-time comparison after fetch to close the SQL-equality timing oracle. + // GetSession does `WHERE token = $1` which is not constant-time in PostgreSQL; + // align with ValidateUserAPIKey / validateResetToken (issue #392 PR #837). + if subtle.ConstantTimeCompare([]byte(session.Token), []byte(hashedToken)) != 1 { + return nil, fmt.Errorf("session not found") + } + ``` +- suggested fix: Remove the four compares and their tests, or replace the comment with the true statement (the lookup key is a SHA-256 of a 256-bit random token, so equality timing reveals nothing usable). +- verdict: CONFIRMED — every compared value is the same hash the row was just selected by: hashSessionToken is SHA-256 of a 32-byte random token (service_helpers.go:51-54, :121-127) and GetSession selects `WHERE token = $1` on it (store_postgres.go:670), ValidateUserAPIKey selects by the SHA-256 keyHash (service_apikeys.go:289-292) and validateResetToken by the hashed token (service_password.go:444-446), so the compares at service.go:334, service_apikeys.go:303, service_password.go:424 and :458 can only fail on a store bug and the comments' timing-oracle claim is not what the code closes. +- issue: (pending cross-reference) + +### A05-018 `collectWithService` and `dedupeRaw` are inert +- category: over-engineering +- severity: low +- location: internal/commitmentopts/probe.go:594 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: `collectWithService(service, raw)` is `return collect(service, raw)` with nothing else, and its comment explains a distinction that does not exist — `collect` already takes the service name as its first argument. `dedupeRaw` (probe.go:570) dedupes on `(durationSeconds, payment)` immediately before `collect`, which dedupes on the normalized `(term, payment)`; every pair `dedupeRaw` removes would be removed again one call later, so deleting it changes no output. Both are correct and tested, and both guard nothing. +- evidence: + ```go + func collectWithService(service string, raw []rawOffer) []Combo { + return collect(service, raw) + } + ``` +- suggested fix: call `collect(serviceKey, raw)` directly at probe.go:562 and drop both helpers. +- verdict: CONFIRMED — collectWithService (probe.go:594-596) is a one-line pass-through to collect, whose own first parameter is already the service name (probe.go:105). dedupeRaw keys on `(durationSeconds, payment)` (probe.go:570-585) and collect keys on `(durationToTerm(durationSeconds), normalizePayment(payment))` (probe.go:112-124); since both normalizers are pure functions of those same two fields, every pair dedupeRaw drops maps to a key collect would drop one call later, and collect logs nothing and has no other side effect, so removing dedupeRaw cannot change the output. +- issue: (pending cross-reference) + +### A12-062 `canAccess('admin','*')` in the cancel and revoke gates is dead +- category: over-engineering +- severity: low +- location: frontend/src/history.ts:540 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: `cancel-any:purchases` and `revoke-any:purchases` are not in `ADMIN_CARVED_OUTS` (`permissions.ts:173`), so `canAccess('cancel-any','purchases')` already returns true for an `admin:*` holder on both the effective-permissions path and the loading fallback. Deleting the `canAccess('admin','*') ||` term changes no outcome. Its presence here but deliberate absence from `rbacAllowsApprove` and `canRetryFailedRow` reads as an intentional asymmetry, inviting a future edit to "fix" the approve path by adding it back and reopening the #923 carve-out. +- evidence: + ```ts + if (canAccess('admin', '*') || canAccess('cancel-any', 'purchases')) return true; + // line 677 + if (canAccess('admin', '*') || canAccess('revoke-any', 'purchases')) return true; + ``` +- suggested fix: Drop the redundant term from both so all four row predicates gate on the verb alone. +- verdict: CONFIRMED — cancel-any:purchases and revoke-any:purchases are absent from ADMIN_CARVED_OUTS (frontend/src/permissions.ts:173-182), so canAccess already returns true for an admin:* holder on both the effective-permissions path and the loading fallback (frontend/src/permissions.ts:333, :350), making the leading term at frontend/src/history.ts:540 and :677 outcome-neutral. +- issue: (pending cross-reference) + +### Category: ops + +37 findings: 4 high, 15 medium, 18 low. + +### A13c-002 Lambda log group uses `name_prefix`, so Lambda's real log group is unmanaged — retention never applies and the migration alarm can never fire +- category: ops +- severity: high +- location: terraform/modules/compute/aws/lambda/main.tf:506 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: Lambda writes to `/aws/lambda/` exactly. `name_prefix` makes + Terraform create `/aws/lambda/-` instead, and `aws_lambda_function.main` + carries no `logging_config` block redirecting output. Two consequences: (a) `retention_in_days` + is applied to a permanently empty group while the real group is auto-created by Lambda with + never-expire retention, so logs accumulate forever and `lambda_log_retention_days` is a no-op; + (b) `migration-alarm.tf:33` points the metric filter at `aws_cloudwatch_log_group.lambda.name`, + so even with `enable_migration_alarm = true` the filter watches the empty group and the alarm + can never leave `notBreaching`. +- evidence: + ```hcl + resource "aws_cloudwatch_log_group" "lambda" { + name_prefix = "/aws/lambda/${aws_lambda_function.main.function_name}-" + retention_in_days = var.log_retention_days + tags = var.tags + } + ``` +- suggested fix: use `name = "/aws/lambda/${aws_lambda_function.main.function_name}"` (importing + the existing auto-created group), or set `logging_config { log_group = ... }` on the function. +- verdict: CONFIRMED — Lambda writes to `/aws/lambda/${var.stack_name}-api` (function_name at + lambda/main.tf:29) and holds `AWSLambdaBasicExecutionRole` (main.tf:221) so it auto-creates that + group with never-expire retention, while `name_prefix` at main.tf:507 makes Terraform manage a + different, suffixed group; `/usr/bin/grep -rn logging_config terraform/` returns nothing, so + nothing redirects the function's output, and migration-alarm.tf:33 points the metric filter at + `aws_cloudwatch_log_group.lambda.name` — the suffixed, permanently empty group. +- issue: (pending cross-reference) + +### A14-002 The rollback workflow pins a Terraform version the modules reject +- category: ops +- severity: high +- location: .github/workflows/rollback.yml:49 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: `TF_VERSION: '1.6.0'` is used by all four rollback jobs (lines 234, 338, 408, 499), each of which runs `terraform init` inside `terraform/environments/{aws,gcp,azure}`. Every one of those roots declares `required_version = ">= 1.10.0"` (aws/main.tf:5, gcp/main.tf:5, azure/main.tf:5). `terraform init` aborts with "Unsupported Terraform Core version" before the backend is even configured, so an operator invoking the emergency rollback path during an incident gets a hard failure on every cloud. rollback.yml is the sole outlier: every other workflow pins 1.10.0 or 1.10.5. +- evidence: + ```yaml + env: + TF_VERSION: '1.6.0' + ``` +- suggested fix: Change to `'1.10.0'` to match the other deploy/destroy workflows. +- verdict: CONFIRMED — rollback.yml:49 sets `TF_VERSION: '1.6.0'`, consumed by setup-terraform at lines 234/338/408/499, and all four jobs `terraform init` in roots declaring `required_version = ">= 1.10.0"` (terraform/environments/{aws,gcp,azure}/main.tf:5); every other workflow pins 1.10.0 or 1.10.5. +- issue: (pending cross-reference) + +### A14-003 database-migration.yml runs `terraform init` with no backend config, so it can never read a DB endpoint +- category: ops +- severity: high +- location: .github/workflows/database-migration.yml:296 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: All three `Get database endpoint from Terraform` steps (296, 410, 502) run a bare `terraform init`. `terraform/environments/{aws,gcp,azure}/backend.tf` each declare an empty partial backend block ("Configuration provided via -backend-config flag"), so `init` fails on the missing required `bucket`/`storage_account_name` argument. `TF_BACKEND_AWS/GCP/AZURE` is never referenced anywhere in this file. Even if init somehow succeeded it would attach to a single default state object with no `github-` key, so a `prod` migration would read a `dev` endpoint. The result is that the only documented migration path (README.md:293-358) cannot run at all, and no guard covers it: `scripts/test-aws-tfstate-platform-key.sh` only inspects jobs that apply a `compute_platform`, which these do not. +- evidence: + ```bash + cd terraform/environments/aws + terraform init + DB_ENDPOINT=$(terraform output -raw database_proxy_endpoint 2>/dev/null || echo "") + if [ -z "$DB_ENDPOINT" ]; then + echo "Failed to get database endpoint" + exit 1 + fi + ``` +- suggested fix: Write `/tmp/backend.tfbackend` from `secrets.TF_BACKEND_` plus the environment-scoped key, the way every deploy job does, and pass `-backend-config`. +- verdict: CONFIRMED — database-migration.yml:296/410/502 run a bare `terraform init` against partial backends (terraform/environments/{aws,gcp,azure}/backend.tf each declare an empty block "provided via -backend-config"), `TF_BACKEND` appears nowhere in the file, and scripts/test-aws-tfstate-platform-key.sh:4 only scans jobs applying a `compute_platform`. +- issue: (pending cross-reference) + +### A15-001 npm audit fails on the pinned commit: fast-uri 3.1.5 carries four high-severity advisories +- category: ops +- severity: high +- location: frontend/package-lock.json:5754 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: The committed lockfile pins `fast-uri@3.1.5`, which is inside the advisory range `>= 3.1.2, < 3.1.6`. `.github/workflows/ci.yml:663` runs `npm audit --audit-level=high`; run against this lockfile it exits 1, so the Security job's npm audit step is red on this commit. Measured locally after `npm ci --ignore-scripts`: `npm audit --audit-level=high` -> exit 1, `1 high severity vulnerability`. +- evidence: + ```json + "node_modules/fast-uri": { + "version": "3.1.5", + "dev": true, + ``` + Four GHSAs: GHSA-5jgf-p345-68v8, GHSA-f65p-4m7j-42xc, GHSA-fph4-wmhf-6fwf, GHSA-jqff-g426-hqxp (host confusion / SSRF via IDN, IPv6 and percent-decoding normalization). First patched version 3.1.6 per the GitHub advisory API. + Reachability: `npm ls fast-uri --omit=dev` returns empty. It is reached only through devDependencies (`babel-loader` -> `schema-utils` -> `ajv`, and `serve` -> `ajv`), and production `dependencies` are only `@types/qrcode`, `chart.js`, `qrcode`. It does not enter the shipped webpack bundle, so no end user of the deployed frontend is exposed. The live impact is the red CI gate, not runtime SSRF. +- suggested fix: Refresh the lockfile so `fast-uri` resolves to `3.1.6` (a patch bump inside the existing semver range, no dependency change needed). +- verdict: CONFIRMED — reproduced in the pinned worktree: `npm audit --audit-level=high` in `frontend/` exits 1 with `1 high severity vulnerability` and all four GHSAs against `fast-uri@3.1.5` (frontend/package-lock.json:5755, `"dev": true`), while `npm ls fast-uri --omit=dev` returns empty and frontend/package.json:49-53 lists only `@types/qrcode`, `chart.js`, `qrcode` in `dependencies`, so the impact is the red gate at .github/workflows/ci.yml:663 and not runtime SSRF; the only correction is the advisory range, which npm reports as `3.0.0 - 3.1.5` (per-GHSA `>=3.1.3 <3.1.6`) rather than the finding's `>= 3.1.2, < 3.1.6`. +- issue: (pending cross-reference) + +### A07-017 Three Cost Explorer pagination loops have no page cap +- category: ops +- severity: medium +- location: providers/aws/recommendations/coverage.go:266 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: `fetchCoveragePaged`, `GetRIUtilization` (utilization.go:69) and `fetchDailyCoverage` (usage_history.go:98) loop on `NextPageToken` until the token is empty, with no ceiling. Every other CE loop in the package guards with `maxRecommendationPages`, `maxSPCoveragePages` or `maxOnDemandSeriesPages` and the comments on those constants state the reason: a token loop or API misbehaviour otherwise spins while billing per call. A repeated token from CE turns any of these three into an unbounded billed loop that only ends when the context deadline fires. +- evidence: + ```go + var token *string + for { + if err := ctx.Err(); err != nil { + return fmt.Errorf("coverage: pagination cancelled: %w", err) + } + input.NextPageToken = token + result, err := c.fetchCoveragePage(ctx, input) + ``` +- suggested fix: add the same page-index cap and diagnostic error the sibling loops use to all three. +- verdict: CONFIRMED — `fetchCoveragePaged` (providers/aws/recommendations/coverage.go:266-288), `GetRIUtilization` (utilization.go:68-89) and `fetchDailyCoverage` (usage_history.go:98-111) all loop on the token with no page index, while the siblings cap at client.go:181 (`maxRecommendationPages`), sp_coverage.go:334 (`maxSPCoveragePages`) and ondemand_series.go:130 (`maxOnDemandSeriesPages`); `fetchDailyCoverage` additionally has no `ctx.Err()` check at the top of its loop. +- issue: (pending cross-reference) + +### A08b-003 Hardened transport drops proxy support, connection reuse limits and HTTP/2 +- category: ops +- severity: medium +- location: pkg/httpclient/httpclient.go:60 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: the transport is constructed with only `DialContext` and `TLSHandshakeTimeout`. `Proxy` is nil, so `HTTPS_PROXY` is ignored and every Azure/pricing call fails in a network that requires an egress proxy. `IdleConnTimeout` is zero, so idle keep-alive connections are never reaped, and `MaxIdleConns`/`MaxIdleConnsPerHost` default to 2 per host, throttling the multi-page pricing walks. Setting a custom `DialContext` also disables the automatic HTTP/2 upgrade that `http.DefaultTransport` gets via `ForceAttemptHTTP2`. +- evidence: + ```go + transport := &http.Transport{ + DialContext: dialer.DialContext, + TLSHandshakeTimeout: tlsHandshakeTimeout, + } + ``` +- suggested fix: start from `http.DefaultTransport.(*http.Transport).Clone()`, then override `DialContext`, so proxy, idle-pool and HTTP/2 defaults are preserved. +- verdict: CONFIRMED — the transport at pkg/httpclient/httpclient.go:60-63 sets only `DialContext` and `TLSHandshakeTimeout`, leaving `Proxy` nil, `IdleConnTimeout` zero and `ForceAttemptHTTP2` false; one correction, zero `MaxIdleConns` means unlimited, it is `MaxIdleConnsPerHost` alone that falls back to 2. +- issue: (pending cross-reference) + +### A09-003 Hardened transport silently disables proxy support, HTTP/2 and idle-connection expiry +- category: ops +- severity: medium +- location: pkg/httpclient/httpclient.go:60 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: a hand-built `&http.Transport{}` has `Proxy: nil`, unlike `http.DefaultTransport` which uses `http.ProxyFromEnvironment`. In a VPC or Lambda deployment whose only egress path is an `HTTPS_PROXY`, every call through this client dials the origin directly and times out after 10s at the dialer, with no indication that the proxy was ignored. `ForceAttemptHTTP2` also defaults false here, and the absent `IdleConnTimeout`/`MaxIdleConns` means pooled connections are never reaped in a long-lived process. +- evidence: + ```go + transport := &http.Transport{ + DialContext: dialer.DialContext, + TLSHandshakeTimeout: tlsHandshakeTimeout, + } + ``` +- suggested fix: start from `http.DefaultTransport.(*http.Transport).Clone()` and override only `DialContext` and `TLSHandshakeTimeout`, so the proxy, HTTP/2 and idle-pool defaults survive. +- verdict: CONFIRMED — `New` builds a bare `&http.Transport{}` with only `DialContext` and `TLSHandshakeTimeout` set (pkg/httpclient/httpclient.go:60-63), so `Proxy`, `ForceAttemptHTTP2`, `IdleConnTimeout` and `MaxIdleConns` all take their zero values rather than `http.DefaultTransport`'s. +- issue: (pending cross-reference) + +### A13-003 The legacy cross-account CloudFormation stack is missing three grants and is excluded from the parity guard +- category: ops +- severity: medium +- location: cloudformation/stacks/CUDly-CrossAccount/template.yaml:159 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: A customer onboarded with this template gets a role lacking `savingsplans:DescribeSavingsPlansOfferings`, `ec2:GetReservedInstancesExchangeQuote` and `ec2:AcceptReservedInstancesExchangeQuote`, all three of which the other four federation flavors grant. Savings Plans purchase fails at offering lookup and RI exchange fails at quote time, in that account only, with an AccessDenied that looks like a customer misconfiguration. `scripts/check-aws-iam-parity.sh` compares seven files (lines 120-152) and this is not one of them, so the drift is invisible in CI. +- evidence: + ```yaml + - savingsplans:DescribeSavingsPlans + - savingsplans:CreateSavingsPlan + - savingsplans:DescribeSavingsPlansOfferingRates + ``` +- suggested fix: Either add this template to comparison 3 in `scripts/check-aws-iam-parity.sh` and bring its action list into parity, or delete it if `iac/federation/aws-cross-account/` has superseded it. +- verdict: CONFIRMED — a per-file grep count for the three actions returns 1 in all six files the parity script compares (iac/federation/aws-{cross-account,target}/{cloudformation,terraform} and internal/iacfiles/templates/aws-{cross-account,wif}-cli.sh.tmpl) and 0 in cloudformation/stacks/CUDly-CrossAccount/template.yaml, whose SavingsPlans statement stops at `DescribeSavingsPlansOfferingRates` (lines 152-158) and whose EC2 statement has no exchange verbs (lines 97-107). The script's three comparisons (scripts/check-aws-iam-parity.sh:120-152) never name this file. +- issue: (pending cross-reference) + +### A13-006 The production Azure Key Vault is default-deny with an empty allowlist, so no principal can reach it +- category: ops +- severity: high +- location: terraform/environments/azure/github-prod.tfvars:52 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: `github-prod.tfvars` sets neither `key_vault_default_network_acl_action` (default `"Deny"`, variables.tf:167) nor `allowed_ip_addresses` (default `[]`), and `create_private_subnet` defaults to `false`, so `allowed_subnet_ids` is `[]` too. The rendered `network_acls` block is `default_action = "Deny"`, `ip_rules = []`, `virtual_network_subnet_ids = []`, `bypass = "AzureServices"`. A GitHub Actions runner is not an Azure service, so every `azurerm_key_vault_secret` write in the prod apply is refused and the deploy fails; the Container App cannot read the vault at runtime either. The staging file's own comment (github-staging.tfvars:56) says production "should keep the default Deny and supply allowed_ip_addresses" and prod supplies none. +- evidence: + ```hcl + # github-prod.tfvars — the whole Key Vault block + key_vault_sku = "standard" + soft_delete_retention_days = 90 + purge_protection_enabled = true + ``` +- suggested fix: Add `allowed_ip_addresses` for the deploy runner egress to `github-prod.tfvars`, or set `create_private_subnet = true` there so the private subnet's `Microsoft.KeyVault` service endpoint is allowlisted. +- verdict: CONFIRMED — the whole of terraform/environments/azure/github-prod.tfvars sets only `key_vault_sku`, `soft_delete_retention_days` and `purge_protection_enabled`; it never sets `key_vault_default_network_acl_action` (default `"Deny"`, variables.tf:167-175), `allowed_ip_addresses` (default `[]`, variables.tf:161-165) or `create_private_subnet` (default `false`, variables.tf:96-100), so secrets.tf:37-38 renders `ip_rules = []` and `virtual_network_subnet_ids = []` into terraform/modules/secrets/azure/main.tf:46-51. github-staging.tfvars:56-58 sets `"Allow"` with the comment that prod "should keep the default Deny and supply allowed_ip_addresses", and .github/workflows/deploy-azure.yml:207 selects the file by environment name, so the prod apply really does run against a default-deny vault with an empty allowlist. +- severity-adjusted: medium — the failure is a hard, immediately visible apply failure on a path that has evidently not been exercised, not a silent security or money defect. +- issue: (pending cross-reference) + +### A13-011 `timestamp()` in the image tag forces a rebuild and push on every apply, making the content-hash triggers dead +- category: ops +- severity: medium +- location: terraform/modules/build/main.tf:39 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: `local.timestamp` is `formatdate(..., timestamp())`, which is re-evaluated on every plan, so `terraform_data.image_tag.output` and hence `local.image_tag` change on every run. `image_tag` is itself one of `terraform_data.docker_build.triggers_replace`, so the build resource is replaced on every apply regardless of the four file-hash triggers above it. Those hashes (`go_mod`, `go_sum`, `dockerfile`, `cmd_files`, `pkg_files`) can never prevent a rebuild, so every apply pays a full multi-stage Docker build and pushes a new image layer set to ECR/ACR/Artifact Registry even when nothing changed. The hashes also cover only `cmd/` and `pkg/`, not `internal/` or `frontend/`, so they would miss most real source changes if they were load-bearing. +- evidence: + ```hcl + timestamp = var.skip_docker_build ? "skip" : formatdate("YYYYMMDDhhmmss", timestamp()) + ... + triggers_replace = { + go_mod = fileexists(...) ? filemd5(...) : "none" + image_tag = local.image_tag + platform = local.effective_platform + } + ``` +- suggested fix: Derive the tag from the content hash (the existing `sha256(...)` of the source files plus `git_commit`) instead of `timestamp()`, and drop the now-redundant per-file triggers or extend them to `internal/` and `frontend/`. +- verdict: CONFIRMED — terraform/modules/build/main.tf:39 sets `timestamp = formatdate("YYYYMMDDhhmmss", timestamp())`, feeding `terraform_data.image_tag.input` at line 24 and `local.image_tag` at line 43, which is itself a `triggers_replace` key at line 46 alongside the five file hashes at lines 57-62. The escape hatch is not used: `var.custom_image_tag` is referenced only at main.tf:24 and set by no environment (`/usr/bin/grep -rn custom_image_tag terraform .github` finds only the declaration, the default `""`, and a comment in deploy-aws-lambda.yml:185), so the timestamp branch is always taken and the hashes can never gate a rebuild. The coverage gap is real too: the fileset globs cover `cmd/` and `pkg/` only, not `internal/` or `frontend/`. +- issue: (pending cross-reference) + +### A13-012 `enable_migration_alarm` promises a bootstrap grant that the bootstrap policy does not contain +- category: ops +- severity: medium +- location: terraform/modules/compute/aws/lambda/variables.tf:19 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: The variable description tells the operator that `logs:PutMetricFilter` "is granted via the ci-cd-permissions bootstrap" and to "set true only after re-applying the bootstrap". The `CloudWatchLogs` statement in `policy_compute.tf:33-47` grants only `CreateLogGroup`, `DeleteLogGroup`, `ListTagsForResource`, `PutRetentionPolicy`, `TagResource` and `UntagResource`; no metric-filter action exists anywhere under `ci-cd-permissions/`. An operator who re-applies the bootstrap and then flips the flag gets an AccessDenied on `aws_cloudwatch_log_metric_filter.migration_failed`, which fails the apply and blocks every deploy, the exact outcome the gate was added to prevent. +- evidence: + ```hcl + description = "... requires logs:PutMetricFilter on the deploy SA, which is granted via the ci-cd-permissions bootstrap (root CLAUDE.md CI/CD IAM split). ... set true only after re-applying the bootstrap so the deploy role can manage the filter." + ``` +- suggested fix: Add `logs:PutMetricFilter`, `logs:DeleteMetricFilter` and `logs:DescribeMetricFilters` scoped to the existing log-group ARNs in `policy_compute.tf`, or reword the description to say the grant does not exist yet. +- verdict: CONFIRMED — `/usr/bin/grep -rn 'PutMetricFilter|DescribeMetricFilters|DeleteMetricFilter' terraform cloudformation iac` matches only the two prose mentions (lambda/variables.tf:19 and lambda/migration-alarm.tf:25); the `CloudWatchLogs` statement at policy_compute.tf:32-48 stops at `CreateLogGroup`, `DeleteLogGroup`, `ListTagsForResource`, `PutRetentionPolicy`, `TagResource`, `UntagResource`, with `logs:DescribeLogGroups` split out at :49-53. `logs:*` at policy_boundary.tf:173 is the workload permissions boundary, which caps rather than grants and does not apply to the deploy role. So an operator who re-applies the bootstrap and flips `enable_migration_alarm = true` still 403s on `aws_cloudwatch_log_metric_filter.migration_failed` (migration-alarm.tf:24-45). +- issue: (pending cross-reference) + +### A13c-007 Azure Logic App recurrence triggers embed `timestamp()`, producing a diff on every plan +- category: ops +- severity: medium +- location: terraform/modules/compute/azure/container-apps/scheduled-tasks.tf:143 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: `timestamp()` is evaluated at every plan and apply, so `start_time` changes + on each run and all three recurrence triggers (`daily` at :143, `ri_exchange` at :263, + `cleanup_daily` at :358) show a permanent update. `terraform plan -detailed-exitcode` can never + return "no changes" on an Azure deployment, which removes drift detection as a signal and + re-anchors the schedule window on every deploy. +- evidence: + ```hcl + start_time = "${formatdate("YYYY-MM-DD", timestamp())}T${format("%02s", local.schedule_hour)}:00:00Z" + ``` +- suggested fix: drop `start_time` (Logic Apps starts the recurrence at creation) or make it a + fixed input date, and add `lifecycle { ignore_changes = [start_time] }` for existing state. +- verdict: CONFIRMED — all three recurrence triggers embed `timestamp()` in `start_time` + (scheduled-tasks.tf:143, 263, 358) and the file's only `lifecycle` block is at line 55, on an + unrelated resource, so no `ignore_changes` suppresses the diff. +- issue: (pending cross-reference) + +### A13c-013 Azure PostgreSQL Flexible Server has no deletion protection of any kind, while the AWS and GCP twins do +- category: ops +- severity: medium +- location: terraform/modules/database/azure/main.tf:19 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: `aws_db_instance.main` carries `deletion_protection` (default true) plus + `final_snapshot_identifier`; `google_sql_database_instance.main` carries + `deletion_protection`. `azurerm_postgresql_flexible_server` has neither an equivalent argument + set nor a `lifecycle { prevent_destroy }`, so a `terraform destroy`, a `-target` mistake, or any + ForceNew change (`administrator_login`, `delegated_subnet_id`, `zone` outside the ignore) drops + the production database with only the automatic backup as recourse. +- evidence: + ```hcl + resource "azurerm_postgresql_flexible_server" "main" { + name = "${var.app_name}-postgres" + administrator_login = var.administrator_login + backup_retention_days = var.backup_retention_days + lifecycle { + ignore_changes = [zone] + } + } + ``` +- suggested fix: add `prevent_destroy = var.deletion_protection`-equivalent guarding (a + `lifecycle { prevent_destroy = true }` in the module, or a `azurerm_management_lock` on the + server gated by an input) so the three providers share a posture. +- verdict: CONFIRMED — `azurerm_postgresql_flexible_server.main` spans database/azure/main.tf:19-76 + and its only `lifecycle` block is `ignore_changes = [zone]` at :73-75; a repo-wide grep for + `deletion_protection|prevent_destroy|management_lock` under `terraform/modules/database/` matches + only the AWS (main.tf:152, variables.tf:87 default true) and GCP (main.tf:42) modules. +- issue: (pending cross-reference) + +### A14-007 The README's GitHub Environment names do not match the ones the workflows bind to +- category: ops +- severity: medium +- location: .github/workflows/README.md:521 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: The setup instructions tell the operator to create `aws-lambda-dev/staging/prod`, but `deploy-aws-lambda.yml:231` binds `environment: ${{ needs.prepare.outputs.target_environment }}`, i.e. plain `dev`/`staging`/`prod`. The rollback jobs bind `aws-lambda--rollback` / `aws-fargate--rollback` / `gcp--rollback` / `azure--rollback`, and the migration jobs bind `aws-db-` / `gcp-db-` / `azure-db-`; none of those appear in the README list. Required-reviewer and deployment-branch-policy rules configured per these instructions land on environments nothing binds to, while the environments that credentialed jobs actually use are auto-created bare. Several workflow headers state that the environment binding is the only control that can gate an unapproved production deploy or destroy. +- evidence: + ```markdown + 2. Create environments: + - `aws-lambda-dev`, `aws-lambda-staging`, `aws-lambda-prod` + - `aws-fargate-dev`, `aws-fargate-staging`, `aws-fargate-prod` + - `gcp-dev`, `gcp-staging`, `gcp-prod` + - `azure-dev`, `azure-staging`, `azure-prod` + ``` +- suggested fix: Regenerate the list from the actual `environment:` expressions across the workflows, including the `-rollback` and `-db-` families. +- verdict: CONFIRMED — README.md:521-524 lists `aws-lambda-*`/`gcp-*`/`azure-*`, but deploy-aws-lambda.yml:231,377 bind plain `dev`/`staging`/`prod`, rollback.yml:192,296,376,458 and database-migration.yml:265,376,470 bind the `-rollback` and `-db-` families, and deploy-gcp.yml and deploy-azure.yml carry no job-level `environment:` at all; only `aws-fargate-*` (deploy-aws-fargate.yml:142) matches the README. +- issue: (pending cross-reference) + +### A14-012 `make security-scan-go` swallows gosec's verdict and excludes eight rule classes +- category: ops +- severity: medium +- location: Makefile:159 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: gosec exits non-zero on findings, but the recipe line ends with `; echo "✓ Go security scan complete"`, so the recipe's status is echo's. `security-scan-go` therefore always succeeds, and `make ci` (line 216) — which chains `security-scan` — reports a clean pipeline with findings present. The `-exclude=G101,G104,G115,G204,G301,G304,G402,G505` list also silences InsecureSkipVerify-adjacent and TLS-version rules that ci.yml's authoritative gosec run (ci.yml:694) does **not** exclude, so the local `make ci` result cannot predict CI. When gosec is not installed the recipe prints a hint and still exits 0. +- evidence: + ```make + security-scan-go: + @if command -v gosec > /dev/null; then \ + gosec -fmt=json -out=gosec-report.json -exclude=G101,G104,G115,G204,G301,G304,G402,G505 ./...; \ + echo "✓ Go security scan complete: gosec-report.json"; \ + else \ + echo "gosec not installed. Install: make install-dev-tools"; \ + fi + ``` +- suggested fix: Drop the trailing `echo` so gosec's status propagates, align the exclude list with ci.yml (which uses none), and `exit 1` when the tool is missing, as the `complexity` target already does. +- verdict: CONFIRMED — Makefile:156-163 is a single `@if ...; then gosec ...; echo ...; fi`, so the recipe's exit status is the trailing echo's; ci.yml:694-705 runs gosec with no `-exclude` at all, and `ci:` at Makefile:216 chains `security-scan`. +- issue: (pending cross-reference) + +### A14-032 gcp-import-dev-state.sh hardcodes a live project ID and state bucket +- category: ops +- severity: medium +- location: scripts/gcp-import-dev-state.sh:22 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: The GCP project (`serene-bazaar-666`) and the Terraform state bucket (`cudly-terraform-state-cloudprowess`) are literals in a committed script, unlike every other GCP consumer which reads `vars.GCP_PROJECT_ID` and `secrets.TF_BACKEND_GCP`. Anyone running the script against a different environment silently imports resources from the wrong project into whatever state they initialised, and the two names — which the deploy pipeline treats as configuration worth keeping in repository variables and secrets — are published in the repository. +- evidence: + ```bash + PROJECT="serene-bazaar-666" + REGION="us-central1" + SERVICE_NAME="cudly-dev" + TF_DIR="$(cd "$(dirname "$0")/.." && pwd)/terraform/environments/gcp" + BACKEND_CONFIG="bucket = \"cudly-terraform-state-cloudprowess\"\nprefix = \"github-dev\"" + ``` +- suggested fix: Take the project and bucket from required arguments or environment variables, defaulting to nothing and failing loudly when unset. +- verdict: CONFIRMED — `PROJECT="serene-bazaar-666"` and the `cudly-terraform-state-cloudprowess` bucket are unconditional literals with no argument or environment override (scripts/gcp-import-dev-state.sh:22,26), while deploy-gcp.yml:184 reads `vars.GCP_PROJECT_ID` and the workflows take the bucket from `secrets.TF_BACKEND_GCP`. +- issue: (pending cross-reference) + +### A14-034 `make deploy` points at a directory layout the repository does not have +- category: ops +- severity: medium +- location: scripts/tf-deploy.sh:62 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: `ENV_DIR` is `terraform/environments//`, e.g. `terraform/environments/aws/dev`, which does not exist — the repository has a single root at `terraform/environments/aws` selected by `-var-file`. The script creates the missing directory and symlinks only `main.tf` into it (line 83), so the resulting root has no `variables.tf`, `backend.tf` or sibling `.tf` files and `terraform init` at line 113 (run with no `-backend-config`) cannot resolve anything. `Makefile:102` (`deploy`) and every target in `Makefile.terraform` (deploy/plan/destroy/output/docker-skip) route through this script, so the documented local deployment path is inoperable, and it leaves a stray directory behind. `terraform/profiles/aws/` also contains only `*.tfvars.example`, so `make profile-list` reports no profiles. +- evidence: + ```bash + ENV_DIR="${PROJECT_ROOT}/terraform/environments/${PROVIDER}/${PROFILE}" + ... + mkdir -p "$ENV_DIR" + ln -sf "../../${PROVIDER}/main.tf" "${ENV_DIR}/main.tf" + ``` +- suggested fix: Point `ENV_DIR` at `terraform/environments/${PROVIDER}` and pass the profile through `-var-file`, matching the workflows; remove the directory-creation and symlink branch. +- verdict: CONFIRMED — `ENV_DIR="${PROJECT_ROOT}/terraform/environments/${PROVIDER}/${PROFILE}"` (scripts/tf-deploy.sh:62) names a path that does not exist (terraform/environments/aws is a single flat root) and terraform/profiles/aws/ holds only `*.tfvars.example`, so `make deploy` (Makefile:102) exits at the profile check before reaching the mkdir and symlink branch at line 83. +- issue: (pending cross-reference) + +### A14-035 init-backend.sh writes a backend file whose state key conflicts with the workflows' +- category: ops +- severity: medium +- location: scripts/init-backend.sh:208 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: The script creates `terraform/environments/aws//backend.tf` with a hardcoded `key = "/terraform.tfstate"` and bucket `cudly-terraform-state-`. Every workflow instead writes `key = "github-/terraform.tfstate"` (deploy-aws-lambda.yml:259) or `github-fargate-/…` from `secrets.TF_BACKEND_AWS`. An operator following this script initialises a second, empty state at a different key in a different bucket and then applies into it, creating a duplicate stack alongside the CI-managed one. The generated file also lands in a subdirectory that `terraform fmt -check -recursive terraform/` (ci.yml:583) will start walking as a separate root. +- evidence: + ```bash + BACKEND_CONFIG_FILE="terraform/environments/aws/${ENVIRONMENT}/backend.tf" + ... + bucket = "${BUCKET_NAME}" + key = "${ENVIRONMENT}/terraform.tfstate" + ``` +- suggested fix: Emit a `.tfbackend` file matching the `github-/terraform.tfstate` key the workflows use, into `terraform/environments/aws/backends/`, rather than a `backend.tf` in a new root. +- verdict: CONFIRMED — scripts/init-backend.sh:208,220-221 writes `terraform/environments/aws//backend.tf` with bucket `cudly-terraform-state-` (line 78) and `key = "/terraform.tfstate"` against every workflow's `github-/terraform.tfstate` (deploy-aws-lambda.yml:259), and lines 248-254 instruct the operator to cd there and apply; the duplicate-stack outcome additionally requires configuration to be placed in that otherwise-empty directory. +- issue: (pending cross-reference) + +### A15-003 Go tooling scans third-party Go code vendored inside frontend/node_modules +- category: ops +- severity: medium +- location: .pre-commit-config.yaml:19 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: The npm package `flatted` ships a Go package at `frontend/node_modules/flatted/golang/pkg/flatted/flatted.go` with no `go.mod` of its own, so Go absorbs it into the CUDly module. In `.github/workflows/pre-commit.yml` the "Install frontend deps" step (`cd frontend && npm ci`, line 238) runs before "Run pre-commit" (line 241), which runs `go vet ./...`. Verified in the worktree after `npm ci`: `go list ./...` emits `github.com/LeanerCloud/CUDly/frontend/node_modules/flatted/golang/pkg/flatted`. Go's `./...` skips only `testdata` and directories beginning with `.` or `_`, never `node_modules`. Today `go vet` passes on that package, so nothing is broken; the exposure is that a future `flatted` release with vet-unclean or vulnerable Go code reddens the repo's own lint, test and govulncheck gates on code the repo does not own and cannot fix. +- evidence: + ```yaml + - id: go-vet + name: Run go vet + entry: bash -c 'go vet ./...' + ``` +- suggested fix: Have the Go `./...` invocations enumerate real packages, for example `go vet $(go list ./... | grep -v /node_modules/)`, or move the frontend install after the pre-commit step in that workflow. +- verdict: CONFIRMED — reproduced end to end: `frontend/node_modules/flatted/golang/pkg/flatted/flatted.go` exists after `npm ci` and `go list ./...` in the worktree emits `github.com/LeanerCloud/CUDly/frontend/node_modules/flatted/golang/pkg/flatted`, the go-vet hook really is `bash -c 'go vet ./...'` (.pre-commit-config.yaml:16-21, one line above the cited :19), and "Install frontend deps" at .github/workflows/pre-commit.yml:235 does run before "Run pre-commit" at :241, so the third-party package is inside the repo's own gate. +- issue: (pending cross-reference) + +### A02-018 forgot-password has no per-IP limit and its per-email key is not normalized +- category: ops +- severity: low +- location: internal/api/handler_auth.go:267 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: The only limiter is `AllowWithEmail` (10 per 5 min per exact string). One client can submit thousands of distinct addresses per minute, each costing a `GetUserByEmail` lookup and, for real users, an outbound reset email through SES, with no IP budget at all (every other credential endpoint uses `checkRateLimitStrict`). Because the key is the raw string, `Alice@x.com`, `alice@x.com` and `alice@x.com ` are separate buckets, so a single victim can receive well over 10 reset mails per window if the store's lookup is case-insensitive. +- evidence: + ```go + if h.rateLimiter != nil { + allowed, err := h.rateLimiter.AllowWithEmail(ctx, pwdReq.Email, "forgot_password") + ``` +- suggested fix: Add `checkRateLimit(ctx, req, "forgot_password_ip")` (fail-open is fine here) before the email bucket, and key the email bucket on `strings.ToLower(strings.TrimSpace(email))`. +- verdict: CONFIRMED — forgotPassword (handler_auth.go:267-276) has only AllowWithEmail, which keys on the raw string in both limiters (db_rate_limiter.go:217-219, inmemory_rate_limiter.go:143-145), and no checkRateLimit call, unlike every other credential endpoint (handler_auth.go:24,234,316,345,475); the case-variant amplification is weaker than stated because GetUserByEmail is an exact `WHERE email = $1` (internal/auth/store_postgres.go:68), so variants hit distinct lookups rather than the same victim. +- issue: (pending cross-reference) + +### A06-026 OIDC provider is created with an all-zero certificate thumbprint +- category: ops +- severity: low +- location: internal/iacfiles/templates/aws-wif-cli.sh.tmpl:57 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: the generated script creates the IAM OIDC provider with a 40-zero thumbprint and no comment explaining why. For issuers whose certificate chains to a CA in AWS's trust store this value is ignored, but for a CUDly deployment fronted by a private or internal CA the thumbprint is the verification input, and a value that can never match makes every `AssumeRoleWithWebIdentity` fail with an opaque error the operator has no pointer to. It is also a hardcoded magic value in a security-relevant position, in a file that elsewhere goes to considerable length to explain its trust-policy choices. +- evidence: + ```sh + aws iam create-open-id-connect-provider $PROFILE_ARG \ + --url "${OIDC_ISSUER_URL}" \ + --client-id-list "${OIDC_AUDIENCE}" \ + --thumbprint-list "0000000000000000000000000000000000000000" >/dev/null + ``` +- suggested fix: compute the thumbprint from the issuer's TLS chain in the script, or keep the placeholder and add the one-line comment stating that AWS ignores it for trust-store issuers. +- verdict: PLAUSIBLE — the all-zero literal is there with no explanatory comment, in a file that comments its trust-policy choices at length (internal/iacfiles/templates/aws-wif-cli.sh.tmpl:52-58 vs 15-41), so the hardcoded-magic-value half is a verified fact; the operational failure needs the CUDly issuer to be fronted by a CA outside AWS's trust store, which I cannot establish from source since `CUDLY_ISSUER_URL` resolves to the deployment's own Lambda Function URL / Cloud Run domain (internal/api/handler_federation.go:175-183). +- issue: (pending cross-reference) + +### A13-018 `acm:RemoveTagsFromCertificate` is missing, the same gap the file documents for `ec2:DeleteTags` +- category: ops +- severity: low +- location: terraform/environments/aws/ci-cd-permissions/policy_networking.tf:140 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: `terraform/environments/aws/acm.tf:14` tags the certificate with `merge(local.common_tags, ...)`. The AWS provider's tag update path removes dropped keys before adding new ones, so dropping any key from `common_tags` or `default_tags` on an existing certificate calls `acm:RemoveTagsFromCertificate` and fails the apply. This is the identical failure mode the same file documents at lines 53-58 as the reason `ec2:DeleteTags` had to be added, applied to a resource that also carries tags. +- evidence: + ```hcl + Action = [ + "acm:AddTagsToCertificate", + "acm:DeleteCertificate", + "acm:DescribeCertificate", + "acm:GetCertificate", + "acm:ListTagsForCertificate", + "acm:RequestCertificate", + ] + ``` +- suggested fix: Add `acm:RemoveTagsFromCertificate` to the ACM statement. +- verdict: CONFIRMED — the `ACM` statement at policy_networking.tf:139-150 lists six actions and no `RemoveTagsFromCertificate`, while the same file at :52-58 documents the identical provider behaviour as the reason `ec2:DeleteTags` had to be added. terraform/environments/aws/acm.tf:14-16 tags the certificate from `merge(local.common_tags, ...)`, so a dropped key hits the untagged verb. The failure needs two conditions the finding does not state: the certificate exists only when `frontend_domain_names` and `subdomain_zone_name` are both set (acm.tf:5), and a tag key must actually be removed. +- issue: (pending cross-reference) + +### A13-021 Two compose images float on tags while every sibling image is digest-pinned with a written rationale +- category: ops +- severity: low +- location: docker-compose.yml:96 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: `docker-compose.test.yml:7-10` explains that a floating tag "lets the database change under an unchanged repo" and pins Postgres by digest in both files. `nginx:alpine` and `dpage/pgadmin4:9.2` in `docker-compose.yml` are not pinned, so a local dev stack silently picks up a new nginx or pgAdmin build, including one that changes the reverse-proxy behaviour `scripts/nginx.conf` depends on. `nginx:alpine` in particular tracks a moving major. +- evidence: + ```yaml + frontend: + image: nginx:alpine + ... + pgadmin: + image: dpage/pgadmin4:9.2 + ``` +- suggested fix: Pin both by `@sha256:` digest with the same refresh comment the Postgres entries carry. +- verdict: CONFIRMED — docker-compose.yml pins postgres by digest at line 6 but leaves `nginx:alpine` (line 96) and `dpage/pgadmin4:9.2` (line 110) on tags, while docker-compose.test.yml:7-11 states the rationale ("a floating tag lets the database change under an unchanged repo") and pins the same digest. `nginx:alpine` does float a major; `dpage/pgadmin4:9.2` is at least minor-pinned, so only the nginx half floats freely. Scope is the local dev stack only — the test compose file and CI service containers are already pinned. +- issue: (pending cross-reference) + +### A13-022 The dev image installs a golang-migrate release binary the production image was rewritten to avoid +- category: ops +- severity: low +- location: Dockerfile.dev:29 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: `Dockerfile:55-85` explains at length that upstream's prebuilt `migrate` tarballs ship whatever toolchain and dependency versions upstream built them with, that this is how issue #1833's stdlib CVEs reached the runtime image, and that migrate must therefore be built from this module's own `go.mod`. `Dockerfile.dev` still downloads the v4.17.0 release tarball, an older release than the one `go.mod` resolves, so the development image ships exactly the transitive advisories the production image was changed to remove, and a developer running `migrate` locally exercises a different binary from the one that runs in the container. `go install github.com/air-verse/air@v1.61.7` on line 16 resolves air's own `go.mod` for the same reason. +- evidence: + ```dockerfile + curl -Lo migrate.tar.gz "https://github.com/golang-migrate/migrate/releases/download/v4.17.0/migrate.linux-${MIGRATE_ARCH}.tar.gz" && \ + echo "${MIGRATE_SHA256} migrate.tar.gz" | sha256sum -c - && \ + ``` +- suggested fix: Build `migrate` from the main module in `Dockerfile.dev` the way `Dockerfile` does, so the two images run the same binary at the same versions. +- verdict: CONFIRMED — Dockerfile.dev:22-33 downloads the v4.17.0 release tarball, while Dockerfile:55-85 explains that upstream tarballs carry upstream's own toolchain and dependency pins ("how issue #1833's stdlib CVEs reached the runtime image") and builds migrate as a package of this module at the version `go.mod` resolves (v4.19.1, go.mod:106) with `-tags=pgx5`. The divergence is worse than the finding states: the prod binary registers only the `pgx5://` scheme while the dev tarball is the default `postgres://` build, so the same migration command is not portable between the two images. `go install github.com/air-verse/air@v1.61.7` at Dockerfile.dev:16 does resolve air's own go.mod, as claimed. +- issue: (pending cross-reference) + +### A13-023 The email sender publishes to SNS but no IaC grants `sns:Publish` +- category: ops +- severity: low +- location: terraform/modules/compute/aws/lambda/main.tf:381 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: `internal/email/sender.go:192` publishes to `s.topicARN`, sourced from the `SNS_TOPIC_ARN` environment variable (internal/email/factory.go:77). No terraform module or CloudFormation template grants `sns:Publish` to the runtime role, and `sns:` is absent from the workload permissions boundary's `WorkloadServiceCeiling` as well, so a boundaried role would be denied even if an identity policy granted it. Today nothing sets `SNS_TOPIC_ARN` so `SendNotification` takes the empty-topic skip path; the moment an operator wires `module.monitoring.sns_topic_arn` into the Lambda environment, every notification fails with AccessDenied rather than the intended send. +- evidence: + ```go + _, err := s.snsClient.Publish(ctx, &sns.PublishInput{ + TopicArn: aws.String(s.topicARN), + ``` +- suggested fix: Grant `sns:Publish` scoped to the notification topic ARN in the runtime modules and add `sns:Publish` to `WorkloadServiceCeiling`, in the same change that first sets `SNS_TOPIC_ARN`. +- verdict: PLAUSIBLE — the claim that "no terraform module or CloudFormation template grants sns:Publish" is wrong: cloudformation/stacks/CUDly/template.yaml:568-573 has an `SNSPublish` Sid scoped to `!Ref NotificationTopic`, and that stack also creates the topic (:729). The Terraform half and the boundary half do hold: `/usr/bin/grep -rn 'sns:' terraform iac` returns nothing, and `WorkloadServiceCeiling` (policy_boundary.tf:160-198) lists no `sns:` entry. The failure needs the runtime condition the finding names, an operator wiring the topic ARN, and there is a second wrinkle it missed: even the CFN stack sets `NOTIFICATION_TOPIC_ARN` (:607) while internal/email/factory.go:77 reads `SNS_TOPIC_ARN`, so `SendNotification` takes the empty-topic skip path there too. +- issue: (pending cross-reference) + +### A13-024 Customer-facing federation modules use unbounded `>=` provider constraints two majors behind their own lock files +- category: ops +- severity: low +- location: iac/federation/aws-target/terraform/main.tf:6 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: All five federation roots constrain with `>=` (`aws >= 5.0`, `google >= 5.0`, `http >= 3.4`) while their committed `.terraform.lock.hcl` files pin `aws 6.40.0` and `google 7.27.0`. A customer who unpacks the bundle into a directory that already has a lock, runs `terraform init -upgrade`, or is given only the `.tf` files gets whatever major is current at the time, with no floor that reflects what the module was written and tested against. The repo's own workload modules use `~>` pessimistic constraints throughout; only the customer-facing bundles do not. +- evidence: + ```hcl + aws = { + source = "hashicorp/aws" + version = ">= 5.0" + } + ``` +- suggested fix: Change the five federation roots to `~> 6.0` / `~> 7.0` / `~> 3.5` to match their lock files, so a customer's major-version upgrade is a deliberate edit. +- verdict: CONFIRMED — four of the five roots leave their primary provider unbounded (`aws >= 5.0` in aws-cross-account/terraform/main.tf:6 and aws-target/terraform/main.tf:6, `google >= 5.0` in gcp-sa-impersonation and gcp-target main.tf:6) and all five use `http >= 3.4`, against locks that pin `aws 6.40.0`, `google 7.27.0` and `http 3.5.0`. The workload side does use pessimistic constraints (`~> 5.0`, `~> 3.3`, `~> 3.4`, `~> 2.0` in terraform/environments/aws/main.tf:10-22, `~> 5.0` in the compute module versions.tf files). Two corrections: azure-target already uses `~> 3.8` / `~> 4.0` for azuread and azurerm, so only its `http` line is unbounded, and the drift is two majors for google but one for aws. +- issue: (pending cross-reference) + +### A13b-014 CloudFormation WIF template makes the operator keep the issuer URL and its condition-key host in sync by hand +- category: ops +- severity: low +- location: iac/federation/aws-target/cloudformation/template.yaml:20 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: `OIDCIssuerHost` is a second parameter the operator must set to `OIDCIssuerURL` minus `https://`, and the description says outright that keeping them in sync is the operator's job. It is used to build the trust-policy condition keys `:aud` and `:sub` at lines 265-267. If the two disagree, IAM never populates those keys for the incoming token, `StringEquals` fails, and every `AssumeRoleWithWebIdentity` is denied. That fails closed, but it fails after the stack reports success, with no diagnostic pointing at the mismatch. The Terraform module derives the host from the URL in one line (`iac/federation/aws-target/terraform/main.tf:22`) and has no second parameter to desynchronise. +- evidence: + ```yaml + OIDCIssuerHost: + Type: String + Description: > + OIDC issuer host and path WITHOUT the https:// prefix — used as the IAM + condition key. Must equal OIDCIssuerURL with the https:// stripped and no + trailing slash; the operator is responsible for keeping the two in sync. + ``` +- suggested fix: Remove `OIDCIssuerHost` and derive it as `!Select [1, !Split ["https://", !Ref OIDCIssuerURL]]`, matching the Terraform local so one input cannot contradict the other. +- verdict: CONFIRMED — `iac/federation/aws-target/cloudformation/template.yaml:20-28` is a second free-text parameter whose own description hands the operator the sync duty, and its two `AllowedPattern`s constrain shape only, never the relationship to `OIDCIssuerURL` (lines 8-18); the value is spliced into both condition keys at `:265-267`, and `iac/federation/aws-target/terraform/main.tf:22` derives the same host from the URL with `trimsuffix(trimprefix(...))` so the Terraform path has nothing to desynchronise. Fails closed, as the finding states. +- issue: (pending cross-reference) + +### A13b-016 Azure registration fires immediately after the role assignment, ahead of RBAC propagation +- category: ops +- severity: low +- location: iac/federation/azure-target/terraform/registration.tf:30 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: The registration POST depends on `azurerm_role_assignment.cudly_reservations`, so it fires the moment the assignment resource is created. Azure RBAC propagation is asynchronous and takes up to ten minutes, so CUDly receives a registered subscription and starts collecting against a principal whose grant has not landed. The first collection or purchase attempt returns 403 and reads as a permissions misconfiguration rather than a timing artefact. Nothing in the module waits. +- evidence: + ```hcl + # Defer to apply phase — ensure Azure resources are created before registering. + depends_on = [azurerm_role_assignment.cudly_reservations] + ``` +- suggested fix: Insert a `time_sleep` of a few minutes between the role assignment and the registration `data.http`, or have the registration payload flag that RBAC may still be propagating so the backend retries rather than reporting a hard failure. +- verdict: PLAUSIBLE — the code half holds: `iac/federation/azure-target/terraform/registration.tf:30` waits only on resource creation and nothing in the module sleeps, while the module's own comment at `iac/federation/azure-target/terraform/main.tf:76-77` puts RBAC propagation at up to ten minutes. But registration does not start collection: `internal/api/handler_registrations.go:76-78` stores the account as `pending`, and only the admin-driven `approveRegistration` (`:252`, enabling the account at `:288`) creates the cloud account, so the 403 needs the runtime condition that an operator approves the registration inside the propagation window — which I cannot establish from source. +- issue: (pending cross-reference) + +### A13c-014 GCP Cloud SQL module defaults `deletion_protection` to false where the AWS module defaults it to true +- category: ops +- severity: low +- location: terraform/modules/database/gcp/variables.tf:199 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: `modules/database/aws/variables.tf:87` defaults `deletion_protection = true`, + so a caller who forgets the input gets the safe posture. The GCP module defaults false, so the + same omission yields a deletable production database. `environments/gcp/database.tf:39` does + pass `var.database_deletion_protection`, which masks this for the in-tree callers, but the + module contract itself is unsafe by default and differs from its sibling for no stated reason. +- evidence: + ```hcl + variable "deletion_protection" { + description = "Enable deletion protection" + type = bool + default = false + } + ``` +- suggested fix: flip the default to true so the three database modules agree, and let dev + profiles opt out explicitly as `environments/gcp/github-dev.tfvars:50` already does. +- verdict: CONFIRMED — database/gcp/variables.tf:199-202 defaults false while database/aws/ + variables.tf:87-90 defaults true. The masking is exactly as described: environments/gcp/ + database.tf:39 passes `var.database_deletion_protection`, whose env-layer default is true + (environments/gcp/variables.tf:174) and which github-dev.tfvars:50 overrides to false. +- issue: (pending cross-reference) + +### A13c-024 GCP OIDC signing key sets `prevent_destroy = false` where its AWS counterpart was explicitly given `create_before_destroy` +- category: ops +- severity: low +- location: terraform/modules/compute/gcp/cloud-run/signing-key.tf:35 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: `modules/compute/aws/lambda/signing-key.tf:25-27` carries + `create_before_destroy = true` with a comment explaining that replacing the OIDC signing CMK + destroy-first breaks client-assertion JWT minting (PR #1480 follow-up). The GCP key opts the + other way and additionally shortens `destroy_scheduled_duration` to one day "because tests + redeploy often". A ForceNew on this key — changing `purpose`, `version_template.algorithm`, or + the key ring's `location` via `var.region` — schedules the live signing key for destruction and + every federated target cloud rejects assertions until the new JWKS propagates. +- evidence: + ```hcl + resource "google_kms_crypto_key" "signing" { + purpose = "ASYMMETRIC_SIGN" + destroy_scheduled_duration = "86400s" # 1 day — tests redeploy often + lifecycle { + prevent_destroy = false + } + } + ``` +- suggested fix: make `prevent_destroy` and `destroy_scheduled_duration` inputs so production + deployments get the protective values and only test environments opt out. +- verdict: CONFIRMED — the GCP key carries `destroy_scheduled_duration = "86400s"` with the + "tests redeploy often" comment and an explicit `prevent_destroy = false` + (cloud-run/signing-key.tf:24-38), against the AWS twin's `create_before_destroy = true` and its + seven-line rationale citing #1480 (lambda/signing-key.tf:19-27). The two mechanisms are not + equivalent — `create_before_destroy` protects a ForceNew replacement, `prevent_destroy` blocks a + destroy — but the GCP key has neither, and `prevent_destroy = false` is also Terraform's default, + so the line states rather than changes the posture. +- issue: (pending cross-reference) + +### A14-016 The Snyk job cannot fail and is not a dependency of CI Success +- category: ops +- severity: medium +- location: .github/workflows/ci.yml:829 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: `continue-on-error: true` on the only step means `snyk-scan` always reports success, and the job is absent from `ci-success`'s `needs` list (lines 1064-1080), so even a job-level failure would not gate a merge. If `SNYK_TOKEN` is unset the action no-ops as well. The README lists Snyk as CI job 7 (README.md:38) and `make security-scan-all` chains it, so the repo presents dependency scanning as a gate that has no effect on any outcome. +- evidence: + ```yaml + - name: Run Snyk to check for vulnerabilities + uses: snyk/actions/golang@b98d498629f1c368650224d6d212bf7dfa89e4bf # 0.4.0 + continue-on-error: true + env: + SNYK_TOKEN: ${{ secrets.SNYK_TOKEN }} + ``` +- suggested fix: Either drop the job and the README claim, or remove `continue-on-error`, add `snyk-scan` to `ci-success`'s `needs`, and fail loudly when the token is missing. +- verdict: CONFIRMED — ci.yml:829 sets `continue-on-error: true` on the job's only substantive step, and `ci-success`'s `needs` list (ci.yml:1064-1080) enumerates 17 jobs, none of which is `snyk-scan`. +- severity-adjusted: low — .github/workflows/README.md:38,50 already mark `SNYK_TOKEN` as optional, so nothing in the repo relies on this job as a merge gate. +- issue: (pending cross-reference) + +### A14-037 `make complexity` gates test files that ci.yml and .golangci.yml both exempt +- category: ops +- severity: low +- location: Makefile:126 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: The local target runs `gocyclo -over 10 .` across every file, while `ci.yml:70` runs `gocyclo -over 10 -ignore "_test\.go" .` and `.golangci.yml:42-43,86-92` sets `min-complexity: 15` and excludes gocyclo on `_test\.go`. A developer running `make pre-commit` or `make ci` gets failures on table-driven tests that CI accepts, so the local gate is stricter than the merge gate in a direction that trains people to ignore it. The `2>&1` capture also folds a gocyclo tool error into the "complexity issues" branch. +- evidence: + ```make + COMPLEXITY_ISSUES=$$(gocyclo -over 10 . 2>&1 || true); \ + if [ -n "$$COMPLEXITY_ISSUES" ]; then \ + ``` +- suggested fix: Add `-ignore "_test\.go"` so the local target matches ci.yml, and keep stderr separate from the findings list. +- verdict: CONFIRMED — Makefile:126 runs `gocyclo -over 10 .` with no ignore while ci.yml:70 passes `-ignore "_test\.go"` and .golangci.yml sets gocyclo min-complexity 15 plus a `_test\.go` exclusion; both `ci` and `pre-commit` depend on the `complexity` target. +- issue: (pending cross-reference) + +### A14-038 "Check coverage threshold" never fails and never checks a real threshold +- category: ops +- severity: low +- location: .github/workflows/ci.yml:312 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: The step emits `::warning::` and exits 0, so coverage dropping from 80% to 5% does not affect `unit-tests` or `ci-success`. If the `grep total` produces an empty string (a merge that lost the total line), `bc -l` errors and the `(( ))` is false, so the step also passes silently on a broken profile. A step named "Check coverage threshold" that cannot fail reads as a gate in the job list. +- evidence: + ```bash + coverage=$(go tool cover -func="$RUNNER_TEMP/coverage.out" | grep total | awk '{print $3}' | sed 's/%//') + echo "Total coverage: ${coverage}%" + if (( $(echo "$coverage < 80" | bc -l) )); then + echo "::warning::Coverage is below 80% (current: ${coverage}%)" + fi + ``` +- suggested fix: Either fail the step below an agreed floor, or rename it to "Report coverage" so it is not read as a gate; fail loudly when `coverage` is empty. +- verdict: CONFIRMED — ci.yml:312-317 emits only `::warning::` with no exit, so coverage gates neither `unit-tests` nor `ci-success`; the finding's secondary claim is wrong, since GitHub's default `bash -eo pipefail` makes an empty `grep total` fail the assignment rather than pass silently. +- issue: (pending cross-reference) + +### A14-039 The read-only sanity checks are documented as verifying deploy-credential permissions +- category: ops +- severity: low +- location: ci_cd_sanity_tests/pkg/sanity/aws/aws.go:1 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: The package comment says the checks "verify that deploy credentials have sufficient IAM permissions before a real deploy". `aws_sanity.yml:54` assumes `secrets.AWS_CICD_READONLY_ROLE_ARN`, a different role from the `vars.AWS_ROLE_TO_ASSUME` the deploys use, and the four checks are `sts:GetCallerIdentity`, `ec2:DescribeRegions`, `ec2:DescribeInstances` and `rds:DescribeDBInstances` — none of which the deploy role's IAM policy is even asserted against. A green sanity report therefore says nothing about whether a deploy will have the permissions it needs, which is what the comment promises. +- evidence: + ```go + // Package aws implements read-only AWS sanity checks used in CI/CD to verify + // that deploy credentials have sufficient IAM permissions before a real deploy. + package aws + ``` +- suggested fix: Reword the comment to describe what it does (read-only reachability of a small API set under the CI read-only role), or point the check at the deploy role's actual action list. +- verdict: CONFIRMED — ci_cd_sanity_tests/pkg/sanity/aws/aws.go:1-2 promises verification that "deploy credentials have sufficient IAM permissions", but aws_sanity.yml:54 assumes `secrets.AWS_CICD_READONLY_ROLE_ARN` while deploys assume `vars.AWS_ROLE_TO_ASSUME` (deploy-aws-lambda.yml:246), and the only checks are the four read-only calls at aws.go:123-132. +- issue: (pending cross-reference) + +### A14-041 `terraform fmt -check` on tfvars files is downgraded to a note +- category: ops +- severity: low +- location: .github/workflows/ci.yml:602 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: Both tfvars format checks end in `|| echo "Note: … may need formatting"`, so a malformed `github-prod.tfvars` passes the `Validate Terraform` job. `terraform fmt -check` also returns non-zero on a genuine parse error, not only on formatting, so a broken production variables file reaches the deploy workflows undetected while the step it lives in is named "Validate environment tfvars files". +- evidence: + ```bash + if [ -f "github-${env}.tfvars" ]; then + echo "Checking github-${env}.tfvars syntax..." + terraform fmt -check "github-${env}.tfvars" || echo "Note: github-${env}.tfvars may need formatting" + fi + ``` +- suggested fix: Let the non-zero exit propagate, or run `terraform fmt -check -recursive` over the environment directory as part of the existing format step. +- verdict: CONFIRMED — ci.yml:595-608 ends both tfvars checks with `terraform fmt -check ... || echo "Note: ..."`, so a non-zero exit from either formatting or a parse error never fails the "Validate environment tfvars files" step. +- issue: (pending cross-reference) + +### A14-043 `docker-test` and the sanity-skip gates report success for work that did not happen +- category: ops +- severity: low +- location: Makefile:213 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: `make docker-test` builds the image and then runs `--help` with `|| true`, so a container that cannot start still reports the target as passing. The same shape appears in `aws_sanity.yml:80-82` and `azure_sanity.yml:79-81`: when the cloud secrets are absent the job prints "skipped because required secrets are not configured" and succeeds. If a secret is rotated away or a fork's PR runs, the sanity workflows go green forever with no signal that the check stopped running, which is indistinguishable from the check passing. +- evidence: + ```make + docker-test: docker-build + @echo "Testing Docker image..." + docker run --rm cudly:$(VERSION) /app/cudly --help || true + ``` +- suggested fix: Drop the `|| true` from `docker-test`; for the sanity workflows, emit a `::warning::` and surface the skipped state in the job name or summary so a silently disarmed check is visible. +- verdict: CONFIRMED — Makefile:211-213 ends `docker run ... --help` with `|| true`, and the skip paths at aws_sanity.yml:80-82 and azure_sanity.yml:79-81 echo a message on a green job whenever `should_run != 'true'`. +- issue: (pending cross-reference) + +### A15-002 GO-2026-5932 (x/crypto/openpgp) is present, unreachable, and has no version that clears it +- category: ops +- severity: low +- location: go.mod:67 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: `govulncheck ./...` at the repo root reports one advisory at "modules you require" level against `golang.org/x/crypto@v0.55.0`. Its OSV record has ranges `[{"introduced":"0"}]` with no `fixed` event, so every version that will ever exist matches and no `go get` clears it. Anyone treating this as a routine bump would produce a change that claims to resolve the advisory while govulncheck still reports it. +- evidence: + ```text + Vulnerability #1: GO-2026-5932 + Module: golang.org/x/crypto + Found in: golang.org/x/crypto@v0.55.0 + Fixed in: N/A + ``` + Unreachable, verified two ways: govulncheck reports 0 called and 0 imported symbols, and `git grep 'crypto/openpgp' -- '*.go'` returns nothing, so none of the seven affected `openpgp/*` import paths is used. +- suggested fix: No action beyond recording it as accepted. Do not attempt a bump; if the advisory must be silenced, that is an explicit suppression decision for the owner, not a dependency change. +- verdict: CONFIRMED — reproduced with a govulncheck rebuilt as v1.7.0 under the local go1.27.0 (not the stale binary): `govulncheck -show verbose ./...` at the repo root reports exactly one advisory, GO-2026-5932 against `golang.org/x/crypto@v0.55.0` (go.mod:67) with `Fixed in: N/A`, 0 called and 0 imported symbols, and `git grep 'crypto/openpgp' -- '*.go'` returns nothing, so the no-version-clears-it claim and the unreachability claim both hold. +- issue: (pending cross-reference) + +### Category: hygiene + +14 findings: 1 medium, 13 low. + +### A02-011 openapi.yaml omits roughly a third of the routed surface and mis-describes several documented operations +- category: hygiene +- severity: medium +- location: internal/api/openapi.yaml:44 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: The spec is served publicly at `/docs` and is the contract the 403 regression test (openapi_403_test.go) walks, so undocumented routes are exempt from that guard. Routes registered in router.go with no path entry: `/api/recommendations/freshness` (126), `/api/recommendations/{id}/detail` (133), `/api/plans/{id}/accounts` (146-147), `/api/purchases/revoke/{id}` (171-172), `/api/purchases/retry/{id}` (178), `/api/purchases/{id}/revoke` and `/revoke/calculate` (184, 189), `/api/purchases/{id}/marketplace-list|cancel` (194-195), `/api/analytics/trends` (216), `/api/auth/me/permissions` (228), `/api/auth/reset-password/status` (233), all four `/api/auth/mfa/*` (239-242), every `/api/accounts*` route (264-279), `/api/inventory/commitments|coverage` (295, 299), `/api/ladder/configs` (325-326), `/api/notifications/unsubscribe` (330-331), `/api/register`, `/api/register/{token}` and all `/api/registrations*` (334-342), `/api/federation/iac` (347), `/version` (369), and the HEAD methods on docs (373, 375). Documented operations that disagree with the handler: `/api/history` lists an `interval` param the handler ignores and omits `provider`, `account_ids`, `limit` (yaml:1075-1099 vs handler_history.go:666-710); `/api/history/analytics` omits `interval` and `provider` (yaml:1101-1120 vs handler_analytics.go:131-145); `/api/history/breakdown` dimension enum lacks `account` (yaml:1129 vs analytics_postgres.go:72); `PUT /api/config` and `PUT /api/config/service/{service}` document a config body response but return `{"status":"updated"}` (yaml:110-117, 165-171 vs handler_config.go:130, 364); `/api/config/service/{service}` path enum `[ec2, rds, elasticache, opensearch]` but the segment is `provider/service` and GET returns 200 `{}` not 404 (yaml:137, 176 vs handler_config.go:297-314); `/api/recommendations/refresh` documents 200 `CollectResult` but returns `RefreshResponse` and can 409 (yaml:239-257 vs handler_recommendations_refresh.go:37-40, 82); `/api/recommendations` omits `account_ids`, `min_savings_usd`, `min_savings_pct` and lists `account_id` which the handler does not read (yaml:206-237 vs handler_recommendations.go:39-79); `/api/purchases/approve|cancel/{id}` document POST only with the token required in the query (yaml:470-520) while the router also serves GET and reads the token from the body first (router.go:163-166, 568-586); `/api/dashboard/summary` lists StartDate/EndDate the handler never reads (yaml:53-55); `/api/api-keys/{id}` and `/api/users/{id}` document 404 where the handler returns 500 (see A02-012); `/api/auth/setup-admin` omits 409/503, `/api/auth/login` omits 503, `/api/auth/reset-password` omits 429/503 (yaml:1242-1310 vs handler_auth.go:24, 234-247, 345); `/api/docs/openapi.yaml` declares `application/x-yaml` but the handler emits `application/yaml` (yaml:1795 vs handler_docs.go:104). +- evidence: + ```yaml + /api/history: + get: + parameters: + - $ref: '#/components/parameters/AccountID' + - $ref: '#/components/parameters/StartDate' + - $ref: '#/components/parameters/EndDate' + - name: interval + ``` +- suggested fix: Add a test that diffs `registerRoutes()` (path+method) against the spec's paths and fails on any unlisted route, then backfill the missing operations and correct the listed mismatches in one pass. +- verdict: CONFIRMED — the spec's path keys (openapi.yaml:44-1800) contain no /api/accounts*, /api/registrations*, /api/register*, /api/recommendations/freshness, /api/auth/mfa/*, /api/inventory/*, /api/ladder/configs, /api/analytics/trends or /version entries although router.go registers them (e.g. router.go:264-279, 334-342, 369); spot-checked mismatches hold: openapi.yaml:1080-1088 lists `interval` and omits provider/account_ids/limit versus parseHistoryFilters (handler_history.go:666-706), openapi.yaml:137 enumerates bare service names versus the provider/service split at handler_config.go:297-303, and openapi.yaml:1795 says application/x-yaml versus handler_docs.go:104 application/yaml. +- issue: (pending cross-reference) + +### A02-019 Store errors are classified by substring matching in four handlers +- category: hygiene +- severity: low +- location: internal/api/handler_accounts.go:268 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: `isDuplicateKeyError` tests `err.Error()` for "duplicate key" / "23505" while forty lines later `deleteAccount` (lines 644-645) does the typed `errors.As(err, &pgErr)` check; `submitRegistration` matches "duplicate" (handler_registrations.go:106), `mergeServiceConfig` matches "not found" (handler_config.go:219) and `isResetPasswordClientError` keeps a list of five message fragments (handler_auth.go:291-298). A wrapped error whose message is reworded (or a legitimately different error that happens to contain "not found") flips a 409/400 into a 500 or vice versa, and `mergeServiceConfig` would treat a transport error mentioning "not found" as "no existing row" and overwrite filter fields it was meant to preserve. +- evidence: + ```go + s := err.Error() + return strings.Contains(s, "duplicate key") || strings.Contains(s, "23505") + ``` +- suggested fix: Use `pgconn.PgError` codes / `config.ErrNotFound` / exported auth sentinels with `errors.Is`/`errors.As` in all four sites and delete the substring helpers. +- verdict: CONFIRMED — isDuplicateKeyError (handler_accounts.go:262-268), submitRegistration (handler_registrations.go:106), mergeServiceConfig (handler_config.go:217-221) and isResetPasswordClientError (handler_auth.go:291-298) all branch on err.Error() substrings, while deleteAccount (handler_accounts.go:644-645) does the typed pgconn.PgError/errors.As check in the same package; the mergeServiceConfig "not found" branch really does return cfg unchanged on any error whose text contains that phrase. +- issue: (pending cross-reference) + +### A03-021 Test doubles and testify/mock are compiled into the production auth package +- category: hygiene +- severity: low +- location: internal/auth/test_helpers.go:15 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: `test_helpers.go` is not a `_test.go` file, so `MockStore`, `MockEmailSender`, `TestCSRFKey`, `DeriveTestCSRFToken`, the fixed `testCSRFKey` and the `github.com/stretchr/testify/mock` dependency ship in every server, Lambda and CLI binary. A caller can construct a service with the well-known test CSRF key; and the mocks are what let the proof-of-concept for A03-001 through A03-009 be driven from outside the package. +- evidence: + ```go + // MockStore is a mock implementation of the auth store for testing. + type MockStore struct { + mock.Mock + } + ``` +- suggested fix: Move the doubles to an `authtest` package or to `export_test.go`/`_test.go` files; `internal/mocks` already exists for cross-package mocks. +- verdict: CONFIRMED — internal/auth/test_helpers.go is a non-_test file in package auth importing testify/mock, testify/require and "testing" (:3-12) and exporting MockStore, MockEmailSender and DeriveTestCSRFToken with the fixed testCSRFKey (:14-17, :281-303), so it compiles into every binary that imports internal/auth; no non-test code references these symbols (grep outside the file returns nothing), so the "caller can construct a service with the test key" claim describes reachable-by-linking, not an in-tree call. +- issue: (pending cross-reference) + +### A03-024 Group creation records no creator for any actor +- category: hygiene +- severity: low +- location: internal/auth/service_api.go:313 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: `CreateGroupAPI` passes `""` as `createdBy` for every caller because the admin-API-key sentinel is not a UUID. Human admins therefore also leave `created_by = NULL`, so after A03-001/A03-004 there is no audit trail of who created the escalating group. +- evidence: + ```go + // Use empty string for createdBy: the column is a UUID FK and + // actorUserID may be the non-UUID admin-API-key sentinel. + if err := s.CreateGroup(ctx, group, ""); err != nil { + ``` +- suggested fix: Pass `actorUserID` when it is not `AdminAPIKeyActorID` and `""` only for the sentinel. +- verdict: CONFIRMED — CreateGroupAPI receives the real actorUserID from the handler (handler_groups.go:53) and hands CreateGroup a literal "" for every caller (service_api.go:311-313), so human admins leave created_by unset even though the sentinel case (group_ceiling.go:40) is the only one that cannot be stored. +- issue: (pending cross-reference) + +### A04-018 Auto-heal naming and the newMigratorWithRecovery comment promise a recovery that no longer happens +- category: hygiene +- severity: low +- location: internal/database/postgres/migrations/migrate.go:114-118 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd (behaviour at 381-421) +- failure scenario: the call-site comment says a dirty row is cleared "at the CURRENT recorded version so the subsequent Up() re-applies any pending migrations", and the gate is named `CUDLY_MIGRATION_AUTOHEAL` with a default of true. `maybeAutoHealDirty` heals nothing; it returns an error. An operator following the comment (or the memory note describing Force(current)+Up as default-on) leaves `CUDLY_MIGRATION_AUTOHEAL=true` expecting self-recovery and gets a permanently dirty database until they discover `CUDLY_FORCE_MIGRATION_VERSION`. +- evidence: + ```go + // Default-on dirty auto-heal: when the schema_migrations row is dirty, + // clear the dirty flag at the CURRENT recorded version so the subsequent + // Up() re-applies any pending migrations, letting a cold start self-recover + // instead of staying broken until a manual force. + if err := maybeAutoHealDirty(m); err != nil { + ``` +- suggested fix: rename to `maybeRefuseDirty` / `CUDLY_MIGRATION_DIRTY_CHECK` and rewrite the call-site comment to match the fail-loud behaviour. +- verdict: CONFIRMED — the call-site comment promises "clear the dirty flag at the CURRENT recorded version so the subsequent Up() re-applies any pending migrations" (internal/database/postgres/migrations/migrate.go:114-118), but `maybeAutoHealDirty` calls no `Force` and returns an error on every dirty row (migrate.go:381-421), and `autoHealEnabled` defaults the gate to true (migrate.go:428). +- issue: (pending cross-reference) + +### A08b-044 GCP billing service IDs and the 8760-hour year are inline literals across seven files +- category: hygiene +- severity: low +- location: providers/gcp/services/cloudsql/client.go:350 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd (service IDs also at `cloudstorage/client.go:343`, `memorystore/client.go:337`; `8760.0` at `cache/client.go:548`, `cosmosdb/client.go:548`, `database/client.go:567`, `search/client.go:463`, `managedredis/client.go:492`, `synapse/client.go:439`, and the three GCP files) +- failure scenario: the opaque service ID `services/9662-B51E-5089` appears with no name and no comment, so a reader cannot tell which GCP service is being priced and a transposed digit produces an empty catalog that surfaces as "no pricing found". `8760.0` is repeated in nine files as a bare literal on money paths, with no shared constant to change if the term-to-hours convention is ever revisited. +- evidence: + ```go + skus, err := svc.ListSKUs("services/9662-B51E-5089") + if err != nil { + return nil, fmt.Errorf("failed to list SKUs: %w", err) + } + ``` +- suggested fix: name each service ID as a documented package constant and hoist `hoursPerYear = 8760.0` into the shared pricing helper both providers already import. +- verdict: CONFIRMED — the four service IDs are inline and uncommented at cloudsql:350, cloudstorage:343, memorystore:337 and computeengine:961, and the count of bare `8760.0` literals is higher than stated: twelve provider files, adding compute/client.go:709 and savingsplans/client.go:411 to the nine listed. +- issue: (pending cross-reference) + +### A09-030 Retry documents a per-attempt context as independent of the outer context, and jitter escapes MaxDelay +- category: hygiene +- severity: low +- location: pkg/retry/exponential.go:145 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: `context.WithTimeout(ctx, ...)` produces a child, so cancelling the outer context cancels the attempt — the opposite of the "independent of the outer ctx so a slow attempt fails fast and the retry budget continues" claim repeated at lines 36-38 and 96-97. A caller relying on that sentence will size `PerAttemptTimeout` on the assumption that a cancelled parent still lets the current attempt finish. Separately, `backoffFor` applies the ±25% jitter factor after the `MaxDelay` clamp, so with `MaxDelay = 30s` an actual sleep of up to 37.5s occurs, exceeding the documented cap. +- evidence: + ```go + perAttemptCtx, cancel := context.WithTimeout(ctx, cfg.PerAttemptTimeout) + defer cancel() + return op(perAttemptCtx, attempt) + ``` +- suggested fix: reword the doc to say the per-attempt deadline is *additional* to the outer context, and clamp the post-jitter delay to `MaxDelay` so the documented cap holds. +- verdict: CONFIRMED — both halves check out. `runAttempt` derives the per-attempt context from the caller's ctx via `context.WithTimeout(ctx, cfg.PerAttemptTimeout)` (pkg/retry/exponential.go:141-147), so cancelling the parent cancels the attempt, contradicting the "independent of the outer ctx" wording at exponential.go:36-38 and :92-94. And `backoffFor` applies the ±25% factor after both `MaxDelay` clamps (exponential.go:168-186), so a 30s MaxDelay yields sleeps up to 37.5s. +- issue: (pending cross-reference) + +### A12-049 `HOURS_PER_MONTH` is 730 but its comment derives 730.5 +- category: hygiene +- severity: low +- location: frontend/src/modules/savings-history.ts:25 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: The comment states the derivation `365.25 * 24 / 12`, which is 730.5, while the constant is 730. A reader auditing the $/hr conversion cannot tell whether the value or the comment is authoritative, and neither matches the 720 used by `recommendations.ts` (A12-012). +- evidence: + ```ts + const HOURS_PER_MONTH = 730; // 365.25 * 24 / 12 + ``` +- suggested fix: Fix the comment to name the convention actually used (AWS's 730 hours/month), and share the constant with `recommendations.ts`. +- verdict: CONFIRMED — The constant is 730 while its own comment derives 365.25 * 24 / 12 = 730.5 (frontend/src/modules/savings-history.ts:25), and frontend/src/recommendations.ts:1471 independently uses 1/720 with a 24 x 30 comment. +- issue: (pending cross-reference) + +### A12-070 The staleness banner cannot distinguish soft from hard and contradicts its own docstring +- category: hygiene +- severity: low +- location: frontend/src/riexchange.ts:548 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: The function contract at line 509 says soft means "data may be up to 12 h old", but the rendered soft copy says "may be up to 24h old" and the hard copy says "older than 24h". Both banners therefore tell the operator the same thing about price freshness before they act on a cross-family alternative; only the colour differs. +- evidence: + ```ts + const copy = isSoft + ? `Cross-family alternatives are based on Cost Explorer recommendations that may be up to 24h old${ageLabel}. Some prices may be stale.` + : `Cross-family alternatives are based on Cost Explorer recommendations older than 24h${ageLabel}. Prices may be significantly out of date.`; + ``` +- suggested fix: Make the soft copy say 12h, or drop the hardcoded hour figures and rely on `ageLabel`. +- verdict: CONFIRMED — The contract says soft means data may be up to 12 h old (frontend/src/riexchange.ts:509) while the rendered soft copy says 24h and the hard copy says older than 24h (frontend/src/riexchange.ts:547-550), so both banners state the same freshness bound. +- issue: (pending cross-reference) + +### A13-016 `.trivyignore` GCP-0015 describes a variable that does not exist +- category: hygiene +- severity: low +- location: .trivyignore:93 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: The suppression says the GCP database module "exposes require_ssl as a variable" and is "Tracked as a hardening follow-up to set require_ssl = true in the module default". There is no `require_ssl` variable in `terraform/modules/database/gcp/`; both `ip_configuration` blocks hardcode `ssl_mode = "ENCRYPTED_ONLY"` (main.tf:84 and main.tf:183), which already requires TLS. The suppression is either stale against a fixed finding or masking a different one, and it points a future reviewer at a follow-up that cannot be done. +- evidence: + ```hcl + ip_configuration { + ipv4_enabled = var.enable_public_ip + private_network = var.vpc_network_id + ssl_mode = "ENCRYPTED_ONLY" + ``` +- suggested fix: Remove the entry and re-run `trivy config` to confirm it no longer fires; if it does, replace the justification with the real cause. +- verdict: CONFIRMED — `/usr/bin/grep -rn 'require_ssl|ssl_mode' terraform/modules/database/gcp/` returns no `require_ssl` anywhere; both `ip_configuration` blocks hardcode `ssl_mode = "ENCRYPTED_ONLY"` (main.tf:84, main.tf:183) and outputs.tf:55 emits `ssl_mode = "require"` for the connection string. The suppression at .trivyignore:92-98 therefore points a future reviewer at a variable that does not exist and a follow-up that cannot be performed. +- issue: (pending cross-reference) + +### A13-017 `.snyk` declares severity and license policy keys that the Snyk policy format does not read +- category: hygiene +- severity: low +- location: .snyk:26 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: The `.snyk` policy file supports `version`, `ignore`, `patch` and `exclude`. `failOnSeverity` and `license` are not policy-file keys: Snyk takes the severity gate from `--severity-threshold` (which `Makefile:232` and the ci.yml snyk job pass explicitly) and license policy from organization settings. A reader auditing the repo's license posture sees an allow/deny list that nothing enforces, so a GPL-3.0 dependency would be admitted despite the file appearing to forbid it. +- evidence: + ```yaml + # Severity thresholds + failOnSeverity: high + + # License policy + license: + allow: + - MIT + ``` +- suggested fix: Delete the two inert blocks, or move the license policy to a real enforcement point and leave a comment saying where it lives. +- verdict: CONFIRMED — .snyk declares `ignore`, `patch`, `exclude` (and omits `version` entirely), then `failOnSeverity: high` at line 26 and a `license` allow/deny map at lines 29-43, neither of which is a Snyk policy-file key. Both real enforcement points pass the threshold on the command line instead: `Makefile:232` runs `snyk test --severity-threshold=high` and .github/workflows/ci.yml:832 passes `args: --severity-threshold=high`, and no consumer reads the license map, so a GPL-3.0 dependency would be admitted despite the deny list. +- issue: (pending cross-reference) + +### A13b-015 Registration response is explicitly marked non-sensitive while its own description says it carries a reference token +- category: hygiene +- severity: low +- location: iac/federation/aws-target/terraform/outputs.tf:24 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: All five modules declare `registration_response` with `sensitive = false` set explicitly, and describe it as containing the reference token used for status checks. The raw response body is therefore printed by `terraform apply` and lands in CI logs and any stored plan output. The same file marks the far less sensitive `cudly_account_registration` block as sensitive in the Azure module (`iac/federation/azure-target/terraform/outputs.tf:21`), so the two are inconsistent about what deserves redaction. Identical in the `aws-cross-account`, `azure-target`, `gcp-sa-impersonation` and `gcp-target` outputs files. +- evidence: + ```hcl + output "registration_response" { + description = "CUDly registration API response (contains reference_token for status checks)" + value = local.do_register ? data.http.cudly_registration[0].response_body : "Skipped (cudly_api_url or contact_email not set)" + sensitive = false + } + ``` +- suggested fix: Set `sensitive = true` in all five modules and tell the customer to read it with `terraform output -raw registration_response`. +- verdict: CONFIRMED — the `registration_response` output is byte-identical with an explicit `sensitive = false` in all five `outputs.tf`, and the body it prints really does carry the token: `internal/api/handler_registrations.go:128-131` returns `reference_token`, `internal/config/store_postgres_registrations.go:79-81` uses it as the lookup key for `GetAccountRegistrationByToken`, and `internal/api/handler_registrations.go:373` treats it as withheld even from admins; the inconsistency with the redacted `cudly_account_registration` block at `iac/federation/azure-target/terraform/outputs.tf:11-22` is as described. +- issue: (pending cross-reference) + +### A13c-017 A live deployed Lambda Function URL hostname is hardcoded in a tracked tfvars file with no mechanism to keep it current +- category: hygiene +- severity: low +- location: terraform/environments/aws/github-dev.tfvars:32 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: the Function URL ID is server-assigned and only known after apply, so it + cannot be self-referenced without a cycle and has to be pasted by hand — the comment three lines + above says exactly that ("Update the Lambda Function URL entry when the dev environment is + redeployed"). Any redeploy that reissues the URL silently invalidates the CORS allowlist, and + the failure surfaces as a browser CORS error rather than a Terraform diff. The file is + deliberately un-gitignored (`terraform/.gitignore` negates `github-*.tfvars`), so a live + environment's public endpoint is published in the repository. +- evidence: + ```hcl + lambda_allowed_origins = [ + "https://33pz7pombdqwu3bdlxp4lqxyra0bsriy.lambda-url.us-east-1.on.aws", + "http://localhost:3000", + ] + ``` +- suggested fix: supply the origin via `TF_VAR_lambda_allowed_origins` from the deploy workflow + (which can read the previous apply's output) rather than committing the host. +- verdict: CONFIRMED — the hostname is committed at environments/aws/github-dev.tfvars:32, the + manual-update comment is three lines above at :30, and `terraform/.gitignore` ignores `*.tfvars` + then re-admits `!github-*.tfvars`, so the file is tracked by design. The endpoint is a public + URL rather than a credential, which is consistent with the low severity. +- issue: (pending cross-reference) + +### A15-011 Twelve source files exceed 500 lines by more than double, against the project's own limit +- category: hygiene +- severity: low +- location: internal/api/handler_purchases.go:1 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: `CLAUDE.md` states "Keep files under 500 lines". Forty non-test source files exceed it; twelve exceed 1000. The length hides mixed responsibilities rather than merely being long: `handler_purchases.go` holds 85 top-level functions spanning session authorization (`authorizeSession*`), request validation (`validateExecute*`, `validateRevoke*`), notification dispatch (`sendPurchase*`), revocation, cancellation, approval and delayed scheduling. `frontend/src/recommendations.ts` is 5481 lines with 50 exports and 68 functions. Reviewing a money-path change in these files means reading past several unrelated concerns, and the file is a permanent merge-conflict hotspot for concurrent PRs. +- evidence: + ```text + 5481 frontend/src/recommendations.ts 1847 frontend/src/history.ts + 3823 internal/config/store_postgres.go 1784 internal/mocks/stores.go + 3715 frontend/src/settings.ts 1641 internal/api/handler_accounts.go + 3316 internal/api/handler_purchases.go 1536 frontend/src/auth.ts + 2594 internal/api/handler_ri_exchange.go 1519 internal/scheduler/scheduler.go + 2359 frontend/src/plans.ts 1432 providers/gcp/.../computeengine/client.go + 2270 frontend/src/riexchange.ts + ``` +- suggested fix: Split by the concern boundaries already visible in the function-name prefixes, starting with `handler_purchases.go` (authorization, validation and notification each move to their own file) rather than attempting all forty. +- verdict: CONFIRMED — the quoted line counts reproduce exactly, and the two inventory claims check out: `handler_purchases.go` has 85 `^func ` declarations whose prefixes cluster as authorize (7), build (6), resolve (5), validate (4), require (4), approve (4), send (3), cancel (3), revoke (2), and `frontend/src/recommendations.ts` has 50 `^export ` lines across 5481 lines. The counts understate rather than overstate: across tracked non-test `.go` and `frontend/src/*.ts`, 91 files exceed 500 lines and 22 exceed 1000, against the finding's "forty" and "twelve" (its own table already lists 13 files over 1000). Read the verdict with the same caveat as the finding: this rests on function inventories and name prefixes, not on a line-by-line reading of the four largest files, so it evidences mixed responsibilities by naming clusters rather than by tracing each function's behaviour. +- severity-adjusted: unchanged at low, but the scope is roughly double what is reported. +- issue: (pending cross-reference) + +### Category: bug + +1 finding: 1 low. + +This label is outside the report taxonomy; the reviewer wrote it as-is and it is preserved. + +### A10-029 The RDS instance pagination loop has no page cap and never checks context cancellation +- category: bug +- severity: low +- location: cmd/multi_service_engine_versions.go:138 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: `queryRDSInstancesInRegion` loops on the marker with no iteration ceiling and no `ctx.Err()` check. If the API returns a non-advancing marker, the loop calls `DescribeDBInstances` forever, appending duplicate `InstanceEngineVersion` entries into the shared map under the mutex, which inflates `excludedCount` in `adjustRecommendationForExcludedVersions` and can drive recommendation counts to zero. Cancelling the CLI does not stop the worker; only the shared `sem` bounds how many regions spin at once. The sibling loop `fetchMajorEngineVersionsForEngine` (line 235-244) has both guards. +- evidence: + ```go + var marker *string + for { + localVersions, nextMarker, err := queryRDSInstancesPage(ctx, rdsClient, marker, regionName) + ... + if nextMarker == nil { break } + marker = nextMarker + } + ``` +- suggested fix: Add the same `if err := ctx.Err(); err != nil { return }` check and a page cap constant that `fetchMajorEngineVersionsForEngine` already uses. +- verdict: PLAUSIBLE — the missing guards are real: cmd/multi_service_engine_versions.go:137-156 has neither the ctx.Err() check nor the maxEngineVersionPages cap its sibling carries at :235-244 (tested at cmd/multi_service_engine_versions_paginate_test.go:144). The runaway needs the runtime condition of AWS returning a non-advancing marker, and the "cancelling the CLI does not stop the worker" half is wrong: a cancelled context makes DescribeDBInstances error, which hits the break at :140-143. +- issue: (pending cross-reference) + +### Category: bugs + +4 findings: 4 low. + +This label is outside the report taxonomy; the reviewer wrote it as-is and it is preserved. + +### A05-013 `executableByScheduler` dereferences the plan without the nil guard its sibling has +- category: bugs +- severity: low +- location: internal/purchase/manager.go:640 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: `executePurchase` explicitly handles `GetPurchasePlan` returning `(nil, nil)` (execution.go:53) because the store interface permits it; the AutoPurchase gate does not, and `plan.AutoPurchase` panics on a nil plan. The panic is inside `processOneExecution`, called from `ProcessScheduledPurchases` in the scheduler Lambda, with no `recover()` on that path (the only recover in this shard is inside `FanOutWithConcurrency`'s goroutines), so one such row takes down the whole tick and every later due row goes unprocessed. The Postgres store happens to return `ErrNotFound` today, so this is latent rather than live, but the asymmetry with `executePurchase` is exactly what makes it easy to trip on the next store or mock. +- evidence: + ```go + plan, err := m.config.GetPurchasePlan(ctx, exec.PlanID) + if err != nil { + return false, fmt.Errorf("failed to fetch plan %s for AutoPurchase gate: %w", exec.PlanID, err) + } + return plan.AutoPurchase, nil + ``` +- suggested fix: add `if plan == nil { return false, fmt.Errorf(...) }` before the dereference — fail closed, matching the function's own stated "no silent money action on error" policy. +- verdict: PLAUSIBLE — the asymmetry is real (executePurchase guards `if plan == nil` at execution.go:52-54; executableByScheduler dereferences `plan.AutoPurchase` at manager.go:643 with no guard) and the panic would indeed escape, since the only recover() calls in this shard are internal/execution/fanout.go:115 and scheduler.go:1336, neither on the processOneExecution path. Reaching it needs a store returning (nil, nil): PostgresStore.GetPurchasePlan wraps pgx.ErrNoRows as ErrNotFound (store_postgres.go:599-601) and the interface (internal/config/interfaces.go:31) states no contract, so the crash is latent as the finding itself says. +- issue: (pending cross-reference) + +### A07-031 A transient STS failure permanently disables account-ID resolution for the client +- category: bugs +- severity: low +- location: providers/aws/services/redshift/client.go:307 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: `resolveAccountID` caches the outcome in `sync.Once`. If the first call happens under a cancelled context or a throttled STS, `accountErr` is stored and every later call on the same client returns that stale error without retrying. For Redshift this is on the pre-purchase guard path, so `findNodeByIdempotencyToken` fails closed for the client's whole lifetime and blocks every subsequent purchase. The identical construct is at opensearch/client.go:313, where it only disables tagging. +- evidence: + ```go + c.accountOnce.Do(func() { + if c.stsClient == nil { + return + } + out, err := c.stsClient.GetCallerIdentity(ctx, &sts.GetCallerIdentityInput{}) + if err != nil { + c.accountErr = err + return + } + ``` +- suggested fix: cache only a successful resolution (guard with a mutex and re-attempt when `accountID` is still empty) so a transient STS error does not become permanent. +- verdict: CONFIRMED — `resolveAccountID` (providers/aws/services/redshift/client.go:306-320) stores `c.accountErr` inside `sync.Once` and returns it on every later call, and `findNodeByIdempotencyToken` :226-230 converts that into a hard "resolve account ID for idempotency check" failure for the lifetime of that client instance; opensearch/client.go has the same construct on a tagging-only path. +- issue: (pending cross-reference) + +### A08-012 The reservation-order idempotency lookup paginates with no cycle guard and no page cap +- category: bugs +- severity: medium +- location: providers/azure/services/internal/reservations/purchase.go:476 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: a `nextLink` that points back at itself (or an unbounded chain) makes the loop spin forever. This runs before every purchase and before every retry inside `purchaseTwoStepGuarded`, so a purchase with a caller-supplied context that has no deadline hangs indefinitely instead of failing. The repo already owns a walker with both guards for exactly this class of loop: `pricing.FetchAll` enforces a seen-URL guard, a max-pages cap and a per-page timeout (providers/azure/internal/pricing/retail_prices.go:63). +- evidence: + ```go + nextURL := ReservationOrdersListURL() + for nextURL != "" { + page, err := fetchReservationOrdersPage(ctx, httpClient, nextURL, bearerToken) + if err != nil { return "", false, err } + if orderID, found := matchReservationOrderInPage(page, idempotencyToken); found { return orderID, true, nil } + nextURL = page.NextLink + } + ``` +- suggested fix: add the same seen-URL set and page cap this loop's sibling walker already has, erroring on a repeat link rather than looping. +- verdict: CONFIRMED — `FindReservationOrderByIdempotencyToken` (reservations/purchase.go:475-484) follows `nextLink` with no seen-URL set and no page cap, while the sibling `pricing.FetchAll` enforces both plus a per-page timeout (internal/pricing/retail_prices.go:63-86). +- severity-adjusted: low — the scheduler/web purchase path runs every rec under a 30s context (internal/purchase/execution.go:1007), so the unbounded spin needs a caller supplying a deadline-free context. +- issue: (pending cross-reference) + +### A08-028 RI-utilization pagination has neither a page cap nor a cancellation check +- category: bugs +- severity: low +- location: providers/azure/ri_utilization.go:104 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: `getRIUtilizationViaAPI` walks `pager.More()` with no bound and no `ctx.Err()` check between pages, unlike every pagination loop in the compute client, which caps pages and checks cancellation on each iteration (compute/client.go:262-268). A large tenant, or a pager that keeps reporting `More()`, keeps the call running past the caller's intent. Rows with a nil `ReservationID` are also silently discarded by `accumulateSummary` (ri_utilization.go:125), so a partial summary is reported as a complete utilization figure. +- evidence: + ```go + for pager.More() { + page, err := pager.NextPage(ctx) + if err != nil { + return nil, fmt.Errorf("azure ri utilization: failed to fetch summaries page: %w", err) + } + for _, summary := range page.Value { + accumulateSummary(agg, summary) + } + } + ``` +- suggested fix: add the same `pageIdx >= maxPages` cap and per-iteration `ctx.Err()` check the compute client's loops use. +- verdict: PLAUSIBLE — the page cap really is absent at ri_utilization.go:104-113 where the sibling `collectVMReservations` has both guards (compute/client.go:262-268), but `pager.NextPage(ctx)` already fails a cancelled context on every page fetch, so only a pager that keeps reporting `More()` produces the runaway, and the Azure two-argument `GetRIUtilization` (ri_utilization.go:65) has no caller outside ri_utilization_test.go — the API layer calls the three-argument AWS shape (internal/server/ladder_write.go:55). The nil-`ReservationID` skip is documented at ri_utilization.go:118-120, not silent. +- issue: (pending cross-reference) + +## Rejected findings + +The verifier found each of these wrong. They are kept with their reasoning so the same claim is +not re-raised. Do not action anything in this section. + +| ID | Category | Reviewer severity | Location | +|---|---|---|---| +| A03-013 | silent-fallback | medium | `internal/auth/service_group.go:414` | +| A03-017 | security | low | `internal/credentials/resolver.go:266` | +| A03-018 | security | low | `internal/secrets/azure_resolver.go:32` | +| A04-014 | correctness | low | `internal/config/store_postgres.go:1083-1090` | +| A05-016 | concurrency | low | `internal/scheduler/scheduler.go:600` | +| A06-002 | security | high | `internal/server/http.go:207` | +| A06-011 | security | low | `internal/email/smtp_sender.go:327` | +| A06-021 | correctness | low | `internal/accounts/org_discovery.go:48` | +| A06-023 | correctness | low | `internal/server/handler_ri_exchange.go:165` | +| A07-020 | silent-fallback | medium | `providers/aws/recommendations/converters.go:27` | +| A07-024 | money-path | medium | `providers/aws/services/ec2/client.go:1043` | +| A07-025 | silent-fallback | medium | `providers/aws/provider.go:476` | +| A07-030 | silent-fallback | low | `providers/aws/provider.go:419` | +| A08-011 | money-path | high | `providers/azure/services/savingsplans/client.go:490` | +| A08-025 | hygiene | low | `providers/azure/services/compute/client.go:448` | +| A09-017 | money-path | high | `pkg/common/service_details_codec.go:98` | +| A09-027 | security | low | `pkg/ladder/plan.go:195` | +| A09-029 | silent-fallback | low | `pkg/common/identifiers.go:33` | +| A11-003 | money-path | high | `frontend/src/groups/groupModals.ts:269` | +| A11-007 | money-path | high | `frontend/src/app.ts:398` | +| A11-010 | correctness | medium | `frontend/src/state.ts:167` | +| A11-021 | correctness | low | `frontend/src/users/filters.ts:131` | +| A11-025 | hygiene | low | `frontend/src/utils.ts:224` | +| A12-007 | correctness | high | `frontend/src/plans.ts:1253` | +| A12-016 | silent-fallback | medium | `frontend/src/ladder.ts:437` | +| A12-025 | money-path | medium | `frontend/src/dashboard.ts:751` | +| A12-027 | silent-fallback | medium | `frontend/src/plans.ts:1361` | +| A12-036 | test-gap | medium | `frontend/src/__tests__/history-marketplace-sell-button.test.ts:252` | +| A12-046 | silent-fallback | low | `frontend/src/plans.ts:1908` | +| A12-050 | over-engineering | low | `frontend/src/modules/savings-history.ts:576` | +| A12-051 | correctness | low | `frontend/src/plans.ts:2294` | +| A13b-006 | correctness | medium | `iac/federation/aws-cross-account/cloudformation/template.yaml:68` | +| A13c-008 | security | medium | `terraform/modules/compute/gcp/cloud-run/main.tf:454` | +| A13c-010 | ops | medium | `terraform/modules/compute/aws/fargate/main.tf:764` | +| A13c-011 | correctness | low | `terraform/modules/secrets/gcp/main.tf:183` | +| A13c-026 | hygiene | low | `terraform/modules/email/azure/main.tf:10` | +| A14-001 | security | critical | `.github/workflows/deploy-azure.yml:232` | +| A14-013 | ops | medium | `scripts/security-scan.sh:187` | +| A14-014 | ops | medium | `scripts/security-scan.sh:71` | +| A14-019 | ops | medium | `.github/workflows/deploy-aws-fargate.yml:173` | +| A15-005 | test-gap | medium | `internal/scheduler/scheduler_test.go:1048` | +| A15-010 | ops | low | `.github/workflows/ci.yml:646` | +| A16-003 | test-gap | high | `internal/purchase/approvals.go:494` | +| A16-010 | test-gap | medium | `internal/scheduler/scheduler_test.go:1048` | +| A16-013 | test-gap | medium | `internal/purchase/execution.go:766` | + +### A03-013 Constraint matching treats an absent request dimension or a zero amount as satisfied +- category: silent-fallback +- severity: medium +- location: internal/auth/service_group.go:414 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: A group grants `execute:purchases` constrained to `AccountIDs=[acct-A], MaxPurchaseAmount=100`. `permissionsAllow` is called with a request constraint set whose `AccountIDs` is empty (a caller that did not populate it, a region-agnostic Savings Plan path, or a provider whose record lacks the field) or whose amount is 0 (unknown cost). `matchStringListConstraints` returns true because `len(reqList) == 0`, and `matchPurchaseAmountConstraint(100, 0)` returns true because `reqMax > permMax` is false. The constrained grant therefore authorizes any account and any amount whenever the caller under-fills the request side; the code comments push the burden onto "callers that need ..." but nothing in scope enforces it. This is the empty-means-unrestricted shape that #1748 removed from account scope, still present on the permission-constraint axis. +- evidence: + ```go + func matchStringListConstraints(permList, reqList []string) bool { + if len(permList) > 0 && len(reqList) > 0 { + return containsAny(permList, reqList) + } + return true + } + ... + func matchPurchaseAmountConstraint(permMax, reqMax float64) bool { + if permMax > 0 && reqMax > permMax { + return false + } + ``` +- suggested fix: When the permission constrains a dimension, require the request to name it (empty request list against a non-empty permission list is a refusal; zero amount against a positive cap is a refusal), mirroring `listCovers` and `amountCovers` in group_ceiling.go, which already implement the strict polarity. +- verdict: REJECTED — the matcher polarity is as described (service_group.go:414-419, :459-464), but every caller of the constraint check fills the request side: purchaseConstraintSets (handler_purchases.go:2221-2242) always populates Providers/Services/Regions/AccountIDs, substituting the `unattributedAccountConstraint` sentinel (handler.go:466) for a missing account, requireNonZeroCommitment (handler_purchases.go:2197) refuses a zero total before the check, and checkAzureExecuteConstraints (handler_ri_exchange.go:1064-1096) names account/provider/service and still returns 403 when it probes with MaxPurchaseAmount=0; HasPermissionForConstraintsAPI also fails loud on an empty set (service_api.go:415-420), so no in-tree caller reaches the under-filled case. + +### A03-017 Web-identity token file path is tenant-supplied and read from the host filesystem +- category: security +- severity: low +- location: internal/credentials/resolver.go:266 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: `AWSWebIdentityTokenFile` comes from the cloud-account record (writable by any `create:accounts`/`update:accounts` holder). The only check is a substring test for `..` and `filepath.IsAbs`, so any absolute path on the host (`/proc/self/environ`, `/var/run/secrets/...`, another tenant's projected token) is opened by the CUDly process and its contents sent to STS as the web identity token for a role ARN the same tenant chose. The substring check also rejects legitimate names containing `..` inside a component. Host-level path policy belongs to the operator, not to tenant data. +- evidence: + ```go + if strings.Contains(tokenFile, "..") || !filepath.IsAbs(tokenFile) { + return nil, fmt.Errorf("credentials: aws_web_identity_token_file must be an absolute path without '..' (account %s)", account.ID) + } + ``` +- suggested fix: Take the token file exclusively from deployment configuration (`AWS_WEB_IDENTITY_TOKEN_FILE`) or an operator allow-list of directories, and drop the per-account field. +- verdict: REJECTED — the API boundary already enforces the operator allow-list the fix asks for: validateAWSWebIdentityTokenFile (internal/api/validation.go:105-119) rejects ".." and requires the path to start with /var/run/secrets/eks.amazonaws.com/serviceaccount/ or /var/run/secrets/kubernetes.io/serviceaccount/ (awsWebIdentityTokenFilePrefixes, :51-54), and it runs via validateAWSAuthMode:419 from validateCloudAccountRequest on create (handler_accounts.go:316), update (:558) and registration (handler_registrations.go:262), so /proc/self/environ and arbitrary host paths are unreachable from tenant data; the resolver check at resolver.go:266 is a second, weaker layer, not the only one. + +### A03-018 The IMDS block in the Azure secrets client checks literal IPs only +- category: security +- severity: low +- location: internal/secrets/azure_resolver.go:32 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: `blockIMDSDialer` compares the dial `host` string against two literal addresses. A hostname that resolves to 169.254.169.254 (attacker DNS, or a `169.254.169.254.nip.io`-style name), the GCP metadata name `metadata.google.internal`, or a redirect to such a name is dialled normally because resolution happens after the check. The guard therefore blocks only the most literal spelling of the SSRF it documents. +- evidence: + ```go + host, _, err := net.SplitHostPort(addr) + if err != nil { + host = addr + } + if imdsAddresses[host] { + return nil, fmt.Errorf("connection to metadata endpoint %s is blocked", host) + } + return d.inner.DialContext(ctx, network, addr) + ``` +- suggested fix: Resolve the host first (or wrap `Control` on the dialer) and reject link-local (169.254.0.0/16, fe80::/10) and the metadata hostnames on the resolved address; share the implementation with `providers/azure/internal/httpclient`, which solves the same problem. +- verdict: REJECTED — the guard is literal-only as described (azure_resolver.go:32-41), but the client it protects talks only to the vault named by the operator's AZURE_KEY_VAULT_URL env var (secrets/resolver.go:58-81) with secret names as path segments (azure_resolver.go:103, :121), so no tenant-controlled host reaches this dialer and the hostname-resolution bypass has no attacker input; the suggested sibling providers/azure/internal/httpclient delegates to pkg/httpclient, which uses the same literal-address map (pkg/httpclient/httpclient.go:29-47), so it does not solve the stated problem either. + +### A04-014 TransitionExecutionStatus RETURNING projects raw cancelled_by while every other execution read projects COALESCE(canceled_by, cancelled_by) +- category: correctness +- severity: low +- location: internal/config/store_postgres.go:1083-1090 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: rows canceled by new code carry `canceled_by` only (`CancelExecutionAtomic`). A transition out of such a row (any caller listing a canceled spelling in `fromStatuses`) returns `CancelledBy == nil`; a follow-up `SavePurchaseExecution` of that struct writes `cancelled_by = NULL` and, while `canceled_by` is retained, any consumer of the returned record loses the actor. The pgxmock guard `TestPGXMock_GetExecutionByID_ProjectsCoalescedCancelledBy` covers only `GetExecutionByID`, so this projection drifted unnoticed. +- evidence: + ```go + RETURNING plan_id, execution_id, status, step_number, scheduled_date, + notification_sent, approval_token, recommendations, + total_upfront_cost, estimated_savings, completed_at, error, expires_at, + cloud_account_id, source, approved_by, cancelled_by, capacity_percent, + ``` +- suggested fix: project `COALESCE(canceled_by, cancelled_by) AS cancelled_by` in the RETURNING list and extend the pgxmock projection test to every method that feeds `scanExecutionRows`. +- verdict: REJECTED — the projection asymmetry is real (raw `cancelled_by` at internal/config/store_postgres.go:1086 versus `COALESCE(canceled_by, cancelled_by)` at 1554), but the failure needs a transition *out of* a canceled row and no caller passes a canceled spelling in `fromStatuses`: the thirteen call sites are internal/purchase/approvals.go:332, internal/purchase/scheduled_fire.go:81, internal/purchase/reaper.go:175, internal/purchase/manager.go:179/452/498, internal/api/handler_history.go:247 and internal/api/handler_purchases.go:271/295/320/445/827/1448. Both writers of canonical-only `canceled_by` land the row in `canceled` (internal/config/store_postgres.go:1158-1166 and 1214-1222), which is terminal, and `SetCancelledBy` writes both columns (1128-1132), so no record with a lost actor is ever returned. + +### A05-016 `fanOutPerAccount` omits the post-`Wait` ctx check its sibling fan-out has +- category: concurrency +- severity: low +- location: internal/scheduler/scheduler.go:600 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: Every goroutine returns nil to isolate per-account failures, so `g.Wait()` can never surface `context.Canceled`. `collectAllProviders` recognises this and adds an explicit `ctx.Err()` check after its own Wait (scheduler.go:339); `fanOutPerAccount` does not. Today the parent's check catches the cancellation before `persistCollection` runs, so nothing is evicted, but the per-provider collect functions in between (`collectAWSRecommendations` and its Azure/GCP twins) each return `(recs, outcome.SucceededAccountIDs, nil)` — a nil error alongside a roster that authorizes stale-row eviction — for a sweep that was cut short. The safety of the eviction rests entirely on a check two frames up. +- evidence: + ```go + if waitErr := g.Wait(); waitErr != nil { + // Goroutines return nil to isolate per-account failures; non-nil is unexpected. + logging.Warnf("fanOutPerAccount: errgroup.Wait returned unexpected error: %v", waitErr) + } + return all, outcome + ``` +- suggested fix: after Wait, if `ctx.Err() != nil`, clear `outcome.SucceededAccountIDs` (move those IDs to `IncompleteAccountIDs`) so a canceled sweep can never authorize eviction regardless of what the caller does. +- verdict: REJECTED — a guard exists on the only path out. fanOutPerAccount is reached exclusively through collectAWSRecommendations (scheduler.go:495), collectAzure (:802) and collectGCP (:875), all called from collectProviderRecommendations (:419), whose sole caller is the collectAllProviders goroutine at scheduler.go:319; that function does check `ctx.Err()` after Wait and returns early (:338-340), before CollectRecommendations reaches persistCollection (:211). No canceled sweep can authorize eviction, and the finding concedes this itself — what remains is a defense-in-depth preference with no failure scenario. + +### A06-002 scheduledAuthMiddleware falls open to a pass-through when the validator is nil +- category: security +- severity: high +- location: internal/server/http.go:207 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: any `Application` not built by `NewApplicationFromDeps` has `scheduledAuth == nil`, and every `/api/scheduled/*` route (collect_recommendations, process_scheduled_purchases, ladder_run, ri_exchange_reshape — all money-moving) is then served with no authentication at all. This is the exact pattern the project memory records as fixed on PR #620 (`feedback_fail_closed_middleware`), so it is a regression, not a known gap. It is currently load-bearing for tests: internal/server/http_test.go:304 builds `app := &Application{}` and asserts HTTP 200 for `POST /api/scheduled/collect_recommendations` with no `Authorization` header, so the suite passes *because* of the bypass and would catch nothing if a future refactor exposed it in production wiring. +- evidence: + ```go + func (app *Application) scheduledAuthMiddleware(next http.Handler) http.Handler { + if app.scheduledAuth == nil { + return next + } + return app.scheduledAuth.Middleware(next) + } + ``` +- suggested fix: return a handler that writes 403 when `app.scheduledAuth == nil`, and change the tests to inject `scheduledauth.New(Config{Mode: ModeDisabled})` explicitly. +- verdict: REJECTED — the nil state is unreachable in production wiring: the only non-test `Application` literal is internal/server/app.go:502 inside `NewApplicationFromDeps`, which returns early unless `initScheduledAuth` yields a non-nil validator (internal/server/app.go:396-402 and 361-381), `LoadConfig` errors when `SCHEDULED_TASK_AUTH_MODE` is unset, and `scheduledauth.New` never returns `(nil, nil)` (internal/server/scheduledauth/validator.go:109-133); the fail-open default remains a hardening gap but has no reachable failure scenario at this commit. + +### A06-011 SMTP TLS guard protects credentials only, not the token-bearing body it claims to protect +- category: security +- severity: low +- location: internal/email/smtp_sender.go:327 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: the comment above `dispatchSMTP` states the guard exists because a non-TLS connection "exposes credentials ... and token-bearing message bodies in cleartext", but the condition is `auth != nil && !s.useTLS && !s.allowInsecure`. With `Username`/`Password` empty (an open relay or a local MTA) `auth` is nil, the guard does not fire, and `smtp.SendMail` sends purchase-approval and revocation tokens over plaintext port 25. Reachability today is limited: both factory paths hardcode `Port: 587, UseTLS: true`, so only a direct `NewSMTPSender` caller (tests, a future wiring) can hit it — the finding is that the guard's axes do not match its stated purpose. +- evidence: + ```go + if auth != nil && !s.useTLS && !s.allowInsecure { + return fmt.Errorf("SMTP auth over non-TLS connection is refused: ...") + } + ``` +- suggested fix: drop `auth != nil` from the condition so any non-TLS send is refused unless `AllowInsecure` is set. +- verdict: REJECTED — the mismatch between the comment and the condition is real (internal/email/smtp_sender.go:311-330), but no reachable caller can produce a non-TLS send: all four constructors hardcode `Port: 587, UseTLS: true` with non-empty credentials (internal/email/factory.go:140-148, 196-203, 221-230, 239-248), and `NewSMTPSender` additionally forces `UseTLS = true` whenever `Port == 587` (smtp_sender.go:82-84), so there is no in-tree path with `auth == nil` and `useTLS == false`. + +### A06-021 Org discovery silently drops malformed accounts and imports suspended ones as enabled +- category: correctness +- severity: low +- location: internal/accounts/org_discovery.go:48 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: `ListAccounts` returns every member account including ones with `Status: SUSPENDED` or `PENDING_CLOSURE`; each is turned into `CloudAccount{Enabled: true, AWSAuthMode: "role_arn"}`, so CUDly will keep trying to assume a role into a closed account on every collection cycle and surface the failures as account errors. Separately, an entry with a nil `Id` or `Name` is skipped with a bare `continue`, which contradicts the type's own doc ("Discovery is all-or-nothing ... There is no partial-success path") — the caller receives a short list with no indication anything was dropped. `"aws"` and `"role_arn"` are bare literals where the repo has typed provider constants. +- evidence: + ```go + for _, a := range page.Accounts { + if a.Id == nil || a.Name == nil { + continue + } + accounts = append(accounts, config.CloudAccount{ + Provider: "aws", ExternalID: *a.Id, Name: *a.Name, + Enabled: true, AWSAuthMode: "role_arn", + }) + ``` +- suggested fix: skip accounts whose `Status` is not `ACTIVE`, and return an error (or a reported count) instead of silently discarding entries with missing fields. +- verdict: REJECTED — the stated consequence cannot occur: the sole consumer of the discovery result overwrites both fields before persisting, setting `member.Enabled = false` and `member.AWSAuthMode = ""` on every row (internal/api/handler_accounts.go:1591-1595, reached via `runOrgDiscovery` at handler_accounts.go:1540-1549), so a SUSPENDED account is written as a disabled row awaiting operator review and no collection cycle ever assumes a role into it. The residual sub-point survives as a nit: the nil-`Id`/`Name` `continue` (internal/accounts/org_discovery.go:49-51) does silently shrink a list the type documents as all-or-nothing (org_discovery.go:14-16). + +### A06-023 Exchange notification branch re-derives the manual mode from a bare literal +- category: correctness +- severity: low +- location: internal/server/handler_ri_exchange.go:165 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: `result.Mode` is `cfg.RIExchangeMode` copied straight from the DB (`pkg/exchange/auto.go:157`), and `pkg/exchange` owns the manual/auto decision at auto.go:248. This handler re-derives it from the literal `"manual"` even though `exchange.ExchangeModeManual` exists (auto.go:607). If the two ever disagree — a stored value of `"Manual"`, or a future rename — the engine still produces `Pending` outcomes with live approval tokens while this branch falls through to the `Completed`/`Failed` test, which is false for a pending-only run, so no email is sent at all and the approvals expire unnoticed. +- evidence: + ```go + if result.Mode == "manual" && len(result.Pending) > 0 { + data.RecipientEmail = notifyEmail + err = app.Email.SendRIExchangePendingApproval(ctx, data) + } else if len(result.Completed)+len(result.Failed) > 0 { + ``` +- suggested fix: branch on `len(result.Pending) > 0` (the actual precondition) and use `string(exchange.ExchangeModeManual)` if the mode check is kept. +- verdict: REJECTED — the two sides cannot disagree, because the engine's own gate uses the identical bare literal: `if params.Config.Mode == "manual"` (pkg/exchange/auto.go:248) is the only place `result.Pending` is ever appended to (auto.go:253), and `result.Mode` is `params.Config.Mode` copied verbatim (auto.go:156). So `len(result.Pending) > 0` implies `result.Mode == "manual"`; a stored `"Manual"` would make the engine run in auto mode and produce no pendings at all, not the described silent-drop. The unused `ExchangeModeManual` constant (auto.go:607) makes this a style inconsistency with no failure scenario. + +### A07-020 Unmapped service types are sent to Cost Explorer as a raw slug +- category: silent-fallback +- severity: medium +- location: providers/aws/recommendations/converters.go:27 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: `getServiceStringForCostExplorer` returns `string(service)` for anything not in its switch, and `client.go:150` puts that value straight into `GetReservationPurchaseRecommendationInput.Service`. A service type the switch does not cover reaches CE as, for example, `"savingsplans-compute"` instead of a real SERVICE dimension value. CE matches nothing and returns an empty recommendation set with no error, so the run reports "no savings available" for a service that was never actually queried. +- evidence: + ```go + case common.ServiceMemoryDB: + return "Amazon MemoryDB Service" + default: + return string(service) + } + ``` +- suggested fix: add an error-returning variant used by the request builder (the file already establishes this pattern with `convertPaymentOptionE` / `convertTermInYearsE`) so an unmapped service fails loud instead of querying a nonexistent dimension value. +- verdict: REJECTED — client.go:145 diverts every Savings Plans slug through `common.IsSavingsPlan` (pkg/common/types.go:148-157) before reaching the input built at :150, the cited value `"savingsplans-compute"` is not a real constant (the slug is `"savings-plans-compute"`, types.go:115), and every remaining entry in `GetSupportedServices` (provider.go:426-446) has an arm in the switch, so the `default` at converters.go:27 is unreachable from either call site (the other, usage_history.go:151, additionally skips SP recs because they carry no `ResourceType`). + +### A07-024 Marketplace listing accepts an empty idempotency token +- category: money-path +- severity: medium +- location: providers/aws/services/ec2/client.go:1043 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: `CreateMarketplaceListing` validates `PriceSchedule`, `ReservedInstancesID` and `InstanceCount` but not `ClientToken`, then sends `aws.String("")`. `ClientToken` is the only thing preventing a retried listing request from creating a second listing for the same RI. A caller that forgets to populate it gets no error, and a retry after a timeout lists the RI twice. +- evidence: + ```go + input := &ec2.CreateReservedInstancesListingInput{ + ReservedInstancesId: aws.String(req.ReservedInstancesID), + ClientToken: aws.String(req.ClientToken), + InstanceCount: aws.Int32(req.InstanceCount), + PriceSchedules: awsSchedule, + } + ``` +- suggested fix: reject an empty `ClientToken` alongside the other three boundary checks, matching the non-empty idempotency-source rule the EC2 and RDS purchase paths already follow. +- verdict: REJECTED — the missing check is real (providers/aws/services/ec2/client.go:1021-1029 validates the other three fields only), but the sole production caller always supplies `uuid.New().String()` (internal/api/handler_marketplace.go:250) and concurrent creates are serialized by `ClaimMarketplaceListingSlot` (:240-246), whose comment records that AWS gets a fresh token per call anyway, so the described double-listing is unreachable; this is a latent boundary-validation gap, not a live money-path defect. + +### A07-025 GetServiceClient discards the plan-type lookup error and silently falls into umbrella mode +- category: silent-fallback +- severity: medium +- location: providers/aws/provider.go:476 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: `PlanTypeForServiceType` returns `("", false)` for anything outside its four cases and the `ok` value is thrown away. If a fifth SP slug is added to this `case` list without a matching arm in `PlanTypeForServiceType`, `pt` is `""`, which the savingsplans constructor treats as umbrella mode: `GetExistingCommitments` stops partitioning and returns every plan type, and `resolveSPPlanType` stops rejecting a plan-type mismatch (client.go:271). The scope guard that exists to stop the wrong product being bought is disabled by a value nobody checked. +- evidence: + ```go + case common.ServiceSavingsPlansCompute, + common.ServiceSavingsPlansEC2Instance, + common.ServiceSavingsPlansSageMaker, + common.ServiceSavingsPlansDatabase: + pt, _ := savingsplans.PlanTypeForServiceType(service) + return NewSavingsPlansClient(regionalCfg, pt), nil + ``` +- suggested fix: check the boolean and return an error when the slug has no mapped plan type, so the two lists cannot drift into an unintended umbrella client. +- verdict: REJECTED — all four slugs in the case list at providers/aws/provider.go:471-474 have matching arms in `PlanTypeForServiceType` (providers/aws/services/savingsplans/client.go:84-96), so `pt` is never empty at this commit and the umbrella-mode consequences at client.go:120-123 and :271 are unreachable; the discarded `ok` is a genuine drift hazard but describes a future edit, not a present failure. +- severity-adjusted: low — no reachable failure scenario, so this is hardening rather than a live silent fallback. + +### A07-030 GetDefaultRegion falls back to a hardcoded us-east-1 +- category: silent-fallback +- severity: low +- location: providers/aws/provider.go:419 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: when no region is set on the provider, on the config, or in the SDK chain, the function returns `"us-east-1"` rather than reporting that no region could be resolved. `GetServiceClient` copies whatever region it is handed into the regional config, so a misconfigured environment silently targets us-east-1 for offering lookups and purchases instead of failing. +- evidence: + ```go + if p.IsConfigured() && p.cfg.Region != "" { + return p.cfg.Region + } + return "us-east-1" + ``` +- suggested fix: add an error-returning variant used by the purchase and offering paths, keeping the string default (if at all) only for display contexts. +- verdict: REJECTED — the hardcoded return at providers/aws/provider.go:419 is real (as is the dead `IsConfigured()` branch at :417, which repeats the check at :412), but a repo-wide grep finds no production caller of `GetDefaultRegion` anywhere: only the interface declaration at pkg/provider/interface.go:27 and test mocks, so the claimed propagation into `GetServiceClient`'s regional config, offering lookups and purchases does not exist. + +### A08-011 An empty term silently buys a one-year Savings Plan while an empty payment option correctly fails closed +- category: money-path +- severity: high +- location: providers/azure/services/savingsplans/client.go:490 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: `PurchaseCommitment` calls `toAzureTerm(rec.Term)`, which maps `""` to `TermP1Y`. A recommendation row persisted before the term column was populated reaches the purchase path with `Term == ""` and buys a real one-year commitment nobody asked for. The sibling parser in the same package family refuses the equivalent input explicitly and explains why: `BillingPlanForPaymentOption("")` returns an error rather than defaulting because "Azure's own default is upfront and would charge the whole commitment immediately" (reservations/purchase.go:115). `reservations.ParseTermYears` has the same `""` → 1 mapping (purchase.go:70). +- evidence: + ```go + func toAzureTerm(term string) (armbillingbenefits.Term, error) { + switch term { + case "1yr", "1", "P1Y", "": + return armbillingbenefits.TermP1Y, nil + ``` +- suggested fix: remove `""` from the one-year case in both `toAzureTerm` and `reservations.ParseTermYears` and return the same style of explicit error `BillingPlanForPaymentOption` returns for an empty payment option. +- verdict: REJECTED — no caller can deliver `Term == ""` to the purchase path: internal/purchase/execution.go:1041 builds it as `fmt.Sprintf("%dyr", rec.Term)` from an int column (a 0 would yield "0yr", which `toAzureTerm` already rejects), and every Azure-sourced recommendation goes through `normaliseTerm` (internal/recommendations/converter.go:341), which maps nil/empty to "1yr" before the rec exists. The `""` case at savingsplans/client.go:490 and reservations/purchase.go:70 is an unreachable allowance, not a live silent purchase. + +### A08-025 The VM reservation purchase body uses a raw "Shared" literal for a typed SDK enum +- category: hygiene +- severity: low +- location: providers/azure/services/compute/client.go:448 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: `appliedScopeType` is written as the string `"Shared"` while `armreservations.AppliedScopeTypeShared` exists and is used correctly elsewhere in the same package (exchange_operations.go:448). A casing or spelling change in a future API version breaks the purchase body at runtime with no compile error, and the scope is not derived from the recommendation's own `ExtractedFields.Scope`, which the extractor already populates (internal/recommendations/converter.go:51). +- evidence: + ```go + "appliedScopeType": "Shared", + "renew": false, + ``` +- suggested fix: use `string(armreservations.AppliedScopeTypeShared)`. +- verdict: REJECTED — the literal at compute/client.go:448 is byte-identical to `armreservations.AppliedScopeTypeShared` (armreservations@v1.1.0/constants.go:21), so no failure scenario exists today, and the "not derived from ExtractedFields.Scope" half is refuted by the recommendation filter the client actually sends, `"properties/scope eq 'Shared'"` (compute/client.go:199), which is exactly the rationale recorded at providers/azure/internal/recommendations/converter.go:44-50. + +### A09-017 An empty persisted Details payload yields a zero-valued typed pointer that the provider fills with Linux/default +- category: money-path +- severity: high +- location: pkg/common/service_details_codec.go:98 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: for a `purchase_executions` row whose `Details` JSONB is NULL or absent — a legacy row, or any writer that failed to populate it — `DecodeServiceDetailsFor("ec2", nil)` returns `&ComputeDetails{}` with `Platform`, `Tenancy` and `Scope` all empty. The comment states the downstream `buildOfferingFilters` then substitutes `Platform=Linux/UNIX, Tenancy=default, Scope=Region`. A Windows EC2 recommendation re-driven from such a row buys a Linux Reserved Instance that covers none of the Windows demand. That is the exact mis-purchase class issue #453 was opened for, still reachable by design on the legacy path. +- evidence: + ```go + if len(raw) == 0 || bytes.Equal(raw, jsonNullBytes) { + // Legacy row or genuinely absent payload — hand back a zero- + // valued typed pointer so the service client's type-assertion + // succeeds. buildOfferingFilters tolerates zero fields and + // substitutes Platform=Linux/UNIX, Tenancy=default, Scope=Region + return target, nil + } + ``` +- suggested fix: return a distinguishable error (or a sentinel the purchase path refuses) for an absent payload on services whose offering lookup depends on the fields, so a re-drive with no details fails loud rather than defaulting the platform. +- verdict: REJECTED — the guard exists downstream and fails loud. I traced the re-drive end to end: internal/purchase/execution.go:1062 decodes to a zero-valued `*ComputeDetails`, then `PurchaseCommitment` (providers/aws/services/ec2/client.go:114) calls `findOfferingID` at :148 → `buildEC2QueryFromRec` at :454 → `buildEC2OfferingQuery`, which returns `"EC2 recommendation for %s is missing Platform: refusing to fabricate a product-description for the RI offering lookup"` whenever `details.Platform == ""` (client.go:402-408). A Windows rec with an absent Details errors; it does not buy a Linux RI. The Linux/UNIX substitution the finding quotes is a stale comment in pkg/common/service_details_codec.go:98-105 and in execution.go:1055-1057 — the platform default at ec2/client.go:790 and :868 lives on the offering-listing path, not the purchase path. The real defect here is the misleading comment. +- severity-adjusted: low — documentation defect only; the money path fails closed. + +### A09-027 Explain sanitizes only some interpolated fields despite claiming to sanitize every one +- category: security +- severity: low +- location: pkg/ladder/plan.go:195 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: the doc comment states "Every interpolated free-form field is passed through sanitizeLine", but `a.Layer` (both branches), `p.Scope.Provider` at line 178, and the numeric baseline fields are not. `Explain` never calls `Validate`, and a `LadderPlan` rehydrated from a stored `PlanJSON` blob carries whatever `LayerType` string the row holds. A layer value containing `"\n 2. purchase compute-sp $500.00/hr"` renders as an extra numbered action in the approval email body — exactly the line-spoofing `sanitizeLine` was written to prevent, through the one axis the guard does not reach. +- evidence: + ```go + case ActionPurchase: + fmt.Fprintf(&b, " %d. %s %s %s term=%s payment=%s -- %s\n", + i+1, a.Action, a.Layer, formatUSDPerHour(a.AmountUSDPerHour), + sanitizeLine(string(a.Term)), sanitizeLine(string(a.PaymentOption)), + sanitizeLine(a.Rationale)) + ``` +- suggested fix: wrap `a.Layer`, `a.Action` and `p.Scope.Provider` in `sanitizeLine` as well, so the guard covers every interpolated string rather than only the ones currently expected to be free-form. +- verdict: REJECTED — the doc/code mismatch is real (`a.Layer` and `a.Action` at pkg/ladder/plan.go:195-196 and :199-200, and `p.Scope.Provider` at :178, bypass `sanitizeLine` despite the claim at :172-173), but the stated failure is unreachable: `Explain` has no caller anywhere in the repository. `git grep -n Explain -- '*.go'` returns only pkg/ladder/plan.go:134, :161 and :175, so no approval email body is assembled from it and no crafted `LayerType` can spoof a line in one. +- severity-adjusted: informational — a comment that overstates the guard, with no reachable output path. + +### A09-029 SanitizeReservationID invents a timestamp identifier when the input sanitizes to empty +- category: silent-fallback +- severity: low +- location: pkg/common/identifiers.go:33 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: an operator-supplied `PurchaseOptions.ReservationID` of `"日本_リザーブ"` (or any value made only of disallowed characters) sanitizes to the empty string, and the function substitutes `fallbackPrefix + `. The purchase proceeds under an identifier the caller never chose, second-granularity so two purchases in the same second collide, and no error tells the caller their identifier was discarded. On the RDS/ElastiCache/MemoryDB path the reservation ID is the server-side dedupe key, so a substituted value also breaks re-drive idempotency. +- evidence: + ```go + s = strings.Trim(s, "-") + if s == "" { + s = fallbackPrefix + strconv.FormatInt(time.Now().Unix(), 10) + } + return s + ``` +- suggested fix: return an error for a caller-supplied identifier that sanitizes to empty, keeping the generated fallback only for the internal builder paths that have no caller-supplied value to preserve. +- verdict: REJECTED — the timestamp substitution is in the code as quoted (pkg/common/identifiers.go:32-35), but no caller can reach it with a value that sanitizes to empty. The only provider call site is providers/aws/services/rds/client.go:207, reached only when `opts.ReservationID != ""`, and the only production writer of that field is cmd/multi_service_helpers.go:246, which passes `generatePurchaseID` — a machine-composed ASCII string always containing the "ri"/"dryrun" prefix, an RFC-style timestamp and a uuid8 suffix (cmd/main.go:313-330). There is no operator-controlled ReservationID input. The "RDS/ElastiCache/MemoryDB path" claim is also wrong: `git grep -n SanitizeReservationID` shows RDS is the only provider caller. +- severity-adjusted: informational — unreachable from every current caller. + +### A11-003 A stored max_amount of 0 renders as blank and re-saves as no spending cap +- category: money-path +- severity: high +- location: frontend/src/groups/groupModals.ts:269 @ 3c0f8ac94048a2c36fce5ccddee54e6c4849a5cd +- failure scenario: the Max Amount input is populated with `String(permission?.constraints?.max_amount || '')`. For a permission constrained to `max_amount: 0` (spend nothing) the `||` treats 0 as absent and the box renders empty. `collectPermissions` then gates on `if (maxAmount)` (groupModals.ts:450), so the empty box contributes nothing and the saved permission carries no `max_amount` at all. On the backend an absent cap is unrestricted, so a cosmetic rename of the group converts a $0 ceiling into unlimited spend. This is the exact widening the file's own comments guard against for the list constraints, but `unrepresentableDimensions` only walks `LIST_CONSTRAINT_DIMENSIONS` (groupModals.ts:319) and never inspects `max_amount`. No test covers `max_amount: 0`; groups.test.ts only exercises 5000, 10000, Infinity and a negative. +- evidence: + ```typescript + // groups/groupModals.ts:268-270 + + + + +
+

Constraints (Optional)

+
+ +
+
+ + +
+
+ + +
+
+ `; + + permissionsList.appendChild(permDiv); + + // Add event listener for remove button + const removeBtn = permDiv.querySelector('.remove-permission-btn'); + if (removeBtn) { + removeBtn.addEventListener('click', () => { + permDiv.remove(); + }); + } +} + +// The one tokenizer for every comma-separated constraint input. Both the save +// path and the render-time safety check below go through it, deliberately: if +// this drops or alters a value, the check must catch it, and sharing the +// function is what stops the two from drifting apart. +function parseConstraintList(raw: string): string[] { + return raw.split(',').map(s => s.trim()).filter(s => s); +} + +// Names the constraint lists in `constraints` that this form cannot carry +// back unchanged (issue #1629, raised again by CodeRabbit on PR #1875). +// +// The form encodes a list as comma-separated text, and that encoding is not +// injective. A stored [""] renders blank and re-parses as ABSENT; a stored +// [","] re-parses as an empty list; a stored [" acct A "] comes back trimmed, +// which is a different fence, since enforcement compares values exactly. In +// every case the form would re-submit a materially different restriction than +// the one it loaded, and for a constraint list "different" means WIDER: an +// empty list is "no restriction on this dimension" at enforcement +// (matchStringListConstraints). The refusal in saveGroup is the loud +// alternative to that silent widening. +// +// The test is a round trip through parseConstraintList rather than a +// hand-written blank check, for two reasons: it is the same predicate the +// save path uses, so the two cannot disagree; and a per-entry "is it blank" +// test would MISS [","], whose entry is not blank yet still vanishes, because +// the split runs before the filter. +// +// An absent or empty list is NOT flagged. Both mean "no restriction on this +// dimension", both are perfectly normal, and refusing them would make +// ordinary unconstrained groups uneditable. +// The constraint dimensions the form renders as comma-separated text, each in +// an input classed `perm-`. One list, so the render-time check and +// the typed-input check below cannot cover different sets. +const LIST_CONSTRAINT_DIMENSIONS = ['accounts', 'providers', 'services', 'regions'] as const; + +function unrepresentableDimensions(constraints: Permission['constraints']): string[] { + if (!constraints) return []; + + const unsafe: string[] = []; + for (const dimension of LIST_CONSTRAINT_DIMENSIONS) { + const values = constraints[dimension]; + if (!values || values.length === 0) continue; + const reparsed = parseConstraintList(values.join(', ')); + if (reparsed.length !== values.length || reparsed.some((value, i) => value !== values[i])) { + unsafe.push(dimension); + } + } + return unsafe; +} + +// Every reason this form must refuse to save, as one message per problem, +// naming the permission by index and by action:resource and naming the +// constraint list. Mirrors the specificity of the backend's own refusal in +// validateConstraintEntries. +// +// Two directions, one rule and one error path, because they are the same +// defect: the form would send a constraint list that says something different +// from what the operator is looking at, and for a constraint list "different" +// is WIDER, since an empty list is "no restriction on this dimension" at +// enforcement. +// +// 1. STORED: the loaded value cannot survive the text encoding (flagged when +// the row rendered, see unrepresentableDimensions). +// 2. TYPED: the box holds something, but parseConstraintList reduces it to +// nothing -- "," or " , " being the reachable case, since it takes only a +// stray comma. Without this the save would send an empty list and silently +// remove the fence. +// +// Both use parseConstraintList itself as the predicate, never a second +// implementation of "parses to nothing" that could drift from it. +// +// A BLANK box is not an error: it means "no restriction on this dimension", +// which is ordinary. Blank is measured after trimming, so a box holding only +// spaces counts as blank -- it is indistinguishable from empty on screen, and +// erroring on a field that looks empty would be unfixable from the UI. +function unrepresentablePermissionErrors(): string[] { + const permissionsList = document.getElementById('permissions-list'); + if (!permissionsList) return []; + + const errors: string[] = []; + permissionsList.querySelectorAll('.permission-item').forEach((item, index) => { + const action = (item.querySelector('.perm-action') as HTMLSelectElement | null)?.value || ''; + const resource = (item.querySelector('.perm-resource') as HTMLSelectElement | null)?.value || ''; + const label = `permission ${index} (${action}:${resource})`; + + const stored = (item.getAttribute('data-unrepresentable') || '').split(', ').filter(d => d); + if (stored.length > 0) { + errors.push(`${label} has a stored "${stored.join(', ')}" constraint value this form cannot represent`); + } + + for (const dimension of LIST_CONSTRAINT_DIMENSIONS) { + // Already refused for this row on the stored value; do not say it twice. + if (stored.includes(dimension)) continue; + const raw = (item.querySelector(`.perm-${dimension}`) as HTMLInputElement | null)?.value ?? ''; + if (raw.trim() !== '' && parseConstraintList(raw).length === 0) { + errors.push(`${label} has a "${dimension}" value with no usable entries`); + } + } + }); + return errors; +} + +/** + * Render permissions list + */ +function renderPermissions(permissions: Permission[]): void { + const permissionsList = document.getElementById('permissions-list'); + if (!permissionsList) return; + + permissionsList.innerHTML = ''; + + if (permissions.length === 0) { + addPermission(); + } else { + permissions.forEach(perm => addPermission(perm)); + } +} + +/** + * Collect permissions from form + */ +function collectPermissions(): Permission[] { + const permissionsList = document.getElementById('permissions-list'); + if (!permissionsList) return []; + + const permissions: Permission[] = []; + const items = permissionsList.querySelectorAll('.permission-item'); + + items.forEach(item => { + const action = (item.querySelector('.perm-action') as HTMLSelectElement)?.value; + const resource = (item.querySelector('.perm-resource') as HTMLSelectElement)?.value; + + if (!action || !resource) return; + + const permission: Permission = { action, resource }; + + // Collect constraints. Dropping `accounts` here WIDENS the permission + // rather than merely losing data: an empty AccountIDs list means "no + // restriction on this dimension" at enforcement + // (matchStringListConstraints), so a permission scoped to one cloud + // account came back out of a cosmetic rename scoped to all of them + // (issue #1629). + const accounts = (item.querySelector('.perm-accounts') as HTMLInputElement)?.value; + const providers = (item.querySelector('.perm-providers') as HTMLInputElement)?.value; + const services = (item.querySelector('.perm-services') as HTMLInputElement)?.value; + const regions = (item.querySelector('.perm-regions') as HTMLInputElement)?.value; + const maxAmount = (item.querySelector('.perm-max-amount') as HTMLInputElement)?.value; + + if (accounts || providers || services || regions || maxAmount) { + permission.constraints = {}; + // Attach a dimension only when it yields at least one entry, so a box + // that LOOKS empty produces the same payload as one that IS empty. An + // input holding only whitespace would otherwise send [], which means + // the same thing at enforcement but makes the request depend on + // invisible characters. A box that parses to nothing while holding + // something visible never reaches here: saveGroup already refused it. + const accountsList = parseConstraintList(accounts || ''); + const providersList = parseConstraintList(providers || ''); + const servicesList = parseConstraintList(services || ''); + const regionsList = parseConstraintList(regions || ''); + if (accountsList.length > 0) permission.constraints.accounts = accountsList; + if (providersList.length > 0) permission.constraints.providers = providersList; + if (servicesList.length > 0) permission.constraints.services = servicesList; + if (regionsList.length > 0) permission.constraints.regions = regionsList; + if (maxAmount) { + const parsed = parseFloat(maxAmount); + // Reject non-finite or negative values (feedback_nullable_not_zero). + // A malformed entry is silently skipped so the rest of the constraints + // still reach the API; the input's type="number" min="0" already + // prevents browser submission of non-numeric values, but the JS + // path must guard too. + if (Number.isFinite(parsed) && parsed >= 0) { + permission.constraints.max_amount = parsed; + } + } + // Nothing survived: the permission carries no restriction, so send no + // constraints object rather than an empty one. + if (Object.keys(permission.constraints).length === 0) { + delete permission.constraints; + } + } + + permissions.push(permission); + }); + + return permissions; +} + +// --------------------------------------------------------------------------- +// Duplicate group modal +// --------------------------------------------------------------------------- + +const DUP_PROVIDER_PILLS: Array<{ value: string; label: string }> = [ + { value: 'all', label: 'All' }, + { value: 'aws', label: 'AWS' }, + { value: 'azure', label: 'Azure' }, + { value: 'gcp', label: 'GCP' }, +]; + +/** + * Render a read-only badge list of source permissions as "action:resource" + * entries. Uses textContent + createElement to avoid innerHTML with user + * strings. + */ +function renderSourcePermissionBadges(container: HTMLElement, permissions: Permission[]): void { + container.textContent = ''; + if (permissions.length === 0) { + const empty = document.createElement('span'); + empty.className = 'dup-empty'; + empty.textContent = 'No permissions on source group'; + container.appendChild(empty); + return; + } + for (const perm of permissions) { + const badge = document.createElement('span'); + badge.className = 'permission-badge'; + badge.textContent = `${perm.action}:${perm.resource}`; + container.appendChild(badge); + } +} + +/** + * Render the provider filter pills (All / AWS / Azure / GCP). Each pill + * filters the visible account checkboxes by data-provider; selection is + * UI-only and never stored in the created group. + */ +function renderDuplicateProviderPills(container: HTMLElement, accountsList: HTMLElement): void { + container.textContent = ''; + const buttons: HTMLButtonElement[] = []; + + for (const pill of DUP_PROVIDER_PILLS) { + const btn = document.createElement('button'); + btn.type = 'button'; + btn.className = 'btn btn-small target-cloud-pill'; + btn.textContent = pill.label; + btn.setAttribute('data-provider', pill.value); + btn.setAttribute('aria-pressed', 'false'); + btn.addEventListener('click', () => { + for (const b of buttons) { + const selected = b === btn; + b.setAttribute('aria-pressed', selected ? 'true' : 'false'); + b.classList.toggle('selected', selected); + } + applyDuplicateProviderFilter(accountsList, pill.value); + }); + buttons.push(btn); + container.appendChild(btn); + } + + // Default selection: "All" (first option). + const first = buttons[0]; + if (first) { + first.setAttribute('aria-pressed', 'true'); + first.classList.add('selected'); + } + applyDuplicateProviderFilter(accountsList, 'all'); +} + +/** + * Hide/show account checkbox rows by data-provider. "all" shows everything. + */ +function applyDuplicateProviderFilter(accountsList: HTMLElement, provider: string): void { + const labels = accountsList.querySelectorAll('label[data-provider]'); + labels.forEach(label => { + const rowProvider = (label as HTMLElement).getAttribute('data-provider') || ''; + const visible = provider === 'all' || provider === rowProvider; + label.classList.toggle('dup-account-hidden', !visible); + }); +} + +/** + * Render the account checkbox list. Each row is a label + checkbox whose + * value is the account name (names are what the backend matcher accepts + * for human-readable scoping). + */ +function renderDuplicateAccountsList(container: HTMLElement, accounts: api.CloudAccount[]): void { + container.textContent = ''; + if (accounts.length === 0) { + const empty = document.createElement('p'); + empty.className = 'dup-empty'; + empty.textContent = 'No cloud accounts configured yet. Duplicating without scope clones the full source group — add accounts first if you want to restrict.'; + container.appendChild(empty); + return; + } + + for (const acct of accounts) { + const label = document.createElement('label'); + label.setAttribute('data-provider', acct.provider); + + const cb = document.createElement('input'); + cb.type = 'checkbox'; + cb.className = 'dup-account-checkbox'; + cb.value = acct.name; + cb.setAttribute('data-provider', acct.provider); + + const text = document.createElement('span'); + text.textContent = `${acct.name} (${acct.external_id}) [${acct.provider}]`; + + label.appendChild(cb); + label.appendChild(text); + container.appendChild(label); + } +} + +/** + * Open the Duplicate Group modal for the given source group. + * + * Looks up the source in cached `availableGroups` first, falling back to + * a fresh `api.getGroup` fetch. Prefills name (with " (copy)" suffix), + * description, and renders source permissions as read-only badges. + * Populates account checkboxes from `api.listAccounts()`. + */ +export async function openDuplicateGroupModal(groupId: string): Promise { + try { + let source = availableGroups.find(g => g.id === groupId) || null; + if (!source) { + source = await api.getGroup(groupId); + } + duplicateSourceGroup = source; + + const modal = document.getElementById('group-duplicate-modal'); + if (!modal) return; + + const nameInput = document.getElementById('dup-group-name') as HTMLInputElement | null; + const descInput = document.getElementById('dup-group-description') as HTMLTextAreaElement | null; + const permsContainer = document.getElementById('dup-source-permissions'); + const providerFilter = document.getElementById('dup-provider-filter'); + const accountsList = document.getElementById('dup-accounts-list'); + + if (nameInput) nameInput.value = `${source.name} (copy)`; + if (descInput) descInput.value = source.description || ''; + if (permsContainer) renderSourcePermissionBadges(permsContainer, source.permissions); + + // Populate accounts, then wire provider pills to filter them. + let accounts: api.CloudAccount[] = []; + try { + accounts = await api.listAccounts(); + } catch (err) { + console.error('Failed to list accounts for duplicate modal:', err); + accounts = []; + } + if (accountsList) renderDuplicateAccountsList(accountsList, accounts); + if (providerFilter && accountsList) renderDuplicateProviderPills(providerFilter, accountsList); + + openModal(modal); + } catch (error) { + console.error('Failed to open duplicate group modal:', error); + showError('Failed to load group details'); + } +} + +/** + * Close the Duplicate Group modal and clear its module-level state. + */ +export function closeDuplicateGroupModal(): void { + const modal = document.getElementById('group-duplicate-modal'); + if (modal) closeModal(modal); + duplicateSourceGroup = null; +} + +/** + * Save the duplicate group — posts to the existing POST /api/groups + * endpoint. If account checkboxes are ticked, their names become the new + * group's `allowed_accounts`; otherwise the source's `allowed_accounts` + * is inherited as-is. Permissions are copied verbatim from the source. + */ +export async function saveDuplicateGroup(e: Event): Promise { + e.preventDefault(); + + const source = duplicateSourceGroup; + if (!source) { + showError('No source group to duplicate'); + return; + } + + const nameInput = document.getElementById('dup-group-name') as HTMLInputElement | null; + const descInput = document.getElementById('dup-group-description') as HTMLTextAreaElement | null; + const accountsList = document.getElementById('dup-accounts-list'); + + const name = nameInput?.value.trim() || ''; + const description = descInput?.value || ''; + + const tickedNames: string[] = []; + if (accountsList) { + const checked = accountsList.querySelectorAll('.dup-account-checkbox:checked'); + checked.forEach(cb => { + const val = (cb as HTMLInputElement).value; + if (val) tickedNames.push(val); + }); + } + + const allowedAccounts = tickedNames.length > 0 + ? tickedNames + : (source.allowed_accounts || []); + + try { + await api.createGroup({ + name, + description, + permissions: source.permissions, + allowed_accounts: allowedAccounts, + }); + showSuccess('Group duplicated successfully'); + closeDuplicateGroupModal(); + await loadUsers(); + } catch (error) { + console.error('Failed to duplicate group:', error); + const err = error as Error; + showError(`Failed to duplicate group: ${err.message}`); + } +} diff --git a/frontend/src/groups/handlers.ts b/frontend/src/groups/handlers.ts new file mode 100644 index 000000000..32ac4ff4c --- /dev/null +++ b/frontend/src/groups/handlers.ts @@ -0,0 +1,21 @@ +/** + * Event handlers setup for group management + */ + +import { openCreateGroupModal, closeGroupModal, saveGroup, addPermission } from './groupModals'; + +/** + * Setup event handlers for group management + */ +export function setupGroupHandlers(): void { + // Make functions globally available for modal buttons that still need onclick handlers + (window as any).openCreateGroupModal = openCreateGroupModal; + (window as any).closeGroupModal = closeGroupModal; + (window as any).addPermission = () => addPermission(); + + // Setup form handlers + const groupForm = document.getElementById('group-form'); + if (groupForm) { + groupForm.addEventListener('submit', (e) => void saveGroup(e)); + } +} diff --git a/frontend/src/groups/index.ts b/frontend/src/groups/index.ts new file mode 100644 index 000000000..f426184d2 --- /dev/null +++ b/frontend/src/groups/index.ts @@ -0,0 +1,27 @@ +/** + * Groups module barrel export + */ + +// Re-export state +export { currentEditingGroup, setCurrentEditingGroup } from './state'; + +// Re-export group list rendering +export { renderGroups } from './groupList'; + +// Re-export group modals +export { + openCreateGroupModal, + openEditGroupModal, + closeGroupModal, + saveGroup, + addPermission, + openDuplicateGroupModal, + closeDuplicateGroupModal, + saveDuplicateGroup +} from './groupModals'; + +// Re-export group actions +export { deleteGroup } from './groupActions'; + +// Re-export handlers +export { setupGroupHandlers } from './handlers'; diff --git a/frontend/src/groups/state.ts b/frontend/src/groups/state.ts new file mode 100644 index 000000000..1cdec8a21 --- /dev/null +++ b/frontend/src/groups/state.ts @@ -0,0 +1,13 @@ +/** + * Group management state + */ + +import type { APIGroup } from '../api'; + +// State for modal management +export let currentEditingGroup: APIGroup | null = null; + +// Setters for state +export function setCurrentEditingGroup(group: APIGroup | null): void { + currentEditingGroup = group; +} diff --git a/frontend/src/history.ts b/frontend/src/history.ts new file mode 100644 index 000000000..9f42257c5 --- /dev/null +++ b/frontend/src/history.ts @@ -0,0 +1,1847 @@ +/** + * History module for CUDly + */ + +import * as api from './api'; +import * as state from './state'; +import { formatCurrency, formatDate, formatTerm, escapeHtml, escapeHtmlAttr, amortizedMonthly } from './utils'; +import type { HistoryResponse, HistorySummary, HistoryPurchase } from './types'; +import { switchTab } from './navigation'; +import { confirmDialog } from './confirmDialog'; +import { buildApprovalDetailsBody } from './approval-details'; +import { showToast } from './toast'; +import { getCurrentUser } from './state'; +import { canAccess } from './permissions'; +import { showSkeletonRows, teardownSkeleton } from './lib/skeleton'; +import { getAccountName } from './recommendations'; +import { applyColumnFilters } from './lib/column-filters'; +import { + openHistoryColumnPopover, + renderHistoryFilterButton, + closeOpenHistoryPopover, +} from './lib/history-filter-popover'; +import type { + PurchaseHistoryColumnId, + ApprovalQueueColumnId, +} from './state'; + +const VALID_PROVIDERS: api.Provider[] = ['aws', 'azure', 'gcp']; + +// AWS Marketplace fee/pricing constants (must stay in sync with handler_marketplace.go). +// awsMarketplaceFeePercent: transaction fee deducted from listing proceeds (published by AWS). +const AWS_MARKETPLACE_FEE_PERCENT = 12; +// awsMarketplaceNetFactor: fraction of list price the seller receives after the fee. +const AWS_MARKETPLACE_NET_FACTOR = 1 - AWS_MARKETPLACE_FEE_PERCENT / 100; +// awsMarketplaceBuyerDiscountFactor: applied to residual RI value to compute the default list price. +const AWS_MARKETPLACE_BUYER_DISCOUNT = 0.95; + +type StatusFilter = 'all' | 'pending' | 'completed' | 'failed' | 'expired' | 'cancelled'; + +// Cache of the last-rendered purchase list so the status-chip click handler +// can re-render without re-fetching. Cleared on each loadHistory / viewPlanHistory. +let lastPurchases: HistoryPurchase[] = []; +let activeStatusFilter: StatusFilter = 'all'; + +// _fourEyesMode mirrors GlobalConfig.require_different_approver (issue #1005). +// Gates the inline Approve button so a creator can't approve their own pending +// purchase when dual-control is on. +let _fourEyesMode = false; + +/** + * Refresh _fourEyesMode and the dual-control banner from GlobalConfig. + * + * Every path that renders the approval queue must call this first, or the + * queue renders as though dual control were off and offers the creator an + * Approve button the backend will reject. + * + * An unreadable config fails closed, to dual-control ON. The fetch never + * throws, so a config blip cannot block the render; it can only make the UI + * more restrictive than the backend, never less. The cost is a temporarily + * hidden Approve button during an outage, which self-corrects on the next + * successful load. + */ +async function refreshFourEyesMode(): Promise { + const cfgResponse = await api.getConfig().catch(() => null); + _fourEyesMode = cfgResponse?.global + ? cfgResponse.global.require_different_approver === true + : true; + const banner = document.getElementById('four-eyes-banner'); + if (banner) banner.classList.toggle('hidden', !_fourEyesMode); +} + +function normalizeStatus(p: HistoryPurchase): string { + // Absent status → legacy DB row → counts as completed for filtering. + return p.status || 'completed'; +} + +// isInFlight reports whether a status is an approved-but-not-yet-finalised +// execution (issue #621): the synchronous AWS purchase is mid-execution or got +// interrupted (Lambda timeout / crash). These rows MUST NOT render as the green +// "Completed" badge — doing so would tell the user a purchase finished when it +// may not have, tempting a re-approval / double-spend. They are grouped under +// the "Pending" filter chip (not "Completed") for the same reason. +function isInFlightStatus(s: string): boolean { + return s === 'approved' || s === 'running' || s === 'paused'; +} + +// readDeepLinkExecutionID returns the value of the ?execution= +// query parameter inside the current location hash, or '' when +// absent. The app uses hash routing ("#history"), so the "query" piece +// sits inside `window.location.hash` — `window.location.search` is +// empty. Parse it manually: +// #history?execution=abc123 → 'abc123' +// Exported for unit-test coverage. +export function readDeepLinkExecutionID(): string { + const hash = window.location.hash || ''; + const q = hash.split('?')[1]; + if (!q) return ''; + return new URLSearchParams(q).get('execution') || ''; +} + +// applyExecutionDeepLink scrolls the history table to the row matching +// the ?execution= hash query, if any, and flashes a highlight +// class on it. Called after each loadHistory render so links from +// the Recommendations badge AND the scheduled-purchase email's +// Review & Edit button land on the right row. Returns true when a +// match was found + highlighted. +// +// Falsy paths: +// - No execID in URL: return false silently (the common case — no +// deeplink was requested). +// - execID present but no matching row in the rendered list (e.g. +// the user's date filter excludes the execution, or the row hasn't +// been ingested yet): surface a non-blocking toast so the user +// understands why the page didn't jump anywhere, and clear the +// hash so a follow-up loadHistory() doesn't re-toast the same +// miss on every re-render. +// +// Exported for unit-test coverage. +export function applyExecutionDeepLink(): boolean { + const execID = readDeepLinkExecutionID(); + if (!execID) return false; + const row = document.querySelector( + `tr[data-execution-id="${CSS.escape(execID)}"]`, + ); + if (!row) { + // Short-prefix the ID so the toast is readable but the user can + // still cross-reference against the email if needed. + const shortID = execID.length > 8 ? `${execID.slice(0, 8)}…` : execID; + showToast({ + message: `Execution ${shortID} isn't in the current view — clear filters or widen the date range to find it.`, + kind: 'info', + timeout: 8_000, + }); + // Drop the ?execution= from the hash so the next loadHistory() + // (e.g. user changes a filter) doesn't fire this toast again. + if (window.location.hash) { + const baseHash = window.location.hash.split('?')[0] ?? ''; + window.history.replaceState({}, '', window.location.pathname + window.location.search + baseHash); + } + return false; + } + row.classList.add('history-row-highlight'); + row.scrollIntoView({ behavior: 'smooth', block: 'center' }); + // Fade the highlight after a few seconds so the row goes back to + // normal styling — otherwise it stays yellow forever and a future + // user click on a different row looks inconsistent. + window.setTimeout(() => row.classList.remove('history-row-highlight'), 4000); + return true; +} + +/** + * Setup history filter event handlers. + * + * Provider/account filters are global (sourced from state.ts via the + * topbar chips). The history-section's own controls are just date range + * + Load History button — those stay here. + */ +export function setupHistoryHandlers(): void { + state.subscribeProvider(() => void loadHistory()); + state.subscribeAccount(() => void loadHistory()); + // Re-render both tables when the amortize toggle flips (issue #1112). + state.subscribeAmortizeUpfront(() => { + renderHistoryList(lastPurchases); + renderApprovalQueue(lastPurchases); + syncAmortizeCheckbox('history-amortize-checkbox'); + syncAmortizeCheckbox('approval-queue-amortize-checkbox'); + }); +} + +/** + * Mount the "Amortize upfront over term" checkbox into a container element + * (idempotent -- safe to call on every loadHistory). + * + * The checkbox is wired to setAmortizeUpfront so a change here is reflected + * in all other views via the shared localStorage key + subscriber pattern. + */ +function mountAmortizeCheckbox(containerId: string, checkboxId: string): void { + const container = document.getElementById(containerId); + if (!container) return; + if (document.getElementById(checkboxId)) return; // already mounted + + const wrapper = document.createElement('label'); + wrapper.className = 'amortize-toggle-label'; + wrapper.htmlFor = checkboxId; + + const cb = document.createElement('input'); + cb.type = 'checkbox'; + cb.id = checkboxId; + cb.checked = state.getAmortizeUpfront(); + cb.addEventListener('change', () => { + state.setAmortizeUpfront(cb.checked); + }); + + wrapper.appendChild(cb); + wrapper.appendChild(document.createTextNode(' Amortize upfront over term')); + container.appendChild(wrapper); +} + +/** Keep an already-mounted checkbox in sync when state changes externally. */ +function syncAmortizeCheckbox(checkboxId: string): void { + const cb = document.getElementById(checkboxId) as HTMLInputElement | null; + if (cb) cb.checked = state.getAmortizeUpfront(); +} + +/** + * Initialize history date range. + * + * Defaults to a 7-day window because the Purchase events table is a *log* + * view — recent activity is what matters; older days are mostly empty and + * mostly noise. The Savings History card on the same page covers the + * complementary multi-month *trend* view (default 90 days, see + * `#savings-period` in index.html), so the two controls open to their + * own natural windows rather than fighting over one default. + */ +export function initHistoryDateRange(): void { + const end = new Date(); + const start = new Date(); + start.setDate(start.getDate() - 7); + + const startInput = document.getElementById('history-start') as HTMLInputElement | null; + const endInput = document.getElementById('history-end') as HTMLInputElement | null; + + if (startInput && !startInput.value) { + startInput.value = start.toISOString().split('T')[0] || ''; + } + if (endInput && !endInput.value) { + endInput.value = end.toISOString().split('T')[0] || ''; + } +} + +/** + * View plan history. + * + * The plan-history endpoint returns the plan's *full* history regardless + * of date range — `api.getHistory({ planId })` ignores `start`/`end`. + * Don't seed the From/To inputs with the generic 7-day default here: + * they would visibly disagree with the table contents (the tab-level + * default would suggest the table only covers the last week, while in + * fact every purchase the plan ever recorded is shown). Instead, after + * the fetch lands, snap the inputs to the actual min/max purchase + * timestamps so the date pickers reflect what the user is looking at. + * + * If the plan has no purchases yet, leave the inputs untouched — there's + * no meaningful range to display, and clobbering them with `today` + * would be misleading. + */ +export async function viewPlanHistory(planId: string): Promise { + // skipDefaultLoad: the tab's own unscoped 7-day fetch would land after the + // plan-scoped one below and overwrite it, and its date-range seeding is + // exactly what the doc comment above says not to do here. + switchTab('purchases', { skipDefaultLoad: true }); + // switchTab renders a no-access placeholder for sessions without + // view:purchases; don't overwrite it with the plan's purchases. + if (!canAccess('view', 'purchases')) return; + + try { + const [data] = await Promise.all([ + api.getHistory({ planId }) as unknown as Promise, + refreshFourEyesMode(), + ]); + renderHistorySummary(data.summary ?? null); + const purchases = data.purchases || []; + renderApprovalQueue(purchases); + renderHistoryList(purchases); + snapDateInputsToPurchases(purchases); + } catch (error) { + console.error('Failed to load plan history:', error); + const err = error as Error; + const list = document.getElementById('history-list'); + if (list) { + list.innerHTML = `

Failed to load plan history: ${escapeHtml(err.message)}

`; + } + // Mirror the loadHistory catch: clear the approval queue on error + // so stale pending rows from a previous render don't sit on screen + // alongside the failure message and tempt clicks on outdated state. + const queue = document.getElementById('purchases-approval-queue'); + if (queue) { + queue.innerHTML = `

Failed to load approval queue: ${escapeHtml(err.message)}

`; + } + } +} + +/** + * Set the From/To inputs to bracket the purchases that just rendered. + * No-op when the list is empty — keeping the previous values is more + * honest than seeding a fake "today" range. Timestamps are normalised + * to UTC YYYY-MM-DD so they slot directly into ``. + */ +function snapDateInputsToPurchases(purchases: HistoryPurchase[]): void { + if (purchases.length === 0) return; + const epochs = purchases + .map(p => Date.parse(p.timestamp)) + .filter(n => !Number.isNaN(n)); + if (epochs.length === 0) return; + const startInput = document.getElementById('history-start') as HTMLInputElement | null; + const endInput = document.getElementById('history-end') as HTMLInputElement | null; + const minDate = new Date(Math.min(...epochs)).toISOString().split('T')[0] || ''; + const maxDate = new Date(Math.max(...epochs)).toISOString().split('T')[0] || ''; + if (startInput) startInput.value = minDate; + if (endInput) endInput.value = maxDate; +} + +/** + * Load history with filters + */ +export async function loadHistory(): Promise { + // Issue #344 T3: skeleton rows for the purchase-history table. 8 + // rows matches the typical first-page row count so the skeleton + // doesn't shrink dramatically when real data arrives. Column count + // (12) mirrors the rendered table headers in renderHistoryList: + // Status / Date / Provider / Service / Type / Region / Count / + // Term / Upfront Cost / Monthly Cost / Monthly Savings / Plan. + const listEl = document.getElementById('history-list'); + if (listEl) showSkeletonRows(listEl, 8, 12); + // Pending-approval queue card (issue #340 sub-task): 3 rows x 12 + // cols matches the queue table shape (Date / Account / Provider / + // Service / Count / Term / Payment / Monthly Cost / Upfront Cost / + // Monthly Savings / Created by / Actions). The queue is typically + // much shorter than the full history list, so 3 rows is a sensible + // skeleton size. + const queueEl = document.getElementById('purchases-approval-queue'); + if (queueEl) showSkeletonRows(queueEl, 3, 12); + + // Close any open History column-filter popover before re-rendering so the + // popover doesn't sit anchored to a stale button after the table is + // innerHTML-rewritten. The next render with active filters rebinds fresh + // triggers; the popover stays opt-in (user clicks again to re-open). + closeOpenHistoryPopover(); + + try { + // Provider/account filters live in state.ts now (mutated by topbar chips). + const rawProvider = state.getCurrentProvider(); + const provider: api.Provider | undefined = (VALID_PROVIDERS as string[]).includes(rawProvider) + ? (rawProvider as api.Provider) + : undefined; + + const stateAccountIDs = state.getCurrentAccountIDs(); + const accountIDs: string[] | undefined = stateAccountIDs.length > 0 ? stateAccountIDs : undefined; + + const filters: api.HistoryFilters = { + start: (document.getElementById('history-start') as HTMLInputElement | null)?.value, + end: (document.getElementById('history-end') as HTMLInputElement | null)?.value, + provider, + account_ids: accountIDs + }; + const [data] = await Promise.all([ + api.getHistory(filters) as unknown as Promise, + refreshFourEyesMode(), + ]); + renderHistorySummary(data.summary ?? null); + const purchases = data.purchases || []; + renderApprovalQueue(purchases); + renderHistoryList(purchases); + } catch (error) { + console.error('Failed to load history:', error); + const err = error as Error; + const list = document.getElementById('history-list'); + if (list) { + teardownSkeleton(list); + list.innerHTML = `

Failed to load history: ${escapeHtml(err.message)}

`; + } + const queue = document.getElementById('purchases-approval-queue'); + if (queue) { + teardownSkeleton(queue); + queue.innerHTML = `

Failed to load approval queue: ${escapeHtml(err.message)}

`; + } + } +} + +function renderHistorySummary(summary: HistorySummary | null): void { + const container = document.getElementById('history-summary'); + if (!container) return; + + // When the API omits the summary field entirely (older deploy, partial + // response, or an error absorbed upstream), render an explicit unknown + // state on each card rather than fabricating all-zero values that look + // like real financial aggregates. + if (summary === null || summary === undefined) { + container.innerHTML = ` +
+

Total Purchases

+

--

+
+
+

Total Upfront Spent

+

--

+
+
+

Monthly Savings

+

--

+
+
+

Annual Savings

+

--

+
+ `; + return; + } + + // total_completed / total_pending fall back to total_purchases so the + // summary renders sensibly against an older API deploy that hasn't shipped + // the new counters yet. + const total = summary.total_purchases ?? null; + const totalDisplay = total !== null ? String(total) : '--'; + const completed = summary.total_completed ?? total; + const pending = summary.total_pending ?? 0; + const detail = (total !== null && pending > 0) + ? `

${completed} completed · ${pending} pending

` + : ''; + + container.innerHTML = ` +
+

Total Purchases

+

${totalDisplay}

+ ${detail} +
+
+

Total Upfront Spent

+

${formatCurrency(summary.total_upfront ?? null)}

+
+
+

Monthly Savings

+

${formatCurrency(summary.total_monthly_savings ?? null)}

+
+
+

Annual Savings

+

${formatCurrency(summary.total_annual_savings ?? null)}

+
+ `; +} + +function statusBadgeHTML(status: string): string { + const normalized = (status || 'completed').toLowerCase(); + switch (normalized) { + case 'pending': + case 'notified': + return 'Pending'; + case 'approved': + case 'running': + case 'paused': + // In-flight (issue #621): not finished — never show the green Completed + // badge for these, or the user may think the purchase is done. + return 'In Progress'; + case 'canceled': + case 'cancelled': + // Migration 000089 (expand-contract rename): the backend may return + // either the new US spelling ('canceled') or the legacy British + // spelling ('cancelled') during the rolling deploy window. Match both + // so a row written by EITHER old or new code renders the muted Cancelled + // badge instead of falling through to the green Completed default. + // The CONTRACT migration (#1278) will normalize the data once the deploy + // is stable; the British branch can be removed then. + return 'Cancelled'; + case 'partially_completed': + // #642: some commitments succeeded, some failed. Not a clean success + // and never "failed" (real commitments exist) — a distinct warning badge + // so the user knows to read the description and reconcile the failures. + return 'Partial'; + case 'failed': + return 'Failed'; + case 'expired': + return 'Expired'; + default: + return 'Completed'; + } +} + +function buildStatusChipRowHTML(purchases: HistoryPurchase[], active: StatusFilter): string { + const counts: Record = { + all: purchases.length, + pending: 0, + completed: 0, + failed: 0, + expired: 0, + cancelled: 0, + }; + for (const p of purchases) { + const s = normalizeStatus(p).toLowerCase(); + if (s === 'pending' || s === 'notified' || isInFlightStatus(s)) counts.pending++; + // Migration 000089 (expand-contract rename): the backend may return + // either spelling during the rolling deploy window. Counting only the + // British spelling would silently bucket new 'canceled' rows into the + // Completed total, hiding them from the user. + else if (s === 'canceled' || s === 'cancelled') counts.cancelled++; + else if (s === 'failed') counts.failed++; + else if (s === 'expired') counts.expired++; + else counts.completed++; + } + // Only render Failed / Expired / Cancelled chips when there's something in + // them — keeps the row uncluttered on healthy deployments that have never + // seen one. All / Pending / Completed always render so the user has a + // consistent filter affordance even on an empty or fresh dataset. + const allChips: Array<{ key: StatusFilter; label: string }> = [ + { key: 'all', label: 'All' }, + { key: 'pending', label: 'Pending' }, + { key: 'completed', label: 'Completed' }, + { key: 'failed', label: 'Failed' }, + { key: 'expired', label: 'Expired' }, + { key: 'cancelled', label: 'Cancelled' }, + ]; + const chips = allChips.filter(c => c.key === 'all' || c.key === 'pending' || c.key === 'completed' || counts[c.key] > 0); + return ` +
+ ${chips.map(c => ` + + `).join('')} +
+ `; +} + +function providerCell(p: HistoryPurchase): string { + if (!p.provider || p.provider === 'multiple') return 'Multiple'; + // Whitelist provider to prevent stored XSS via class attribute injection (#443). + const safeProvider = (VALID_PROVIDERS as string[]).includes(p.provider) ? p.provider : 'unknown'; + return `${safeProvider.toUpperCase()}`; +} + +// canCancelPendingRow returns true when the current session is permitted +// to cancel the given pending/notified history row via the session-authed +// Cancel button (issue #46). UX gate only — the backend +// authorizeSessionCancel in internal/api/handler_purchases.go remains the +// security boundary; if this helper is wrong-positive the API surfaces +// 403 and the click handler turns that into a "Failed to cancel" toast. +// +// Heuristic: +// * admin → always yes (canAccess('admin', '*')); +// * cancel-any:purchases → yes (operator roles, issue #158); +// * cancel-own:purchases + matching created_by_user_id → yes; +// * anyone else, or legacy row with no created_by_user_id → no. +function canCancelPendingRow(p: HistoryPurchase): boolean { + const status = (p.status || '').toLowerCase(); + if (status !== 'pending' && status !== 'notified') return false; + const user = getCurrentUser(); + if (!user) return false; + if (canAccess('admin', '*') || canAccess('cancel-any', 'purchases')) return true; + // cancel-own: only the original creator. Legacy rows with no + // created_by_user_id can't be cancelled via this UI; the email-token + // path remains the escape hatch. + if (!p.created_by_user_id) return false; + return canAccess('cancel-own', 'purchases') && p.created_by_user_id === user.id; +} + +// canApproveUnder4Eyes returns true when the 4-eyes dual-control policy +// (issue #1005, GlobalConfig.require_different_approver) allows sessionUserId +// to approve a row created by row.created_by_user_id. Mirrors the backend's +// requireDifferentApprover in internal/api/handler_purchases.go: mode off → +// always allowed; mode on → allowed only when the row has a recorded, +// different creator (a NULL/legacy creator is denied, matching the backend's +// fail-closed 403 for rows that predate dual-control). +function canApproveUnder4Eyes(row: HistoryPurchase, sessionUserId: string): boolean { + return _fourEyesMode === false || (row.created_by_user_id != null && row.created_by_user_id !== sessionUserId); +} + +// rbacAllowsApprove is the approve-permission decision (issue #286 / +// #1407) WITHOUT the 4-eyes overlay, so renderPendingActionButtons can +// distinguish "no permission at all" (no button, no badge) from +// "permission would allow it but 4-eyes blocks it" (badge instead of button). +function rbacAllowsApprove(p: HistoryPurchase, user: { id: string }): boolean { + const status = (p.status || '').toLowerCase(); + if (status !== 'pending' && status !== 'notified') return false; + // approve-any:purchases is carved out of admin:* (issue #923) and is + // granted by the seeded Purchaser group OR any custom group that + // explicitly lists the verb in effectivePermissions. Gate on the + // verb directly so a non-seeded role with the same grant still + // approves rows the backend would also let through. + if (canAccess('approve-any', 'purchases')) return true; + // Four-eyes RBAC (issue #1407): the session must hold an explicit + // approve-own grant before ownership is consulted. Ownership alone never + // grants approve. + if (!canAccess('approve-own', 'purchases')) return false; + if (!p.created_by_user_id) return false; + return p.created_by_user_id === user.id; +} + +// canApprovePendingRow returns true when the current session is permitted +// to approve the given pending history row via the inline Approve button +// (issue #286). UX gate only — the backend authorizeSessionApprove in +// internal/api/handler_purchases.go remains the security boundary; a +// false-positive here surfaces as a 403 toast on click rather than a +// successful approve. +// +// Heuristic (four-eyes RBAC — issue #1407; 4-eyes dual-control — issue #1005): +// * status must be "pending" or "notified"; +// * any session with approve-any:purchases (carved-out admin verb, +// seeded on Purchaser group; can also come from a custom group via +// effectivePermissions) → approve-any; shows Approve on every pending row; +// * session must also hold approve-own:purchases before ownership is +// even evaluated (four-eyes: ownership alone does NOT grant approve); +// * only then: the row's created_by_user_id must match the current user; +// * legacy rows with NULL created_by_user_id → no (the email-token +// path remains the escape hatch); +// * finally, canApproveUnder4Eyes must allow it: when dual-control mode is +// on, the session cannot approve a row it created itself. +function canApprovePendingRow(p: HistoryPurchase): boolean { + const user = getCurrentUser(); + if (!user) return false; + if (!rbacAllowsApprove(p, user)) return false; + return canApproveUnder4Eyes(p, user.id); +} + +// canRetryFailedRow returns true when the current session is permitted +// to retry the given failed history row via the inline Retry button +// (issue #47). UX gate only — the backend authorizeSessionRetry in +// internal/api/handler_purchases.go remains the security boundary. +// +// Heuristic: +// * status must be "failed"; +// * row must NOT carry an ops_hint (persistent failure → no retry, +// show the hint instead); +// * row must NOT already have a retry_execution_id (we don't allow +// retrying the same failure twice — the user should retry the +// latest descendant in the chain); +// * any session with retry-any:purchases (carved-out admin verb, +// seeded on Purchaser group; can also come from a custom group via +// effectivePermissions) → retry-any; +// * otherwise the row's created_by_user_id must match the current +// user (retry-own). +function canRetryFailedRow(p: HistoryPurchase): boolean { + const status = (p.status || '').toLowerCase(); + if (status !== 'failed') return false; + if (p.ops_hint) return false; + if (p.retry_execution_id) return false; // already retried — user should act on the descendant + const user = getCurrentUser(); + if (!user) return false; + // retry-any:purchases is carved out of admin:* (issue #923) and is + // granted by the seeded Purchaser group OR any custom group that + // explicitly lists the verb in effectivePermissions. Gate on the + // verb directly so a non-seeded role with the same grant still + // retries rows the backend would also let through. + if (canAccess('retry-any', 'purchases')) return true; + // Mirror canCancelPendingRow / canApprovePendingRow (issue #1418): require + // the retry-own permission explicitly before checking creator match. + // Without this gate a role that holds NO retry permission at all (e.g. a + // Plan Authors group with only plan verbs) would still see the Retry button + // on rows they created, because the creator check alone was a sufficient + // condition. The backend authorizeSessionRetry is the real security boundary; + // this closes the UX gap. + if (!canAccess('retry-own', 'purchases')) return false; + if (!p.created_by_user_id) return false; + return p.created_by_user_id === user.id; +} + +// canRevokeCompletedRow returns true when the current session may revoke the +// given purchase row via the inline Revoke button (issue #290). +// UX gate only -- the backend authorizeSessionRevoke remains the real +// security boundary. +// +// Conditions: +// * status must be "completed", "" (legacy blank), or "scheduled" +// (pre-fire delay: the cloud SDK has not been called yet -- free cancel); +// * provider must be "azure" (AWS and GCP have no direct cancel API); +// * revocation_window_closes_at must be in the future; +// for "scheduled" rows this field is populated with scheduled_execution_at +// by the backend (issue #290, second-wave CR Finding E); +// * row must not already be revoked (revoked_at absent); +// * session must have revoke-any:purchases or revoke-own:purchases. Without +// this the button rendered for every signed-in user and the backend just +// 403d, replicating the same UX-vs-RBAC drift PR #995 caught for the +// approve / delete paths. Mirror the peer predicates (canCancelPendingRow, +// canApprovePendingRow, canRetryFailedRow) which all check canAccess. +function canRevokeCompletedRow(p: HistoryPurchase): boolean { + const status = (p.status || '').toLowerCase(); + if (status !== 'completed' && status !== '' && status !== 'scheduled') return false; + if ((p.provider || '').toLowerCase() !== 'azure') return false; + if (p.revoked_at) return false; // already revoked + if (!p.revocation_window_closes_at) return false; + if (new Date(p.revocation_window_closes_at) <= new Date()) return false; + const user = getCurrentUser(); + if (!user) return false; + // RBAC: admin or revoke-any always; otherwise revoke-own (account-scope + // ownership is enforced server-side, the same model as the backend handler). + if (canAccess('admin', '*') || canAccess('revoke-any', 'purchases')) return true; + return canAccess('revoke-own', 'purchases'); +} + +// retryThresholdReached returns true when the row has hit the soft- +// block threshold (5 attempts). The frontend shows a confirm-with- +// warning dialog and forwards force=true on confirmation. +// +// Kept in sync with retryThreshold in internal/api/handler_purchases.go; +// the backend remains authoritative — if this client predicate disagrees +// with the server, the API surfaces a structured 409 with retry_attempt_n +// + threshold and the toast falls back to the server message. +const RETRY_THRESHOLD = 5; +function retryThresholdReached(p: HistoryPurchase): boolean { + return (p.retry_attempt_n ?? 0) >= RETRY_THRESHOLD; +} + +// canSellOnMarketplace returns true when the current session is permitted +// to list the given completed history row on the AWS RI Marketplace +// (issue #292). UX gate only -- the backend authorizeSessionSell in +// internal/api/handler_marketplace.go remains the security boundary; a +// false-positive here surfaces as a 403 toast on click. +// +// Conditions: +// * row must be a completed purchase (status "completed" or absent); +// * row must be an AWS EC2 RI (marketplace is EC2-only); +// * offering_class must be "standard" OR still unknown (empty). CUDly stamps +// "convertible" on its own EC2 purchases, but externally-created Standard +// RIs (and pre-migration rows) have an empty offering_class until the +// backend lazily populates it from AWS on the list call. The backend +// definitively gates -- it 400s a fetched "convertible" -- so we show the +// button for unknown-class EC2 rows and let the backend decide, otherwise +// the feature is unreachable end-to-end for the very case it targets; +// * no active listing already (listing_state != "active"); and +// * admin, or non-admin user (sell-own covers their own accounts -- +// we can't efficiently check per-account ownership client-side, so +// we show the button for all non-admin users and let the backend 403 +// when the account is out of scope). +function canSellOnMarketplace(p: HistoryPurchase): boolean { + const status = normalizeStatus(p).toLowerCase(); + if (status !== 'completed') return false; + // Marketplace listing is AWS EC2 Standard-RI only. Gate on provider/service + // so unknown-class rows for non-EC2 providers never show the Sell button. + if ((p.provider || '').toLowerCase() !== 'aws') return false; + if ((p.service || '').toLowerCase() !== 'ec2') return false; + const offeringClass = (p.offering_class || '').toLowerCase(); + if (offeringClass !== 'standard' && offeringClass !== '') return false; + if ((p.listing_state || '').toLowerCase() === 'active') return false; + const user = getCurrentUser(); + if (!user) return false; + // Gate on the sell verbs, not bare sign-in, so we don't show Sell to a + // role that lacks sell-own/sell-any and avoid frontend/backend auth drift + // (the backend authorizeSessionSell would 403 anyway). admin:* satisfies + // sell-own here since it is not carved out of admin. + if ( + !canAccess('admin', '*') && + !canAccess('sell-any', 'purchases') && + !canAccess('sell-own', 'purchases') + ) { + return false; + } + // Guard against listing a matured RI: compute remaining months from the + // purchase timestamp and the total term. purchase_history.term is stored in + // YEARS (1 or 3), so convert to months before comparing against elapsed + // months (mirrors the row.Term * 12 conversion in handler_marketplace.go). + // Without the conversion a 3-year RI was treated as 3 months and the Sell + // button vanished after ~3 months. We require at least 1 full month remaining. + const termYears = typeof p.term === 'number' ? p.term : Number(p.term) || 0; + if (termYears <= 0) return false; + const termMonths = termYears * 12; + const purchaseMs = new Date(p.timestamp).getTime(); + if (!Number.isFinite(purchaseMs)) return false; + const elapsedMonths = (Date.now() - purchaseMs) / (1000 * 60 * 60 * 24 * 30.4375); + const remainingMonths = termMonths - elapsedMonths; + if (remainingMonths < 1) return false; + return true; +} + +// canCancelMarketplaceListing returns true when there is an active listing +// that the current session can cancel. +function canCancelMarketplaceListing(p: HistoryPurchase): boolean { + if ((p.listing_state || '').toLowerCase() !== 'active') return false; + const user = getCurrentUser(); + if (!user) return false; + // Same sell-verb gate as canSellOnMarketplace: cancelling a listing is a + // marketplace write, so require sell-own/sell-any (or admin) rather than + // bare sign-in to keep the UX gate aligned with authorizeSessionSell. + if ( + !canAccess('admin', '*') && + !canAccess('sell-any', 'purchases') && + !canAccess('sell-own', 'purchases') + ) { + return false; + } + return true; +} + +// shortExecID renders the first 8 chars of a UUID so inline lineage +// links ("Retried as #abc12345") stay readable in the table cell. The +// full ID is preserved in the data-history-status attribute so the +// click handler can deep-link without truncation surprises. +function shortExecID(id: string): string { + return (id || '').replace(/^urn:.*?:/, '').slice(0, 8); +} + +// sameRowActions returns the set of row-action buttons (Approve, Cancel, +// future siblings) in the same table cell as `btn`. Used by the Approve +// and Cancel click handlers to disable BOTH actions for the in-flight +// row while the API request is pending — issue #286 + CR pass on +// PR #299: clicking Approve disabled only Approve, leaving the +// adjacent Cancel button live; a quick double-click could fire +// conflicting requests on the same row before the reload completes. +// +// Falls back to `[btn]` when the button has no parent (test +// fixtures may render buttons without a wrapping cell). +function sameRowActions(btn: HTMLButtonElement): HTMLButtonElement[] { + const cell = btn.closest('td') || btn.parentElement; + if (!cell) return [btn]; + return Array.from( + cell.querySelectorAll( + '.history-approve-btn, .history-cancel-btn, .history-revoke-btn, .history-marketplace-sell-btn, .history-marketplace-cancel-btn', + ), + ); +} + +// renderPendingActionButtons returns the inline Approve / Cancel +// button HTML for a pending|notified row, or "" when neither verb is +// available to the current session. Extracted from renderActionCell +// so the approval-queue card can emit identical buttons without +// duplicating the predicate logic or the DOM contract. Each predicate +// is checked independently so a custom role with only one of the +// verbs renders just that button; Approve sits to the left as the +// affirmative action. +function renderPendingActionButtons(p: HistoryPurchase): string { + if (!p.purchase_id) return ''; + const buttons: string[] = []; + const user = getCurrentUser(); + if (canApprovePendingRow(p)) { + buttons.push(``); + } else if (user && rbacAllowsApprove(p, user) && !canApproveUnder4Eyes(p, user.id)) { + // RBAC would allow Approve, but 4-eyes dual-control (issue #1005) blocks + // this session from approving its own row. Surface the reason inline + // instead of silently hiding the action. + buttons.push('Awaiting different approver'); + } + if (canCancelPendingRow(p)) { + buttons.push(``); + } + return buttons.join(' '); +} + +// renderActionCell returns the HTML for the Plan / action column on a +// single history row. The column doubles as the per-row action surface +// because pending / failed rows rarely have a meaningful plan_name (the +// pending plan info already shows in StatusDescription) and reusing the +// existing column keeps table width unchanged. +// +// Decision tree (status-driven, mutually exclusive): +// * pending|notified + canCancel → Cancel button (issue #46) +// * failed + ops_hint set → ⚠ ops-hint badge (issue #47, Q3 — no retry) +// * failed + threshold reached → "Retried 5× — confirm to override" Retry button (Q2) +// * failed + canRetry → standard ↻ Retry button +// * any row with retry lineage → inline ↻ Retried as / ↻ Retry of link +// * else → plan_name or "-" +function renderActionCell(p: HistoryPurchase): string { + // Pending → render Approve + Cancel side-by-side when the session + // qualifies for both (the typical case after issue #286 — the same + // approve-own / cancel-own grant lives in DefaultUserPermissions). + const pendingButtons = renderPendingActionButtons(p); + if (pendingButtons) { + return pendingButtons; + } + + // Failed → either ops-hint (no retry possible), Retry (with optional + // threshold-confirm flag), plus lineage link to the successor if it + // exists. We never show ops-hint AND Retry on the same row — the + // hint replaces the action entirely because retrying a persistent + // misconfig is guaranteed to fail again. + if ((p.status || '').toLowerCase() === 'failed') { + if (p.ops_hint) { + return `⚠ ${escapeHtml(p.ops_hint)}`; + } + if (canRetryFailedRow(p) && p.purchase_id) { + const overThreshold = retryThresholdReached(p); + const label = overThreshold ? `⚠ Retried ${p.retry_attempt_n ?? 0}× — click to override` : '↻ Retry'; + const cls = overThreshold ? 'btn-link history-retry-btn history-retry-over-threshold' : 'btn-link history-retry-btn'; + return ``; + } + } + + // Lineage links: a row that was retried (retry_execution_id set) or + // is itself a retry (retry_attempt_n > 0) gets an inline cross-link + // to the other end of the chain so the user can navigate without + // scrolling. Both can be true simultaneously on a middle-of-chain + // row (failed_v2 was retried into v3 AND is itself a retry of v1). + const lineage: string[] = []; + if (p.retry_execution_id) { + lineage.push(`↻ Retried as #${escapeHtml(shortExecID(p.retry_execution_id))}`); + } + if ((p.retry_attempt_n ?? 0) > 0) { + // We don't carry a back-pointer field — a future enhancement + // could surface the predecessor's exec ID via the API. For now + // we render a static badge (no link target) so users at least + // see "this is a retry" provenance. + lineage.push(`↻ Retry #${p.retry_attempt_n}`); + } + // Build the trailing action buttons so they compose with lineage links + // — a retry-descendant can also have an active listing that needs a + // Cancel button, or be an Azure row still inside its revoke window. + const trailingActions: string[] = []; + if (p.purchase_id) { + // Completed Azure row within revocation window: Revoke button (issue + // #290). Only Azure supports direct in-app revocation; AWS and GCP have + // no cancel API so the button is suppressed for those providers. + if (canRevokeCompletedRow(p)) { + trailingActions.push(``); + } + // Completed Standard RI rows (AWS): Cancel listing / Sell on Marketplace + // (issue #292). Mutually exclusive with revoke in practice (revoke is + // Azure-only, marketplace is AWS Standard-RI-only). + if (canCancelMarketplaceListing(p)) { + trailingActions.push(``); + } else if (canSellOnMarketplace(p)) { + trailingActions.push(``); + } + } + + if (lineage.length > 0 || trailingActions.length > 0) { + return [...lineage, ...trailingActions].join(' '); + } + + return escapeHtml(p.plan_name || '-'); +} + +// --------------------------------------------------------------------------- +// Per-column filter wiring for the Purchase History table. +// +// The Status chip-row above stays as-is — it's an enum-driven filter with no +// natural fit for the generic set/expr column-filter shape, and the chip-row +// is the more discoverable affordance for status anyway. Column filters here +// cover the row attributes (provider/service/type/region/term/count/upfront/ +// savings) — Status is intentionally excluded. +// +// Numeric extractors round to 0 decimal places to match formatCurrency's +// default (CURRENCY_DEFAULT_DIGITS), so a user typing the visible "$123" +// matches the row that displays that exact value (issue #484 contract on +// the recommendations table). +// --------------------------------------------------------------------------- + +const PURCHASE_HISTORY_NUMERIC_COLUMNS: ReadonlySet = new Set([ + 'count', 'upfront_cost', 'savings', +]); + +function purchaseHistoryCategoricalCellValue( + p: HistoryPurchase, + col: PurchaseHistoryColumnId, +): string { + switch (col) { + case 'provider': return p.provider ?? ''; + case 'service': return p.service ?? ''; + case 'resource_type': return p.resource_type ?? ''; + case 'region': return p.region ?? ''; + case 'term': return p.term == null ? '' : String(p.term); + case 'count': + case 'upfront_cost': + case 'savings': return ''; + } +} + +function purchaseHistoryNumericCellValue( + p: HistoryPurchase, + col: PurchaseHistoryColumnId, +): number { + switch (col) { + case 'count': return p.count ?? 0; + case 'upfront_cost': return p.upfront_cost ?? 0; + case 'savings': return p.estimated_savings ?? 0; + case 'provider': + case 'service': + case 'resource_type': + case 'region': + case 'term': return Number.NaN; + } +} + +// Round to display precision so typed values match the rendered cell value +// (formatCurrency default of 0 fraction digits, formatTerm renders the +// integer term unchanged). +function roundForDisplay(n: number): number { + if (!Number.isFinite(n)) return n; + return Number(n.toFixed(0)); +} + +export function applyPurchaseHistoryColumnFilters( + purchases: readonly HistoryPurchase[], + filters: state.PurchaseHistoryColumnFilters, +): HistoryPurchase[] { + return applyColumnFilters( + purchases, + filters, + { + categorical: purchaseHistoryCategoricalCellValue, + numeric: (p, col) => roundForDisplay(purchaseHistoryNumericCellValue(p, col)), + }, + ); +} + +const PURCHASE_HISTORY_LABELS: Record = { + provider: 'Provider', + service: 'Service', + resource_type: 'Type', + region: 'Region', + term: 'Term', + count: 'Count', + upfront_cost: 'Upfront Cost', + savings: 'Monthly Savings', +}; + +function purchaseHistoryDistinctValues( + purchases: readonly HistoryPurchase[], + column: PurchaseHistoryColumnId, +): string[] { + const seen = new Set(); + for (const p of purchases) { + seen.add(purchaseHistoryCategoricalCellValue(p, column)); + } + return Array.from(seen).sort((a, b) => { + if (a === '' && b !== '') return -1; + if (a !== '' && b === '') return 1; + return a.localeCompare(b); + }); +} + +function purchaseHistoryDisplayLabel( + column: PurchaseHistoryColumnId, + value: string, +): string { + if (value === '') return '(empty)'; + if (column === 'term') { + const n = Number(value); + return Number.isFinite(n) ? formatTerm(n) : value; + } + return value; +} + +function wirePurchaseHistoryFilterButtons( + container: HTMLElement, + // The pre-column-filter slice — popover lists distinct values from + // every row that survived the status chip, NOT the further-narrowed + // visible slice (otherwise the popover would lose values the user just + // unchecked). + sourceRows: readonly HistoryPurchase[], +): void { + container.querySelectorAll('.history-column-filter-btn').forEach((btn) => { + const column = btn.dataset['column'] as PurchaseHistoryColumnId | undefined; + if (!column) return; + btn.addEventListener('click', (e) => { + e.stopPropagation(); + const isNumeric = PURCHASE_HISTORY_NUMERIC_COLUMNS.has(column); + const filters = state.getPurchaseHistoryColumnFilters(); + openHistoryColumnPopover({ + column, + anchor: btn, + currentFilter: filters[column], + headerLabel: PURCHASE_HISTORY_LABELS[column], + kind: isNumeric ? 'numeric' : 'categorical', + distinctValues: isNumeric ? undefined : purchaseHistoryDistinctValues(sourceRows, column), + displayLabel: (v) => purchaseHistoryDisplayLabel(column, v), + onCommit: (filter) => { + state.setPurchaseHistoryColumnFilter(column, filter); + renderHistoryList(lastPurchases); + }, + }); + }); + }); +} + +function renderHistoryList(purchases: HistoryPurchase[]): void { + const container = document.getElementById('history-list'); + if (!container) return; + + lastPurchases = purchases; + + // Issue #923: inject a read-only notice for sessions that lack the + // carved-out spending verbs the buttons in this table need. Approve / + // Retry are gated by canApprovePendingRow / canRetryFailedRow which + // call canAccess('approve-any','purchases') and + // canAccess('retry-any','purchases'); use the same predicate here so + // the banner stays in lockstep with the visible buttons (a user who + // holds either verb via a custom group sees no contradictory + // notice). + const canApproveAny = canAccess('approve-any', 'purchases'); + const canRetryAny = canAccess('retry-any', 'purchases'); + const hasAnyCarvedOut = canApproveAny || canRetryAny; + const existingBanner = document.getElementById('history-no-purchaser-banner'); + if (!existingBanner && !hasAnyCarvedOut) { + const banner = document.createElement('div'); + banner.id = 'history-no-purchaser-banner'; + banner.className = 'info-banner'; + banner.setAttribute('role', 'note'); + banner.textContent = + 'You can view and plan, but not execute purchases directly. ' + + 'Ask an admin to add you to the Purchaser group (Admin → Users) to execute purchases.'; + container.parentElement?.insertBefore(banner, container); + } else if (existingBanner && hasAnyCarvedOut) { + existingBanner.remove(); + } + + // Reset the filter when the dataset changes so the user isn't stuck on an + // empty "Cancelled" slice after reloading with a fresh query. + if (activeStatusFilter !== 'all' && !purchases.some(p => { + const s = normalizeStatus(p).toLowerCase(); + if (activeStatusFilter === 'pending') return s === 'pending' || s === 'notified' || isInFlightStatus(s); + if (activeStatusFilter === 'completed') return s === 'completed' || s === 'partially_completed' || !p.status; + // Migration 000089: the Cancelled chip key is 'cancelled' (British, kept + // stable for URL/state compatibility) but it must surface BOTH spellings + // during the expand-contract deploy window so new 'canceled' rows aren't + // hidden from the filter. + if (activeStatusFilter === 'cancelled') return s === 'cancelled' || s === 'canceled'; + return s === activeStatusFilter; + })) { + activeStatusFilter = 'all'; + } + + if (!purchases || purchases.length === 0) { + container.innerHTML = '

No purchase history found for the selected period.

'; + return; + } + + const statusFiltered = purchases.filter(p => { + if (activeStatusFilter === 'all') return true; + const s = normalizeStatus(p).toLowerCase(); + if (activeStatusFilter === 'pending') return s === 'pending' || s === 'notified' || isInFlightStatus(s); + if (activeStatusFilter === 'completed') return s === 'completed' || s === 'partially_completed' || !p.status; + // Migration 000089: surface BOTH spellings under the Cancelled chip during + // the expand-contract deploy window (see the equivalent guard above). + if (activeStatusFilter === 'cancelled') return s === 'cancelled' || s === 'canceled'; + return s === activeStatusFilter; + }); + + // Apply the per-column filters AFTER the status chip filter so the + // categorical popover lists only values present in the active status + // slice (e.g. filtering by "Failed" only shows providers/services that + // have failed rows). + const colFilters = state.getPurchaseHistoryColumnFilters(); + const visible = applyPurchaseHistoryColumnFilters(statusFiltered, colFilters); + + const tableRows = visible.map(p => { + const statusCell = (() => { + const badge = statusBadgeHTML(normalizeStatus(p)); + const s = normalizeStatus(p).toLowerCase(); + if ((s === 'pending' || s === 'notified') && p.approver) { + return `${badge}
awaiting approval from ${escapeHtml(p.approver)}
`; + } + if (p.status_description) { + return `${badge}
${escapeHtml(p.status_description)}
`; + } + return badge; + })(); + const execIdAttr = p.purchase_id ? ` data-execution-id="${escapeHtmlAttr(p.purchase_id)}"` : ''; + const planCellContent = renderActionCell(p); + const amortize = state.getAmortizeUpfront(); + const rawMonthly = p.monthly_cost != null ? p.monthly_cost : null; + const displayMonthly = (rawMonthly != null && amortize) + ? amortizedMonthly(rawMonthly, p.upfront_cost, p.term) + : rawMonthly; + const monthlyCostCell = displayMonthly != null + ? formatCurrency(displayMonthly) + : '-'; + return ` + + ${statusCell} + ${formatDate(p.timestamp)} + ${providerCell(p)} + ${escapeHtml(p.service)} + ${escapeHtml(p.resource_type)} + ${escapeHtml(p.region)} + ${p.count} + ${formatTerm(p.term)} + ${formatCurrency(p.upfront_cost)} + ${monthlyCostCell} + ${formatCurrency(p.estimated_savings)} + ${planCellContent} + + `; + }).join(''); + + const amortize = state.getAmortizeUpfront(); + const monthlyColHeader = amortize ? 'Monthly Cost (amortized)' : 'Monthly Cost'; + const fbtn = (col: PurchaseHistoryColumnId): string => renderHistoryFilterButton( + col, PURCHASE_HISTORY_LABELS[col], colFilters[col] != null, + ); + const markup = ` + ${buildStatusChipRowHTML(purchases, activeStatusFilter)} + + + + + + + + + + + + + + + + + + + ${tableRows} + +
StatusDateProvider${fbtn('provider')}Service${fbtn('service')}Type${fbtn('resource_type')}Region${fbtn('region')}Count${fbtn('count')}Term${fbtn('term')}Upfront Cost${fbtn('upfront_cost')}${escapeHtml(monthlyColHeader)}Monthly Savings${fbtn('savings')}Plan
+ `; + container.innerHTML = markup; + + // Mount the amortize checkbox into the controls area (idempotent). + mountAmortizeCheckbox('history-controls', 'history-amortize-checkbox'); + wirePurchaseHistoryFilterButtons(container, statusFiltered); + + container.querySelectorAll('.status-chip[data-history-status]').forEach(btn => { + btn.addEventListener('click', () => { + const next = btn.dataset['historyStatus'] as StatusFilter | undefined; + if (!next || next === activeStatusFilter) return; + activeStatusFilter = next; + renderHistoryList(lastPurchases); + }); + }); + + wireRowActionHandlers(container); + + // Scroll + flash the deep-link target if the URL hash carries a + // ?execution=. The suppression badge on the Recommendations + // view links here so the user lands on the relevant row without + // scrolling through the whole list. + applyExecutionDeepLink(); +} + +// wireRowActionHandlers binds the Approve / Cancel / Retry click handlers +// against the buttons inside `container`. Scoped to the container (NOT +// the document) so multiple mounted lists (Purchase History table + +// Approval queue card) don't cross-fire: clicking the queue's Approve +// button binds and dispatches against the queue's button instance only. +// +// All three handlers terminate with a `loadHistory()` reload on success, +// which re-renders BOTH the history table and the queue card from the +// same fetched dataset. That means a successful approve from the queue +// card removes the row from BOTH views in one shot. +function wireRowActionHandlers(container: HTMLElement): void { + // Wire the inline Approve button on pending rows the current session + // may approve (issue #286). One flow: confirmDialog → POST /approve + // (no token — bearer-session auth on apiRequest) → reload + toast. + // Backend may still 409 on a status race (concurrent cancel landed + // first); the catch surfaces the structured detail. + container.querySelectorAll('.history-approve-btn[data-approve-id]').forEach(btn => { + btn.addEventListener('click', async () => { + const id = btn.dataset['approveId']; + if (!id) return; + // Issue #374: show the per-rec details (service / engine / + // resource / region / count / term + payment / costs) in the + // modal so the user has informed consent before authorising a + // financial commitment. buildApprovalDetailsBody falls back to + // the legacy text sentence if the GET fails. + const detailsBody = await buildApprovalDetailsBody(id); + const ok = await confirmDialog({ + title: 'Approve this pending purchase?', + body: detailsBody, + confirmLabel: 'Approve purchase', + destructive: false, + }); + if (!ok) return; + // Issue #286 + CR pass: Approve and Cancel can render together on + // the same row, so disabling only the clicked button leaves the + // sibling clickable while we await the API. Disable BOTH on + // either click and re-enable both on failure — a successful + // approve triggers a full history reload that re-renders the + // row, so the row-action sibling state doesn't matter on the + // happy path. + const rowActions = sameRowActions(btn); + rowActions.forEach((b) => { b.disabled = true; }); + try { + await api.approvePurchase(id); + } catch (approveError) { + console.error('Failed to approve pending purchase:', approveError); + const err = approveError as Error; + showToast({ message: `Failed to approve: ${err.message || 'unknown error'}`, kind: 'error' }); + rowActions.forEach((b) => { b.disabled = false; }); + return; + } + showToast({ message: 'Purchase approved', kind: 'success', timeout: 5_000 }); + try { + await loadHistory(); + } catch (reloadError) { + console.error('Failed to reload history after approve:', reloadError); + } + }); + }); + + // Wire the inline Cancel button on pending/notified rows the current + // session may cancel (issue #46). confirmDialog → POST → reload. The + // backend remains the security boundary; the canCancelPendingRow + // helper above is a UX gate that hides the button when the call would + // 403, but a stale cache could still surface a 403 — handle it the + // same way as any other failure. + // + // The cancel POST and the follow-up reload are split into separate + // try/catch blocks: a successful cancel + failed reload must not show + // a "Failed to cancel" toast (the purchase IS cancelled), and the + // user should see success-toast first so they don't think their + // click was lost while we re-fetch the table. + container.querySelectorAll('.history-cancel-btn[data-cancel-id]').forEach(btn => { + btn.addEventListener('click', async () => { + const id = btn.dataset['cancelId']; + if (!id) return; + const ok = await confirmDialog({ + title: 'Cancel this pending purchase?', + body: 'This will permanently abort the approval flow. The pending email approval link will stop working. This action cannot be undone.', + confirmLabel: 'Cancel purchase', + destructive: true, + }); + if (!ok) return; + // Symmetric with the Approve handler above: disable both row + // actions while the API is in flight (CR pass on PR #299). + const rowActions = sameRowActions(btn); + rowActions.forEach((b) => { b.disabled = true; }); + try { + await api.cancelPurchase(id); + } catch (cancelError) { + console.error('Failed to cancel pending purchase:', cancelError); + const err = cancelError as Error; + showToast({ message: `Failed to cancel: ${err.message || 'unknown error'}`, kind: 'error' }); + rowActions.forEach((b) => { b.disabled = false; }); + return; + } + // Cancel succeeded — surface success regardless of whether the + // refresh works. A reload failure leaves the row in its previous + // pending state on screen (stale-but-correct: the next manual + // reload corrects it). + showToast({ message: 'Purchase cancelled', kind: 'success', timeout: 5_000 }); + try { + await loadHistory(); + } catch (reloadError) { + console.error('Failed to reload history after cancel:', reloadError); + // Don't downgrade the success toast; loadHistory's own catch + // already paints an error message into the list area. + } + }); + }); + + // Wire the inline Retry button on failed rows the current session + // may retry (issue #47). Two flows: + // * normal: confirmDialog → POST /retry → reload + toast. + // * over-threshold: confirmDialog with stronger warning → POST + // with ?force=true → reload + toast. + // The backend may still 409 with an ops_hint or threshold response; + // the catch block surfaces the structured detail when present. + container.querySelectorAll('.history-retry-btn[data-retry-id]').forEach(btn => { + btn.addEventListener('click', async () => { + const id = btn.dataset['retryId']; + if (!id) return; + const overThreshold = btn.classList.contains('history-retry-over-threshold'); + const ok = await confirmDialog({ + title: overThreshold ? 'Retry past threshold?' : 'Retry this failed purchase?', + body: overThreshold + ? 'The same recommendations have already failed multiple times. Are you sure you want to retry again? This may not succeed.' + : 'This will create a new purchase execution from the same recommendations. The original failed row will be linked to the new attempt.', + confirmLabel: overThreshold ? 'Retry anyway' : 'Retry purchase', + destructive: false, + }); + if (!ok) return; + btn.disabled = true; + let retryResult: Awaited>; + try { + retryResult = await api.retryPurchase(id, overThreshold ? { force: true } : undefined); + } catch (retryError) { + console.error('Failed to retry purchase:', retryError); + // Surface structured retry hints from the backend (issue #47): + // * ops_hint — operator-actionable reason; takes priority + // * retry_attempt_n + threshold — soft-block message + // * else — fall back to the raw error message + const err = retryError as Error & { details?: Record }; + const opsHint = typeof err.details?.['ops_hint'] === 'string' ? err.details['ops_hint'] : ''; + const retryAttemptN = typeof err.details?.['retry_attempt_n'] === 'number' ? err.details['retry_attempt_n'] : undefined; + const threshold = typeof err.details?.['threshold'] === 'number' ? err.details['threshold'] : undefined; + let detailMessage = ''; + if (opsHint) { + detailMessage = opsHint; + } else if (retryAttemptN != null && threshold != null) { + detailMessage = `already retried ${retryAttemptN} times (threshold ${threshold}) — confirm the override prompt to force`; + } + const finalMessage = detailMessage || err.message || 'unknown error'; + showToast({ message: `Failed to retry: ${finalMessage}`, kind: 'error' }); + btn.disabled = false; + return; + } + // Gate the toast on the approval email outcome reported by the backend. + // email_sent===false is an explicit failure signal that overrides any + // status-based inference; show a warning even when status==='pending'. + // email_sent===true or a pending/notified status (with email_sent absent) + // means the approval request is in the queue. + const emailExplicitlyFailed = retryResult.email_sent === false; + const emailOk = !emailExplicitlyFailed && ( + retryResult.email_sent === true + || retryResult.status === 'pending' + || retryResult.status === 'notified' + ); + if (emailOk) { + showToast({ message: 'Purchase request sent for approval', kind: 'success', timeout: 5_000 }); + } else { + showToast({ message: 'Retry created but approval email failed - check your notification settings', kind: 'warning', timeout: 8_000 }); + } + try { + await loadHistory(); + } catch (reloadError) { + console.error('Failed to reload history after retry:', reloadError); + } + }); + }); + + // Wire the inline Revoke button on completed Azure rows within the + // free-cancel window (issue #290). confirmDialog -> POST -> reload. + // The backend is the security boundary; canRevokeCompletedRow is a + // UX gate that hides the button when the call would fail, but a stale + // cache can still surface a 4xx -- handle it like any other failure. + container.querySelectorAll('.history-revoke-btn[data-revoke-id]').forEach(btn => { + btn.addEventListener('click', async () => { + const id = btn.dataset['revokeId']; + if (!id) return; + const ok = await confirmDialog({ + title: 'Revoke this purchase within the free-cancel window?', + body: 'This will request an Azure reservation return. The charge will be refunded if the request is within the 7-day window. This action cannot be undone.', + confirmLabel: 'Revoke purchase', + destructive: true, + }); + if (!ok) return; + const rowActions = sameRowActions(btn); + rowActions.forEach((b) => { b.disabled = true; }); + try { + await api.revokePurchase(id); + } catch (revokeError) { + console.error('Failed to revoke purchase:', revokeError); + const err = revokeError as Error; + showToast({ message: `Failed to revoke: ${err.message || 'unknown error'}`, kind: 'error' }); + rowActions.forEach((b) => { b.disabled = false; }); + return; + } + showToast({ message: 'Purchase revocation submitted', kind: 'success', timeout: 5_000 }); + try { + await loadHistory(); + } catch (reloadError) { + console.error('Failed to reload history after revoke:', reloadError); + } + }); + }); + + // Wire Sell on Marketplace button (issue #292). + // Flow: pricing/schedule modal (RI summary + default price + 12% fee) → + // user confirms → createMarketplaceListing. We never skip the pricing + // modal (CR finding: going straight from confirmDialog to the API call + // denies the user informed consent about the price and fee). + container.querySelectorAll('.history-marketplace-sell-btn[data-marketplace-sell-id]').forEach(btn => { + btn.addEventListener('click', async () => { + const id = btn.dataset['marketplaceSellId']; + if (!id) return; + + // Look up the purchase record so we can show a meaningful price summary. + const purchase = lastPurchases.find(p => p.purchase_id === id); + + // Build a pricing modal body with RI summary and fee breakdown. + const bodyEl = document.createElement('div'); + bodyEl.className = 'marketplace-pricing-modal-body'; + + if (purchase) { + // purchase_history.term is stored in YEARS (1 or 3); convert to months + // before computing the remaining term and residual so the price summary + // shown to the user reflects real remaining value rather than ~1/3 of it + // (a 3-year RI was previously treated as 3 months). Mirrors the + // row.Term * 12 conversion in internal/api/handler_marketplace.go. + const termYears = typeof purchase.term === 'number' ? purchase.term : Number(purchase.term) || 0; + const termMonths = termYears > 0 ? termYears * 12 : 0; + const purchaseMs = new Date(purchase.timestamp).getTime(); + const elapsedMonths = Number.isFinite(purchaseMs) + ? (Date.now() - purchaseMs) / (1000 * 60 * 60 * 24 * 30.4375) + : 0; + const remainingMonths = Math.max(0, Math.round(termMonths - elapsedMonths)); + const upfront = purchase.upfront_cost ?? 0; + const count = purchase.count > 0 ? purchase.count : 1; + // Mirror marketplaceResidualPerUnit + resolveMarketplacePriceSchedule's + // default branch in internal/api/handler_marketplace.go EXACTLY, so + // this preview can never diverge from what the backend actually lists: + // - upfront-only: recurring (monthly) cost is deliberately excluded + // because the buyer assumes the recurring obligation post-transfer; + // - per instance: upfront_cost is the row total for `count` instances, + // but the AWS Marketplace price is per instance, so divide by count; + // - prorated: the upfront residual is scaled by remaining/original + // term (a 36-month RI at month 6 retains only 30/36 of its value); + // - zero when unpriceable: a no-upfront RI (upfront <= 0) or an + // unknown term (termMonths <= 0) has no residual to prorate, which + // is exactly when the backend now rejects the default schedule + // with an error instead of silently listing at $0. + const perUnitResidual = termMonths > 0 && upfront > 0 + ? (upfront * (remainingMonths / termMonths)) / count + : 0; + const listPricePerUnit = perUnitResidual * AWS_MARKETPLACE_BUYER_DISCOUNT; + const listPriceTotal = listPricePerUnit * count; + const netProceedsTotal = listPriceTotal * AWS_MARKETPLACE_NET_FACTOR; + + const summaryEl = document.createElement('dl'); + summaryEl.className = 'marketplace-pricing-summary'; + const addRow = (label: string, value: string): void => { + const dt = document.createElement('dt'); + dt.textContent = label; + const dd = document.createElement('dd'); + dd.textContent = value; + summaryEl.appendChild(dt); + summaryEl.appendChild(dd); + }; + addRow('RI ID', id); + addRow('Region', purchase.region || '-'); + addRow('Resource type', purchase.resource_type || '-'); + addRow('Remaining term', remainingMonths === 1 ? '1 month' : `${remainingMonths} months`); + if (listPricePerUnit > 0) { + addRow('Default list price', count > 1 + ? `${formatCurrency(listPricePerUnit)}/unit (${formatCurrency(listPriceTotal)} total for ${count} units)` + : formatCurrency(listPriceTotal)); + addRow(`AWS fee (${AWS_MARKETPLACE_FEE_PERCENT}%)`, formatCurrency(listPriceTotal * (AWS_MARKETPLACE_FEE_PERCENT / 100))); + addRow('Estimated net proceeds', formatCurrency(netProceedsTotal)); + } else { + // No default price can be computed (no upfront cost or unknown + // term) -- listing will be rejected server-side unless a custom + // price_schedule is supplied. Say so instead of showing a + // misleading $0 or fabricated price. + addRow('Default list price', 'unavailable (no upfront cost or unknown term)'); + } + bodyEl.appendChild(summaryEl); + } + + const noteEl = document.createElement('p'); + noteEl.className = 'marketplace-pricing-note'; + noteEl.textContent = `AWS charges a ${AWS_MARKETPLACE_FEE_PERCENT}% transaction fee on proceeds. The default schedule prices the listing at ${(1 - AWS_MARKETPLACE_BUYER_DISCOUNT) * 100}% below remaining value. You can adjust pricing by contacting your administrator or modifying the schedule via the API. This action cannot be undone without cancelling the listing.`; + bodyEl.appendChild(noteEl); + + const ok = await confirmDialog({ + title: 'List this RI on the AWS Marketplace?', + body: bodyEl, + confirmLabel: 'Confirm listing', + destructive: false, + }); + if (!ok) return; + + const rowActions = sameRowActions(btn); + rowActions.forEach(b => { b.disabled = true; }); + try { + await api.createMarketplaceListing(id); + } catch (sellError) { + console.error('Failed to list RI on Marketplace:', sellError); + const err = sellError as Error; + showToast({ message: `Failed to list on Marketplace: ${err.message || 'unknown error'}`, kind: 'error' }); + rowActions.forEach(b => { b.disabled = false; }); + return; + } + showToast({ message: 'RI listed on Marketplace successfully', kind: 'success', timeout: 5_000 }); + try { + await loadHistory(); + } catch (reloadError) { + console.error('Failed to reload history after Marketplace listing:', reloadError); + } + }); + }); + + // Wire Cancel listing button (issue #292) + container.querySelectorAll('.history-marketplace-cancel-btn[data-marketplace-cancel-id]').forEach(btn => { + btn.addEventListener('click', async () => { + const id = btn.dataset['marketplaceCancelId']; + if (!id) return; + const ok = await confirmDialog({ + title: 'Cancel this Marketplace listing?', + body: 'This will remove the listing from the AWS Marketplace. Any existing buyer negotiations will be cancelled. You can relist the RI at any time.', + confirmLabel: 'Cancel listing', + destructive: true, + }); + if (!ok) return; + const rowActions = sameRowActions(btn); + rowActions.forEach(b => { b.disabled = true; }); + try { + await api.cancelMarketplaceListing(id); + } catch (cancelError) { + console.error('Failed to cancel Marketplace listing:', cancelError); + const err = cancelError as Error; + showToast({ message: `Failed to cancel listing: ${err.message || 'unknown error'}`, kind: 'error' }); + rowActions.forEach(b => { b.disabled = false; }); + return; + } + showToast({ message: 'Marketplace listing cancelled', kind: 'success', timeout: 5_000 }); + try { + await loadHistory(); + } catch (reloadError) { + console.error('Failed to reload history after Marketplace cancel:', reloadError); + } + }); + }); +} + +// isPendingRow returns true when a history row represents a purchase +// awaiting approval. The Approval queue card filters on this predicate. +// Mirrors the badge / button-eligibility logic elsewhere in this file: +// "pending" is the freshly-created state, "notified" is the post-SES +// state; both render the same Approve / Cancel affordances. +function isPendingRow(p: HistoryPurchase): boolean { + const s = (p.status || '').toLowerCase(); + return s === 'pending' || s === 'notified'; +} + +// renderApprovalQueue paints the pending-approval card at the top of the +// Purchases tab (issue #340 sub-task). Filters the full history slice +// down to pending|notified rows and renders a compact action-focused +// table. When the filtered slice is empty, the card shows a friendly +// "No pending approvals" message so the section stays visible (stable +// layout, screen-reader-discoverable) without rendering an empty table. +// +// The row-action buttons reuse renderPendingActionButtons and the click +// handlers are wired by the shared wireRowActionHandlers helper, so +// approving from the queue card runs the exact same flow as approving +// from the history table (confirmDialog → API → toast → reload). The +// reload re-renders both views from one fetch, which removes the +// approved row from BOTH lists in one shot. +// --------------------------------------------------------------------------- +// Per-column filter wiring for the Approval Queue table. +// +// The queue scope is already narrow (pending|notified rows only); column +// filters add inline narrowing on the queue's own columns. As with the +// Purchase History wiring, Status is excluded because the queue's row set +// is status-defined and the parent loadHistory loop is the authoritative +// status source. +// +// Numeric extractors round to 0 decimal places (CURRENCY_DEFAULT_DIGITS) +// so a "$X" filter targets the same value the cell renders. +// --------------------------------------------------------------------------- + +const APPROVAL_QUEUE_NUMERIC_COLUMNS: ReadonlySet = new Set([ + 'count', 'monthly_cost', 'upfront_cost', 'savings', +]); + +const APPROVAL_QUEUE_LABELS: Record = { + provider: 'Provider', + account: 'Account', + service: 'Service', + term: 'Term', + payment: 'Payment', + created_by: 'Created by', + count: 'Count', + monthly_cost: 'Monthly Cost', + upfront_cost: 'Upfront Cost', + savings: 'Monthly Savings', +}; + +function approvalQueueCategoricalCellValue( + p: HistoryPurchase, + col: ApprovalQueueColumnId, +): string { + switch (col) { + case 'provider': return p.provider ?? ''; + case 'account': return p.account_id ?? ''; + case 'service': return p.service ?? ''; + case 'term': return p.term == null ? '' : String(p.term); + case 'payment': return p.payment ?? ''; + case 'created_by': return p.created_by_user_email ?? p.created_by_user_id ?? ''; + case 'count': + case 'monthly_cost': + case 'upfront_cost': + case 'savings': return ''; + } +} + +function approvalQueueNumericCellValue( + p: HistoryPurchase, + col: ApprovalQueueColumnId, +): number { + switch (col) { + case 'count': return p.count ?? 0; + // Return NaN for null monthly_cost so numeric predicates (e.g. "= 0") + // don't match rows where the provider didn't report a monthly cost. + case 'monthly_cost': return p.monthly_cost == null ? Number.NaN : p.monthly_cost; + case 'upfront_cost': return p.upfront_cost ?? 0; + case 'savings': return p.estimated_savings ?? 0; + case 'provider': + case 'account': + case 'service': + case 'term': + case 'payment': + case 'created_by': return Number.NaN; + } +} + +export function applyApprovalQueueColumnFilters( + purchases: readonly HistoryPurchase[], + filters: state.ApprovalQueueColumnFilters, +): HistoryPurchase[] { + return applyColumnFilters( + purchases, + filters, + { + categorical: approvalQueueCategoricalCellValue, + numeric: (p, col) => roundForDisplay(approvalQueueNumericCellValue(p, col)), + }, + ); +} + +function approvalQueueDistinctValues( + purchases: readonly HistoryPurchase[], + column: ApprovalQueueColumnId, +): string[] { + const seen = new Set(); + for (const p of purchases) { + seen.add(approvalQueueCategoricalCellValue(p, column)); + } + return Array.from(seen).sort((a, b) => { + if (a === '' && b !== '') return -1; + if (a !== '' && b === '') return 1; + return a.localeCompare(b); + }); +} + +function approvalQueueDisplayLabel( + column: ApprovalQueueColumnId, + value: string, +): string { + if (value === '') return '(empty)'; + if (column === 'term') { + const n = Number(value); + return Number.isFinite(n) ? formatTerm(n) : value; + } + if (column === 'account') { + // Account cells render via getAccountName() — mirror the same display. + return getAccountName(value); + } + return value; +} + +function wireApprovalQueueFilterButtons( + container: HTMLElement, + // Source for the popover's distinct-values list — the pre-column-filter + // pending slice. Same reasoning as Purchase History: the popover must + // list every value that exists in the broader (un-narrowed) set so + // the user can re-check a value after unchecking it. + sourceRows: readonly HistoryPurchase[], +): void { + container.querySelectorAll('.history-column-filter-btn').forEach((btn) => { + const column = btn.dataset['column'] as ApprovalQueueColumnId | undefined; + if (!column) return; + btn.addEventListener('click', (e) => { + e.stopPropagation(); + const isNumeric = APPROVAL_QUEUE_NUMERIC_COLUMNS.has(column); + const filters = state.getApprovalQueueColumnFilters(); + openHistoryColumnPopover({ + column, + anchor: btn, + currentFilter: filters[column], + headerLabel: APPROVAL_QUEUE_LABELS[column], + kind: isNumeric ? 'numeric' : 'categorical', + distinctValues: isNumeric ? undefined : approvalQueueDistinctValues(sourceRows, column), + displayLabel: (v) => approvalQueueDisplayLabel(column, v), + onCommit: (filter) => { + state.setApprovalQueueColumnFilter(column, filter); + renderApprovalQueue(lastPendingForQueue); + }, + }); + }); + }); +} + +// Cache of the last pre-column-filter pending list so the popover-driven +// re-render path can rebuild the table without re-fetching. +let lastPendingForQueue: HistoryPurchase[] = []; + +export function renderApprovalQueue(purchases: HistoryPurchase[]): void { + const container = document.getElementById('purchases-approval-queue'); + if (!container) return; + + const pending = (purchases || []).filter(isPendingRow); + lastPendingForQueue = pending; + + if (pending.length === 0) { + container.innerHTML = '

No pending approvals.

'; + return; + } + + const colFilters = state.getApprovalQueueColumnFilters(); + const visible = applyApprovalQueueColumnFilters(pending, colFilters); + + const rows = visible.map(p => { + const actions = renderPendingActionButtons(p); + const actionsCell = actions || '-'; + // Show email when resolved; fall back to UUID so the cancel-own gate still + // has something human-readable to show. Fall back to "-" for scheduler rows. + const createdBy = p.created_by_user_email + ? escapeHtml(p.created_by_user_email) + : p.created_by_user_id + ? escapeHtml(p.created_by_user_id) + : '-'; + const accountCell = p.account_id + ? escapeHtml(getAccountName(p.account_id)) + : '-'; + const termCell = p.term ? escapeHtml(formatTerm(p.term)) : '-'; + const paymentCell = p.payment ? escapeHtml(p.payment) : '-'; + const amortize = state.getAmortizeUpfront(); + const rawMonthly = p.monthly_cost != null ? p.monthly_cost : null; + const displayMonthly = (rawMonthly != null && amortize) + ? amortizedMonthly(rawMonthly, p.upfront_cost, p.term) + : rawMonthly; + const monthlyCostCell = displayMonthly != null + ? formatCurrency(displayMonthly) + : '-'; + const execIdAttr = p.purchase_id ? ` data-execution-id="${escapeHtmlAttr(p.purchase_id)}"` : ''; + return ` + + ${formatDate(p.timestamp)} + ${accountCell} + ${providerCell(p)} + ${escapeHtml(p.service)} + ${p.count} + ${termCell} + ${paymentCell} + ${monthlyCostCell} + ${formatCurrency(p.upfront_cost)} + ${formatCurrency(p.estimated_savings)} + ${createdBy} + ${actionsCell} + + `; + }).join(''); + + const amortize = state.getAmortizeUpfront(); + const monthlyColHeader = amortize ? 'Monthly Cost (amortized)' : 'Monthly Cost'; + // monthlyColHeader is a hardcoded constant string (no user data), so + // interpolating it directly into the template is safe. + const fbtn = (col: ApprovalQueueColumnId): string => renderHistoryFilterButton( + col, APPROVAL_QUEUE_LABELS[col], colFilters[col] != null, + ); + container.innerHTML = ` + + + + + + + + + + + + + + + + + + + ${rows} + +
DateAccount${fbtn('account')}Provider${fbtn('provider')}Service${fbtn('service')}Count${fbtn('count')}Term${fbtn('term')}Payment${fbtn('payment')}${monthlyColHeader}${fbtn('monthly_cost')}Upfront Cost${fbtn('upfront_cost')}Monthly Savings${fbtn('savings')}Created by${fbtn('created_by')}Actions
+ `; + + // Mount the amortize checkbox into the approval queue section (idempotent). + mountAmortizeCheckbox('purchases-approval-queue-section', 'approval-queue-amortize-checkbox'); + + wireRowActionHandlers(container); + wireApprovalQueueFilterButtons(container, pending); +} diff --git a/frontend/src/index.html b/frontend/src/index.html new file mode 100644 index 000000000..cb019881a --- /dev/null +++ b/frontend/src/index.html @@ -0,0 +1,1348 @@ + + + + + + + + + + + + + CUDly - Cloud Commitment Optimizer + + + Skip to main content +
+
+ + +
+ +

CUDly

+
+
+
+ + + API Docs + + +
+
+ + +
+ +
+ +
+
+
+
+

Savings over time

+
+ + + + +
+
+ + +
+
+

Potential savings range per service

+ + +
+
+

Upcoming Scheduled Purchases

+
+
+
+ + +
+
+ +
+ +
+
+ + +
+
+

Purchase Plans

+ + +
+
+ +
+

Planned Purchases

+

Individual purchases scheduled from your plans. You can run, pause, edit, or delete each purchase.

+
+
+
+ + +
+ + + + +
+

Approval queue

+

Purchases awaiting approval. Review and approve or cancel each one. The same actions are available from the Purchase History table below.

+
+
+ + +
+
+

Savings History

+
+ + + +
+
+ + +
+
+

Period Savings

+

$0.00

+

shown in monthly equivalents

+
+
+

Avg Monthly Savings

+

$0.00/mo

+
+
+

Peak Savings

+

$0.00/mo

+
+
+ + +
+ +
+ + + +
+ + +
+

Purchase History

+
+
+ + + + +
+
+
+
+
+
+ + +
+
+ + + +
+ + +
+
+
+

Active commitments

+
+ +
+
+

All non-expired Reserved Instances, Savings Plans, and Compute Units committed across your registered accounts. Soonest-expiring first.

+
+ +
+
+
+
+ + + + + + +
+ + +
+
+ + + + +
+
+

Global Configuration

+

Configure CUDly settings for commitment purchases across all cloud providers.

+
Loading settings...
+ + +
+ + + + + + + + + + + + +
+
+
+ + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + +
+ + diff --git a/frontend/src/index.ts b/frontend/src/index.ts new file mode 100644 index 000000000..962d0ba26 --- /dev/null +++ b/frontend/src/index.ts @@ -0,0 +1,77 @@ +/** + * CUDly - Cloud Commitment Optimizer Dashboard + * Main entry point + */ + +import './styles.css'; +import * as api from './api'; +import * as utils from './utils'; +import { init } from './app'; +import { refreshRecommendations } from './recommendations'; +import { openCreatePlanModal, openNewPlanModal, closePlanModal, closePurchaseModal } from './plans'; +import { resetSettings } from './settings'; +import { loadHistory } from './history'; +import { logout } from './auth'; +import { openCreateUserModal, closeUserModal } from './users/userModals'; +import { openCreateGroupModal, closeGroupModal, addPermission, closeDuplicateGroupModal, saveDuplicateGroup } from './groups/groupModals'; +import { initTopbarFilters } from './topbar-filters'; + +// Re-export for external use +export { api, utils }; + +// Import types for global window declarations +import './types'; + +// Set up global window functions for HTML onclick handlers +window.refreshRecommendations = refreshRecommendations; +window.openCreatePlanModal = openCreatePlanModal; +window.openNewPlanModal = openNewPlanModal; +window.closePlanModal = closePlanModal; +window.closePurchaseModal = closePurchaseModal; +window.resetSettings = resetSettings; +window.loadHistory = loadHistory; +window.logout = logout; +window.openCreateUserModal = openCreateUserModal; +window.closeUserModal = closeUserModal; +window.openCreateGroupModal = openCreateGroupModal; +window.closeGroupModal = closeGroupModal; +window.addPermission = addPermission; + +// Wire event listeners for buttons that previously used inline onclick +// (CSP blocks inline event handlers when script-src is 'self' without 'unsafe-inline') +document.addEventListener('DOMContentLoaded', () => { + document.getElementById('create-user-btn')?.addEventListener('click', openCreateUserModal); + document.getElementById('create-group-btn')?.addEventListener('click', openCreateGroupModal); + document.getElementById('close-user-modal-btn')?.addEventListener('click', closeUserModal); + document.getElementById('close-group-modal-btn')?.addEventListener('click', closeGroupModal); + document.getElementById('add-permission-btn')?.addEventListener('click', () => addPermission()); + document.getElementById('close-group-duplicate-modal-btn')?.addEventListener('click', closeDuplicateGroupModal); + document.getElementById('cancel-group-duplicate-btn')?.addEventListener('click', closeDuplicateGroupModal); + document.getElementById('group-duplicate-form')?.addEventListener('submit', (e) => void saveDuplicateGroup(e)); + + // Sidebar collapse toggle (issue #340). Persisted in localStorage so the + // user's preference survives reloads. + const sidebar = document.querySelector('.app-sidebar'); + const sidebarToggle = document.querySelector('.app-sidebar-toggle'); + if (sidebar && sidebarToggle) { + const STORAGE_KEY = 'cudly_sidebar_collapsed'; + if (localStorage.getItem(STORAGE_KEY) === '1') { + sidebar.classList.add('collapsed'); + sidebarToggle.setAttribute('aria-expanded', 'false'); + } + sidebarToggle.addEventListener('click', () => { + const isCollapsed = sidebar.classList.toggle('collapsed'); + sidebarToggle.setAttribute('aria-expanded', isCollapsed ? 'false' : 'true'); + localStorage.setItem(STORAGE_KEY, isCollapsed ? '1' : '0'); + }); + } + + // Global filter chips in the topbar (issue #344 T2). Replaces the + // per-section provider/account dropdowns that Home / Plans / Purchases + // used to carry; sections subscribe to state.subscribeProvider / + // subscribeAccount and reload themselves when the filter changes. + initTopbarFilters(); +}); + +// Initialize on page load +document.addEventListener('DOMContentLoaded', () => void init()); diff --git a/frontend/src/inventory.ts b/frontend/src/inventory.ts new file mode 100644 index 000000000..a63ddf4de --- /dev/null +++ b/frontend/src/inventory.ts @@ -0,0 +1,645 @@ +/** + * Inventory & Coverage section (issue #340 T4, #754). + * + * Umbrella section that folds the former top-level "RI Exchange" tab into + * a sub-section of a broader Inventory & Coverage view. Sub-sections: + * - active-commitments — per-commitment list backed by + * /api/inventory/commitments + * - coverage — per-provider coverage breakdowns backed by + * /api/inventory/coverage (issue #754) + * - ri-exchange — hosts the existing RI Exchange UI unchanged + */ + +import * as api from './api'; +import type { ProviderCoverageSection, CoverageServiceRow } from './api'; +import { loadRIExchange } from './riexchange'; +import { showSkeletonRows, teardownSkeleton } from './lib/skeleton'; +import { formatCurrency, formatDate, amortizedMonthly } from './utils'; +import * as state from './state'; +import { switchInventorySubTab } from './navigation'; + +type InventorySubSection = 'active-commitments' | 'coverage' | 'ri-exchange'; + +const SUB_SECTION_IDS: Record = { + 'active-commitments': 'inventory-active-commitments', + 'coverage': 'inventory-coverage', + 'ri-exchange': 'inventory-ri-exchange', +}; + +export const DEFAULT_INVENTORY_SUB_SECTION: InventorySubSection = 'active-commitments'; + +let currentSubSection: InventorySubSection | undefined; +let listenersWired = false; + +/** + * Type guard for the Inventory sub-section identifiers. Exported so the + * router (navigation.ts) can validate the `/inventory/` path + * segment without duplicating the closed set. + */ +export function isValidInventorySubSection(name: string): name is InventorySubSection { + return name === 'active-commitments' || name === 'coverage' || name === 'ri-exchange'; +} + +/** + * Show one sub-section, hide the others. Activates the matching sub-nav + * button and (for ri-exchange) triggers the RI exchange data load so the + * existing flow stays identical to its pre-#340 behaviour. + * + * This is the pure view switcher: it does NOT touch the URL. URL history + * (the `/inventory/` addressing from QA A.4) is owned by + * navigation.ts' switchInventorySubTab, mirroring how switchSettingsSubTab + * owns the `/admin/` history so a single counter (historyId) stays + * authoritative for the back/forward dirty-guard. + * + * Returns the resolved (validated, default-substituted) sub-section so the + * caller can reflect the same value in the URL. + */ +export function switchInventorySubSection(name: string): InventorySubSection { + const target: InventorySubSection = isValidInventorySubSection(name) + ? name + : DEFAULT_INVENTORY_SUB_SECTION; + + document.querySelectorAll('#inventory-tab .sub-tab-btn').forEach((btn) => { + const isActive = btn.dataset['invSubtab'] === target; + btn.classList.toggle('active', isActive); + btn.setAttribute('aria-selected', isActive ? 'true' : 'false'); + }); + + for (const key of Object.keys(SUB_SECTION_IDS) as InventorySubSection[]) { + const el = document.getElementById(SUB_SECTION_IDS[key]); + if (el) el.classList.toggle('hidden', key !== target); + } + + if (target === 'ri-exchange') { + void loadRIExchange(); + } else if (target === 'active-commitments') { + void loadActiveCommitments(); + } else if (target === 'coverage') { + void loadCoverageBreakdown(); + } + + currentSubSection = target; + return target; +} + +// ────────────────────────────────────────────── +// Active commitments +// ────────────────────────────────────────────── + +const ACTIVE_COMMITMENTS_LIST_ID = 'active-commitments-list'; +const ACTIVE_COMMITMENTS_REFRESH_BTN_ID = 'active-commitments-refresh-btn'; +const ACTIVE_COMMITMENTS_COLS = 11; + +/** + * Fetch and render the active-commitments table. Replaces #active-commitments-list + * children with a shimmer skeleton on entry, then either the rendered + * table or an empty-state / error paragraph on completion. Idempotent — + * safe to call on every sub-tab switch and on every refresh click. + * + * Reads the current provider/account chips from state so that changing a + * chip while on this sub-tab re-fetches with the new scope (issue #866). + */ +export async function loadActiveCommitments(): Promise { + const container = document.getElementById(ACTIVE_COMMITMENTS_LIST_ID); + if (!container) return; + + wireRefreshButton(); + wireAmortizeSubscription(); + + const provider = state.getCurrentProvider(); + const accountIDs = state.getCurrentAccountIDs(); + // account chip is single-select; forward only when exactly one is active. + const accountID = accountIDs.length === 1 ? accountIDs[0] : undefined; + + // 5 rows × 11 cols matches the rendered table shape (see + // renderActiveCommitmentsTable). The renderer wipes the container's + // children for a clean handoff from the skeleton. + showSkeletonRows(container, 5, ACTIVE_COMMITMENTS_COLS); + + try { + const commitments = await api.listActiveCommitments({ provider: provider || undefined, accountID }); + // Cache for amortize-toggle re-renders (issue #1112). + lastCommitments = commitments; + lastCommitmentsProvider = provider || undefined; + lastCommitmentsAccountID = accountID; + renderActiveCommitmentsTable(container, commitments, provider, accountID); + } catch (error) { + teardownSkeleton(container); + const err = error as Error; + renderErrorParagraph(container, `Failed to load active commitments: ${err.message}`); + } +} + +function wireRefreshButton(): void { + // Idempotency is tracked on the element itself rather than a + // module-level flag — when the section is re-rendered (e.g. between + // tests, or after a hot-swap in dev), the new button is unwired and + // a stale flag would block the rebind. The dataset marker travels + // with the element so it can't drift out of sync. + const btn = document.getElementById(ACTIVE_COMMITMENTS_REFRESH_BTN_ID); + if (!btn) return; + if (btn.dataset['wired'] === '1') return; + btn.addEventListener('click', () => { + void loadActiveCommitments(); + }); + btn.dataset['wired'] = '1'; +} + +function clearChildren(el: HTMLElement): void { + while (el.firstChild) el.removeChild(el.firstChild); +} + +function renderErrorParagraph(container: HTMLElement, message: string): void { + clearChildren(container); + const p = document.createElement('p'); + p.className = 'error'; + p.textContent = message; + container.appendChild(p); +} + +function renderEmptyParagraph(container: HTMLElement, message: string): void { + clearChildren(container); + const p = document.createElement('p'); + p.className = 'empty'; + p.textContent = message; + container.appendChild(p); +} + +/** + * Build a context-aware empty-state message for the active-commitments + * table. When chip filters are active the message names the scope so + * the user knows the result is filtered rather than globally empty. + */ +function buildActiveCommitmentsEmptyMessage(provider?: string, accountID?: string): string { + if (provider && accountID) { + return `No active commitments for provider "${provider}" and account ${accountID}.`; + } + if (provider) { + return `No active commitments for provider "${provider}".`; + } + if (accountID) { + return `No active commitments for account ${accountID}.`; + } + return 'No active commitments found across your registered accounts.'; +} + +/** + * Build a context-aware empty-state message for a per-provider Coverage card. + * When chip filters are active the message names the scope so the user knows + * the result is filtered rather than genuinely absent usage. + * + * @param providerLabel - Display name of the provider (e.g. "AWS"). + * @param activeProvider - The provider chip value, if one is set. + * @param activeAccountID - The account chip value, if exactly one is selected. + */ +function buildCoverageEmptyMessage( + providerLabel: string, + activeProvider?: string, + activeAccountID?: string, +): string { + if (activeProvider || activeAccountID) { + return `No ${providerLabel} usage for the selected account/filter.`; + } + return `No usage detected for ${providerLabel}.`; +} + +/** + * Render the per-commitment table into `container`. Empty list yields + * an inline `.empty` paragraph instead of an empty table so the user + * gets a real message ("no active commitments"), not a blank header. + * + * When a chip filter is active the empty-state message names the scope + * so the user understands the result is filtered, not globally empty. + * + * All text uses textContent / DOM construction — no innerHTML — to + * keep the section safe by default against any unescaped backend + * field (issue #340 XSS posture). + */ +function renderActiveCommitmentsTable( + container: HTMLElement, + commitments: api.InventoryCommitment[], + provider?: string, + accountID?: string, +): void { + if (!commitments || commitments.length === 0) { + const msg = buildActiveCommitmentsEmptyMessage(provider, accountID); + renderEmptyParagraph(container, msg); + return; + } + + clearChildren(container); + const table = document.createElement('table'); + + const thead = document.createElement('thead'); + const headerRow = document.createElement('tr'); + const amortize = state.getAmortizeUpfront(); + const monthlyLabel = amortize ? 'Monthly cost (amortized)' : 'Monthly cost'; + const headers = ['Provider', 'Account', 'Service', 'Resource type', 'Region', 'Count', 'Term', 'Payment', monthlyLabel, 'Monthly savings', 'Expires']; + for (const label of headers) { + const th = document.createElement('th'); + th.textContent = label; + headerRow.appendChild(th); + } + thead.appendChild(headerRow); + table.appendChild(thead); + + const tbody = document.createElement('tbody'); + for (const c of commitments) { + tbody.appendChild(buildCommitmentRow(c)); + } + table.appendChild(tbody); + + container.appendChild(table); + + // Mount the amortize toggle into the section-header-actions area (idempotent). + mountInventoryAmortizeCheckbox(); +} + +function buildCommitmentRow(c: api.InventoryCommitment): HTMLTableRowElement { + const tr = document.createElement('tr'); + + appendCell(tr, c.provider); + tr.appendChild(buildAccountCell(c)); + appendCell(tr, c.service); + appendCell(tr, c.resource_type ?? ''); + appendCell(tr, c.region); + appendCell(tr, String(c.count)); + appendCell(tr, `${c.term_years}y`); + appendCell(tr, c.payment_option ?? ''); + + // When amortize is on, fold the upfront cost over the term years. + const amortize = state.getAmortizeUpfront(); + let displayMonthly: number | null = c.monthly_cost; + if (displayMonthly != null && amortize) { + displayMonthly = amortizedMonthly(displayMonthly, c.upfront_cost, c.term_years); + } + appendCell(tr, displayMonthly != null ? formatCurrency(displayMonthly) : '—'); + + appendCell(tr, formatCurrency(c.estimated_savings)); + appendCell(tr, formatDate(c.end_date)); + + return tr; +} + +function appendCell(tr: HTMLTableRowElement, text: string): void { + const td = document.createElement('td'); + td.textContent = text; + tr.appendChild(td); +} + +/** + * Mount the "Amortize upfront over term" checkbox into the active-commitments + * section-header-actions area (idempotent). Wires to setAmortizeUpfront so + * the same localStorage key is shared with all other views (issue #1112). + */ +function mountInventoryAmortizeCheckbox(): void { + const actions = document.querySelector('#inventory-active-commitments .section-header-actions'); + if (!actions) return; + const checkboxId = 'inventory-amortize-checkbox'; + if (document.getElementById(checkboxId)) { + // Already mounted -- sync checked state in case another view changed it. + const cb = document.getElementById(checkboxId) as HTMLInputElement; + cb.checked = state.getAmortizeUpfront(); + return; + } + + const wrapper = document.createElement('label'); + wrapper.className = 'amortize-toggle-label'; + wrapper.htmlFor = checkboxId; + + const cb = document.createElement('input'); + cb.type = 'checkbox'; + cb.id = checkboxId; + cb.checked = state.getAmortizeUpfront(); + cb.addEventListener('change', () => { + state.setAmortizeUpfront(cb.checked); + }); + + wrapper.appendChild(cb); + wrapper.appendChild(document.createTextNode(' Amortize upfront over term')); + actions.appendChild(wrapper); +} + +function buildAccountCell(c: api.InventoryCommitment): HTMLTableCellElement { + const td = document.createElement('td'); + if (c.account_name) { + td.appendChild(document.createTextNode(c.account_name + ' ')); + const id = document.createElement('span'); + id.className = 'monospace'; + id.textContent = `(${c.account_id})`; + td.appendChild(id); + } else { + const id = document.createElement('span'); + id.className = 'monospace'; + id.textContent = c.account_id; + td.appendChild(id); + } + return td; +} + +// ────────────────────────────────────────────── +// Coverage breakdown +// ────────────────────────────────────────────── + +const COVERAGE_CONTAINER_ID = 'coverage-providers'; +const COVERAGE_REFRESH_BTN_ID = 'coverage-refresh-btn'; + +const PROVIDER_DISPLAY_NAMES: Record = { + aws: 'AWS', + azure: 'Azure', + gcp: 'GCP', +}; + +/** + * Fetch and render per-provider coverage breakdowns into #coverage-providers. + * Shows a skeleton on entry, then either the rendered sections or an error. + * Idempotent — safe to call on every sub-tab switch and on every refresh click. + * + * Reads the current provider/account chips from state so that changing a + * chip while on this sub-tab re-fetches with the new scope (issue #866). + */ +export async function loadCoverageBreakdown(): Promise { + const container = document.getElementById(COVERAGE_CONTAINER_ID); + if (!container) return; + + wireCoverageRefreshButton(); + + const provider = state.getCurrentProvider(); + const accountIDs = state.getCurrentAccountIDs(); + const accountID = accountIDs.length === 1 ? accountIDs[0] : undefined; + + // One skeleton row per known provider while loading. + showSkeletonRows(container, 3, 1); + + try { + const data = await api.getCoverageBreakdown({ provider: provider || undefined, accountID }); + renderCoverageBreakdown(container, data.providers, provider || undefined, accountID); + } catch (error) { + teardownSkeleton(container); + const err = error as Error; + renderErrorParagraph(container, `Failed to load coverage data: ${err.message}`); + } +} + +function wireCoverageRefreshButton(): void { + const btn = document.getElementById(COVERAGE_REFRESH_BTN_ID); + if (!btn) return; + if (btn.dataset['wired'] === '1') return; + btn.addEventListener('click', () => { + void loadCoverageBreakdown(); + }); + btn.dataset['wired'] = '1'; +} + +/** + * Render coverage sections. Each provider gets its own card. Providers + * with services=null show an empty-state paragraph. All text is set via + * textContent -- no innerHTML -- so no escaping helper is needed (XSS posture + * matches the active-commitments section per issue #340). + */ +function renderCoverageBreakdown( + container: HTMLElement, + providers: ProviderCoverageSection[], + activeProvider?: string, + activeAccountID?: string, +): void { + clearChildren(container); + + if (!providers || providers.length === 0) { + renderEmptyParagraph(container, 'No coverage data available.'); + return; + } + + for (const section of providers) { + container.appendChild(buildProviderSection(section, activeProvider, activeAccountID)); + } +} + +function buildProviderSection( + section: ProviderCoverageSection, + activeProvider?: string, + activeAccountID?: string, +): HTMLElement { + const card = document.createElement('section'); + card.className = 'card coverage-provider-card'; + + // Header row: provider name + overall coverage badge. + const header = document.createElement('div'); + header.className = 'section-header'; + + const providerLabel = PROVIDER_DISPLAY_NAMES[section.provider] ?? section.provider.toUpperCase(); + + const title = document.createElement('h3'); + title.textContent = providerLabel; + header.appendChild(title); + + if (section.overall_coverage_pct !== null && section.overall_coverage_pct !== undefined) { + const badge = document.createElement('span'); + badge.className = 'coverage-overall-badge'; + badge.textContent = `Overall: ${section.overall_coverage_pct.toFixed(1)}% covered`; + header.appendChild(badge); + } + card.appendChild(header); + + // Body: empty-state or per-service table. + if (!section.services || section.services.length === 0) { + const empty = document.createElement('p'); + empty.className = 'empty'; + empty.textContent = buildCoverageEmptyMessage(providerLabel, activeProvider, activeAccountID); + card.appendChild(empty); + return card; + } + + card.appendChild(buildServiceTable(section.services)); + return card; +} + +function buildServiceTable(rows: CoverageServiceRow[]): HTMLTableElement { + const table = document.createElement('table'); + table.className = 'coverage-service-table'; + + const thead = document.createElement('thead'); + const headerRow = document.createElement('tr'); + for (const label of ['Service', 'Covered/mo', 'On-demand gap/mo', 'Coverage %', 'Coverage bar']) { + const th = document.createElement('th'); + th.textContent = label; + if (label === 'Coverage bar') { + th.setAttribute('aria-label', 'Coverage bar'); + } + headerRow.appendChild(th); + } + thead.appendChild(headerRow); + table.appendChild(thead); + + const tbody = document.createElement('tbody'); + for (const row of rows) { + tbody.appendChild(buildServiceRow(row)); + } + table.appendChild(tbody); + return table; +} + +function buildServiceRow(row: CoverageServiceRow): HTMLTableRowElement { + const tr = document.createElement('tr'); + + appendCell(tr, row.service); + appendCell(tr, formatCurrency(row.covered_monthly)); + appendCell(tr, formatCurrency(row.on_demand_monthly)); + appendCell(tr, row.coverage_pct !== null && row.coverage_pct !== undefined + ? `${row.coverage_pct.toFixed(1)}%` + : 'N/A'); + // Bar cell: visual coverage indicator. + const barTd = document.createElement('td'); + barTd.className = 'coverage-bar-cell'; + if (row.coverage_pct !== null && row.coverage_pct !== undefined) { + const bar = document.createElement('div'); + bar.className = 'coverage-bar'; + // The bar carries no text, so screen readers need the value spelled out. + bar.setAttribute('role', 'img'); + bar.setAttribute('aria-label', `${row.coverage_pct.toFixed(1)}% covered`); + const fill = document.createElement('div'); + fill.className = 'coverage-bar-fill'; + // Clamp to [0, 100] so a misconfigured value can't overflow. + const pct = Math.min(100, Math.max(0, row.coverage_pct)); + fill.style.width = `${pct}%`; + bar.appendChild(fill); + barTd.appendChild(bar); + } else { + // No coverage figure: say so rather than leaving the cell blank, which + // reads as a rendering failure. A 0%-width bar is not an option -- it + // would claim we measured zero coverage. + const absent = document.createElement('span'); + absent.className = 'coverage-bar-absent'; + absent.textContent = 'N/A'; + barTd.appendChild(absent); + } + tr.appendChild(barTd); + + return tr; +} + +/** + * Wire sub-nav button clicks. Idempotent — calling this more than once + * doesn't double-bind handlers. + */ +function wireSubNavListeners(): void { + if (listenersWired) return; + const buttons = document.querySelectorAll('#inventory-tab .sub-tab-btn'); + if (buttons.length === 0) return; + buttons.forEach((btn) => { + btn.addEventListener('click', () => { + const name = btn.dataset['invSubtab'] ?? DEFAULT_INVENTORY_SUB_SECTION; + // Route through the router so the click both switches the view AND + // pushes /inventory/ (QA A.4), keeping history consistent + // with the Admin sub-tab flow. + switchInventorySubTab(name); + }); + }); + listenersWired = true; +} + +/** + * True when the Inventory & Coverage tab is the currently-visible top-level + * tab. The chip-subscription reload skips the fetch when this returns false + * so we don't burn an API call (or trigger a skeleton flash) for a section + * the user isn't looking at — switchTab('inventory') runs loadInventory() + * on next entry anyway, which re-fetches with the current chip state. + */ +function isInventoryTabActive(): boolean { + return document.getElementById('inventory-tab')?.classList.contains('active') === true; +} + +// Unsubscribe handles for the chip subscriptions. Re-assigned each time +// loadInventory() wires them so repeated tab-switches don't stack duplicate +// listeners -- the old pair is torn down before a new pair is registered. +let unsubscribeProvider: (() => void) | null = null; +let unsubscribeAccount: (() => void) | null = null; + +// Cache of the last-fetched commitments so the amortize toggle can +// re-render without a round-trip to the API (issue #1112). Also caches +// the provider/accountID context so the empty-state message stays accurate. +let lastCommitments: api.InventoryCommitment[] | null = null; +let lastCommitmentsProvider: string | undefined; +let lastCommitmentsAccountID: string | undefined; + +// Wired once; tracks whether the amortize subscriber has been registered +// for this module so repeated loadInventory() calls don't stack listeners. +let amortizeUnsubscribe: (() => void) | null = null; + +function wireAmortizeSubscription(): void { + if (amortizeUnsubscribe) return; // already wired + amortizeUnsubscribe = state.subscribeAmortizeUpfront(() => { + const container = document.getElementById(ACTIVE_COMMITMENTS_LIST_ID); + if (!container || lastCommitments === null) return; + renderActiveCommitmentsTable( + container, + lastCommitments, + lastCommitmentsProvider, + lastCommitmentsAccountID, + ); + }); +} + +/** + * Wire provider + account chip subscriptions (issue #866). + * + * Mirrors the pattern from PR #741 (Purchases) and PR #747 (Home): + * - Active-tab guard: only fire when the Inventory tab is active. + * - queueMicrotask coalescing: topbar-filters.ts fires BOTH the + * account-clear AND the provider-set subscribers synchronously on a + * single chip change. Without coalescing the two back-to-back fires + * would kick off two fetches; with it they collapse into one. + * - Re-check the active-tab guard inside the microtask: a tab switch + * between the chip change and the microtask flush cancels the + * now-unneeded fetch. + * + * Called from loadInventory() on every Inventory tab-switch. Tears down + * the previous subscription pair first so repeated switches don't stack + * duplicate listeners. + */ +function wireChipSubscriptions(): void { + // Tear down any existing subscriptions to avoid stacking on repeated + // tab-switches. + if (unsubscribeProvider) { unsubscribeProvider(); unsubscribeProvider = null; } + if (unsubscribeAccount) { unsubscribeAccount(); unsubscribeAccount = null; } + + let reloadQueued = false; + const scheduleReload = (): void => { + if (!isInventoryTabActive() || reloadQueued) return; + reloadQueued = true; + queueMicrotask(() => { + reloadQueued = false; + if (!isInventoryTabActive()) return; + if (currentSubSection === 'active-commitments') { + void loadActiveCommitments(); + } else if (currentSubSection === 'coverage') { + void loadCoverageBreakdown(); + } + }); + }; + + unsubscribeProvider = state.subscribeProvider(scheduleReload); + unsubscribeAccount = state.subscribeAccount(scheduleReload); +} + +/** + * Initialize the Inventory & Coverage section. Called by navigation.ts' + * switchTab when 'inventory' is selected, passing the sub-section parsed + * from the `/inventory/` URL path (QA A.4). + * + * The sub-section comes from the URL, not hidden session state: a fresh + * `/inventory` with no sub-segment lands on the default (active-commitments) + * and a `/inventory/` deep link lands on that sub-section. The + * switch is URL-driven (push: false) so re-entering the tab doesn't stack + * a redundant history entry on top of the one switchTab already pushed. + */ +export function loadInventory(subSection?: string): void { + wireSubNavListeners(); + wireChipSubscriptions(); + const target = subSection !== undefined && isValidInventorySubSection(subSection) + ? subSection + : DEFAULT_INVENTORY_SUB_SECTION; + // Pure view switch (no history push): switchTab already pushed the + // canonical /inventory/ URL when this tab was entered. + switchInventorySubSection(target); +} diff --git a/frontend/src/ladder.ts b/frontend/src/ladder.ts new file mode 100644 index 000000000..0cfccc47a --- /dev/null +++ b/frontend/src/ladder.ts @@ -0,0 +1,498 @@ +/** + * Commitment Laddering settings section (issue #1333 phase 3). + * + * Renders the per-account ladder config editor inside the + * #commitment-laddering-settings placeholder div that lives inside the + * Purchasing Settings panel. The section is flag-gated default-off: + * + * - Global kill-switch: global_config.laddering_enabled (bool) + * - Per-account: LadderConfig.enabled (bool) per (account, provider) pair + * + * No laddering engine runs fire until both flags are true. This PR (phase 3, + * schema foundation) wires the UI to read and write configs; actual engine + * invocation is in a later phase. + */ + +import * as api from './api'; +import { escapeHtml, formatDate } from './utils'; +import { showToast } from './toast'; +import { canAccess } from './permissions'; + +// ========================================== +// MODULE STATE +// ========================================== + +// Separate in-flight guards for the two independent save paths. Sharing one +// flag let the kill-switch toggle and the per-account modal save block/revert +// each other even though they touch unrelated backend endpoints. +let killSwitchSaveInFlight = false; +let configModalSaveInFlight = false; + +// Cached configs from the last successful load. +let cachedConfigs: api.LadderConfig[] = []; + +// ========================================== +// PUBLIC API +// ========================================== + +/** + * Initialize the Commitment Laddering settings section. + * + * Renders the HTML form into #commitment-laddering-settings and populates + * it with data from the API. Should be called from loadGlobalSettings after + * the global config response is available. + * + * The globalEnabled parameter is the current value of + * global_config.laddering_enabled so the kill-switch toggle reflects the + * persisted state without a second API call. + */ +export async function initLadderingSettings(globalEnabled: boolean): Promise { + const container = document.getElementById('commitment-laddering-settings'); + if (!container) return; + + container.innerHTML = renderLadderingSection(globalEnabled); + wireKillSwitchToggle(); + wireModalCloseButtons(); + + try { + cachedConfigs = await api.getLadderConfigs(); + renderConfigTable(cachedConfigs); + } catch (err) { + console.error('Failed to load ladder configs:', err); + const tableContainer = document.getElementById('ladder-configs-table-container'); + if (tableContainer) { + tableContainer.innerHTML = '

Failed to load per-account configurations.

'; + } + } + + if (canAccess('update', 'config')) { + document.getElementById('ladder-add-config-btn')?.addEventListener('click', () => openLadderConfigModal()); + } +} + +// ========================================== +// RENDERING +// ========================================== + +function renderLadderingSection(globalEnabled: boolean): string { + const canEdit = canAccess('update', 'config'); + const disabledAttr = canEdit ? '' : ' disabled'; + + return ` +
+ Commitment Laddering +

+ Commitment Laddering automatically ramps up cloud commitments towards a + target coverage level using a configurable ramp schedule. It is feature-gated: + enable the global kill-switch below, then enable individual accounts via the + per-account configuration table. +

+ +
+
+ + + Global kill-switch. Must be on before any per-account config can fire. +
+
+ +
+
+ +
+

Per-Account Configurations

+
+

Loading…

+
+ ${canEdit ? ` + ` : ''} +
+
+ + + +`; +} + +// Exported for unit testing (mirrors the apikeys.ts convention of exposing +// render/action helpers so tests can drive them directly). +export function renderConfigTable(configs: api.LadderConfig[]): void { + const container = document.getElementById('ladder-configs-table-container'); + if (!container) return; + + if (configs.length === 0) { + container.innerHTML = '

No per-account configurations yet. Click Add Account Config to create one.

'; + return; + } + + const canEdit = canAccess('update', 'config'); + const rows = configs.map(cfg => ` + + ${escapeHtml(cfg.cloud_account_id)} + ${escapeHtml(cfg.provider.toUpperCase())} + ${cfg.enabled ? 'Yes' : 'No'} + ${escapeHtml(cfg.mode === 'email_approval' ? 'Email Approval' : 'Auto Approve')} + ${escapeHtml(cfg.cadence === 'daily' ? 'Daily' : 'Weekly')} + ${cfg.target_coverage.toFixed(1)}% + ${cfg.updated_at ? escapeHtml(formatDate(cfg.updated_at)) : 'N/A'} + ${canEdit + ? `` + : ''} + + `).join(''); + + container.innerHTML = ` + + + + + + + + + + + + + + ${rows} +
Account IDProviderEnabledModeCadenceTarget CoverageUpdated
+ `; + + // Wire edit buttons via querySelectorAll so no user data is ever injected + // into a JS string context (data-* attributes are HTML-attribute-context only). + container.querySelectorAll('.ladder-edit-btn').forEach(btn => { + btn.addEventListener('click', () => { + const accountID = btn.dataset['accountId'] ?? ''; + const provider = btn.dataset['provider'] ?? ''; + const cfg = cachedConfigs.find(c => c.cloud_account_id === accountID && c.provider === provider); + if (cfg) openLadderConfigModal(cfg); + }); + }); +} + +// ========================================== +// MODAL CLOSE WIRING +// ========================================== + +// closeLadderModal hides the per-account config modal. +function closeLadderModal(): void { + document.getElementById('ladder-config-modal')?.classList.add('hidden'); +} + +// wireModalCloseButtons attaches click listeners to the modal's × and Cancel +// buttons. Uses addEventListener rather than inline onclick for CSP +// consistency with the dynamically-wired edit buttons (no inline handlers +// anywhere in this module). +function wireModalCloseButtons(): void { + document.getElementById('ladder-modal-close-btn')?.addEventListener('click', closeLadderModal); + document.getElementById('ladder-modal-cancel-btn')?.addEventListener('click', closeLadderModal); +} + +// ========================================== +// KILL-SWITCH TOGGLE +// ========================================== + +function wireKillSwitchToggle(): void { + const toggle = document.getElementById('setting-laddering-enabled') as HTMLInputElement | null; + if (!toggle || !canAccess('update', 'config')) return; + + toggle.addEventListener('change', async () => { + if (killSwitchSaveInFlight) { + toggle.checked = !toggle.checked; // revert optimistic + return; + } + killSwitchSaveInFlight = true; + try { + // The backend merges a partial PUT over the stored config (json.Unmarshal + // only sets keys present in the body), so we send ONLY the field we are + // changing. Every other setting is preserved server-side; enumerating the + // whole config here was fragile and dropped fields it did not list. + await api.updateConfig({ laddering_enabled: toggle.checked }); + showToast({ + message: `Commitment Laddering ${toggle.checked ? 'enabled' : 'disabled'} globally.`, + kind: 'success', + }); + } catch (err) { + toggle.checked = !toggle.checked; // revert on error + showToast({ message: 'Failed to update laddering kill-switch.', kind: 'error' }); + console.error('Failed to toggle laddering_enabled:', err); + } finally { + killSwitchSaveInFlight = false; + } + }); +} + +// ========================================== +// PER-ACCOUNT CONFIG MODAL +// ========================================== + +function openLadderConfigModal(existing?: api.LadderConfig): void { + const modal = document.getElementById('ladder-config-modal'); + if (!modal) return; + + // Populate modal fields. + setValue('ladder-cfg-id', existing?.id ?? ''); + setValue('ladder-cfg-account', existing?.cloud_account_id ?? ''); + setSelectValue('ladder-cfg-provider', existing?.provider ?? 'aws'); + setChecked('ladder-cfg-enabled', existing?.enabled ?? false); + setSelectValue('ladder-cfg-mode', existing?.mode ?? 'email_approval'); + setSelectValue('ladder-cfg-cadence', existing?.cadence ?? 'daily'); + setValue('ladder-cfg-target-coverage', String(existing?.target_coverage ?? 100)); + setValue('ladder-cfg-buffer-fraction', String(existing?.buffer_fraction ?? 0.10)); + setValue('ladder-cfg-baseline-percentile', String(existing?.baseline_percentile ?? 5.0)); + setValue('ladder-cfg-lookback-days', String(existing?.lookback_days ?? 30)); + setValue('ladder-cfg-buf-util-threshold', String(existing?.buffer_utilization_threshold ?? 90)); + setValue('ladder-cfg-max-hourly', existing?.max_hourly_commit_per_run != null + ? String(existing.max_hourly_commit_per_run) + : ''); + setValue('ladder-cfg-max-actions', String(existing?.max_actions_per_run ?? 10)); + + const defaultRamp = JSON.stringify({ steps: [{ after_days: 0, fraction: 1.0 }] }, null, 2); + setValue('ladder-cfg-ramp-schedule', + existing?.ramp_schedule ? JSON.stringify(existing.ramp_schedule, null, 2) : defaultRamp); + + // Disable account/provider fields when editing an existing row (they form + // the upsert key and cannot be changed without a delete+re-create). + const accountInput = document.getElementById('ladder-cfg-account') as HTMLInputElement | null; + const providerSelect = document.getElementById('ladder-cfg-provider') as HTMLSelectElement | null; + if (accountInput) accountInput.readOnly = !!existing; + if (providerSelect) providerSelect.disabled = !!existing; + + // Wire the save button (remove previous listener by replacing the element + // clone so duplicate-listener accumulation cannot occur). + const form = document.getElementById('ladder-config-form'); + if (form) { + const newForm = form.cloneNode(true) as HTMLElement; + form.replaceWith(newForm); + newForm.addEventListener('submit', (e) => { + e.preventDefault(); + saveLadderConfig(); + }); + // The Cancel button lives inside the form, so cloneNode drops its + // listener (cloneNode does not copy event handlers). Re-wire it on the + // fresh node. The × close button sits outside the form and keeps the + // listener attached once in wireModalCloseButtons. + newForm.querySelector('#ladder-modal-cancel-btn') + ?.addEventListener('click', closeLadderModal); + } + + modal.classList.remove('hidden'); +} + +// Exported for unit testing (see renderConfigTable note). +export async function saveLadderConfig(): Promise { + if (configModalSaveInFlight) return; + + const maxHourlyRaw = (document.getElementById('ladder-cfg-max-hourly') as HTMLInputElement | null)?.value?.trim() ?? ''; + const maxHourly: number | null = maxHourlyRaw === '' ? null : Number(maxHourlyRaw); + if (maxHourlyRaw !== '' && (!Number.isFinite(maxHourly) || (maxHourly as number) <= 0)) { + showToast({ message: 'Max hourly commit must be a positive number (or blank for no cap).', kind: 'error' }); + return; + } + + const rampRaw = (document.getElementById('ladder-cfg-ramp-schedule') as HTMLTextAreaElement | null)?.value?.trim() ?? ''; + let rampSchedule: api.LadderConfig['ramp_schedule']; + try { + rampSchedule = JSON.parse(rampRaw); + } catch { + showToast({ message: 'Ramp schedule is not valid JSON.', kind: 'error' }); + return; + } + + const cfg: api.LadderConfig = { + id: (document.getElementById('ladder-cfg-id') as HTMLInputElement | null)?.value || undefined, + cloud_account_id: (document.getElementById('ladder-cfg-account') as HTMLInputElement | null)?.value?.trim() ?? '', + provider: (document.getElementById('ladder-cfg-provider') as HTMLSelectElement | null)?.value ?? 'aws', + enabled: (document.getElementById('ladder-cfg-enabled') as HTMLInputElement | null)?.checked ?? false, + mode: ((document.getElementById('ladder-cfg-mode') as HTMLSelectElement | null)?.value ?? 'email_approval') as api.LadderConfig['mode'], + cadence: ((document.getElementById('ladder-cfg-cadence') as HTMLSelectElement | null)?.value ?? 'daily') as api.LadderConfig['cadence'], + target_coverage: Number((document.getElementById('ladder-cfg-target-coverage') as HTMLInputElement | null)?.value ?? '100'), + buffer_fraction: Number((document.getElementById('ladder-cfg-buffer-fraction') as HTMLInputElement | null)?.value ?? '0.10'), + baseline_percentile: Number((document.getElementById('ladder-cfg-baseline-percentile') as HTMLInputElement | null)?.value ?? '5'), + lookback_days: Number((document.getElementById('ladder-cfg-lookback-days') as HTMLInputElement | null)?.value ?? '30'), + buffer_utilization_threshold: Number((document.getElementById('ladder-cfg-buf-util-threshold') as HTMLInputElement | null)?.value ?? '90'), + max_hourly_commit_per_run: maxHourly, + max_actions_per_run: Number((document.getElementById('ladder-cfg-max-actions') as HTMLInputElement | null)?.value ?? '10'), + ramp_schedule: rampSchedule, + }; + + if (!cfg.cloud_account_id) { + showToast({ message: 'Cloud Account ID is required.', kind: 'error' }); + return; + } + + configModalSaveInFlight = true; + const saveBtn = document.getElementById('ladder-config-save-btn') as HTMLButtonElement | null; + if (saveBtn) saveBtn.disabled = true; + + try { + const saved = await api.upsertLadderConfig(cfg); + + // Update the in-memory cache and re-render the table. + const idx = cachedConfigs.findIndex( + c => c.cloud_account_id === saved.cloud_account_id && c.provider === saved.provider + ); + if (idx >= 0) { + cachedConfigs[idx] = saved; + } else { + cachedConfigs.push(saved); + } + renderConfigTable(cachedConfigs); + + closeLadderModal(); + showToast({ message: 'Ladder config saved.', kind: 'success' }); + } catch (err) { + showToast({ message: 'Failed to save ladder config. Check inputs and try again.', kind: 'error' }); + console.error('saveLadderConfig:', err); + } finally { + configModalSaveInFlight = false; + if (saveBtn) saveBtn.disabled = false; + } +} + +// ========================================== +// DOM HELPERS +// ========================================== + +function setValue(id: string, value: string): void { + const el = document.getElementById(id) as HTMLInputElement | HTMLTextAreaElement | null; + if (el) el.value = value; +} + +function setSelectValue(id: string, value: string): void { + const el = document.getElementById(id) as HTMLSelectElement | null; + if (el) el.value = value; +} + +function setChecked(id: string, checked: boolean): void { + const el = document.getElementById(id) as HTMLInputElement | null; + if (el) el.checked = checked; +} diff --git a/frontend/src/lib/capacity.ts b/frontend/src/lib/capacity.ts new file mode 100644 index 000000000..aabe86ae8 --- /dev/null +++ b/frontend/src/lib/capacity.ts @@ -0,0 +1,111 @@ +/** + * Cross-provider compute capacity ranking utilities. + * + * Provides a comparator for sorting cloud commitment recommendations by their + * underlying compute capacity (vCPU first, then memory in GB, then instance + * type lexically). This lets a mixed AWS/Azure/GCP recommendation table be + * sorted by actual size rather than by opaque type-name strings whose + * lexicographic order carries no capacity meaning. + * + * The fields below mirror the JSON keys emitted by ComputeDetails in + * pkg/common/types.go (vcpu / memory_gb) — the shapes must stay in sync. + * + * Missing capacity values (0 or null/undefined from the API) sort to the + * end of the list rather than the front, so well-populated rows stay visible + * at the top when the user sorts ascending by size. + */ + +/** + * Minimum capability representation carried by a recommendation's details + * payload when the service is "compute". Matches the JSON produced by + * ComputeDetails (pkg/common/types.go): vcpu and memory_gb are omitempty so + * they may be absent from the wire payload. + */ +export interface ComputeCapacity { + vcpu?: number | null; + memory_gb?: number | null; + [key: string]: unknown; +} + +/** + * compareByCapacity ranks two compute recommendations by capacity. + * + * Sort key priority: + * 1. vCPU count (ascending, unknowns / 0 sort last) + * 2. MemoryGB (ascending, unknowns / 0 sort last) + * 3. resource_type lexical (ascending, stable tie-break across providers) + * + * "Unknown" means the capacity value is absent, null, or 0 (see the + * ComputeDetails godoc: "0 = unknown"). Nulls sort to the end so that + * well-populated rows stay at the top when the user chooses ascending order. + * (Rationale: feedback_nullable_not_zero.md — absent numeric fields must not + * be treated as zero in sort/rank paths.) + * + * @param a - first recommendation (only details + resource_type used) + * @param b - second recommendation (same) + * @returns negative / 0 / positive per Array.prototype.sort contract + */ +export function compareByCapacity( + a: { details?: unknown; resource_type?: string }, + b: { details?: unknown; resource_type?: string } +): number { + const ca = extractCapacity(a.details); + const cb = extractCapacity(b.details); + + const vcpuCmp = compareNullable(ca.vcpu ?? null, cb.vcpu ?? null); + if (vcpuCmp !== 0) return vcpuCmp; + + const memCmp = compareNullable(ca.memory_gb ?? null, cb.memory_gb ?? null); + if (memCmp !== 0) return memCmp; + + // Stable tie-break: lexical by resource_type across providers. + const ta = a.resource_type ?? ''; + const tb = b.resource_type ?? ''; + return ta < tb ? -1 : ta > tb ? 1 : 0; +} + +/** + * extractCapacity pulls vcpu / memory_gb from an opaque details blob. + * Returns null for non-compute payloads or absent/non-finite fields, + * matching the ComputeDetails omitempty behaviour on the wire. + * + * Note: Number.isFinite() is used (not typeof === 'number') to reject NaN, + * which passes the typeof check but would produce NaN subtraction results in + * compareNullable and cause an unstable sort. NaN falls into the unknown bucket + * and sorts last alongside null/absent/0 values. + * + * Intended consumer: the recommendations table capacity sort (clicking the + * vCPU/Memory column header). Wiring that column requires a UX decision + * (adding a new RecommendationsColumnId entry and a rendered column header); + * see issue #82 for context. + */ +function extractCapacity(details: unknown): ComputeCapacity { + if ( + details !== null && + typeof details === 'object' && + !Array.isArray(details) + ) { + const d = details as Record; + return { + vcpu: Number.isFinite(d['vcpu']) ? (d['vcpu'] as number) : null, + memory_gb: Number.isFinite(d['memory_gb']) ? (d['memory_gb'] as number) : null, + }; + } + return { vcpu: null, memory_gb: null }; +} + +/** + * compareNullable sorts null/zero values after all known-positive values. + * For two positive numbers it falls back to numeric ascending order. + * A value of 0 is treated as "unknown" per the ComputeDetails convention. + */ +function compareNullable(a: number | null, b: number | null): number { + const aUnknown = a === null || a === 0; + const bUnknown = b === null || b === 0; + + if (aUnknown && bUnknown) return 0; + if (aUnknown) return 1; // a is unknown -> sort after b + if (bUnknown) return -1; // b is unknown -> sort before it + + return (a as number) - (b as number); +} diff --git a/frontend/src/lib/chip-select.ts b/frontend/src/lib/chip-select.ts new file mode 100644 index 000000000..05d968a5a --- /dev/null +++ b/frontend/src/lib/chip-select.ts @@ -0,0 +1,372 @@ +/** + * chip-select — a filter chip that opens a popover menu (issue #344 T1). + * + * Replaces native ` lists (issue #1629). Before this, those lists were a *third*, +// independently hand-maintained copy of the vocabulary in groupModals.ts +// that had drifted 13 actions and 2 resources behind this file, so any +// group edit silently dropped or widened permissions the form couldn't +// represent. Deriving from Action/Resource instead of hand-writing this +// array again keeps there being exactly one place to update. +// +// The Record / Record assignments below are a +// compile-time exhaustiveness check: TS rejects the assignment if a key is +// missing OR if a key doesn't belong to the union, so adding a new Action +// or Resource variant without updating these objects fails the build +// instead of silently reintroducing the drift this issue closes. +const ACTION_EXHAUSTIVENESS_CHECK: Record = { + view: true, + create: true, + update: true, + delete: true, + execute: true, + approve: true, + 'cancel-own': true, + 'cancel-any': true, + 'retry-own': true, + 'retry-any': true, + 'approve-own': true, + 'approve-any': true, + 'execute-own': true, + 'execute-any': true, + 'update-any': true, + 'revoke-own': true, + 'revoke-any': true, + 'sell-own': true, + 'sell-any': true, + admin: true, +}; +export const ALL_ACTIONS: readonly Action[] = Object.keys(ACTION_EXHAUSTIVENESS_CHECK) as Action[]; + +const RESOURCE_EXHAUSTIVENESS_CHECK: Record = { + '*': true, + recommendations: true, + plans: true, + purchases: true, + history: true, + accounts: true, + config: true, + users: true, + groups: true, + 'api-keys': true, + 'ri-exchange': true, +}; +export const ALL_RESOURCES: readonly Resource[] = Object.keys(RESOURCE_EXHAUSTIVENESS_CHECK) as Resource[]; + +/** + * Well-known group UUID for the Administrators group seeded by + * migration 000057. Being a member of this group is the frontend + * equivalent of the backend's HasPermissionAPI(admin, *) == true. + */ +export const ADMINISTRATORS_GROUP_ID = '00000000-0000-5000-8000-000000000001'; + +/** + * Well-known group UUID for the Purchaser group relocated by migration + * 000064 (issue #942; originally seeded for issue #923). The three + * money-spending verbs (execute:purchases, approve-any:purchases, + * retry-any:purchases) are carved out of the admin:* wildcard and + * require explicit membership in this group (or a custom group that + * grants the same verbs). + */ +export const PURCHASER_GROUP_ID = '00000000-0000-5000-8000-000000000007'; + +/** + * Well-known group UUID for the RI Exchanger group seeded by migration + * 000096 (issue #1644). execute:ri-exchange is carved out of the admin:* + * wildcard and requires explicit membership in this group (or a custom + * group granting the same verb). Mirrors DefaultRIExchangerGroupID in + * internal/auth/types.go. + */ +export const RI_EXCHANGER_GROUP_ID = '00000000-0000-5000-8000-000000000008'; + +/** + * The set of (action, resource) pairs carved out of the admin:* + * wildcard. Mirrors adminCarvedOuts in internal/auth/types.go. Which + * group's membership grants each key back during the fallback path + * (effectivePermissions not yet loaded) is NOT uniform across this set -- + * see CARVE_OUT_FALLBACK_CHECK below, which every entry here must also + * appear in. + */ +const ADMIN_CARVED_OUTS: ReadonlySet = new Set([ + 'execute:purchases', + 'approve-any:purchases', + 'retry-any:purchases', + // execute:ri-exchange is carved out by issue #1644 and granted by the + // seeded RI Exchanger group (migration 000096), not by admin:*. If this + // set drifts from adminCarvedOuts the UI offers an action the backend + // then refuses with a 403. + 'execute:ri-exchange', +]); + +/** + * Subset of ADMIN_CARVED_OUTS specific to the three money-spending purchase + * verbs (issue #923). isPurchaser() consults only these -- NOT the full + * ADMIN_CARVED_OUTS set -- so that holding execute:ri-exchange alone (issue + * #1644, a disjoint carve-out with its own group) does not also satisfy the + * "can spend money" predicate the no-Purchaser banners key off. + */ +const PURCHASER_CARVED_OUTS: ReadonlySet = new Set([ + 'execute:purchases', + 'approve-any:purchases', + 'retry-any:purchases', +]); + +/** + * Return true when the current session user is a member of the + * Administrators group. This replaces the former `user.role === "admin"` + * check that PR #912 removed from both the backend and the API response. + * + * A null user (logged out, pre-init race) returns false. + */ +export function isAdmin(): boolean { + const user = state.getCurrentUser(); + if (!user) return false; + return Array.isArray(user.groups) && user.groups.includes(ADMINISTRATORS_GROUP_ID); +} + +/** + * Return true when the current session is authorised to execute the + * three carved-out money-spending verbs (execute:purchases, + * approve-any:purchases, retry-any:purchases). When the backend has + * delivered effectivePermissions (post-bootstrap) we drive off the + * permission set itself so a user who holds any of those verbs via a + * custom group (not just the seeded Purchaser group) also returns + * true. While effectivePermissions is still loading we fall back to + * seeded-group membership so the helper agrees with the canAccess() + * carve-out fallback in the same window. + * + * Callers that need a hard verb-specific gate should prefer + * canAccess('execute', 'purchases'). isPurchaser() is the + * verb-agnostic "can spend money at all" predicate (true if ANY of + * the three carved-out verbs is granted), which is what the + * no-Purchaser banners use. + */ +export function isPurchaser(): boolean { + const user = state.getCurrentUser(); + if (!user) return false; + if (user.effectivePermissions) { + // Match canAccess()'s semantics: a permission entry with + // resource '*' satisfies the carved-out verb on 'purchases' the + // same way the backend's HasPermission accepts ResourceAll. Walk + // each carved-out key and accept either an exact match or a + // wildcard-resource match on the same action. + for (const key of PURCHASER_CARVED_OUTS) { + const colon = key.indexOf(':'); + if (colon < 0) continue; + const action = key.slice(0, colon); + const resource = key.slice(colon + 1); + for (const p of user.effectivePermissions) { + if (p.action === action && (p.resource === resource || p.resource === '*')) { + return true; + } + } + } + return false; + } + return Array.isArray(user.groups) && user.groups.includes(PURCHASER_GROUP_ID); +} + +/** + * Return true when the current session is authorised for the + * execute:ri-exchange carved-out verb (issue #1644). Mirrors isPurchaser()'s + * shape: when effectivePermissions has loaded, drive off the permission set + * itself so a user granted the verb via a custom group (not just the seeded + * RI Exchanger group) also returns true; while it is still loading, fall + * back to seeded RI-Exchanger-group membership so this helper agrees with + * canAccess()'s fallback in the same window. + */ +export function isRIExchanger(): boolean { + const user = state.getCurrentUser(); + if (!user) return false; + if (user.effectivePermissions) { + for (const p of user.effectivePermissions) { + if (p.action === 'execute' && (p.resource === 'ri-exchange' || p.resource === '*')) { + return true; + } + } + return false; + } + return Array.isArray(user.groups) && user.groups.includes(RI_EXCHANGER_GROUP_ID); +} + +/** + * Maps each carved-out (action:resource) key to the predicate that grants it + * back during the fallback (effectivePermissions not yet loaded) path. + * ADMIN_CARVED_OUTS mirrors the backend's *set* of carved-out verbs; this map + * mirrors which group's membership grants each one back, which is NOT + * uniform (Purchaser for the three money-spending verbs, RI Exchanger for + * execute:ri-exchange). A carved-out key missing from this map would be + * silently hardcoded to the wrong predicate here, which is exactly the bug + * this map replaces: canAccess() used to route every carved-out verb through + * isPurchaser() regardless of which group actually granted it (PR #1758 + * review). + */ +const CARVE_OUT_FALLBACK_CHECK: ReadonlyMap boolean> = new Map([ + ['execute:purchases', isPurchaser], + ['approve-any:purchases', isPurchaser], + ['retry-any:purchases', isPurchaser], + ['execute:ri-exchange', isRIExchanger], +]); + +/** + * Returns true when the current session's effective permissions grant + * the specified action on the specified resource. + * + * When effectivePermissions is populated (fetched from + * GET /api/auth/me/permissions on login/bootstrap) the set is + * consulted directly: admin:* grants everything EXCEPT the verbs carved + * out of admin:* by the backend (the three money-spending verbs from + * issue #923, plus execute:ri-exchange from issue #1644) -- those require + * an explicit (action, resource) entry in effectivePermissions (which the + * backend only returns when the user is in the group that grants the verb, + * or a custom group that grants it directly). For non-admin entries an + * exact action:resource match (or matching action with resource '*') is + * required. + * + * While effectivePermissions is not yet loaded (e.g. during the first + * render before the async fetch completes) the function falls back to + * group-membership checks via CARVE_OUT_FALLBACK_CHECK: Administrators- + * group members pass every check EXCEPT the carved-out verbs, each of which + * requires membership in the specific group that grants it (Purchaser for + * the money-spending verbs, RI Exchanger for execute:ri-exchange). This + * mirrors the backend's HasPermission carve-out so UX and enforcement agree + * on the same verbs whether or not effectivePermissions has loaded yet. + * + * UX-only gate. The backend still enforces on every request; a + * wrong-positive surfaces as a 403 on click, a wrong-negative just + * hides a button. + */ +export function canAccess(action: Action, resource: Resource): boolean { + const user = state.getCurrentUser(); + if (!user) return false; + + const key = `${action}:${resource}`; + const isCarvedOut = ADMIN_CARVED_OUTS.has(key); + + // Use the server-provided effective permission set when available. + if (user.effectivePermissions) { + for (const p of user.effectivePermissions) { + // admin:* covers everything EXCEPT the carved-out verbs. + if (p.action === 'admin' && p.resource === '*' && !isCarvedOut) { + return true; + } + if (p.action === action && (p.resource === resource || p.resource === '*')) { + return true; + } + } + return false; + } + + // Fallback while permissions are still loading. Mirror the backend's + // carve-out: admin grants everything except the carved-out verbs, each of + // which requires membership in the specific group that grants it back. + if (isCarvedOut) { + const check = CARVE_OUT_FALLBACK_CHECK.get(key); + return check !== undefined && check(); + } + return isAdmin(); +} + +/** + * Return the well-known permission set for a legacy role name. Kept + * for the effective-permissions display on the admin Users page, which + * shows what permissions the built-in role-mirror groups carry. + * No longer used for session gating. + */ +export function getRolePermissions(role: string | undefined | null): ReadonlySet { + switch (role) { + case 'admin': + return ADMIN_PERMS; + case 'user': + return USER_PERMS; + case 'readonly': + return READONLY_PERMS; + default: + return new Set(); + } +} diff --git a/frontend/src/plans.ts b/frontend/src/plans.ts new file mode 100644 index 000000000..4a9e677cd --- /dev/null +++ b/frontend/src/plans.ts @@ -0,0 +1,2359 @@ +/** + * Plans module for CUDly + */ + +import * as api from './api'; +import * as state from './state'; +import { formatDate, formatTerm, getStatusBadge, escapeHtml, escapeHtmlAttr, formatCurrency, CURRENCY_DEFAULT_DIGITS, providerBadgeHtml } from './utils'; +import { showToast } from './toast'; +import { confirmDialog } from './confirmDialog'; +import type { PlansResponse, LocalPlan, SavePlanData } from './types'; +import { viewPlanHistory } from './history'; +import type { PlannedPurchase } from './api'; +import { populateTermSelect, populatePaymentSelect, isValidCombination, normalizePaymentValue } from './commitmentOptions'; +import { openModal, closeModal } from './modal'; +import { showSkeletonTiles, showSkeletonRows, teardownSkeleton } from './lib/skeleton'; +import { canAccess } from './permissions'; +import { parseNumericFilter, applyColumnFilters as applyColumnFiltersLib } from './lib/column-filters'; + +// pendingPlanRecommendations holds the resolved plan target captured at +// "Plan from N selected" button-click time. The Plan flow used to re-derive +// its target from state.getVisibleRecommendations() + getSelectedRecommendation +// IDs() at savePlan time, but state mutations between modal-open and +// modal-Save (Refresh, filter changes, deselections) could silently shrink +// or replace the planned set. This snapshot is stamped by openCreatePlan- +// Modal(snapshot) and consumed by savePlan; openNewPlanModal() clears it +// (the New-Plan-from-scratch path has no pre-resolved target). See #273 +// CR follow-up. +let pendingPlanRecommendations: api.Recommendation[] = []; + +// Install-once guard for setupRampScheduleHandlers. The elements it binds are +// static modal singletons that are never replaced, so adding listeners on +// every modal open stacks duplicate handlers. The flag ensures we bind once; +// reset to false only in tests via resetRampHandlersForTest(). +let rampHandlersInstalled = false; + +/** + * Load plans and planned purchases + */ +export async function loadPlans(): Promise { + // Issue #344 T3: skeleton tiles for the plans list. Synchronous + // render before fetch so the page doesn't sit blank during the + // round-trip. The planned-purchases skeleton lives in + // loadPlannedPurchases so direct callers of that fetch (not via + // loadPlans) get the same loading affordance. + const plansList = document.getElementById('plans-list'); + if (plansList) showSkeletonTiles(plansList, 3); + + // Issue #365: hide the top-level "New Plan" button for sessions that + // can't create plans. Readonly users hit a 403 on click otherwise. + // The button itself stays in the DOM (HTML keeps the static markup) + // so admin/user sessions get the same layout they always did. + const newPlanBtn = document.getElementById('new-plan-btn'); + if (newPlanBtn) newPlanBtn.hidden = !canAccess('create', 'plans'); + + try { + // Account filter: pass account_ids to the backend so it JOINs + // plan_accounts and returns only plans that reference one of the + // selected accounts. Empty array means "all plans" — the backend + // omits the JOIN entirely in that case. Mirrors the pattern used by + // getRecommendations (see recommendations.ts, issue #705). + const accountIDs = state.getCurrentAccountIDs(); + const data = await api.getPlans( + accountIDs.length > 0 ? { account_ids: accountIDs } : {} + ) as unknown as PlansResponse; + let plans = data.plans || []; + + // Client-side provider filter. Backend `config.PurchasePlan` has no + // top-level `provider` field — the plan's provider is derived from + // its first service entry (see extractPlanInfo below). Filtering on + // `p.provider` directly silently returned zero rows for every + // non-empty filter value. + // Filter source is the global topbar (state.ts), shared across + // sections. The topbar's "All Providers" chip writes '' to state, + // which is falsy so the filter is naturally skipped. + const providerFilter = state.getCurrentProvider(); + if (providerFilter) { + plans = plans.filter(p => extractPlanInfo(p as unknown as BackendPlan).provider === providerFilter); + } + + renderPlans(plans); + } catch (error) { + console.error('Failed to load plans:', error); + const list = document.getElementById('plans-list'); + if (list) { + teardownSkeleton(list); + const err = error as Error; + list.innerHTML = `

Failed to load plans: ${escapeHtml(err.message)}

`; + } + } + + // Load planned purchases + await loadPlannedPurchases(); +} + +/** + * Load planned purchases + */ +async function loadPlannedPurchases(): Promise { + const container = document.getElementById('planned-purchases-list'); + if (!container) return; + + // Issue #344 T3 (CR follow-up on PR #346): skeleton lives here, not + // in loadPlans, so direct callers (e.g. follow-up refresh paths after + // a single purchase action) also get the loading affordance. 5 rows + // × 11 cols matches the rendered table — see renderPlannedPurchases. + showSkeletonRows(container, 5, 11); + + try { + const data = await api.getPlannedPurchases(); + renderPlannedPurchases(data.purchases || []); + } catch (error) { + console.error('Failed to load planned purchases:', error); + teardownSkeleton(container); + const err = error as Error; + container.innerHTML = `

Failed to load planned purchases: ${escapeHtml(err.message)}

`; + } +} + +// Cached last-fetched planned purchases. The filter popover commits re-render +// without re-fetching, so we hold the unfiltered set in module scope. Reset +// on every successful loadPlannedPurchases() and consumed by +// rerenderPlannedPurchases() (called by popover commits + Clear button). +let lastLoadedPurchases: PlannedPurchase[] = []; + +/** + * Render planned purchases list + */ +function renderPlannedPurchases(purchases: PlannedPurchase[]): void { + lastLoadedPurchases = [...purchases]; + renderPlannedPurchasesInternal(); +} + +// renderPlannedPurchasesInternal is the actual render — separate from +// renderPlannedPurchases() so popover commits can re-render the table +// against the cached unfiltered set without re-fetching from the API. +function renderPlannedPurchasesInternal(): void { + const container = document.getElementById('planned-purchases-list'); + if (!container) return; + + const purchases = lastLoadedPurchases; + if (!purchases || purchases.length === 0) { + container.innerHTML = '

No planned purchases. Create a purchase plan to schedule automatic purchases.

'; + return; + } + + const filters = state.getPlansColumnFilters(); + const filtered = applyPlansColumnFilters(purchases, filters); + + const filterBtn = (column: state.PlansColumnId, lbl: string): string => { + const active = filters[column] ? ' active' : ''; + const label = filters[column] ? `Filter ${lbl} \u2014 currently active` : `Filter ${lbl}`; + return ``; + }; + + const tbody = filtered.length === 0 + ? `No rows match these filters.` + : filtered.map(purchase => renderPlannedPurchaseRow(purchase)).join(''); + + container.innerHTML = ` + + + + + + + + + + + + + + + + + + ${tbody} + +
PlanScheduled DateProvider${filterBtn('provider', 'Provider')}Service${filterBtn('service', 'Service')}Resource Type${filterBtn('resource_type', 'Resource Type')}Count${filterBtn('count', 'Count')}Term${filterBtn('term', 'Term')}${filterBtn('payment', 'Payment')}Upfront${filterBtn('upfront_cost', 'Upfront')}Est. Savings${filterBtn('estimated_savings', 'Est. Savings')}Status${filterBtn('status', 'Status')}Actions
+ `; + + // Add event listeners for row action buttons + container.querySelectorAll('[data-action]').forEach(btn => { + btn.addEventListener('click', () => void handlePlannedPurchaseAction( + btn.dataset['action'] || '', + btn.dataset['id'] || '', + btn.dataset['planId'] || '' + )); + }); + + // Per-column filter trigger buttons. e.stopPropagation prevents any future + // surrounding-th handlers (sort etc.) from firing on the same click. + container.querySelectorAll('.column-filter-btn').forEach((btn) => { + const column = btn.dataset['column'] as state.PlansColumnId | undefined; + if (!column) return; + btn.addEventListener('click', (e) => { + e.stopPropagation(); + openPlansColumnPopover(column, btn); + }); + }); + + // Re-anchor any open popover to the freshly-rendered trigger so the + // popover survives table re-renders triggered by other state changes. + rebindOpenPlansPopoverAnchor(); +} + +// --------------------------------------------------------------------------- +// Per-column filter pipeline (issue #166 follow-up to #570). +// +// Mirrors the canonical recommendations.ts wiring: cell extractors map each +// row + column id to its raw value, numeric values are rounded to the cell's +// display precision so filter predicates match what the user sees. +// --------------------------------------------------------------------------- + +function applyPlansColumnFilters( + purchases: readonly PlannedPurchase[], + filters: state.PlansColumnFilters, +): PlannedPurchase[] { + return applyColumnFiltersLib( + purchases, + filters, + { + categorical: categoricalCellValueForPlan, + numeric: (p, col) => roundForDisplay(numericCellValueForPlan(p, col), displayPrecisionForPlan(col)), + }, + ); +} + +function categoricalCellValueForPlan(p: PlannedPurchase, col: state.PlansColumnId): string { + switch (col) { + case 'provider': return p.provider ?? ''; + case 'service': return p.service ?? ''; + case 'resource_type': return p.resource_type ?? ''; + case 'term': return p.term == null ? '' : String(p.term); + case 'payment': return p.payment ?? ''; + case 'status': return p.status ?? ''; + // Numeric columns shouldn't reach this branch; return empty for type-safety. + case 'count': + case 'upfront_cost': + case 'estimated_savings': + return ''; + } +} + +function numericCellValueForPlan(p: PlannedPurchase, col: state.PlansColumnId): number { + switch (col) { + case 'count': return p.count ?? 0; + case 'upfront_cost': return p.upfront_cost ?? 0; + case 'estimated_savings': return p.estimated_savings ?? 0; + case 'provider': + case 'service': + case 'resource_type': + case 'term': + case 'payment': + case 'status': + return Number.NaN; + } +} + +// Issue #484 parity: filter predicates compare against the rounded display +// value so "exact-match" filters work for rows whose raw value rounds to the +// typed value. The planned-purchases table renders count as integer and +// currency cells via formatCurrency (CURRENCY_DEFAULT_DIGITS = 0). +function displayPrecisionForPlan(col: state.PlansColumnId): number { + switch (col) { + case 'count': + return 0; + case 'upfront_cost': + case 'estimated_savings': + return CURRENCY_DEFAULT_DIGITS; + case 'provider': + case 'service': + case 'resource_type': + case 'term': + case 'payment': + case 'status': + return CURRENCY_DEFAULT_DIGITS; + } +} + +function roundForDisplay(n: number, precision: number): number { + if (!Number.isFinite(n)) return n; + return Number(n.toFixed(precision)); +} + +// --------------------------------------------------------------------------- +// Plans column-filter popover (portal pattern — sibling of the one in +// recommendations.ts). Lives appended to document.body so it survives the +// table's innerHTML rewrite on every render. +// --------------------------------------------------------------------------- + +const PLANS_NUMERIC_COLUMNS: ReadonlySet = new Set([ + 'count', 'upfront_cost', 'estimated_savings', +]); + +interface PlansPopoverState { + column: state.PlansColumnId; + el: HTMLDivElement; + checkboxes: Map; + input: HTMLInputElement | null; + errorEl: HTMLElement | null; +} + +let openPlansPopover: PlansPopoverState | null = null; +let plansOutsideClickHandler: ((e: MouseEvent) => void) | null = null; +let plansEscKeyHandler: ((e: KeyboardEvent) => void) | null = null; +let plansResizeHandler: (() => void) | null = null; + +function plansColumnLabel(column: state.PlansColumnId): string { + switch (column) { + case 'provider': return 'Provider'; + case 'service': return 'Service'; + case 'resource_type': return 'Resource Type'; + case 'term': return 'Term'; + case 'payment': return 'Payment'; + case 'status': return 'Status'; + case 'count': return 'Count'; + case 'upfront_cost': return 'Upfront'; + case 'estimated_savings': return 'Est. Savings'; + } +} + +function plansCategoricalDisplayLabel(column: state.PlansColumnId, value: string): string { + if (value === '') return '(empty)'; + if (column === 'term') { + const n = Number(value); + return Number.isFinite(n) && n > 0 ? formatTerm(n) : value; + } + if (column === 'provider') { + return value.toUpperCase(); + } + return value; +} + +function getPlansColumnTriggerButton(column: state.PlansColumnId): HTMLButtonElement | null { + return document.querySelector( + `#planned-purchases-list th .column-filter-btn[data-column="${column}"]`, + ); +} + +function positionPlansPopover(popover: HTMLElement, anchor: HTMLElement): void { + const rect = anchor.getBoundingClientRect(); + popover.style.display = 'block'; + const popRect = popover.getBoundingClientRect(); + const margin = 8; + + let top = rect.bottom + 4; + if (top + popRect.height > window.innerHeight - margin) { + top = Math.max(margin, rect.top - popRect.height - 4); + } + let left = rect.left; + if (left + popRect.width > window.innerWidth - margin) { + left = Math.max(margin, window.innerWidth - margin - popRect.width); + } + popover.style.position = 'absolute'; + popover.style.top = `${top + window.scrollY}px`; + popover.style.left = `${left + window.scrollX}px`; +} + +function plansDistinctValuesForColumn( + purchases: readonly PlannedPurchase[], + column: state.PlansColumnId, +): string[] { + const seen = new Set(); + for (const p of purchases) { + seen.add(categoricalCellValueForPlan(p, column)); + } + return Array.from(seen).sort((a, b) => { + if (a === '' && b !== '') return -1; + if (a !== '' && b === '') return 1; + return a.localeCompare(b); + }); +} + +function buildPlansPopoverContent( + column: state.PlansColumnId, + purchases: readonly PlannedPurchase[], +): { el: HTMLDivElement; checkboxes: Map; input: HTMLInputElement | null; errorEl: HTMLElement | null } { + const popover = document.createElement('div'); + popover.className = 'column-filter-popover'; + popover.setAttribute('role', 'dialog'); + popover.setAttribute('aria-modal', 'false'); + + const headingId = `plans-column-filter-heading-${column}`; + popover.setAttribute('aria-labelledby', headingId); + + const heading = document.createElement('h3'); + heading.id = headingId; + heading.className = 'column-filter-heading'; + heading.textContent = `Filter ${plansColumnLabel(column)}`; + popover.appendChild(heading); + + const checkboxes = new Map(); + let input: HTMLInputElement | null = null; + let errorEl: HTMLElement | null = null; + let commitAllRef: ((target: boolean) => void) | null = null; + + if (PLANS_NUMERIC_COLUMNS.has(column)) { + const label = document.createElement('label'); + label.className = 'column-filter-numeric-label'; + label.textContent = 'Expression'; + input = document.createElement('input'); + input.type = 'text'; + input.className = 'column-filter-numeric-input'; + input.placeholder = 'e.g. >100, 50..200, 5'; + input.setAttribute('aria-describedby', `plans-column-filter-error-${column}`); + label.appendChild(input); + popover.appendChild(label); + + errorEl = document.createElement('div'); + errorEl.id = `plans-column-filter-error-${column}`; + errorEl.className = 'column-filter-error'; + errorEl.setAttribute('role', 'status'); + popover.appendChild(errorEl); + + const commit = (): void => { + const expr = input!.value.trim(); + if (expr === '') { + state.setPlansColumnFilter(column, null); + errorEl!.textContent = ''; + rerenderPlannedPurchases(); + return; + } + const parsed = parseNumericFilter(expr); + if (!parsed.ok) { + errorEl!.textContent = parsed.error; + return; + } + errorEl!.textContent = ''; + state.setPlansColumnFilter(column, { kind: 'expr', expr }); + rerenderPlannedPurchases(); + }; + input.addEventListener('blur', commit); + input.addEventListener('keydown', (e) => { + if (e.key === 'Enter') { + e.preventDefault(); + commit(); + } + }); + } else { + const distinct = plansDistinctValuesForColumn(purchases, column); + + const allLabel = document.createElement('label'); + allLabel.className = 'column-filter-all'; + const allBox = document.createElement('input'); + allBox.type = 'checkbox'; + allBox.dataset['role'] = 'all'; + allLabel.appendChild(allBox); + const allText = document.createElement('span'); + allText.textContent = '(All)'; + allLabel.appendChild(allText); + popover.appendChild(allLabel); + + const list = document.createElement('div'); + list.className = 'column-filter-list'; + for (const value of distinct) { + const itemLabel = document.createElement('label'); + itemLabel.className = 'column-filter-item'; + const cb = document.createElement('input'); + cb.type = 'checkbox'; + cb.dataset['value'] = value; + itemLabel.appendChild(cb); + const text = document.createElement('span'); + text.textContent = plansCategoricalDisplayLabel(column, value); + itemLabel.appendChild(text); + list.appendChild(itemLabel); + checkboxes.set(value, cb); + } + popover.appendChild(list); + + const updateAllTriState = (): void => { + const total = checkboxes.size; + let checked = 0; + checkboxes.forEach((cb) => { if (cb.checked) checked++; }); + allBox.indeterminate = checked > 0 && checked < total; + allBox.checked = checked === total && total > 0; + }; + + const commit = (): void => { + const selected: string[] = []; + checkboxes.forEach((cb, value) => { if (cb.checked) selected.push(value); }); + if (selected.length === checkboxes.size) { + state.setPlansColumnFilter(column, null); + } else { + state.setPlansColumnFilter(column, { kind: 'set', values: selected }); + } + updateAllTriState(); + rerenderPlannedPurchases(); + }; + + const commitAll = (target: boolean): void => { + checkboxes.forEach((cb) => { cb.checked = target; }); + if (target) { + state.setPlansColumnFilter(column, null); + } else { + state.setPlansColumnFilter(column, { kind: 'set', values: [] }); + } + updateAllTriState(); + rerenderPlannedPurchases(); + }; + commitAllRef = commitAll; + + checkboxes.forEach((cb) => { + cb.addEventListener('change', commit); + }); + allBox.addEventListener('change', () => { + commitAll(allBox.checked); + }); + } + + const footer = document.createElement('div'); + footer.className = 'column-filter-footer'; + const clearBtn = document.createElement('button'); + clearBtn.type = 'button'; + clearBtn.className = 'column-filter-clear'; + clearBtn.textContent = 'Clear'; + clearBtn.addEventListener('click', () => { + if (input) { + state.setPlansColumnFilter(column, null); + input.value = ''; + if (errorEl) errorEl.textContent = ''; + rerenderPlannedPurchases(); + } else { + commitAllRef?.(false); + } + }); + footer.appendChild(clearBtn); + popover.appendChild(footer); + + return { el: popover, checkboxes, input, errorEl }; +} + +function resyncOpenPlansPopover(): void { + if (!openPlansPopover) return; + const f = state.getPlansColumnFilters()[openPlansPopover.column]; + if (openPlansPopover.input) { + if (document.activeElement !== openPlansPopover.input) { + const expr = f && f.kind === 'expr' ? f.expr : ''; + openPlansPopover.input.value = expr; + if (openPlansPopover.errorEl) openPlansPopover.errorEl.textContent = ''; + } + return; + } + if (f == null) { + openPlansPopover.checkboxes.forEach((cb) => { cb.checked = true; }); + } else { + const values: ReadonlySet = f.kind === 'set' ? new Set(f.values) : new Set(); + openPlansPopover.checkboxes.forEach((cb, value) => { + cb.checked = values.has(value); + }); + } + const allBox = openPlansPopover.el.querySelector('input[data-role="all"]'); + if (allBox) { + const total = openPlansPopover.checkboxes.size; + let checked = 0; + openPlansPopover.checkboxes.forEach((cb) => { if (cb.checked) checked++; }); + allBox.indeterminate = checked > 0 && checked < total; + allBox.checked = checked === total && total > 0; + } +} + +function attachPlansPopoverGlobalListeners(): void { + if (plansOutsideClickHandler) return; + plansOutsideClickHandler = (e: MouseEvent): void => { + if (!openPlansPopover) return; + const target = e.target as Node | null; + if (!target) return; + if (openPlansPopover.el.contains(target)) return; + if (target instanceof Element && target.closest('.column-filter-btn')) return; + closePlansPopover(); + }; + plansEscKeyHandler = (e: KeyboardEvent): void => { + if (!openPlansPopover) return; + if (e.key === 'Escape') { + e.preventDefault(); + closePlansPopover(true); + } + }; + plansResizeHandler = (): void => { + if (!openPlansPopover) return; + const trigger = getPlansColumnTriggerButton(openPlansPopover.column); + if (!trigger) { + closePlansPopover(); + return; + } + positionPlansPopover(openPlansPopover.el, trigger); + }; + document.addEventListener('mousedown', plansOutsideClickHandler); + document.addEventListener('keydown', plansEscKeyHandler); + window.addEventListener('resize', plansResizeHandler); +} + +function detachPlansPopoverGlobalListeners(): void { + if (plansOutsideClickHandler) document.removeEventListener('mousedown', plansOutsideClickHandler); + if (plansEscKeyHandler) document.removeEventListener('keydown', plansEscKeyHandler); + if (plansResizeHandler) window.removeEventListener('resize', plansResizeHandler); + plansOutsideClickHandler = null; + plansEscKeyHandler = null; + plansResizeHandler = null; +} + +function openPlansColumnPopover(column: state.PlansColumnId, anchor: HTMLElement): void { + if (openPlansPopover && !openPlansPopover.el.isConnected) { + detachPlansPopoverGlobalListeners(); + openPlansPopover = null; + } + if (openPlansPopover && openPlansPopover.column === column) { + closePlansPopover(true); + return; + } + if (openPlansPopover) closePlansPopover(); + + const built = buildPlansPopoverContent(column, lastLoadedPurchases); + document.body.appendChild(built.el); + openPlansPopover = { + column, + el: built.el, + checkboxes: built.checkboxes, + input: built.input, + errorEl: built.errorEl, + }; + resyncOpenPlansPopover(); + positionPlansPopover(built.el, anchor); + anchor.setAttribute('aria-expanded', 'true'); + attachPlansPopoverGlobalListeners(); + + const firstFocusable = built.input + ?? built.el.querySelector('input[type="checkbox"]'); + firstFocusable?.focus(); +} + +function closePlansPopover(restoreFocus = false): void { + if (!openPlansPopover) return; + const { column, el } = openPlansPopover; + el.remove(); + openPlansPopover = null; + detachPlansPopoverGlobalListeners(); + const trigger = getPlansColumnTriggerButton(column); + if (trigger) { + trigger.setAttribute('aria-expanded', 'false'); + if (restoreFocus) trigger.focus(); + } +} + +function rebindOpenPlansPopoverAnchor(): void { + if (!openPlansPopover) return; + const trigger = getPlansColumnTriggerButton(openPlansPopover.column); + if (!trigger) { + closePlansPopover(); + return; + } + trigger.setAttribute('aria-expanded', 'true'); + positionPlansPopover(openPlansPopover.el, trigger); + resyncOpenPlansPopover(); +} + +function rerenderPlannedPurchases(): void { + renderPlannedPurchasesInternal(); +} + +// canManageScheduledPurchase returns true when the current session is +// permitted to act on the given scheduled purchase's row buttons (issue #950). +// UX gate only -- the backend authorizeExecutionManagement in +// internal/api/handler_purchases.go remains the security boundary; a +// false-positive here surfaces as a 403 toast on click rather than a +// successful mutation. +// +// Heuristic (mirrors the creator-scope model on History rows): +// * admin (admin:* wildcard) or update-any:purchases -> manage anyone's row; +// * otherwise the row's created_by_user_id must match the current user; +// * legacy rows with a NULL created_by_user_id -> no buttons for non- +// privileged users (out of reach without update-any). +function canManageScheduledPurchase(purchase: PlannedPurchase): boolean { + if (canAccess('admin', '*') || canAccess('update-any', 'purchases')) return true; + const user = state.getCurrentUser(); + if (!user) return false; + if (!purchase.created_by_user_id) return false; + return purchase.created_by_user_id === user.id; +} + +/** + * Render a single planned purchase row + */ +function renderPlannedPurchaseRow(purchase: PlannedPurchase): string { + const statusClass = getPlannedPurchaseStatusClass(purchase.status); + const isPaused = purchase.status === 'paused'; + const isPending = purchase.status === 'pending'; + const canRun = isPending || isPaused; + const termCell = purchase.term > 0 + ? `${purchase.term}yr ${escapeHtml((purchase.payment ?? '').replace('-', ' '))}` + : '—'; + // Show an em-dash for upfront=0 unless the plan truly is all-upfront; + // $0 upfront on "partial" or "no-upfront" is informative ($0 is real). + // A zero on an all-upfront term almost always means missing data. + const upfrontCell = purchase.upfront_cost > 0 || purchase.payment !== 'all-upfront' + ? formatCurrency(purchase.upfront_cost) + : '—'; + + // Issue #365: gate row actions by the same plan-management permissions + // a click on each button would require. Readonly users see no buttons + // (status badge only); user role sees Run/Pause/Resume/Edit but not + // Disable; admins see everything. + // + // Issue #950: AND in creator-scope ownership. A non-creator who lacks + // update-any:purchases (a standard user looking at someone else's row) + // sees NO action buttons, mirroring the backend ownership gate. This is + // a UX gate; the backend authorizePlannedPurchaseCancel is the real + // boundary. + // + // Issue #1442: the Disable button (cancel) must be shown to the creator + // when they hold cancel-own:purchases or cancel-any:purchases, not only + // when they hold delete:purchases. This mirrors the cancel-verb logic in + // canCancelUpcomingPurchase (dashboard.ts) and the backend + // requireDeleteOrCancelPurchasePermission gate introduced by PR #1421. + // canManagePurchase already enforces the ownership check, so accepting + // cancel-own here only widens the Disable button for the creator's own row. + const canManagePurchase = canManageScheduledPurchase(purchase); + const canRunPurchase = canManagePurchase && canAccess('execute', 'purchases') && canRun; + const canPauseOrResumePurchase = canManagePurchase && canAccess('update', 'purchases'); + const canEditPlan = canManagePurchase && canAccess('update', 'plans'); + const canDisablePlan = canManagePurchase && ( + canAccess('delete', 'purchases') || + canAccess('cancel-any', 'purchases') || + canAccess('cancel-own', 'purchases') + ); + + return ` + + + ${escapeHtml(purchase.plan_name)} + Step ${purchase.step_number}/${purchase.total_steps} + + ${formatDate(purchase.scheduled_date)} + ${providerBadgeHtml(purchase.provider)} + ${escapeHtml(purchase.service)} + ${escapeHtml(purchase.resource_type)} (${escapeHtml(purchase.region)}) + ${purchase.count} + ${termCell} + ${upfrontCell} + ${formatCurrency(purchase.estimated_savings)}/mo + ${escapeHtml(purchase.status)} + + ${canRunPurchase ? `` : ''} + ${canPauseOrResumePurchase && isPending ? `` : ''} + ${canPauseOrResumePurchase && isPaused ? `` : ''} + ${canEditPlan ? `` : ''} + ${canDisablePlan ? `` : ''} + + + `; +} + +/** + * Get CSS class for planned purchase status + */ +function getPlannedPurchaseStatusClass(status: string): string { + switch (status) { + case 'pending': return 'status-pending'; + case 'paused': return 'status-paused'; + case 'running': return 'status-running'; + case 'completed': return 'status-completed'; + case 'failed': return 'status-failed'; + default: return ''; + } +} + +/** + * Handle planned purchase action + */ +async function handlePlannedPurchaseAction(action: string, purchaseId: string, planId = ''): Promise { + try { + switch (action) { + case 'run': { + // Use styled async dialog (11-L2) instead of blocking browser confirm(). + const runOk = await confirmDialog({ + title: 'Run purchase now?', + body: 'This will immediately execute the purchase.', + confirmLabel: 'Run now', + destructive: true, + }); + if (runOk) { + await api.runPlannedPurchase(purchaseId); + showToast({ message: 'Purchase executed successfully', kind: 'success', timeout: 5_000 }); + } + break; + } + case 'pause': + // Pause is reversible and scoped to a single execution: the plan stays + // enabled (unlike Disable plan) and the row stays listed with a Paused + // badge (unlike the old silent-removal behaviour). + await api.pausePlannedPurchase(purchaseId); + showToast({ message: 'Purchase paused', kind: 'success', timeout: 5_000 }); + break; + case 'resume': + await api.resumePlannedPurchase(purchaseId); + showToast({ message: 'Purchase resumed', kind: 'success', timeout: 5_000 }); + break; + case 'edit': + // Open edit modal for the parent plan using plan_id, not the purchase id. + // The purchase row's data-plan-id attribute carries the plan FK (#773). + if (!planId) { + console.warn('edit action ignored: missing plan id'); + return; + } + // If the plan can't be loaded (deleted or no longer accessible while + // the row was on screen), reconcile the list so the orphaned row is + // dropped instead of leaving a dead Edit button (issue #1403). + if (!(await editPlan(planId))) { + await loadPlannedPurchases(); + } + return; + case 'disable': { + // Use styled async dialog (11-L2) instead of blocking browser confirm(). + const disableOk = await confirmDialog({ + title: 'Disable this plan?', + body: 'The plan will be paused and no purchases will be scheduled. You can re-enable it later from the Plans list.', + confirmLabel: 'Disable plan', + destructive: true, + }); + if (disableOk) { + await api.deletePlannedPurchase(purchaseId); + // Reload full plans list since we disabled a plan + await loadPlans(); + return; + } + break; + } + } + await loadPlannedPurchases(); + } catch (error) { + console.error(`Failed to ${action} planned purchase:`, error); + const err = error as Error; + showToast({ message: `Failed to ${action} purchase: ${err.message}`, kind: 'error' }); + } +} + +// Backend plan type (as returned from API) +interface BackendPlan { + id: string; + name: string; + enabled: boolean; + auto_purchase: boolean; + notification_days_before: number; + services?: Record; + ramp_schedule: { + type: string; + percent_per_step: number; + step_interval_days: number; + current_step: number; + total_steps: number; + }; + next_execution_date?: string; + // unassigned is true for legacy plans that have zero plan_accounts rows + // (issue #973). The backend sets this flag when an account filter is + // active so the frontend can bucket them under an "Unassigned" section. + unassigned?: boolean; + // Health score/factors are computed at read time on GET /plans (never + // persisted) -- see computePlanHealth in internal/api/plan_health.go + // (issue #340 follow-up). Optional so an older cached response served + // during a partial deploy still renders the card cleanly; explicitly + // null when the backend could not compute the score. + health_score?: number | null; + health_factors?: api.PlanHealthFactor[]; +} + +// Score bands for the per-plan health badge (issue #340 follow-up). Green +// >= 80 (healthy), amber 50-79 (needs attention), red < 50 (action needed). +// Reuses the existing status-badge palette (badge-success/badge-warning/ +// badge-danger) already used elsewhere on this card rather than introducing +// new CSS. +const HEALTH_SCORE_GOOD_THRESHOLD = 80; +const HEALTH_SCORE_WARN_THRESHOLD = 50; + +// Inclusive bounds of the score the API contracts to send (openapi.yaml +// PlanWithHealth.health_score: minimum 0, maximum 100). Anything outside is +// treated as unusable rather than banded. +const HEALTH_SCORE_MIN = 0; +const HEALTH_SCORE_MAX = 100; + +function healthBadgeClass(score: number): string { + if (score >= HEALTH_SCORE_GOOD_THRESHOLD) return 'badge-success'; + if (score >= HEALTH_SCORE_WARN_THRESHOLD) return 'badge-warning'; + return 'badge-danger'; +} + +// healthBadgeHtml renders the per-plan health-score badge, distinguishing +// three states the API can report: +// +// - field absent: an older API response served during a partial deploy. +// Render nothing so the card still looks intentional. +// - field null: the backend could not compute the score (execution counts +// unavailable). Render an explicit "unknown" badge -- never a stand-in +// number, which an operator could not tell apart from a measured score. +// - a number within the contract's 0-100 range: the score, banded into the +// green/amber/red palette. +// - anything else (NaN, Infinity, out of range, a non-number off the wire): +// also "unknown". A score outside 0-100 violates the range the API +// declares, so it is no more trustworthy than a NaN, and banding it would +// paint -1 red and 101 green as though both were measured. Silently +// dropping the badge instead would read as "this deploy predates the +// feature". +// +// The tooltip enumerates every penalty factor so a bad score is actionable +// instead of opaque; every factor note is escaped since it renders inside an +// HTML attribute (title) via innerHTML (feedback_innerhtml_xss: all +// API-sourced values must be escaped here even though the backend generates +// these notes itself). +function healthBadgeHtml(plan: BackendPlan): string { + if (plan.health_score === undefined) return ''; + // The typeof/isFinite check is defensive: the type says number, but this + // value comes off the wire and is interpolated into innerHTML below. + if (plan.health_score === null + || typeof plan.health_score !== 'number' + || !Number.isFinite(plan.health_score) + || plan.health_score < HEALTH_SCORE_MIN + || plan.health_score > HEALTH_SCORE_MAX) { + // Neutral wording on purpose: null means the backend could not read the + // execution history, while a non-finite value means the number itself + // arrived unusable. Naming only the first would send an operator + // chasing a database problem that may not exist. + return `Health: unknown`; + } + const factors = plan.health_factors || []; + const factorLines = factors.length > 0 + ? factors.map(f => `-${f.penalty}: ${f.note}`) + : ['No issues detected']; + const tooltip = [`Plan health: ${plan.health_score}/100`, ...factorLines].join('\n'); + return `Health: ${plan.health_score}`; +} + +// Pretty label for a service slug used inside the plan card. +// +// SP slugs are abbreviated ("Compute SP") rather than spelled out +// ("Compute Savings Plans") so a multi-SP plan with 3-4 entries still +// fits in the summary line. Non-SP slugs pass through unchanged so +// existing single-service plans render exactly as before. +function planServiceLabel(slug: string): string { + switch (slug) { + case 'savings-plans-compute': return 'Compute SP'; + case 'savings-plans-ec2instance': return 'EC2 Instance SP'; + case 'savings-plans-sagemaker': return 'SageMaker SP'; + case 'savings-plans-database': return 'Database SP'; + default: return slug; + } +} + +// Extract provider/service info from plan's services map. +// +// `service` is now a comma-separated list of all services covered by +// the plan, not just the first map entry. Pre-PR #123 a plan only ever +// had one service slug; post-split a plan targeting multiple SP plan +// types has up to four entries (savings-plans-{compute,ec2instance, +// sagemaker,database}) but the old "first entry wins" rendering hid +// all but one. See issue #131 for the bug; this fix shows them all. +// +// `term` and `coverage` continue to come from the first entry — they +// are plan-level today, not per-service, so picking any entry is +// correct. If the model ever differentiates per service, this needs +// to render the same way. +function extractPlanInfo(plan: BackendPlan): { provider: string | null; service: string; term: number | null; coverage: number | null } { + const services = plan.services || {}; + const serviceValues = Object.values(services); + const firstService = serviceValues[0]; + if (firstService) { + const service = serviceValues.length === 0 + ? '—' + : serviceValues + .map(s => planServiceLabel(s.service || '—')) + .join(', '); + // H-4: never fabricate provider/term/coverage when absent. + // Return null so callers can surface '--'/'Unknown' in the display + // layer rather than silently committing to aws/3yr/80% defaults. + return { + provider: firstService.provider || null, + service, + term: firstService.term || null, + coverage: firstService.coverage ?? null + }; + } + // No services at all: all fields are unknown. + return { provider: null, service: '—', term: null, coverage: null }; +} + +// Format a Date as YYYY-MM-DD in the user's *local* calendar — the format +// expected by .value and .min. Built from local +// year/month/day components so we don't shift by the user's UTC offset +// (e.g. a US user at 6pm local already crosses into UTC's next day, so +// toISOString() would return the wrong calendar day for a date-picker). +function toLocalDateInputValue(d: Date): string { + const y = d.getFullYear(); + const m = String(d.getMonth() + 1).padStart(2, '0'); + const day = String(d.getDate()).padStart(2, '0'); + return `${y}-${m}-${day}`; +} + +// Returns true when the plan's next_execution_date is strictly before today. +function isPlanOverdue(plan: BackendPlan): boolean { + if (!plan.next_execution_date) return false; + const nextDate = new Date(plan.next_execution_date); + if (isNaN(nextDate.getTime())) return false; + const today = new Date(); + today.setHours(0, 0, 0, 0); + return nextDate < today; +} + +// Format ramp schedule from backend struct +function formatBackendRampSchedule(ramp: BackendPlan['ramp_schedule']): string { + if (!ramp) return 'Immediate'; + switch (ramp.type) { + case 'immediate': return 'Immediate'; + case 'weekly': return `Weekly ${ramp.percent_per_step}%`; + case 'monthly': return `Monthly ${ramp.percent_per_step}%`; + case 'custom': return `Custom ${ramp.percent_per_step}% every ${ramp.step_interval_days} days`; + default: return ramp.type || 'Unknown'; + } +} + +async function loadPlanAccountNames(planId: string, cardEl: Element): Promise { + try { + const accounts = await api.listPlanAccounts(planId); + if (accounts.length === 0) return; + const detailsEl = cardEl.querySelector('.plan-details'); + if (!detailsEl) return; + const names = accounts.map(a => escapeHtml(a.name)).join(', '); + const div = document.createElement('div'); + div.className = 'plan-detail'; + div.innerHTML = `Accounts${names}`; + detailsEl.appendChild(div); + } catch { + // Non-critical --- just do not show account names + } +} + +// renderPlanCard generates the HTML for a single plan card. +// When the plan is unassigned (no plan_accounts rows) account-scoped +// actions that require an account (Add Purchases, Edit) are suppressed +// — only History and Delete remain — and a read-only "Unassigned" badge +// is shown so operators can identify and re-scope the plan. +function renderPlanCard(plan: BackendPlan, canManagePlan: boolean, canDeletePlan: boolean, canAddPurchases: boolean): string { + const info = extractPlanInfo(plan); + const status = getStatusBadge(plan.enabled, plan.auto_purchase); + const rampSchedule = plan.ramp_schedule || { type: 'immediate', current_step: 0, total_steps: 1 }; + const overdue = isPlanOverdue(plan); + // Hide the stale next_execution_date for disabled plans — keeping it + // visible implies the plan will still run on that date, which it won't. + const showNextDate = Boolean(plan.next_execution_date) && plan.enabled; + const overdueBadge = overdue && plan.enabled + ? 'Overdue' + : ''; + // Read-only mode: unassigned plans cannot be purchased against until an + // operator re-assigns them to at least one account. + const isUnassigned = Boolean(plan.unassigned); + + return ` +
+
+

${escapeHtml(plan.name)}

+
+ ${status.label} + ${healthBadgeHtml(plan)} + ${overdueBadge} + ${canManagePlan && !isUnassigned ? ` + + ` : ''} +
+
+
+
+
+ Provider + ${providerBadgeHtml(info.provider)} +
+
+ Service + ${escapeHtml(info.service)} +
+
+ Term + ${info.term !== null ? formatTerm(info.term) : '--'} +
+
+ Coverage + ${info.coverage !== null ? `${info.coverage}%` : '--'} +
+
+ Ramp Schedule + ${formatBackendRampSchedule(rampSchedule)} +
+
+ Progress + ${rampSchedule.current_step || 0}/${rampSchedule.total_steps || 1} steps +
+ ${showNextDate ? ` +
+ Next Purchase + ${formatDate(plan.next_execution_date || '')} +
+ ` : ''} +
+
+ ${canAddPurchases && !isUnassigned ? `` : ''} + ${canManagePlan && !isUnassigned ? `` : ''} + + ${canDeletePlan ? `` : ''} +
+
+
+ `; +} + +function renderPlans(plans: LocalPlan[]): void { + const container = document.getElementById('plans-list'); + if (!container) return; + + if (!plans || plans.length === 0) { + container.innerHTML = '

No purchase plans configured. Create one to automate your commitment purchases.

'; + return; + } + + // Issue #365: cache permission checks once per render rather than + // per-card so a 100-plan list doesn't bounce through the helper + // 600 times. Action buttons hidden for sessions that lack the verb. + const canManagePlan = canAccess('update', 'plans'); + const canDeletePlan = canAccess('delete', 'plans'); + // Issue #1406 / #1418: "Add Purchases" requires BOTH plan-management AND + // purchase-write permission. Plan Authors hold update:plans but not + // update:purchases; Standard Users hold both. Gate the button on the + // conjunction so Plan Authors cannot schedule purchases. + const canAddPurchases = canManagePlan && canAccess('update', 'purchases'); + + // Issue #973: split plans into assigned (have plan_accounts rows) and + // unassigned (legacy plans with zero plan_accounts rows, flagged by the + // backend). Unassigned plans are rendered under a separate read-only + // section so operators can discover and re-scope them. + const assignedPlans = plans.filter(p => !(p as unknown as BackendPlan).unassigned); + const unassignedPlans = plans.filter(p => (p as unknown as BackendPlan).unassigned); + + const assignedHtml = assignedPlans.map(rawPlan => + renderPlanCard(rawPlan as unknown as BackendPlan, canManagePlan, canDeletePlan, canAddPurchases) + ).join(''); + + let unassignedHtml = ''; + if (unassignedPlans.length > 0) { + const cards = unassignedPlans.map(rawPlan => + renderPlanCard(rawPlan as unknown as BackendPlan, canManagePlan, canDeletePlan, canAddPurchases) + ).join(''); + unassignedHtml = ` +
+

Unassigned

+ These legacy plans have no associated accounts and cannot be purchased against. Assign accounts or delete them. +
+ ${cards} + `; + } + + container.innerHTML = assignedHtml + unassignedHtml; + + // Asynchronously populate account names per assigned plan card. + // Unassigned plans intentionally skip this: they have no plan_accounts + // rows so the API call would return an empty list. + container.querySelectorAll('.plan-card').forEach((card) => { + const planId = card.querySelector('[data-id]')?.dataset['id']; + // Find the matching plan to check unassigned status. + const allPlans = [...assignedPlans, ...unassignedPlans]; + const matchedPlan = allPlans.find(p => (p as unknown as BackendPlan).id === planId) as unknown as BackendPlan | undefined; + if (planId && matchedPlan && !matchedPlan.unassigned) { + void loadPlanAccountNames(planId, card); + } + }); + + // Add event listeners + container.querySelectorAll('[data-action="toggle-plan"]').forEach(toggle => { + toggle.addEventListener('change', () => void togglePlan(toggle.dataset['id'] || '', toggle.checked)); + }); + container.querySelectorAll('[data-action="add-purchases"]').forEach(btn => { + btn.addEventListener('click', () => void openAddPurchasesModal(btn.dataset['id'] || '', btn.dataset['name'] || '')); + }); + container.querySelectorAll('[data-action="edit-plan"]').forEach(btn => { + btn.addEventListener('click', () => void editPlan(btn.dataset['id'] || '')); + }); + container.querySelectorAll('[data-action="view-history"]').forEach(btn => { + btn.addEventListener('click', () => void viewPlanHistory(btn.dataset['id'] || '')); + }); + container.querySelectorAll('[data-action="delete-plan"]').forEach(btn => { + btn.addEventListener('click', () => void deletePlanAction(btn.dataset['id'] || '')); + }); +} + +async function togglePlan(planId: string, enabled: boolean): Promise { + try { + await api.patchPlan(planId, { enabled } as Partial); + await loadPlans(); + } catch (error) { + console.error('Failed to toggle plan:', error); + showToast({ message: 'Failed to update plan', kind: 'error' }); + await loadPlans(); + } +} + +// editPlan loads the plan and opens the edit modal pre-filled. Returns true +// when the modal opened, false when the plan could not be loaded (e.g. it was +// deleted or is no longer accessible). Callers that render the plan in a list +// use the false result to reconcile a now-stale row (issue #1403). +async function editPlan(planId: string): Promise { + try { + const backendPlan = await api.getPlan(planId) as unknown as BackendPlan; + + // Extract info from the backend plan format + const info = extractPlanInfo(backendPlan); + const rampSchedule = backendPlan.ramp_schedule || { type: 'immediate', percent_per_step: 100, step_interval_days: 0 }; + + // Map ramp schedule type to frontend value + let rampValue = 'immediate'; + if (rampSchedule.type === 'weekly' && rampSchedule.percent_per_step === 25) { + rampValue = 'weekly-25pct'; + } else if (rampSchedule.type === 'monthly' && rampSchedule.percent_per_step === 10) { + rampValue = 'monthly-10pct'; + } else if (rampSchedule.type === 'custom' || (rampSchedule.type !== 'immediate' && rampSchedule.type !== 'weekly' && rampSchedule.type !== 'monthly')) { + rampValue = 'custom'; + } + + // Get payment option from services and normalize for provider. + // H-5: do not pre-select 'no-upfront' when the payment field is absent. + // An absent payment field leaves the select empty so the user must + // explicitly choose rather than silently inheriting a fabricated default + // that could change the plan's payment type on re-save. + const firstService = Object.values(backendPlan.services || {})[0]; + const rawPayment = firstService?.payment || null; + const payment = rawPayment !== null ? normalizePaymentValue(rawPayment, info.provider ?? '') : ''; + + const titleEl = document.getElementById('plan-modal-title'); + if (titleEl) titleEl.textContent = 'Edit Purchase Plan'; + + (document.getElementById('plan-id') as HTMLInputElement).value = backendPlan.id; + (document.getElementById('plan-name') as HTMLInputElement).value = backendPlan.name; + (document.getElementById('plan-description') as HTMLTextAreaElement).value = ''; + + // Set provider and service first. When provider is absent from the API + // response, leave the select at its default empty/first option rather + // than fabricating 'aws' (H-4: never default provider silently). + const providerSelect = document.getElementById('plan-provider') as HTMLSelectElement; + providerSelect.value = info.provider ?? ''; + (document.getElementById('plan-service') as HTMLSelectElement).value = info.service; + + // Update term/payment options based on provider/service + const termSelect = document.getElementById('plan-term') as HTMLSelectElement; + const paymentSelect = document.getElementById('plan-payment') as HTMLSelectElement; + populateTermSelect(termSelect, info.provider ?? '', info.service); + populatePaymentSelect(paymentSelect, info.provider ?? '', info.service); + + // Set term only when present; absent term leaves the select unset so + // the user must explicitly choose rather than silently inheriting a + // fabricated 3yr default (H-4). + termSelect.value = info.term !== null ? String(info.term) : ''; + + paymentSelect.value = payment; + + // Set coverage only when present; absent coverage leaves the input + // blank so the user sees the field is missing (H-4). + (document.getElementById('plan-coverage') as HTMLInputElement).value = info.coverage !== null ? String(info.coverage) : ''; + (document.getElementById('plan-auto-purchase') as HTMLInputElement).checked = backendPlan.auto_purchase; + (document.getElementById('plan-notify-days') as HTMLInputElement).value = String(backendPlan.notification_days_before || 3); + (document.getElementById('plan-enabled') as HTMLInputElement).checked = backendPlan.enabled; + + const rampRadio = document.querySelector(`input[name="ramp-schedule"][value="${rampValue}"]`); + if (rampRadio) rampRadio.checked = true; + + const customConfig = document.getElementById('custom-ramp-config'); + if (customConfig) { + customConfig.classList.toggle('hidden', rampValue !== 'custom'); + } + + if (rampValue === 'custom') { + (document.getElementById('ramp-step-percent') as HTMLInputElement).value = String(rampSchedule.percent_per_step || 20); + (document.getElementById('ramp-interval-days') as HTMLInputElement).value = String(rampSchedule.step_interval_days || 7); + } + + void setupPlanAccountsSection(backendPlan.id); + // Wire live range validation on all five numeric inputs (#702). + wirePlanRangeInputs(); + const planModal = document.getElementById('plan-modal'); + if (planModal) openModal(planModal); + return true; + } catch (error) { + console.error('Failed to load plan:', error); + // A missing/inaccessible plan comes back as 404 (issue #1403): the + // scheduled-purchase row outlived its plan (deleted, or the caller's + // account scope changed). Surface an actionable message instead of the + // generic "Failed to load plan details", and signal failure so the + // caller can reconcile the stale row. Other errors keep the generic + // message (network blip, transient 5xx) since the plan may still exist. + const status = (error as { status?: number }).status; + const message = status === 404 + ? 'This plan is no longer available. It may have been deleted.' + : 'Failed to load plan details'; + showToast({ message, kind: 'error' }); + return false; + } +} + +async function deletePlanAction(planId: string): Promise { + const ok = await confirmDialog({ + title: 'Delete this plan?', + body: 'This removes the plan and cancels all its scheduled purchases. This action cannot be undone.', + confirmLabel: 'Delete plan', + destructive: true, + }); + if (!ok) return; + + try { + await api.deletePlan(planId); + await loadPlans(); + showToast({ message: 'Plan deleted', kind: 'success', timeout: 5_000 }); + } catch (error) { + console.error('Failed to delete plan:', error); + showToast({ message: 'Failed to delete plan', kind: 'error' }); + } +} + +/** + * Save plan (create or update) + */ +export async function savePlan(e: Event): Promise { + e.preventDefault(); + + const planId = (document.getElementById('plan-id') as HTMLInputElement).value; + const rampScheduleRadio = document.querySelector('input[name="ramp-schedule"]:checked'); + const rampSchedule = rampScheduleRadio?.value || 'immediate'; + + // Parse and validate integer fields up front. Use Number() not parseInt so + // fractions like "2.5" fail Number.isInteger() rather than silently truncating + // to 2. Mirrors the strict parse pattern from handleAddPurchases and settings.ts + // validatePurchasingSettings (feedback_strict_int_parse, finding 11-M1). + const rawTerm = Number((document.getElementById('plan-term') as HTMLSelectElement).value); + const rawCoverage = Number((document.getElementById('plan-coverage') as HTMLInputElement).value); + const rawNotifyDays = Number((document.getElementById('plan-notify-days') as HTMLInputElement).value); + + if (!Number.isFinite(rawTerm) || !Number.isInteger(rawTerm) || rawTerm < 1) { + showToast({ message: 'Term must be a valid whole number of years', kind: 'error' }); + return; + } + if (!Number.isFinite(rawCoverage) || !Number.isInteger(rawCoverage) || rawCoverage < 0 || rawCoverage > 100) { + showToast({ message: 'Target Coverage must be a whole number between 0 and 100', kind: 'error' }); + return; + } + if (!Number.isFinite(rawNotifyDays) || !Number.isInteger(rawNotifyDays) || rawNotifyDays < 1 || rawNotifyDays > 30) { + showToast({ message: 'Notification Days must be a whole number between 1 and 30', kind: 'error' }); + return; + } + + const plan: SavePlanData = { + name: (document.getElementById('plan-name') as HTMLInputElement).value, + description: (document.getElementById('plan-description') as HTMLTextAreaElement).value, + provider: (document.getElementById('plan-provider') as HTMLSelectElement).value, + service: (document.getElementById('plan-service') as HTMLSelectElement).value, + term: rawTerm, + payment: (document.getElementById('plan-payment') as HTMLSelectElement).value, + target_coverage: rawCoverage, + ramp_schedule: rampSchedule, + auto_purchase: (document.getElementById('plan-auto-purchase') as HTMLInputElement).checked, + notification_days_before: rawNotifyDays, + enabled: (document.getElementById('plan-enabled') as HTMLInputElement).checked + }; + + if (rampSchedule === 'custom') { + const rawStepPercent = Number((document.getElementById('ramp-step-percent') as HTMLInputElement).value); + const rawIntervalDays = Number((document.getElementById('ramp-interval-days') as HTMLInputElement).value); + if (!Number.isFinite(rawStepPercent) || !Number.isInteger(rawStepPercent) || rawStepPercent < 1 || rawStepPercent > 100) { + showToast({ message: 'Ramp Step Percent must be a whole number between 1 and 100', kind: 'error' }); + return; + } + if (!Number.isFinite(rawIntervalDays) || !Number.isInteger(rawIntervalDays) || rawIntervalDays < 1 || rawIntervalDays > 365) { + showToast({ message: 'Ramp Interval Days must be a whole number between 1 and 365', kind: 'error' }); + return; + } + plan.custom_step_percent = rawStepPercent; + plan.custom_interval_days = rawIntervalDays; + } + + // Use the snapshot stamped at Plan-button click time (#273 CR follow-up). + // Reading state.getVisibleRecommendations() / getSelectedRecommendation + // IDs() here would re-derive the target at Save time — racing Refresh, + // filter changes, and deselections that happen while the modal is open. + // openCreatePlanModal(snapshot) freezes the Plan target the moment the + // user clicked "Plan from N selected"; we read it back here. + // openNewPlanModal() clears the snapshot for the New-Plan-from-scratch + // path (no pre-resolved target — plan submits without `recommendations`). + if (pendingPlanRecommendations.length > 0) { + plan.recommendations = [...pendingPlanRecommendations]; + } + + // Universal-plans fix: read the selected account chips and reject submit + // when the list is empty. The Save button is also disabled in the same + // condition via refreshPlanSaveButtonState() so this branch is mostly a + // belt-and-suspenders against scripted form submission; the toast keeps + // the failure mode loud either way. + const accountIdsField = document.getElementById('plan-account-ids') as HTMLInputElement | null; + const accountIds = accountIdsField?.value ? accountIdsField.value.split(',').filter(Boolean) : []; + if (accountIds.length === 0) { + showToast({ + message: 'Target Accounts is required: pick at least one account before saving the plan.', + kind: 'error', + }); + return; + } + plan.target_accounts = accountIds; + + try { + if (planId) { + await api.updatePlan(planId, plan as unknown as api.CreatePlanRequest); + // Update flow: push the selected account list via the dedicated endpoint. + await api.setPlanAccounts(planId, accountIds); + } else { + await api.createPlan(plan as unknown as api.CreatePlanRequest); + // Create flow: backend inserted plan_accounts atomically from + // target_accounts in the POST body (see internal/api/handler_plans.go + // createPlan). No follow-up account-write needed. + } + + closePlanModal(); + await loadPlans(); + showToast({ message: planId ? 'Plan updated successfully' : 'Plan created successfully', kind: 'success', timeout: 5_000 }); + } catch (error) { + console.error('Failed to save plan:', error); + const err = error as Error; + showToast({ message: `Failed to save plan: ${err.message}`, kind: 'error' }); + } +} + +// refreshPlanSaveButtonState toggles the Save button's disabled state based +// on whether at least one Target Account is selected. Universal plans (rows +// in purchase_plans with no plan_accounts row) are no longer allowed by the +// API; surfacing the failure in the disabled state is friendlier than +// letting the user fill in every other field and get rejected at submit. +function refreshPlanSaveButtonState(): void { + const form = document.getElementById('plan-form') as HTMLFormElement | null; + if (!form) return; + const submitBtn = form.querySelector('button[type="submit"]'); + if (!submitBtn) return; + const hasAccounts = planSelectedAccounts.length > 0; + submitBtn.disabled = !hasAccounts; + submitBtn.title = hasAccounts ? '' : 'Select at least one Target Account to save the plan'; +} + +/** + * Close plan modal + */ +export function closePlanModal(): void { + const planModal = document.getElementById('plan-modal'); + if (planModal) closeModal(planModal); + // Invalidate the resolved-target snapshot stamped by openCreatePlanModal + // so a subsequent flow doesn't accidentally inherit it. The snapshot + // ties the plan to a specific button-click moment; once the modal + // closes — by Save, Cancel, or any other path — that moment is over. + pendingPlanRecommendations = []; +} + +// Selected accounts for the plan modal +let planSelectedAccounts: Array<{ id: string; name: string; external_id: string }> = []; + +// Monotonically incrementing counter scoped to the plan modal lifecycle. +// Incremented each time the create modal opens so that async callbacks +// from a previous session (stale promises) can detect they're out-of-date +// and discard their results rather than mutating state in the new session. +let planModalSession = 0; + +/** + * Render selected account chips in the plan modal + */ +function renderPlanAccountChips(): void { + const container = document.getElementById('plan-accounts-selected'); + if (!container) return; + container.textContent = ''; + planSelectedAccounts.forEach(acct => { + const chip = document.createElement('span'); + chip.className = 'account-chip'; + chip.textContent = `${acct.name} (${acct.external_id})`; + + const removeBtn = document.createElement('button'); + removeBtn.type = 'button'; + removeBtn.textContent = '\u00d7'; + removeBtn.addEventListener('click', () => { + planSelectedAccounts = planSelectedAccounts.filter(a => a.id !== acct.id); + renderPlanAccountChips(); + updatePlanAccountIdsField(); + }); + chip.appendChild(removeBtn); + container.appendChild(chip); + }); +} + +/** + * Update hidden plan-account-ids field + */ +function updatePlanAccountIdsField(): void { + const field = document.getElementById('plan-account-ids') as HTMLInputElement | null; + if (field) field.value = planSelectedAccounts.map(a => a.id).join(','); + // Recalc Save-button disabled state every time the account list changes + // so the user gets immediate feedback when they remove the last chip. + refreshPlanSaveButtonState(); +} + +let planAccountSearchTimer: ReturnType | null = null; + +/** + * Handle plan account search input + */ +async function handlePlanAccountSearch(value: string): Promise { + const suggestions = document.getElementById('plan-account-suggestions'); + if (!suggestions) return; + + if (!value.trim()) { + suggestions.classList.add('hidden'); + return; + } + + try { + const providerSelect = document.getElementById('plan-provider') as HTMLSelectElement | null; + const provider = providerSelect?.value as api.Provider | undefined; + // Minimal-disclosure list (view:recommendations) so plan-account search + // works for Standard / Read-Only users; the full view:accounts list 403s + // for them. See issues #949/#951. + const accounts = await api.listAccountsMinimal({ search: value, ...(provider ? { provider } : {}) }); + suggestions.textContent = ''; + if (accounts.length === 0) { + suggestions.classList.add('hidden'); + return; + } + accounts.forEach(a => { + if (planSelectedAccounts.some(s => s.id === a.id)) return; + const item = document.createElement('div'); + item.className = 'account-suggestion-item'; + item.textContent = `${a.name} (${a.external_id})`; + item.addEventListener('click', () => { + planSelectedAccounts.push({ id: a.id, name: a.name, external_id: a.external_id }); + renderPlanAccountChips(); + updatePlanAccountIdsField(); + suggestions.classList.add('hidden'); + (document.getElementById('plan-account-search') as HTMLInputElement).value = ''; + }); + suggestions.appendChild(item); + }); + suggestions.classList.remove('hidden'); + } catch { + suggestions.classList.add('hidden'); + } +} + +/** + * Set up plan accounts section in the modal + */ +async function setupPlanAccountsSection(planId?: string): Promise { + planSelectedAccounts = []; + + const planProvider = (document.getElementById('plan-provider') as HTMLSelectElement | null)?.value; + + if (planId) { + try { + const existingAccounts = await api.listPlanAccounts(planId); + // Filter out any account whose provider does not match the current plan + // provider. This prevents stale cross-provider assignments from silently + // surviving a provider switch on an existing plan. + planSelectedAccounts = existingAccounts + .filter(a => !planProvider || a.provider === planProvider) + .map(a => ({ id: a.id, name: a.name, external_id: a.external_id })); + } catch { + // Non-critical — section just starts empty + } + } + + renderPlanAccountChips(); + updatePlanAccountIdsField(); + + const searchInput = document.getElementById('plan-account-search') as HTMLInputElement | null; + if (searchInput) { + // Disable the search input until a provider is selected. + searchInput.disabled = !planProvider; + + // Remove previous listeners by replacing node + const newInput = searchInput.cloneNode(true) as HTMLInputElement; + searchInput.parentNode?.replaceChild(newInput, searchInput); + newInput.addEventListener('input', () => { + if (planAccountSearchTimer) clearTimeout(planAccountSearchTimer); + planAccountSearchTimer = setTimeout(() => { + void handlePlanAccountSearch(newInput.value); + }, 300); + }); + } +} + +/** + * Prefill the Purchase Configuration section (provider / service / term / + * payment) from a representative selected commitment. Called after + * form.reset() so the defaults are already in place; each field is still + * editable. (#770) + * + * For a multi-commitment selection the caller passes the first commitment as + * the representative: the "Plan from N selected" button is only enabled when + * the selection is homogeneous (same provider/service/term/payment — see + * isHomogeneousSelection in recommendations.ts), so any element shares those + * four values. (#898) + */ +function prefillPurchaseConfigFromCommitment(rec: api.Recommendation): void { + const providerSelect = document.getElementById('plan-provider') as HTMLSelectElement | null; + const serviceSelect = document.getElementById('plan-service') as HTMLSelectElement | null; + const termSelect = document.getElementById('plan-term') as HTMLSelectElement | null; + const paymentSelect = document.getElementById('plan-payment') as HTMLSelectElement | null; + + if (!providerSelect || !serviceSelect || !termSelect || !paymentSelect) return; + + const provider = rec.provider ?? ''; + const service = rec.service ?? ''; + + if (provider) providerSelect.value = provider; + if (service) serviceSelect.value = service; + + // Repopulate term/payment options for the chosen provider+service, then + // apply the commitment's own values so the dropdowns are consistent. + if (provider && service) { + populateTermSelect(termSelect, provider, service); + populatePaymentSelect(paymentSelect, provider, service); + } + + if (rec.term != null) termSelect.value = String(rec.term); + const normalizedPayment = rec.payment ? normalizePaymentValue(rec.payment, provider) : ''; + if (normalizedPayment) paymentSelect.value = normalizedPayment; +} + +/** + * Fetch the account with the given internal UUID and add it as a pre-selected + * chip in the plan modal accounts section. Runs after setupPlanAccountsSection + * has reset the chip list for the create flow. Silently no-ops on failure so + * the user can still pick the account manually. (#770) + * + * @param accountId - Internal UUID of the account to prefill. + * @param session - planModalSession value captured at call time. If the + * modal is closed and reopened before this promise resolves, + * the counter will have advanced and the stale result is + * discarded to prevent wrong-modal pollution. (#770 CR) + */ +async function prefillAccountChipFromId(accountId: string, session: number): Promise { + try { + // Resolve via the minimal-disclosure list (view:recommendations) instead of + // GET /api/accounts/:id (view:accounts) so the target prefills for + // Standard / Read-Only users too — the per-id endpoint 403s for them, which + // previously left the target empty and made Save Plan silently no-op once + // the empty-target guard rejected submit. See issues #949/#951. + const account = (await api.listAccountsMinimal()).find((a) => a.id === accountId); + // Session guard: discard the result if the modal was closed and reopened + // while this promise was in-flight. planModalSession is incremented on + // each new modal open, so a mismatch means this callback is stale. + if (session !== planModalSession) return; + // Guard: only add if the chip is not already present (e.g. a concurrent + // edit flow somehow set it) and the account record is usable. + if (account && account.id && !planSelectedAccounts.some(a => a.id === account.id)) { + planSelectedAccounts.push({ id: account.id, name: account.name, external_id: account.external_id }); + renderPlanAccountChips(); + updatePlanAccountIdsField(); + } + } catch { + // Non-critical: the user can still search and add the account manually. + } +} + +/** + * Open create plan modal with selected recommendations. + * + * When the user has no selection (issue #17 reproducer: filter + * active, no checkboxes ticked), fall through to the plain new-plan + * flow instead of silently noop-ing behind a toast the user may + * miss. Same UX as the dedicated "New Plan" button — the modal + * always opens, and the user fills in provider/service from scratch. + */ +export function openCreatePlanModal(snapshot?: readonly api.Recommendation[]): void { + // Stamp the resolved-target snapshot from the caller so savePlan can + // consume it without re-deriving from global state at Save time. The + // Bottom Action Box passes the result of resolvePurchaseTarget(); a + // missing arg falls back to the legacy behaviour (no captured target, + // savePlan submits a plan without `recommendations`). See #273 CR. + pendingPlanRecommendations = snapshot ? [...snapshot] : []; + + const titleEl = document.getElementById('plan-modal-title'); + const hasSelection = pendingPlanRecommendations.length > 0; + if (titleEl) { + titleEl.textContent = hasSelection ? 'Create Purchase Plan' : 'New Purchase Plan'; + } + (document.getElementById('plan-id') as HTMLInputElement).value = ''; + (document.getElementById('plan-form') as HTMLFormElement | null)?.reset(); + + // When one or more commitments are selected, prefill the Purchase + // Configuration fields so the user does not have to re-enter them. The + // first commitment is the representative: for a multi-selection the + // "Plan from N selected" button only enables on a homogeneous selection, + // so every element shares provider/service/term/payment. Fields are still + // fully editable after prefill. (#770, #898) + if (pendingPlanRecommendations.length >= 1) { + prefillPurchaseConfigFromCommitment(pendingPlanRecommendations[0]!); + } + + // Set up ramp schedule change handlers for dynamic plan name + setupRampScheduleHandlers(); + + // Wire live range validation on all five numeric inputs (#702). + wirePlanRangeInputs(); + + // Generate initial plan name + updatePlanNameFromSchedule(); + + // Stamp a new session so any in-flight prefillAccountChipFromId promise + // from a prior modal open can detect it belongs to a stale session and + // discard its result without mutating planSelectedAccounts. (#770 CR) + planModalSession += 1; + + // setupPlanAccountsSection clears planSelectedAccounts and re-renders. + // When every selected commitment carries the SAME cloud_account_id, we look + // up that account after the section has reset and add it as a pre-selected + // chip. Homogeneity of provider/service/term/payment (enforced by the + // "Plan from N selected" gate) does not imply a single account, so a + // multi-account selection leaves the chip empty for the user to fill. (#898) + void setupPlanAccountsSection(); + if (pendingPlanRecommendations.length >= 1) { + const firstAccountId = pendingPlanRecommendations[0]!.cloud_account_id; + const sharedAccountId = + firstAccountId && + pendingPlanRecommendations.every((r) => r.cloud_account_id === firstAccountId) + ? firstAccountId + : undefined; + if (sharedAccountId) { + void prefillAccountChipFromId(sharedAccountId, planModalSession); + } + } + + const planModal = document.getElementById('plan-modal'); + if (planModal) { + openModal(planModal); + } +} + +/** + * Open new plan modal (without pre-selected recommendations) + */ +export function openNewPlanModal(): void { + // No pre-resolved target — this is the "New Plan from scratch" path. + // Clear any stale snapshot from a prior openCreatePlanModal call so a + // subsequent savePlan doesn't accidentally inherit a previous flow's + // recs. See #273 CR. + pendingPlanRecommendations = []; + + const titleEl = document.getElementById('plan-modal-title'); + if (titleEl) titleEl.textContent = 'New Purchase Plan'; + (document.getElementById('plan-id') as HTMLInputElement).value = ''; + (document.getElementById('plan-form') as HTMLFormElement | null)?.reset(); + + // Set up ramp schedule change handlers for dynamic plan name + setupRampScheduleHandlers(); + + // Wire live range validation on all five numeric inputs (#702). + wirePlanRangeInputs(); + + // Generate initial plan name + updatePlanNameFromSchedule(); + + void setupPlanAccountsSection(); + + const planModal = document.getElementById('plan-modal'); + if (planModal) { + openModal(planModal); + } +} + +/** + * Generate a plan name based on the selected ramp schedule + */ +function generatePlanName(rampSchedule: string, customStepPercent?: number, customIntervalDays?: number): string { + const service = (document.getElementById('plan-service') as HTMLSelectElement)?.value || 'EC2'; + const serviceUpper = service.toUpperCase(); + + switch (rampSchedule) { + case 'immediate': + return `${serviceUpper} Full Coverage Purchase`; + case 'weekly-25pct': + return `${serviceUpper} Weekly 25% Ramp-up (4 weeks)`; + case 'monthly-10pct': + return `${serviceUpper} Monthly 10% Ramp-up (10 months)`; + case 'custom': + if (customStepPercent && customIntervalDays) { + const totalSteps = Math.ceil(100 / customStepPercent); + const intervalLabel = customIntervalDays === 7 ? 'weekly' : + customIntervalDays === 30 ? 'monthly' : + `every ${customIntervalDays} days`; + return `${serviceUpper} Custom ${customStepPercent}% ${intervalLabel} (${totalSteps} steps)`; + } + return `${serviceUpper} Custom Ramp-up Plan`; + default: + return `${serviceUpper} Purchase Plan`; + } +} + +/** + * Update plan name field based on current ramp schedule selection + */ +function updatePlanNameFromSchedule(): void { + const planNameInput = document.getElementById('plan-name') as HTMLInputElement; + const planIdInput = document.getElementById('plan-id') as HTMLInputElement; + + // Only auto-generate name for new plans (not editing existing ones) + if (planIdInput?.value) return; + + const rampScheduleRadio = document.querySelector('input[name="ramp-schedule"]:checked'); + const rampSchedule = rampScheduleRadio?.value || 'immediate'; + + let customStepPercent: number | undefined; + let customIntervalDays: number | undefined; + + if (rampSchedule === 'custom') { + customStepPercent = parseInt((document.getElementById('ramp-step-percent') as HTMLInputElement)?.value || '20', 10); + customIntervalDays = parseInt((document.getElementById('ramp-interval-days') as HTMLInputElement)?.value || '7', 10); + } + + if (planNameInput) { + planNameInput.value = generatePlanName(rampSchedule, customStepPercent, customIntervalDays); + } +} + +/** + * Wire live range + integer-only validation on a plan numeric input. + * + * Registers `input` and `blur` event handlers that: + * - reject non-integer values via the regex `^\d+$` (blocks scientific + * notation such as `1e+30` and decimal fractions) + * - show a sibling `.field-error` span when the value is out of [min, max] + * or non-integer, and hide it when the value is valid or the field is empty + * - set / clear `aria-invalid` for screen-reader accessibility + * + * Idempotent: the `data-range-wired` attribute guards against re-registering + * duplicate listeners when the modal is closed and reopened. The error span + * is created once on the first call and reused thereafter. Explicit `min` / + * `max` parameters are used instead of reading HTML attributes so callers are + * the source of truth and the helper stays self-contained. + */ +function wireRangeInput(inputId: string, min: number, max: number): void { + const input = document.getElementById(inputId) as HTMLInputElement | null; + if (!input) return; + + // Idempotency guard: skip registration if already wired during a prior + // modal open. The error span, aria-describedby, and listeners persist + // across opens because the modal node stays in the DOM. + if (input.dataset['rangeWired']) { + // Re-trigger validation so stale error UI is reconciled on modal reopen. + input.dispatchEvent(new Event('input')); + return; + } + input.dataset['rangeWired'] = '1'; + + const errorId = `${inputId}-range-error`; + let errorEl = document.getElementById(errorId); + if (!errorEl) { + errorEl = document.createElement('small'); + errorEl.id = errorId; + errorEl.className = 'field-error hidden'; + errorEl.setAttribute('role', 'status'); + errorEl.setAttribute('aria-live', 'polite'); + input.insertAdjacentElement('afterend', errorEl); + const existing = input.getAttribute('aria-describedby'); + input.setAttribute( + 'aria-describedby', + existing ? `${existing} ${errorId}` : errorId, + ); + } + const error = errorEl; + const message = `Must be a whole number between ${min} and ${max}`; + const integerPattern = /^\d+$/; + + const check = (): void => { + const raw = input.value.trim(); + if (raw === '') { + input.removeAttribute('aria-invalid'); + error.classList.add('hidden'); + return; + } + if (!integerPattern.test(raw)) { + input.setAttribute('aria-invalid', 'true'); + error.textContent = message; + error.classList.remove('hidden'); + return; + } + const parsed = parseInt(raw, 10); + if (parsed < min || parsed > max) { + input.setAttribute('aria-invalid', 'true'); + error.textContent = message; + error.classList.remove('hidden'); + } else { + input.removeAttribute('aria-invalid'); + error.classList.add('hidden'); + } + }; + + const clampOnBlur = (): void => { + const raw = input.value.trim(); + if (!integerPattern.test(raw) || raw === '') return; + const parsed = parseInt(raw, 10); + if (parsed < min || parsed > max) { + input.value = String(Math.min(max, Math.max(min, parsed))); + input.removeAttribute('aria-invalid'); + error.classList.add('hidden'); + } + }; + + input.addEventListener('input', check); + input.addEventListener('blur', clampOnBlur); +} + +/** + * Wire live range validation on all five plan-creation number inputs. + * Called every time the plan modal opens so validation is active for both + * the create and edit flows. + */ +function wirePlanRangeInputs(): void { + wireRangeInput('plan-coverage', 0, 100); + wireRangeInput('ramp-step-percent', 1, 100); + wireRangeInput('ramp-interval-days', 1, 365); + wireRangeInput('plan-notify-days', 1, 30); +} + +/** + * Set up event handlers for ramp schedule changes. + * Guarded by rampHandlersInstalled so re-opening the plan modal does not + * stack duplicate listeners on the static modal elements (H3, feedback_event_listener_dedup). + */ +function setupRampScheduleHandlers(): void { + if (rampHandlersInstalled) return; + rampHandlersInstalled = true; + + // Listen to ramp schedule radio changes + document.querySelectorAll('input[name="ramp-schedule"]').forEach(radio => { + radio.addEventListener('change', () => { + // Update custom config fields based on selected preset + updateCustomConfigFromPreset(radio.value); + + updatePlanNameFromSchedule(); + + // Show/hide custom config + const customConfig = document.getElementById('custom-ramp-config'); + if (customConfig) { + customConfig.classList.toggle('hidden', radio.value !== 'custom'); + } + }); + }); + + // Listen to custom schedule field changes + const stepPercentInput = document.getElementById('ramp-step-percent'); + const intervalDaysInput = document.getElementById('ramp-interval-days'); + + stepPercentInput?.addEventListener('input', updatePlanNameFromSchedule); + intervalDaysInput?.addEventListener('input', updatePlanNameFromSchedule); + + // Listen to provider/service changes to update payment/term options + const providerSelect = document.getElementById('plan-provider') as HTMLSelectElement | null; + const serviceSelect = document.getElementById('plan-service') as HTMLSelectElement | null; + const termSelect = document.getElementById('plan-term') as HTMLSelectElement | null; + const paymentSelect = document.getElementById('plan-payment') as HTMLSelectElement | null; + + providerSelect?.addEventListener('change', () => { + updateCommitmentOptions(); + updatePlanNameFromSchedule(); + // Clear all selected accounts when the provider changes. Selected account + // entries only carry id/name/external_id (no provider field), so we cannot + // filter by provider directly. Clearing on change is the safe default: + // an account valid for the old provider is almost never valid for the new + // one, and it prevents a cross-provider assignment from silently reaching + // the backend validator. + planSelectedAccounts = []; + renderPlanAccountChips(); + updatePlanAccountIdsField(); + // Clear and hide any open account suggestion dropdown so stale suggestions + // from the previous provider cannot be clicked and add a mismatched account. + const suggestions = document.getElementById('plan-account-suggestions') as HTMLElement | null; + if (suggestions) { + suggestions.textContent = ''; + suggestions.classList.add('hidden'); + } + // Re-enable/disable the account search input to match the new provider state. + const accountSearchInput = document.getElementById('plan-account-search') as HTMLInputElement | null; + if (accountSearchInput) { + accountSearchInput.value = ''; + accountSearchInput.disabled = !providerSelect.value; + } + }); + + serviceSelect?.addEventListener('change', () => { + updateCommitmentOptions(); + updatePlanNameFromSchedule(); + }); + + termSelect?.addEventListener('change', () => { + updatePaymentOptionsForTerm(); + }); + + paymentSelect?.addEventListener('change', () => { + updateTermOptionsForPayment(); + }); + + // Initialize options based on default provider/service + updateCommitmentOptions(); +} + +/** + * Update term and payment options based on current provider/service selection + */ +function updateCommitmentOptions(): void { + const providerSelect = document.getElementById('plan-provider') as HTMLSelectElement | null; + const serviceSelect = document.getElementById('plan-service') as HTMLSelectElement | null; + const termSelect = document.getElementById('plan-term') as HTMLSelectElement | null; + const paymentSelect = document.getElementById('plan-payment') as HTMLSelectElement | null; + + if (!providerSelect || !serviceSelect || !termSelect || !paymentSelect) return; + + const provider = providerSelect.value; + const service = serviceSelect.value; + + // Populate both selects with provider/service specific options + populateTermSelect(termSelect, provider, service); + populatePaymentSelect(paymentSelect, provider, service); + + // Validate current selection + validateAndFixCombination(); +} + +/** + * Update payment options based on selected term + */ +function updatePaymentOptionsForTerm(): void { + const providerSelect = document.getElementById('plan-provider') as HTMLSelectElement | null; + const serviceSelect = document.getElementById('plan-service') as HTMLSelectElement | null; + const termSelect = document.getElementById('plan-term') as HTMLSelectElement | null; + const paymentSelect = document.getElementById('plan-payment') as HTMLSelectElement | null; + + if (!providerSelect || !termSelect || !paymentSelect) return; + + const provider = providerSelect.value; + const service = serviceSelect?.value; + const term = parseInt(termSelect.value, 10); + + populatePaymentSelect(paymentSelect, provider, service, term); +} + +/** + * Update term options based on selected payment + */ +function updateTermOptionsForPayment(): void { + const providerSelect = document.getElementById('plan-provider') as HTMLSelectElement | null; + const serviceSelect = document.getElementById('plan-service') as HTMLSelectElement | null; + const termSelect = document.getElementById('plan-term') as HTMLSelectElement | null; + const paymentSelect = document.getElementById('plan-payment') as HTMLSelectElement | null; + + if (!providerSelect || !termSelect || !paymentSelect) return; + + const provider = providerSelect.value; + const service = serviceSelect?.value; + const payment = paymentSelect.value; + + populateTermSelect(termSelect, provider, service, payment); +} + +/** + * Validate and fix invalid term/payment combinations + */ +function validateAndFixCombination(): void { + const providerSelect = document.getElementById('plan-provider') as HTMLSelectElement | null; + const serviceSelect = document.getElementById('plan-service') as HTMLSelectElement | null; + const termSelect = document.getElementById('plan-term') as HTMLSelectElement | null; + const paymentSelect = document.getElementById('plan-payment') as HTMLSelectElement | null; + + if (!providerSelect || !termSelect || !paymentSelect) return; + + const provider = providerSelect.value; + const service = serviceSelect?.value; + const term = parseInt(termSelect.value, 10); + const payment = paymentSelect.value; + + // Check if current combination is valid + if (!isValidCombination(provider, service, term, payment)) { + // Invalid combination - update payment to first valid option + updatePaymentOptionsForTerm(); + } +} + +/** + * Update custom config fields based on the selected ramp schedule preset + */ +function updateCustomConfigFromPreset(rampSchedule: string): void { + const stepPercentInput = document.getElementById('ramp-step-percent') as HTMLInputElement; + const intervalDaysInput = document.getElementById('ramp-interval-days') as HTMLInputElement; + + if (!stepPercentInput || !intervalDaysInput) return; + + switch (rampSchedule) { + case 'immediate': + stepPercentInput.value = '100'; + intervalDaysInput.value = '0'; + break; + case 'weekly-25pct': + stepPercentInput.value = '25'; + intervalDaysInput.value = '7'; + break; + case 'monthly-10pct': + stepPercentInput.value = '10'; + intervalDaysInput.value = '30'; + break; + case 'custom': + // Don't change values when switching to custom - let user modify + break; + } +} + +/** + * Close purchase modal + */ +export function closePurchaseModal(): void { + const purchaseModal = document.getElementById('purchase-modal'); + if (purchaseModal) closeModal(purchaseModal); +} + +/** + * Open modal to add planned purchases for a plan + */ +export async function openAddPurchasesModal(planId: string, planName: string): Promise { + // Remove existing modal if present + document.getElementById('add-purchases-modal')?.remove(); + + const modal = document.createElement('div'); + modal.id = 'add-purchases-modal'; + modal.innerHTML = ` + + `; + document.body.appendChild(modal); + + // Set default start date to tomorrow. + // + // Date inputs use the user's *local* calendar day, so we must build the + // ISO string from local components (toISOString returns UTC). For a user + // west of UTC on the evening of day D, `new Date().toISOString()` returns + // D+1's calendar date, which would make `min` reject "today" in the + // picker and contradict the QA 5.6 promise that same-day purchases stay + // selectable. Mirrors isPlanOverdue's local-midnight pattern above. + const startDateInput = document.getElementById('add-purchases-start-date') as HTMLInputElement; + const tomorrow = new Date(); + tomorrow.setDate(tomorrow.getDate() + 1); + startDateInput.value = toLocalDateInputValue(tomorrow); + startDateInput.min = toLocalDateInputValue(new Date()); + + // Add event listeners + document.getElementById('add-purchases-cancel')?.addEventListener('click', closeAddPurchasesModal); + document.getElementById('add-purchases-form')?.addEventListener('submit', (e) => void handleAddPurchases(e)); + + // Inline range validation on the count field: surfaces a + // "Must be a whole number between 1 and 52" message under the input + // as the user types, instead of waiting for the API call to reject + // the value (mirrors the wireRangeInput pattern from #702/#714, + // closes #771). The modal is built fresh on every open so no + // data-range-wired guard is needed here. + wireRangeInput('add-purchases-count', 1, 52); + + // Keep the submit button disabled while the count field is invalid. + // Checked after every input/blur so the button reflects the live + // validation state set by wireRangeInput above. + const countInput = document.getElementById('add-purchases-count') as HTMLInputElement | null; + const submitBtn = modal.querySelector('button[type="submit"]'); + if (countInput && submitBtn) { + const syncSubmitBtn = (): void => { + submitBtn.disabled = countInput.getAttribute('aria-invalid') === 'true'; + }; + countInput.addEventListener('input', syncSubmitBtn); + countInput.addEventListener('blur', syncSubmitBtn); + } + + // Engage focus trap + Escape handler. The modal element itself is + // removed from the DOM on close (see closeAddPurchasesModal) instead + // of just toggling .hidden, so the closeModal call there is what + // actually triggers focus restoration to the trigger. + openModal(modal); +} + +/** + * Close add purchases modal — restore focus first (closeModal), then + * remove the dynamically-injected element from the DOM. + */ +function closeAddPurchasesModal(): void { + const modal = document.getElementById('add-purchases-modal'); + if (modal) closeModal(modal); + modal?.remove(); +} + +/** + * Handle form submission for adding planned purchases + */ +async function handleAddPurchases(e: Event): Promise { + e.preventDefault(); + const errorDiv = document.getElementById('add-purchases-error'); + errorDiv?.classList.add('hidden'); + + try { + const planId = (document.getElementById('add-purchases-plan-id') as HTMLInputElement).value; + // Use Number() (not parseInt) so fractional input like "2.5" fails + // Number.isInteger() instead of silently truncating to 2. Mirrors + // the strict parse pattern from #471/#702 (feedback_strict_int_parse). + const rawCount = Number((document.getElementById('add-purchases-count') as HTMLInputElement).value); + if (!Number.isFinite(rawCount) || !Number.isInteger(rawCount) || rawCount < 1 || rawCount > 52) { + if (errorDiv) { + errorDiv.textContent = 'Number of Purchases must be a whole number between 1 and 52'; + errorDiv.classList.remove('hidden'); + } + return; + } + const count = rawCount; + const startDate = (document.getElementById('add-purchases-start-date') as HTMLInputElement).value; + + await api.createPlannedPurchases(planId, count, startDate); + + closeAddPurchasesModal(); + await loadPlannedPurchases(); + showToast({ message: `Successfully scheduled ${count} purchase${count > 1 ? 's' : ''}`, kind: 'success', timeout: 5_000 }); + } catch (error) { + const err = error as Error; + if (errorDiv) { + errorDiv.textContent = err.message; + errorDiv.classList.remove('hidden'); + } + } +} + +// Unsubscribe handles kept at module scope so setupPlanHandlers stays +// idempotent: calling it more than once (e.g. in tests or after HMR) +// replaces the old subscriptions rather than stacking duplicates. +let _unsubProvider: (() => void) | null = null; +let _unsubAccount: (() => void) | null = null; + +/** + * Setup plan form event handlers (provider-aware service dropdown). + * + * Provider/account filter source-of-truth is the global topbar (state.ts); + * subscribe so the plans list re-renders when the user changes filters at + * the page level. + */ +export function setupPlanHandlers(): void { + // Drop any previous subscriptions before re-registering so we don't + // accumulate duplicate loadPlans() calls on each invocation. + _unsubProvider?.(); + _unsubAccount?.(); + _unsubProvider = state.subscribeProvider(() => void loadPlans()); + _unsubAccount = state.subscribeAccount(() => void loadPlans()); + + const providerSelect = document.getElementById('plan-provider') as HTMLSelectElement | null; + const serviceSelect = document.getElementById('plan-service') as HTMLSelectElement | null; + + if (providerSelect && serviceSelect) { + // Update service dropdown visibility when provider changes + providerSelect.addEventListener('change', () => { + updateServiceDropdownForProvider(providerSelect.value); + }); + + // Initialize with current provider value + updateServiceDropdownForProvider(providerSelect.value); + } +} + +/** + * Update service dropdown to show only services for selected provider + */ +function updateServiceDropdownForProvider(provider: string): void { + const serviceSelect = document.getElementById('plan-service') as HTMLSelectElement | null; + if (!serviceSelect) return; + + // Show/hide optgroups based on selected provider. + // + // Toggle the `hidden` class (same class the HTML starts with) so the + // DOM state tracks a single source of truth — previously we flipped + // `style.display`, which doesn't clear the pre-existing `hidden` + // class, so Azure/GCP stayed hidden forever even when selected. + // + // Also flip `optgroup.disabled`: Chrome has a long-standing quirk + // where a `display: none` still renders its