diff --git a/.circleci/config.yml b/.circleci/config.yml index a1eda8638..38ccd0d0b 100644 --- a/.circleci/config.yml +++ b/.circleci/config.yml @@ -108,6 +108,9 @@ deploy_steps: &deploy_steps ./psvar-processor.sh -t appenv -p /config/${APPNAME}/deployvar source deployvar_env ./master_deploy.sh -d CFRONT -e $DEPLOY_ENV -c $ENABLE_CACHE + # Switch the uncached Gigs shell only after all referenced assets are deployed. + aws s3 cp "${AWS_S3_SOURCE_SYNC_PATH%/}/index.html" "s3://${AWS_S3_BUCKET%/}/gigs/index.html" \ + --content-type text/html --cache-control 'no-cache,no-store,max-age=0' jobs: lint-dev: @@ -229,8 +232,8 @@ workflows: only: - dev - copilot_reviewer - - support-app - PM-5460 + - aws_analytics tags: only: /^dev-.*/ diff --git a/.environments/.env.dev b/.environments/.env.dev index ab1b994f6..38b99ddbb 100644 --- a/.environments/.env.dev +++ b/.environments/.env.dev @@ -1,5 +1,10 @@ REACT_APP_HOST_ENV=dev +# First-party AWS Clickstream analytics +REACT_APP_ANALYTICS_API_URL=https://api.topcoder-dev.com/v1/analytics +REACT_APP_AWS_ANALYTICS_APP_ID=topcoder_web +REACT_APP_AWS_ANALYTICS_ENDPOINT=https://events.topcoder-dev.com/collect + REACT_APP_ENABLE_TCA_CERT_MONETIZATION=false # Stripe configs @@ -11,8 +16,6 @@ REACT_APP_DATADOG_PUBLIC_TOKEN=puba0825671e469d16f940c5a30dc738f11 REACT_APP_MEMBER_VERIFY_LOOKER=3322 -REACT_APP_SPRIG_ENV_ID=bUcousVQ0-yF - # Filestack configuration for uploading Submissions REACT_APP_FILESTACK_API_KEY='AzFINuQoqTmqw0QEoaw9az' REACT_APP_FILESTACK_REGION='us-east-1' diff --git a/.environments/.env.prod b/.environments/.env.prod index 1a460a189..0c3a4bab5 100644 --- a/.environments/.env.prod +++ b/.environments/.env.prod @@ -1,5 +1,10 @@ REACT_APP_HOST_ENV=prod +# First-party AWS Clickstream analytics +REACT_APP_ANALYTICS_API_URL= +REACT_APP_AWS_ANALYTICS_APP_ID= +REACT_APP_AWS_ANALYTICS_ENDPOINT= + REACT_APP_ENABLE_TCA_CERT_MONETIZATION=false # Stripe configs @@ -11,8 +16,6 @@ REACT_APP_DATADOG_PUBLIC_TOKEN=puba0825671e469d16f940c5a30dc738f11 REACT_APP_MEMBER_VERIFY_LOOKER=3322 -REACT_APP_SPRIG_ENV_ID=a-IZBZ6-r7bU - # Filestack configuration for uploading Submissions REACT_APP_FILESTACK_API_KEY='AzFINuQoqTmqw0QEoaw9az' REACT_APP_FILESTACK_REGION=us-east-1 diff --git a/.environments/.env.qa b/.environments/.env.qa index 96332984b..091977cd2 100644 --- a/.environments/.env.qa +++ b/.environments/.env.qa @@ -1,5 +1,10 @@ REACT_APP_HOST_ENV=qa +# First-party AWS Clickstream analytics +REACT_APP_ANALYTICS_API_URL= +REACT_APP_AWS_ANALYTICS_APP_ID= +REACT_APP_AWS_ANALYTICS_ENDPOINT= + REACT_APP_ENABLE_TCA_CERT_MONETIZATION=false # Stripe configs @@ -11,8 +16,6 @@ REACT_APP_DATADOG_PUBLIC_TOKEN=puba0825671e469d16f940c5a30dc738f11 REACT_APP_MEMBER_VERIFY_LOOKER=3322 -REACT_APP_SPRIG_ENV_ID=bUcousVQ0-yF - # Filestack configuration for uploading Submissions REACT_APP_FILESTACK_API_KEY= REACT_APP_FILESTACK_REGION= diff --git a/.gitignore b/.gitignore index ada8e1f55..cf9575e76 100644 --- a/.gitignore +++ b/.gitignore @@ -27,3 +27,7 @@ yarn-debug.log* yarn-error.log* storybook-static + +# Python build artifacts +__pycache__/ +*.py[cod] diff --git a/docs/PM-6100-footer-verification.md b/docs/PM-6100-footer-verification.md new file mode 100644 index 000000000..4f3e9fcad --- /dev/null +++ b/docs/PM-6100-footer-verification.md @@ -0,0 +1,49 @@ +# PM-6100: universal navigation footer colors + +Verified against `dev` at `9d879b43d` on 2026-09-13. + +The reported green footer links are already corrected in this revision. The +ticket screenshot shows `/challenges` in the Work app. Both Work and Opportunities +load the shared 2026 foundations, so the investigation covers both layouts. + +## Cause and existing fixes + +The 2026 foundations originally applied `body.work-app a { color: #00797a; }`. +Because the universal footer is also inside `body`, that rule overrode the white +color inherited from the footer navigation sections. + +- `cb108c147a` (2026-09-09, PM-6091) added the scoped + `#footer-nav-el a:not([class*='cta'])` override for normal, hover, and keyboard + focus states. It leaves the white CTA's dark text intact. +- `9321313974` (2026-08-15) placed Opportunities' theme on its content wrapper, + `.opportunities-app.tc-2026`, and changed its body marker to + `opportunities-page`. `AppFooter` is a sibling of the platform route container, + so Opportunities' teal anchor default cannot select footer links. +- The explicit theme allowlist remains scoped when its stylesheet persists + during client-side navigation. Removing only an import would not be sufficient. + +No additional CSS override is needed on current `dev`. + +## Verification + +An isolated Chromium fixture compiled the actual shared reset, 2026 foundation, +and installed universal-navigation footer Sass. It retained the content/footer +sibling layout and the footer component's selectors. Checks at 1440 px and +390 px widths covered normal, hover, and focus states while changing the body +class from Opportunities to Work and back. All 18 checks passed: + +| Element | Computed color | +| --- | --- | +| Opportunity content link | `rgb(0, 121, 122)` | +| Footer navigation link | `rgb(255, 255, 255)` | +| Footer CTA text | `rgb(11, 61, 86)` | + +`yarn lint` and `NODE_OPTIONS=--max-old-space-size=16384 yarn run build` +also passed under Node 22.13.0. The initial build exhausted Node's default heap; +the larger heap completed the unchanged production build. + +This verifies the CSS cascade and app scoping; it is not a live deployment test. +For release QA, open Work's challenge listing and Opportunities at desktop and +mobile widths, hover and keyboard-focus footer links, then navigate between the +apps without reloading. Footer links should remain white, content links teal, +and the "Talk to an expert" CTA should retain readable dark text. diff --git a/docs/PM-6285-waitlist-verification.md b/docs/PM-6285-waitlist-verification.md new file mode 100644 index 000000000..f438b6759 --- /dev/null +++ b/docs/PM-6285-waitlist-verification.md @@ -0,0 +1,42 @@ +# PM-6285: review waitlist and badge verification + +Verified platform-ui `origin/dev` at `9d879b43d` and review-api-v6 +`origin/develop` at `41d6853`. The ticket's changes are already present: platform +commit `5aed5245b` (merged PR #2255) added the waitlist clock and corrected track +colors. Later QA copy changes retain the capacity and status behavior. This new +PR records verification without duplicating the existing implementation. + +## Coverage + +| Requirement | Current implementation | +| --- | --- | +| Full opportunity can accept a waitlist application | Eligible reviewers receive `canApply: true`; the POST persists `PENDING` even when approved reviewers fill capacity. | +| Capacity excludes pending applications | Review API reports approved count and clamps remaining positions at zero on both list and detail responses. | +| Waitlisted status | Caller-owned `PENDING` plus zero remaining positions becomes `Waitlisted` in Browse, My Work, and the detail CTA. An explicit future `WAITLISTED` status is also understood. | +| Later application decisions | Approval and rejection keep their distinct statuses. If capacity reopens, a pending application's derived label returns to `Applied`. | +| Clear application flow | The full-capacity CTA reads “Apply to be a reviewer (waitlist)” and explains Support's replacement-reviewer process before and after submission. | +| Eligibility remains enforced | Closed/expired opportunities, inactive/inaccessible challenges, invalid roles, and duplicate applications remain rejected. The unique database key also prevents concurrent duplicates. | +| Waitlist administration | Existing approval/rejection endpoints operate on pending applications. Approval creates the reviewer resource, publishes its event, and sends the notification. Selection remains manual. | +| Status appearance | Open is neutral; Applied/Waitlisted use green outlines with check/clock icons; Approved uses the solid green accepted style; Rejected uses red and an X. | +| Track backgrounds | Design `#d0e2ff`, Development `#a7f0ba`, AI `#e8daff`, Data Science `#ffd9be`, consistently selected through the shared track class mapping. | + +The backend intentionally has no persisted `WAITLISTED` enum: pending +applications are the waitlist, and live approved capacity determines the label. +An old API response explicitly denying capacity remains disabled, because that +server cannot honor a waitlist POST. The deployed API and UI must both contain +the already-merged contract. + +## Validation + +- 57 frontend tests pass across review opportunity utilities, cards, My Work, + and details, including full-capacity submission, disabled legacy responses, + caller waitlist state, and track class mapping. +- 31 backend tests pass across opportunity service, application service, and + application controller, including aggregated capacity, lifecycle/privacy, + full-capacity pending creation, and concurrent duplicate rejection. +- 18 Chromium checks of the compiled repository Sass pass at 1440px and 390px, + covering the five status appearances and four requested track backgrounds. + +The browser checks use a reduced fixture with the actual stylesheet and waitlist +SVG; they are not a live authenticated application test. The verification found +no additional runtime change required by PM-6285. diff --git a/docs/adr/0001-topscout-rag-management.md b/docs/adr/0001-topscout-rag-management.md new file mode 100644 index 000000000..51627469d --- /dev/null +++ b/docs/adr/0001-topscout-rag-management.md @@ -0,0 +1,223 @@ +# ADR 0001 — TopScout RAG Management admin page + +- **Status:** Proposed — for review +- **Date:** 2026-08-27 +- **Repos affected:** `platform-ui` (this ADR, admin UI), `tc-ai-api` (two new HTTP endpoints — companion implementation work tracked here since the UI cannot ship without them) +- **Related:** + - `tc-ai-api` ADR 0004 — role-based access control for agents/workflows/tools (Proposed, same date). Restricts `challenge-ingestion`/`challenge-bulk-ingestion` to `roles: ['administrator']` / `scopes: ['challengesRAG:admin']` by default and introduces the reusable `checkAccess`/`toAuthenticatedCaller` policy core this ADR depends on. + - `tc-ai-api/src/mastra/agents/challenge/challenge-search-agent.ts` — the TopScout agent this page's index feeds. + - `tc-ai-api/src/mastra/workflows/challenge/challenge-ingestion-workflow.ts`, `challenge-bulk-ingestion-workflow.ts` — the two workflows this page drives. + - `platform-ui/src/libs/shared/lib/services/ai-workflows.tsx` — existing shared workflow-run/poll helper, already used for single-challenge ingestion. + - `platform-ui/src/apps/admin/src/lib/components/ChallengeList/ChallengeList.tsx` — existing single-challenge "Upsert for TopScout" action (toast-driven), the nearest prior art in this repo. + - `platform-ui/src/apps/admin/src/ai/review-workflows/AiReviewWorkflowsPage.tsx` — page-shell precedent this ADR follows (list + filter + confirm-modal architecture). + - `platform-ui/src/apps/customer-portal/src/pages/talent-search/TalentSearchPage/TalentSearchPage.tsx` — loading-state precedent for a blocking AI action (`isExtractingSkills` boolean + label swap), reused for the bulk-ingestion "run" state. + +## Context + +### What TopScout's RAG pipeline looks like today + +`challenge-search-agent` (id `challenge-search-agent`) answers questions by querying a PgVector index (`PgVector`, `@mastra/pg`) whose default name is **`challenge_embeddings`**, in schema `ai` (`MASTRA_DB_SCHEMA`, default `ai`), created and dimension-guarded lazily by `ensureChallengeIndex()` (`tc-ai-api/src/mastra/vector/challenge-vector-store.ts`). One challenge produces many rows — one per description chunk — each row's `metadata` JSONB carrying `challengeId`, `name`, `type`, `track`, `skills`, `groups`, `projectId`, `chunkIndex`/`totalChunks`, `text`, `ingestedAt`. `type`/`track` are free-form strings, never enum-validated (`rag.config.ts`'s D12), though `rag.config.ts` does export an *informational* `KNOWN_TYPES`/`KNOWN_TRACKS` pair for readability. `metadataIndexes: ['challengeId', 'projectId', 'track']` are the only three JSONB paths currently btree-indexed. + +Two Mastra workflows populate this index: +- `challenge-ingestion` (id `challenge-ingestion`) — one challenge, by id or inline record: resolve → chunk/embed the **public** description only → `store.upsert({ ..., deleteFilter: { challengeId } })` (atomic per-challenge replace). +- `challenge-bulk-ingestion` (id `challenge-bulk-ingestion`) — paginates `searchChallengesTool` over a filter (`status`, `projectId`, `types`, `tracks`, `tags`, `groups`, `updatedDateStart`), fans out `challenge-ingestion` per match (bounded concurrency), aggregates a report (`processed`/`succeeded`/`failed`/`skipped`/`totalChunks`/`forceSplits`/per-challenge `results[]`). + +Both are exposed over Mastra's native HTTP surface (`POST /v6/ai/workflows/:workflowId/create-run` → `.../start?runId=` → poll `GET /v6/ai/workflows/:workflowId/runs/:runId`), and `tc-ai-api` ADR 0004 (Proposed) is independently restricting both to `administrator`/`challengesRAG:admin` by default, precisely because a bulk run "re-embeds and rewrites the shared vector index for every challenge matching a filter." + +**What's missing, concretely:** +- No bulk-ingestion trigger anywhere in `platform-ui` — only the single-challenge path exists (`ingestChallengeInRag()` in the shared `ai-workflows.tsx`, wired into `ChallengeList.tsx`'s row-action dropdown as "Upsert for TopScout," toast-driven, one challenge at a time). +- No visibility into what's actually indexed. `PgVector` has no "list distinct challenges" API — it's a chunk-grained key-value/vector store, not a relational read model — so nothing today can answer "which challenges are in the RAG index, for which project, how many chunks." +- No delete path. `PgVector.deleteVectors({ indexName, filter })` (metadata-filtered delete) already exists in the SDK the code depends on, but nothing in `tc-ai-api` calls it, and there's no HTTP route to reach it. A challenge that's cancelled, deleted, or re-scoped in the source system stays searchable by TopScout indefinitely with no way to remove it short of a manual DB operation. + +### Where this page belongs + +`platform-ui`'s admin app already has a system-admin "AI" tab group (`aiRouteId = 'ai'` in `src/apps/admin/src/config/routes.config.ts`) with two children — Review Workflows (`aiReviewWorkflowsRouteId`) and Review Templates (`aiReviewTemplatesRouteId`) — registered as an ``-wrapped `children` block in `admin-app.routes.tsx`, gated by `rolesRequired: administratorOnlyRoles` on that parent node, and surfaced in `SystemAdminTabsConfig` (`src/apps/admin/src/lib/components/common/Tab/config/system-admin-tabs-config.ts`) gated a second time by `isAdministrator(roles)` in `getSystemAdminTabs()`. This is the exact shape the user's requested location (`/system-admin/ai/TopScout-RAG`) already matches — a third child under the same `aiRouteId` node, inheriting the same `administrator`-only gate with zero new frontend authorization code. (This ADR normalizes the path segment to `topscout-rag`, lower-kebab-case, matching the sibling `review-workflows`/`review-templates` convention rather than the mixed-case form in the request.) + +### Access control is a genuine cross-repo dependency, not a UI nicety + +The two capabilities this page adds — bulk-(re)ingest and delete-from-index — are at least as sensitive as `challenge-bulk-ingestion` itself (same shared-index blast radius; delete is one-way against the only copy of that data). `tc-ai-api` ADR 0004's own enforcement design only recognizes `/v6/ai/agents/:id/*`, `/v6/ai/workflows/:id/*`, and `/v6/ai-chat/:agentId` shapes in its URL-pattern matcher (`authorizeAccessPolicy`) — a plain custom route like `/v6/ai/challenge-embeddings/*` falls through that matcher's "no match → true" branch and would be **authenticated-but-unrestricted** by default if nothing else is done, exactly the kind of silent gap ADR 0004 itself flags as a standing footgun for future resources. This ADR does not relitigate ADR 0004's design; it reuses its policy primitives directly at the two new route handlers' call sites (see Decision 2), the same way ADR 0004 itself wraps tools at their own export site rather than relying purely on the HTTP-boundary matcher. + +## Scope + +**In scope:** +- A new admin page, `TopScout RAG` (`/system-admin/ai/topscout-rag`), as a third child of the existing `aiRouteId` system-admin tab. +- **Ingestion panel**: trigger a single-challenge or filtered bulk-ingestion run against `challenge-ingestion`/`challenge-bulk-ingestion`, monitor it to completion via the existing poll helper, and present a completion summary (counts, per-challenge failures) or a clear error. +- **Indexed-challenges panel**: a paginated, filterable (project, track, type), searchable list of what `challenge_embeddings` currently holds, with a per-row Delete action (confirm-modal gated) that removes that challenge's vectors from the index. +- Exactly two new `tc-ai-api` HTTP endpoints, both under the existing `/v6/ai` prefix: + - `DELETE /v6/ai/challenge-embeddings/:challengeId` + - `GET /v6/ai/challenge-embeddings` (list + count + filtering + search — the single "stats/count" endpoint requested; see Decision 2 for why this is one endpoint, not a separate stats/facets endpoint) +- Extending the shared `platform-ui` workflow-run service with bulk-ingestion support, and a small new admin-app service for the two new endpoints. +- Reusing (not redefining) `tc-ai-api` ADR 0004's access-control primitives to gate the two new endpoints with the same default policy as the two ingestion workflows. + +**Out of scope (deferred, not rejected):** +- Automatic re-ingestion (e.g., a challenge-update webhook calling `challenge-ingestion`) — this page is manual/on-demand only. +- Scheduled/cron-driven bulk ingestion. +- A resolved project-name picker for the project filter — v1 filters by bare project id (see Decision 4); TopScout's own agent treats `projectId` as an opaque, never-dereferenced reference (D10) for the same reason, and this page follows that precedent rather than adding a new projects-api dependency for a filter control. +- In-place editing of an indexed challenge's content — re-ingestion is the only way to change what's indexed; this page never writes challenge content itself. +- Any change to ADR 0004's own decisions — this ADR is a consumer of that policy core, not a revision of it (see Decision 2 and Open Questions for sequencing). + +## Decision + +### 1. Page placement and route wiring (`platform-ui`) + +- `src/apps/admin/src/config/routes.config.ts`: add `export const topScoutRagRouteId = 'topscout-rag'`. +- `src/apps/admin/src/admin-app.routes.tsx`: lazy-load `TopScoutRagPage` (`./ai/topscout-rag/TopScoutRagPage`) and add it as a third entry in the existing `aiRouteId` block's `children` array, alongside `AiReviewWorkflowsPage`/`AiReviewTemplatesPage` — no new `rolesRequired` needed; it inherits the parent `aiRouteId` node's `administratorOnlyRoles`. +- `src/apps/admin/src/lib/components/common/Tab/config/system-admin-tabs-config.ts`: add `{ id: `${aiRouteId}/${topScoutRagRouteId}`, title: 'TopScout RAG' }` to the existing `aiRouteId` tab's `children` array (next to "AI Review Workflows"/"AI Review Templates"). +- Page files: `src/apps/admin/src/ai/topscout-rag/TopScoutRagPage.tsx` (+ `.module.scss`, `index.ts`), following `AiReviewWorkflowsPage.tsx`'s shell exactly: `PageWrapper` wrapping (a) the ingestion panel and (b) `TableLoading` / `TableNoRecord` / `TableWrapper` / `Table` / `TableMobile` for the indexed-challenges list, all pulled from the admin app's existing `../../lib` barrel — no new shared UI primitives required. + +### 2. `tc-ai-api`: exactly two new endpoints, and why not three + +The request asks for delete-by-id plus "a stats/count endpoint which should provide the list of challenges, count and project, [with] filtering... and searching." That is one endpoint doing list+count+filter+search, not a separate facets endpoint for populating filter dropdowns. To keep it at exactly two endpoints: + +- **Project filter**: plain project-id text/number input, not a resolved picker (Scope, Decision 4) — no facet needed. +- **Type/Track filter options**: sourced from a small duplicated constant in `platform-ui` (`TOPSCOUT_RAG_KNOWN_TYPES`/`TOPSCOUT_RAG_KNOWN_TRACKS`, mirroring `tc-ai-api`'s `rag.config.ts` `KNOWN_TYPES`/`KNOWN_TRACKS` values verbatim, with a comment pointing at that file as source of truth) rather than a live `SELECT DISTINCT` facet query. This is safe specifically because `rag.config.ts` already documents these two lists as **informational only, not enforced** (D12) — a stale dropdown option is a minor UX gap (an admin can't filter by a value that predates this constant list), never a correctness bug, since the list/search query itself matches whatever free-form value a row actually has regardless of what the dropdown offers. (Note this dropdown is intentionally broader than the vector-*search* tool's own type enum — `challenge-search-agent`'s instructions restrict its search-time `type` filter to only `"Challenge"`/`"Marathon Match"` — because this page browses everything that was ever indexed, not just what the chat agent can query by type.) + +New module `tc-ai-api/src/mastra/vector/challenge-embeddings-admin.ts` (co-located with the existing `challenge-vector-store.ts`, reusing its lazy `getChallengeVectorStore()`/`ensureChallengeIndex()` singleton): + +```ts +export async function deleteChallengeEmbeddings(challengeId: string): Promise<{ deletedChunks: number }> { + const config = getRagConfig(); + const store = await ensureChallengeIndex(); + const { rows } = await store.pool.query( + `SELECT COUNT(*)::int AS n FROM "${schemaIdent(config)}"."${config.vectorIndexName}" WHERE metadata->>'challengeId' = $1`, + [challengeId], + ); + const deletedChunks = rows[0]?.n ?? 0; + if (deletedChunks > 0) { + await store.deleteVectors({ indexName: config.vectorIndexName, filter: { challengeId } }); + } + return { deletedChunks }; +} + +export async function listChallengeEmbeddings(params: { + projectId?: string; track?: string; type?: string; search?: string; + page: number; perPage: number; +}): Promise<{ items: ChallengeEmbeddingSummary[]; total: number }> { + // GROUP BY metadata->>'challengeId', with a parameterized WHERE built from + // the optional filters (never string-concatenated — only the validated + // schema/index identifiers are interpolated; every filter *value* is a + // bound $n param); a second COUNT(DISTINCT metadata->>'challengeId') + // query (same WHERE, no GROUP BY / LIMIT) for `total`. +} +``` + +`PgVector.deleteVectors({ indexName, filter, ids, namespace })` and the public `pool: pg.Pool` field are both already part of the installed `@mastra/pg@1.22.0` surface — confirmed by reading `dist/vector/index.d.ts` directly rather than assumed. `deleteVectors` gives the atomic metadata-filtered delete the DELETE endpoint needs for free; `pool` is the only way to do the chunk-grained-to-challenge-grained `GROUP BY` aggregation the list endpoint needs, since `PgVector` itself has no "distinct records" query shape. + +Two follow-on, additive, idempotent changes this decision requires: +- `ensureChallengeIndex()`'s existing `metadataIndexes: ['challengeId', 'projectId', 'track']` gains `'type'`, so the new list/filter query isn't an unindexed scan on the one JSONB path it's missing. `createIndex()` is already called unconditionally on every boot per that function's own doc comment ("safe to call every time"), so this is a one-line, zero-migration addition. +- The schema name (`MASTRA_DB_SCHEMA`, default `ai`) is validated with the same `validateSqlIdentifier()` regex `rag.config.ts` already applies to `VECTOR_INDEX_NAME`, at the point this new module builds its raw SQL identifiers — `rag.config.ts` itself does not currently validate `schemaName`, and this module is the first place in the codebase to interpolate it directly into a SQL string, so it must not inherit that gap silently. + +**Route registration** (`tc-ai-api/src/mastra/index.ts`, `server.apiRoutes` array, alongside the existing `chatRoute(...)` entry): both new routes are registered via Mastra's `registerApiRoute` (`@mastra/core/server`) under `${API_PREFIX}/challenge-embeddings` (GET, list) and `${API_PREFIX}/challenge-embeddings/:challengeId` (DELETE). Both are automatically **authenticated** by `apiAuthLayer`'s existing `protected: ['${API_PREFIX}/*']` blanket rule — no change needed there. Neither is automatically **authorized** by ADR 0004's `authorizeAccessPolicy`, since its path matcher only recognizes `/agents/`, `/workflows/`, and chatRoute shapes (Context, above). Each handler therefore calls the same policy check ADR 0004 exports, at the top of its own body — mirroring how ADR 0004 wraps tools at their *own* export site rather than relying solely on the HTTP boundary: + +```ts +// inside each of the two new route handlers +if (process.env.DISABLE_AUTH !== 'true') { + const user = /* from context, same as resourceIdMiddleware's resolution */; + const policy = resolveAccessPolicy('route', 'challenge-embeddings-admin'); // new AccessCategory + if (!user || !checkAccess(toAuthenticatedCaller(user), policy)) { + return c.json({ error: 'Forbidden' }, 403); + } +} +``` + +This adds one new entry to ADR 0004's own `DEFAULT_ACCESS_POLICIES` registry — a third `AccessCategory` (`'route'`) alongside `agent`/`workflow`/`tool` — rather than a silently-hardcoded, unregistered check, so the new capability stays visible in the same reviewable place ADR 0004 established: + +```ts +route: { + 'challenge-embeddings-admin': { mode: 'restricted', roles: ['administrator'], scopes: ['challengesRAG:admin'] }, +}, +``` + +Both the list (`GET`) and delete (`DELETE`) endpoints use the **same** restricted policy — the list endpoint is not left `public` even though it's read-only, because it lets a caller enumerate every currently-indexed challenge (name/project/track/type) in one shot, which is a wider disclosure surface than the per-challenge, already-public data `challenge-vector-query` exposes one hit at a time. This mirrors ADR 0004's own choice to restrict both ingestion workflows uniformly rather than splitting by read/write. + +**Response shape** (`GET`): body is a **plain array** of `{ challengeId, name, type, track, projectId, groups, skills, chunkCount, ingestedAt }`; pagination rides on `X-Page`/`X-Per-Page`/`X-Total`/`X-Total-Pages` **response headers** — already present in `tc-ai-api`'s existing CORS `exposeHeaders` list (`src/mastra/index.ts`) and exactly what `platform-ui`'s existing `xhrGetPaginatedAsync` (`src/libs/core/lib/xhr/xhr-functions/xhr.functions.ts`, reading `x-page`/`x-per-page`/`x-total`) already parses. The frontend needs zero new pagination-parsing code for this endpoint — only a new URL and query params. + +### 3. `platform-ui`: service layer + +- `src/libs/shared/lib/services/ai-workflows.tsx`: add `ingestChallengesBulkInRag(filters, workflowId?)`, mirroring `ingestChallengeInRag`'s existing shape (`startWorkflowRun` → `pollWorkflowRunStatus`, both already generic over any workflow id) and a new `normalizeChallengeBulkIngestionResult()`/`ChallengeBulkIngestionResult` type mirroring the workflow's own `bulkReportSchema` (`processed`/`succeeded`/`failed`/`skipped`/`totalChunks`/`results[]`). No changes to `startWorkflowRun`/`pollWorkflowRunStatus` themselves — both are already workflow-agnostic. +- New env var `RAG_CHALLENGE_BULK_INGESTION_WORKFLOW_ID` (`default.env.ts` + `global-config.model.ts`), defaulting to **`'challenge-bulk-ingestion'`** — the workflow's own `.id`. This is a deliberate departure from the sibling `RAG_CHALLENGE_INGESTION_WORKFLOW_ID`'s existing default (`'challengeIngestionWorkflow'`, the Mastra *registry key*, not the `.id`) already shipped in this codebase: per ADR 0004's own documented footgun, keying on the registry key instead of `.id` only happens to work today because `Mastra.getAgentById()`/workflow routing falls back from `.id` to registry key on a miss — it is not something a new addition should copy forward as if it were the correct convention. +- New `src/apps/admin/src/lib/services/challenge-embeddings.service.ts`, matching the existing `ai-workflows.service.ts` pattern in the same directory: `getChallengeEmbeddings(params)` via `xhrGetPaginatedAsync(...)`, `deleteChallengeEmbeddings(challengeId)` via `xhrDeleteAsync`. + +### 4. Page behavior + +**Ingestion panel** — a small form: either a single free-text challenge-id field, or the bulk filter set (`projectId`, `track`, `type`, `status`, `updatedDateStart` — mirroring `challenge-bulk-ingestion`'s own `bulkInputSchema`), mutually exclusive the same way `challenge-ingestion`'s own resolve step requires exactly one source (challengeId XOR inline record) — the page enforces the analogous rule client-side (single-id field XOR any filter field) before allowing Run. A "Dry run" toggle passes straight through to the workflow's existing `dryRun` input (chunk+embed, skip the upsert) — useful for previewing a bulk run's scope before committing it. Flow on Run: +1. `ConfirmModal` (bulk ingestion mutates the shared index for every match — the same reason ADR 0004 restricts it server-side; the UI asks before firing, not as a substitute for that server-side enforcement, but because it's a slow, hard-to-abort action worth a deliberate click). +2. `isRunning` boolean disables the form and swaps the button label (`"Ingesting…"` / `"Running bulk ingestion…"`), matching `TalentSearchPage.tsx`'s `isExtractingSkills` idiom — no separate blocking modal is needed, since `pollWorkflowRunStatus` already blocks the returned promise until the run resolves. +3. On resolve: a per-run **completion summary** stays visible in the panel (processed / succeeded / failed / skipped / total chunks, plus an expandable list of `results.filter(r => r.status === 'failed')` with each failure's message) — richer than a toast alone, since a bulk run has much more to report than the single-challenge case. `toast.success`/`toast.error` still fires for the at-a-glance case, matching `ChallengeList.tsx`'s existing toast idiom for the single-ingest action. +4. A successful run (bulk or single) triggers a refetch of the indexed-challenges list below, so newly (re-)ingested challenges show up without a manual refresh. + +**Indexed-challenges panel** — `getChallengeEmbeddings({ page, perPage, projectId, track, type, search })` on mount and on any filter change; columns: Challenge (name, linking to `${EnvironmentConfig.ADMIN.CHALLENGE_URL}/${challengeId}`, matching `ChallengeList.tsx`'s existing external-link convention), Type, Track, Project (bare id, per Decision 4/Scope), Chunks, Ingested At, and a Delete action. Delete opens the same `ConfirmModal` pattern `AiReviewWorkflowsPage.tsx` already uses for its activate/deactivate confirmation, calls `deleteChallengeEmbeddings(challengeId)`, toasts on result, and removes the row (or refetches the current page) on success. + +## Implementation plan + +### Phase 0 — `tc-ai-api`: data-layer additions +- `challenge-vector-store.ts`: add `'type'` to `metadataIndexes`. +- New `challenge-embeddings-admin.ts`: `deleteChallengeEmbeddings()`, `listChallengeEmbeddings()`, schema-identifier validation at the point of use. +- Unit tests mirroring `challenge-vector-store.test.ts`'s existing conventions: delete removes exactly the matched challenge's rows and reports an accurate `deletedChunks` (including the "nothing matched" / `0` case); list respects each filter independently and in combination; search matches by name substring and by exact challengeId; pagination math (`total`, `page`/`perPage` boundaries) is correct against a seeded multi-challenge, multi-chunk fixture. + +### Phase 1 — `tc-ai-api`: routes and access control +- Extend ADR 0004's `AccessCategory`/`DEFAULT_ACCESS_POLICIES` with the `route` category and the `challenge-embeddings-admin` entry (Decision 2). If ADR 0004 has not yet merged when this phase starts, land the equivalent inline `checkAccess`/`toAuthenticatedCaller` call directly in the two new handlers with the hardcoded restricted policy, and switch to the shared registry in a follow-up once ADR 0004 ships — sequencing note, not a blocker (see Open Questions). +- Register the two routes in `mastra/index.ts`'s `apiRoutes`. +- Tests mirroring ADR 0004's own planned `resourceIdMiddleware.test.ts` cases: member without `administrator` → 403; member with it → 200; M2M without `challengesRAG:admin` → 403; M2M with it → 200; `DISABLE_AUTH=true` bypasses both. + +### Phase 2 — `platform-ui`: service layer +- `ai-workflows.tsx`: `ingestChallengesBulkInRag()`, `ChallengeBulkIngestionResult`. +- `default.env.ts` / `global-config.model.ts`: `RAG_CHALLENGE_BULK_INGESTION_WORKFLOW_ID`. +- New `challenge-embeddings.service.ts` in the admin app. + +### Phase 3 — `platform-ui`: page +- Route/tab wiring (Decision 1). +- `TopScoutRagPage.tsx` — ingestion panel + indexed-challenges panel, per Decision 4. +- Manual smoke test in a real environment (per this repo's own "verify against the running system, don't assume" precedent): run a small bulk ingestion with a narrow filter, confirm the summary matches what actually landed in the index, confirm a subsequent list-page fetch shows it, delete it, confirm it disappears from the list and TopScout itself can no longer surface it in a follow-up chat query. + +### Phase 4 — Documentation +- `tc-ai-api README.md`: document the two new routes next to the existing workflow/agent route documentation. +- `platform-ui`: no README currently documents the admin AI tab group's routes; none added here either, consistent with the existing sibling pages. + +## File-level mapping + +| Repo | File | Change | +| --- | --- | --- | +| tc-ai-api | `src/mastra/vector/challenge-vector-store.ts` | Modified — add `'type'` to `metadataIndexes` | +| tc-ai-api | `src/mastra/vector/challenge-embeddings-admin.ts` | New — `deleteChallengeEmbeddings`, `listChallengeEmbeddings` | +| tc-ai-api | `src/mastra/vector/challenge-embeddings-admin.test.ts` | New | +| tc-ai-api | `src/config/access-control.config.ts` | Modified (ADR 0004 dependency) — `route` category, `challenge-embeddings-admin` policy entry | +| tc-ai-api | `src/mastra/index.ts` | Modified — register the two new `apiRoutes` | +| tc-ai-api | `README.md` | Modified — document the two new routes | +| platform-ui | `src/libs/shared/lib/services/ai-workflows.tsx` | Modified — `ingestChallengesBulkInRag`, `ChallengeBulkIngestionResult` | +| platform-ui | `src/config/environments/default.env.ts` | Modified — `RAG_CHALLENGE_BULK_INGESTION_WORKFLOW_ID` | +| platform-ui | `src/config/environments/global-config.model.ts` | Modified — same | +| platform-ui | `src/apps/admin/src/lib/services/challenge-embeddings.service.ts` | New | +| platform-ui | `src/apps/admin/src/config/routes.config.ts` | Modified — `topScoutRagRouteId` | +| platform-ui | `src/apps/admin/src/admin-app.routes.tsx` | Modified — new lazy route under `aiRouteId` | +| platform-ui | `src/apps/admin/src/lib/components/common/Tab/config/system-admin-tabs-config.ts` | Modified — new child tab | +| platform-ui | `src/apps/admin/src/ai/topscout-rag/TopScoutRagPage.tsx` (+ `.module.scss`, `index.ts`) | New | +| platform-ui | `src/apps/admin/src/lib/components/ChallengeList/ChallengeList.tsx` | **Unchanged** — its existing single-ingest action is independent of this page and stays as-is | + +## Consequences + +**Positive** +- Admins get a real operational surface for the RAG pipeline — bulk backfill/re-ingest, visibility into what's actually indexed, and a way to remove stale entries — none of which exists today short of a manual DB operation. +- Reuses, rather than duplicates, three separate pieces of prior art in this codebase: the shared workflow-run/poll service, the admin app's existing list-page shell (`AiReviewWorkflowsPage.tsx`), and the existing `xhrGetPaginatedAsync` header-based pagination contract — net-new frontend code is small. +- The two new endpoints stay inside `tc-ai-api`'s existing route/auth/CORS conventions (same `apiPrefix`, same `exposeHeaders`, same `apiRoutes` registration point as `chatRoute`) rather than introducing a parallel pattern. +- Explicitly re-uses ADR 0004's access-control model instead of inventing a second one, keeping "who can touch the RAG index" answerable from one registry. + +**Negative / risk** +- **Two-repo dependency for one feature.** This page cannot function until both new `tc-ai-api` endpoints ship; sequencing must be tracked explicitly (see Implementation plan phasing) rather than assumed. +- **Depends on an ADR that is itself still Proposed.** If ADR 0004's shape changes before it merges, Decision 2's `route` category / registry entry needs to move with it. Mitigated by keeping the inline-fallback option explicit in Phase 1 rather than hard-blocking on ADR 0004 merging first. +- **`KNOWN_TYPES`/`KNOWN_TRACKS` duplication is a manually-maintained mirror**, not a shared import (the two repos don't share a package for this). A future addition to `rag.config.ts`'s lists won't automatically appear in this page's filter dropdown. Accepted per Decision 2's reasoning (informational-only, non-enforced, UX-only staleness) rather than adding a third endpoint to keep them in sync live. +- **Raw SQL against `store.pool`** is new surface in a codebase that otherwise only talks to Postgres through `PgVector`'s own methods. Every filter/search *value* is a bound parameter; only the already-validated `vectorIndexName`/`schemaName` identifiers are ever interpolated — but this is still the first hand-written SQL in the vector layer, worth an explicit review pass, not just a "matches existing patterns" assumption. +- **Delete is irreversible.** There is no soft-delete or undo — removing a challenge's vectors means the next chat query about it returns nothing until it's re-ingested. The confirm-modal step is the only guard; this ADR does not add an audit trail beyond `tc-ai-api`'s existing `tcAILogger` usage (matching ADR 0004's own "single structured log line" precedent for denials — an actual delete, not just a denial, should log at `info` with the challengeId and caller for the same reason). + +## Open questions + +- **Sequencing with ADR 0004**: does this ADR's `tc-ai-api` work land before, after, or alongside ADR 0004's own implementation? Phase 1 above describes both orders; whichever happens first should not block the other, but the final registry shape (Decision 2) is only settled once ADR 0004 itself is no longer Proposed. +- Should the project filter graduate from a bare-id input to a resolved project-name picker once real admin usage shows bare ids are hard to work with? Deferred per Scope; revisit with actual usage feedback rather than speculatively building it now. +- Should `type`/`track` filter options move from the duplicated constant to a live facet query if the two lists drift in practice? Deferred per Decision 2/Consequences; only worth the extra endpoint if the duplication actually causes friction. +- Does a future webhook-driven auto-(re)ingestion pipeline reduce this page's bulk-ingestion panel to a "manual override / backfill only" tool? Out of scope here either way, but worth naming so a future ADR doesn't have to rediscover the relationship. + +## Prerequisites to confirm before implementation starts + +- The `administrator` role / `challengesRAG:admin` M2M scope convention from `tc-ai-api` ADR 0004 must actually exist in the relevant Auth0 tenant before either new endpoint's restriction is meaningful — this ADR reuses that prerequisite rather than re-verifying it independently. +- Confirm `challenge-bulk-ingestion` is reachable by its own `.id` (`RAG_CHALLENGE_BULK_INGESTION_WORKFLOW_ID`'s default) against a real deployed `tc-ai-api` — expected to work per Mastra's documented `.id`-first routing, but this ADR's own Decision 3 calls out that a sibling env var's default currently relies on registry-key fallback instead, so a live smoke test is cheap insurance rather than a formality. +- Reviewer sign-off on: adding a third `route` category to ADR 0004's access-control registry from a second ADR (Decision 2) rather than that decision living entirely inside ADR 0004 itself; the "duplicate the two known-value constants instead of a third facets endpoint" trade-off (Decision 2/Consequences); and the "list endpoint is restricted, not public" choice given it's read-only (Decision 2). diff --git a/docs/adr/0002-aws-product-analytics.md b/docs/adr/0002-aws-product-analytics.md new file mode 100644 index 000000000..eac05d52a --- /dev/null +++ b/docs/adr/0002-aws-product-analytics.md @@ -0,0 +1,285 @@ +# ADR 0002: AWS-native product analytics + +- Status: Accepted +- Date: 2026-08-30 +- Owners: Platform and website engineering + +## Context + +Topcoder needs one first-party analytics path across topcoder-website and +platform-ui. It must attribute page views and clicks to standard UTM values and +measure the same person's progression from a marketing landing page through +challenge registration and submission. The development deployment should +normally remain below USD 200 per month and the production design below USD 400 +per month. + +All prior product-analytics and survey integrations are removed. Existing +operational error logging remains out of scope and is not an event source for +product analytics. + +## Decision + +Use AWS Guidance for Clickstream Analytics on AWS, version 1.2.1, in us-east-1. +The control plane provisions a regional ingestion service on ECS with EC2 +capacity, stores raw events in S3, runs the built-in transformer on a +daily EMR Serverless schedule, loads modeled data into Redshift Serverless, and +exposes direct-query datasets to Amazon QuickSight. + +One Clickstream project and one web app ID are shared by the two UI surfaces in +each environment. The global surface attribute distinguishes platform_ui from +topcoder_website. Production must use a separate project, application, +ingestion endpoint, S3 prefixes, and Redshift namespace; production clients must +never send data to the development project. + +The browser integration uses the AWS Clickstream Web SDK directly in each host +application. Universal Navigation owns only first-touch UTM persistence and +signup-link propagation. It must not initialize another SDK instance because +that would double-count every host page. + +## Development topology and cost controls + +| Layer | Development setting | Production starting point | +| --- | --- | --- | +| Ingestion | One active ECS/EC2 instance, scale to two | Two instances, scale to four | +| Delivery | Direct S3 sink, 10 MB or 300 second buffering | Same | +| Network | Existing VPC and NAT gateway; no Global Accelerator | Reuse an existing production VPC/NAT | +| Logs | Default service logs; no ALB access-log bucket | Enable only when an operational need justifies it | +| Processing | Built-in transform plus user-agent enrichment once per day | Daily; increase only after a freshness review | +| Location enrichment | Disabled | Disabled unless approved for a documented use case | +| Warehouse | Redshift Serverless at the 8 RPU minimum | Start at 8 RPU and observe workload | +| Reporting | One QuickSight Enterprise author, direct query, no SPICE | Add readers/authors only as needed | +| Storage | Expire temporary artifacts; retain raw and modeled data to approved policy | Same with production retention policy | + +The deployed development environment is in AWS account 811668436784 in +us-east-1. Project topcoder_web_dev and app topcoder_web are active at +https://events.topcoder-dev.com/collect. The pipeline ID is +ead9a39a35334fafb77428073704ad7b. It uses the existing development VPC, a +dedicated second private ingestion subnet in a supported availability zone, +the existing Redshift/QuickSight subnets, daily processing, and the S3 bucket +topcoder-clickstream-data-dev-811668436784. A synthetic SDK-format event was +accepted by the collector before the client configuration was enabled. A +second event with source codex, medium integration, and campaign aws_analytics +was then processed through S3 and EMR and verified in both Athena and Redshift. + +The collector's durable client hostname is +https://events.topcoder-dev.com/collect. The original +analytics.topcoder-dev.com ingestion alias remains available only during the +cutover observation window; that hostname now belongs to the reporting UI. + +The development control-plane API has a narrow patch over upstream v1.2.1: +endpoint security-group discovery treats a gateway VPC endpoint without a +Groups array as an empty list. This fixes project provisioning in a VPC that +contains S3 gateway endpoints. CloudFormation does not own this code change, so +an upgrade or stack repair can overwrite it; either carry the patch forward or +upgrade only after the upstream implementation handles gateway endpoints. + +The upstream v1.2.1 Redshift schema also creates six PL/Python UDFs. Redshift +[stopped allowing new PL/Python UDFs on 2025-10-30](https://docs.aws.amazon.com/redshift/latest/mgmt/behavior-changes.html), +so the original schema run failed before creating the merge procedures. The +development database uses the compatibility Lambda +topcoder-clickstream-redshift-udf-dev for only those six preserved functions. +The Redshift-associated role has permission to invoke only that Lambda, and the +Lambda execution role can write only its logs. Redshift administrator access +uses the namespace's managed Secrets Manager secret rather than a copied +password. The secret adds a small fixed monthly charge; Lambda invocation cost +at the daily processing frequency is negligible relative to ingestion and +Redshift. + +The patched 114-statement installer is stored at the version-labeled prefix +s3://topcoder-clickstream-data-dev-811668436784/clickstream/topcoder_web_dev/data/load-workflow/tmp/topcoder_web_dev/sqls/topcoder_web-20260830T030115510Z-lambda-udf-v2/. +The rebuild scripts, deployed package, external-function definitions, +checksums, patched SQL, and runbook are archived at +s3://topcoder-clickstream-templates-dev-811668436784/custom-fixes/redshift-lambda-udf/v1/. +These resources are tagged and encrypted. CloudFormation does not own this +repair; reapply it after a Clickstream schema or application upgrade unless the +new release has removed the PL/Python dependency. Start a new load execution +with a unique name after repair because the v1.2.1 parent redrive reuses child +names and collides with their prior executions. + +At low event volume, the design target is approximately USD 130–190 per month +for development and USD 240–360 per month for production. These are operating +guardrails, not an invoice guarantee: ingestion traffic, EMR runtime, Redshift +query duration, log volume, cross-AZ transfer, and additional QuickSight users +are variable. Resources use Application, Environment, and CostCenter tags. +After the payer account activates those cost-allocation tags, review Cost +Explorer by tag weekly during the first month and create tag-scoped budgets. +The linked development account cannot activate those tags or create the +tag-scoped payer budget itself. Investigate development at USD 160 forecast and +stop nonessential processing before USD 200; use USD 320 and USD 400 as the +equivalent production thresholds. + +Do not increase the ingestion minimum, processing frequency, Redshift base RPU, +or QuickSight user count without recording the expected monthly delta. + +## Identity and attribution + +The clients create a random tc_analytics_id first-party cookie with a one-year +lifetime. On topcoder.com, topcoder-dev.com, and topcoder-qa.com it is scoped to +the registrable domain, allowing a landing-page visit and a later platform-ui +conversion to use the same Clickstream user ID. It is pseudonymous and contains +no member data. + +After authentication, platform-ui adds member_id as a global event attribute. +It does not send handle, name, email, form values, or rendered click text. + +Universal Navigation stores the first valid visit containing any of these +standard parameters in the tc_utm cookie for 30 days: + +- utm_source +- utm_medium +- utm_campaign +- utm_id +- utm_term +- utm_content + +Values are restricted to 100 characters and the characters A-Z, a-z, 0-9, +period, underscore, tilde, and hyphen. The two clients map them to Clickstream's +traffic_source columns, preferring values on the current URL and otherwise using +the first-touch cookie. + +## Event contract + +| Event | Producer | Required dimensions | Meaning | +| --- | --- | --- | --- | +| _page_view | AWS SDK in both apps | surface, environment, traffic source | A browser route became active | +| _user_engagement | AWS SDK in both apps | surface, environment, traffic source | Foreground engagement of at least one second | +| ui_click | Both apps | page_path, element_type, click_x_percent, click_y_percent | A click on an interactive element | +| ui_click | Both apps, when available | element_id, placement, destination_host, destination_path | Semantic click location and query-free destination | +| challenge_registered | platform-ui after API success | challenge_id, challenge_track, member_id | The Submitter resource was successfully created | +| challenge_submitted | platform-ui after API success | challenge_id, challenge_track, member_id, submission_type | The Review API successfully created a submission | + +Clickable conversion controls should have stable data-analytics-id and +data-analytics-placement attributes. Do not derive element_id from rendered +copy, because copy changes and may contain user-provided text. + +The AWS SDK adds its reserved current-page URL to events. UTM query values are +therefore present in raw/modelled events as well as normalized traffic-source +columns. Application routes must not place credentials, email addresses, or +other secrets in query parameters. The custom ui_click destination fields +deliberately omit destination queries. + +Global Privacy Control and Do Not Track prevent SDK initialization. Analytics +errors are swallowed so collection can never block navigation, registration, or +submission. + +## Funnel definition + +The product_analytics_events_v1 Redshift view normalizes the event contract. +The challenge_funnel_daily_v1 view counts distinct coalesced +user_id/user_pseudo_id values, grouped by traffic source and first landing page, +with these ordered stages: + +1. A _page_view event with surface topcoder_website. +2. A ui_click event on that surface. +3. A challenge_registered event after the click. +4. A challenge_submitted event after registration. + +The challenge_conversion_daily_v1 view groups a person's first registration +and later submission by challenge_id. The click_location_daily_v1 view groups +clicks by stable placement and element dimensions, query-free destination, and +10-percentage-point viewport buckets. All three reporting views enforce the +event contract centrally; simple independent event counts would overstate +conversion. + +QuickSight dashboard topcoder_product_analytics_dev_v1 provides landing users, +landing clickers, registrations after a click, submissions after registration, +the ordered UTM funnel, challenge conversion, and click-location tables. It +uses the direct-query datasets topcoder_challenge_funnel_v1, +topcoder_challenge_conversion_v1, and topcoder_click_location_v1, so it does not +create a SPICE copy. The reporting SQL, dataset requests, dashboard request, and +runbook are archived at +s3://topcoder-clickstream-templates-dev-811668436784/custom-assets/product-analytics/v1/. + +Platform UI also provides an operator-facing application at +https://analytics.topcoder-dev.com. Every app route requires an authenticated +profile with the exact analytics role. Its read-only HTTP API is exposed at +https://api.topcoder-dev.com/v1/analytics. API Gateway validates the +development Auth0 issuer and human-client audience; Lambda independently checks +a verified Topcoder roles claim before issuing fixed, parameterized, bounded +Redshift Data API queries through the analytics_api_reader database role. The +API returns aggregate data only and marks every response private and +non-cacheable. + +The Campaigns tab exposes the ordered landing, click, registration, and +submission funnel with UTM, landing-page, and privacy-safe click-location +breakdowns. Click locations are ranked and paginated twenty rows at a time; +the API combines viewport-position buckets for the same semantic item, and the +table does not display position. +The General tab exposes page views, visitors, and clicks in +separate daily charts plus paginated page and traffic-source tables. Both +report warehouse freshness and limit callers to 366 inclusive days. QuickSight +remains the AWS native exploratory dashboard; the Platform UI app is the +narrowly scoped daily operational interface. + +To avoid making a user wait for Redshift Serverless to resume after idle, an +EventBridge rule starts the default Campaigns and filter-option statements at +the beginning of each four-hour Data API idempotency window. Interactive +requests reuse the completed statements by their query fingerprint. This adds +only bounded scheduled queries and does not keep development RPUs continuously +active. + +## Configuration + +Platform UI reads: + +- REACT_APP_AWS_ANALYTICS_APP_ID +- REACT_APP_AWS_ANALYTICS_ENDPOINT + +topcoder-website reads: + +- NEXT_PUBLIC_AWS_ANALYTICS_APP_ID +- NEXT_PUBLIC_AWS_ANALYTICS_ENDPOINT + +The values are public ingestion configuration, not AWS credentials. Store them +in each deployment environment and leave them empty to disable analytics in +local or unprovisioned environments. Never expose the control-plane login, +Redshift credentials, or AWS credentials to either client. + +Development uses app ID topcoder_web and endpoint +https://events.topcoder-dev.com/collect. QA and production remain empty and +disabled until their separate projects and endpoints are provisioned. + +## Validation and operations + +For each environment: + +1. Open a landing URL with a unique test UTM campaign. +2. Confirm tc_utm and tc_analytics_id are first-party cookies. +3. Click a challenge CTA, register, and submit with a test account. +4. After the daily processing run, verify all four ordered stages share the same + user identity and traffic-source values in Redshift. +5. Verify the QuickSight funnel and UTM breakdown against the Redshift counts. +6. Repeat with Global Privacy Control or Do Not Track enabled and confirm no + browser requests reach the ingestion endpoint. +7. Review failed ingestion responses, EMR jobs, Redshift load state, S3 growth, + and the monthly cost forecast. +8. Verify the reporting API returns `401` without a JWT, `403` for a verified + account without the analytics role, and aggregate JSON for an authorized + account; confirm that its logs contain no tokens, filters, SQL, or records. + +The deployment smoke test completed this path on 2026-08-30. Both synthetic +events are present in event_v2; the tagged event retained codex, integration, +and aws_analytics while the control event remained Direct. The default AWS +Clickstream dashboard, Topcoder reporting dashboard, Redshift data source, and +all three custom datasets reported successful creation status. + +On 2026-09-02, the development collector received a larger pseudonymous fixture +for campaign `aws_analytics` and UTM ID `dev_fixture_20260902`. The standard +transform and Redshift load completed successfully, and both the reporting view +and role-gated API returned an ordered 30 landing visitors, 22 clickers, 14 +registrations, and 8 submissions. Three landing paths and three semantic click +placements provide non-empty table data. These aggregates are synthetic UI +test data and must not be interpreted as member traffic. + +On 2026-09-04, the same standard pipeline loaded a second pseudonymous fixture, +UTM ID `dev_fixture_multiday_20260904`, into the September 2 and 3 cohorts. The +reporting views verified 38 landing visitors, 29 clickers, 19 registrations, and +13 submissions across those dates, plus 29 distinct semantic clicked items. +Combined with the original fixture, `aws_analytics` now spans three reporting +dates and contains 68 landing visitors, 51 clickers, 33 registrations, and 21 +submissions. The expanded click set intentionally exercises the Campaigns +table's twenty-row pagination. These aggregates are also synthetic UI test data. + +If the daily pipeline misses its freshness objective, first inspect failures and +job duration. Increasing processing frequency is a cost-bearing design change, +not the default incident response. diff --git a/infrastructure/analytics-api/README.md b/infrastructure/analytics-api/README.md new file mode 100644 index 000000000..84a246a55 --- /dev/null +++ b/infrastructure/analytics-api/README.md @@ -0,0 +1,157 @@ +# Topcoder analytics API + +This directory contains the development infrastructure and Lambda code for the +role-gated Analytics UI. The API is a read-only adapter over the existing AWS +Clickstream Redshift reporting relations; it is not an ingestion endpoint. + +## Architecture and security boundary + +```text +Platform UI + -> api. shared CloudFront/API Gateway HTTP API + -> analytics-route JWT authorizer + -> Lambda exact analytics-role check and fixed queries + -> Redshift Data API + -> analytics_api_reader database role + -> approved reporting views and materialized projection +``` + +API Gateway validates the configured Auth0 issuer, audience, signature, and +standard JWT time claims. Lambda then requires `analytics` in a verified +Topcoder roles claim. The handler accepts only four fixed `GET` routes, +strict dates, bounded UTM/surface tokens, and an exact query-free path. SQL is server-owned and uses Data +API named parameters; callers cannot provide SQL, object names, sort clauses, +or result limits. + +Reports are limited to 366 inclusive days and 2,000 decoded rows. Query waits +leave time for a sanitized response. A failed or aborted statement is retried +once within the same deadline. Clients can opt into resumable requests with +`async=true`. If Redshift Serverless needs longer than one HTTP request, the API +returns `202`, `Retry-After`, and a server-generated query token. The browser +polls with that token until the original statement completes instead of +starting another warehouse query. Tokens are accepted only for the exact SQL, +validated parameters, retry attempt, and current or previous four-hour window. +The four-hour window remains below the Data API's eight-hour idempotency +retention and permits one complete boundary-crossing window without making +tokens reusable for other reports. An EventBridge schedule invokes the default +Campaigns report, filter-option query, and `/opportunities` route report at each +new window, allowing their statements to finish before an interactive request +while preserving Redshift Serverless idle cost controls. Browser polling is +bounded. Concurrency and API throttles cap warehouse pressure, and successful +responses use `Cache-Control: private, no-store`. Logs contain request IDs and +service-owned error categories only. + +The General report applies an inclusive timestamp range and uses Redshift +`GROUPING SETS` to calculate its summary, daily, page, traffic-source, and +surface sections in one scan of the reporting view. This avoids repeating the +view's JSON-derived field work for each section while preserving exact visitor +counts and the existing response contract. + +The Route report reads an auto-refreshed, timestamp-sorted materialized +projection containing only its approved event types and fields. Replicated +distribution keeps the report's repeated session and funnel joins local and +avoids re-extracting JSON fields from the wide Clickstream table. The response +continues to expose `dataThrough` so consumers can see the latest included route +page-view date while Redshift schedules incremental refreshes. + +## Files + +- `template.yaml` registers protected analytics routes on the shared API and + provisions the JWT authorizer, Lambda, least-privilege query role, scheduled + default-report prewarm, and logs. + The former dedicated API remains during the cutover observation window. +- `src/handler.py` validates and shapes filter, campaign, general, and exact-route reports. +- `bootstrap.sql` creates the read-only Redshift database role and grants only + the reporting objects required by the handler. Its route event view and + materialized projection expose only timestamp, pseudonymous join key, + session, source group, semantic click, challenge join key, and form lifecycle + fields; the Lambda role cannot select the raw event table. +- `collector-host-migration.yaml` creates `events.` on the existing + ingestion ALB so `analytics.` can become the reporting UI host. +- `tests/test_handler.py` verifies authorization, validation, parameterization, + privacy-safe click shaping, and response contracts without AWS access. + +## Development deployment order + +Use `us-east-1` and account `811668436784`. Resolve every ARN and hosted-zone ID +from AWS immediately before deployment; do not paste credentials or managed +secret values into parameters, source files, or shell history. + +1. Validate both templates and run the local tests. +2. Deploy `collector-host-migration.yaml` against the existing HTTPS listener + and ingestion target group. +3. Verify `https://events.topcoder-dev.com/ping?appId=topcoder_web` and a browser + preflight/request to `/collect` before changing either client configuration. +4. Package `src/handler.py` as a versioned zip in the encrypted Clickstream + templates bucket. +5. Run `bootstrap.sql` through the Data API using the Redshift namespace's + managed administrator secret before deploying code that reads a new + reporting relation. On reapplication, omit existing `CREATE ROLE` and + `CREATE MATERIALIZED VIEW` statements, then run the replaceable view and + idempotent `GRANT` statements. +6. Deploy `template.yaml` with `CAPABILITY_NAMED_IAM`, the shared HTTP API ID, + and the exact workgroup, wildcard certificate, public hosted zone, code + bucket, and code key. +7. Exercise Lambda directly with missing, wrong, and exact role claims, then + exercise the public API with no token, an unauthorized token, and an + authorized token. A direct invocation does not replace the positive public + JWT test. +8. Add `analytics.topcoder-dev.com` to the Platform UI CloudFront distribution, + deploy the verified Platform UI build, and only then change its Route 53 + alias from the ingestion ALB to CloudFront. +9. Keep the former collector listener rules during the observation window. + Remove them only after both clients use `events.topcoder-dev.com` and ALB + traffic confirms the old host is idle. + +The collector move and reporting-host cutover are deliberately separate. If +the new collector fails, leave `analytics.topcoder-dev.com` on the ALB and roll +the client endpoint back. If the UI deployment fails after the collector move, +the `events` hostname can remain active without changing reporting DNS. + +## Validation commands + +```bash +python3 -m unittest discover -s infrastructure/analytics-api/tests -v +python3 -m py_compile infrastructure/analytics-api/src/handler.py +aws cloudformation validate-template \ + --template-body file://infrastructure/analytics-api/collector-host-migration.yaml +aws cloudformation validate-template \ + --template-body file://infrastructure/analytics-api/template.yaml +``` + +After deployment, expected public authorization behavior is: + +```text +no or malformed bearer token -> 401 from API Gateway +valid token without analytics -> 403 from Lambda +valid token with analytics -> 200 aggregate JSON, or 202 until the query completes +``` + +The canonical development endpoint is +`https://api.topcoder-dev.com/v1/analytics`. Preserve existing shared-stage +route settings when adding 5 requests/second, burst 10, detailed metrics for +the four `GET` routes and the public `OPTIONS /v1/analytics/{proxy+}` route. + +`GET /v1/analytics/route` requires `path=/...` and supports the same inclusive +date range and optional surface as General Analytics. It returns only aggregate +route totals, mutually exclusive visitor-source buckets, new/returning counts, +semantic click locations, form lifecycle totals, abandonment field IDs, and an +ordered challenge funnel. Bounce is defined as a one-page entrance session, +average time is focused engagement per page view, and conversion is a unique +form completer or a visitor who registers for the exact challenge clicked from +the route, divided by route visitors. Submission likewise requires the same +challenge ID and follows that registration. +Winner attribution is intentionally null until a trusted challenge-results +event is added upstream. + +Also verify an invalid date returns `400`, an unsupported route returns `404`, +an opted-in cold query returns resumable `202` responses instead of `504`, +responses are `private, no-store`, and CloudWatch logs do not contain tokens, +filters, SQL, or record values. + +## Production promotion + +Provision production as a separate stack and database role. Change the shared +API ID/domain, workgroup, database, Auth0 issuer, audience, and role claim to +production values. Do not reuse the development Clickstream project, app ID, +S3 prefixes, Redshift namespace, Lambda role, or collector hostname. diff --git a/infrastructure/analytics-api/bootstrap.sql b/infrastructure/analytics-api/bootstrap.sql new file mode 100644 index 000000000..9682fdd71 --- /dev/null +++ b/infrastructure/analytics-api/bootstrap.sql @@ -0,0 +1,122 @@ +-- Run as the managed Redshift administrator after deploying the Lambda role. +-- CREATE ROLE is intentionally separate because Redshift does not support +-- CREATE ROLE IF NOT EXISTS. Reapplication may begin at the GRANT statements. + +CREATE ROLE analytics_api_reader; + +-- Expose only the event fields required by the route report. Keeping this +-- projection separate from event_v2 prevents the Lambda database role from +-- reading device, location, free-text, or other raw event attributes. +CREATE OR REPLACE VIEW topcoder_web.route_analytics_events_v1 AS +SELECT + event_timestamp, + event_timestamp::date AS event_date, + COALESCE(NULLIF(user_id, ''), user_pseudo_id) AS analytics_user_id, + event_name, + NULLIF(JSON_EXTRACT_PATH_TEXT(custom_parameters_json_str, 'surface', true), '') AS surface, + COALESCE( + NULLIF(JSON_EXTRACT_PATH_TEXT(custom_parameters_json_str, 'page_path', true), ''), + NULLIF(page_view_page_url_path, '') + ) AS page_path, + CASE + WHEN LOWER(COALESCE(traffic_source_channel_group, '')) LIKE '%email%' + OR LOWER(COALESCE(traffic_source_medium, '')) IN ('email', 'e-mail', 'newsletter') + THEN 'email' + WHEN LOWER(COALESCE(traffic_source_channel_group, '')) LIKE '%paid%' + OR LOWER(COALESCE(traffic_source_medium, '')) IN ( + 'affiliate', 'cpc', 'cpm', 'cpv', 'display', 'paid', 'paid_search', 'ppc' + ) + OR NULLIF(traffic_source_clid, '') IS NOT NULL + THEN 'paid' + WHEN LOWER(COALESCE(traffic_source_channel_group, '')) LIKE '%social%' + OR LOWER(COALESCE(traffic_source_medium, '')) IN ('social', 'social-media', 'social_media') + THEN 'social' + WHEN LOWER(COALESCE(traffic_source_channel_group, '')) LIKE '%organic%' + OR LOWER(COALESCE(traffic_source_medium, '')) IN ('organic', 'organic_search', 'seo') + OR LOWER(COALESCE(traffic_source_category, '')) = 'search' + THEN 'organic' + ELSE 'other' + END AS source_group, + session_id, + session_number, + page_view_entrances, + user_engagement_time_msec, + NULLIF(JSON_EXTRACT_PATH_TEXT(custom_parameters_json_str, 'placement', true), '') AS placement, + NULLIF(JSON_EXTRACT_PATH_TEXT(custom_parameters_json_str, 'element_id', true), '') AS element_id, + NULLIF(JSON_EXTRACT_PATH_TEXT(custom_parameters_json_str, 'element_type', true), '') AS element_type, + NULLIF(JSON_EXTRACT_PATH_TEXT(custom_parameters_json_str, 'destination_host', true), '') AS destination_host, + NULLIF(JSON_EXTRACT_PATH_TEXT(custom_parameters_json_str, 'destination_path', true), '') AS destination_path, + NULLIF(JSON_EXTRACT_PATH_TEXT(custom_parameters_json_str, 'form_id', true), '') AS form_id, + NULLIF(JSON_EXTRACT_PATH_TEXT(custom_parameters_json_str, 'field_id', true), '') AS field_id, + NULLIF(JSON_EXTRACT_PATH_TEXT(custom_parameters_json_str, 'challenge_id', true), '') AS challenge_id +FROM topcoder_web.event_v2; + +-- Route reports reuse the same event subsets many times for session, form, +-- source, and challenge-funnel calculations. Materialize this narrow, +-- read-only projection so those calculations do not repeatedly extract JSON +-- fields from the wide Clickstream table. DISTSTYLE ALL is appropriate for +-- this small reporting relation and keeps its self-joins local. +CREATE MATERIALIZED VIEW topcoder_web.route_analytics_events_mv_v1 +BACKUP NO +DISTSTYLE ALL +SORTKEY(event_timestamp, page_path) +AUTO REFRESH YES AS +SELECT + event_timestamp, + event_timestamp::date AS event_date, + COALESCE(NULLIF(user_id, ''), user_pseudo_id) AS analytics_user_id, + event_name, + NULLIF(JSON_EXTRACT_PATH_TEXT(custom_parameters_json_str, 'surface', true), '') AS surface, + COALESCE( + NULLIF(JSON_EXTRACT_PATH_TEXT(custom_parameters_json_str, 'page_path', true), ''), + NULLIF(page_view_page_url_path, '') + ) AS page_path, + CASE + WHEN LOWER(COALESCE(traffic_source_channel_group, '')) LIKE '%email%' + OR LOWER(COALESCE(traffic_source_medium, '')) IN ('email', 'e-mail', 'newsletter') + THEN 'email' + WHEN LOWER(COALESCE(traffic_source_channel_group, '')) LIKE '%paid%' + OR LOWER(COALESCE(traffic_source_medium, '')) IN ( + 'affiliate', 'cpc', 'cpm', 'cpv', 'display', 'paid', 'paid_search', 'ppc' + ) + OR NULLIF(traffic_source_clid, '') IS NOT NULL + THEN 'paid' + WHEN LOWER(COALESCE(traffic_source_channel_group, '')) LIKE '%social%' + OR LOWER(COALESCE(traffic_source_medium, '')) IN ('social', 'social-media', 'social_media') + THEN 'social' + WHEN LOWER(COALESCE(traffic_source_channel_group, '')) LIKE '%organic%' + OR LOWER(COALESCE(traffic_source_medium, '')) IN ('organic', 'organic_search', 'seo') + OR LOWER(COALESCE(traffic_source_category, '')) = 'search' + THEN 'organic' + ELSE 'other' + END AS source_group, + session_id, + session_number, + page_view_entrances, + user_engagement_time_msec, + NULLIF(JSON_EXTRACT_PATH_TEXT(custom_parameters_json_str, 'placement', true), '') AS placement, + NULLIF(JSON_EXTRACT_PATH_TEXT(custom_parameters_json_str, 'element_id', true), '') AS element_id, + NULLIF(JSON_EXTRACT_PATH_TEXT(custom_parameters_json_str, 'element_type', true), '') AS element_type, + NULLIF(JSON_EXTRACT_PATH_TEXT(custom_parameters_json_str, 'destination_host', true), '') AS destination_host, + NULLIF(JSON_EXTRACT_PATH_TEXT(custom_parameters_json_str, 'destination_path', true), '') AS destination_path, + NULLIF(JSON_EXTRACT_PATH_TEXT(custom_parameters_json_str, 'form_id', true), '') AS form_id, + NULLIF(JSON_EXTRACT_PATH_TEXT(custom_parameters_json_str, 'field_id', true), '') AS field_id, + NULLIF(JSON_EXTRACT_PATH_TEXT(custom_parameters_json_str, 'challenge_id', true), '') AS challenge_id +FROM topcoder_web.event_v2 +WHERE event_name IN ( + '_page_view', + '_user_engagement', + 'ui_click', + 'form_viewed', + 'form_started', + 'form_completed', + 'form_abandoned', + 'challenge_registered', + 'challenge_submitted' +); + +GRANT USAGE ON SCHEMA topcoder_web TO ROLE analytics_api_reader; +GRANT SELECT ON topcoder_web.product_analytics_events_v1 TO ROLE analytics_api_reader; +GRANT SELECT ON topcoder_web.challenge_funnel_daily_v1 TO ROLE analytics_api_reader; +GRANT SELECT ON topcoder_web.route_analytics_events_v1 TO ROLE analytics_api_reader; +GRANT SELECT ON topcoder_web.route_analytics_events_mv_v1 TO ROLE analytics_api_reader; diff --git a/infrastructure/analytics-api/collector-host-migration.yaml b/infrastructure/analytics-api/collector-host-migration.yaml new file mode 100644 index 000000000..869815923 --- /dev/null +++ b/infrastructure/analytics-api/collector-host-migration.yaml @@ -0,0 +1,83 @@ +AWSTemplateFormatVersion: '2010-09-09' +Description: Preserves AWS Clickstream ingestion while analytics.topcoder-dev.com becomes a UI host. + +Parameters: + ApplicationId: + Type: String + Default: topcoder_web + EventsDomainName: + Type: String + Default: events.topcoder-dev.com + HostedZoneId: + Type: String + Description: Public Route 53 hosted zone ID for topcoder-dev.com. + IngestionAlbDnsName: + Type: String + IngestionAlbHostedZoneId: + Type: String + IngestionHttpsListenerArn: + Type: String + IngestionTargetGroupArn: + Type: String + +Resources: + EventsDns: + Type: AWS::Route53::RecordSet + Properties: + AliasTarget: + DNSName: !Ref IngestionAlbDnsName + HostedZoneId: !Ref IngestionAlbHostedZoneId + EvaluateTargetHealth: true + HostedZoneId: !Ref HostedZoneId + Name: !Ref EventsDomainName + Type: A + + EventsCollectRule: + Type: AWS::ElasticLoadBalancingV2::ListenerRule + Properties: + Actions: + - Type: forward + TargetGroupArn: !Ref IngestionTargetGroupArn + Conditions: + - Field: host-header + HostHeaderConfig: + Values: + - !Ref EventsDomainName + - Field: path-pattern + PathPatternConfig: + Values: + - /collect + - Field: query-string + QueryStringConfig: + Values: + - Key: appId + Value: !Ref ApplicationId + ListenerArn: !Ref IngestionHttpsListenerArn + Priority: 7 + + EventsPingRule: + Type: AWS::ElasticLoadBalancingV2::ListenerRule + Properties: + Actions: + - Type: forward + TargetGroupArn: !Ref IngestionTargetGroupArn + Conditions: + - Field: host-header + HostHeaderConfig: + Values: + - !Ref EventsDomainName + - Field: path-pattern + PathPatternConfig: + Values: + - /ping + - Field: query-string + QueryStringConfig: + Values: + - Key: appId + Value: !Ref ApplicationId + ListenerArn: !Ref IngestionHttpsListenerArn + Priority: 8 + +Outputs: + CollectorEndpoint: + Value: !Sub 'https://${EventsDomainName}/collect' diff --git a/infrastructure/analytics-api/src/handler.py b/infrastructure/analytics-api/src/handler.py new file mode 100644 index 000000000..c4f612589 --- /dev/null +++ b/infrastructure/analytics-api/src/handler.py @@ -0,0 +1,1853 @@ +"""Read-only Topcoder campaign and site analytics HTTP API. + +API Gateway verifies the current Topcoder Auth0 access token before invoking +this Lambda. The handler independently requires the exact ``analytics`` role, +validates every filter, and executes fixed parameterized queries against the +Topcoder AWS Clickstream reporting relations. +""" + +from __future__ import annotations + +import hashlib +import json +import os +import re +import time +from datetime import date, datetime, timedelta, timezone +from typing import Any + +import boto3 + + +MAX_DATE_RANGE_DAYS = 366 +MAX_RESULT_ROWS = 2_000 +MAX_QUERY_ATTEMPTS = 2 +REPORT_CACHE_SECONDS = 60 +FILTER_CACHE_SECONDS = 300 +# Redshift Data API retains idempotency tokens for eight hours. Four-hour +# windows leave a complete prior window available for safe boundary-crossing +# polls while allowing scheduled requests to prepare the default reports. +QUERY_TOKEN_WINDOW_SECONDS = 14_400 +QUERY_POLL_DELAY_MILLISECONDS = 1_000 +SAFE_FILTER_PATTERN = re.compile(r"^[A-Za-z0-9._~-]{1,100}$") +QUERY_TOKEN_PATTERN = re.compile(r"^[0-9a-f]{64}$") +NO_FILTER_PARAMETER = "*" + +_redshift_data = boto3.client("redshift-data") +_cache: dict[str, tuple[float, dict[str, Any]]] = {} + + +FILTERS_SQL = """ +WITH recent_events AS ( + SELECT event_date, utm_campaign, utm_campaign_id, utm_source, utm_medium, surface + FROM topcoder_web.product_analytics_events_v1 + WHERE event_date >= DATEADD(day, -365, CURRENT_DATE) +), +option_rows AS ( + SELECT 'campaign'::varchar AS row_type, utm_campaign::varchar AS value, COUNT(*)::bigint AS usage_count + FROM recent_events + WHERE utm_campaign IS NOT NULL + GROUP BY utm_campaign + UNION ALL + SELECT 'campaign_id', utm_campaign_id, COUNT(*)::bigint + FROM recent_events + WHERE utm_campaign_id IS NOT NULL + GROUP BY utm_campaign_id + UNION ALL + SELECT 'source', utm_source, COUNT(*)::bigint + FROM recent_events + WHERE utm_source IS NOT NULL + GROUP BY utm_source + UNION ALL + SELECT 'medium', utm_medium, COUNT(*)::bigint + FROM recent_events + WHERE utm_medium IS NOT NULL + GROUP BY utm_medium + UNION ALL + SELECT 'surface', surface, COUNT(*)::bigint + FROM recent_events + WHERE surface IS NOT NULL + GROUP BY surface +), +ranked_options AS ( + SELECT + row_type, + value, + usage_count, + ROW_NUMBER() OVER ( + PARTITION BY row_type + ORDER BY usage_count DESC, value + ) AS option_rank + FROM option_rows +) +SELECT 'meta' AS row_type, + CAST(MIN(event_date) AS varchar(10)) AS value, + CAST(MAX(event_date) AS varchar(10)) AS secondary_value, + COUNT(*)::bigint AS usage_count +FROM recent_events +UNION ALL +SELECT row_type, value, NULL, usage_count +FROM ranked_options +WHERE option_rank <= 200 +ORDER BY row_type, usage_count DESC, value +""" + + +CAMPAIGN_SQL = """ +WITH filtered_funnel AS ( + SELECT * + FROM topcoder_web.challenge_funnel_daily_v1 + WHERE cohort_date BETWEEN CAST(:from_date AS date) AND CAST(:to_date AS date) + AND (:campaign = '*' OR utm_campaign = :campaign) + AND (:campaign_id = '*' OR utm_campaign_id = :campaign_id) + AND (:source = '*' OR utm_source = :source) + AND (:medium = '*' OR utm_medium = :medium) +), +filtered_clicks AS ( + SELECT + event_date, + page_path, + placement, + element_id, + element_type, + destination_host, + destination_path, + analytics_user_id + FROM topcoder_web.product_analytics_events_v1 + WHERE event_name = 'ui_click' + AND event_date BETWEEN CAST(:from_date AS date) AND CAST(:to_date AS date) + AND (:campaign = '*' OR utm_campaign = :campaign) + AND (:campaign_id = '*' OR utm_campaign_id = :campaign_id) + AND (:source = '*' OR utm_source = :source) + AND (:medium = '*' OR utm_medium = :medium) +), +summary_row AS ( + SELECT + CAST(MAX(cohort_date) AS varchar(10)) AS data_through, + COALESCE(SUM(landing_users), 0)::bigint AS landing_users, + COALESCE(SUM(landing_clickers), 0)::bigint AS landing_clickers, + COALESCE(SUM(registered_after_click), 0)::bigint AS registrations, + COALESCE(SUM(submitted_after_registration), 0)::bigint AS submissions + FROM filtered_funnel +), +daily_rows AS ( + SELECT + cohort_date, + SUM(landing_users)::bigint AS landing_users, + SUM(landing_clickers)::bigint AS landing_clickers, + SUM(registered_after_click)::bigint AS registrations, + SUM(submitted_after_registration)::bigint AS submissions + FROM filtered_funnel + GROUP BY cohort_date +), +campaign_rows AS ( + SELECT + utm_campaign, + utm_campaign_id, + utm_source, + utm_medium, + SUM(landing_users)::bigint AS landing_users, + SUM(landing_clickers)::bigint AS landing_clickers, + SUM(registered_after_click)::bigint AS registrations, + SUM(submitted_after_registration)::bigint AS submissions + FROM filtered_funnel + GROUP BY utm_campaign, utm_campaign_id, utm_source, utm_medium + ORDER BY landing_users DESC, utm_campaign + LIMIT 100 +), +landing_rows AS ( + SELECT + landing_page_path, + SUM(landing_users)::bigint AS landing_users, + SUM(landing_clickers)::bigint AS landing_clickers, + SUM(registered_after_click)::bigint AS registrations, + SUM(submitted_after_registration)::bigint AS submissions + FROM filtered_funnel + GROUP BY landing_page_path + ORDER BY landing_users DESC, landing_page_path + LIMIT 50 +), +click_rows AS ( + SELECT + page_path, + placement, + element_id, + element_type, + destination_host, + destination_path, + COUNT(*)::bigint AS click_count, + COUNT(DISTINCT analytics_user_id)::bigint AS click_users + FROM filtered_clicks + GROUP BY + page_path, + placement, + element_id, + element_type, + destination_host, + destination_path + ORDER BY click_count DESC, page_path + LIMIT 100 +) +SELECT + 'summary'::varchar AS row_type, + NULL::varchar AS date_value, + data_through::varchar AS dimension_1, + NULL::varchar AS dimension_2, + NULL::varchar AS dimension_3, + NULL::varchar AS dimension_4, + NULL::varchar AS dimension_5, + NULL::varchar AS dimension_6, + NULL::varchar AS dimension_7, + landing_users::double precision AS metric_1, + landing_clickers::double precision AS metric_2, + registrations::double precision AS metric_3, + submissions::double precision AS metric_4, + NULL::double precision AS metric_5, + NULL::double precision AS metric_6, + NULL::double precision AS metric_7, + NULL::double precision AS metric_8 +FROM summary_row +UNION ALL +SELECT + 'daily', + CAST(cohort_date AS varchar(10)), + NULL, NULL, NULL, NULL, NULL, NULL, NULL, + landing_users, landing_clickers, registrations, submissions, + NULL, NULL, NULL, NULL +FROM daily_rows +UNION ALL +SELECT + 'campaign', + NULL, + utm_campaign, + utm_campaign_id, + utm_source, + utm_medium, + NULL, NULL, NULL, + landing_users, landing_clickers, registrations, submissions, + NULL, NULL, NULL, NULL +FROM campaign_rows +UNION ALL +SELECT + 'landing_page', + NULL, + landing_page_path, + NULL, NULL, NULL, NULL, NULL, NULL, + landing_users, landing_clickers, registrations, submissions, + NULL, NULL, NULL, NULL +FROM landing_rows +UNION ALL +SELECT + 'click_location', + NULL, + page_path, + placement, + element_id, + element_type, + destination_host, + destination_path, + NULL, + click_count, click_users, NULL, NULL, NULL, NULL, NULL, NULL +FROM click_rows +ORDER BY row_type, date_value, metric_1 DESC +""" + + +GENERAL_SQL = """ +WITH filtered_events AS ( + SELECT + event_timestamp::date AS event_date, + event_name, + analytics_user_id, + surface, + page_path, + utm_source + FROM topcoder_web.product_analytics_events_v1 + WHERE event_timestamp >= CAST(:from_date AS timestamp) + AND event_timestamp < DATEADD(day, 1, CAST(:to_date AS timestamp)) + AND (:surface = '*' OR surface = :surface) +), +aggregated_rows AS ( + SELECT + event_date, + surface, + page_path, + utm_source, + GROUPING(event_date) AS date_is_grouped, + GROUPING(surface) AS surface_is_grouped, + GROUPING(page_path) AS page_is_grouped, + GROUPING(utm_source) AS source_is_grouped, + MAX(event_date) AS data_through, + COUNT(CASE WHEN event_name = '_page_view' THEN 1 END)::bigint AS page_views, + COUNT(DISTINCT CASE WHEN event_name = '_page_view' THEN analytics_user_id END)::bigint AS visitors, + COUNT(CASE WHEN event_name = 'ui_click' THEN 1 END)::bigint AS clicks, + COUNT(DISTINCT CASE WHEN event_name = 'ui_click' THEN analytics_user_id END)::bigint AS clickers + FROM filtered_events + GROUP BY GROUPING SETS ( + (), + (event_date), + (surface, page_path), + (utm_source), + (surface) + ) +), +typed_rows AS ( + SELECT + CASE + WHEN date_is_grouped = 0 THEN 'daily' + WHEN page_is_grouped = 0 THEN 'page' + WHEN source_is_grouped = 0 THEN 'source' + WHEN surface_is_grouped = 0 THEN 'surface' + ELSE 'summary' + END::varchar AS row_type, + CASE + WHEN date_is_grouped = 0 THEN CAST(event_date AS varchar(10)) + END AS date_value, + CASE + WHEN date_is_grouped = 0 THEN NULL + WHEN page_is_grouped = 0 THEN surface + WHEN source_is_grouped = 0 THEN utm_source + WHEN surface_is_grouped = 0 THEN surface + ELSE CAST(data_through AS varchar(10)) + END::varchar AS dimension_1, + CASE + WHEN page_is_grouped = 0 THEN page_path + END::varchar AS dimension_2, + NULL::varchar AS dimension_3, + NULL::varchar AS dimension_4, + page_views::double precision AS metric_1, + visitors::double precision AS metric_2, + CASE + WHEN page_is_grouped = 0 OR source_is_grouped = 0 THEN NULL + ELSE clicks + END::double precision AS metric_3, + CASE + WHEN date_is_grouped = 0 + OR ( + date_is_grouped = 1 + AND surface_is_grouped = 1 + AND page_is_grouped = 1 + AND source_is_grouped = 1 + ) + THEN clickers + END::double precision AS metric_4 + FROM aggregated_rows + WHERE + ( + date_is_grouped = 1 + AND surface_is_grouped = 1 + AND page_is_grouped = 1 + AND source_is_grouped = 1 + ) + OR date_is_grouped = 0 + OR (page_is_grouped = 0 AND page_path IS NOT NULL AND page_views > 0) + OR (source_is_grouped = 0 AND page_views > 0) + OR ( + surface_is_grouped = 0 + AND page_is_grouped = 1 + AND surface IS NOT NULL + ) +), +ranked_rows AS ( + SELECT + *, + ROW_NUMBER() OVER ( + PARTITION BY row_type + ORDER BY metric_1 DESC, dimension_2, dimension_1 + ) AS row_rank + FROM typed_rows +) +SELECT + row_type, + date_value, + dimension_1, + dimension_2, + dimension_3, + dimension_4, + metric_1, + metric_2, + metric_3, + metric_4 +FROM ranked_rows +WHERE (row_type <> 'page' OR row_rank <= 50) + AND (row_type <> 'source' OR row_rank <= 25) + AND (row_type <> 'surface' OR row_rank <= 25) +ORDER BY row_type, date_value, metric_1 DESC +""" + + +ROUTE_SQL = """ +WITH date_events AS ( + SELECT * + FROM topcoder_web.route_analytics_events_mv_v1 + WHERE event_timestamp >= CAST(:from_date AS timestamp) + AND event_timestamp < DATEADD(day, 1, CAST(:to_date AS timestamp)) + AND event_name IN ( + '_page_view', + '_user_engagement', + 'ui_click', + 'form_viewed', + 'form_started', + 'form_completed', + 'form_abandoned', + 'challenge_registered', + 'challenge_submitted' + ) +), +route_events AS ( + SELECT * + FROM date_events + WHERE page_path = :path +), +route_page_views AS ( + SELECT * + FROM route_events + WHERE event_name = '_page_view' +), +ranked_first_views AS ( + SELECT + analytics_user_id, + event_timestamp, + session_number, + source_group, + ROW_NUMBER() OVER ( + PARTITION BY analytics_user_id + ORDER BY event_timestamp + ) AS view_rank + FROM route_page_views +), +first_views AS ( + SELECT analytics_user_id, event_timestamp, session_number, source_group + FROM ranked_first_views + WHERE view_rank = 1 +), +route_visitors AS ( + SELECT analytics_user_id, event_timestamp AS viewed_at + FROM first_views +), +session_page_counts AS ( + SELECT analytics_user_id, session_id, COUNT(*)::bigint AS page_views + FROM date_events + WHERE event_name = '_page_view' AND session_id IS NOT NULL + GROUP BY analytics_user_id, session_id +), +entry_sessions AS ( + SELECT DISTINCT analytics_user_id, session_id + FROM route_page_views + WHERE page_view_entrances IS TRUE AND session_id IS NOT NULL +), +bounce_sessions AS ( + SELECT entry.analytics_user_id, entry.session_id + FROM entry_sessions entry + JOIN session_page_counts session + ON session.analytics_user_id = entry.analytics_user_id + AND session.session_id = entry.session_id + WHERE session.page_views = 1 +), +route_engagement AS ( + SELECT user_engagement_time_msec + FROM route_events + WHERE event_name = '_user_engagement' +), +route_clicks AS ( + SELECT * + FROM route_events + WHERE event_name = 'ui_click' +), +challenge_click_candidates AS ( + SELECT + visitor.analytics_user_id, + click.event_timestamp, + CASE + WHEN click.destination_path LIKE '/opportunities/challenge/%' + THEN NULLIF(SPLIT_PART(click.destination_path, '/', 4), '') + WHEN click.destination_path LIKE '/challenges/%' + THEN NULLIF(SPLIT_PART(click.destination_path, '/', 3), '') + WHEN click.destination_path LIKE '/earn/challenges/%' + THEN NULLIF(SPLIT_PART(click.destination_path, '/', 4), '') + ELSE NULL + END AS challenge_id + FROM route_visitors visitor + JOIN route_clicks click + ON click.analytics_user_id = visitor.analytics_user_id + AND click.event_timestamp >= visitor.viewed_at + WHERE click.destination_path LIKE '/opportunities/challenge/%' + OR click.destination_path LIKE '/challenges/%' + OR click.destination_path LIKE '/earn/challenges/%' +), +challenge_cta_clicks AS ( + SELECT analytics_user_id, challenge_id, MIN(event_timestamp) AS clicked_at + FROM challenge_click_candidates + WHERE challenge_id IS NOT NULL + GROUP BY analytics_user_id, challenge_id +), +registrations AS ( + SELECT + click.analytics_user_id, + click.challenge_id, + MIN(event.event_timestamp) AS registered_at + FROM challenge_cta_clicks click + JOIN date_events event + ON event.analytics_user_id = click.analytics_user_id + AND event.event_name = 'challenge_registered' + AND event.challenge_id = click.challenge_id + AND event.event_timestamp >= click.clicked_at + GROUP BY click.analytics_user_id, click.challenge_id +), +submissions AS ( + SELECT + registration.analytics_user_id, + registration.challenge_id, + MIN(event.event_timestamp) AS submitted_at + FROM registrations registration + JOIN date_events event + ON event.analytics_user_id = registration.analytics_user_id + AND event.event_name = 'challenge_submitted' + AND event.challenge_id = registration.challenge_id + AND event.event_timestamp >= registration.registered_at + GROUP BY registration.analytics_user_id, registration.challenge_id +), +route_form_events AS ( + SELECT * + FROM route_events + WHERE event_name IN ('form_viewed', 'form_started', 'form_completed', 'form_abandoned') + AND form_id IS NOT NULL +), +conversion_users AS ( + SELECT DISTINCT event.analytics_user_id + FROM route_form_events event + JOIN route_visitors visitor + ON visitor.analytics_user_id = event.analytics_user_id + AND event.event_timestamp >= visitor.viewed_at + WHERE event.event_name = 'form_completed' + UNION + SELECT analytics_user_id + FROM registrations +), +summary_row AS ( + SELECT + CAST(MAX(event_date) AS varchar(10)) AS data_through, + COUNT(*)::bigint AS page_views, + COUNT(DISTINCT analytics_user_id)::bigint AS visitors, + (SELECT COUNT(*) FROM route_clicks)::bigint AS clicks, + (SELECT COUNT(DISTINCT analytics_user_id) FROM route_clicks)::bigint AS clickers, + (SELECT COUNT(*) FROM first_views WHERE session_number = 1)::bigint AS new_visitors, + (SELECT COUNT(*) FROM first_views WHERE session_number > 1)::bigint AS returning_visitors, + ROUND( + COALESCE((SELECT SUM(user_engagement_time_msec) FROM route_engagement), 0)::decimal(20, 2) + / NULLIF(COUNT(*), 0) + / 1000, + 2 + )::double precision AS avg_engagement_seconds, + (SELECT COUNT(*) FROM entry_sessions)::bigint AS entrances, + (SELECT COUNT(*) FROM bounce_sessions)::bigint AS bounces, + (SELECT COUNT(*) FROM conversion_users)::bigint AS conversions, + (SELECT COUNT(*) FROM route_form_events WHERE event_name = 'form_started')::bigint AS form_starts, + (SELECT COUNT(*) FROM route_form_events WHERE event_name = 'form_completed')::bigint AS form_completions, + (SELECT COUNT(*) FROM route_form_events WHERE event_name = 'form_abandoned')::bigint AS form_abandonments + FROM route_page_views +), +source_rows AS ( + SELECT source_group, COUNT(*)::bigint AS visitors + FROM first_views + GROUP BY source_group +), +click_rows AS ( + SELECT + placement, + element_id, + element_type, + destination_host, + destination_path, + COUNT(*)::bigint AS clicks, + COUNT(DISTINCT analytics_user_id)::bigint AS clickers + FROM route_clicks + GROUP BY placement, element_id, element_type, destination_host, destination_path + ORDER BY clicks DESC, element_id, destination_path + LIMIT 100 +), +form_rows AS ( + SELECT + form_id, + COUNT(CASE WHEN event_name = 'form_viewed' THEN 1 END)::bigint AS form_views, + COUNT(DISTINCT CASE WHEN event_name = 'form_viewed' THEN analytics_user_id END)::bigint AS form_viewers, + COUNT(CASE WHEN event_name = 'form_started' THEN 1 END)::bigint AS form_starts, + COUNT(DISTINCT CASE WHEN event_name = 'form_started' THEN analytics_user_id END)::bigint AS form_starters, + COUNT(CASE WHEN event_name = 'form_completed' THEN 1 END)::bigint AS form_completions, + COUNT(DISTINCT CASE WHEN event_name = 'form_completed' THEN analytics_user_id END)::bigint AS form_completers, + COUNT(CASE WHEN event_name = 'form_abandoned' THEN 1 END)::bigint AS form_abandonments, + COUNT(DISTINCT CASE WHEN event_name = 'form_abandoned' THEN analytics_user_id END)::bigint AS form_abandoners + FROM route_form_events + GROUP BY form_id + ORDER BY form_views DESC, form_id + LIMIT 50 +), +abandonment_rows AS ( + SELECT + form_id, + COALESCE(field_id, 'Unknown field') AS field_id, + COUNT(*)::bigint AS abandonments, + COUNT(DISTINCT analytics_user_id)::bigint AS visitors + FROM route_form_events + WHERE event_name = 'form_abandoned' + GROUP BY form_id, COALESCE(field_id, 'Unknown field') + ORDER BY abandonments DESC, form_id, field_id + LIMIT 100 +), +funnel_row AS ( + SELECT + (SELECT COUNT(*) FROM route_visitors)::bigint AS page_visitors, + (SELECT COUNT(DISTINCT analytics_user_id) FROM challenge_cta_clicks)::bigint AS challenge_cta_clickers, + (SELECT COUNT(DISTINCT analytics_user_id) FROM registrations)::bigint AS registrations, + (SELECT COUNT(DISTINCT analytics_user_id) FROM submissions)::bigint AS submissions +) +SELECT + 'summary'::varchar AS row_type, + NULL::varchar AS dimension_1, + NULL::varchar AS dimension_2, + NULL::varchar AS dimension_3, + NULL::varchar AS dimension_4, + NULL::varchar AS dimension_5, + data_through::varchar AS dimension_6, + page_views::double precision AS metric_1, + visitors::double precision AS metric_2, + clicks::double precision AS metric_3, + clickers::double precision AS metric_4, + new_visitors::double precision AS metric_5, + returning_visitors::double precision AS metric_6, + avg_engagement_seconds::double precision AS metric_7, + entrances::double precision AS metric_8, + bounces::double precision AS metric_9, + conversions::double precision AS metric_10, + form_starts::double precision AS metric_11, + form_completions::double precision AS metric_12, + form_abandonments::double precision AS metric_13 +FROM summary_row +UNION ALL +SELECT + 'source', source_group, NULL, NULL, NULL, NULL, NULL, + visitors, + NULL, NULL, NULL, NULL, NULL, NULL, + NULL, NULL, NULL, NULL, NULL, NULL +FROM source_rows +UNION ALL +SELECT + 'click_location', placement, element_id, element_type, destination_host, destination_path, NULL, + clicks, clickers, + NULL, NULL, NULL, NULL, NULL, NULL, + NULL, NULL, NULL, NULL, NULL +FROM click_rows +UNION ALL +SELECT + 'form', form_id, NULL, NULL, NULL, NULL, NULL, + form_views, form_viewers, form_starts, form_starters, form_completions, form_completers, + form_abandonments, form_abandoners, + NULL, NULL, NULL, NULL, NULL +FROM form_rows +UNION ALL +SELECT + 'form_abandonment', form_id, field_id, NULL, NULL, NULL, NULL, + abandonments, visitors, + NULL, NULL, NULL, NULL, NULL, NULL, + NULL, NULL, NULL, NULL, NULL +FROM abandonment_rows +UNION ALL +SELECT + 'funnel', NULL, NULL, NULL, NULL, NULL, NULL, + page_visitors, challenge_cta_clickers, registrations, submissions, + NULL, NULL, NULL, NULL, NULL, NULL, + NULL, NULL, NULL +FROM funnel_row +ORDER BY row_type, metric_1 DESC, dimension_1 +""" + +ROUTE_SURFACE_SQL = ROUTE_SQL.replace( + " WHERE page_path = :path\n", + " WHERE page_path = :path\n AND surface = :surface\n", + 1, +) + + +class QueryFailure(RuntimeError): + """Raised when Redshift rejects or aborts a reporting query.""" + + +class QueryTimeout(RuntimeError): + """Raised with a reusable query token when work outlives one HTTP deadline.""" + + +def handler(event: dict[str, Any], context: Any) -> dict[str, Any]: + """Authorize and serve one analytics endpoint. + + Args: + event: API Gateway HTTP API v2 proxy event. + context: Lambda invocation context used to bound query wait time. + + Returns: + An API Gateway proxy response containing a JSON analytics document. + + Raises: + No exceptions escape; failures are converted to sanitized HTTP errors. + """ + + request_id = _request_id(event, context) + asynchronous = False + try: + route_key = str(event.get("routeKey", "")) + if route_key.startswith("OPTIONS "): + return _preflight_response() + + if not _has_required_role(event): + return _response(403, {"message": "Analytics access is not permitted", "requestId": request_id}) + + query = event.get("queryStringParameters") or {} + asynchronous = _asynchronous_query_requested(query) + query_token = _resume_query_token(query, asynchronous) + + if route_key == "GET /v1/analytics/filters": + return _response( + 200, + _cached_report( + "filters", + FILTER_CACHE_SECONDS, + lambda: _filters_report(context, query_token), + context, + ), + ) + if route_key == "GET /v1/analytics/campaign": + filters = _campaign_filters(query) + cache_key = f"campaign:{json.dumps(filters, sort_keys=True)}" + report = _cached_report( + cache_key, + REPORT_CACHE_SECONDS, + lambda: _campaign_report(filters, context, query_token), + context, + ) + return _response(200, report) + if route_key == "GET /v1/analytics/general": + filters = _general_filters(query) + cache_key = f"general:{json.dumps(filters, sort_keys=True)}" + report = _cached_report( + cache_key, + REPORT_CACHE_SECONDS, + lambda: _general_report(filters, context, query_token), + context, + ) + return _response(200, report) + if route_key == "GET /v1/analytics/route": + filters = _route_filters(query) + cache_key = f"route:{json.dumps(filters, sort_keys=True)}" + report = _cached_report( + cache_key, + REPORT_CACHE_SECONDS, + lambda: _route_report(filters, context, query_token), + context, + ) + return _response(200, report) + + return _response(404, {"message": "Analytics route not found", "requestId": request_id}) + except ValueError as error: + return _response(400, {"message": str(error), "requestId": request_id}) + except QueryTimeout as error: + if asynchronous: + return _response( + 202, + { + "status": "pending", + "queryToken": str(error), + "retryAfterMs": QUERY_POLL_DELAY_MILLISECONDS, + "requestId": request_id, + }, + {"Retry-After": str(QUERY_POLL_DELAY_MILLISECONDS // 1_000)}, + ) + return _response(504, {"message": "Analytics data is still being prepared. Please retry.", "requestId": request_id}) + except QueryFailure: + _log_error(request_id, "redshift-query-failed") + return _response(502, {"message": "Analytics data could not be loaded", "requestId": request_id}) + except Exception: + _log_error(request_id, "unhandled-analytics-error") + return _response(500, {"message": "Analytics data could not be loaded", "requestId": request_id}) + + +def _cached_report( + key: str, + lifetime_seconds: int, + loader: Any, + context: Any, +) -> dict[str, Any]: + """Return a short-lived cached report or invoke its loader. + + Args: + key: Stable cache key containing only validated filters. + lifetime_seconds: Maximum age of a cached response. + loader: Zero-argument callable that loads the response. + context: Lambda context retained for a uniform loader signature. + + Returns: + Cached or newly loaded analytics report. + + Raises: + Propagates loader failures so the handler can sanitize them. + """ + + del context + now = time.monotonic() + cached = _cache.get(key) + if cached and cached[0] > now: + return cached[1] + result = loader() + if len(_cache) >= 50: + _cache.clear() + _cache[key] = (now + lifetime_seconds, result) + return result + + +def _filters_report(context: Any, query_token: str | None = None) -> dict[str, Any]: + """Load bounded campaign and surface filter options. + + Args: + context: Lambda context used to respect the remaining deadline. + query_token: Optional server-issued token that resumes a pending statement. + + Returns: + Filter option arrays and available event-date bounds. + + Raises: + QueryFailure or QueryTimeout when Redshift cannot return data. + """ + + rows = _execute_query(FILTERS_SQL, [], context, query_token) + options: dict[str, list[str]] = { + "campaigns": [], + "campaignIds": [], + "sources": [], + "mediums": [], + "surfaces": [], + } + min_date = None + max_date = None + key_by_row_type = { + "campaign": "campaigns", + "campaign_id": "campaignIds", + "source": "sources", + "medium": "mediums", + "surface": "surfaces", + } + for row in rows: + row_type = row.get("row_type") + if row_type == "meta": + min_date = row.get("value") + max_date = row.get("secondary_value") + continue + option_key = key_by_row_type.get(str(row_type)) + value = row.get("value") + if option_key and isinstance(value, str) and value: + options[option_key].append(value) + + return { + **options, + "generatedAt": _now_iso(), + "minDate": min_date, + "maxDate": max_date, + "dataThrough": max_date, + } + + +def _campaign_report( + filters: dict[str, str], + context: Any, + query_token: str | None = None, +) -> dict[str, Any]: + """Load and shape the ordered campaign funnel report. + + Args: + filters: Validated date and UTM filter values. + context: Lambda context used to respect the remaining deadline. + query_token: Optional server-issued token that resumes a pending statement. + + Returns: + Funnel totals, daily series, campaign/landing breakdowns, and click locations. + + Raises: + QueryFailure or QueryTimeout when Redshift cannot return data. + """ + + rows = _execute_query(CAMPAIGN_SQL, _sql_parameters(filters), context, query_token) + summary = next((row for row in rows if row.get("row_type") == "summary"), {}) + totals = { + "landingUsers": _integer(summary.get("metric_1")), + "landingClickers": _integer(summary.get("metric_2")), + "registrations": _integer(summary.get("metric_3")), + "submissions": _integer(summary.get("metric_4")), + } + totals.update({ + "clickThroughPercent": _percentage(totals["landingClickers"], totals["landingUsers"]), + "clickToRegistrationPercent": _percentage(totals["registrations"], totals["landingClickers"]), + "registrationToSubmissionPercent": _percentage(totals["submissions"], totals["registrations"]), + "landingToSubmissionPercent": _percentage(totals["submissions"], totals["landingUsers"]), + }) + + return { + "generatedAt": _now_iso(), + "dataThrough": summary.get("dimension_1"), + "filters": filters, + "totals": totals, + "series": [ + { + "date": row.get("date_value"), + "landingUsers": _integer(row.get("metric_1")), + "landingClickers": _integer(row.get("metric_2")), + "registrations": _integer(row.get("metric_3")), + "submissions": _integer(row.get("metric_4")), + } + for row in rows if row.get("row_type") == "daily" + ], + "campaigns": [ + { + "campaign": row.get("dimension_1") or "Direct", + "campaignId": row.get("dimension_2"), + "source": row.get("dimension_3") or "Direct", + "medium": row.get("dimension_4") or "None", + "landingUsers": _integer(row.get("metric_1")), + "landingClickers": _integer(row.get("metric_2")), + "registrations": _integer(row.get("metric_3")), + "submissions": _integer(row.get("metric_4")), + } + for row in rows if row.get("row_type") == "campaign" + ], + "landingPages": [ + { + "path": row.get("dimension_1") or "Unknown", + "landingUsers": _integer(row.get("metric_1")), + "landingClickers": _integer(row.get("metric_2")), + "registrations": _integer(row.get("metric_3")), + "submissions": _integer(row.get("metric_4")), + } + for row in rows if row.get("row_type") == "landing_page" + ], + "clickLocations": [ + _click_location(row) + for row in rows if row.get("row_type") == "click_location" + ], + } + + +def _general_report( + filters: dict[str, str], + context: Any, + query_token: str | None = None, +) -> dict[str, Any]: + """Load and shape general Topcoder site engagement analytics. + + Args: + filters: Validated date and optional surface filter values. + context: Lambda context used to respect the remaining deadline. + query_token: Optional server-issued token that resumes a pending statement. + + Returns: + General totals, daily series, pages, traffic sources, and surfaces. + + Raises: + QueryFailure or QueryTimeout when Redshift cannot return data. + """ + + rows = _execute_query(GENERAL_SQL, _sql_parameters(filters), context, query_token) + summary = next((row for row in rows if row.get("row_type") == "summary"), {}) + return { + "generatedAt": _now_iso(), + "dataThrough": summary.get("dimension_1"), + "filters": filters, + "totals": { + "pageViews": _integer(summary.get("metric_1")), + "visitors": _integer(summary.get("metric_2")), + "clicks": _integer(summary.get("metric_3")), + "clickers": _integer(summary.get("metric_4")), + }, + "series": [ + { + "date": row.get("date_value"), + "pageViews": _integer(row.get("metric_1")), + "visitors": _integer(row.get("metric_2")), + "clicks": _integer(row.get("metric_3")), + "clickers": _integer(row.get("metric_4")), + } + for row in rows if row.get("row_type") == "daily" + ], + "pages": [ + { + "surface": row.get("dimension_1") or "Unknown", + "path": row.get("dimension_2") or "Unknown", + "pageViews": _integer(row.get("metric_1")), + "visitors": _integer(row.get("metric_2")), + } + for row in rows if row.get("row_type") == "page" + ], + "sources": [ + { + "source": row.get("dimension_1") or "Direct", + "pageViews": _integer(row.get("metric_1")), + "visitors": _integer(row.get("metric_2")), + } + for row in rows if row.get("row_type") == "source" + ], + "surfaces": [ + { + "surface": row.get("dimension_1") or "Unknown", + "pageViews": _integer(row.get("metric_1")), + "visitors": _integer(row.get("metric_2")), + "clicks": _integer(row.get("metric_3")), + } + for row in rows if row.get("row_type") == "surface" + ], + } + + +def _route_report( + filters: dict[str, str], + context: Any, + query_token: str | None = None, +) -> dict[str, Any]: + """Load and shape detailed engagement analytics for one exact page path. + + Requests without a surface use a fixed query that omits the surface + predicate and parameter; filtered requests use the exact-match variant. + This lets Redshift prune the replicated route projection without evaluating + a bound wildcard branch during each reused event scan. + + Args: + filters: Validated date, surface, and query-free path filters. + context: Lambda context used to respect the remaining deadline. + query_token: Optional server-issued token that resumes a pending statement. + + Returns: + Route totals, visitor-source groups, click locations, form activity, and + the ordered challenge funnel. No visitor identifiers are returned. + + Raises: + QueryFailure or QueryTimeout when Redshift cannot return data. + """ + + query_filters = dict(filters) + query_sql = ROUTE_SQL + if filters["surface"]: + query_sql = ROUTE_SURFACE_SQL + else: + query_filters.pop("surface") + + rows = _execute_query(query_sql, _sql_parameters(query_filters), context, query_token) + summary = next((row for row in rows if row.get("row_type") == "summary"), {}) + funnel_row = next((row for row in rows if row.get("row_type") == "funnel"), {}) + visitors = _integer(summary.get("metric_2")) + clickers = _integer(summary.get("metric_4")) + entrances = _integer(summary.get("metric_8")) + bounces = _integer(summary.get("metric_9")) + conversions = _integer(summary.get("metric_10")) + new_visitors = _integer(summary.get("metric_5")) + returning_visitors = _integer(summary.get("metric_6")) + visitor_sources = { + "organic": 0, + "paid": 0, + "social": 0, + "email": 0, + "other": 0, + } + for row in rows: + source_group = row.get("dimension_1") + if row.get("row_type") == "source" and source_group in visitor_sources: + visitor_sources[source_group] = _integer(row.get("metric_1")) + + page_visitors = _integer(funnel_row.get("metric_1")) + challenge_cta_clickers = _integer(funnel_row.get("metric_2")) + registrations = _integer(funnel_row.get("metric_3")) + submissions = _integer(funnel_row.get("metric_4")) + + return { + "generatedAt": _now_iso(), + "dataThrough": summary.get("dimension_6"), + "filters": filters, + "totals": { + "pageViews": _integer(summary.get("metric_1")), + "visitors": visitors, + "clicks": _integer(summary.get("metric_3")), + "clickers": clickers, + "clickThroughPercent": _percentage(clickers, visitors), + "newVisitors": new_visitors, + "returningVisitors": returning_visitors, + "unknownVisitorType": max(0, visitors - new_visitors - returning_visitors), + "averageEngagementSeconds": _number(summary.get("metric_7")), + "entrances": entrances, + "bounces": bounces, + "bounceRatePercent": _percentage(bounces, entrances), + "conversions": conversions, + "conversionRatePercent": _percentage(conversions, visitors), + "formStarts": _integer(summary.get("metric_11")), + "formCompletions": _integer(summary.get("metric_12")), + "formAbandonments": _integer(summary.get("metric_13")), + }, + "visitorSources": [ + { + "source": source, + "visitors": source_visitors, + "percent": _percentage(source_visitors, visitors), + } + for source, source_visitors in visitor_sources.items() + ], + "clickLocations": [ + _route_click_location(row, visitors) + for row in rows if row.get("row_type") == "click_location" + ], + "forms": [ + _route_form(row) + for row in rows if row.get("row_type") == "form" + ], + "formAbandonments": [ + { + "formId": row.get("dimension_1") or "Unknown form", + "fieldId": row.get("dimension_2") or "Unknown field", + "abandonments": _integer(row.get("metric_1")), + "visitors": _integer(row.get("metric_2")), + } + for row in rows if row.get("row_type") == "form_abandonment" + ], + "funnel": { + "pageVisitors": page_visitors, + "challengeCtaClickers": challenge_cta_clickers, + "registrations": registrations, + "submissions": submissions, + "wins": None, + "clickThroughPercent": _percentage(challenge_cta_clickers, page_visitors), + "clickToRegistrationPercent": _percentage(registrations, challenge_cta_clickers), + "registrationToSubmissionPercent": _percentage(submissions, registrations), + "winTrackingAvailable": False, + }, + } + + +def _route_click_location(row: dict[str, Any], visitors: int) -> dict[str, Any]: + """Convert one route click aggregate to its privacy-safe wire shape. + + Args: + row: Query row containing semantic element and destination dimensions. + visitors: Unique route visitors used as the location CTR denominator. + + Returns: + Aggregate click-location object with no raw coordinates or rendered text. + + Raises: + Does not raise. + """ + + clickers = _integer(row.get("metric_2")) + return { + "placement": row.get("dimension_1"), + "elementId": row.get("dimension_2"), + "elementType": row.get("dimension_3"), + "destinationHost": row.get("dimension_4"), + "destinationPath": row.get("dimension_5"), + "clicks": _integer(row.get("metric_1")), + "clickers": clickers, + "clickThroughPercent": _percentage(clickers, visitors), + } + + +def _route_form(row: dict[str, Any]) -> dict[str, Any]: + """Convert one route form aggregate to completion and abandonment metrics. + + Args: + row: Query row containing a safe form ID and lifecycle counts. + + Returns: + Aggregate form activity with rates based on form starts. + + Raises: + Does not raise. + """ + + starts = _integer(row.get("metric_3")) + completions = _integer(row.get("metric_5")) + abandonments = _integer(row.get("metric_7")) + return { + "formId": row.get("dimension_1") or "Unknown form", + "views": _integer(row.get("metric_1")), + "viewers": _integer(row.get("metric_2")), + "starts": starts, + "starters": _integer(row.get("metric_4")), + "completions": completions, + "completers": _integer(row.get("metric_6")), + "abandonments": abandonments, + "abandoners": _integer(row.get("metric_8")), + "completionRatePercent": _percentage(completions, starts), + "abandonmentRatePercent": _percentage(abandonments, starts), + } + + +def _click_location(row: dict[str, Any]) -> dict[str, Any]: + """Convert one normalized Redshift click-location row to the wire contract. + + Args: + row: Query result row with safe, aggregate click dimensions. + + Returns: + Camel-cased click-location object grouped by semantic element fields. + + Raises: + Does not raise. + """ + + return { + "pagePath": row.get("dimension_1") or "Unknown", + "placement": row.get("dimension_2"), + "elementId": row.get("dimension_3"), + "elementType": row.get("dimension_4"), + "destinationHost": row.get("dimension_5"), + "destinationPath": row.get("dimension_6"), + "clicks": _integer(row.get("metric_1")), + "clickers": _integer(row.get("metric_2")), + } + + +def _campaign_filters(query: dict[str, Any]) -> dict[str, str]: + """Validate campaign report dates and UTM dimensions. + + Args: + query: Untrusted API Gateway query string values. + + Returns: + Complete normalized filter dictionary. + + Raises: + ValueError for malformed dates, excessive ranges, or unsafe dimensions. + """ + + date_filters = _date_filters(query) + return { + **date_filters, + "campaign": _safe_filter(query.get("campaign"), "campaign"), + "campaignId": _safe_filter(query.get("campaignId"), "campaign ID"), + "source": _safe_filter(query.get("source"), "source"), + "medium": _safe_filter(query.get("medium"), "medium"), + } + + +def _general_filters(query: dict[str, Any]) -> dict[str, str]: + """Validate general report dates and surface. + + Args: + query: Untrusted API Gateway query string values. + + Returns: + Complete normalized filter dictionary. + + Raises: + ValueError for malformed dates, excessive ranges, or unsafe surface. + """ + + return { + **_date_filters(query), + "surface": _safe_filter(query.get("surface"), "surface"), + } + + +def _route_filters(query: dict[str, Any]) -> dict[str, str]: + """Validate the date, surface, and exact page path for a route report. + + Args: + query: Untrusted API Gateway query string values. + + Returns: + Complete normalized route filter dictionary. + + Raises: + ValueError for malformed dates, unsafe surfaces, or invalid paths. + """ + + return { + **_date_filters(query), + "surface": _safe_filter(query.get("surface"), "surface"), + "path": _safe_path(query.get("path")), + } + + +def _asynchronous_query_requested(query: dict[str, Any]) -> bool: + """Validate and resolve the opt-in asynchronous warehouse protocol. + + Args: + query: Untrusted API Gateway query string values. + + Returns: + True only when the caller explicitly requests ``async=true``. + + Raises: + ValueError when an unsupported async value is supplied. + """ + + value = query.get("async") + if value in (None, ""): + return False + if value != "true": + raise ValueError("The async query option must be true") + return True + + +def _resume_query_token(query: dict[str, Any], asynchronous: bool) -> str | None: + """Validate an opaque server-issued token used to resume one statement. + + Args: + query: Untrusted API Gateway query string values. + asynchronous: Whether the caller opted into asynchronous polling. + + Returns: + A normalized SHA-256 token, or null when starting a new query. + + Raises: + ValueError when a token is malformed or supplied without async polling. + """ + + value = query.get("queryToken") + if value in (None, ""): + return None + if not asynchronous: + raise ValueError("A query token requires async=true") + if not isinstance(value, str) or not QUERY_TOKEN_PATTERN.fullmatch(value): + raise ValueError("The query token is invalid") + return value + + +def _date_filters(query: dict[str, Any]) -> dict[str, str]: + """Parse an inclusive UTC date range with a safe 30-day default. + + Args: + query: Untrusted query values containing optional ``from`` and ``to``. + + Returns: + ISO date strings under ``from`` and ``to``. + + Raises: + ValueError when dates are invalid, reversed, future, or over 366 days. + """ + + today = datetime.now(timezone.utc).date() + to_date = _parse_date(query.get("to"), "to") if query.get("to") else today + from_date = _parse_date(query.get("from"), "from") if query.get("from") else to_date - timedelta(days=29) + if from_date > to_date: + raise ValueError("The from date must not be after the to date") + if to_date > today: + raise ValueError("The to date must not be in the future") + if (to_date - from_date).days + 1 > MAX_DATE_RANGE_DAYS: + raise ValueError(f"Analytics date ranges cannot exceed {MAX_DATE_RANGE_DAYS} days") + return {"from": from_date.isoformat(), "to": to_date.isoformat()} + + +def _parse_date(value: Any, label: str) -> date: + """Parse one strict ISO calendar date. + + Args: + value: Untrusted candidate date. + label: Field label used in the validation message. + + Returns: + Parsed date. + + Raises: + ValueError when the candidate is not exactly YYYY-MM-DD. + """ + + if not isinstance(value, str) or not re.fullmatch(r"\d{4}-\d{2}-\d{2}", value): + raise ValueError(f"The {label} date must use YYYY-MM-DD") + try: + return date.fromisoformat(value) + except ValueError as error: + raise ValueError(f"The {label} date must be valid") from error + + +def _safe_filter(value: Any, label: str) -> str: + """Validate one optional UTM or surface token. + + Args: + value: Untrusted query value. + label: Human-readable field label. + + Returns: + Empty string for no filter or the unchanged safe token. + + Raises: + ValueError when the value is not a bounded marketing token. + """ + + if value in (None, ""): + return "" + if not isinstance(value, str) or not SAFE_FILTER_PATTERN.fullmatch(value): + raise ValueError(f"The {label} filter contains unsupported characters") + return value + + +def _safe_path(value: Any) -> str: + """Validate one exact, query-free URL path used by the route lookup. + + Args: + value: Untrusted query value after API Gateway URL decoding. + + Returns: + Unchanged absolute path suitable for a named Data API parameter. + + Raises: + ValueError when the path is missing, relative, unbounded, contains a + query/fragment, or includes control characters. + """ + + if not isinstance(value, str) or not value: + raise ValueError("The page path is required") + if len(value) > 500: + raise ValueError("The page path cannot exceed 500 characters") + if not value.startswith("/") or "?" in value or "#" in value: + raise ValueError("The page path must be an absolute query-free path") + if any(ord(character) < 32 or ord(character) == 127 for character in value): + raise ValueError("The page path contains unsupported characters") + return value + + +def _sql_parameters(filters: dict[str, str]) -> list[dict[str, str]]: + """Map wire filters to Redshift Data API named parameters. + + Args: + filters: Validated campaign or general filter dictionary. + + Returns: + Data API parameter objects for keys referenced by the SQL template. Empty optional filters use + a non-empty sentinel that cannot pass public filter validation because Data API rejects empty values. + + Raises: + Does not raise. + """ + + names = { + "from": "from_date", + "to": "to_date", + "campaign": "campaign", + "campaignId": "campaign_id", + "source": "source", + "medium": "medium", + "surface": "surface", + "path": "path", + } + return [ + {"name": names[key], "value": value or NO_FILTER_PARAMETER} + for key, value in filters.items() + if key in names + ] + + +def _execute_query( + sql: str, + parameters: list[dict[str, str]], + context: Any, + query_token: str | None = None, +) -> list[dict[str, Any]]: + """Execute a fixed parameterized query with resumable timeout handling. + + Args: + sql: Server-owned SQL template. + parameters: Validated Data API named parameters. + context: Lambda context or null for the cached filter loader. + query_token: Optional server-issued token that resumes a pending statement. + + Returns: + Query rows keyed by Redshift column name. + + Raises: + QueryFailure after repeated provider failure or excess output and + QueryTimeout containing the reusable token when the shared deadline + expires. ValueError when a token does not match this query and its + validated parameters. + """ + + request: dict[str, Any] = { + "Database": os.environ["REDSHIFT_DATABASE"], + "Sql": sql, + "StatementName": "topcoder-analytics-read", + "WithEvent": False, + "WorkgroupName": os.environ["REDSHIFT_WORKGROUP"], + } + if parameters: + request["Parameters"] = parameters + deadline = time.monotonic() + _query_wait_seconds(context) + initial_attempt = _query_token_attempt(sql, parameters, query_token) if query_token else 0 + for attempt in range(initial_attempt, MAX_QUERY_ATTEMPTS): + client_token = query_token if query_token and attempt == initial_attempt else _query_client_token( + sql, + parameters, + attempt, + ) + request["ClientToken"] = client_token + statement_id = _redshift_data.execute_statement(**request)["Id"] + delay = 0.2 + while time.monotonic() < deadline: + status = _redshift_data.describe_statement(Id=statement_id) + if status["Status"] == "FINISHED": + return _statement_rows(statement_id) + if status["Status"] in {"FAILED", "ABORTED"}: + if attempt + 1 < MAX_QUERY_ATTEMPTS: + break + raise QueryFailure("Redshift reporting query failed") + time.sleep(delay) + delay = min(delay * 1.5, 1.0) + else: + raise QueryTimeout(client_token) + + raise QueryFailure("Redshift reporting query failed") + + +def _query_client_token( + sql: str, + parameters: list[dict[str, str]], + attempt: int, + token_window: int | None = None, +) -> str: + """Build a bounded idempotency key for one fixed reporting query. + + Args: + sql: Server-owned SQL template. + parameters: Validated named parameters. + attempt: Provider retry index; failed statements receive a new token. + token_window: Optional explicit four-hour bucket used to validate a resume token. + + Returns: + Sixty-four-character SHA-256 token stable within a four-hour window. + + Raises: + Does not raise for validated handler inputs and configured environment values. + """ + + effective_window = token_window if token_window is not None else int(time.time() // QUERY_TOKEN_WINDOW_SECONDS) + fingerprint = json.dumps({ + "attempt": attempt, + "database": os.environ["REDSHIFT_DATABASE"], + "parameters": parameters, + "sql": sql, + "tokenWindow": effective_window, + "workgroup": os.environ["REDSHIFT_WORKGROUP"], + }, separators=(",", ":"), sort_keys=True) + return hashlib.sha256(fingerprint.encode("utf-8")).hexdigest() + + +def _query_token_attempt( + sql: str, + parameters: list[dict[str, str]], + query_token: str, +) -> int: + """Verify a resume token belongs to this exact server-owned query. + + Args: + sql: Server-owned SQL template. + parameters: Validated Data API named parameters. + query_token: SHA-256 token returned by a pending response. + + Returns: + Provider attempt encoded into the valid current or prior-window token. + + Raises: + ValueError when the token belongs to another query, filter set, or expired window. + """ + + current_window = int(time.time() // QUERY_TOKEN_WINDOW_SECONDS) + for token_window in (current_window, current_window - 1): + for attempt in range(MAX_QUERY_ATTEMPTS): + expected = _query_client_token(sql, parameters, attempt, token_window) + if query_token == expected: + return attempt + raise ValueError("The query token is invalid or expired") + + +def _query_wait_seconds(context: Any) -> float: + """Calculate a provider wait that leaves time for a sanitized HTTP response. + + Args: + context: Lambda invocation context, or null in direct unit calls. + + Returns: + Wait duration between one and 24 seconds. + + Raises: + Does not raise. + """ + + if context and hasattr(context, "get_remaining_time_in_millis"): + return max(1.0, min(24.0, (context.get_remaining_time_in_millis() / 1_000) - 2.0)) + return 24.0 + + +def _statement_rows(statement_id: str) -> list[dict[str, Any]]: + """Page through one Data API result without exceeding the response contract. + + Args: + statement_id: Completed Data API statement identifier. + + Returns: + Decoded query rows. + + Raises: + QueryFailure when the query returns more than the allowed row bound. + """ + + rows: list[dict[str, Any]] = [] + next_token = None + while True: + request = {"Id": statement_id} + if next_token: + request["NextToken"] = next_token + page = _redshift_data.get_statement_result(**request) + columns = [column["name"] for column in page.get("ColumnMetadata", [])] + rows.extend({name: _field_value(field) for name, field in zip(columns, record)} + for record in page.get("Records", [])) + if len(rows) > MAX_RESULT_ROWS: + raise QueryFailure("Analytics query exceeded the result row bound") + next_token = page.get("NextToken") + if not next_token: + return rows + + +def _field_value(field: dict[str, Any]) -> Any: + """Decode one Redshift Data API union field. + + Args: + field: Data API field object. + + Returns: + Native scalar value or null. + + Raises: + Does not raise for supported Data API field shapes. + """ + + if field.get("isNull"): + return None + for key in ("stringValue", "longValue", "doubleValue", "booleanValue", "blobValue"): + if key in field: + return field[key] + return None + + +def _has_required_role(event: dict[str, Any]) -> bool: + """Require the exact role from supported API Gateway-verified Topcoder claims. + + Args: + event: API Gateway event containing JWT authorizer claims. + + Returns: + True only when the normalized role set contains the required role. + + Raises: + Does not raise; malformed claims deny access. + """ + + claims = (((event.get("requestContext") or {}).get("authorizer") or {}).get("jwt") or {}).get("claims") or {} + if not isinstance(claims, dict): + return False + + configured_claim = os.environ.get("HUMAN_ROLE_CLAIM", "https://topcoder-dev.com/roles") + claim_names = [configured_claim] + claim_names.extend( + key for key in claims + if isinstance(key, str) + and key != configured_claim + and (key == "roles" or key.endswith("/roles")) + ) + roles: list[str] = [] + for claim_name in claim_names: + roles.extend(_role_values(claims.get(claim_name))) + required = os.environ.get("REQUIRED_ROLE", "analytics").strip() + return required in {role.strip() for role in roles} + + +def _role_values(raw_roles: Any) -> list[str]: + """Normalize one verified JWT role-claim representation. + + Args: + raw_roles: List or string claim supplied by the API Gateway JWT authorizer. + + Returns: + String role values, preserving their original case for exact matching. + + Raises: + Does not raise; malformed claim values produce an empty list. + """ + + if isinstance(raw_roles, list): + return [role for role in raw_roles if isinstance(role, str)] + if isinstance(raw_roles, str): + try: + parsed = json.loads(raw_roles) + if isinstance(parsed, list): + return [role for role in parsed if isinstance(role, str)] + if isinstance(parsed, str): + return [parsed] + except json.JSONDecodeError: + return [ + part.strip("[]\"'") + for part in re.split(r"[\s,]+", raw_roles) + if part.strip("[]\"'") + ] + return [] + + +def _preflight_response() -> dict[str, Any]: + """Build the empty response used by the shared API's analytics preflight route. + + Returns: + HTTP API v2 response; the shared CloudFront response policy adds the + environment's public CORS headers. + + Raises: + Does not raise. + """ + + return { + "statusCode": 204, + "headers": { + "Cache-Control": "no-store", + "Vary": "Origin", + }, + "body": "", + } + + +def _response( + status_code: int, + body: dict[str, Any], + additional_headers: dict[str, str] | None = None, +) -> dict[str, Any]: + """Build a private JSON API Gateway proxy response. + + Args: + status_code: HTTP response status. + body: JSON-serializable response document. + additional_headers: Optional service-owned headers merged into the secure defaults. + + Returns: + HTTP API v2 Lambda proxy response. + + Raises: + Does not raise for the service-owned response shapes. + """ + + headers = { + "Cache-Control": "private, no-store", + "Content-Type": "application/json; charset=utf-8", + "Referrer-Policy": "no-referrer", + "Vary": "Authorization, Origin", + "X-Content-Type-Options": "nosniff", + "X-Frame-Options": "DENY", + } + if additional_headers: + headers.update(additional_headers) + return { + "statusCode": status_code, + "headers": headers, + "body": json.dumps(body, separators=(",", ":")), + } + + +def _request_id(event: dict[str, Any], context: Any) -> str: + """Resolve a provider-generated request identifier. + + Args: + event: API Gateway event. + context: Lambda context fallback. + + Returns: + Request identifier suitable for support correlation. + + Raises: + Does not raise. + """ + + gateway_id = (event.get("requestContext") or {}).get("requestId") + lambda_id = getattr(context, "aws_request_id", None) + return str(gateway_id or lambda_id or "unknown")[:128] + + +def _log_error(request_id: str, error_type: str) -> None: + """Write a bounded diagnostic without tokens, filters, or SQL. + + Args: + request_id: Provider-generated correlation identifier. + error_type: Service-owned error category. + + Returns: + Nothing after writing one structured log line. + + Raises: + Does not raise. + """ + + print(json.dumps({"level": "error", "requestId": request_id, "type": error_type})) + + +def _integer(value: Any) -> int: + """Convert a numeric aggregate to a non-negative integer. + + Args: + value: Data API numeric field. + + Returns: + Non-negative integer, defaulting to zero. + + Raises: + Does not raise for malformed provider values. + """ + + try: + return max(0, int(float(value or 0))) + except (TypeError, ValueError): + return 0 + + +def _number(value: Any) -> float: + """Convert a numeric aggregate to a non-negative two-decimal number. + + Args: + value: Data API numeric field. + + Returns: + Non-negative float rounded to two decimals, defaulting to zero. + + Raises: + Does not raise for malformed provider values. + """ + + try: + return round(max(0.0, float(value or 0)), 2) + except (TypeError, ValueError): + return 0.0 + + +def _percentage(numerator: int, denominator: int) -> float: + """Calculate a bounded conversion percentage. + + Args: + numerator: Successful later-stage count. + denominator: Eligible earlier-stage count. + + Returns: + Percentage rounded to two decimals, or zero for an empty denominator. + + Raises: + Does not raise. + """ + + if denominator <= 0: + return 0.0 + return round((numerator / denominator) * 100, 2) + + +def _now_iso() -> str: + """Return the current UTC timestamp for response freshness metadata. + + Returns: + ISO-8601 timestamp ending in ``Z``. + + Raises: + Does not raise. + """ + + return datetime.now(timezone.utc).isoformat().replace("+00:00", "Z") diff --git a/infrastructure/analytics-api/template.yaml b/infrastructure/analytics-api/template.yaml new file mode 100644 index 000000000..f376e579d --- /dev/null +++ b/infrastructure/analytics-api/template.yaml @@ -0,0 +1,412 @@ +AWSTemplateFormatVersion: '2010-09-09' +Description: Role-gated read-only Topcoder product analytics API. + +Parameters: + CodeBucket: + Type: String + Description: S3 bucket containing the versioned Lambda zip. + CodeKey: + Type: String + Description: S3 key containing the versioned Lambda zip. + CertificateArn: + Type: String + Description: us-east-1 ACM wildcard certificate for the custom API domain. + DatabaseName: + Type: String + Default: topcoder_web_dev + DomainName: + Type: String + Default: analytics-api.topcoder-dev.com + HostedZoneId: + Type: String + Description: Public Route 53 hosted zone ID for topcoder-dev.com. + HumanJwtAudience: + Type: String + Default: BXWXUWnilVUPdN01t2Se29Tw2ZYNGZvH + HumanJwtIssuer: + Type: String + Default: https://auth.topcoder-dev.com/ + HumanRoleClaim: + Type: String + Default: https://topcoder-dev.com/roles + SharedApiDomainName: + Type: String + Default: api.topcoder-dev.com + Description: Canonical public API hostname backed by the shared HTTP API. + SharedApiId: + Type: String + Default: oid1rrke97 + Description: Existing environment HTTP API behind the public CloudFront distribution. + WorkgroupArn: + Type: String + Description: Exact Redshift Serverless workgroup ARN allowed by the query role. + WorkgroupName: + Type: String + Default: clickstream-topcoder-web-dev + +Resources: + AnalyticsFunctionLogGroup: + Type: AWS::Logs::LogGroup + DeletionPolicy: Retain + UpdateReplacePolicy: Retain + Properties: + LogGroupName: /aws/lambda/topcoder-analytics-api-dev + RetentionInDays: 30 + Tags: + - Key: Application + Value: topcoder-web-analytics + - Key: CostCenter + Value: product-analytics + - Key: Environment + Value: dev + + AnalyticsFunctionRole: + Type: AWS::IAM::Role + Properties: + RoleName: topcoder-analytics-api-dev-query-role + AssumeRolePolicyDocument: + Version: '2012-10-17' + Statement: + - Effect: Allow + Principal: + Service: lambda.amazonaws.com + Action: sts:AssumeRole + Policies: + - PolicyName: analytics-read-only-query + PolicyDocument: + Version: '2012-10-17' + Statement: + - Sid: WriteOwnLogs + Effect: Allow + Action: + - logs:CreateLogStream + - logs:PutLogEvents + Resource: !Sub '${AnalyticsFunctionLogGroup.Arn}:*' + - Sid: ExecuteAgainstAnalyticsWorkgroup + Effect: Allow + Action: redshift-data:ExecuteStatement + Resource: !Ref WorkgroupArn + - Sid: ReadOwnStatement + Effect: Allow + Action: + - redshift-data:CancelStatement + - redshift-data:DescribeStatement + - redshift-data:GetStatementResult + Resource: '*' + Condition: + StringEquals: + redshift-data:statement-owner-iam-userid: '${aws:userid}' + - Sid: GetExactWorkgroupCredentials + Effect: Allow + Action: redshift-serverless:GetCredentials + Resource: !Ref WorkgroupArn + - Sid: ResolveDatabaseRoleTag + Effect: Allow + Action: + - tag:GetResources + - tag:GetTagKeys + Resource: '*' + Tags: + - Key: Application + Value: topcoder-web-analytics + - Key: CostCenter + Value: product-analytics + - Key: Environment + Value: dev + - Key: RedshiftDbRoles + Value: analytics_api_reader + + AnalyticsFunction: + Type: AWS::Lambda::Function + DependsOn: AnalyticsFunctionLogGroup + Properties: + FunctionName: topcoder-analytics-api-dev + Architectures: + - arm64 + Code: + S3Bucket: !Ref CodeBucket + S3Key: !Ref CodeKey + Description: Read-only role-gated Topcoder campaign and site analytics. + Environment: + Variables: + HUMAN_ROLE_CLAIM: !Ref HumanRoleClaim + REDSHIFT_DATABASE: !Ref DatabaseName + REDSHIFT_WORKGROUP: !Ref WorkgroupName + REQUIRED_ROLE: analytics + Handler: handler.handler + MemorySize: 256 + ReservedConcurrentExecutions: 5 + Role: !GetAtt AnalyticsFunctionRole.Arn + Runtime: python3.12 + Timeout: 28 + Tags: + - Key: Application + Value: topcoder-web-analytics + - Key: CostCenter + Value: product-analytics + - Key: Environment + Value: dev + + AnalyticsPrewarmRule: + Type: AWS::Events::Rule + Properties: + Description: Prepares default analytics results once per four-hour Data API token window. + ScheduleExpression: cron(0 0/4 * * ? *) + State: ENABLED + Targets: + - Arn: !GetAtt AnalyticsFunction.Arn + Id: analytics-filter-options + Input: !Sub >- + {"routeKey":"GET /v1/analytics/filters","queryStringParameters":{"async":"true"},"requestContext":{"requestId":"scheduled-filter-prewarm","authorizer":{"jwt":{"claims":{"${HumanRoleClaim}":["analytics"]}}}}} + - Arn: !GetAtt AnalyticsFunction.Arn + Id: analytics-default-campaign + Input: !Sub >- + {"routeKey":"GET /v1/analytics/campaign","queryStringParameters":{"async":"true"},"requestContext":{"requestId":"scheduled-campaign-prewarm","authorizer":{"jwt":{"claims":{"${HumanRoleClaim}":["analytics"]}}}}} + - Arn: !GetAtt AnalyticsFunction.Arn + Id: analytics-default-opportunities-route + Input: !Sub >- + {"routeKey":"GET /v1/analytics/route","queryStringParameters":{"path":"/opportunities","async":"true"},"requestContext":{"requestId":"scheduled-route-prewarm","authorizer":{"jwt":{"claims":{"${HumanRoleClaim}":["analytics"]}}}}} + + AnalyticsPrewarmInvokePermission: + Type: AWS::Lambda::Permission + Properties: + Action: lambda:InvokeFunction + FunctionName: !Ref AnalyticsFunction + Principal: events.amazonaws.com + SourceArn: !GetAtt AnalyticsPrewarmRule.Arn + + AnalyticsApiAccessLogGroup: + Type: AWS::Logs::LogGroup + DeletionPolicy: Retain + UpdateReplacePolicy: Retain + Properties: + LogGroupName: /aws/apigateway/topcoder-analytics-api-dev + RetentionInDays: 30 + Tags: + - Key: Application + Value: topcoder-web-analytics + - Key: CostCenter + Value: product-analytics + - Key: Environment + Value: dev + + AnalyticsHttpApi: + Type: AWS::ApiGatewayV2::Api + Properties: + Name: topcoder-analytics-api-dev + ProtocolType: HTTP + CorsConfiguration: + AllowHeaders: + - authorization + - content-type + AllowMethods: + - GET + AllowOrigins: + - https://analytics.topcoder-dev.com + - https://platform-ui.topcoder-dev.com + - https://local.topcoder-dev.com + MaxAge: 300 + Tags: + Application: topcoder-web-analytics + CostCenter: product-analytics + Environment: dev + + AnalyticsJwtAuthorizer: + Type: AWS::ApiGatewayV2::Authorizer + Properties: + ApiId: !Ref AnalyticsHttpApi + AuthorizerType: JWT + IdentitySource: + - $request.header.Authorization + JwtConfiguration: + Audience: + - !Ref HumanJwtAudience + Issuer: !Ref HumanJwtIssuer + Name: topcoder-human-jwt + + AnalyticsIntegration: + Type: AWS::ApiGatewayV2::Integration + Properties: + ApiId: !Ref AnalyticsHttpApi + IntegrationType: AWS_PROXY + IntegrationUri: !GetAtt AnalyticsFunction.Arn + PayloadFormatVersion: '2.0' + TimeoutInMillis: 29000 + + FiltersRoute: + Type: AWS::ApiGatewayV2::Route + Properties: + ApiId: !Ref AnalyticsHttpApi + AuthorizationType: JWT + AuthorizerId: !Ref AnalyticsJwtAuthorizer + RouteKey: GET /v1/analytics/filters + Target: !Sub 'integrations/${AnalyticsIntegration}' + + CampaignRoute: + Type: AWS::ApiGatewayV2::Route + Properties: + ApiId: !Ref AnalyticsHttpApi + AuthorizationType: JWT + AuthorizerId: !Ref AnalyticsJwtAuthorizer + RouteKey: GET /v1/analytics/campaign + Target: !Sub 'integrations/${AnalyticsIntegration}' + + GeneralRoute: + Type: AWS::ApiGatewayV2::Route + Properties: + ApiId: !Ref AnalyticsHttpApi + AuthorizationType: JWT + AuthorizerId: !Ref AnalyticsJwtAuthorizer + RouteKey: GET /v1/analytics/general + Target: !Sub 'integrations/${AnalyticsIntegration}' + + RouteDetailRoute: + Type: AWS::ApiGatewayV2::Route + Properties: + ApiId: !Ref AnalyticsHttpApi + AuthorizationType: JWT + AuthorizerId: !Ref AnalyticsJwtAuthorizer + RouteKey: GET /v1/analytics/route + Target: !Sub 'integrations/${AnalyticsIntegration}' + + DefaultStage: + Type: AWS::ApiGatewayV2::Stage + DependsOn: AnalyticsApiAccessLogGroup + Properties: + ApiId: !Ref AnalyticsHttpApi + StageName: $default + AutoDeploy: true + AccessLogSettings: + DestinationArn: !Sub 'arn:${AWS::Partition}:logs:${AWS::Region}:${AWS::AccountId}:log-group:/aws/apigateway/topcoder-analytics-api-dev' + Format: >- + {"requestId":"$context.requestId","routeKey":"$context.routeKey","status":"$context.status","responseLength":"$context.responseLength","integrationError":"$context.integrationErrorMessage"} + DefaultRouteSettings: + DetailedMetricsEnabled: true + ThrottlingBurstLimit: 10 + ThrottlingRateLimit: 5 + Tags: + Application: topcoder-web-analytics + CostCenter: product-analytics + Environment: dev + + AnalyticsInvokePermission: + Type: AWS::Lambda::Permission + Properties: + Action: lambda:InvokeFunction + FunctionName: !Ref AnalyticsFunction + Principal: apigateway.amazonaws.com + SourceArn: !Sub 'arn:${AWS::Partition}:execute-api:${AWS::Region}:${AWS::AccountId}:${AnalyticsHttpApi}/*/GET/v1/analytics/*' + + SharedAnalyticsJwtAuthorizer: + Type: AWS::ApiGatewayV2::Authorizer + Properties: + ApiId: !Ref SharedApiId + AuthorizerType: JWT + IdentitySource: + - $request.header.Authorization + JwtConfiguration: + Audience: + - !Ref HumanJwtAudience + Issuer: !Ref HumanJwtIssuer + Name: topcoder-human-jwt-analytics + + SharedAnalyticsIntegration: + Type: AWS::ApiGatewayV2::Integration + Properties: + ApiId: !Ref SharedApiId + IntegrationType: AWS_PROXY + IntegrationUri: !GetAtt AnalyticsFunction.Arn + PayloadFormatVersion: '2.0' + TimeoutInMillis: 29000 + + SharedFiltersRoute: + Type: AWS::ApiGatewayV2::Route + Properties: + ApiId: !Ref SharedApiId + AuthorizationType: JWT + AuthorizerId: !Ref SharedAnalyticsJwtAuthorizer + RouteKey: GET /v1/analytics/filters + Target: !Sub 'integrations/${SharedAnalyticsIntegration}' + + SharedCampaignRoute: + Type: AWS::ApiGatewayV2::Route + Properties: + ApiId: !Ref SharedApiId + AuthorizationType: JWT + AuthorizerId: !Ref SharedAnalyticsJwtAuthorizer + RouteKey: GET /v1/analytics/campaign + Target: !Sub 'integrations/${SharedAnalyticsIntegration}' + + SharedGeneralRoute: + Type: AWS::ApiGatewayV2::Route + Properties: + ApiId: !Ref SharedApiId + AuthorizationType: JWT + AuthorizerId: !Ref SharedAnalyticsJwtAuthorizer + RouteKey: GET /v1/analytics/general + Target: !Sub 'integrations/${SharedAnalyticsIntegration}' + + SharedRouteDetailRoute: + Type: AWS::ApiGatewayV2::Route + Properties: + ApiId: !Ref SharedApiId + AuthorizationType: JWT + AuthorizerId: !Ref SharedAnalyticsJwtAuthorizer + RouteKey: GET /v1/analytics/route + Target: !Sub 'integrations/${SharedAnalyticsIntegration}' + + SharedPreflightRoute: + Type: AWS::ApiGatewayV2::Route + Properties: + ApiId: !Ref SharedApiId + AuthorizationType: NONE + RouteKey: OPTIONS /v1/analytics/{proxy+} + Target: !Sub 'integrations/${SharedAnalyticsIntegration}' + + SharedAnalyticsInvokePermission: + Type: AWS::Lambda::Permission + Properties: + Action: lambda:InvokeFunction + FunctionName: !Ref AnalyticsFunction + Principal: apigateway.amazonaws.com + SourceArn: !Sub 'arn:${AWS::Partition}:execute-api:${AWS::Region}:${AWS::AccountId}:${SharedApiId}/*/*/v1/analytics/*' + + AnalyticsApiDomain: + Type: AWS::ApiGatewayV2::DomainName + Properties: + DomainName: !Ref DomainName + DomainNameConfigurations: + - CertificateArn: !Ref CertificateArn + EndpointType: REGIONAL + SecurityPolicy: TLS_1_2 + Tags: + Application: topcoder-web-analytics + CostCenter: product-analytics + Environment: dev + + AnalyticsApiMapping: + Type: AWS::ApiGatewayV2::ApiMapping + DependsOn: DefaultStage + Properties: + ApiId: !Ref AnalyticsHttpApi + DomainName: !Ref AnalyticsApiDomain + Stage: !Ref DefaultStage + + AnalyticsApiDns: + Type: AWS::Route53::RecordSet + Properties: + AliasTarget: + DNSName: !GetAtt AnalyticsApiDomain.RegionalDomainName + HostedZoneId: !GetAtt AnalyticsApiDomain.RegionalHostedZoneId + EvaluateTargetHealth: false + HostedZoneId: !Ref HostedZoneId + Name: !Ref DomainName + Type: A + +Outputs: + ApiEndpoint: + Value: !Sub 'https://${SharedApiDomainName}/v1/analytics' + FunctionArn: + Value: !GetAtt AnalyticsFunction.Arn + FunctionRoleName: + Value: !Ref AnalyticsFunctionRole diff --git a/infrastructure/analytics-api/tests/test_handler.py b/infrastructure/analytics-api/tests/test_handler.py new file mode 100644 index 000000000..2c0d8ff83 --- /dev/null +++ b/infrastructure/analytics-api/tests/test_handler.py @@ -0,0 +1,690 @@ +"""Unit tests for the role-gated analytics Lambda contract.""" + +from __future__ import annotations + +import importlib.util +import json +import os +import sys +import types +import unittest +from pathlib import Path +from unittest.mock import Mock, patch + + +HANDLER_PATH = Path(__file__).parents[1] / "src" / "handler.py" + + +class _FakeBoto3(types.ModuleType): + """Minimal boto3 module used while importing the dependency-free handler.""" + + def __init__(self) -> None: + super().__init__("boto3") + self.redshift = Mock() + + def client(self, service: str) -> Mock: + """Return the fake Redshift Data API client. + + Args: + service: Requested AWS service name. + + Returns: + Shared mock client. + + Raises: + AssertionError when the handler requests an unexpected service. + """ + + if service != "redshift-data": + raise AssertionError(f"Unexpected AWS client: {service}") + return self.redshift + + +def _load_handler() -> types.ModuleType: + """Import the Lambda handler with a fake boto3 module. + + Returns: + Fresh handler module. + + Raises: + RuntimeError when the module cannot be loaded from disk. + """ + + fake_boto3 = _FakeBoto3() + sys.modules["boto3"] = fake_boto3 + specification = importlib.util.spec_from_file_location("analytics_handler", HANDLER_PATH) + if specification is None or specification.loader is None: + raise RuntimeError("Unable to load analytics handler") + module = importlib.util.module_from_spec(specification) + specification.loader.exec_module(module) + return module + + +class AnalyticsHandlerTests(unittest.TestCase): + """Covers authorization, filter validation, and wire-shape transformations.""" + + @classmethod + def setUpClass(cls) -> None: + """Load the handler once with deterministic runtime configuration.""" + + os.environ.update({ + "HUMAN_ROLE_CLAIM": "https://topcoder-dev.com/roles", + "REDSHIFT_DATABASE": "topcoder_web_dev", + "REDSHIFT_WORKGROUP": "clickstream-topcoder-web-dev", + "REQUIRED_ROLE": "analytics", + }) + cls.module = _load_handler() + + def setUp(self) -> None: + """Clear the warm Lambda cache before each test.""" + + self.module._cache.clear() + + def _event( + self, + route_key: str, + roles: object = None, + query: dict[str, str] | None = None, + role_claim: str = "https://topcoder-dev.com/roles", + ) -> dict[str, object]: + """Build one API Gateway v2 event. + + Args: + route_key: API Gateway route key. + roles: Namespaced role claim value. + query: Optional query string parameters. + role_claim: Verified JWT claim name containing the roles. + + Returns: + Synthetic API Gateway event. + + Raises: + Does not raise. + """ + + claims = {} + if roles is not None: + claims[role_claim] = roles + return { + "queryStringParameters": query, + "requestContext": { + "authorizer": {"jwt": {"claims": claims}}, + "requestId": "request-123", + }, + "routeKey": route_key, + } + + def test_denies_missing_or_wrong_role_without_querying(self) -> None: + """A verified token still needs the exact analytics role.""" + + with patch.object(self.module, "_execute_query") as execute: + missing = self.module.handler(self._event("GET /v1/analytics/filters"), None) + wrong = self.module.handler( + self._event("GET /v1/analytics/filters", json.dumps(["administrator"])), + None, + ) + wrong_case = self.module.handler( + self._event("GET /v1/analytics/filters", json.dumps(["Analytics"])), + None, + ) + + self.assertEqual(403, missing["statusCode"]) + self.assertEqual(403, wrong["statusCode"]) + self.assertEqual(403, wrong_case["statusCode"]) + execute.assert_not_called() + + def test_accepts_json_array_role_claim_and_returns_private_filters(self) -> None: + """The API Gateway string form of an array role claim is supported.""" + + rows = [ + {"row_type": "meta", "value": "2026-08-01", "secondary_value": "2026-08-30"}, + {"row_type": "campaign", "value": "launch", "usage_count": 4}, + {"row_type": "source", "value": "newsletter", "usage_count": 4}, + ] + with patch.object(self.module, "_execute_query", return_value=rows): + response = self.module.handler( + self._event("GET /v1/analytics/filters", json.dumps(["analytics"])), + None, + ) + + body = json.loads(response["body"]) + self.assertEqual(200, response["statusCode"]) + self.assertEqual("private, no-store", response["headers"]["Cache-Control"]) + self.assertEqual(["launch"], body["campaigns"]) + self.assertEqual("2026-08-30", body["dataThrough"]) + + def test_accepts_verified_topcoder_role_claim_variants(self) -> None: + """V2/V3 Topcoder role namespaces and API Gateway string forms authorize identically.""" + + variants = [ + ("https://topcoder.com/roles", ["analytics"]), + ("roles", "analytics"), + ("roles", "[analytics]"), + ] + with patch.object(self.module, "_execute_query", return_value=[]): + for claim_name, roles in variants: + with self.subTest(claim_name=claim_name, roles=roles): + self.module._cache.clear() + response = self.module.handler( + self._event( + "GET /v1/analytics/filters", + roles, + role_claim=claim_name, + ), + None, + ) + self.assertEqual(200, response["statusCode"]) + + def test_async_timeout_returns_a_private_pending_poll_response(self) -> None: + """Opted-in clients receive a reusable token instead of an HTTP failure.""" + + query_token = "a" * 64 + with patch.object( + self.module, + "_execute_query", + side_effect=self.module.QueryTimeout(query_token), + ): + pending = self.module.handler( + self._event( + "GET /v1/analytics/filters", + ["analytics"], + {"async": "true"}, + ), + None, + ) + legacy = self.module.handler( + self._event("GET /v1/analytics/filters", ["analytics"]), + None, + ) + + body = json.loads(pending["body"]) + self.assertEqual(202, pending["statusCode"]) + self.assertEqual("1", pending["headers"]["Retry-After"]) + self.assertEqual("pending", body["status"]) + self.assertEqual(query_token, body["queryToken"]) + self.assertEqual(1_000, body["retryAfterMs"]) + self.assertEqual(504, legacy["statusCode"]) + + def test_rejects_invalid_async_options_and_resume_tokens(self) -> None: + """Only opted-in clients can send a bounded server-issued query token.""" + + cases = [ + {"async": "false"}, + {"async": "true", "queryToken": "not-a-token"}, + {"queryToken": "a" * 64}, + ] + with patch.object(self.module, "_execute_query") as execute: + responses = [ + self.module.handler( + self._event("GET /v1/analytics/filters", ["analytics"], query), + None, + ) + for query in cases + ] + + self.assertTrue(all(response["statusCode"] == 400 for response in responses)) + execute.assert_not_called() + + def test_shared_api_preflight_does_not_require_a_role(self) -> None: + """The public OPTIONS route returns no data and never queries Redshift.""" + + with patch.object(self.module, "_execute_query") as execute: + response = self.module.handler( + self._event("OPTIONS /v1/analytics/{proxy+}"), + None, + ) + + self.assertEqual(204, response["statusCode"]) + self.assertEqual("", response["body"]) + execute.assert_not_called() + + def test_rejects_invalid_or_excessive_date_ranges(self) -> None: + """Malformed and unbounded reporting requests fail before Redshift.""" + + with patch.object(self.module, "_execute_query") as execute: + invalid = self.module.handler( + self._event( + "GET /v1/analytics/general", + ["analytics"], + {"from": "2026-99-01", "to": "2026-08-30"}, + ), + None, + ) + excessive = self.module.handler( + self._event( + "GET /v1/analytics/general", + ["analytics"], + {"from": "2025-01-01", "to": "2026-01-02"}, + ), + None, + ) + + self.assertEqual(400, invalid["statusCode"]) + self.assertEqual(400, excessive["statusCode"]) + execute.assert_not_called() + + def test_rejects_unsafe_utm_values_before_querying(self) -> None: + """UTM values cannot alter fixed SQL templates.""" + + with patch.object(self.module, "_execute_query") as execute: + response = self.module.handler( + self._event( + "GET /v1/analytics/campaign", + ["analytics"], + {"campaign": "launch' OR 1=1 --"}, + ), + None, + ) + + self.assertEqual(400, response["statusCode"]) + execute.assert_not_called() + + def test_route_lookup_requires_a_bounded_query_free_absolute_path(self) -> None: + """Route reports reject missing, full-URL, query, and control-character values.""" + + invalid_paths = [None, "challenges", "https://www.topcoder.com/challenges", "/x?q=1", "/x\n"] + with patch.object(self.module, "_execute_query") as execute: + responses = [ + self.module.handler( + self._event( + "GET /v1/analytics/route", + ["analytics"], + {"path": path} if path is not None else {}, + ), + None, + ) + for path in invalid_paths + ] + + self.assertTrue(all(response["statusCode"] == 400 for response in responses)) + execute.assert_not_called() + + def test_unset_filters_use_nonempty_data_api_parameters(self) -> None: + """Optional filters use an unreachable sentinel because Data API rejects empty values.""" + + parameters = self.module._sql_parameters({ + "from": "2026-08-01", + "to": "2026-08-30", + "campaign": "", + "source": "newsletter", + }) + values = {parameter["name"]: parameter["value"] for parameter in parameters} + + self.assertEqual("*", values["campaign"]) + self.assertEqual("newsletter", values["source"]) + self.assertTrue(all(parameter["value"] for parameter in parameters)) + self.assertIsNone(self.module.SAFE_FILTER_PATTERN.fullmatch(self.module.NO_FILTER_PARAMETER)) + self.assertIn(":campaign = '*' OR", self.module.CAMPAIGN_SQL) + self.assertIn(":surface = '*' OR", self.module.GENERAL_SQL) + route_parameters = self.module._sql_parameters({"path": "/opportunities/challenge/example"}) + self.assertEqual( + [{"name": "path", "value": "/opportunities/challenge/example"}], + route_parameters, + ) + + def test_retries_one_failed_redshift_statement(self) -> None: + """A transient failed statement is retried once inside the request deadline.""" + + context = Mock() + context.get_remaining_time_in_millis.return_value = 28_000 + with ( + patch.object( + self.module._redshift_data, + "execute_statement", + side_effect=[{"Id": "failed"}, {"Id": "finished"}], + ) as execute, + patch.object( + self.module._redshift_data, + "describe_statement", + side_effect=[{"Status": "FAILED"}, {"Status": "FINISHED"}], + ), + patch.object( + self.module._redshift_data, + "get_statement_result", + return_value={ + "ColumnMetadata": [{"name": "value"}], + "Records": [[{"stringValue": "recovered"}]], + }, + ), + ): + rows = self.module._execute_query("SELECT 1", [], context) + + self.assertEqual([{"value": "recovered"}], rows) + self.assertEqual(2, execute.call_count) + self.assertNotEqual( + execute.call_args_list[0].kwargs["ClientToken"], + execute.call_args_list[1].kwargs["ClientToken"], + ) + + def test_stops_after_bounded_redshift_retries(self) -> None: + """Repeated failed statements surface an error after two attempts.""" + + context = Mock() + context.get_remaining_time_in_millis.return_value = 28_000 + with ( + patch.object( + self.module._redshift_data, + "execute_statement", + side_effect=[{"Id": "failed-1"}, {"Id": "failed-2"}], + ) as execute, + patch.object( + self.module._redshift_data, + "describe_statement", + side_effect=[{"Status": "FAILED"}, {"Status": "FAILED"}], + ), + self.assertRaises(self.module.QueryFailure), + ): + self.module._execute_query("SELECT 1", [], context) + + self.assertEqual(2, execute.call_count) + + def test_timeout_token_resumes_across_an_idempotency_window_boundary(self) -> None: + """A pending response pins its statement even when the clock enters a new token window.""" + + context = Mock() + context.get_remaining_time_in_millis.return_value = 28_000 + with ( + patch.object(self.module, "_query_wait_seconds", return_value=0), + patch.object( + self.module._redshift_data, + "execute_statement", + return_value={"Id": "still-running"}, + ) as execute, + patch.object(self.module._redshift_data, "cancel_statement") as cancel, + ): + with ( + patch.object( + self.module.time, + "time", + return_value=self.module.QUERY_TOKEN_WINDOW_SECONDS - 1, + ), + self.assertRaises(self.module.QueryTimeout) as initial_timeout, + ): + self.module._execute_query("SELECT 1", [], context) + query_token = str(initial_timeout.exception) + with ( + patch.object( + self.module.time, + "time", + return_value=self.module.QUERY_TOKEN_WINDOW_SECONDS + 1, + ), + self.assertRaises(self.module.QueryTimeout) as resumed_timeout, + ): + self.module._execute_query("SELECT 1", [], context, query_token) + + self.assertEqual(query_token, str(resumed_timeout.exception)) + self.assertRegex(query_token, r"^[0-9a-f]{64}$") + self.assertEqual(2, execute.call_count) + self.assertEqual( + execute.call_args_list[0].kwargs["ClientToken"], + execute.call_args_list[1].kwargs["ClientToken"], + ) + cancel.assert_not_called() + + def test_resume_token_cannot_be_reused_for_another_query(self) -> None: + """A server token is bound to the fixed SQL and validated parameter set.""" + + with patch.object(self.module.time, "time", return_value=1_000): + query_token = self.module._query_client_token("SELECT 1", [], 0) + with self.assertRaisesRegex(ValueError, "invalid or expired"): + self.module._query_token_attempt("SELECT 2", [], query_token) + + def test_query_client_token_is_scoped_by_query_attempt_and_time_window(self) -> None: + """Idempotency tokens reuse only the same query attempt inside one bounded window.""" + + parameters = [{"name": "from_date", "value": "2026-08-01"}] + with patch.object(self.module.time, "time", return_value=1_000): + original = self.module._query_client_token("SELECT 1", parameters, 0) + same_query = self.module._query_client_token("SELECT 1", parameters, 0) + new_attempt = self.module._query_client_token("SELECT 1", parameters, 1) + with patch.object( + self.module.time, + "time", + return_value=1_000 + self.module.QUERY_TOKEN_WINDOW_SECONDS, + ): + next_window = self.module._query_client_token("SELECT 1", parameters, 0) + + self.assertRegex(original, r"^[0-9a-f]{64}$") + self.assertEqual(original, same_query) + self.assertNotEqual(original, new_attempt) + self.assertNotEqual(original, next_window) + + def test_shapes_campaign_funnel_and_click_location(self) -> None: + """Campaign rows become totals, conversion rates, series, and safe click dimensions.""" + + rows = [ + { + "row_type": "summary", + "dimension_1": "2026-08-29", + "metric_1": 100, + "metric_2": 40, + "metric_3": 10, + "metric_4": 5, + }, + { + "row_type": "daily", + "date_value": "2026-08-29", + "metric_1": 100, + "metric_2": 40, + "metric_3": 10, + "metric_4": 5, + }, + { + "row_type": "click_location", + "dimension_1": "/challenges", + "dimension_2": "main", + "dimension_3": "register", + "dimension_4": "button", + "dimension_5": "www.topcoder-dev.com", + "dimension_6": "/challenges/123", + "metric_1": 8, + "metric_2": 6, + }, + ] + with patch.object(self.module, "_execute_query", return_value=rows): + response = self.module.handler( + self._event( + "GET /v1/analytics/campaign", + ["analytics"], + {"from": "2026-08-01", "to": "2026-08-30"}, + ), + None, + ) + + body = json.loads(response["body"]) + self.assertEqual(200, response["statusCode"]) + self.assertEqual(40.0, body["totals"]["clickThroughPercent"]) + self.assertEqual(50.0, body["totals"]["registrationToSubmissionPercent"]) + self.assertNotIn("xBucket", body["clickLocations"][0]) + self.assertNotIn("yBucket", body["clickLocations"][0]) + + def test_campaign_query_groups_clicks_by_semantic_item(self) -> None: + """Campaign click rankings do not split one item by viewport position.""" + + self.assertNotIn("click_x_bucket", self.module.CAMPAIGN_SQL) + self.assertNotIn("click_y_bucket", self.module.CAMPAIGN_SQL) + + def test_general_query_uses_one_timestamp_pruned_event_scan(self) -> None: + """General report dimensions share one grouping-sets scan of the event view.""" + + self.assertEqual( + 1, + self.module.GENERAL_SQL.count("topcoder_web.product_analytics_events_v1"), + ) + self.assertIn("GROUP BY GROUPING SETS", self.module.GENERAL_SQL) + self.assertIn("event_timestamp >= CAST(:from_date AS timestamp)", self.module.GENERAL_SQL) + self.assertIn( + "event_timestamp < DATEADD(day, 1, CAST(:to_date AS timestamp))", + self.module.GENERAL_SQL, + ) + + def test_route_funnel_matches_the_exact_clicked_challenge(self) -> None: + """Route registrations and submissions retain the clicked challenge ID.""" + + self.assertIn("topcoder_web.route_analytics_events_mv_v1", self.module.ROUTE_SQL) + self.assertNotIn(":surface", self.module.ROUTE_SQL) + self.assertIn("AND surface = :surface", self.module.ROUTE_SURFACE_SQL) + self.assertIn("event.challenge_id = click.challenge_id", self.module.ROUTE_SQL) + self.assertIn("event.challenge_id = registration.challenge_id", self.module.ROUTE_SQL) + + def test_route_query_omits_unused_surface_wildcard(self) -> None: + """Unfiltered routes avoid a bound wildcard branch in each warehouse scan.""" + + with patch.object(self.module, "_execute_query", return_value=[]) as execute: + response = self.module.handler( + self._event( + "GET /v1/analytics/route", + ["analytics"], + {"path": "/opportunities"}, + ), + None, + ) + + self.assertEqual(200, response["statusCode"]) + query_sql, parameters = execute.call_args.args[:2] + self.assertIs(self.module.ROUTE_SQL, query_sql) + self.assertNotIn("surface", {parameter["name"] for parameter in parameters}) + + self.module._cache.clear() + with patch.object(self.module, "_execute_query", return_value=[]) as execute: + response = self.module.handler( + self._event( + "GET /v1/analytics/route", + ["analytics"], + {"path": "/opportunities", "surface": "platform_ui"}, + ), + None, + ) + + self.assertEqual(200, response["statusCode"]) + query_sql, parameters = execute.call_args.args[:2] + self.assertEqual(self.module.ROUTE_SURFACE_SQL, query_sql) + parameter_values = { + parameter["name"]: parameter["value"] for parameter in parameters + } + self.assertEqual("platform_ui", parameter_values["surface"]) + + def test_shapes_general_report(self) -> None: + """General rows retain page, source, and surface dimensions.""" + + rows = [ + { + "row_type": "summary", + "dimension_1": "2026-08-30", + "metric_1": 30, + "metric_2": 20, + "metric_3": 10, + "metric_4": 8, + }, + { + "row_type": "page", + "dimension_1": "topcoder_website", + "dimension_2": "/challenges", + "metric_1": 12, + "metric_2": 9, + }, + ] + with patch.object(self.module, "_execute_query", return_value=rows): + response = self.module.handler( + self._event( + "GET /v1/analytics/general", + ["analytics"], + {"from": "2026-08-01", "to": "2026-08-30"}, + ), + None, + ) + + body = json.loads(response["body"]) + self.assertEqual(30, body["totals"]["pageViews"]) + self.assertEqual("/challenges", body["pages"][0]["path"]) + + def test_shapes_route_report_without_exposing_visitor_identifiers(self) -> None: + """Route rows become source, behavior, form, and ordered funnel aggregates.""" + + rows = [ + { + "row_type": "summary", + "dimension_6": "2026-09-10", + "metric_1": 150, + "metric_2": 100, + "metric_3": 60, + "metric_4": 40, + "metric_5": 35, + "metric_6": 60, + "metric_7": 12.345, + "metric_8": 80, + "metric_9": 20, + "metric_10": 15, + "metric_11": 12, + "metric_12": 8, + "metric_13": 4, + }, + {"row_type": "source", "dimension_1": "organic", "metric_1": 30}, + {"row_type": "source", "dimension_1": "paid", "metric_1": 20}, + { + "row_type": "click_location", + "dimension_1": "hero", + "dimension_2": "view-challenge", + "dimension_3": "a", + "dimension_4": "platform-ui.topcoder-dev.com", + "dimension_5": "/opportunities/challenge/123", + "metric_1": 25, + "metric_2": 20, + }, + { + "row_type": "form", + "dimension_1": "contact-us", + "metric_1": 50, + "metric_2": 45, + "metric_3": 12, + "metric_4": 11, + "metric_5": 8, + "metric_6": 8, + "metric_7": 4, + "metric_8": 4, + }, + { + "row_type": "form_abandonment", + "dimension_1": "contact-us", + "dimension_2": "company", + "metric_1": 3, + "metric_2": 3, + }, + { + "row_type": "funnel", + "metric_1": 100, + "metric_2": 30, + "metric_3": 12, + "metric_4": 6, + }, + ] + with patch.object(self.module, "_execute_query", return_value=rows): + response = self.module.handler( + self._event( + "GET /v1/analytics/route", + ["analytics"], + { + "from": "2026-09-01", + "to": "2026-09-10", + "path": "/landing", + }, + ), + None, + ) + + body = json.loads(response["body"]) + self.assertEqual(200, response["statusCode"]) + self.assertEqual(40.0, body["totals"]["clickThroughPercent"]) + self.assertEqual(25.0, body["totals"]["bounceRatePercent"]) + self.assertEqual(15.0, body["totals"]["conversionRatePercent"]) + self.assertEqual(5, body["totals"]["unknownVisitorType"]) + self.assertEqual(20.0, body["clickLocations"][0]["clickThroughPercent"]) + self.assertEqual(66.67, body["forms"][0]["completionRatePercent"]) + self.assertEqual(30.0, body["funnel"]["clickThroughPercent"]) + self.assertIsNone(body["funnel"]["wins"]) + self.assertFalse(body["funnel"]["winTrackingAvailable"]) + self.assertNotIn("analyticsUserId", response["body"]) + + +if __name__ == "__main__": + unittest.main() diff --git a/infrastructure/gigs-routing/README.md b/infrastructure/gigs-routing/README.md new file mode 100644 index 000000000..8d9efe7a5 --- /dev/null +++ b/infrastructure/gigs-routing/README.md @@ -0,0 +1,101 @@ +# Gigs routing + +The public routes `/gigs` and `/gigs/*` belong to platform-ui. Both apex and `www` +website aliases use the same website CloudFront distribution. This additive +CloudFormation change preserves the site's apex-to-www redirect, including encoded +and repeated query parameters. On the canonical host, it serves the platform-ui +S3 `gigs/index.html` with a signed origin request. The browser retains its path and query, so React handles listing, +details, and applications. Other routes, particularly `/api/recruit/*` and the +Payload compatibility API, retain their existing origins. The website's existing +`/static/*`, `/global.css`, and manifest handoffs serve the platform assets. + +`route-template.py` transforms the **current** website stack template instead of +checking a stale copy of that stack into this repository. It adds two exact path +behaviors, one S3 origin, an origin access control, and a small request function. +`prepare.py` discovers the platform bucket and validates the environment, then +writes local before/after templates, policies and a manifest. `apply.py` checks +for stale state and limits execution to the reviewed routing resources. Python 3, +PyYAML, and AWS CLI are required. Use a session for the intended account. Never +commit credentials or generated account snapshots. + +The deployment pipeline copies its built shell to `gigs/index.html` only after +the normal asset deployment and platform invalidation finish. This keeps the +uncached Gigs route on a complete release while the existing deployment script +uploads the next release's assets. The website needs no shell invalidation for +subsequent UI releases. Keep the dedicated shell when pruning old build files. + +## Release sequence + +1. Run platform-ui lint, build, Gigs tests and browser checks. Publish the `gigs` + branch, open its PR to `dev`, and deploy the resulting platform-ui build using + the existing deployment pipeline. The routing must not move until that build + actually includes the Gigs routes. Production needs its production build and + credentials; do not reuse a development build or session. +2. Prepare the routing artifacts with the account's website stack and platform + distribution. Development uses `topcoder-website-dev` and `EFA7R1KH3UX5M`: + + ```sh + python3 infrastructure/gigs-routing/prepare.py \ + --stack topcoder-website-dev --platform-distribution EFA7R1KH3UX5M \ + --output /tmp/gigs-routing-plan + python3 -c 'import json; p="/tmp/gigs-routing-plan/"; open(p+"template-deploy.json","w").write(json.dumps(json.load(open(p+"template-after.json")),separators=(",",":")))' + aws cloudformation create-change-set --stack-name topcoder-website-dev \ + --change-set-name gigs-routing --change-set-type UPDATE \ + --template-body file:///tmp/gigs-routing-plan/template-deploy.json \ + --parameters file:///tmp/gigs-routing-plan/parameters.json \ + --capabilities CAPABILITY_NAMED_IAM + ``` + +3. Inspect `describe-change-set`. Expected changes are the two new Gigs resources, + the website distribution (no replacement), and its derived distribution-domain + SSM parameter. Existing API functions, bucket policies, roles, and the original + viewer functions must not change. The separate platform-bucket grant allows + the website distribution to read **only `gigs/index.html`**, preserving all existing + grants and the deny-insecure-transport statement. +4. Execute the reviewed plan after the platform release is verified: + + ```sh + python3 infrastructure/gigs-routing/apply.py \ + --artifacts /tmp/gigs-routing-plan --changeset gigs-routing + aws cloudformation describe-stacks --stack-name topcoder-website-dev \ + --query 'Stacks[0].StackStatus' + aws cloudfront create-invalidation --distribution-id E3P6DRXC192WA \ + --paths '/gigs' '/gigs/*' + ``` + +5. Wait for `UPDATE_COMPLETE`, CloudFront `Deployed`, and invalidation completion. + Verify HTTPS apex and `www` `/gigs`, `/gigs/`, a real job, its `/apply` URL, and + a search query. Check that the shell and its referenced JS/CSS all return 200 + with the expected content types (an HTML fallback is not a valid JS response), + browser routing works, a fulfilled gig is not applicable, and `/api/recruit/jobs` + still returns JSON. Check an unrelated website page and `/opportunities`. + +The website stack is the infrastructure owner. Apply this same transform to its +source template when updating that stack from topcoder-website; a future stack +update from an older template would otherwise remove the Gigs routing resources. +Routine website content releases do not replace the stack template. + +## Rollback + +Create and inspect a CloudFormation change set using `template-before.json` and +`parameters.json`, then execute it and wait for completion. Restore the platform +bucket's `bucket-policy-before.json` only after comparing the current policy with +the prepared after-policy; if other grants changed, remove only the matching +`AllowWebsiteGigsPublishedShell` statement. Invalidate `/gigs` and `/gigs/*` again. This +restores the community-app handoff without removing platform assets or changing +Recruit data. Preserve prior platform deployment artifacts for application-level +rollback using the normal deployment process. + +## Checks + +```sh +nvm use +python3 -m unittest discover -s infrastructure/gigs-routing -p 'test_*.py' +``` + +The transform tests prove route isolation, preservation of existing behaviors, +idempotency, rejection of conflicting route ownership, and canonical redirects +without losing or double-encoding query parameters. The request test uses Node +from `.nvmrc` to execute the generated edge handler. The change-set review +and live browser checks remain necessary because unit tests cannot prove an AWS +account's actual deployment state. diff --git a/infrastructure/gigs-routing/apply.py b/infrastructure/gigs-routing/apply.py new file mode 100644 index 000000000..2c242814b --- /dev/null +++ b/infrastructure/gigs-routing/apply.py @@ -0,0 +1,72 @@ +#!/usr/bin/env python3 +"""Execute a reviewed Gigs change set after verifying its scope and current state. + +Uses the configured AWS CLI session. Requires the directory created by prepare.py +and the name of an already-created CloudFormation change set. Does not deploy UI +code or wait for propagation; follow the runbook for release order and verification. +""" +import argparse +import json +from pathlib import Path +import subprocess + +from prepare import aws + + +ALLOWED_CHANGES = { + 'GigsOriginAccessControl': 'Add', + 'GigsViewerRequestFunction': 'Add', + 'WebsiteDistribution': 'Modify', + # Derived from the distribution; CloudFormation recalculates this parameter. + 'DistributionDomainDeployParameter': 'Modify', +} + + +def apply(directory, changeset): + """Grant private shell access and execute a reviewed, unchanged Gigs plan. + + directory contains prepare.py artifacts; changeset identifies the pending + change. Raises on credential mismatch, stale state, unexpected resource edits, + replacements, or failed CLI commands. Rolls back the policy if execution fails. + """ + manifest = json.loads((directory / 'manifest.json').read_text()) + before = json.loads((directory / 'bucket-policy-before.json').read_text()) + after = json.loads((directory / 'bucket-policy-after.json').read_text()) + if aws('sts', 'get-caller-identity')['Account'] != manifest['account']: + raise ValueError('AWS session does not match the prepared account') + current_stack = aws('cloudformation', 'describe-stacks', '--stack-name', manifest['stack'])['Stacks'][0] + updated = str(current_stack.get('LastUpdatedTime', current_stack['CreationTime'])) + if updated != manifest['stackLastUpdated']: + raise ValueError('The website stack changed; prepare and review a new plan') + change = aws('cloudformation', 'describe-change-set', '--stack-name', manifest['stack'], '--change-set-name', changeset) + if change['Status'] != 'CREATE_COMPLETE' or change['ExecutionStatus'] != 'AVAILABLE': + raise ValueError('Change set is not ready to execute') + for item in change.get('Changes', []): + resource = item['ResourceChange'] + if ALLOWED_CHANGES.get(resource['LogicalResourceId']) != resource['Action']: + raise ValueError('Unexpected change: ' + resource['LogicalResourceId']) + if resource.get('Replacement') not in (None, 'False'): + raise ValueError('Resource replacement is not allowed') + current_policy = json.loads(aws('s3api', 'get-bucket-policy', '--bucket', manifest['platformBucket'])['Policy']) + if current_policy != before: + raise ValueError('Bucket policy changed; prepare and review a new plan') + subprocess.run(['aws', 's3api', 'put-bucket-policy', '--bucket', manifest['platformBucket'], + '--policy', 'file://' + str((directory / 'bucket-policy-after.json').resolve())], check=True) + try: + subprocess.run(['aws', 'cloudformation', 'execute-change-set', '--stack-name', manifest['stack'], + '--change-set-name', changeset], check=True) + except subprocess.CalledProcessError: + latest_policy = json.loads(aws('s3api', 'get-bucket-policy', '--bucket', manifest['platformBucket'])['Policy']) + if latest_policy == after: + subprocess.run(['aws', 's3api', 'put-bucket-policy', '--bucket', manifest['platformBucket'], + '--policy', 'file://' + str((directory / 'bucket-policy-before.json').resolve())], check=True) + raise + print(json.dumps({'status': 'UPDATE_STARTED', 'stack': manifest['stack'], 'distribution': manifest['websiteDistribution']})) + + +if __name__ == '__main__': + parser = argparse.ArgumentParser(description=__doc__) + parser.add_argument('--artifacts', required=True, type=Path) + parser.add_argument('--changeset', required=True) + args = parser.parse_args() + apply(args.artifacts, args.changeset) diff --git a/infrastructure/gigs-routing/prepare.py b/infrastructure/gigs-routing/prepare.py new file mode 100644 index 000000000..6b653e65f --- /dev/null +++ b/infrastructure/gigs-routing/prepare.py @@ -0,0 +1,107 @@ +#!/usr/bin/env python3 +"""Prepare a reviewable Gigs CloudFormation change and a narrow bucket-policy grant. + +Uses the configured AWS CLI session without reading or writing credentials. This +command is read-only in AWS: output consists of local artifacts and a manifest. +Requires PyYAML for parsing CloudFormation's intrinsic YAML tags. +""" +import argparse +import importlib.util +import json +from pathlib import Path +import subprocess + +import yaml + + +class CloudFormationLoader(yaml.SafeLoader): + """Parse CloudFormation YAML intrinsics as their JSON equivalents.""" + + +def intrinsic(loader, tag, node): + """Convert one scalar, list, or mapping intrinsic to CloudFormation JSON.""" + if isinstance(node, yaml.ScalarNode): + value = loader.construct_scalar(node) + elif isinstance(node, yaml.SequenceNode): + value = loader.construct_sequence(node) + else: + value = loader.construct_mapping(node) + return {tag if tag in ('Ref', 'Condition') else 'Fn::' + tag: value} + + +CloudFormationLoader.add_multi_constructor('!', intrinsic) + + +def aws(*args): + """Run a read-only AWS CLI request; return JSON or raise on command failure.""" + return json.loads(subprocess.check_output(['aws', *args, '--output', 'json'], text=True)) + + +def prepare(stack_name, platform_distribution, directory): + """Write current/proposed templates and policies for an existing website stack. + + Returns an artifact manifest. Raises on nonterminal stack state, unexpected + origins, aliases, or policy conflicts. Preserves all existing policy grants. + """ + directory.mkdir(parents=True, exist_ok=False) + stack = aws('cloudformation', 'describe-stacks', '--stack-name', stack_name)['Stacks'][0] + if stack['StackStatus'] not in ('CREATE_COMPLETE', 'UPDATE_COMPLETE', 'UPDATE_ROLLBACK_COMPLETE'): + raise ValueError('Website stack is not stable: ' + stack['StackStatus']) + template_body = aws('cloudformation', 'get-template', '--stack-name', stack_name)['TemplateBody'] + template = yaml.load(template_body, Loader=CloudFormationLoader) if isinstance(template_body, str) else template_body + website = aws('cloudformation', 'describe-stack-resource', '--stack-name', stack_name, + '--logical-resource-id', 'WebsiteDistribution')['StackResourceDetail']['PhysicalResourceId'] + platform = aws('cloudfront', 'get-distribution-config', '--id', platform_distribution)['DistributionConfig'] + origin_id = platform['DefaultCacheBehavior']['TargetOriginId'] + origin = next(item for item in platform['Origins']['Items'] if item['Id'] == origin_id) + if 'S3OriginConfig' not in origin or origin.get('OriginPath'): + raise ValueError('Expected an unprefixed platform-ui S3 origin') + bucket = origin['DomainName'].split('.s3.')[0] + identity = aws('sts', 'get-caller-identity') + params = {item['ParameterKey']: item['ParameterValue'] for item in stack['Parameters']} + if 'platform-ui.' + params['ApexDomainName'] not in platform['Aliases']['Items']: + raise ValueError('Platform distribution and website environment do not match') + spec = importlib.util.spec_from_file_location('route_template', Path(__file__).with_name('route-template.py')) + module = importlib.util.module_from_spec(spec) + spec.loader.exec_module(module) + proposed = module.add_gigs_routes(template, bucket) + policy = json.loads(aws('s3api', 'get-bucket-policy', '--bucket', bucket)['Policy']) + grant = { + 'Sid': 'AllowWebsiteGigsPublishedShell', + 'Effect': 'Allow', + 'Principal': {'Service': 'cloudfront.amazonaws.com'}, + 'Action': 's3:GetObject', + 'Resource': f'arn:aws:s3:::{bucket}/gigs/index.html', + 'Condition': {'StringEquals': {'AWS:SourceArn': f'arn:aws:cloudfront::{identity["Account"]}:distribution/{website}'}}, + } + next_policy = json.loads(json.dumps(policy)) + existing = next((item for item in policy['Statement'] if item.get('Sid') == grant['Sid']), None) + if existing and existing != grant: + raise ValueError('Conflicting Gigs shell policy grant') + if not existing: + next_policy['Statement'].append(grant) + artifacts = { + 'template-before.json': template, + 'template-after.json': proposed, + 'bucket-policy-before.json': policy, + 'bucket-policy-after.json': next_policy, + 'parameters.json': [{'ParameterKey': key, 'UsePreviousValue': True} for key in params], + } + for filename, value in artifacts.items(): + (directory / filename).write_text(json.dumps(value, indent=2) + '\n') + manifest = { + 'account': identity['Account'], 'stack': stack_name, 'websiteDistribution': website, + 'platformDistribution': platform_distribution, 'platformBucket': bucket, + 'apex': params['ApexDomainName'], 'stackLastUpdated': str(stack.get('LastUpdatedTime', stack['CreationTime'])), + } + (directory / 'manifest.json').write_text(json.dumps(manifest, indent=2) + '\n') + return manifest + + +if __name__ == '__main__': + parser = argparse.ArgumentParser(description=__doc__) + parser.add_argument('--stack', required=True) + parser.add_argument('--platform-distribution', required=True) + parser.add_argument('--output', required=True, type=Path) + args = parser.parse_args() + print(json.dumps(prepare(args.stack, args.platform_distribution, args.output), indent=2)) diff --git a/infrastructure/gigs-routing/route-template.py b/infrastructure/gigs-routing/route-template.py new file mode 100644 index 000000000..d98a81cbe --- /dev/null +++ b/infrastructure/gigs-routing/route-template.py @@ -0,0 +1,126 @@ +#!/usr/bin/env python3 +"""Extend the existing website CloudFormation template with the Gigs SPA routes. + +This is a pure transform: read the current template as JSON on stdin and write +JSON on stdout. Supply the platform-ui bucket name as the sole argument. Deploy +with the accompanying AWS CLI runbook; no credentials are read by this program. +""" +import copy +import json +import sys + + +def add_gigs_routes(template, bucket): + """Return a copy with exact/deep Gigs behaviors and a private S3 origin. + + template is a CloudFormation JSON object, bucket is the existing platform-ui + bucket. Raises ValueError for conflicting resources or unexpected website + topology. All existing parameters, resources, policies, and behaviors remain. + """ + output = copy.deepcopy(template) + resources = output['Resources'] + config = resources['WebsiteDistribution']['Properties']['DistributionConfig'] + new_resources = { + 'GigsOriginAccessControl': { + 'Type': 'AWS::CloudFront::OriginAccessControl', + 'Properties': { + 'OriginAccessControlConfig': { + 'Name': {'Fn::Sub': '${AppName}-${Environment}-gigs'}, + 'OriginAccessControlOriginType': 's3', + 'SigningBehavior': 'always', + 'SigningProtocol': 'sigv4', + }, + }, + }, + 'GigsViewerRequestFunction': { + 'Type': 'AWS::CloudFront::Function', + 'Properties': { + 'Name': {'Fn::Sub': '${AppName}-${Environment}-gigs-request'}, + 'AutoPublish': True, + 'FunctionConfig': { + 'Comment': 'Serve the platform-ui shell for exact and nested Gigs pages', + 'Runtime': 'cloudfront-js-2.0', + }, + 'FunctionCode': {'Fn::Sub': ( + 'function handler(event) {\n' + ' var request = event.request;\n' + ' var host = request.headers.host ? request.headers.host.value.toLowerCase() : "";\n' + ' if (host === "${ApexDomainName}") {\n' + ' var query = [];\n' + ' Object.keys(request.querystring).forEach(function (name) {\n' + ' var item = request.querystring[name];\n' + ' var values = item.multiValue && item.multiValue.length ? item.multiValue : [item];\n' + ' values.forEach(function (entry) {\n' + # CloudFront passes percent encoding through; do not encode it a second time. + ' query.push(name + "=" + (entry.value || ""));\n' + ' });\n' + ' });\n' + ' return {\n' + ' statusCode: 301,\n' + ' statusDescription: "Moved Permanently",\n' + ' headers: {\n' + ' location: { value: "https://${CanonicalDomainName}" + request.uri\n' + ' + (query.length ? "?" + query.join("&") : "") },\n' + ' "cache-control": { value: "public, max-age=300" }\n' + ' }\n' + ' };\n' + ' }\n' + ' request.uri = "/index.html";\n' + ' return request;\n' + '}\n' + )}, + }, + }, + } + for key, value in new_resources.items(): + if key in resources and resources[key] != value: + raise ValueError('Conflicting resource: ' + key) + resources[key] = value + origin = { + 'Id': 'GigsPlatformOrigin', + 'DomainName': {'Fn::Sub': bucket + '.s3.${AWS::Region}.${AWS::URLSuffix}'}, + 'OriginPath': '/gigs', + 'OriginAccessControlId': {'Fn::GetAtt': ['GigsOriginAccessControl', 'Id']}, + 'S3OriginConfig': {'OriginAccessIdentity': ''}, + } + origins = config['Origins'] + existing_origin = next((item for item in origins if item['Id'] == origin['Id']), None) + if existing_origin and existing_origin != origin: + # Upgrade the initial root-shell route without altering any other origin settings. + initial_origin = {key: value for key, value in origin.items() if key != 'OriginPath'} + if existing_origin != initial_origin: + raise ValueError('Conflicting Gigs platform origin') + existing_origin['OriginPath'] = origin['OriginPath'] + if not existing_origin: + origins.append(origin) + behavior = { + 'TargetOriginId': origin['Id'], + 'ViewerProtocolPolicy': 'redirect-to-https', + 'AllowedMethods': ['GET', 'HEAD'], + 'CachedMethods': ['GET', 'HEAD'], + 'Compress': True, + # The shell must follow the latest platform-ui release immediately. + 'CachePolicyId': '4135ea2d-6df8-44a3-9df3-4b5a84be39ad', + 'FunctionAssociations': [{ + 'EventType': 'viewer-request', + 'FunctionARN': {'Fn::GetAtt': ['GigsViewerRequestFunction', 'FunctionARN']}, + }], + } + if config['DefaultCacheBehavior'].get('ResponseHeadersPolicyId'): + behavior['ResponseHeadersPolicyId'] = config['DefaultCacheBehavior']['ResponseHeadersPolicyId'] + behaviors = config.get('CacheBehaviors', []) + for path in ['/gigs', '/gigs/*']: + desired = dict(behavior, PathPattern=path) + existing = next((item for item in behaviors if item['PathPattern'] == path), None) + if existing and existing != desired: + raise ValueError('Conflicting route behavior: ' + path) + if not existing: + behaviors.append(desired) + config['CacheBehaviors'] = behaviors + return output + + +if __name__ == '__main__': + if len(sys.argv) != 2 or not sys.argv[1].replace('.', '').replace('-', '').isalnum(): + sys.exit('Usage: route-template.py PLATFORM_UI_BUCKET < current-template.json') + json.dump(add_gigs_routes(json.load(sys.stdin), sys.argv[1]), sys.stdout, indent=2) diff --git a/infrastructure/gigs-routing/test_routes.py b/infrastructure/gigs-routing/test_routes.py new file mode 100644 index 000000000..ceb2480bc --- /dev/null +++ b/infrastructure/gigs-routing/test_routes.py @@ -0,0 +1,73 @@ +"""Regression checks for isolated Gigs routing and private shell access.""" +import importlib.util +import json +from pathlib import Path +import subprocess +import unittest + +spec = importlib.util.spec_from_file_location('route_template', Path(__file__).with_name('route-template.py')) +module = importlib.util.module_from_spec(spec) +spec.loader.exec_module(module) + + +class GigsRoutesTest(unittest.TestCase): + """Verify the template transform never takes ownership of APIs or unrelated routes.""" + + def test_preserves_existing_resources_and_is_idempotent(self): + """Only two Gigs behaviors, one origin and two resources may be added.""" + template = {'Resources': {'WebsiteDistribution': {'Properties': {'DistributionConfig': { + 'Origins': [{'Id': 'LegacyAlbOrigin'}], + 'DefaultCacheBehavior': {'TargetOriginId': 'StaticSiteOrigin'}, + 'CacheBehaviors': [{'PathPattern': '/__api/*', 'TargetOriginId': 'WebsiteApiOrigin'}], + }}}}} + result = module.add_gigs_routes(template, 'platform-ui.example.com') + config = result['Resources']['WebsiteDistribution']['Properties']['DistributionConfig'] + self.assertEqual(['/__api/*', '/gigs', '/gigs/*'], [item['PathPattern'] for item in config['CacheBehaviors']]) + self.assertEqual({'TargetOriginId': 'StaticSiteOrigin'}, config['DefaultCacheBehavior']) + self.assertEqual(1, len(template['Resources'])) + self.assertEqual(result, module.add_gigs_routes(result, 'platform-ui.example.com')) + self.assertEqual('', config['Origins'][-1]['S3OriginConfig']['OriginAccessIdentity']) + self.assertIn('OriginAccessControlId', config['Origins'][-1]) + self.assertEqual('/gigs', config['Origins'][-1]['OriginPath']) + initial = json.loads(json.dumps(result)) + del initial['Resources']['WebsiteDistribution']['Properties']['DistributionConfig']['Origins'][-1]['OriginPath'] + self.assertEqual(result, module.add_gigs_routes(initial, 'platform-ui.example.com')) + + def test_refuses_conflicting_route_ownership(self): + """An existing incompatible /gigs behavior must be reviewed instead of overwritten.""" + template = {'Resources': {'WebsiteDistribution': {'Properties': {'DistributionConfig': { + 'Origins': [], 'DefaultCacheBehavior': {}, + 'CacheBehaviors': [{'PathPattern': '/gigs', 'TargetOriginId': 'OtherOrigin'}], + }}}}} + with self.assertRaises(ValueError): + module.add_gigs_routes(template, 'platform-ui.example.com') + + def test_request_preserves_canonical_host_and_encoded_query(self): + """Execute the edge handler with apex/deep links and repeated encoded query values.""" + template = {'Resources': {'WebsiteDistribution': {'Properties': {'DistributionConfig': { + 'Origins': [], 'DefaultCacheBehavior': {}, + }}}}} + result = module.add_gigs_routes(template, 'platform-ui.example.com') + code = result['Resources']['GigsViewerRequestFunction']['Properties']['FunctionCode']['Fn::Sub'] + code = code.replace('${ApexDomainName}', 'example.com').replace('${CanonicalDomainName}', 'www.example.com') + request = {'uri': '/gigs/software-engineer/apply', 'headers': {'host': {'value': 'example.com'}}, + 'querystring': {'search': {'value': 'C%2B%2B+Developer'}, + 'ref': {'multiValue': [{'value': 'a%26b'}, {'value': 'campaign'}]}, + 'empty': {'value': ''}}} + redirect = json.loads(subprocess.check_output( + ['node', '-e', code + '\nconsole.log(JSON.stringify(handler(' + json.dumps({'request': request}) + ')));'], + text=True)) + self.assertEqual(301, redirect['statusCode']) + self.assertEqual('https://www.example.com/gigs/software-engineer/apply' + '?search=C%2B%2B+Developer&ref=a%26b&ref=campaign&empty=', + redirect['headers']['location']['value']) + request['headers']['host']['value'] = 'www.example.com' + rewritten = json.loads(subprocess.check_output( + ['node', '-e', code + '\nconsole.log(JSON.stringify(handler(' + json.dumps({'request': request}) + ')));'], + text=True)) + self.assertEqual('/index.html', rewritten['uri']) + self.assertEqual(request['querystring'], rewritten['querystring']) + + +if __name__ == '__main__': + unittest.main() diff --git a/package.json b/package.json index 667c8db70..837b54149 100644 --- a/package.json +++ b/package.json @@ -20,6 +20,11 @@ "deploy:dev": "BRANCH=$(git rev-parse --abbrev-ref HEAD) && TAG=\"dev-${BRANCH}\" && git tag -d \"$TAG\" 2>/dev/null; git push origin \":refs/tags/$TAG\" 2>/dev/null; git tag \"$TAG\" && git push origin \"$TAG\"" }, "dependencies": { + "@ai-sdk/react": "4.0.82", + "@assistant-ui/react": "0.15.16", + "@assistant-ui/react-ai-sdk": "1.4.7", + "@assistant-ui/react-markdown": "0.14.12", + "@aws/clickstream-web": "^0.12.6", "@codemirror/autocomplete": "^6.20.1", "@codemirror/lang-java": "^6.0.2", "@codemirror/language": "^6.12.2", @@ -34,7 +39,6 @@ "@highcharts/map-collection": "^2.3.3", "@hookform/resolvers": "^4.1.3", "@popperjs/core": "^2.11.8", - "@sprig-technologies/sprig-browser": "^2.39.0", "@storybook/addon-actions": "7.6.20", "@storybook/react": "7.6.20", "@stripe/react-stripe-js": "1.16.5", @@ -42,13 +46,13 @@ "@tinymce/tinymce-react": "^6.3.0", "@types/codemirror": "5.60.17", "@uiw/react-codemirror": "^4.25.8", + "ai": "7.0.79", "amazon-s3-uri": "^0.1.1", "apexcharts": "^3.54.1", "axios": "^1.19.0", "browser-cookies": "^1.2.0", "city-timezones": "^1.3.2", "classnames": "^2.5.1", - "contentful": "^9.3.7", "country-calling-code": "0.0.3", "crypto-js": "^4.2.0", "customize-cra": "^1.0.0", @@ -64,10 +68,14 @@ "express-interceptor": "^1.2.0", "fflate": "^0.8.2", "filestack-js": "^3.44.2", + "flag-icons": "^6.7.0", + "grapesjs": "0.22.14", + "grapesjs-preset-newsletter": "1.0.2", "highcharts": "^10.3.3", "highcharts-react-official": "^3.2.3", "highlight.js": "^11.11.1", "html2canvas": "^1.4.1", + "i18n-iso-countries": "^3.7.1", "lodash": "^4.18.1", "markdown-it": "^14.3.0", "marked": "4.3.0", @@ -88,7 +96,6 @@ "react-dom": "^18.3.1", "react-dropzone": "^11.7.1", "react-elastic-carousel": "^0.11.5", - "react-gtm-module": "^2.0.11", "react-helmet": "^6.1.0", "react-hook-form": "^7.68.0", "react-markdown": "8.0.7", @@ -111,10 +118,12 @@ "redux-promise-middleware": "^6.2.0", "redux-thunk": "^2.4.2", "rehype-raw": "^7.0.0", + "rehype-sanitize": "^5.0.1", "rehype-stringify": "^10.0.1", "remark-breaks": "^3.0.3", "remark-frontmatter": "^4.0.1", "remark-gfm": "^3.0.1", + "remark-gfm-v4": "npm:remark-gfm@^4.0.1", "remark-parse": "^11.0.0", "remove": "^0.1.5", "sanitize-html": "^2.17.6", @@ -123,12 +132,10 @@ "swr": "^1.3.0", "tc-auth-lib": "topcoder-platform/tc-auth-lib#v2.0", "tinymce": "^7.9.1", - "typescript": "^4.9.5", + "typescript": "5.5.4", "universal-navigation": "https://github.com/topcoder-platform/universal-navigation#master", "uuid": "^11.1.0", - "yup": "^1.7.1", - "flag-icons": "^6.7.0", - "i18n-iso-countries": "^3.7.1" + "yup": "^1.7.1" }, "devDependencies": { "@babel/core": "^7.29.6", @@ -160,7 +167,6 @@ "@types/react": "18.3.27", "@types/react-datepicker": "^4.19.6", "@types/react-dom": "^18.3.7", - "@types/react-gtm-module": "^2.0.4", "@types/react-helmet": "^6.1.11", "@types/react-redux-toastr": "^7.6.6", "@types/redux-actions": "2.6.5", diff --git a/src/apps/accounts/README.md b/src/apps/accounts/README.md index a5bb372fc..23f2655fe 100644 --- a/src/apps/accounts/README.md +++ b/src/apps/accounts/README.md @@ -1 +1,17 @@ -# Accounts App \ No newline at end of file +# Accounts App + +The Preferences tab loads marketing email categories from +`EnvironmentConfig.CONTACT_API/me/subscriptions` using the member's normal bearer +token. The member chooses categories explicitly, then saves the complete selection +with `PUT`; an empty selection opts out of every active marketing category. The API +derives member identity from authentication, so the browser never supplies a target +member ID. Loading and save failures preserve the member's choices and offer retry. +Existing delivery suppression is shown and is not cleared by a subscription change. +The separate forum settings link and account/security email behavior remain available. + +`getMarketingPreferences()` returns the authenticated member's categories and +suppression flag. `saveMarketingPreferences(subscriptionTypeIds)` stores selected +active category IDs and returns refreshed preferences; both reject on API failure. +`MarketingPreferences` renders these choices, saves only on the member's explicit +action, and handles loading, failure and success feedback. Method inputs, outputs, +usage and failure behavior are documented in source. diff --git a/src/apps/accounts/src/accounts.routes.spec.tsx b/src/apps/accounts/src/accounts.routes.spec.tsx new file mode 100644 index 000000000..2e5b2bebf --- /dev/null +++ b/src/apps/accounts/src/accounts.routes.spec.tsx @@ -0,0 +1,32 @@ +/* eslint-disable import/no-extraneous-dependencies, ordered-imports/ordered-imports */ +import { accountsRoutes } from './accounts.routes' + +jest.mock('~/config', () => ({ + AppSubdomain: { accounts: 'account-settings' }, + EnvironmentConfig: { SUBDOMAIN: 'platform-ui' }, + ToolTitle: { accounts: 'Account Settings' }, +}), { virtual: true }) + +jest.mock('~/libs/core', () => ({ + lazyLoad: () => (): JSX.Element =>
, +}), { virtual: true }) + +describe('Account Settings routes', () => { + it('protects settings while allowing validation links to work logged out', () => { + const [root] = accountsRoutes + const settingsRoute = root.children?.find(route => route.route === '') + const validationRoute = root.children + ?.find(route => route.route === 'email-change/verify') + const legacyValidationRoute = root.children + ?.find(route => route.route === 'changeEmail') + + expect(root.authRequired) + .toBeUndefined() + expect(settingsRoute?.authRequired) + .toBe(true) + expect(validationRoute?.authRequired) + .toBeUndefined() + expect(legacyValidationRoute?.authRequired) + .toBeUndefined() + }) +}) diff --git a/src/apps/accounts/src/accounts.routes.tsx b/src/apps/accounts/src/accounts.routes.tsx index ea60df349..4ad2060dd 100644 --- a/src/apps/accounts/src/accounts.routes.tsx +++ b/src/apps/accounts/src/accounts.routes.tsx @@ -3,6 +3,10 @@ import { AppSubdomain, EnvironmentConfig, ToolTitle } from '~/config' const AccountsApp: LazyLoadedComponent = lazyLoad(() => import('./AccountsApp')) const AccountSettingsPage: LazyLoadedComponent = lazyLoad(() => import('./settings'), 'AccountSettingsPage') +const ChangeEmailVerificationPage: LazyLoadedComponent = lazyLoad( + () => import('./settings/change-email-verification'), + 'ChangeEmailVerificationPage', +) export const rootRoute: string = ( EnvironmentConfig.SUBDOMAIN === AppSubdomain.accounts ? '' : `/${AppSubdomain.accounts}` @@ -13,14 +17,26 @@ export const absoluteRootRoute: string = `${window.location.origin}${rootRoute}` export const accountsRoutes: ReadonlyArray = [ { - authRequired: true, children: [ { + authRequired: true, children: [], element: , id: 'Account Settings', route: '', }, + { + children: [], + element: , + id: 'Change Email Verification', + route: 'email-change/verify', + }, + { + children: [], + element: , + id: 'Legacy Change Email Verification', + route: 'changeEmail', + }, ], domain: AppSubdomain.accounts, element: , diff --git a/src/apps/accounts/src/lib/index.ts b/src/apps/accounts/src/lib/index.ts index a435df51a..fe7f884cd 100644 --- a/src/apps/accounts/src/lib/index.ts +++ b/src/apps/accounts/src/lib/index.ts @@ -1,3 +1,4 @@ export * from './accounts-swr' export * from './components' +export * from './services' export * from './assets' diff --git a/src/apps/accounts/src/lib/services/email-change.service.spec.ts b/src/apps/accounts/src/lib/services/email-change.service.spec.ts new file mode 100644 index 000000000..9a9d736f6 --- /dev/null +++ b/src/apps/accounts/src/lib/services/email-change.service.spec.ts @@ -0,0 +1,58 @@ +/* eslint-disable import/no-extraneous-dependencies, ordered-imports/ordered-imports */ +import { xhrGetAsync, xhrPostAsync } from '~/libs/core' + +import { + completeEmailChangeAsync, + initiateEmailChangeAsync, + requestEmailChangeOtpAsync, + verifyEmailChangeOtpAsync, +} from './email-change.service' + +jest.mock('~/config', () => ({ + EnvironmentConfig: { API: { V6: 'https://api.example.test/v6' } }, +}), { virtual: true }) + +jest.mock('~/libs/core', () => ({ + xhrGetAsync: jest.fn(), + xhrPostAsync: jest.fn(), +}), { virtual: true }) + +const mockedGet = xhrGetAsync as jest.Mock +const mockedPost = xhrPostAsync as jest.Mock + +describe('email change API service', () => { + beforeEach(() => { + mockedGet.mockReset() + mockedPost.mockReset() + }) + + it('uses the ownership, proof, validation, and completion endpoints', async () => { + mockedPost + .mockResolvedValueOnce({ expiresIn: 600 }) + .mockResolvedValueOnce({ expiresIn: 600, verificationToken: 'proof' }) + .mockResolvedValueOnce({ email: 'new@example.com' }) + mockedGet.mockResolvedValueOnce({ email: 'new@example.com' }) + + await requestEmailChangeOtpAsync(123) + await verifyEmailChangeOtpAsync(123, '012345') + await initiateEmailChangeAsync(123, 'new@example.com', 'proof') + await completeEmailChangeAsync('signed/token') + + expect(mockedPost.mock.calls) + .toEqual([ + ['https://api.example.test/v6/users/123/email-change/otp', {}], + [ + 'https://api.example.test/v6/users/123/email-change/verify-otp', + { param: { otp: '012345' } }, + ], + [ + 'https://api.example.test/v6/users/123/email-change', + { param: { email: 'new@example.com', verificationToken: 'proof' } }, + ], + ]) + expect(mockedGet) + .toHaveBeenCalledWith( + 'https://api.example.test/v6/users/email-change/verify?code=signed%2Ftoken', + ) + }) +}) diff --git a/src/apps/accounts/src/lib/services/email-change.service.ts b/src/apps/accounts/src/lib/services/email-change.service.ts new file mode 100644 index 000000000..1ec6b26ac --- /dev/null +++ b/src/apps/accounts/src/lib/services/email-change.service.ts @@ -0,0 +1,117 @@ +import { AxiosError } from 'axios' + +import { EnvironmentConfig } from '~/config' +import { xhrGetAsync, xhrPostAsync } from '~/libs/core' + +export interface EmailChangeOtpResponse { + expiresIn: number +} + +export interface EmailChangeOtpVerificationResponse extends EmailChangeOtpResponse { + verificationToken: string +} + +export interface EmailChangeResponse { + email: string +} + +const usersUrl: string = `${EnvironmentConfig.API.V6}/users` + +/** + * Requests a six-digit ownership code at the member's current primary email. + * + * @param userId member ID whose email will be changed. + * @returns the code lifetime in seconds. + * @throws rejects when the identity API cannot send the code. + */ +export async function requestEmailChangeOtpAsync( + userId: number, +): Promise { + return xhrPostAsync, EmailChangeOtpResponse>( + `${usersUrl}/${userId}/email-change/otp`, + {}, + ) +} + +/** + * Verifies the ownership code sent to the member's current primary email. + * + * @param userId member ID that requested the code. + * @param otp six-digit ownership code. + * @returns a short-lived proof used to submit a new address. + * @throws rejects when the code is invalid, expired, or blocked. + */ +export async function verifyEmailChangeOtpAsync( + userId: number, + otp: string, +): Promise { + return xhrPostAsync< + { param: { otp: string } }, + EmailChangeOtpVerificationResponse + >( + `${usersUrl}/${userId}/email-change/verify-otp`, + { param: { otp } }, + ) +} + +/** + * Sends a validation link to the proposed new primary email. + * + * @param userId member ID whose email will be changed. + * @param email proposed new primary email. + * @param verificationToken proof that the current email OTP was verified. + * @returns the normalized address that received the validation link. + * @throws rejects when the address or proof is invalid. + */ +export async function initiateEmailChangeAsync( + userId: number, + email: string, + verificationToken: string, +): Promise { + return xhrPostAsync< + { param: { email: string, verificationToken: string } }, + EmailChangeResponse + >( + `${usersUrl}/${userId}/email-change`, + { param: { email, verificationToken } }, + ) +} + +/** + * Completes the deferred email update from the validation link. + * + * @param validationCode one-time code delivered in the proposed-email link. + * @returns the email address that is now primary. + * @throws rejects when the validation link is invalid, expired, or already used. + */ +export async function completeEmailChangeAsync( + validationCode: string, +): Promise { + return xhrGetAsync( + `${usersUrl}/email-change/verify?code=${encodeURIComponent(validationCode)}`, + ) +} + +/** + * Extracts a user-facing message from an email-change API error. + * + * @param error unknown error caught from an API request. + * @param fallback message used when the response has no useful detail. + * @returns a concise user-facing error message. + */ +export function getEmailChangeErrorMessage(error: unknown, fallback: string): string { + if (!(error instanceof AxiosError)) { + return error instanceof Error && error.message ? error.message : fallback + } + + const responseMessage: unknown = error.response?.data?.message + ?? error.response?.data?.error?.message + + if (Array.isArray(responseMessage)) { + return responseMessage.join(' ') + } + + return typeof responseMessage === 'string' && responseMessage.trim() + ? responseMessage + : (error.message || fallback) +} diff --git a/src/apps/accounts/src/lib/services/index.ts b/src/apps/accounts/src/lib/services/index.ts new file mode 100644 index 000000000..410e86c47 --- /dev/null +++ b/src/apps/accounts/src/lib/services/index.ts @@ -0,0 +1 @@ +export * from './email-change.service' diff --git a/src/apps/accounts/src/lib/services/marketing-preferences.service.ts b/src/apps/accounts/src/lib/services/marketing-preferences.service.ts new file mode 100644 index 000000000..90bc89c65 --- /dev/null +++ b/src/apps/accounts/src/lib/services/marketing-preferences.service.ts @@ -0,0 +1,41 @@ +import { EnvironmentConfig } from '~/config' +import { xhrGetAsync, xhrPutAsync } from '~/libs/core' + +export interface MarketingSubscription { + active: boolean + description: string + name: string + source: string + subscribed: boolean + subscriptionTypeId: string +} + +export interface MarketingPreferences { + memberId: string + subscriptions: MarketingSubscription[] + suppressed: boolean +} + +const preferencesUrl: string = `${EnvironmentConfig.CONTACT_API}/me/subscriptions` + +/** + * Loads the authenticated member's current email category choices in Accounts settings. + * @returns Recorded choices and delivery suppression state, without inferring subscriptions. + * @throws Rejects when the authenticated Contact API request fails. + */ +export async function getMarketingPreferences(): Promise { + return xhrGetAsync(preferencesUrl) +} + +/** + * Saves the member's complete selection of active marketing email categories. + * @param subscriptionTypeIds Selected category IDs; an empty array opts out of every active category. + * @returns The server's saved preferences; identity is derived from the member JWT. + * @throws Rejects when authentication, category validation or persistence fails. + */ +export async function saveMarketingPreferences(subscriptionTypeIds: string[]): Promise { + return xhrPutAsync<{ subscriptionTypeIds: string[] }, MarketingPreferences>( + preferencesUrl, + { subscriptionTypeIds }, + ) +} diff --git a/src/apps/accounts/src/settings/change-email-verification/ChangeEmailVerificationPage.module.scss b/src/apps/accounts/src/settings/change-email-verification/ChangeEmailVerificationPage.module.scss new file mode 100644 index 000000000..d3965df1f --- /dev/null +++ b/src/apps/accounts/src/settings/change-email-verification/ChangeEmailVerificationPage.module.scss @@ -0,0 +1,23 @@ +@import '@libs/ui/styles/includes'; + +.layout { + margin: $sp-8 auto !important; +} + +.card { + align-items: center; + display: flex; + flex-direction: column; + gap: $sp-5; + margin: $sp-10 auto; + max-width: 620px; + text-align: center; + + p { + margin: 0; + } +} + +.error { + color: $red-100; +} diff --git a/src/apps/accounts/src/settings/change-email-verification/ChangeEmailVerificationPage.spec.tsx b/src/apps/accounts/src/settings/change-email-verification/ChangeEmailVerificationPage.spec.tsx new file mode 100644 index 000000000..c0c571d0b --- /dev/null +++ b/src/apps/accounts/src/settings/change-email-verification/ChangeEmailVerificationPage.spec.tsx @@ -0,0 +1,88 @@ +/* eslint-disable import/no-extraneous-dependencies, ordered-imports/ordered-imports */ +import '@testing-library/jest-dom' +import { render, screen, waitFor } from '@testing-library/react' +import type { PropsWithChildren } from 'react' + +import { + completeEmailChangeAsync, + getEmailChangeErrorMessage, +} from '~/apps/accounts/src/lib/services' + +import ChangeEmailVerificationPage from './ChangeEmailVerificationPage' + +let mockSearchParams = new URLSearchParams() + +jest.mock('react-router-dom', () => ({ + useSearchParams: (): [URLSearchParams] => [mockSearchParams], +})) + +jest.mock('~/apps/accounts/src/lib/services', () => ({ + completeEmailChangeAsync: jest.fn(), + getEmailChangeErrorMessage: jest.fn(), +}), { virtual: true }) + +jest.mock('~/libs/ui', () => ({ + ContentLayout: (props: PropsWithChildren): JSX.Element => ( +
{props.children}
+ ), + LinkButton: (props: { label: string, to: string }): JSX.Element => ( + {props.label} + ), + LoadingSpinner: (): JSX.Element => Loading, + PageTitle: (props: PropsWithChildren): JSX.Element => ( +

{props.children}

+ ), +}), { virtual: true }) + +const mockedCompleteEmailChange = completeEmailChangeAsync as jest.MockedFunction< + typeof completeEmailChangeAsync +> +const mockedGetErrorMessage = getEmailChangeErrorMessage as jest.MockedFunction< + typeof getEmailChangeErrorMessage +> + +describe('ChangeEmailVerificationPage', () => { + beforeEach(() => { + jest.clearAllMocks() + mockSearchParams = new URLSearchParams() + mockedGetErrorMessage.mockReturnValue('Validation failed.') + }) + + it('forwards the validation code and reports the changed address', async () => { + mockSearchParams = new URLSearchParams('code=signed%2Fcode') + mockedCompleteEmailChange.mockResolvedValue({ + email: 'new@example.com', + }) + + render() + + await waitFor(() => expect(mockedCompleteEmailChange) + .toHaveBeenCalledWith('signed/code')) + expect(await screen.findByText('Email changed')) + .toBeInTheDocument() + expect(screen.getByText('new@example.com is now your primary email address.')) + .toBeInTheDocument() + }) + + it('continues to accept validation tokens from legacy links', async () => { + mockSearchParams = new URLSearchParams('token=legacy-token') + mockedCompleteEmailChange.mockResolvedValue({ + email: 'new@example.com', + }) + + render() + + await waitFor(() => expect(mockedCompleteEmailChange) + .toHaveBeenCalledWith('legacy-token')) + }) + + it('does not call identity API when the link has no code', () => { + render() + + expect(screen.getByText('This email validation link is incomplete.')) + .toBeInTheDocument() + expect(mockedCompleteEmailChange) + .not + .toHaveBeenCalled() + }) +}) diff --git a/src/apps/accounts/src/settings/change-email-verification/ChangeEmailVerificationPage.tsx b/src/apps/accounts/src/settings/change-email-verification/ChangeEmailVerificationPage.tsx new file mode 100644 index 000000000..4e4abd4a9 --- /dev/null +++ b/src/apps/accounts/src/settings/change-email-verification/ChangeEmailVerificationPage.tsx @@ -0,0 +1,76 @@ +import { FC, useEffect, useRef, useState } from 'react' +import { useSearchParams } from 'react-router-dom' + +import { + completeEmailChangeAsync, + getEmailChangeErrorMessage, +} from '~/apps/accounts/src/lib/services' +import { ContentLayout, LinkButton, LoadingSpinner, PageTitle } from '~/libs/ui' + +import styles from './ChangeEmailVerificationPage.module.scss' + +type VerificationStatus = 'error' | 'loading' | 'success' + +/** + * Completes a pending email change when the member follows the validation link. + */ +const ChangeEmailVerificationPage: FC = () => { + const [searchParams] = useSearchParams() + const validationCode: string | null = searchParams.get('code') + ?? searchParams.get('token') + const [status, setStatus] = useState('loading') + const [message, setMessage] = useState('Validating your new email address…') + const requestedCode = useRef() + + useEffect(() => { + if (!validationCode) { + requestedCode.current = undefined + setStatus('error') + setMessage('This email validation link is incomplete.') + return + } + + if (requestedCode.current === validationCode) { + return + } + + requestedCode.current = validationCode + completeEmailChangeAsync(validationCode) + .then(response => { + setStatus('success') + setMessage(`${response.email} is now your primary email address.`) + }) + .catch(error => { + setStatus('error') + setMessage(getEmailChangeErrorMessage( + error, + 'This email validation link is invalid or has expired.', + )) + }) + }, [validationCode]) + + return ( + +
+ + {status === 'success' ? 'Email changed' : 'Validate email change'} + + {status === 'loading' && } +

+ {message} +

+ {status !== 'loading' && ( + + )} +
+
+ ) +} + +export default ChangeEmailVerificationPage diff --git a/src/apps/accounts/src/settings/change-email-verification/index.ts b/src/apps/accounts/src/settings/change-email-verification/index.ts new file mode 100644 index 000000000..af89295a3 --- /dev/null +++ b/src/apps/accounts/src/settings/change-email-verification/index.ts @@ -0,0 +1 @@ +export { default as ChangeEmailVerificationPage } from './ChangeEmailVerificationPage' diff --git a/src/apps/accounts/src/settings/tabs/account/user-and-pass/UserAndPassword.module.scss b/src/apps/accounts/src/settings/tabs/account/user-and-pass/UserAndPassword.module.scss index 09192f31e..ae1e34b16 100644 --- a/src/apps/accounts/src/settings/tabs/account/user-and-pass/UserAndPassword.module.scss +++ b/src/apps/accounts/src/settings/tabs/account/user-and-pass/UserAndPassword.module.scss @@ -16,4 +16,11 @@ max-width: 380px; } } -} \ No newline at end of file +} + +.changeEmailLink { + font-size: 12px !important; + line-height: 14px !important; + min-height: 0 !important; + padding: 0 !important; +} diff --git a/src/apps/accounts/src/settings/tabs/account/user-and-pass/UserAndPassword.tsx b/src/apps/accounts/src/settings/tabs/account/user-and-pass/UserAndPassword.tsx index fa3f9aab7..5d08c302b 100644 --- a/src/apps/accounts/src/settings/tabs/account/user-and-pass/UserAndPassword.tsx +++ b/src/apps/accounts/src/settings/tabs/account/user-and-pass/UserAndPassword.tsx @@ -19,8 +19,15 @@ import { UserTraits, } from '~/libs/core' import { SettingSection } from '~/apps/accounts/src/lib' +import { + getEmailChangeErrorMessage, + initiateEmailChangeAsync, + requestEmailChangeOtpAsync, + verifyEmailChangeOtpAsync, +} from '~/apps/accounts/src/lib/services' -import { UserAndPassFromConfig } from './user-and-pass.form.config' +import { ChangeEmailModal, ChangeEmailOtpModal } from './change-email' +import { createUserAndPassFormConfig } from './user-and-pass.form.config' import styles from './UserAndPassword.module.scss' interface UserAndPasswordProps { @@ -42,6 +49,41 @@ const UserAndPassword: FC = (props: UserAndPasswordProps) const { mutate: mutateTraits }: { mutate: KeyedMutator } = useMemberTraits(props.profile.handle) const [userConsent, setUserConsent]: [boolean, Dispatch] = useState(false) + const [isRequestingOtp, setIsRequestingOtp] = useState(false) + const [isResendingOtp, setIsResendingOtp] = useState(false) + const [isVerifyingOtp, setIsVerifyingOtp] = useState(false) + const [isSubmittingEmail, setIsSubmittingEmail] = useState(false) + const [isOtpModalOpen, setIsOtpModalOpen] = useState(false) + const [isChangeEmailModalOpen, setIsChangeEmailModalOpen] = useState(false) + const [otpError, setOtpError] = useState() + const [changeEmailError, setChangeEmailError] = useState() + const [emailChangeProof, setEmailChangeProof] = useState() + + /** + * Requests an ownership code at the current primary email and opens the OTP dialog. + * @returns a promise resolved after the code request finishes. + */ + const handleChangeEmailClick = useCallback(async (): Promise => { + setIsRequestingOtp(true) + setOtpError(undefined) + try { + await requestEmailChangeOtpAsync(props.profile.userId) + setIsOtpModalOpen(true) + toast.success(`Verification code sent to ${props.profile.email}.`) + } catch (error) { + toast.error(getEmailChangeErrorMessage( + error, + 'Unable to send a verification code. Please try again.', + )) + } finally { + setIsRequestingOtp(false) + } + }, [props.profile.email, props.profile.userId]) + + const userAndPassFormConfig = useMemo( + () => createUserAndPassFormConfig(handleChangeEmailClick, isRequestingOtp), + [handleChangeEmailClick, isRequestingOtp], + ) const requestGenerator: (inputs: ReadonlyArray) => any = useCallback((inputs: ReadonlyArray) => { @@ -90,6 +132,83 @@ const UserAndPassword: FC = (props: UserAndPasswordProps) }) } + /** + * Verifies a completed current-email OTP and opens the new-address dialog. + * @param otp six-digit code entered by the member. + * @returns a promise resolved after identity verification finishes. + */ + async function handleVerifyOtp(otp: string): Promise { + setIsVerifyingOtp(true) + setOtpError(undefined) + try { + const response = await verifyEmailChangeOtpAsync(props.profile.userId, otp) + setEmailChangeProof(response.verificationToken) + setIsOtpModalOpen(false) + setIsChangeEmailModalOpen(true) + } catch (error) { + setOtpError(getEmailChangeErrorMessage( + error, + 'The verification code could not be verified.', + )) + } finally { + setIsVerifyingOtp(false) + } + } + + /** + * Sends a replacement ownership code to the current primary email. + * @returns a promise resolved after the resend request finishes. + */ + async function handleResendOtp(): Promise { + setIsResendingOtp(true) + setOtpError(undefined) + try { + await requestEmailChangeOtpAsync(props.profile.userId) + toast.success(`A new verification code was sent to ${props.profile.email}.`) + } catch (error) { + setOtpError(getEmailChangeErrorMessage( + error, + 'Unable to resend the verification code.', + )) + } finally { + setIsResendingOtp(false) + } + } + + /** + * Sends the final validation link to the proposed new email address. + * @param email normalized address entered in the change-email dialog. + * @returns a promise resolved after the validation email request finishes. + */ + async function handleSubmitNewEmail(email: string): Promise { + if (!emailChangeProof) { + setChangeEmailError('Your verification expired. Start the email change again.') + return + } + + setIsSubmittingEmail(true) + setChangeEmailError(undefined) + try { + const response = await initiateEmailChangeAsync( + props.profile.userId, + email, + emailChangeProof, + ) + toast.success( + `Validation email sent to ${response.email}. Your primary email will change after validation.`, + ) + setIsChangeEmailModalOpen(false) + setEmailChangeProof(undefined) + } catch (error) { + setChangeEmailError(getEmailChangeErrorMessage( + error, + 'Unable to start the email change. Please try again.', + )) + } finally { + setIsSubmittingEmail(false) + } + } + function shouldDisableChangePasswordButton(): boolean { // pass reset form validation const specialChars: any = /[`!@#$%^&*()_+\-=[\]{};':"\\|,.<>/?~]/ @@ -127,14 +246,14 @@ const UserAndPassword: FC = (props: UserAndPasswordProps) contentClass={styles.content} >

- While your Topcoder handle or username and your email cannot be changed, - we encourage to change your password frequently. + While your Topcoder handle or username cannot be changed, + we encourage you to change your password frequently.

= (props: UserAndPasswordProps) )} />
+ + + + ) } diff --git a/src/apps/accounts/src/settings/tabs/account/user-and-pass/change-email/ChangeEmailModal.module.scss b/src/apps/accounts/src/settings/tabs/account/user-and-pass/change-email/ChangeEmailModal.module.scss new file mode 100644 index 000000000..ea4b11c0f --- /dev/null +++ b/src/apps/accounts/src/settings/tabs/account/user-and-pass/change-email/ChangeEmailModal.module.scss @@ -0,0 +1,29 @@ +@import '@libs/ui/styles/includes'; + +.container { + display: flex; + flex-direction: column; + gap: $sp-3; + min-width: 440px; + + > p { + margin: 0 0 $sp-2; + } + + @include ltesm { + min-width: 0; + } +} + +.actions { + align-items: center; + display: flex; + flex-wrap: wrap; + gap: $sp-2; + justify-content: flex-end; +} + +.spinner { + margin-right: auto; + width: 48px; +} diff --git a/src/apps/accounts/src/settings/tabs/account/user-and-pass/change-email/ChangeEmailModal.tsx b/src/apps/accounts/src/settings/tabs/account/user-and-pass/change-email/ChangeEmailModal.tsx new file mode 100644 index 000000000..2272e0e29 --- /dev/null +++ b/src/apps/accounts/src/settings/tabs/account/user-and-pass/change-email/ChangeEmailModal.tsx @@ -0,0 +1,134 @@ +import { FC, useEffect, useMemo } from 'react' +import { noop } from 'lodash' +import { useForm, UseFormReturn } from 'react-hook-form' +import { object, ObjectSchema, string } from 'yup' + +import { yupResolver } from '@hookform/resolvers/yup' +import { BaseModal, Button, InputText, LoadingSpinner } from '~/libs/ui' + +import styles from './ChangeEmailModal.module.scss' + +interface ChangeEmailForm { + email: string +} + +interface ChangeEmailModalProps { + currentEmail: string + error?: string + isOpen: boolean + isSubmitting: boolean + onClose: () => void + onSubmit: (email: string) => Promise +} + +/** + * Shows the member's current address and collects a different valid address. + */ +const ChangeEmailModal: FC = ( + props: ChangeEmailModalProps, +) => { + const schema: ObjectSchema = useMemo(() => object({ + email: string() + .trim() + .email('Enter a valid email address.') + .required('New email is required.') + .test( + 'different-email', + 'The new email must be different from your current email.', + value => value?.toLowerCase() !== props.currentEmail.toLowerCase(), + ), + }), [props.currentEmail]) + + const { + formState: { errors, isValid }, + handleSubmit, + register, + reset, + }: UseFormReturn = useForm({ + defaultValues: { email: '' }, + mode: 'all', + resolver: yupResolver(schema), + }) + + useEffect(() => { + if (props.isOpen) { + reset({ email: '' }) + } + }, [props.isOpen, reset]) + + /** + * Normalizes and forwards a valid proposed email. + * @param values validated modal form values. + * @returns a promise resolved after the validation email request completes. + */ + async function submit(values: ChangeEmailForm): Promise { + await props.onSubmit(values.email.trim() + .toLowerCase()) + } + + return ( + + +

+ We'll send a validation link to the new address. Your + primary email will stay unchanged until that link is used. +

+ + + + +
+ {props.isSubmitting && ( + + )} +
+ +
+ ) +} + +export default ChangeEmailModal diff --git a/src/apps/accounts/src/settings/tabs/account/user-and-pass/change-email/ChangeEmailModals.spec.tsx b/src/apps/accounts/src/settings/tabs/account/user-and-pass/change-email/ChangeEmailModals.spec.tsx new file mode 100644 index 000000000..d6705e5b9 --- /dev/null +++ b/src/apps/accounts/src/settings/tabs/account/user-and-pass/change-email/ChangeEmailModals.spec.tsx @@ -0,0 +1,110 @@ +/* eslint-disable import/no-extraneous-dependencies, ordered-imports/ordered-imports, react/jsx-no-bind */ +import '@testing-library/jest-dom' +import { fireEvent, render, screen } from '@testing-library/react' +import type { PropsWithChildren, ReactNode } from 'react' + +import ChangeEmailModal from './ChangeEmailModal' +import ChangeEmailOtpModal from './ChangeEmailOtpModal' + +jest.mock('~/libs/ui', () => ({ + BaseModal: (props: PropsWithChildren<{ + closeOnOverlayClick?: boolean + onClose: () => void + open: boolean + size?: string + title?: ReactNode + }>): JSX.Element => (props.open ? ( +
{ + if (props.closeOnOverlayClick !== false) { + props.onClose() + } + }} + > +
+ {props.title} + + {props.children} +
+
+ ) : <>), + Button: (props: { + disabled?: boolean + label: ReactNode + onClick?: () => void + }): JSX.Element => ( + + ), + InputText: (props: { + disabled?: boolean + label: string + placeholder?: string + value?: string + }): JSX.Element => ( + + ), + LoadingCircles: (): JSX.Element => Verifying, + LoadingSpinner: (): JSX.Element => Submitting, +}), { virtual: true }) + +describe('change email modals', () => { + it('uses the large modal width for the change email form', async () => { + render( + , + ) + + expect(await screen.findByTestId('modal')) + .toHaveClass('modal-lg') + }) + + it('keeps the OTP modal open after a backdrop click', async () => { + const onClose = jest.fn() + + render( + , + ) + + fireEvent.click(await screen.findByTestId('modal-container')) + + expect(onClose) + .not + .toHaveBeenCalled() + expect(screen.getByText('CHECK YOUR EMAIL FOR A CODE')) + .toBeInTheDocument() + + fireEvent.click(screen.getByTestId('close-button')) + expect(onClose) + .toHaveBeenCalledTimes(1) + }) +}) diff --git a/src/apps/accounts/src/settings/tabs/account/user-and-pass/change-email/ChangeEmailOtpModal.module.scss b/src/apps/accounts/src/settings/tabs/account/user-and-pass/change-email/ChangeEmailOtpModal.module.scss new file mode 100644 index 000000000..5c232a5e1 --- /dev/null +++ b/src/apps/accounts/src/settings/tabs/account/user-and-pass/change-email/ChangeEmailOtpModal.module.scss @@ -0,0 +1,56 @@ +@import '@libs/ui/styles/includes'; + +.container { + align-items: center; + display: flex; + flex-direction: column; + justify-content: center; + padding: $sp-5 $sp-5 0; + + p { + color: $black-80; + margin: $sp-2 0 $sp-4; + max-width: 480px; + text-align: center; + } + + .error { + color: $red-100; + margin-top: 0; + } +} + +.otpInput { + border: $border solid $black-40; + border-radius: $sp-1; + font-size: 20px; + height: 48px; + margin: $sp-1; + text-align: center; + width: 48px !important; + + &:focus { + border-color: $turq-160; + outline: none; + } + + &:disabled { + background: $black-10; + } + + &::-webkit-inner-spin-button, + &::-webkit-outer-spin-button { + -webkit-appearance: none; + margin: 0; + } + + &[type='number'] { + -moz-appearance: textfield; + } + + @include ltesm { + height: 40px; + margin: 2px; + width: 40px !important; + } +} diff --git a/src/apps/accounts/src/settings/tabs/account/user-and-pass/change-email/ChangeEmailOtpModal.tsx b/src/apps/accounts/src/settings/tabs/account/user-and-pass/change-email/ChangeEmailOtpModal.tsx new file mode 100644 index 000000000..8fa2e20fd --- /dev/null +++ b/src/apps/accounts/src/settings/tabs/account/user-and-pass/change-email/ChangeEmailOtpModal.tsx @@ -0,0 +1,138 @@ +import { FC, useEffect, useState } from 'react' +import OTPInput, { InputProps } from 'react-otp-input' + +import { BaseModal, Button, LoadingCircles } from '~/libs/ui' + +import styles from './ChangeEmailOtpModal.module.scss' + +const RESEND_DELAY_MS: number = 60_000 + +interface ChangeEmailOtpModalProps { + email: string + error?: string + isOpen: boolean + isResending: boolean + isVerifying: boolean + onClose: () => void + onResend: () => Promise + onVerify: (otp: string) => Promise +} + +/** + * Collects the ownership code sent to the member's current primary email. + */ +const ChangeEmailOtpModal: FC = ( + props: ChangeEmailOtpModalProps, +) => { + const [otp, setOtp] = useState('') + const [canResend, setCanResend] = useState(false) + const [resendSequence, setResendSequence] = useState(0) + + useEffect(() => { + if (!props.isOpen) { + setOtp('') + setCanResend(false) + return undefined + } + + const timer: NodeJS.Timeout = setTimeout(() => { + setCanResend(true) + }, RESEND_DELAY_MS) + + return () => clearTimeout(timer) + }, [props.isOpen, resendSequence]) + + useEffect(() => { + if (props.error) { + setOtp('') + } + }, [props.error]) + + /** + * Updates the six OTP inputs and verifies a complete code. + * @param value digits currently entered by the member. + * @returns a promise resolved after any complete-code verification request. + */ + async function handleOtpChange(value: string): Promise { + const digits: string = value.replace(/\D/g, '') + .slice(0, 6) + setOtp(digits) + if (digits.length === 6 && !props.isVerifying) { + await props.onVerify(digits) + } + } + + /** + * Requests another code and restarts the resend cooldown. + * @returns a promise resolved after the resend request completes. + */ + async function handleResend(): Promise { + setCanResend(false) + setOtp('') + await props.onResend() + setResendSequence(value => value + 1) + } + + /** + * Renders one accessible OTP digit input. + * @param inputProps properties supplied by react-otp-input. + * @returns a styled input element for one code digit. + */ + function renderOtpInput(inputProps: InputProps): JSX.Element { + return ( + + ) + } + + return ( + +
+

+ For added security, we sent a 6-digit code to + {' '} + {props.email} + . Enter it below before changing your email address. +

+ + {props.error && ( +

{props.error}

+ )} + + + + {props.isVerifying && } + +

Can't find the code? Check your spam folder.

+
+
+ ) +} + +export default ChangeEmailOtpModal diff --git a/src/apps/accounts/src/settings/tabs/account/user-and-pass/change-email/index.ts b/src/apps/accounts/src/settings/tabs/account/user-and-pass/change-email/index.ts new file mode 100644 index 000000000..c58ef8d45 --- /dev/null +++ b/src/apps/accounts/src/settings/tabs/account/user-and-pass/change-email/index.ts @@ -0,0 +1,2 @@ +export { default as ChangeEmailModal } from './ChangeEmailModal' +export { default as ChangeEmailOtpModal } from './ChangeEmailOtpModal' diff --git a/src/apps/accounts/src/settings/tabs/account/user-and-pass/user-and-pass.form.config.tsx b/src/apps/accounts/src/settings/tabs/account/user-and-pass/user-and-pass.form.config.tsx index 7b2ae8fe1..8265eab9a 100644 --- a/src/apps/accounts/src/settings/tabs/account/user-and-pass/user-and-pass.form.config.tsx +++ b/src/apps/accounts/src/settings/tabs/account/user-and-pass/user-and-pass.form.config.tsx @@ -1,96 +1,127 @@ +import { MouseEvent } from 'react' import { noop } from 'lodash' -import { FormDefinition, validatorRequired } from '~/libs/ui' +import { Button, FormDefinition, validatorRequired } from '~/libs/ui' import PasswordTips from './password-tips' +import styles from './UserAndPassword.module.scss' -export const UserAndPassFromConfig: FormDefinition = { - buttons: { - primaryGroup: [], - secondaryGroup: [ - { - buttonStyle: 'secondary', - label: 'Change Password', - onClick: noop, - type: 'submit', - }, - ], - }, - groups: [ - { - inputs: [ +/** + * Builds the username/password form and attaches the email-change action. + * + * @param onChangeEmail callback that starts current-email ownership verification. + * @param isChangeEmailLoading whether the ownership-code request is in progress. + * @returns the account form definition used by the shared Form component. + */ +export function createUserAndPassFormConfig( + onChangeEmail: () => void, + isChangeEmailLoading: boolean, +): FormDefinition { + return { + buttons: { + primaryGroup: [], + secondaryGroup: [ { - disabled: true, - label: 'Username', - name: 'handle', - type: 'text', + buttonStyle: 'secondary', + label: 'Change Password', + onClick: noop, + type: 'submit', }, - { - disabled: true, - label: 'Primary Email', - name: 'email', - type: 'text', - }, - { - hideInlineErrors: true, - label: 'Current Password', - name: 'currentPassword', - placeholder: 'Type your current password', - type: 'password', - validators: [ - { - validator: validatorRequired, - }, - ], - }, - { - hideInlineErrors: true, - label: 'New Password', - name: 'newPassword', - placeholder: 'Type your new password', - tooltip: { - className: 'passTooltip', - content: , - place: 'bottom', + ], + }, + groups: [ + { + inputs: [ + { + disabled: true, + label: 'Username', + name: 'handle', + type: 'text', + }, + { + actionElement: ( + + ), +}), { virtual: true }) + +const mockedGet = xhrGetAsync as jest.Mock +const mockedPut = xhrPutAsync as jest.Mock +const preferences = { + memberId: '123', + subscriptions: [ + { + active: true, + description: 'Community news', + name: 'Community Newsletter', + source: 'Member preference', + subscribed: true, + subscriptionTypeId: 'newsletter', + }, + { + active: true, + description: 'Match announcements', + name: 'Marathon Match', + source: 'No recorded preference', + subscribed: false, + subscriptionTypeId: 'marathon', + }, + ], + suppressed: false, +} + +describe('member marketing preferences', () => { + beforeEach(() => { + mockedGet.mockReset() + mockedPut.mockReset() + }) + + it('loads actual choices and explicitly saves an empty selection without a target member ID', async () => { + mockedGet.mockResolvedValue(preferences) + mockedPut.mockResolvedValue({ + ...preferences, + subscriptions: preferences.subscriptions.map(item => ({ ...item, subscribed: false })), + }) + render() + const newsletter = await screen.findByRole('checkbox', { name: /Community Newsletter/ }) + expect(newsletter) + .toBeChecked() + expect(screen.getByRole('checkbox', { name: /Marathon Match/ })) + .not.toBeChecked() + expect(mockedPut) + .not.toHaveBeenCalled() + fireEvent.click(newsletter) + fireEvent.click(screen.getByRole('button', { name: 'Save email preferences' })) + await screen.findByText('Your email preferences are saved.') + expect(mockedPut) + .toHaveBeenCalledWith('https://api.example.test/v6/contact/me/subscriptions', { subscriptionTypeIds: [] }) + }) + + it('keeps choices unavailable after load failure and permits an explicit retry', async () => { + mockedGet.mockRejectedValueOnce(new Error('Unavailable')) + .mockResolvedValueOnce(preferences) + render() + await screen.findByRole('alert') + expect(screen.queryByRole('checkbox')) + .not.toBeInTheDocument() + expect(mockedPut) + .not.toHaveBeenCalled() + fireEvent.click(screen.getByRole('button', { name: 'Try again' })) + await screen.findByRole('checkbox', { name: /Community Newsletter/ }) + }) + + it('retains unsaved choices on save failure and displays real suppression without clearing it', async () => { + mockedGet.mockResolvedValue({ ...preferences, suppressed: true }) + mockedPut.mockRejectedValue(new Error('Unavailable')) + render() + const marathon = await screen.findByRole('checkbox', { name: /Marathon Match/ }) + expect(screen.getByText(/Delivery to your email address is paused/)) + .toBeInTheDocument() + fireEvent.click(marathon) + fireEvent.click(screen.getByRole('button', { name: 'Save email preferences' })) + await screen.findByRole('alert') + expect(marathon) + .toBeChecked() + await waitFor(() => expect(screen.getByRole('button', { name: 'Save email preferences' })) + .toBeEnabled()) + expect(screen.queryByText('Your email preferences are saved.')) + .not.toBeInTheDocument() + }) +}) diff --git a/src/apps/accounts/src/settings/tabs/preferences/MarketingPreferences.tsx b/src/apps/accounts/src/settings/tabs/preferences/MarketingPreferences.tsx new file mode 100644 index 000000000..a665553a3 --- /dev/null +++ b/src/apps/accounts/src/settings/tabs/preferences/MarketingPreferences.tsx @@ -0,0 +1,142 @@ +import { ChangeEvent, FC, useCallback, useEffect, useState } from 'react' + +import { Button } from '~/libs/ui' + +import { + getMarketingPreferences, + MarketingPreferences as Preferences, + saveMarketingPreferences, +} from '../../../lib/services/marketing-preferences.service' + +import styles from './PreferencesTab.module.scss' + +/** + * Shows and saves the signed-in member's marketing email choices in Accounts preferences. + * @returns A category form with explicit opt-in, loading/error states and delivery status. + * @throws No render exceptions for API failures; rejected reads/writes show a retryable message. + */ +const MarketingPreferences: FC = () => { + const [preferences, setPreferences] = useState() + const [selected, setSelected] = useState([]) + const [loading, setLoading] = useState(true) + const [saving, setSaving] = useState(false) + const [error, setError] = useState('') + const [saved, setSaved] = useState(false) + const [reload, setReload] = useState(0) + + useEffect(() => { + let mounted = true + setLoading(true) + setError('') + getMarketingPreferences() + .then(response => { + if (!mounted) return + setPreferences(response) + setSelected(response.subscriptions.filter(item => item.active && item.subscribed) + .map(item => item.subscriptionTypeId)) + }) + .catch(() => { + if (mounted) setError('Email preferences could not be loaded. Please try again.') + }) + .finally(() => { + if (mounted) setLoading(false) + }) + return () => { mounted = false } + }, [reload]) + + /** Retry a failed preference read; takes no input, returns void and does not throw. */ + const handleReload = useCallback((): void => { setReload(value => value + 1) }, []) + + /** + * Change one local category choice before save. + * @param event Checkbox change containing its category ID and checked state. + * @returns Nothing; updates the pending selection and clears an old saved confirmation. + * @throws No exceptions. + */ + const handleChange = useCallback((event: ChangeEvent): void => { + const { checked, value }: { checked: boolean, value: string } = event.currentTarget + setSelected(previous => (checked ? [...previous, value] : previous.filter(id => id !== value))) + setSaved(false) + }, []) + + /** + * Persist the full pending selection, including an intentional empty selection. + * @returns Completion after the API confirms the server's saved choices. + * @throws No API exceptions outward; failures preserve the pending selection for retry. + */ + const handleSave = useCallback(async (): Promise => { + setSaving(true) + setError('') + setSaved(false) + try { + const response = await saveMarketingPreferences(selected) + setPreferences(response) + setSelected(response.subscriptions.filter(item => item.active && item.subscribed) + .map(item => item.subscriptionTypeId)) + setSaved(true) + } catch { + setError('Your email preferences were not saved. Please try again.') + } finally { + setSaving(false) + } + }, [selected]) + + const categories = preferences?.subscriptions.filter(item => item.active) ?? [] + const current = categories.filter(item => item.subscribed) + .map(item => item.subscriptionTypeId) + const changed = current.length !== selected.length || selected.some(id => !current.includes(id)) + + return ( +
+

Email preferences

+

Choose the Topcoder newsletters and updates you would like to receive.

+ {loading &&

Loading your email preferences…

} + {error &&

{error}

} + {!loading && !preferences && ( +
+ ) +} + +export default MarketingPreferences diff --git a/src/apps/accounts/src/settings/tabs/preferences/PreferencesTab.module.scss b/src/apps/accounts/src/settings/tabs/preferences/PreferencesTab.module.scss index c078309f0..309e695ea 100644 --- a/src/apps/accounts/src/settings/tabs/preferences/PreferencesTab.module.scss +++ b/src/apps/accounts/src/settings/tabs/preferences/PreferencesTab.module.scss @@ -36,4 +36,71 @@ } } } -} \ No newline at end of file +} + +.marketing { + border-bottom: 1px solid $black-10; + margin-bottom: $sp-8; + padding-bottom: $sp-8; + + h4 { + margin-bottom: $sp-3; + } + + p { + margin: $sp-3 0; + } +} + +.emailCategories { + border: 0; + margin: $sp-5 0; + padding: 0; +} + +.preferenceLegend { + font-weight: 700; + margin-bottom: $sp-3; +} + +.emailCategory { + display: flex; + align-items: flex-start; + gap: $sp-3; + padding: $sp-3 0; + cursor: pointer; + + input { + accent-color: $turq-160; + flex-shrink: 0; + height: 20px; + margin-top: 2px; + width: 20px; + } + + span span { + display: block; + margin-top: $sp-1; + color: $black-60; + } +} + +.preferenceError { + color: #a32727; +} + +.deliveryNotice { + background: $black-5; + border-radius: 4px; + padding: $sp-3; +} + +.preferenceHint { + color: $black-60; + font-size: 14px; + padding-bottom: $sp-3; +} + +.preferenceSaved { + color: $turq-160; +} diff --git a/src/apps/accounts/src/settings/tabs/preferences/PreferencesTab.tsx b/src/apps/accounts/src/settings/tabs/preferences/PreferencesTab.tsx index d1a17044d..e9d28a569 100644 --- a/src/apps/accounts/src/settings/tabs/preferences/PreferencesTab.tsx +++ b/src/apps/accounts/src/settings/tabs/preferences/PreferencesTab.tsx @@ -6,8 +6,10 @@ import { EnvironmentConfig } from '~/config' import { ForumIcon, SettingSection } from '../../../lib' +import MarketingPreferences from './MarketingPreferences' import styles from './PreferencesTab.module.scss' +/** Renders member email choices and the existing forum preferences link; has no inputs and handles no API errors directly. */ const PreferencesTab: FC = () => { function handleGoToForumPreferences(): void { window.open( @@ -23,6 +25,7 @@ const PreferencesTab: FC = () => {

PLATFORM PREFERENCES

+ diff --git a/src/apps/admin/src/admin-app.routes.tsx b/src/apps/admin/src/admin-app.routes.tsx index 1d6da0bd2..829f2f696 100644 --- a/src/apps/admin/src/admin-app.routes.tsx +++ b/src/apps/admin/src/admin-app.routes.tsx @@ -10,6 +10,7 @@ import { aiReviewTemplatesRouteId, aiReviewWorkflowsRouteId, aiRouteId, + aiTopScoutRagRouteId, billingAccountRouteId, defaultReviewersRouteId, gamificationAdminRouteId, @@ -186,6 +187,11 @@ const AiReviewTemplatesPage: LazyLoadedComponent = lazyLoad( 'AiReviewTemplatesPage', ) +const TopScoutRagPage: LazyLoadedComponent = lazyLoad( + () => import('./ai/topscout-rag/TopScoutRagPage'), + 'TopScoutRagPage', +) + export const toolTitle: string = ToolTitle.admin export const adminRoutes: ReadonlyArray = [ @@ -444,6 +450,11 @@ export const adminRoutes: ReadonlyArray = [ id: 'ai-review-templates-page', route: aiReviewTemplatesRouteId, }, + { + element: , + id: 'ai-topscout-rag-page', + route: aiTopScoutRagRouteId, + }, ], element: , id: aiRouteId, diff --git a/src/apps/admin/src/ai/topscout-rag/FormFields.module.scss b/src/apps/admin/src/ai/topscout-rag/FormFields.module.scss new file mode 100644 index 000000000..88c1c9eda --- /dev/null +++ b/src/apps/admin/src/ai/topscout-rag/FormFields.module.scss @@ -0,0 +1,33 @@ +@import '@libs/ui/styles/includes'; + +.field { + display: flex; + flex-direction: column; + min-width: 0; +} + +.fieldLabel { + margin-bottom: $sp-1; + font-family: $font-barlow; + font-size: 12px; + line-height: 16px; + color: $black-100; +} + +.fieldHint { + margin-top: $sp-2; + font-size: 12px; + color: $black-60; +} + +.input { + width: 100%; + border: none; + outline: none; + background: transparent; + + &:disabled { + background: transparent; + cursor: not-allowed; + } +} diff --git a/src/apps/admin/src/ai/topscout-rag/FormFields.tsx b/src/apps/admin/src/ai/topscout-rag/FormFields.tsx new file mode 100644 index 000000000..99ed4b6c7 --- /dev/null +++ b/src/apps/admin/src/ai/topscout-rag/FormFields.tsx @@ -0,0 +1,54 @@ +/** + * Field primitives for this page. + * + * The platform's `InputText`/`InputSelect` render their label *inside* the + * bordered box (InputWrapper puts it in the `