{"configuration":{},"description":"Describes solution architecture for OP Solutions' EHR called Compass","documentation":{"sections":[{"content":"# -\n\n## Introduction\n\nThis repository contains architectural description of OP Solutions' Compass EHR application, and possibly more in the future.\nIt uses [C4 model](https://c4model.com) to describe the architecture of a system in the context, and [Structurizr](https://structurizr.com/) diagram-as-a-code tool to generate the diagrams, along with necessary documentation and [ADRs](https://cognitect.com/blog/2011/11/15/documenting-architecture-decisions) in [Markdown](https://www.markdownguide.org/cheat-sheet/), occasionally using [PlantUML](https://plantuml.com/) to describe more in-the-trenches things.\n\n## Where to see\n\nThe architecture is available in the following places:\n\n- [architecture.opsehr.com](https://architecture.opsehr.com/) - our self-hosted server\n\n","filename":"landscape.md","format":"Markdown","order":1,"title":""}]},"id":1,"lastModifiedAgent":"structurizr-cli/","lastModifiedDate":"2026-08-15T10:31:13Z","lastModifiedUser":"root@29d677caf17d","model":{"deploymentNodes":[{"children":[{"containerInstances":[{"containerId":"16","deploymentGroups":["Default"],"environment":"Production","id":"34","instanceId":1,"properties":{"structurizr.dsl.identifier":"49b1584b-f09b-48a6-bec2-6929213f9c3c"},"relationships":[{"description":"calls","destinationId":"44","id":"46","linkedRelationshipId":"25","sourceId":"34","technology":"REST/HTTPS"}],"tags":"Container Instance"}],"environment":"Production","id":"33","instances":"1","name":"Browser","properties":{"structurizr.dsl.identifier":"dn_browser"},"relationships":[{"description":"calls","destinationId":"35","id":"61","sourceId":"33","tags":"Relationship,DeploymentLevel","technology":"HTTPS"}],"tags":"Element,Deployment Node","technology":"Chrome/Firefox/Safari"}],"environment":"Production","id":"32","instances":"1","name":"User's PC","properties":{"structurizr.dsl.identifier":"dn_userPC"},"tags":"Element,Deployment Node","technology":"Windows/MacOS/Linux"},{"environment":"Production","id":"35","infrastructureNodes":[{"description":"DNS/DDoS protection/SSL termination service","environment":"Production","id":"36","name":"CloudFlare","properties":{"structurizr.dsl.identifier":"in_cloudFlare"},"tags":"Element,Infrastructure Node","technology":"CloudFlare"}],"instances":"1","name":"CloudFlare","properties":{"structurizr.dsl.identifier":"dn_cloudFlare"},"relationships":[{"description":"proxies to","destinationId":"39","id":"62","sourceId":"35","tags":"Relationship,DeploymentLevel","technology":"HTTPS"}],"tags":"Element,Deployment Node","technology":"CloudFlare"},{"children":[{"children":[{"children":[{"environment":"Production","id":"41","infrastructureNodes":[{"description":"A server that serves the SPA UI","environment":"Production","id":"42","name":"Frontend server","properties":{"structurizr.dsl.identifier":"in_compass_frontendServer"},"tags":"Element,Infrastructure Node","technology":"nginx"}],"instances":"1","name":"Frontend server","properties":{"structurizr.dsl.identifier":"dn_frontendServer"},"tags":"Element,Deployment Node,Kubernetes - pod","technology":"k8s pod"},{"containerInstances":[{"containerId":"18","deploymentGroups":["Default"],"environment":"Production","id":"44","instanceId":1,"properties":{"structurizr.dsl.identifier":"ci_compass_api"},"relationships":[{"description":"notifies","destinationId":"34","id":"45","linkedRelationshipId":"26","sourceId":"44","technology":"SignalR"},{"description":"indexes/searches the data","destinationId":"48","id":"49","linkedRelationshipId":"30","sourceId":"44","technology":"REST/HTTPS"},{"description":"reads/writes data","destinationId":"51","id":"52","linkedRelationshipId":"27","sourceId":"44","technology":"TCP"},{"description":"saves/retrieves blobs","destinationId":"54","id":"55","linkedRelationshipId":"28","sourceId":"44","technology":"REST/HTTPS"},{"description":"saves/reads table rows","destinationId":"56","id":"57","linkedRelationshipId":"29","sourceId":"44","technology":"REST/HTTPS"},{"description":"sends emails","destinationId":"59","id":"60","linkedRelationshipId":"31","sourceId":"44","technology":"REST/HTTPS"}],"tags":"Container Instance"}],"environment":"Production","id":"43","instances":"1..N","name":"Backend replica-set","properties":{"structurizr.dsl.identifier":"dn_backendServer"},"relationships":[{"description":"uses","destinationId":"50","id":"65","sourceId":"43","tags":"Relationship,DeploymentLevel","technology":"TCP"},{"description":"uses","destinationId":"47","id":"66","sourceId":"43","tags":"Relationship,DeploymentLevel","technology":"REST/HTTP"},{"description":"uses","destinationId":"53","id":"67","sourceId":"43","tags":"Relationship,DeploymentLevel","technology":"REST/HTTPS"},{"description":"uses","destinationId":"58","id":"68","sourceId":"43","tags":"Relationship,DeploymentLevel","technology":"REST/HTTPS"}],"tags":"Element,Deployment Node,Kubernetes - pod","technology":"k8s pod"},{"containerInstances":[{"containerId":"21","deploymentGroups":["Default"],"environment":"Production","id":"48","instanceId":1,"properties":{"structurizr.dsl.identifier":"in_compass_search"},"tags":"Container Instance"}],"environment":"Production","id":"47","instances":"1..N","name":"Storage engine replica-set","properties":{"structurizr.dsl.identifier":"dn_searchEngine"},"tags":"Element,Deployment Node,Kubernetes - pod","technology":"k8s pod"}],"environment":"Production","id":"40","instances":"1..N","name":"AKS VMSS VM","properties":{"structurizr.dsl.identifier":"dn_aks_vm"},"tags":"Element,Deployment Node,Microsoft Azure - VM Scale Sets","technology":"Azure Linux"}],"environment":"Production","id":"38","infrastructureNodes":[{"description":"An ingress into the application cluster","environment":"Production","id":"39","name":"Ingress","properties":{"structurizr.dsl.identifier":"compass_ingress"},"relationships":[{"description":"/","destinationId":"41","id":"63","sourceId":"39","tags":"Relationship,DeploymentLevel","technology":"HTTP"},{"description":"/api","destinationId":"43","id":"64","sourceId":"39","tags":"Relationship,DeploymentLevel","technology":"HTTP"}],"tags":"Element,Infrastructure Node,Kubernetes - ing","technology":"Azure Application Gateway"}],"instances":"1","name":"Kubernetes cluster","properties":{"structurizr.dsl.identifier":"dn_aks"},"tags":"Element,Deployment Node,Microsoft Azure - Kubernetes Services","technology":"Azure Kubernetes Service"},{"containerInstances":[{"containerId":"17","deploymentGroups":["Default"],"environment":"Production","id":"51","instanceId":1,"properties":{"structurizr.dsl.identifier":"a1eba011-196f-4ffb-9b84-b007bef3f26c"},"tags":"Container Instance"}],"environment":"Production","id":"50","instances":"1","name":"Azure SQL Server","properties":{"structurizr.dsl.identifier":"dn_azsql"},"tags":"Element,Deployment Node,Microsoft Azure - SQL Server","technology":"Azure/east-us"},{"containerInstances":[{"containerId":"19","deploymentGroups":["Default"],"environment":"Production","id":"54","instanceId":1,"properties":{"structurizr.dsl.identifier":"90368ada-685e-462a-9cb4-aa60bdd93262"},"tags":"Container Instance"},{"containerId":"20","deploymentGroups":["Default"],"environment":"Production","id":"56","instanceId":1,"properties":{"structurizr.dsl.identifier":"ebc07338-3b74-45fb-9516-d0ee9a451dc5"},"tags":"Container Instance"}],"environment":"Production","id":"53","instances":"1","name":"Azure Storage Account","properties":{"structurizr.dsl.identifier":"az_azstorage"},"tags":"Element,Deployment Node,Microsoft Azure - Storage Accounts","technology":"Azure/east-us"},{"containerInstances":[{"containerId":"22","deploymentGroups":["Default"],"environment":"Production","id":"59","instanceId":1,"properties":{"structurizr.dsl.identifier":"7fb66887-bbfe-4980-8eb8-3bd3367f31f0"},"tags":"Container Instance"}],"environment":"Production","id":"58","instances":"1","name":"Azure Communication Services","properties":{"structurizr.dsl.identifier":"az_azcommservices"},"tags":"Element,Deployment Node,Microsoft Azure - Communication Services","technology":"Azure/global"}],"environment":"Production","id":"37","instances":"1","name":"Azure resource group","properties":{"structurizr.dsl.identifier":"dn_azure"},"tags":"Element,Deployment Node,Microsoft Azure - Resource Groups","technology":"Azure/east-us"}],"people":[{"description":"A generic clinic staff person","group":"OP Clinic","id":"1","name":"Staff","properties":{"structurizr.dsl.identifier":"person_op_generic"},"relationships":[{"description":"uses the system","destinationId":"15","id":"72","sourceId":"1","tags":"Relationship,ContextLevel"}],"tags":"Element,Person,OP Clinic Person,Generic"},{"description":"Front-office administrator","group":"OP Clinic","id":"2","name":"FOA","properties":{"structurizr.dsl.identifier":"person_op_foa"},"relationships":[{"description":"uses the office portal","destinationId":"15","id":"73","sourceId":"2","tags":"Relationship,ContextLevel"}],"tags":"Element,Person,OP Clinic Person"},{"description":"Back-office administrator","group":"OP Clinic","id":"3","name":"BOA","properties":{"structurizr.dsl.identifier":"person_op_boa"},"relationships":[{"description":"uses the office portal","destinationId":"15","id":"74","sourceId":"3","tags":"Relationship,ContextLevel"}],"tags":"Element,Person,OP Clinic Person"},{"description":"Person who provides OP services","group":"OP Clinic","id":"4","name":"OP Clinician","properties":{"structurizr.dsl.identifier":"person_op_opclinician"},"relationships":[{"description":"uses the office portal","destinationId":"15","id":"75","sourceId":"4","tags":"Relationship,ContextLevel"}],"tags":"Element,Person,OP Clinic Person"},{"description":"An actual physician/doctor that refers the patient to our system","group":"External users","id":"8","name":"Treating practitioner","properties":{"structurizr.dsl.identifier":"person_physician"},"relationships":[{"description":"refers to the OP clinic","destinationId":"9","id":"77","sourceId":"8","tags":"Relationship,ContextLevel"}],"tags":"Element,Person,External Person"},{"description":"Healthcare service receiver","group":"External users","id":"9","name":"Patient","properties":{"structurizr.dsl.identifier":"person_patient"},"relationships":[{"description":"uses the patient portal","destinationId":"15","id":"76","sourceId":"9","tags":"Relationship,ContextLevel"}],"tags":"Element,Person,External Person"},{"description":"Anyone with access to the system, e.g., clinic staff or patient","group":"External users","id":"10","name":"User","properties":{"structurizr.dsl.identifier":"person_generic"},"relationships":[{"description":"uses the UI","destinationId":"16","id":"23","sourceId":"10","tags":"Relationship,ComponentLevel","technology":"HTTPS"},{"description":"uses the UI","destinationId":"15","id":"24","linkedRelationshipId":"23","sourceId":"10","technology":"HTTPS"}],"tags":"Element,Person,External Person,Generic"},{"description":"Handles customer support tickets.","group":"OP Solutions","id":"11","name":"Customer Support","properties":{"structurizr.dsl.identifier":"person_ops_customerSupport"},"relationships":[{"description":"handles customer tickets","destinationId":"14","id":"79","sourceId":"11","tags":"Relationship,ContextLevel","technology":"Freshdesk"}],"tags":"Element,Person,Internal Person"},{"description":"Develops/maintains the system","group":"OP Solutions","id":"12","name":"Developer","properties":{"structurizr.dsl.identifier":"person_ops_developer"},"relationships":[{"description":"develops the system","destinationId":"13","id":"80","sourceId":"12","tags":"Relationship,ContextLevel","technology":"Azure DevOps"}],"tags":"Element,Person,Internal Person"}],"properties":{"structurizr.sort":"type","structurizr.groupSeparator":"/"},"softwareSystems":[{"description":"Claim exporting","documentation":{},"group":"Partners","id":"5","name":"Encoda","properties":{"structurizr.dsl.identifier":"ss_encoda"},"tags":"Element,Software System,External System"},{"description":"Data exporting","documentation":{},"group":"Partners","id":"6","name":"Mayo LLPR","properties":{"structurizr.dsl.identifier":"ss_mayo"},"tags":"Element,Software System,External System"},{"description":"Inventory system","documentation":{},"group":"Partners","id":"7","name":"Cascade","properties":{"structurizr.dsl.identifier":"ss_cascade"},"tags":"Element,Software System,External System"},{"description":"Code repository, wiki, issue tracker, CI/CD","documentation":{},"group":"OP Solutions","id":"13","name":"Azure DevOps","properties":{"structurizr.dsl.identifier":"ss_ops_devOps"},"relationships":[{"description":"CI/CD","destinationId":"15","id":"81","sourceId":"13","tags":"Relationship,ContextLevel","technology":"Azure DevOps"}],"tags":"Element,Software System,Internal System"},{"description":"Customer support ticketing service","documentation":{},"group":"OP Solutions","id":"14","name":"Freshdesk","properties":{"structurizr.dsl.identifier":"ss_ops_helpdesk"},"tags":"Element,Software System,Internal System"},{"containers":[{"description":"The Compass frontend application the users interact with","documentation":{},"id":"16","name":"SPA UI","properties":{"structurizr.dsl.identifier":"compass_frontend"},"relationships":[{"description":"calls","destinationId":"18","id":"25","sourceId":"16","tags":"Relationship,ComponentLevel","technology":"REST/HTTPS"}],"tags":"Element,Container,UI","technology":"JS/React"},{"description":"Primary relational data storage","documentation":{},"id":"17","name":"Database","properties":{"structurizr.dsl.identifier":"compass_db"},"tags":"Element,Container,Storage","technology":"MSSQL"},{"description":"Request-serving backend","documentation":{},"id":"18","name":"Compass API","properties":{"structurizr.dsl.identifier":"compass_api"},"relationships":[{"description":"notifies","destinationId":"16","id":"26","sourceId":"18","tags":"Relationship,ComponentLevel","technology":"SignalR"},{"description":"reads/writes data","destinationId":"17","id":"27","sourceId":"18","tags":"Relationship,ComponentLevel","technology":"TCP"},{"description":"saves/retrieves blobs","destinationId":"19","id":"28","sourceId":"18","tags":"Relationship,ComponentLevel","technology":"REST/HTTPS"},{"description":"saves/reads table rows","destinationId":"20","id":"29","sourceId":"18","tags":"Relationship,ComponentLevel","technology":"REST/HTTPS"},{"description":"indexes/searches the data","destinationId":"21","id":"30","sourceId":"18","tags":"Relationship,ComponentLevel","technology":"REST/HTTPS"},{"description":"sends emails","destinationId":"22","id":"31","sourceId":"18","tags":"Relationship,ComponentLevel","technology":"REST/HTTPS"}],"tags":"Element,Container","technology":"ASP.NET Core"},{"description":"Container for unstructured blob data","documentation":{},"id":"19","name":"Blob Storage","properties":{"structurizr.dsl.identifier":"compass_blobStorage"},"tags":"Element,Container,Storage","technology":"Azure Blob Storage"},{"description":"Container for denormalized tabular data","documentation":{},"id":"20","name":"Table Storage","properties":{"structurizr.dsl.identifier":"compass_tableStorage"},"tags":"Element,Container,Storage","technology":"Azure Blob Storage"},{"description":"Enables fast fuzzy searching for common queries","documentation":{},"id":"21","name":"Search engine","properties":{"structurizr.dsl.identifier":"compass_search"},"tags":"Element,Container,Storage","technology":"ElasticSearch"},{"description":"Enables communication through email\\SMS\\chats\\etc","documentation":{},"id":"22","name":"Communication Services","properties":{"structurizr.dsl.identifier":"compass_commServices"},"tags":"Element,Container,External System","technology":"Azure Communication Services"}],"description":"OPSolutions' EHR system","documentation":{"decisions":[{"content":"# 1. Record architecture decisions\n\nDate: 2024-01-22\n\n## Status\n\nAccepted\n\n## Context\n\nWe need to record the architectural decisions made on this project.\n\n## Decision\n\nWe will use Architecture Decision Records, as [described by Michael Nygard](http://thinkrelevance.com/blog/2011/11/15/documenting-architecture-decisions).\n\n## Consequences\n\nSee Michael Nygard's article, linked above. For a lightweight ADR toolset, see Nat Pryce's [adr-tools](https://github.com/npryce/adr-tools).\n","date":"2024-01-22T00:00:00Z","format":"Markdown","id":"1","status":"Accepted","title":"Record architecture decisions"},{"content":"# 10. Replace MediatR with Wolverine for in-process messaging\n\nDate: 2026-06-10\n\n## Status\n\nAccepted\n\nWork item [#4629](https://dev.azure.com/opsolutionsus/EHR/_workitems/edit/4629). Companion to [[0009-eventual-consistency-worker-service-rabbitmq-wolverine|0009]], which adopts Wolverine for distributed messaging over RabbitMQ.\n\n## Context\n\nThe backend uses MediatR 14 as its in-process mediator: ~83 `IRequest`/`IRequestHandler` pairs (the per-Part `Handlers/` convention — sealed records with nested Validator + Handler), ~130 `INotification` types, four pipeline behaviors (`ValidationPipelineBehavior`, `AccessControlPipelineBehavior`, `LoggingPipelineBehavior`, `ScopeLoggingPipelineBehavior`), a custom sequential notification publisher (`LoggingForeachAwaitPublisher`), and `IScopedPublisher` for scope-isolated publishing with user-context propagation.\n\nTwo forces make MediatR the wrong long-term home:\n\n1. **Licensing.** MediatR went commercial at v13; we run it under the Lucky Penny key embedded in `EHRv2.Backend.Parts/ServiceCollectionExtensions.cs`. Note this is *not* primarily a cost argument: the same key licenses AutoMapper, which we keep, so dropping MediatR alone does not end the subscription. The argument is dependency strategy — we don't want the core dispatch mechanism of the application on a commercial treadmill we don't control.\n2. **Two mediator models after [[0009-eventual-consistency-worker-service-rabbitmq-wolverine|0009]].** With Wolverine adopted for RabbitMQ messaging, keeping MediatR in-process means two handler conventions, two middleware/behavior pipelines, two testing stories, and a permanent seam where a notification's handlers are split across frameworks depending on where they execute.\n\nWolverine is natively both: an in-process mediator (no broker required for local handlers) and a message bus. Its core is MIT-licensed, with JasperFx's stated business model being paid support and commercial add-ons around an open core — the opposite trajectory of MediatR and MassTransit, which relicensed the core itself.\n\nAlternatives considered: keep MediatR alongside Wolverine (rejected — permanent dual model, ongoing commercial dependency); migrate request handlers to plain injected services (rejected — loses the uniform pipeline for validation/access control and the colocated handler convention the codebase is built around); MassTransit.Mediator (rejected with MassTransit overall, see [[0009-eventual-consistency-worker-service-rabbitmq-wolverine|0009]]).\n\n## Decision\n\n**1. Wolverine becomes the only mediator.** Request/response dispatch (`IMediator.Send`) and notification publishing (`IScopedPublisher.Publish`) both move to Wolverine's `IMessageBus`. Local handlers keep executing in-process with no broker involved; handlers designated for the worker ride RabbitMQ per [[0009-eventual-consistency-worker-service-rabbitmq-wolverine|0009]]. Whether a handler is local or remote becomes endpoint configuration, not a different programming model.\n\n**2. The four pipeline behaviors become Wolverine middleware**, preserving order and semantics: validation (FluentValidation — Wolverine has first-class integration), access control, logging, scope logging. This is the highest-risk porting step and is done once, centrally, before any handler migrates.\n\n**3. `IScopedPublisher` semantics are preserved by middleware, then the abstraction is retired.** Scope isolation comes from Wolverine's per-message scoping; user-context propagation moves to the envelope-header middleware defined in [[0009-eventual-consistency-worker-service-rabbitmq-wolverine|0009]] so it works identically for local and remote handlers. Call sites migrate from `IScopedPublisher.Publish` to `IMessageBus.PublishAsync`.\n\n**4. Handler conventions stay.** The per-Part layout (`Handlers/` with colocated request + validator + handler, `Messaging/Notifications/` for events) is unchanged; only the framework types underneath change (`IRequestHandler<TReq,TRes>.Handle` → Wolverine handler method). Notification records become Wolverine messages as-is; the existing base-class hierarchy (`PatientNotification`, `TreatmentNotification`, …) continues to give fan-out, since Wolverine supports publishing by base type the way MediatR multi-handler dispatch does today.\n\n**5. Migration is incremental, per Part.** MediatR and Wolverine coexist during the transition: new handlers are written as Wolverine handlers from day one; existing Parts migrate one at a time (smallest first to prove the middleware stack); MediatR and `LoggingForeachAwaitPublisher`/`ScopedPublisher` are deleted when the last Part is converted. No big-bang cutover.\n\n**6. AutoMapper is out of scope.** It stays on the Lucky Penny license. Any future decision to replace it is its own ADR.\n\n## Consequences\n\n**Positive**\n\n- One handler model, one middleware pipeline, one test harness across in-process and distributed messaging; moving a handler to the worker stops being a rewrite.\n- Core dispatch dependency returns to MIT-licensed open source with an explicit open-core commitment from its maintainer.\n- Wolverine's codegen produces leaner dispatch than reflection-based pipelines; pipeline behavior cost is paid at build time.\n\n**Negative**\n\n- ~83 request handlers and ~130 notification publishes to touch. Mechanical, but wide — per-Part migration with e2e coverage per slice is mandatory, not optional.\n- Wolverine's convention-based discovery and code generation are a real learning curve after MediatR's explicit interfaces; misconfigured conventions fail less obviously than a missing interface implementation. Mitigation: central middleware and conventions are established and reviewed once, before Parts migrate.\n- During the transition window, two mediators coexist — the per-Part migration order must be tracked visibly (work-item checklist) so the window stays short.\n- Smaller community than MediatR's historical one; JasperFx paid support exists if needed.\n\n**Follow-ups**\n\n- Prove the middleware stack (validation, access control, user-context headers) on one small Part before scheduling the rest.\n- Define the Wolverine handler conventions in the backend `CLAUDE.md` / contributor docs as soon as the first Part lands, so new code follows one pattern only.\n","date":"2026-06-10T00:00:00Z","format":"Markdown","id":"10","status":"Accepted","title":"Replace MediatR with Wolverine for in-process messaging"},{"content":"# 11. Environment-tagged container images are pushed by release pipelines, not CI\n\nDate: 2026-06-12\n\n## Status\n\nAccepted\n\nWork item [#4598](https://dev.azure.com/opsolutionsus/EHR/_workitems/edit/4598).\n\n## Context\n\nEHRv2 is moving to a two-environment delivery model: `develop` deploys to staging, `main` to production. Container images in the shared ACR (`ehrv2sharedcontainers`) were tagged with bare Azure DevOps build ids, so the deploy scripts (`infra/environment/`) could only ever pick \"the newest image\" — there was no way to deploy staging and production from different branches.\n\nThe delivery pipeline is split in two:\n\n- **CI (YAML, `azure_pipelines.yml` in each repo)** builds the app and publishes a build artifact (vite `build/` for frontend, `dotnet publish` output for backend, plus `pack.Dockerfile`).\n- **Classic release pipelines** (\"Publish Frontend Container\" / \"Publish Backend Container\", UI-defined) consume the artifact, build the image with the Container Build task against the `ehrv2sharedcontainers` service connection, and push to ACR.\n\nThe first implementation of #4598 moved the docker build-and-push into the CI YAML (`Docker@2 buildAndPush`), tagging images `<env>-$(Build.BuildId)`. That was rejected for two reasons:\n\n1. **The ACR service connection is not available to CI.** Authorizing the `acrServiceConnection` for YAML pipelines requires a secret/permission grant we cannot obtain.\n2. **It collapses release into CI.** Pushing a deployable image is a release act. Keeping it in release pipelines preserves the existing separation (CI = build + artifact, release = image + deploy gates) instead of duplicating release behavior inside build YAML.\n\n## Decision\n\n**1. CI stays build-only.** `azure_pipelines.yml` in backend/frontend triggers on `main` *and* `develop` and publishes the same build artifact as before. No registry credentials in CI.\n\n**2. Release pipelines own the image.** The classic \"Publish * Container\" release pipelines build and push the image, tagged `<env>-<buildId>` where `<env>` is derived from the artifact's source branch: `develop` → `staging`, `main` → `production`. The build id stays in the tag so the deployed commit remains resolvable through the Build API.\n\n**3. Deploys select images by environment prefix.** `getLatestImageTag(repository, env)` picks the newest `staging-*` / `production-*` tag; `main.mjs` requires `--environment=staging|production`; `tagDeployedCommit.mjs` strips the prefix before resolving the build to a commit.\n\n## Consequences\n\n**Positive**\n\n- Staging and production deploy independently from their own branches; an image's environment and originating build are readable from its tag.\n- No registry secrets enter CI; the release pipelines keep their existing service connection, approvals, and audit trail.\n\n**Negative**\n\n- The environment-prefix mapping lives in UI-defined release pipelines (a Bash task setting `envPrefix` from the artifact's source branch), which is not version-controlled. The CI YAML carries a comment documenting the contract.\n- Deploys fail loudly until at least one `<env>-` tagged image exists per repository — old bare-build-id tags are ignored. The release pipelines must run once per environment before the new deploy scripts are used.\n\n**Follow-ups**\n\n- Update the release pipelines per the work-item guideline (branch filter for `develop`, `envPrefix` Bash step, tag field `$(envPrefix)-$(Build.BuildId)`).\n- `wiki/Development/Deployments.md` still describes staging deploys as driven by merges to `main`; update it once the develop=staging flow is live.\n","date":"2026-06-12T00:00:00Z","format":"Markdown","id":"11","status":"Accepted","title":"Environment-tagged container images are pushed by release pipelines, not CI"},{"content":"# 12. Dependent patients: shared login via ParentPatientId, and overridable duplicate detection\n\nDate: 2026-06-24\n\n## Status\n\nAccepted\n\n## Context\n\nWork item [#4244](https://dev.azure.com/opsolutionsus/EHR/_workitems/edit/4244) surfaced two\nrelated problems with patient registration:\n\n1. **False duplicates.** Patient creation hard-blocks when a new patient matches an existing one on\n   `name+DOB` **or** `email` **or** `phone`. Some of these matches are legitimately different people\n   (a shared household phone, a common name). Staff had no way to say \"I checked — this is a different\n   person, create them anyway.\"\n\n2. **Dependents without their own email.** A child (or other dependent) is registered with a parent's\n   email. Jason asked whether we could simply allow duplicate emails. We **cannot**: in EHRv2 the email\n   *is* the login identity. `EHRUserRegistryService` sets `UserName = email`\n   (`backend/EHRv2.Auth/Services/EHRUserRegistryService.cs`), `NormalizedUserName` is uniquely indexed,\n   and the registry explicitly rejects a second active user on the same email. A second independent\n   account on a shared email is therefore impossible by construction — and that constraint is\n   *desirable* (users sign in by email).\n\nThe naïve \"relax the dedup check\" or \"make email non-unique\" options were both rejected: the first\nlets real duplicates through, the second breaks login identity. We needed a model that lets a dependent\n**reuse** a parent's contact email and portal access without minting a second login.\n\nA sibling option — changing `Person ↔ User` from one-to-one to many-to-one so several patients could\nhang off one user — was considered and rejected: it touches a load-bearing Identity relationship,\nripples through every place that resolves \"the user's patient,\" and buys nothing over a narrower link.\n\n## Decision\n\n**1. A dependent is a patient linked to a parent patient via `Patient.ParentPatientId` (nullable\nself-FK), and has no `EHRUser` of its own.** The parent's `Person` owns the login; the dependent's\n`Person` carries the parent's email as *contact only*. One level — a dependent is never itself a\nparent; the `CreatePatient` validator enforces this explicitly by requiring the link target to have a\nuser **and** a null `ParentPatientId`. (`backend/EHRv2.Database/Entities/Patients/Patient.cs`.)\n\n**2. Duplicate detection becomes typed and overridable, reusing the existing\n`IWarnableRequest` / `IgnoreWarnings` mechanism** (`ValidationPipelineBehavior`). `CreatePatient` now\nsplits the single collapsed failure into:\n\n- **`PatientPossibleDuplicate`** (`Severity.Warning`) for a `name+DOB` or `phone` match — overridable\n  by resubmitting with `IgnoreWarnings=true` (\"create anyway\").\n- **`PatientEmailInUse`** (`Severity.Error`, **not** a warning) for an email match against an existing\n  *patient* that owns a user — **not** bypassable by `IgnoreWarnings`. It can only be resolved by\n  linking as a dependent (resubmit with `ParentPatientId`) or by changing the email. `CustomState`\n  carries the existing patient so the UI can offer the link.\n\nWhen `ParentPatientId` is set, dedup is skipped entirely (the link is an explicit, intentional\ndecision) — the validator only verifies the parent exists and has a user.\n\n**3. The dependent's user is simply not provisioned.** The create handler guards user creation with\n`patient.Person.Emails.Any() && r.Patient.ParentPatientId is null`. The parent's email is stored as the\nchild's contact; no `EHRUser`, no credentials are returned.\n\n**4. Portal authorization treats the parent's user as the dependent's *effective owner*.** Patient-portal\nrequests are scoped by `PatientScopeResolver`, which resolves the owning user **from the database** (so a\nclient-supplied `patientId` can't be spoofed) and `AccessChecker` compares it to the current user. Since\na dependent's `Person.User` is null, the resolver now falls back to\n`patient.ParentPatient.Person.User?.Id`. This makes the *existing* equality check authorize a parent for\ntheir dependent's data **without touching `AccessChecker`**; unrelated patients still get 403.\n\n**5. The patient portal supports a user owning several patients.** A new\n`GET …/patients/byUserId/{userId}/all` returns the user's own patient plus its dependents; the portal\nrenders a patient switcher (shown only when >1) and scopes the active patient via a selection store. The\nhandler resolves the **authenticated caller**, not the route `userId`: the request carries no `IHasPatientId`,\nso the access pipeline scopes it to the caller (it can't bind a route id to the caller), and trusting the route\nid would let a portal user read another user's patients. System admins (who bypass the access check) keep the\nexplicit by-user lookup. This upholds the same \"owner from the database, not from a client-supplied id\"\nprinciple as decision 4.\n\n## Consequences\n\n**Positive**\n\n- Login identity stays intact — email remains unique-per-account; no schema change to Identity.\n- Real duplicates are still caught; staff get a deliberate, auditable override for the soft cases.\n- Dependents reuse a parent's addressing/login with a single nullable FK and no new entity.\n- Portal authorization extends to dependents with a one-line change in the resolver — the database-first\n  ownership check (and its spoofing protection) is preserved.\n\n**Negative**\n\n- `ParentPatientId` conflates two ideas — \"shares the parent's contact info\" and \"portal access is owned\n  by the parent.\" Acceptable today; if they need to diverge, model them separately later.\n- One level of linking only. Grandparent→parent→child chains are not resolved; the portal list and the\n  effective-owner fallback look exactly one hop up. Revisit if multi-level households appear.\n- A dependent who later gets their own email needs a \"promote to own account\" flow (create an `EHRUser`,\n  clear `ParentPatientId`). Out of scope for #4244 — noted as follow-up.\n- The email-in-use branch only offers linking when the email belongs to an existing **patient**. If the\n  email belongs to a non-patient user (e.g. a clinician), it stays a plain hard error (\"choose another\n  email\"), because there is no patient to point `ParentPatientId` at.\n\n## How to apply\n\n- **Linking on create:** send `CreatePatientDto.ParentPatientId`. The handler skips user creation and\n  dedup; the validator checks the parent exists and has a user.\n- **Overriding a soft duplicate:** resubmit `CreatePatient` with `IgnoreWarnings=true`\n  (`POST …/patients/{id}?ignoreWarnings=true`). This bypasses `PatientPossibleDuplicate` but **never**\n  `PatientEmailInUse`.\n- **Authorizing dependent data on the portal:** nothing extra per-endpoint — `PatientScopeResolver`\n  resolves the effective owner. When adding a new patient-scoped resolver branch, `Include` the\n  `ParentPatient.Person.User` and use the `EffectiveOwnerUserId` helper.\n- **Automated/seed patient creation** (OpenPractice sync, data seeders, test seeders) passes `IgnoreWarnings: false` — i.e. it does\n  **not** override duplicate warnings. Seed data is unique, so it never trips the soft-duplicate check (the\n  behaviour predates this change); keeping `false` avoids silently masking a genuine collision introduced by a\n  seed. Email collisions hard-fail regardless of the flag.\n\n## Related\n\n- Backend: `backend/EHRv2.Backend.Parts/MediatR/Handlers/CreatePatient.cs`,\n  `backend/EHRv2.Backend.Parts/Access/PatientScopeResolver.cs`,\n  `backend/EHRv2.Backend.Parts.Patients/Handlers/GetPatientsByUserId.cs`.\n- Frontend: `frontend/src/pages/patient/trees/create/index.tsx` (warning/link flow),\n  `frontend/src/pages/patientPortal/layout/PatientPortalRoot.tsx` + `components/PatientSwitcher.tsx`.\n- Why email is effectively unique (the load-bearing constraint): `EHRUserRegistryService` —\n  `UserName == Email` + unique `NormalizedUserName`. Captured in the wiki under `Development/`.\n","date":"2026-06-24T00:00:00Z","format":"Markdown","id":"12","status":"Accepted","title":"Dependent patients: shared login via ParentPatientId, and overridable duplicate detection"},{"content":"# 13. Local kind environment: run the real Helm chart locally, emulate Azure services in Docker\n\nDate: 2026-07-23\n\n## Status\n\nAccepted\n\n## Context\n\nUntil now \"local dev\" was a docker-compose of dependencies\n(`backend/docker/dev-services.docker-compose.yaml`) plus a host-run backend and the Vite dev\nserver. Nothing about it was Kubernetes-shaped, so local conditions diverged from the real AKS\nenvironments in exactly the places that bite late: ingress path routing, in-cluster DNS,\nmulti-replica statelessness, the Azure-Table-based config flow, and OpenSearch's cert-manager TLS.\nThere was no way to run integration/e2e/stress suites against a prod-shaped stack without pointing\nthem at a real (shared) environment.\n\n## Decision\n\n`infrastructure/environment/local/` provides a **parity environment** alongside (not replacing)\nthe fast loop:\n\n- The **whole app stack runs in a local kind cluster from the same `opsehr-chart`** used by\n  staging/production. `values.local.yaml` overrides only images (locally built, `kind load`-ed),\n  resources, storage class, and ingress class (ingress-nginx instead of AGIC). The chart was\n  parameterized for these knobs with rendered output for real environments unchanged.\n- **Managed Azure services stay outside the cluster**, as in real envs, emulated by Docker\n  containers: Azure SQL → `mssql/server:2022`, Azure Storage → Azurite. Backend configuration\n  flows through the same mechanism as real deploys: one `app-config` secret\n  (`AppConfigurationSource`) pointing at a storage account whose `config` table\n  (PartitionKey=`backend`) is seeded by a local variant of\n  `installSecretsIntoStorageAccountTable.mjs`.\n- Tests target it via `ENV_URL=http://localhost:31080` (e2e) / `BaseUrl` (integration); logs are\n  captured per-pod by `logs.mjs` (CI-artifact-friendly dump mode).\n- The environment uses a **self-contained kubeconfig** (`environment/local/.kubeconfig`); every\n  kubectl/helm call pins it plus `--context kind-ehrv2-local`, so the tooling cannot touch a real\n  AKS context.\n\nSee `infrastructure/environment/local/README.md` for the full parity map and usage.\n\n## Non-obvious findings (why some of it looks the way it does)\n\n1. **cert-manager's default private-key encoding breaks current OpenSearch images.** The chart's\n   `opensearch-tls` Certificate previously inherited cert-manager's default **PKCS#1** encoding;\n   the security plugin in current `opensearchproject/opensearch:2` images only parses **PKCS#8**\n   and crash-loops at boot. The Certificate now pins `privateKey.encoding: PKCS8`. This is a\n   latent bug for real environments too: any AKS cluster that re-pulls a newer `opensearch:2`\n   image (node replacement, pod reschedule) would hit it on the next cert issuance. Note that\n   cert-manager does **not** re-issue an existing secret when only the encoding changes — delete\n   the `opensearch-tls` secret to force a re-issue.\n2. **The Azure Blob SDK silently corrupts blob URIs for non-IP hostnames.** With a path-style\n   endpoint like `http://azurite:31100/devstoreaccount1`, `BlobUriBuilder` treats the dotless\n   hostname as host-style (account-in-host) and **drops the container segment** when building\n   blob URIs (`keys/backend` → `backend`), producing 400/404s only on blob-level operations —\n   container-level calls append paths and work, which makes the failure look impossible. This is\n   why the local storage connection strings use the Docker host-gateway **IP** (IPs get correct\n   path-style handling) while SQL keeps a clean `mssql` ExternalName alias (TDS does not parse\n   URLs).\n3. **First-boot seeding races between backend replicas are expected.** With `replicas: 2`, both\n   pods run migrations + seeding concurrently on a fresh database (same as real non-prod envs);\n   the loser exits on a unique-key violation and passes on restart once the winner finishes.\n\n## Consequences\n\n- Real-env-shaped bugs (ingress routing, config-table flow, multi-replica behavior, OpenSearch\n  TLS) reproduce locally before they reach staging.\n- Stress/load tests can run locally after restoring the real-env resource values in\n  `values.local.yaml` (the committed defaults are laptop-sized).\n- The fast inner loop is untouched; both stacks run concurrently (31xxx port split, separate\n  volumes/networks).\n- The chart is now values-driven for ingress class/annotations, storage class, resources, and\n  imagePullSecrets — future environments (e.g. a CI cluster) need only a values file.\n","date":"2026-07-23T00:00:00Z","format":"Markdown","id":"13","status":"Accepted","title":"Local kind environment: run the real Helm chart locally, emulate Azure services in Docker"},{"content":"# 14. Private AI node for PHI-bearing generation; public-data AI pipelines stay on standard hosted inference\n\nDate: 2026-08-06\n\n## Status\n\nProposed\n\n## Context\n\nWork item [#4914](https://dev.azure.com/opsolutionsus/EHR/_workitems/edit/4914) (\"AI - Features\nin Compass\") is the umbrella story for bringing AI into Compass, with three child stories:\n[#4915](https://dev.azure.com/opsolutionsus/EHR/_workitems/edit/4915) (Device Recommendation),\n[#4941](https://dev.azure.com/opsolutionsus/EHR/_workitems/edit/4941) (Ambient voice recognition),\nand [#4945](https://dev.azure.com/opsolutionsus/EHR/_workitems/edit/4945) (Generate documents).\nOn #4945, Ivan Yeremenko commented: *\"thats definitely a job for a private AI node since we need\nto use PHI to build docs.\"* That comment is correct but under-specifies which parts of the AI\ninitiative it applies to, and the workspace currently has no infrastructure pattern to hang a\n\"private AI node\" on. This ADR draws the line and records the constraint.\n\n**Not every AI workload here touches PHI.** The build plan attached to #4915\n(`00_build_plan.md`, `01_evidence_extraction_prompt.md`, `02_payer_rule_extraction_prompt.md`)\ndescribes two pipelines that are deliberately designed to never see patient data:\n\n- **Evidence database pipeline** — extracts structured, HCPCS-mapped benefit statements from\n  peer-reviewed literature (PubMed/Europe PMC/OpenAlex). Input is public journal text; output is a\n  shared reference database, not a patient document.\n- **Payer-rule database pipeline** — extracts machine-readable coverage criteria from public CMS\n  LCDs, Policy Articles, and payer medical policies. Input is public policy text.\n\nBoth pipelines explicitly forbid the model from retrieving or inferring facts not present in the\nsupplied document, and route every extraction through a code-enforced verbatim-quote check before\npublish — a sound pattern, and one that is only viable because the source documents are public.\n\n**The PHI-bearing work is narrower and specific:** the device-recommendation narrative generation\nin #4915 §1 (which prompts on the patient's Evaluation, outside evaluations, referring physician's\nclinical notes, and outcome-measure summaries to produce insurance-reviewer and patient\nnarratives) and all of #4945 (narrative summaries, \"Dear Physician\" letters, HCPCS justifications —\nall built from \"the prosthetic evaluation\"). #4941's ambient-voice concept is PHI-adjacent by\nnature (it listens to a clinician-patient encounter) but is scoped today as integrating a\nthird-party vendor (O&P Assist, Claim Grade, Ever Labs) rather than building in-house inference, so\nthe PHI handling boundary there is a vendor/BAA question, not an infra question — out of scope for\nthis ADR.\n\n**A planning call between Jason Kahle and Taras Hrytsenko sharpened this scope further:**\n\n- **Evidence-database pipeline — confirmed never PHI, by design, permanently.** Per Jason's review\n  of this ADR: it is not currently, and will never be, linked to PHI — it operates on HCPCS-related,\n  generalized population-level evidence, not patient-specific data. The build approach has changed,\n  though: rather than a fully in-house extraction pipeline, the solution is a **combination of\n  continued manual curation and Scite.ai** (a $40/month third-party literature-analysis service,\n  itself Claude-backed, that produces benefit-statement-plus-citation output) — not a full\n  replacement of the manual process, a supplement to it. This doesn't change the PHI classification\n  — it's a build-vs-buy refinement.\n- **Payer-rule pipeline — confirmed never PHI, by design, permanently.** Per the same review: not\n  currently, and will never be, linked to PHI. Its scope is the O&P specialty (DMEOPS), device\n  type, specific device components, HCPCS codes, and HCPCS reimbursement/cost — payer policy and\n  cost data, not patient records. Approach remains otherwise unresolved and exploratory.\n- **#4941's scope was finalized and is more specific than \"vendor integration.\"** OPS will not\n  build in-house ambient listening or AI note-generation for at least the next ~6 months. Instead\n  it's running a pilot with three third-party vendors — O&P Assist, Claim Grade, and Ever Labs —\n  who own that PHI-touching workload on their own infrastructure; Compass's own build is limited to\n  (a) a guided clinical-question \"scrolling prompt\" script, usable standalone or to drive the vendor\n  tools, and (b) a basic first-party voice-to-text side panel for simple dictation. This is a\n  deliberate, time-boxed business decision with an **acknowledged but not resolved** privacy\n  tradeoff: Taras raised that vendor-hosted processing means loss of control over PHI once it\n  leaves Compass's infrastructure, and no BAA (or equivalent data-processing agreement) was\n  discussed or confirmed with any of the three vendors.\n- The core patient-data narrative-generation workload was reconfirmed as PHI-dependent **by\n  design**, not by omission — Jason: *\"If we don't feed PHI in it, we've got nothing.\"* Insurance\n  reviewers require patient-specific facts, not population statistics.\n- A concrete mitigation was proposed and agreed in principle on the same call: strip direct patient\n  identifiers (name, SSN) before constructing any prompt, send the model only de-identified\n  clinical content keyed to a placeholder, then substitute real identifiers back into the model's\n  output afterward in a separate, non-AI step. This is folded into the Decision below as a required\n  control — with a caveat that name+SSN alone is far short of full HIPAA Safe Harbor\n  de-identification (18 identifier categories, including DOB, dates, geographic subdivisions, MRNs,\n  and more), so the exact field list needs compliance sign-off before it's built.\n\n**Work item [#4950](https://dev.azure.com/opsolutionsus/EHR/_workitems/edit/4950) (\"PHI - Claude\nBlueprint\") is the concrete BAA/de-identification/cost research this ADR's vendor decision now\nleans on:**\n\n- **Only two Claude products are BAA-eligible: Claude Enterprise (chat, seat-based) and the Claude\n  Platform/API (usage-based).** Claude.ai's consumer plans (Free/Pro/Max) cannot get a BAA at all.\n  Enterprise fits a workflow of staff manually pasting de-identified summaries into a chat UI; the\n  API fits the actual repeatable \"feed clinical data, compare to payer rules, generate output\"\n  pipeline this initiative needs — which is the more natural end state here.\n- **The BAA does not cover everything, even on a covered plan.** MCP connectors, Web Search/Web\n  Fetch, the Batch API, the Files API, Code Execution, Computer Use, and beta features are all\n  explicitly carved out of BAA protection. This directly conflicts with one of #4950's own\n  cost-saving suggestions (Batch API gets a 50% discount) — **whichever inference path is chosen\n  must avoid these carve-outs for any call that carries real PHI**, which needs confirming with\n  Anthropic Sales rather than assumed either way.\n- **Cost is not expected to be the blocker.** A worked example in #4950: ~3,500 input + 500 output\n  tokens per claim evaluation on Sonnet 5's introductory rate comes to roughly $0.012/evaluation —\n  about $12/month at 1,000 evaluations/month, before prompt caching (payer rule sets are static and\n  reused across patients — exactly the shape caching is built for, cutting cached-context cost by\n  ~90%). Claude Enterprise's $20/seat/month with a 20-seat minimum ($400/month floor) would likely\n  dominate total cost over token usage at OPS's expected volume either way. This significantly\n  de-risks the \"sign a BAA with Anthropic\" path on cost grounds specifically — the open question per\n  Jason is now the *non-cost* tradeoffs against self-hosting (latency, data residency, vendor\n  dependency), not price.\n- **De-identification method confirmed: HIPAA Safe Harbor**, matching decision 5 below — the same\n  18-identifier list. #4950 adds two concrete techniques: use an internal token/case ID that is\n  **not derived from** any real identifier, and represent dates as relative offsets (\"day 3 of\n  treatment\") rather than absolute dates wherever the payer rule doesn't specifically require an\n  absolute date. Where a payer rule genuinely needs an absolute date or another Safe Harbor\n  identifier, that case falls back to the BAA-covered path rather than de-identification alone —\n  the two controls are complementary, not either/or.\n- **Operational caution, unrelated to Compass's own architecture but worth recording:** #4950 itself\n  notes that \"Cowork\"-style conversational Claude sessions — the kind used to produce #4950's own\n  analysis — are excluded from BAA coverage in every configuration, regardless of the org's plan.\n  Any exploratory research on this initiative (future blueprint sessions, prompt experiments, etc.)\n  should stick to synthetic or de-identified data for that reason, independent of what gets built\n  into Compass itself.\n\n**Current production infra has no isolation pattern to reuse for this.** Inspecting\n`ehrv2-production2` (resource group) and its AKS cluster `ehrv2-production2-aks` directly:\n\n- The cluster is a **single system node pool, one node** (`Standard_D4ds_v4`), **kubenet**\n  networking, **`networkPolicy: none`**, and is **not** a private cluster (no restricted API\n  server access).\n- Every application pod — `backend-deployment`, `frontend-deployment`, `opensearch-0`,\n  `opensearch-dashboards`, `redis-0`, cert-manager — runs in the single `default` namespace.\n  There is no namespace or network-policy boundary between any two workloads today.\n- `ehrv2production2sqlserver` has `publicNetworkAccess: Enabled` with firewall rules that allowlist\n  specific developer IPs plus the legacy `AllowAllWindowsAzureIps` (`0.0.0.0`–`0.0.0.0`) rule — no\n  Private Link/private endpoint.\n- The storage account disables anonymous blob access but is not otherwise network-restricted.\n- Gatekeeper (OPA) is deployed for policy admission, but with `networkPolicy: none` at the CNI\n  level it cannot enforce pod-to-pod network segmentation — only admission-time policy.\n\nIn short: there is no existing \"trusted zone\" in this cluster to place a PHI-handling AI workload\ninto. Building the private AI node Ivan describes is net-new infrastructure work, not a config\nchange to something already segmented.\n\n## Decision\n\n**1. AI workloads are classified by whether PHI enters the prompt, not by feature name.**\n\n- **Public-data pipelines** (payer-rule extraction, and any future pipeline whose input is limited\n  to literature/regulatory/public-policy text — evidence-database work is in this same\n  never-touches-PHI category but is handled by manual curation plus a Scite.ai subscription rather\n  than an in-house pipeline, see Context) may call a standard hosted LLM API directly. No patient\n  identifier or clinical detail is ever constructed into\n  these prompts by design — enforced by the existing verbatim-substring validation in the build\n  plan, which fails the record if the model asserts anything not present in the supplied public\n  document.\n- **Patient-data workloads** — today: #4915's narrative generation, and all of #4945 — must never\n  send PHI to a public/multi-tenant hosted inference endpoint. These route exclusively through the\n  private AI node defined below.\n\n**2. The private AI node is a dedicated, network-isolated inference path**, not a flag on the\nexisting backend deployment. At minimum:\n\n- A separate node pool (or separate compute outside the shared `agentpool`), tainted so only the\n  AI-inference workload schedules there.\n- No inbound path from the public internet; reachable only from the backend's PHI-handling code\n  path, not from `frontend` directly.\n- Egress limited to the approved inference endpoint (self-hosted model, or a hosted model reached\n  under a signed BAA with private connectivity — e.g. Private Link to a single-tenant deployment).\n  No default internet egress.\n- Every PHI-bearing prompt/response pair logged for compliance audit, separately from application\n  logs, with retention aligned to the practice's compliance requirements.\n\nThis requirement applies to AI processing **OPS builds and hosts itself**. Where PHI-bearing\nprocessing is instead delegated to a third-party vendor — as #4941's ambient-voice/note-generation\npipeline now is — the governing control is a **signed BAA (or equivalent data-processing agreement)\nplus a vendor security review**, not this ADR's infra pattern. The non-negotiable is the same\neither way: PHI does not reach any party without an appropriate compliance agreement in place.\nJason is now confirming BAA status directly with O&P Assist, Claim Grade, and Ever Labs — tracked\nin Follow-ups.\n\n**The current working plan for \"the approved inference endpoint\" is a signed BAA with Anthropic**\n(Claude Enterprise or the Claude Platform/API — see #4950), not self-hosting by default. Whichever\nof the two BAA-eligible products is chosen, the specific API surface used for any PHI-bearing call\nmatters: MCP connectors, Web Search/Web Fetch, the Batch API, the Files API, Code Execution,\nComputer Use, and beta features are all carved out of BAA coverage even on an otherwise-covered\nplan. The AI node's egress and integration design (above) needs to route PHI-bearing calls only\nthrough the specific covered surface, not incidentally through one of these excluded features for\nconvenience or cost (Batch API's discount, in particular, is not usable for PHI without confirming\notherwise with Anthropic Sales).\n\n**3. Enforcing network isolation requires an AKS network-plugin change first.** The cluster's\ncurrent `kubenet` + `networkPolicy: none` configuration cannot enforce any NetworkPolicy — kubenet\nonly supports policy enforcement via Calico, which is not currently installed. Standing up the\nprivate AI node's isolation therefore has a prerequisite: install and validate a NetworkPolicy\nprovider (Calico, compatible with the existing kubenet plugin) before the AI node pool is\nschedulable, and cover it with a policy that denies-by-default and allows only the specific\nbackend → AI-node path.\n\n**4. Vendor/model selection is still open, but cost is very unlikely to be the deciding factor.**\nThis ADR fixes the *boundary* (PHI never reaches a non-private endpoint) and the *infra shape*\n(isolated node pool, restricted egress, audited), not the specific model host. #4950's cost\nmodeling puts the BAA-covered Claude API at roughly $12/month at 1,000 evaluations/month on\nSonnet 5 (before prompt caching, which fits this workload well since payer rule sets are static and\nreused across patients) — cheap enough that Jason does not expect price to block this decision. The\nremaining open question, per Jason, is the *non-cost* tradeoffs of self-hosting vs. the BAA-covered\nAPI (latency, data residency, vendor dependency, model-quality/maintenance burden) — that comparison\nis a follow-up, informed by the practice's compliance counsel review (per the build plan's\nguardrail: compliance/DMEPOS-audit review of note-generation logic before it touches a live claim).\n\n**5. De-identify before generation, re-identify after — required in addition to the private node\n(and its BAA), not instead of either.** For the in-house patient-data path (#4915 §1, #4945):\nauto-generate a de-identified version of the source data directly from the Patient O/P evaluation\n— Jason and #4950 both independently landed on this as \"very doable\" — strip identifiers per HIPAA\n**Safe Harbor** (the confirmed applicable method; 18 categories — name and SSN alone are nowhere\nnear sufficient, also DOB, all dates but year, geographic subdivisions smaller than state, MRNs,\nand more), key the de-identified content to an internal token/case ID **not derived from** any real\nidentifier, and represent dates as relative offsets (\"day 3 of treatment\") rather than absolute\ndates wherever the payer rule permits it. Substitute real identifiers back into the model's output\nafterward in a separate, non-AI step that never touches the model. This is a defense-in-depth layer\non top of the BAA-covered/private-node path, not a substitute for it — the de-identification step\nitself operates on raw PHI and must run somewhere already trusted with it, and any case that\ngenuinely needs an absolute date or another Safe Harbor identifier (a payer rule with a hard\ntiming requirement, for instance) falls back to relying on the BAA-covered path rather than\nde-identification alone. The exact field list still needs compliance sign-off before this is\nimplemented — Safe Harbor gives the checklist, not automatic clearance for OPS's specific data.\n\n## Consequences\n\n**Positive**\n\n- The payer-rule pipeline can move forward immediately on standard hosted inference, and the\n  evidence-database work proceeds via manual curation plus Scite.ai — both carry no PHI risk, so\n  neither is blocked waiting on private-node infrastructure.\n- Drawing the line at \"does PHI enter the prompt\" rather than \"which work item\" keeps the\n  classification robust as scope evolves — a future feature added to #4915 or elsewhere is\n  automatically routed by the same test.\n- Recording the current infra gaps (kubenet without Calico, single node pool, no namespace\n  segmentation, SQL public network access) now means the private-AI-node build-out and the\n  cluster's broader network-hardening can be planned as one piece of infra work instead of being\n  discovered piecemeal.\n\n**Negative**\n\n- This is genuinely new infrastructure, not a config toggle — expect a dedicated node pool, a\n  NetworkPolicy provider install and validation pass, and new Helm chart templates in\n  `infra/environment/app/templates/`. Sizing and cost are not yet estimated.\n- Until the network-plugin prerequisite (Calico) lands, no PHI-bearing AI workload should be\n  deployed even experimentally in this cluster — there is no way to guarantee its egress or\n  pod-to-pod isolation today.\n- The vendor/model decision (self-hosted vs. BAA-covered Claude) is still open on non-cost grounds,\n  which means #4945 and #4915 §1 cannot move past prototype until that follow-up decision is made.\n- The de-identify/re-identify control (decision 5) adds a tokenization-and-resubstitution layer on\n  top of the network-isolation work already scoped — more to build and test, not less.\n\n**Follow-ups**\n\n- Select and document the private-node model host (self-hosted vs. BAA-covered Claude\n  Enterprise/API) — new ADR when decided. Per Jason, this is now mainly a non-cost tradeoff\n  comparison (latency, data residency, vendor dependency) since #4950 shows cost is unlikely to be\n  the deciding factor — he's gathering that comparison.\n- Confirm with Anthropic Sales: which specific models are on the current Covered Models list for\n  the chosen BAA product, and whether prompt caching and/or the Batch API's discount are actually\n  usable for PHI-bearing calls under the BAA — #4950 flags Batch API as a general BAA carve-out,\n  which conflicts with its own cost-saving suggestion, so this needs a direct answer rather than an\n  assumption either way.\n- Install a NetworkPolicy provider (Calico) on `ehrv2-production2-aks` and validate default-deny\n  before the AI node pool is created.\n- Design the Helm chart addition for the AI node pool + backend integration path in\n  `infra/environment/app/`.\n- Revisit SQL Server public network access / firewall-by-IP as part of the same network-hardening\n  pass — not directly required for this decision, but the same audit that found it is relevant to\n  any future PHI-in-transit hardening.\n- Compliance defines the exact PHI fields to strip/tokenize for the de-identify/re-identify control\n  (decision 5) before it's implemented — do not ship based on an engineering guess at the field\n  list.\n- **In progress:** Jason is confirming BAA (or equivalent data-processing agreement) status with\n  O&P Assist, Claim Grade, and Ever Labs before real patient audio/notes flow through their systems\n  during the #4941 pilot.\n- Evidence-database pipeline solution is manual curation plus a Scite.ai subscription, not a full\n  in-house build; payer-rule pipeline remains exploratory — update scope tracking accordingly.\n\n## How to apply\n\n- Building the payer-rule-database pipeline from #4915: use a standard hosted LLM API call. No\n  private-node work is a blocker. (The evidence-database piece is manual curation plus a Scite.ai\n  subscription, not an in-house build — nothing to implement here.)\n- Building anything that constructs a prompt containing Evaluation data, clinical notes, outcome\n  measures, or any other patient-identifiable content (#4915 §1's narrative generation, all of\n  #4945): do not implement against a public hosted endpoint. This is blocked on the private AI\n  node landing — flag it rather than working around it with a \"temporary\" hosted call.\n- If a new AI feature is proposed anywhere in Compass, ask \"does the prompt contain PHI\" first;\n  that answer picks which of the two paths above applies.\n\n## Related\n\n- The evidence-pipeline build-vs-buy change, #4941's finalized scope, and the initial\n  de-identification mitigation idea (decision 5) were surfaced in a planning call between Jason\n  Kahle and Taras Hrytsenko, not in any written source at the time — captured here for the first\n  time. The BAA specifics, carve-outs, cost modeling, and confirmed Safe Harbor method are written\n  up in [#4950](https://dev.azure.com/opsolutionsus/EHR/_workitems/edit/4950).\n- [#4914](https://dev.azure.com/opsolutionsus/EHR/_workitems/edit/4914),\n  [#4915](https://dev.azure.com/opsolutionsus/EHR/_workitems/edit/4915),\n  [#4941](https://dev.azure.com/opsolutionsus/EHR/_workitems/edit/4941),\n  [#4945](https://dev.azure.com/opsolutionsus/EHR/_workitems/edit/4945),\n  [#4950](https://dev.azure.com/opsolutionsus/EHR/_workitems/edit/4950).\n- Build plan attached to #4915: `00_build_plan.md`, `01_evidence_extraction_prompt.md`,\n  `02_payer_rule_extraction_prompt.md` (Azure DevOps attachments on that work item).\n- Infra conventions: `infra/CLAUDE.md` (\"Mandatory patterns\"), particularly that the Helm chart in\n  `infra/environment/app/` is the source of truth for in-cluster topology and that Bicep does not\n  touch in-cluster resources — the AI node pool's cluster-level definition (Bicep) and its\n  in-cluster deployment (Helm) will need to be split accordingly.\n","date":"2026-08-06T00:00:00Z","format":"Markdown","id":"14","status":"Proposed","title":"Private AI node for PHI-bearing generation; public-data AI pipelines stay on standard hosted inference"},{"content":"# 15. Move CI/CD from classic Release pipelines to pipeline-as-code\n\nDate: 2026-08-15\n\n## Status\n\nAccepted\n\n## Context\n\nCI/CD across EHRv2 is split: build automation is version-controlled YAML (`azure_pipelines.yml`\nin backend/frontend/e2e), but every actual release/deploy step still runs through classic,\nUI-only Azure DevOps Release definitions - architecture's Structurizr DSL push, backend's and\nfrontend's container image push, and infra's shared-resources/staging/production deploys. None\nof that is checked into any repo.\n\nThis has the same drawbacks any hand-configured, non-declarative infrastructure has, independent\nof what it happens to deploy:\n\n- **Not version controlled.** No diff, no code review, no history beyond ADO's own UI-side audit\n  log - a change to a release definition is invisible until something breaks.\n- **Not declarative or reproducible.** State lives entirely in the ADO UI and can't be recreated\n  from source. It also drifts silently: release definition 15's `id`/`api`/`key`/`secret`\n  pipeline variables turned out to be plain (non-secret) variables, printing the real Structurizr\n  API key and secret in cleartext in every run's log - discovered only by reading a failure log\n  years after the fact, not by anyone reviewing a diff.\n- **No pre-merge validation.** A classic Release only runs _after_ a merge, by construction, so\n  the earliest anyone finds out something is broken is in production.\n\nThat last point is what turned this from a standing weakness into an actual incident: release\nrun 11822 failed the \"Publish Structurizr DSL to https://architecture.opsehr.com\" stage because\ntwo independently-merged PRs each added an ADR file numbered `0013-*.md`. Nothing validated this\nbefore merge - there was no PR-time check to catch it. This is one concrete instance of the\ngeneral problem above, not the whole reason for this decision: the same weaknesses apply equally\nto backend/frontend's image push and infra's environment deploys, whether or not any of them has\nbroken yet.\n\n## Decision\n\nMove CI/CD off classic Release pipelines onto pipeline-as-code (YAML) for every repo that has\none, under one policy: **a merge to a repo's production branch validates AND deploys; every\nother branch (PRs) only validates.** The benefits are the same ones that motivate infrastructure-\nas-code generally - versionability (diffable, reviewable, revertible), declarativeness\n(reproducible from source, no click-ops drift), and ease of modification (a config change is a\nPR like any other, not a manual edit to a UI form) - not specific to Structurizr or to this repo.\n\nTracked as Feature #4993 (\"Transition classic Release pipelines to CI/CD-as-code\"), with one\nchild Task per repo that currently has a classic Release definition:\n\n| Repo         | Release def(s)                                        | Task                                  |\n| ------------ | ----------------------------------------------------- | ------------------------------------- |\n| architecture | 15 (Structurizr DSL push)                             | #4994 - converted now, in this PR     |\n| backend      | 7 (Publish Backend Container)                         | #4995 - investigate first (see below) |\n| frontend     | 8 (Publish Frontend Container)                        | #4996 - investigate first (see below) |\n| infra        | 6 / 14 / 16 (shared resources / staging / production) | #4997 - needs its own design          |\n\n`e2e` has no classic Release definition (its pipelines are test-only) and needs no task.\n\n**architecture converts first**, in this PR, because it has no blocker: unlike backend/frontend,\n[[0011-env-tagged-images-via-release-pipelines|ADR 0011]] documents that their release stayed\nclassic specifically because the ACR service connection isn't authorized for use in YAML\npipelines - a real blocker their own tasks (#4995, #4996) need to resolve or reconfirm before\nconverting. Architecture's release def 15 uses plain pipeline variables (no service connection),\nand its approval step is already `isAutomated: true`, so there's no deploy gate being collapsed\nby moving it into CI either. `infra` (#4997) has no in-repo pipeline YAML at all today and\ndeploys three different targets off a single trunk, so it needs its own multi-environment design\nrather than a copy of this repo's pattern.\n\n### This repo's implementation\n\n**1. One YAML file** (`azure_pipelines.yml`), not the `azure_pipelines.yml`/`azure_pipelines_pr.yml`\nsplit backend/frontend use for CI, because the validate step here must be identical whether\ntriggered by a PR or a push to `main`, and splitting would duplicate it.\n\n**2. A `Validate` job always runs**, on both PRs and pushes to `main`:\n\n- A fast pre-check for duplicate leading `NNNN` prefixes among `docs/compass/adr/*.md`, so a\n  numbering collision fails fast with a clear message.\n- `structurizr-cli validate -workspace model.dsl` - a real, documented structurizr-cli\n  subcommand that needs no credentials or network access. It shares the exact\n  `AbstractCommand.loadWorkspace` -> `StructurizrDslParser.parse` code path that threw the\n  production error (`model.dsl`'s `!adrs \"docs/compass/adr\"` directive imports the ADR folder as\n  decisions at parse time), so it reproduces the real failure locally without a fake deploy or\n  hand-written duplicate-detection logic.\n\n**3. A `Publish` job runs only on a push to `main`** (`Build.Reason != 'PullRequest'` and\n`Build.SourceBranch == refs/heads/main`), running the existing `structurizr-cli push` command,\nnow sourcing `id`/`api`/`key`/`secret` through each step's `env:` block from secret pipeline\nvariables - the only way Azure DevOps' log-masking actually applies to a script-interpolated\nvariable, which fixes the plaintext leak.\n\n**4. A local husky pre-commit hook adds a faster feedback loop**: `lint-staged` formats staged\nMarkdown with Prettier, and the same `structurizr-cli validate` command runs unconditionally.\nThe pre-commit hook does **not** check ADR numbering - a local hook only sees one branch's\nfiles, so it structurally cannot catch the actual failure mode (two different branches each\nindependently claiming the same number); that check only makes sense at PR/CI time against\n`main`, which the `Validate` job already covers.\n\n**5. Release definition 15 is disabled, not deleted**, once the YAML pipeline is registered and\nverified - its release history remains the audit trail for how long the credential leak existed.\n\n## Consequences\n\n**Positive**\n\n- Release configuration for every migrated repo becomes diffable, reviewable, and revertible\n  like any other code change, instead of a UI-only edit no one sees until it breaks something.\n- A duplicate ADR number (or any DSL syntax error) now fails a PR check before merge, instead of\n  breaking the next merge to `main` silently - the concrete failure mode this ADR was triggered\n  by.\n- The Structurizr API key/secret are masked in pipeline logs going forward.\n- Contributors get the same feedback locally (pre-commit) and in CI (PR check), using the same\n  underlying command.\n\n**Negative**\n\n- backend/frontend/infra are not converted by this ADR - their tasks (#4995/#4996/#4997) may\n  conclude \"stays classic for now\" if their respective blockers (ACR authorization, multi-\n  environment design) aren't resolved, leaving CI/CD as a mix of YAML and classic pipelines for a\n  while.\n- The already-leaked Structurizr `key`/`secret` values remain valid until rotated on the\n  Structurizr server side; that rotation is a separate action, not performed by this change.\n\n**Follow-ups**\n\n- PR #5379 (Freshdesk support integration) is still open and currently claims ADR number 0014,\n  which is already taken (merged separately as\n  [[0014-private-ai-node-for-phi-bearing-generation|0014]]). Once #5379 merges it will need its\n  own renumber, now to 0016 since this ADR claims 0015 - the same collision pattern fixed once\n  already in PR #5399, about to repeat unless flagged.\n- backend (#4995), frontend (#4996), and infra (#4997) each start with investigation, not a\n  mechanical copy of this repo's pattern - see the Feature and each task for specifics.\n","date":"2026-08-15T00:00:00Z","format":"Markdown","id":"15","status":"Accepted","title":"Move CI/CD from classic Release pipelines to pipeline-as-code"},{"content":"# 2. Moving from Encounters to Treatments\n\nDate: 2024-01-22\n\n## Status\n\nAccepted\n\n## Context\n\nWe have carried over some design decisions from [V1](/FAQ/Non-technical-FAQ/Why-the-rewrite), and here are the facts that create the issue we're facing:\n\n- Our conditions are known at the moment of filling \"Objective\" accordions, which happens during the Evaluation visit.\n- The process of us treating the patient is modeled as an Encounter, which is conceptually a collection of visits along with all necessary information:\n  - Patient evaluation\n  - Patient insurance\n  - Patient recommendation\n- Our visits not connected to appointments (#461)\n- The encounter gets created when the patient is scheduled for the Evaluation visit.\n- Creating new encounter means we'll have another evaluation and a set of patient visits.\n- Encounters are not explicitly tied to the devices we're going to provide the patients with.\n- Claim items (HCPCS, their price and quantity) are not tied to the condition we are treating.\n\nDue to these facts, we have following issues:\n\n- We can't effectively support a multiple device scenario for different conditions (when we're making multiple devices to treat different patient conditions).\n- We can't effectively support a multiple device scenario for a single condition (when we're making the evening device and a day device of the same purpose for the patient).\n- We have a lot of duplication with evaluation and insurances because they are a property of an Encounter, not the Patient.\n- Overall, such design hurts conceptual integrity (by having things in the wrong boxes) and maintainability (by creating a God-object - Encounter).\n\nHere is the breakdown of current situation represented by ERD and state diagrams to illustrate the problem:\n\n### Baseline ERD\n\n![ERD](https://www.planttext.com/api/plantuml/svg/XL91Ri8m4BplAxPSkF602Y4uj1BgXHDt4rQmQhAsx8sbYF1td3WHAOtQcsV7CpiUUHlKUAsh4jxqIXMXsWPWYU6RnHblAYnPI1j7QBrUBG29iZQuE1ZbT5wW2UWKRu3eycX_ViSJNrWKr-l3rsO3zwViGmfRYvBlXIH5hwHn-bixR_lvQXjDMQTxLh9l_CzseZrouFoEh8eTdWioR_SQPUUSJKmrbcE6Tinp_cUSlOJQ6oceLxE47zsEvAn5sdItgrNRTb6XAxTjCumz6iW8q8KQEATJ565wCsq7s9ASqmzv0000)\n\nAs evident from the diagram, key issues are:\n\n- Disconnect between visits and appointments\n- Encounter \"owns\" Evaluation and Insurances, among other things\n\n### Baseline single device flow\n\n![Baseline single device flow](https://www.planttext.com/api/plantuml/svg/VLHDRzim3BthLx3PhK3GUmz33bklBTfZKGIQn2GiPCdGHuPWo7yVfTYoxAIz923v-FZu8Mz2b3wcpeYxUC0E8RgJ4EoC2AiN6Gbj1EMHRRq26Q2_-BQ3JmZnVly1w_NFwyZ0yjigbsn43pyA0rRxdm0OGKTf4XCu2xBl7TdOvSf17L0dCzH61cshYvCN9OkCEGWUHmwoUv0ckA6RHWSJtHIYNDSRZ6tnwlegvqY1OiMN3YliBVtvH1NfY13oFyDtPCRc0aUt5xjDUluknzrCUKMMoUFJ5qatylmCqLAK7-kOnjF-4C7jcY8yp1LPPPBEmrJEQihvuK6Jt5iNAAmgns8DUJDN7KvsYI-wSVFervflHgPRQ7TsdktjoHdZ_gvJlxYpNPfTcO05KU6-1bsnJiUmApvspGvYGagbdDTvbksQk4JJPbjmL533GsErpb9XNft7dYy_4tVIdY1Inw-0BOkekUB0aDe1Qc86wVSujVThT1M94-0iH3LABdyNX9YVqlJqMGuUqP_Y8PyIsK0QWPf7kGxAzLcLsF1c-GFr7m00)\n\nApart from conceptual integrity and maintainability issues, not much is wrong here, until we consider the multiple device flow.\n\n### Baseline multiple device flow\n\n![Baseline multidevice flow](https://www.planttext.com/api/plantuml/svg/hPR1ZjCm48RlVefXkI4gb3boW3qiN45mHAZAEavhvDfHOdU5qBux4xU9OyTkeN19bV4_uz_ZoQVU3xRkhJieziBknK60DJyPew0LSFUvjb9e2xmNzDxSGJr0Tufjpp76sBTNts6pURTUELVveBbnseh-pOCuIYxWPUKhYIbUmIy6CAcFjN9KoMYekyv8RG-ZXO7lHUFKsOGWqNOELPITREuDhLNPRRWvT4hOPUavL4mwoa7QEU5qWbdtrJt-4DpAeu1X_2LzAzLaVg3LxUczMcPzkfHayZbV15cb1ZKdLvd4Fb94DVn36whJoztUWcaZR_0M6jSfoNrcHcnAsjSWDLUkaNf7tTMD78D3faBUNBAnpqlkqj_lkf5BzwCYxPf9rYyqfs1jdgjeH9wZCVbeb39F6zSbKzkwrJGcfNpvZ6nLE8nVwsEkIqOmI6lEJT-C5x0S2zfaUTfepyb549pF5_Og-0n9Nj584jJzJbJX30zJpDjOADWp2sy-dS32Bb04OuhWFr84_uWYU0j5YaanHV16Yk0D5C7_G15CAA94cq84In2PjAOalSQGgK3UsNM4O8eGJ1622K8mGGYM4O9bX6068HW-iR-Zo9WOTIEfIsruT1Pg2qzxUd0psFSvyE3gWyQ2bpLuy4Rmu0ey68DdqPMmLy5dI8h1WrV1Wn5uC83ddYwE_XVz1W00)\n\nIt becomes evident that we can't support this due to design decisions also taken on the UI side of things, as \"View last DOS\" button always takes us to the \"last\" encounter, while we really want the master-detail there to support having multiple devices for conditions. However, even if the UI wasn't a problem, we would have duplication issues.\n\n## Decision\n\nReorganize the entity relationships to reduce duplication, improve conceptual integrity and maintainability, and support multi-device scenario more easily.\n\n### Target ERD\n\n![Target ERD](https://www.planttext.com/api/plantuml/svg/bLHDRy8m3BtdLtXSTc7bNY64n8U6Ta4QsiwXpI0Yn5MIWYhAVryQcgOKniJDZjFtdb_iZhMXokHxGLxCbkqP62m8UGMzupAZYkv1SCbCaJ50PRP829E6cm9wIsguZNj0DMIN64u4VBn8OrZp3RUdm-7oOpGYv_3jb1rumWOhnQZPUn3ZCmVJPBT0zpdcDOT4KtMH0Vwq86F8b2N5N8kY3pEPO2uDKq7Ix415Rc5HEZ6iIPsQa3ub1w3TzGHboYkCmPJJZKJDjUDAVUQeckn9fZyapZlMVt7DBtuVgiXHkkXPdptWDwgTG9ewp6ETFEYabdlBMHVkFtnL19aBGTU2FUwsVSthMTa1_dTQ6l4nC6wtGRLO-dRyWRO6wWU5mEDRdWKd5bBdYig4EIkupwwsmXy-wxRVi3D6KLa67Tr2mzejD8znWnMQ964IopAAyDIkaSkyr8KQepuybmx96qWQ9qjqjuFSgyerUIYLNSN1TiTqQgrC30bAoHAZOJ6Pshy8mljaEVaaPCnaQdzilm00)\n\nThe main differences are:\n\n- Evaluation is now a history of answers given during the Evaluation visit, and is a property of a patient\n- Insurances are a property of a patient\n- A visit (unified with appointments) is now a collection of workloads\n- A workload is a combination of a visit, a device, and the work we're going to do\n  - Which makes it possible to include multiple workloads for a single visit (e.g., do a follow-up for lower extremity, and cast/fitting for an upper extremity - different work done for different diagnoses/conditions of a patient)\n- A device/treatment is effective and more granular replacement for an \"Encounter\" object\n- A condition/diagnosis can be treated by multiple devices (evening/day prosthesis), but a device can also treat multiple conditions (e.g., an exoskeleton treating scoliosis and tunnel syndrome)\n- Evaluation is a special type of workload\n\n### Target single device flow\n\n![Target single-device flow](https://www.planttext.com/api/plantuml/svg/VLJBRjim4BphApOwfO6DtdCeKg39DK7GMniOBBcM250amOSDjyY_Tob5b6ZJ0GoOddtCx7Be1n-O2t5GpKteSK08vjGq10Q4zkdvEt27TFPWMP2eGmiidtTJ3Fur01yLBrC410wcSypsadlOIwLG59KflflmIh5adJPUrYifE5U-PwMF1wOYPHWDp5eZTHXI9yzx575kWPIqon3nLdh2TlljFB7vTVS61DV4Lx2nHKrkHUpjGNePlcHv7t3QwFIvB3cWy-cRcy4g3ElPCaI5vGpBMEkSfVyh4auh4nD7sTiLmKkM38midsHhGrnETdq0ix-Q2FTm94avt98deSscQ3VWBJjggJjWq8RamM5qvc_Dm3h3qVkfFrFoRqeZPv-nmSexF_0WENxzEJU2M-i8Cj6hqmOUDJTH1tJJ9MMwjUkVZ5Bck8tsp73x4O-sCDtLiTFoBYje6mAv5FG4QGqZFHYDooCfxd5HmZWtJX13yOfo__psAeo5a9oupbqPlQfuFWEZA-_H9SLfuwK1bbtUa-EY4l07OOWgIDe8M2hBGsuW-iYKEesTyir98cAkmt4sCVzgQZgNQ3wDEBtjfdcWww3_mFy0)\n\nNotable detail is that \"Treatment\" as a process of getting the patient something (a device) to \"treat\" one or more of his existing diagnoses/conditions is spawned by filling the Recommendation section.\n\n### Target multi-device flow\n\n![Target multi-device flow](https://www.planttext.com/api/plantuml/svg/fLJ1JiCm3BtdAqnFW9YGUkI0XWHSGOWRe2bAlJPIcaH9kWaG_uxRRbcxPWZ6gUdpUxRpucJk0tUXgHKx2HNM2C6yCqPWGCZk7sv0EAIr1yk3H1qou76zw6FmoG0sYiS-0WNMfdJSLb9uM4gbi1Wfs_YYnnDYoKjjl4mhARXLViLExrPSDSGm6hYrHkfGjvcygB1ejYcGI8i8iKH6cCskzsdivVmD24wrdi2w5Abc4wsUhgINuImsju47VWx8AtI-_GfJMuLXkAS8aN3S_Mv3Gyuk_nCIJbOLVOrnRZSAJgj4eAlo0riRgkLaMGKmouD4k88rjHJ36SqYP74OiWlmbg-rr2Lmo48xUtKtaQ-QWtL6eFTzVgxaHqkZYTwh1eEPkSML57Zvl4q4fzOHP1gltY_WJWlJw82kFvvORdWwzqOfivurTc_GbXtUsC4KmU5jj-Ob5PCgN0ZZ2bAFRaRKnliqpJOkCYKEPZSXlehx0qvdMQ43e_7VLkR7UPd_tSpyQ3Vpf9rvoizyQcX-g9zVM2g9m5s9NLV5ze-bZknWDDyBnrM7qyF6s9nCfEV_6m00)\n\nNotable detail is that now variation happens not on the top level, but where it matters - multiple treatments will have their own WIP progress, and their own documentation building logic that will account for the type of diagnosis we're treating and what type of device we are getting the patient, even though that logic will rely on the same answers provided during the Evaluation visit, and same set of Insurances that the patient owns.\n\n### Patient chart changes\n\nTBD: elaborate how Patient Chart will change\n\n## Consequences\n\n- It improves conceptual integrity and maintainability\n- It reduces duplication\n- It enables us to support multi-device scenarios more efficiently\n- It costs a lot, since:\n  - it produces database changes (that we have to migrate carefully since we have real data now, can't just drop it)\n  - it significantly affects the backend and it's endpoints\n  - it significantly affects the frontend, especially the backend endpoints it uses, and some of the screens (Patient Chart)\n","date":"2024-01-22T00:00:00Z","format":"Markdown","id":"2","status":"Accepted","title":"Moving from Encounters to Treatments"},{"content":"# 3. End to end testing\n\nDate: 2024-07-01\n\n## Status\n\nDraft\n\n## Context\n\nOPSEHR is a EHR system that requires a robust testing strategy to ensure its functionality, reliability, and performance. Given the complexity of healthcare applications, it's crucial to implement testing that verifies the entire application workflow from the user interface down to the database interactions. More than that, it's especially crucial in our effort to gather clients.\n\n## Decision\n\nWe have decided to use end-to-end (E2E) testing for OPSEHR and employ the Playwright framework for implementing these tests.\n\n### Rationale\n\n1. **Safety net**: E2E test (and any automated test really) provides a crucial thing for a big project, which OPSEHR is - a _safety net_. A comprehensive test suite that runs regularly and on-demand is a great way to find bugs early, and make sure that the bugs we have fixed remain fixed. This way we resolve the \"whack-a-mole\" issue typical for big projects with low Maintainability - we will only have to deal with every bug once, and we'll never introduce bugs by fixing bugs without immediately detecting it and being able to act upon it.\n2. **Comprehensive Testing**: E2E tests cover the entire application stack, ensuring that all components work together as expected.\n3. **User Perspective**: E2E tests simulate real user scenarios, providing confidence that the system will perform correctly under real-world conditions.\n4. **Lower development effort**: E2E tests are not as granular and don't cover 100% of the code base, however their thoroughness strikes a great balance between the tested area of the application and the effort required.\n5. **Playwright Framework**:\n   - **Cross-Browser Testing**: Playwright supports testing across multiple browsers (Chromium, Firefox, and WebKit).\n   - **Auto-Wait**: Playwright automatically waits for the UI elements to be ready before interacting with them, reducing flakiness in tests, which is usually a major concern in E2E testing frameworks.\n   - **API Testing**: Playwright provides capabilities to test REST APIs and intercept network requests, allowing comprehensive testing of both frontend and backend, should we need it.\n   - **Parallel Execution**: Playwright supports parallel test execution, speeding up the testing process and improving efficiency which is crucial to get quick feedback from each test session.\n\n## Alternatives Considered\n\n1. **Manual Testing**: While manual testing is necessary for exploratory and usability testing, it is time-consuming, error-prone, and not scalable for regression testing.\n   - Note: we will use manual exploratory testing to find new test cases for automation. We will also use manual testing to test scenarios that have not been automated yet.\n2. **Unit and Integration Tests**: Although these tests are faster and easier to write, they do not cover the full application workflow. They are necessary but not sufficient for ensuring end-to-end functionality.\n3. **Other E2E Frameworks (e.g., Selenium, Cypress)**:\n   - **Selenium**: While mature and widely used, Selenium requires more setup and maintenance, and lacks some of the modern features provided by Playwright.\n   - **Cypress**: Cypress is a strong contender but has limitations in terms of cross-browser testing and support for certain browser APIs.\n   - **Other JS-based Testing Frameworks (e.g., TestCafe, Puppeteer)**:\n     - **TestCafe**: Offers simplicity and ease of use, but does not support native mobile browser testing and has fewer features for handling complex user interactions compared to Playwright.\n     - **Puppeteer**: Provides excellent support for headless browser testing with Chrome, but lacks the cross-browser support (e.g., Firefox, WebKit) that Playwright offers, and requires more custom setup to achieve the same level of functionality.\n\n## Implications\n\n### Initial Learning Curve\n\nTeam members will need to learn the Playwright framework and its best practices. While not substantial (due to Playwright being JS/TS, same as our frontend) - some effort is still required. The process has to be carried out by developers, because Playwright provides [rich documentation](https://playwright.dev/docs/writing-tests). Also, more seasoned developers can help and answer any upcoming questions, and provide at least code review.\n\n### Infrastructure\n\nSetting up and maintaining the infrastructure for running E2E tests (e.g., CI/CD pipelines) will require additional resources. This has already been handled by setting up a CI/CD pipeline that runs e2e tests nightly.\n\n### Test Maintenance\n\nE2E tests require regular updates to reflect changes in the application UI and workflows. In case a developer breaks a test unintentionally, a failed nightly e2e test session will notify about this. Fixing the tests have to be a part of completing a task, i.e. a task _cannot_ be considered complete if it made e2e tests fail. This will help us ensure high quality of the application.\n\n### Part of every task\n\nFor each task, a suite of test cases have to be created. For each bug, a test case reproducing it has to be created. All of these test cases have to be automated going forward so that we could ensure that whatever was broke - is _still_ fixed, and whatever was implemented - still works.\n\n## Implementation Plan\n\n- DONE: **Initial Setup**: Configure Playwright in the existing codebase, set up the necessary infrastructure for running tests locally and in CI/CD pipelines.\n- DONE: **Pilot Testing**: Start with a small set of critical user journeys to validate the approach and gather feedback.\n- **Process Enhancement**: Development process has to be enhanced in a way that test case creation and automation is mandatory. No feature or a bug fix can be considered completely done without an e2e test case.\n  - In order not to halt development until an e2e test is created, creating an e2e test has to be a sub-task of every development work-item.\n- **Scaling Up**: Gradually increase test coverage to include all major functionalities and workflows.\n- **Continuous Improvement**: Regularly review and refine test cases and infrastructure based on feedback and evolving requirements.\n\n## Consequences\n\nAdopting end-to-end testing with the Playwright framework will significantly enhance our ability to deliver a reliable and high-quality EHR system. This decision aligns with our goals of ensuring comprehensive testing coverage, improving testing efficiency, and ultimately providing a better user experience for healthcare providers using OPSEHR.\n","date":"2024-07-01T00:00:00Z","format":"Markdown","id":"3","status":"Draft","title":"End to end testing"},{"content":"# 4. End to end test format\n\nDate: 2024-07-04\n\n## Status\n\nDraft\n\nExpands on [3. End to end testing](#3)\n\n## Summary\n\nThis ADR defines the format and structure for writing end-to-end (E2E) tests for OPSEHR using the Playwright framework. The goal is to optimize for maintainability and development speed of e2e tests cases without sacrificing quality.\n\n## Definitions and abbreviations\n\n| Term                      | Meaning                                                                                                   |\n|---------------------------|-----------------------------------------------------------------------------------------------------------|\n| POM                       | Page Object Model                                                                                         |\n| Page Object Model (POM)   | Design pattern for organizing and encapsulating page-specific locators in E2E web testing. [More](https://www.selenium.dev/documentation/test_practices/encouraged/page_object_models/). |\n| Locator                   | A string formatted in a specific way (e.g., CSS, XPath) used to search for an element in the DOM          |\n| DOM                       | Document Object Model                                                                                     |\n| Document Object Model (DOM) | A tree structure that represents all HTML elements on a page, such as buttons or labels, enabling access and manipulation. |\n| SPA                       | Single Page Application                                                                                   |\n| Command Design Pattern    | A behavioral design pattern that encapsulates a request as an object, thereby allowing for parameterization of clients with different requests, queuing of requests, and logging of the requests.|\n| Composite Design Pattern  | A structural design pattern that allows you to compose objects into tree structures to represent part-whole hierarchies. It lets clients treat individual objects and compositions uniformly. |\n\n## Context\n\nSince we have decided to use end-to-end (E2E) tests, it is essential to define the exact format and structure for writing these tests. The purpose of this Architectural Decision Record (ADR) is to establish a standard format for our automated E2E test cases to ensure they are maintainable and reusable.\n\nE2E tests typically address the following common tasks:\n\n- Logging into the application with a specific user account and role.\n- Reusing navigation steps to move across different pages.\n- Generating fake data to populate forms.\n- Reading real data that is relevant to the specific test case.\n\n## Decision\n\n1. Utilize the _Page Object Model_ (POM) design pattern to encapsulate raw page data and provide useful abstractions. This will improve the reusability of our test code.\n2. Design, implement, and document a chainable pipeline of commands using a hybrid approach of the _Composite_ and _Command_ design patterns, with a functional programming perspective.\n3. Clearly describe how common tasks are addressed and provide examples within this ADR to ensure understanding and consistency.\n\n### Source Code Organization\n\nOur End-to-End (E2E) test code is stored in the [following repository](https://dev.azure.com/opsolutionsus/EHR/_git/EHRv2.End-To-End).\n\n- **Tests Directory** (`/tests`):\n  - Tests related to the patient portal are located in `./patient/`\n  - Tests related to the office portal are located in `./office/`\n  - Tests related to the login screen are located in the root directory `./` (no subfolder)\n  - Test resources containing pre-defined test data in JSON format are located in `./resources`.\n\n- **Source Code Directory** (`/src`):\n  - **Commands**: Utility functions and actions used in tests are located in `./commands`\n  - **POM Classes**: Classes following the Page Object Model pattern are located in `./pom`\n    - **Components**: Reusable UI components are located in `./components`\n    - **Pages**: Specific page classes are located in `./pages`\n  - **Additional Utilities**: Other useful functions and helpers are located in `/utilities`\n    - **Fakers**: Fakers (random data generators) are located in `./fakers`\n\n### Page Object Model\n\n[Page Object Model (POM)](https://www.selenium.dev/documentation/test_practices/encouraged/page_object_models/) is a design pattern that encapsulates locators within so-called \"page\" classes. E.g., `DashboardPage` contains locators that will work on the \"Dashboard\" page, but won't work on other pages.\n\n#### Base class hierarchy\n\nIn a Single Page Application (SPA), most pages typically contain common components (such as a navbar). Therefore, the class hierarchy should be divided appropriately: a `Page` can be seen as a type of `Component`, but a `Component` cannot be a `Page`. This separation ensures clear organization and reusability of code.\n\n![Figure 1. Base class hierarchy](https://www.planttext.com/api/plantuml/svg/dPFTRi8m38Nl-nGMkm6D9oZJ1FkHDebfrSOUmAI6HWfD5PinX7ZtEMtHO1HfsasRnZudNqwQCGi6MQzSak2S9Q0HC0wPuV7fxTwlAbzIAR1B0AwWmYMbaEapsIT9wON0qKB0BqwxK-Z5rXvOXddm6wO0a-mP5i6l87EutGIxB6G8Q0mnsxaZ40-ci2uFL7QXn4M1leJAl0DjTn3ihs594fipbA8_66bHf_pCnxd-GD4oXR1CDr9Olctg6xJoKOfrKuyvvPVQn2CB1KvXreocbcKKepZdVR3e_F9lnCPvfiZQ4Mhham9NP7GC1fbY3S4S35jLdMahg-DAWT0KzQcfh8G2FqbrxJQwg2ThATJG6XJAVqTUV-_Q9Aex3367-Fu1FqOuzsfvyBZGqFNDEkZ1YziEb99ho0hc68Pgjh16TnoqKsoCh_91iFD_VQrxVtEVWpIRO9lJ2vLhvMg4x-t4XIFDclN_OTygJ_p7gjThzmq0)\n\n#### Defining POM page/component classes\n\nA Page Object Model (POM) class is designed to encapsulate all the necessary information needed to interact with a specific page in our application. This includes locators (such as CSS or XPath selectors) and URLs. However, it is important that this information remains hidden within the class.\n\nWhen using a POM class, the class should not accept or return locators, URLs, or any other internal details. Instead, the class should provide simple and clear methods that allow you to interact with the page. These methods can be used to perform actions like typing into fields, clicking buttons, or reading text from the page.\n\nFor example, a POM class for a login page might have methods like `enterCredentials` and `clickLoginButton`. These methods use the internal locators to perform their tasks, but the locators themselves are not exposed to the user of the class.\n\nThis approach ensures that the details of how to find elements on the page are kept separate from the code that uses those elements, making the tests easier to maintain and understand.\n\nLet's evaluate an example, [`LoginPage`](https://dev.azure.com/opsolutionsus/EHR/_git/EHRv2.End-To-End?path=/src/pom/pages/LoginPage.ts):\n\n```typescript\n// imports\n\nconst selectors = {\n loginField: `//input[@name='login']`,\n passwordField: `//input[@name='password']`,\n loginButton: `//button[@data-e2e='submit']`,\n} as const;\n\nexport default class LoginPage extends EHRPage {\n constructor(page: Page) {\n  super(page);\n }\n\n // a necessary overload so that the engine would know where are we\n get pagePath(): string {\n  return '/login';\n }\n\n async enterCredentials(credentials: Credentials) {\n  await this.page.fill(selectors.loginField, credentials.login);\n  await this.page.fill(selectors.passwordField, credentials.password);\n }\n\n async clickOfficeLogin(): Promise<OfficeWelcomePage> {\n  await this.page.click(selectors.loginButton);\n  return new OfficeWelcomePage(this.page);\n }\n\n //...\n}\n```\n\nNotes:\n\n- Locators are defined outside the class itself. This optimization technique in JavaScript is similar to making a field `static`. It prevents the recreation of locators in memory each time the class is used and allows for separating locators and pages across different files, while protecting locators from outside access.\n- Public methods in a POM class provide ways to interact with the page, similar to how a user would. For example, methods for entering credentials and clicking a \"Login\" button on a login page encapsulate the interactions without exposing the underlying locators.\n\n### Reusability through Chainable commands\n\nCommands are defined as actions or functions that require the browser to be on a specific page. After executing a command, the page the browser is on may or may not change:\n\n```typescript\nexport type Command<TCurrentPage extends EHRPage, TNextPage extends EHRPage> = (\n currentPage: TCurrentPage\n) => Promise<TNextPage>;\n```\n\nFor example, you can create a command that starts on a Login page and ends on the same Login page if the login attempt is unsuccessful. Alternatively, you can create a command that performs a successful login and navigates to a Welcome page.\n\nIf a login that is expected to be successful fails, an exception will be thrown.\n\n---\n\nBecause commands are functions, they can be composed into _Composite Commands_, which we refer to as _pipes_. These pipes can be reused:\n\n```typescript\nawait executeCommands(page, commandPipe(\n        loginAs(superAdmin), \n        logout,\n        loginAs(superAdmin),\n        logout,\n        ...\n    )\n);\n```\n\nThe commandPipe function is responsible for building the chain of commands, while executeCommands executes the chain, starting from a specific page.\n\nUsing TypeScript [compile-time checks](https://dev.azure.com/opsolutionsus/EHR/_git/EHRv2.End-To-End?path=/src/commands/base/commandPipe.ts), the proper sequence of pages is enforced at compile-time. For example, it is impossible to start the loginAs command from any page other than the Login page; as in order to login, you would need to sign out first. This technique provides early feedback and ensures that the sequence of commands is correct and executable.\n\n---\n\nThese _pipes_ can also be reused and nested at various levels of depth, as long as the start and end pages are correct:\n\n```typescript\nconst loginAndLogOut = commandPipe(loginAs(superAdmin), logout);\nconst doItTwice = commandPipe(loginAndLogOut, loginAndLogOut);\nconst doItFourTimes = commandPipe(doItTwice, doItTwice);\nawait executeCommands(page, doItFourTimes);\n```\n\nThis approach allows for creating concise tests that compose pre-defined commands into pipes in the correct sequence. For example (pseudocode):\n\n- Log in (a composite command that at the minimum enters credentials and clicks the login button)\n- Create a patient (a composite command that navigates through relevant pages and performs necessary actions)\n- Schedule an appointment (a composite command that handles the scheduling process)\nAnd so on.\n\nThe ability to nest and reuse commands like building blocks means we can model the application with POM, embed appropriate sequences of actions into commands, and chain these commands in tests to verify complex scenarios with minimal code.\n\n> Note:\n> One may ask: why are we mixing functional programming with OOP here? Can't we make all actions nestable commands and eliminate the need for POM?\n>\n> The answer is that we still need POM. In the Command design pattern, POM classes act as `Receivers`, providing a clear interface for commands to interact with. This separation of concerns enforces proper boundaries and sequences of actions. POM encapsulates the details of page interactions, ensuring that commands focus solely on the logic of the actions, not on the specifics of element locations or page structure. This combination enhances maintainability, readability, and reusability of our test code through proper separation of concerns - see figure below.\n\n![Figure 2. Separation of knowledge and responsibility](https://www.planttext.com/api/plantuml/svg/VP6nZW8n34JxV8L5Zww_uWJ55LUWexWVSBBM4k5DLh4XmDVZiCSYAEYYpBpnE5c9Oj73mCxvV89rq9WJx5EkJ5rFSCc9t6YM6EA8IU6FH045r57gm9W9tFvktb7_kSRX2yTuhYNsE_tW751paNSvYpOdC8eiMjYOX-UuxvDIISYmtlx82pbFcj3wBFkIgr1fu4ttZs25vHSWVDgrqE2PR0lJ0ZBRwRQPE6mcwsEsIHX8TxaJiBNdqcITpDASTJO-YitCGMBgOIpnY4fmVnXdaA7-JZgtiexsTrS0)\n\n### Using Real and Fake Data in Tests\n\n#### Introduction\n\nEffective testing often requires working with specific data to trigger various behaviors in the system. Additionally, it is beneficial to automatically generate test data to enhance development efficiency. From the perspective of an `action`, there should be no difference between real and fake data. Therefore, this data should be created, loaded, and injected into `actions` by the respective `Test`.\n\n#### Fake Data Generation\n\nTo automate the generation of fake data for our tests, we will define a `Faker` type as follows:\n\n```typescript\ntype Faker<T> = () => T\ntype ParametrizedFaker<TArgument, T> = (arg: TArgument) => Faker<T>;\n\n// example of a simpler Faker\ninterface PatientInfo { /* ... */ }\nconst patientInfoFaker: Faker<PatientInfo> = () => generatePatient();\n\nenum PatientType = { UpperExtremity, LowerExtremity, /* ... */ };\n\n// example of using a parametrized Faker, result is Faker<T>\nconst parametrizedFaker: ParametrizedFaker<PatientType, PatientInfo> = (type) => () => generatePatientType(type);\n```\n\nFakers can utilize a library or be custom-written, as long as they conform to this interface. This ensures flexibility and consistency in how fake data is generated.\n\n#### Real data usage\n\nIn order to use real data, we will use approach of importing JSON files as objects into Typescript source file.\n\nE.g.:\n\n```typescript\nimport * as data from './resources/data.json' assert { type: 'json' };\n```\n\n##### Naming and foldering policy (or lack thereof)\n\nThe files have to be named, and the naming policy will not be addressed in this ADR. Reasons for that are:\n\n- It's hard to decide whether to group by test or by an object type.\n- It's possible to misuse objects intended for different tests.\n- The folder will grow too big until files will be split by an arbitrary category and neither will be good.\n- Approach of storing tests and resources in a separate folder will just create a lot of folders with tests, and reusing resources becomes whacky.\n\nAnd thus, I encourage you to use your best judgement. In case a pattern emerges that is going to fit all our needs, it will be solidified in a superceding ADR.\n\n#### Benefits\n\nBenefits of such approach as follows:\n\n- Separation of concerns and no \"mixing contexts\" for better Maintainability.\n  - Actions don't know about where the data is coming from, same as POM classes.\n- The ability to mix pre-defined data with random data.\n\n```typescript\nconst data = { ...predefinedData, ...generatedData };\n```\n\n## Consequences\n\nBy implementing this structured approach for writing end-to-end (E2E) tests, we ensure that our tests for the OPSEHR system are maintainable, reusable, and efficient. Utilizing the Page Object Model (POM) design pattern encapsulates page-specific details, which helps in organizing our test code and promoting reusability. The chainable command pipeline, leveraging both Composite and Command design patterns, allows for flexible and modular test execution.\n\nFurthermore, our strategy for handling both real and fake data ensures that tests can easily adapt to different scenarios, improving development speed without compromising on quality. The use of fakers for generating test data and the clear separation of concerns between commands and data sources enhance the robustness and maintainability of our test suite.\n\nThis approach not only streamlines the testing process but also ensures that tests are reliable and reflective of real-world usage, ultimately leading to a higher quality and more resilient OPSEHR application.\n\nBy following these guidelines and best practices, we can achieve a well-organized, efficient, and effective E2E testing framework that supports continuous development and quality assurance for the OPSEHR system.\n","date":"2024-07-04T00:00:00Z","format":"Markdown","id":"4","links":[{"description":"Expands on","id":"3"}],"status":"Draft","title":"End to end test format"},{"content":"# 5. End to end testing - frontend support\n\nDate: 2024-07-10\n\n## Status\n\nDraft\n\nExpands on [End to end testing - general](#3)\n\n## Summary\n\nThis ADR defines guidelines for placing selectors in the frontend application to facilitate end-to-end (E2E) testing. The goal is to ensure that selectors are consistently and appropriately applied to elements, making it easier for E2E tests to interact with the page. This will improve the maintainability and reliability of our tests using the Playwright framework.\n\n## Context\n\nWe are using the React framework with Material-UI (MUI) components, all wrapped by our custom wrappers to facilitate potential future framework migrations. Our E2E tests rely on robust and consistent selectors to interact with page elements. For instance, the following selector is already in use:\n\n```xpath\n//button[@data-e2e='submit']\n```\n\nTo streamline the testing process, we need a standardized approach for placing selectors on elements that E2E tests will interact with, such as buttons, table rows, form fields, labels, menus, etc.\n\n## Decision\n\n### 1. Define naming policy for e2e locators\n\n1. We are going to use `data-` namespace for custom attributes.\n2. We are going to use common `data-e2e-` prefix for attributes that are e2e-related.\n   1. We are going to use `data-e2e-action` attribute to mark actionable elements (e.g., a \"submit\" button), and set it's value to the kind of action this element does. E.g., `data-e2e-action=\"submit\"`.\n   2. We are going to use `data-e2e-control` attribute to mark controls (editable fields on a form), and set it's value to the name of the form combined with the name of the control. E.g., `data-e2e-control={formName + '/' + controlName}`\n   3. We are going to use `data-e2e-`... TBD\n\n### 1. Including Selectors on Clickable Buttons\n\nAll buttons that need to be interacted with during E2E testing should include a `data-e2e-action` attribute with a value. This ensures that Playwright can reliably locate and interact with these buttons.\n\nExample for a submit button:\n\n```jsx\nimport { Button } from '@mui/material';\n\nconst SubmitButton = () => (\n  <Button data-e2e=\"submit\">Submit</Button>\n);\n\nexport default SubmitButton;\n```\n\n### 2. Including Selectors on Identifiable Elements\n\nElements such as table rows and form fields should also include a `data-e2e` attribute to facilitate E2E testing. Each element should have a unique and descriptive identifier.\n\nExample for a table row:\n\n```jsx\nimport { TableRow } from '@mui/material';\n\nconst UserTableRow = ({ user }) => (\n  <TableRow data-e2e={`user-row-${user.id}`}>\n    {/*table cells*/}\n  </TableRow>\n);\n\nexport default UserTableRow;\n```\n\nExample for a form field:\n\n```jsx\nimport { TextField } from '@mui/material';\n\nconst EHRField = ({fieldName, label}) => (\n  <TextField data-e2e={'textField-' + fieldName} label={label} />\n);\n\nexport default UserNameField;\n```\n\n## Consequences\n\n### Benefits\n\n- Consistency: A standardized approach ensures that selectors are applied consistently across the application, making tests more reliable.\n- Maintainability: Easier to maintain and update tests when selectors are applied in a predictable manner.\n- Readability: Test scripts become more readable with descriptive selectors, improving overall code quality.\n\n### Drawbacks\n\n- Development Overhead: Developers need to remember to include `data-e2e` attributes, which may introduce slight overhead.\n  - Can be mitigated with a custom linter that warns on selector omission.\n- Selector Clutter: Adding additional attributes to elements might clutter the JSX, but this is minimal compared to the benefits.\n\n### Implementation\n\n- Documentation and Training: Provide documentation and training for developers on how to use the `data-e2e` attribute effectively.\n- Code Review Process: Integrate checks into the code review process to ensure that all interactive elements have appropriate selectors.\n- Refactoring Existing Code: Refactor existing components to include `data-e2e` attributes where necessary.\n\n## Conclusion\n\nBy implementing a standardized approach for placing selectors on page elements, we ensure that our E2E tests are more robust, maintainable, and reliable. This ADR provides clear guidelines for developers to follow, improving the overall quality and efficiency of our testing process.\n","date":"2024-07-10T00:00:00Z","format":"Markdown","id":"5","links":[{"description":"Expands on","id":"3"}],"status":"Draft","title":"End to end testing - frontend support"},{"content":"# 6. NOC Codes Support\n\nDate: 2024-07-24\n\n## Status\n\nDraft\n\n## Context\n\nCurrently, Compass uses HCPCS codes and has a single table called `HCPCSCodes` with the following fields:\n\n- `HCPCSCode` (primary key, e.g., L5599)\n- `Description` (string)\n- Service fields (various)\n\nWe need to start supporting NOC (Not Otherwise Covered) codes, which are managed by the government but have descriptions and prices provided by practitioners. The following are the key points and requirements for NOC codes:\n\n- NOC codes are HCPCS codes, which are global and provided by the government.\n- NOC codes are \"umbrella\" items, and disambiguated with descriptions.\n- Each NOC description has its own price set by the medical company.\n- NOC descriptions are organization-local.\n- NOC descriptions and their prices need to be importable.\n- NOC descriptions and their prices are imported into an organization, not into the supported insurance.\n- NOC descriptions need to be fuzzy-searchable.\n- A sample list of NOC codes will be provided for organizations as a global opt-in list that they can import.\n- Insurance companies validate NOC descriptions upon payment.\n\n## Decision\n\n### Database\n\nWe will implement a separate `NOCDescriptions` table to support NOC codes. This table will be keyed with `OrganizationId` and `NOCDescriptionId (guid)` and will include a `CHECK` constraint to ensure that `HCPCSCodeId` ends with `99` (indicating an unspecified code). The table will also include fields for description and fee.\n\n![idea](https://www.planttext.com/api/plantuml/svg/TP3H2e8m58RlprESUr7lGqIOM1Aa83t0s8ODR8jjHQIzUyimJUZodFFpdOy_iuuQTprt0AoZrkAErAGXcWkBFGJpM7BSOEECL2qcIRrFKt_DXML6NfpKwdk5C0mXZb4u1i-9UgZ88lj1_-v6_lPOvXFzZGcmCYrLya7NaS97q7yvKOjCtyJe9HKNz__InOITxTPUmn15Gxye0I0JYlj-NW00)\n\n**Pros:**\n\n- Decouples NOC descriptions from HCPCS codes, providing a clear separation of concerns.\n- Fees are part of organization-scoped description, and not network insurance.\n\n**Cons:**\n\n- Requires joins when loading NOC descriptions. However, this is not a significant issue since NOCs are used separately from non-NOCs.\n- Requires NOC import to be done differently from non-NOC import.\n- Increases complexity in database schema with an additional table to manage.\n- Potential performance overhead due to additional joins and lookups.\n\n### Back\n\nTo support the new `NOCDescriptions` table, we will introduce a separate API suite specifically for managing NOC descriptions. This new API will handle all operations related to NOC descriptions, including linking to HCPCS codes, managing descriptions, and handling fees.\n\n**Approach:**\n\n- Develop a dedicated API suite for managing NOC descriptions.\n- Keep NOC description management separate from the general lookup interface.\n- Ensure the new API handles all operations related to NOC descriptions.\n\n**Pros:**\n\n- Clear separation of concerns, making the system easier to understand and maintain.\n- Allows for more specialized handling of NOC-specific logic and validations.\n- Can be optimized independently, ensuring better performance and scalability for NOC-related operations.\n- Facilitates independent scaling and deployment, improving system modularity.\n\n**Cons:**\n\n- Requires additional development effort to create and maintain the new API suite.\n- Increases the overall complexity of the system with an additional API layer.\n- Potential redundancy in endpoints if there are overlapping functionalities with the existing lookup interface.\n- Necessitates coordination between teams to ensure consistency and integration with existing systems.\n\n#### Details\n\nEndpoints we'll have support for NOCs are as follows:\n\n- Import NOC Ddescriptions\n- Fetch all descriptions\n\n### Frontend\n\n(To be completed)\n\n## Conclusion\n\nThe decision to implement a separate `NOCDescriptions` table and a dedicated API for managing NOC descriptions ensures a clear separation of concerns and enhances the maintainability and scalability of the system. This approach facilitates efficient management of NOC-specific data while maintaining the integrity and performance of the overall system architecture. Further frontend implementation details will be addressed to complete the integration process.\n","date":"2024-07-24T00:00:00Z","format":"Markdown","id":"6","status":"Draft","title":"NOC Codes Support"},{"content":"# 7. ADR conventions: Mermaid diagrams and cross-repo sync\n\nDate: 2026-05-18\n\n## Status\n\nAccepted\n\n## Context\n\nADRs for the EHRv2 project live in two repositories:\n\n- **`EHRv2.Knowledgebase`** at `adrs/` — read in the team's Obsidian vault, searched by Claude via the `/recall` slash command, edited day-to-day.\n- **`EHRv2.Architecture`** at `docs/compass/adr/` — colocated with formal architecture artifacts (the Structurizr model, landscape diagrams), maintained alongside the canonical system definition.\n\nBoth audiences need the same content. When the two diverge, knowledge consumers — humans in Obsidian, Claude, architecture reviewers, future maintainers — see inconsistent histories of the same decision.\n\nAdditionally, ADRs [[0002-moving-from-encounters-to-treatments|0002]], [[0003-end-to-end-testing|0003]], and [[0004-end-to-end-test-format|0004]] embed diagrams as external image URLs (PlantUML via planttext.com). These render but are opaque: they can't be diffed or reviewed without round-tripping through the renderer, they tie this repository to a third-party service whose lifetime we don't control, and Claude can't read them as structured content.\n\n## Decision\n\n**1. ADRs are dual-written.** Every new or modified ADR is committed to **both** repositories with identical content under the same filename (`NNNN-slug.md`). A change in one without the corresponding change in the other is a bug. This file (ADR 0007) is itself an example — it exists at `knowledgebase/adrs/0007-…md` and at `architecture/docs/compass/adr/0007-…md`.\n\n**2. Diagrams use Mermaid.** All diagrams in new ADRs are written as fenced ` ```mermaid ` code blocks inline in the markdown. No external image services. Existing PlantUML/planttext URLs in ADRs 0002–0004 are grandfathered but should be migrated to Mermaid opportunistically when those ADRs are otherwise touched.\n\n**3. Mermaid is validated before commit.** Author renders the diagram in [Mermaid Live Editor](https://mermaid.live/) **or** runs locally:\n\n```bash\nnpx -y @mermaid-js/mermaid-cli -i adrs/NNNN-slug.md -o /tmp/check.svg\n```\n\nA diagram that errors in the renderer doesn't get committed.\n\n## Consequences\n\n- **Positive:**\n  - Diffs become meaningful — reviewers see what changed in the diagram, not just that an image URL changed.\n  - ADRs survive the death of planttext.com (or any other external service).\n  - Knowledge stays consistent across the two repositories.\n  - Claude can read Mermaid source directly when answering `/recall` queries.\n- **Negative:**\n  - Authors do a little more work per ADR: two file paths to update, one render-check on diagrams.\n  - The cross-repo sync isn't enforced by tooling yet — discipline is required.\n- **Follow-ups:**\n  - **Backfill:** convert PlantUML in [[0002-moving-from-encounters-to-treatments|0002]], [[0003-end-to-end-testing|0003]], and [[0004-end-to-end-test-format|0004]] to Mermaid when those ADRs are next edited.\n  - **Tooling:** consider a pre-commit hook in both repos that lints any `mermaid` block via `mmdc` and warns if the twin file is missing from the sibling repo.\n\n## Note (2026-08-15)\n\nThe dual-write target in Decision 1 (`EHRv2.Knowledgebase` at `adrs/`) is stale: that repo has\nsince been repurposed as `graphify-out/`, a **generated** code knowledge graph regenerated by\ngit hooks from `repos/`. Hand-editing it would be silently overwritten. Hand-written knowledge,\nincluding any future dual-write target for ADRs, now lives in `wiki/`. New ADRs (e.g.\n[[0015-cicd-as-code-for-release-pipelines|0015]]) are not being dual-written until this\ndecision is revisited.\n","date":"2026-05-18T00:00:00Z","format":"Markdown","id":"7","status":"Accepted","title":"ADR conventions: Mermaid diagrams and cross-repo sync"},{"content":"# 8. Autosave drafts pattern: blob-backed, IsDraft request flag\n\nDate: 2026-05-20\n\n## Status\n\nAccepted\n\n## Context\n\nWork item [#2987](https://dev.azure.com/opsolutionsus/EHR/_workitems/edit/2987) asked us to extend autosave / draft restoration to the high-priority clinician surfaces: Demographics (patient edit), Insurance accordions, ADMIN Pre-determination / Authorization accordions, and Notes. The team had already built a draft-storage primitive (`IDraftStorageService`) and a frontend hook (`useEHRFormWithDraft`) for the evaluation and treatment workflows. The question was whether to keep extending the same pattern — drafts written to Azure Blob, gated by an `IsDraft` flag on the request — or invent something new (a per-entity `*_draft` SQL table, a generic Drafts DB schema, etc.).\n\nThe motivation behind the original pattern was a strong constraint from clinical: **the relational DB must not carry incomplete / half-filled records.** Validation rules on the persisted entities are tuned for \"this is a complete claim / note / insurance row,\" and pretending an in-progress form is a complete row pollutes downstream consumers (claim builders, notification publishers, history tables, search indexes). Drafts must live somewhere the rest of the system doesn't read by default.\n\nThe pattern was working but had reliability bugs surfaced by Jason in #2987 (\"It currently isn't working consistently\"). Before extending to more surfaces, we needed to harden the existing primitive and codify the contract so future surfaces follow it predictably.\n\n## Decision\n\n**1. Drafts continue to live in Azure Blob Storage, never in the relational DB.** `EHRBlobType.Drafts` is the canonical container. Blob name is `{DtoTypeName}/{...resourceIds}_{userId}` — keyed by `DraftStorageKey(UserId, params Guid?[] KeyParts)`. The user is part of the key so two clinicians editing the same record see their own draft. The DB schema is not touched; this means no migration churn when a new feature opts into autosave, and no risk of half-filled records leaking into validation, search, or notifications.\n\n**2. The wire contract is `IDraftEnabledRequest`.** Any PUT (and GET, when a draft load is supported) that supports drafts carries `bool IsDraft { get; }` and implements `IDraftEnabledRequest`. The controller takes `[FromQuery] bool? isDraft` and threads it. There is no separate `/draft` endpoint pair — same URL, query-string discriminator. Reasons: keeps the URL space small; clients can flip the flag without round-trips through a different code path; symmetric with existing Patient demographics and Evaluation handlers.\n\n**3. Validation is skipped when `IsDraft=true`.** For request-level validators (`SmartRequestValidator<T>`), wrap inner DTO rules with `.When(x => !x.IsDraft)`. For inline `IValidator<TDto>` calls in the handler, guard with `if (!request.IsDraft)`. Draft writes must accept any shape the user can produce — that's the whole point.\n\n**4. Side effects only fire on the non-draft branch.** Notification publishing (`IScopedPublisher`), claim-item rebuilding, patient-note auto-creation, history audit rows, search index updates — none of these run when `IsDraft=true`. The handler returns `OkResult` after the blob write and nothing else.\n\n**5. No shared base class. Inline the draft branch.** Earlier in #2987 we considered extracting a generalized `PutDraftableHandler<TRequest, TDto, TTable, TKey>` base out of `PutEvaluationHandler`. We decided against it. The variance between handlers is high (Notes has bespoke per-section permission checks and a custom many-to-many upsert callback; Insurance has multi-step note synthesis tied to `IHasPatientNote`; Pre-det/Auth wires in `noteData` reads first). Three lines of inline draft branching beats a base class with five customization points. `PutEvaluationHandler` stays as-is — it inherits drafts because it was the first surface that needed them, not because every draftable handler should.\n\n```csharp\n// The pattern, inline, in every draftable handler:\nvar draftKey = new DraftStorageKey(userAccessor.UserId, oid, pid, /* resource ids */);\nif (request.IsDraft)\n{\n    await draftStorage.PutDraftAsync(draftKey, dto, cancellationToken);\n    return new OkResult();\n}\nawait draftStorage.DeleteDraftAsync<TDto>(draftKey, cancellationToken);\n// ...existing non-draft logic...\n```\n\n**6. The frontend hook is `useEHRFormWithDraft`.** One generic 1-second-debounced autosave hook lives at `frontend/src/components/forms/hooks/useEHRFormWithDraft.ts`. It loads both the draft and the non-draft, picks the newer by `modifiedAt`, flushes on `beforeunload`, and queues the submit behind any in-flight draft save. Per-feature thin wrappers (`usePatientFormWithDraft`, `useTreatmentFormWithDraft`, `usePatientInsuranceFormWithDraft`, `useTreatmentInsuranceFormWithDraft`, `usePatientNoteFormWithDraft`) inject the resource IDs and shape the `[orgId, patientId, …, userId]` query key. Per-component hooks never reach into TanStack Query directly for drafts — they go through the wrapper.\n\n**7. Drafts are deleted on non-draft submit. No TTL today.** A successful `PUT ?isDraft=false` deletes the corresponding blob via `DraftStorageService.DeleteDraftAsync`. Abandoned drafts accumulate indefinitely. This is intentional for now — until we have a Drafts navigation folder (deferred), users have no way to *find* their old drafts, so silent cleanup would be confusing. When that folder lands, we'll add a TTL or surface a \"discard all drafts\" action there. Until then, blob storage costs are dominated by other data.\n\n## Consequences\n\n**Positive**\n\n- DB stays clean. Every row in the relational store is a fully-validated record. Downstream consumers (claims, notifications, search) need no draft awareness.\n- New autosave surfaces are cheap: ~5 lines in the BE handler, one thin FE wrapper hook, no schema migrations.\n- Multi-device / multi-tab sessions get correct draft scoping for free — the blob name includes `userId`.\n- No new surface area for the team to learn. The pattern is already partially present and now generalized.\n\n**Negative**\n\n- Draft blobs survive forever today. Acceptable while volume is low and there's no UI to enumerate drafts; will need a sweep mechanism when those constraints flip.\n- The handler still has to remember to skip validation, skip side-effect publishing, and skip patient-note synthesis when `IsDraft=true`. There's no compiler enforcement of this — it's a per-handler convention. Mitigation: the [[frontend/form-system|form-system reference]] and [[backend/how-to-recipes|backend recipes]] both document the rule; PR reviewers know what to look for.\n- Two reads happen on every form mount (the draft GET and the non-draft GET). Wasted bandwidth when no draft exists. Acceptable for now; could be optimized via a single endpoint that returns both, or a `HEAD`-style \"does a draft exist\" probe.\n\n## How to apply\n\nWhen adding autosave to a new surface, both ends do these in lockstep:\n\n### Backend\n\n1. Add `bool IsDraft = false` to the request record and mark it `IDraftEnabledRequest`.\n2. Add `IDraftStorageService draftStorage, IUserAccessor userAccessor` to the handler constructor.\n3. At the top of `Handle`, build a `DraftStorageKey` and short-circuit the draft branch:\n   ```csharp\n   var draftKey = new DraftStorageKey(userAccessor.UserId, /* scoping ids */);\n   if (request.IsDraft) {\n       await draftStorage.PutDraftAsync(draftKey, request.Dto, ct);\n       return new OkResult();\n   }\n   await draftStorage.DeleteDraftAsync<TDto>(draftKey, ct);\n   ```\n4. Skip request-level validation rules with `.When(x => !x.IsDraft)`; skip inline `IValidator<TDto>` calls with `if (!request.IsDraft)`.\n5. Mirror the same on the GET handler if the surface needs draft loading.\n6. Controller PUT/GET methods take `[FromQuery] bool? isDraft` and pass `isDraft.GetValueOrDefault()` to the request.\n\n### Frontend\n\n1. If a feature-scoped wrapper hook exists (`usePatientFormWithDraft`, `useTreatmentFormWithDraft`, `usePatientInsuranceFormWithDraft`, `useTreatmentInsuranceFormWithDraft`, `usePatientNoteFormWithDraft`), use it. Otherwise add one — it's ~50 lines.\n2. The wrapper delegates to `useEHRFormWithDraft` with `querySuffix: [orgId, ...resourceIds]` and `loadForm` / `saveForm` callbacks that take `(isDraft: boolean, ...)`.\n3. Both branches go through the same NSwag client method, with `isDraft` flipped:\n   ```ts\n   loadForm: (oid, pid, piid, isDraft) => patientInsuranceClient.insurancesGET(oid, pid, piid, isDraft),\n   saveForm: (oid, pid, piid, isDraft, form) =>\n       patientInsuranceClient.insurancesPUT(oid, pid, piid, isDraft, form).then(() => {\n           if (!isDraft) { /* refetch / notify */ }\n       }),\n   ```\n4. Don't refetch lists, fire toasts, or invalidate other queries on the draft branch. Save the user surprise for when they actually submit.\n\n## Section 9: Drafts folder for modal flows (Procurement, Fax, Email)\n\nPhase 2 (PR series after the initial #2987 ship) extends the pattern to the modal compose flows that didn't fit the single-form shape:\n\n- **Procurement** — `GenerateOrderDialog` (multi-step order creation).\n- **Procurement** — `OrderedProductDialog` (new-product-order builder). Added after `GenerateOrderDialog`; see [Section 9a](#section-9a-orderedproductdialog-product-order-drafts).\n- **Fax** — `SendFaxDialog`, plus `SendGeneratedFaxDialog` (document-scoped — see the contract's scope-keys list below).\n- **Email** — `SendEmailDialog` (always opens from a patient's documents tab), plus `SendGeneratedEmailDialog` (document-scoped).\n\n### Why a drafts *folder*, not a single draft slot\n\nThe single-form surfaces from Phase 1 have exactly one in-flight draft per (user, scope) because there's only one record being edited. Modal compose flows are different — a clinician can start three different faxes throughout a shift and want each one preserved. So each modal needs to enumerate a user's drafts under that modal's scope.\n\n### The contract\n\n1. **Per-modal scope keys.** Each modal carries the smallest scope that makes sense:\n   - Procurement (`GenerateOrderDialog`): `(userId, orgId)` — all your order drafts in this org.\n   - Procurement (`OrderedProductDialog`): `(userId, orgId)` — all your new-product-order drafts in this org.\n   - Fax: `(userId, orgId)` — all your fax drafts in this org.\n   - Email: `(userId, orgId, patientId)` — drafts of emails started from a specific patient's documents tab. Picked over org-wide because the modal is always opened from a patient context and cross-patient drafts would surface confusingly.\n   - Generated-document dialogs (`SendGeneratedEmailDialog` / `SendGeneratedFaxDialog`): the email/fax key plus `hash(documentKey)` — drafts scoped to one concrete document. `documentKey` is the stable client-side key from `usePatientDocuments`, hashed to a Guid because key parts must be Guids. List handlers match the key **length exactly** (not just by prefix), or document-scoped drafts leak into the regular list and vice versa.\n\n   > **Same-key-shape families stay isolated by DTO type name, not by key.** Two draft families can share an identical key shape — e.g. `GenerateOrderDialog` (`CreateProductsOrderDto`) and `OrderedProductDialog` (`ProductOrderDraftDto`) are both `(userId, orgId, draftId)`. They don't collide only because the blob path is prefixed with `{DtoTypeName}/` (Decision 1). This is a hidden invariant: if draft storage ever drops the type-name prefix, every same-shaped family would merge. Keep distinct draft DTO types per modal even when the scope key matches.\n\n2. **Per-modal endpoints, four methods each.** Same shape across all three modals: `GET /drafts` (list), `GET /drafts/{draftId}`, `PUT /drafts/{draftId}`, `DELETE /drafts/{draftId}`. The list endpoint reads each blob and projects a small preview DTO (e.g. `FaxDraftListItemDto { DraftId, ModifiedAt, RecipientName, ToFaxNumber }`) so the FE can render rows without N follow-up fetches.\n\n3. **Submit cleanup is frontend-driven.** The send POST (`POST /exchange/fax`, `POST /exchange/email`, `POST /procurement/orders/{id}`) is untouched; the FE calls `DELETE /drafts/{draftId}` in the dialog's `submitHandler` after the POST resolves. Mirrors how the Notes flow handles cleanup in Phase 1 and keeps the send handlers free of `?fromDraftId=` plumbing.\n\n4. **Tree-nav placement.** Each modal already uses `EHRTreeDialog` with a `nodes` map. The drafts folder is a **sibling node** keyed `drafts`, built via `buildDraftsFolderNode<TListItem>(...)`. The user opens it via a `<DraftsFolderTitleButton draftsNodeId=\"drafts\" draftCount={N} />` in the title bar. Prefer the **right slot** (`dialog.slotProps.title.childRight`); when that slot is already occupied by a modal-specific control (`OrderPunchoutPanel` in `OrderedProductDialog`, `IssueFaxNumberButton` in the fax dialogs), put the drafts button in the **left slot** (`childLeft`) instead — both render. `draftsNodeId` is resolved from the tree root, so a top-level sibling uses the bare key `\"drafts\"`. The button shows a badge with the draft count and routes the tree to `drafts` on click. Clicking a row in the drafts folder calls the modal's load callback and routes back to the parent compose node.\n\n### Storage primitive\n\n`IDraftStorageService` gained `ListDraftsAsync<TDto>(Guid userId, IReadOnlyList<Guid> prefixParts, ct)` which returns `DraftMetadata[]` (`{ KeyParts, ModifiedAt, Size }`). Implementation lists Azure blobs with prefix `{TDto.Name}/{prefix[0]}_{prefix[1]}_...` and filters by `_{userId}` suffix. No DB index needed.\n\nThe existing `PutDraftAsync<TDto>` constraint was relaxed from `AuditableDto` to `notnull`. Audit-field stamping is now a runtime check — DTOs that don't inherit `AuditableDto` (like `CreateProductsOrderDto`, and the dedicated full-form draft DTOs `EmailDraftDto`, `FaxDraftDto`, `ProductOrderDraftDto`) write verbatim, which is the desired behavior since they have no audit fields to fill. Modal drafts deliberately use **dedicated, all-nullable draft DTOs** rather than the send/submit DTOs, so they can round-trip UI-only form state (contact-selector mode, sender fields, target, partial picker values) without tripping submit-time validation.\n\n### Frontend composition\n\nTwo tiny composable hooks in `components/ui/dialog/trees/useModalDraftFormState.ts`:\n\n- **`useFormDraftAutosave({ form, draftId, saveDraft, suppressRef })`** — watches a RHF form (any) and debounces a save 1s after the last change. The `suppressRef` lets the load path skip the autosave that would otherwise fire when we programmatically `form.reset(loaded)`.\n- **`useModalDraftsList({ queryKey, listDrafts, loadDraft, deleteDraft })`** — `useQuery` over the list endpoint plus `currentDraftId` state, `setCurrentDraftId`, `discardDraft`, `startNewDraft` helpers.\n\nWhy two hooks instead of one larger `useModalDraftFormState`? The three modals already wire their forms through `useBasicForm` with custom toast strategies and submit handlers. Layering autosave + list on top — rather than rewriting their form initialization — keeps the diff small and the existing modal behavior intact.\n\nThe shared UI surfaces are:\n- **`buildDraftsFolderNode<TListItem>(props)`** — returns an `EHRDialogNode` for the drafts folder. Caller passes the preview-renderer and load/discard callbacks.\n- **`<DraftsFolderTitleButton draftsNodeId draftCount />`** — IconButton with a folder icon and badge. Goes in the parent compose node's `slotProps.title.childRight`.\n\n### Cleanup policy (unchanged from Section 7)\n\nDrafts are deleted on a successful non-draft submit (POST send for the modal flows, PUT with `IsDraft=false` for the single-form flows). Abandoned drafts persist indefinitely until the team adds a TTL sweeper. Volume is dominated by other blob types; revisit if that flips.\n\n### Section 9a: `OrderedProductDialog` product-order drafts\n\n`OrderedProductDialog` is the new-product-order builder (distinct from `GenerateOrderDialog`, which turns cart items into supplier orders). It follows the Section 9 contract with three modal-specific points worth recording:\n\n1. **Drafts are gated to the new-order mode only.** The same dialog also opens to **edit** an existing product and to **reorder**, both of which hydrate the form from a server record (`getOrganizationProduct`). Autosaving in those modes would fight the load and let a stale draft overwrite a real record on reopen. The node-factory computes `draftsEnabled = !isEditingExistingProduct` and threads it into `useFormDraftAutosave({ enabled })`, the drafts node, and the title button — so edit/reorder show no drafts UI and never autosave.\n\n2. **Full-form draft DTO.** `ProductOrderDraftDto` mirrors the dialog's whole RHF `Form` (`products[]`, `units[]`, `patientDueDate`, `target`) with every field nullable. Endpoints live on `OrderedProductsController` under `products/drafts` (note: `\"drafts\"` is not a Guid, so it doesn't collide with the `products/{productId:guid}` route). Date fields are stored as **strings** to round-trip picker values as-is, the same way `FaxDraftDto` handles its date. The list-item preview (`ProductOrderDraftListItemDto`) carries `ProductCount` and `Target`.\n\n3. **Caveat — empty draft on open.** Unlike the other modal flows, this dialog's mount effects call `form.setValue('units', …, { shouldDirty: true })` when the target is `Patient`. That marks the form dirty before the user types, so autosave can persist an empty draft just from opening a new patient order. Accepted for parity with the pattern; if it becomes noisy, add a content guard in `saveDraft` (skip the PUT until a product has a description/supplier) rather than disabling autosave.\n\n### How to apply to a new modal\n\n1. Pick the scope: smallest natural key that includes `userId` and whatever IDs the modal already routes by.\n2. Backend: add four handlers (`List`, `Get`, `Put`, `Delete`) routed under `{existing-modal-base}/drafts`. The list-item DTO carries `DraftId`, `ModifiedAt`, plus 2–3 preview fields the user would scan in the drafts list. Use a **dedicated draft DTO** distinct from the modal's submit DTO, even if the scope key matches another modal's (isolation rides on the DTO type name — see Section 9 point 1).\n3. Frontend: inside the node-factory hook (`useXxxDialogNodes`), call `useModalDraftsList` + `useFormDraftAutosave` after `useBasicForm`. Add the drafts node to the nodes map via `buildDraftsFolderNode`. Wire the title button into `slotProps.title.childRight` of the root compose node (fall back to `childLeft` if the right slot is taken). Delete the draft in the modal's existing `submitHandler` after a successful send.\n4. If the modal doubles as an editor for existing records, **gate drafts to the create path** with an `enabled` flag (see Section 9a point 1) — never autosave a form that was hydrated from a server record.\n\n## Related\n\n- The original blob-draft primitive: `backend/EHRv2.Storage/Services/IDraftStorageService.cs`, `DraftStorageKey.cs`.\n- The frontend hooks: `frontend/src/components/forms/hooks/useEHRFormWithDraft.ts`, `frontend/src/components/ui/dialog/trees/useModalDraftFormState.ts`.\n- Tree dialog primitives: `frontend/src/components/ui/dialog/trees/index.tsx`, `EHRDialogTitle.tsx` (the `childRight` slot).\n- Reference docs touched by #2987: [[frontend/form-system]], [[backend/how-to-recipes]].\n","date":"2026-05-20T00:00:00Z","format":"Markdown","id":"8","status":"Accepted","title":"Autosave drafts pattern: blob-backed, IsDraft request flag"},{"content":"# 9. Eventual consistency: background jobs and notifications move to a worker service over RabbitMQ, using Wolverine\n\nDate: 2026-06-10\n\n## Status\n\nAccepted\n\nWork item [#4629](https://dev.azure.com/opsolutionsus/EHR/_workitems/edit/4629). Related: [[0010-replace-mediatr-with-wolverine|0010]] (in-process mediator follows the same library).\n\n## Context\n\nToday every side effect of a write — search reindexing, audit/history rows, workflow transitions, communication history, real-time pushes — runs **in-process, inside the HTTP request scope** of the backend API. Request handlers publish MediatR notifications (~130 types under `EHRv2.Backend.Parts/Messaging/Notifications/`) through `IScopedPublisher`, which opens a fresh DI scope, copies the user context (`IUserAccessor.EnrichWithUserScope`), and dispatches sequentially via a custom `LoggingForeachAwaitPublisher`. Recurring and on-demand background jobs run on Quartz (SQL Server persistent store, `ehr_Jobs.` tables) **hosted inside the same API pods**.\n\nThis couples three things that should fail and scale independently:\n\n- **Durability.** A pod restart between `SaveChangesAsync()` and the completion of notification handlers silently loses side effects (at-most-once). There is no retry, no dead-letter, no record that the work was owed.\n- **Latency.** The HTTP response waits on handler chains that the caller doesn't need (audit rows, history exports, workflow bookkeeping).\n- **Capacity.** Job spikes (org-wide exports, bulk reindex) compete for CPU/memory with interactive traffic inside the same deployment.\n\nWe want **eventual consistency with guaranteed delivery** for this class of work, executed in a separately deployable and scalable AKS workload. One deliberate exception: **OpenSearch indexing stays synchronous in the API** — search-backed list views are read immediately after writes, and read-your-writes there is a UX requirement.\n\nTwo constraints shaped the technology choices:\n\n1. **Cloud-agnostic direction.** The team wants to reduce dependence on vendor-specific cloud services. This ruled out Azure Service Bus despite it being operationally attractive on AKS.\n2. **Licensing.** MediatR v13+ is commercial (we hold a Lucky Penny license, shared with AutoMapper). MassTransit v9 is commercial (~$4.8k–14.4k/yr) with v8 on security-patch-only support through roughly end of 2026. We do not want to adopt a core dependency six months before its support cliff, nor trade one license bill for another.\n\nAlternatives considered for the messaging layer:\n\n| Option | Verdict |\n|---|---|\n| Raw `RabbitMQ.Client` + keep MediatR | Rejected. \"Just sending\" is the trivial 10%; we would hand-roll the outbox, consumer DI scoping, retries, dead-lettering, idempotency, and contract versioning — a homegrown framework nobody budgets to maintain. |\n| MassTransit v8 | Rejected. Excellent library, but v8 reaches end of committed support ~6 months after adoption; staying current means v9 commercial licensing. |\n| Rebus / CAP | Viable, free (MIT), but solve only the bus side — we would still need a separate answer for replacing MediatR in-process. |\n| **Wolverine** | **Chosen.** MIT-licensed core (JasperFx's stated model: open core, paid support/add-ons), RabbitMQ transport, built-in durable transactional outbox/inbox with SQL Server + EF Core, and it doubles as the in-process mediator — one programming model for local and distributed handlers (see [[0010-replace-mediatr-with-wolverine|0010]]). |\n\nFor the broker: **RabbitMQ self-hosted in AKS**, over Azure Service Bus (vendor-specific semantics leak into code; conflicts with the cloud-agnostic direction) and over a SQL-backed transport (viable fallback, but RabbitMQ gives queue-native fan-out matching our polymorphic notification hierarchy, and a managed escape hatch exists later via any vanilla-RabbitMQ host without code changes).\n\n## Decision\n\n**1. A new `EHRv2.Worker` host, deployed as its own AKS Deployment.** It consumes from RabbitMQ and hosts the Quartz scheduler. It shares `OPSolutionsDbContext`, configuration source (`AppConfigurationSource`), and the Parts service registrations — which requires first decoupling Parts DI composition from `IMvcCoreBuilder` (today `AddEHRBasePart()` is MVC-bound).\n\n**2. RabbitMQ runs in-cluster** (official `rabbitmq:4-management` image — not Bitnami), as a StatefulSet with a PVC and durable queues, following the existing Redis/OpenSearch Helm pattern in `EHRv2.Infrastructure` (`environment/app/templates/`). Single replica initially — equivalent durability to everything else on the single-node cluster; the growth path is a 3-node quorum-queue cluster with no application change. Credentials are injected via `populateAksSecrets.mjs` like other in-cluster services. The management UI is reachable only by `kubectl port-forward`, not the ingress.\n\n**3. Wolverine with the transactional outbox is the delivery mechanism.** Events are written to outbox tables in the same `OPSolutionsDbContext` transaction as the data change, and relayed to RabbitMQ after commit. The worker uses Wolverine's inbox for deduplication. This converts today's silent at-most-once into **at-least-once with idempotent consumers** — the dual-write problem (commit succeeds / publish lost, or publish succeeds / commit rolled back) is eliminated rather than relocated.\n\n**4. What moves to the worker, what stays in the API:**\n\n| Moves to worker | Stays in API |\n|---|---|\n| Audit/history handlers (`OrganizationHistoryHandler`, procurement/inventory history) | OpenSearch reindex handlers (14) + reporting indexers — synchronous for read-your-writes |\n| Communication history (`EmailSentHistoryHandler`, `SmsHistoryHandler`) | Real-time SignalR pushes (initially; revisit once the Redis backplane path from the worker is proven) |\n| Workflow task handlers, inventory reorder | HTTP request handling (all `IRequest` handlers) |\n| Quartz scheduler host — all recurring and on-demand jobs (same SQL store; API keeps a non-started scheduler handle for `IEHRJobScheduler` call sites) | |\n\n**5. User context travels in message headers.** `ScopedPublisher`'s `IUserAccessor` enrichment is reproduced as Wolverine middleware: a send-side envelope filter serializes user/org context into headers; a consume-side filter rehydrates `IUserAccessor` in the handler scope. Audit handlers must keep recording *who* acted.\n\n**6. Phased rollout, each phase shippable:**\n\n```mermaid\nflowchart LR\n    P0[\"Phase 0\\nRabbitMQ Helm chart\\nWorker skeleton\\nParts DI decoupled from MVC\"]\n    P1[\"Phase 1\\nOutbox tables on OPSolutionsDbContext\\nShadow publishing\\n(nothing consumes yet)\"]\n    P2[\"Phase 2\\nMove handler groups\\none at a time as\\nidempotent consumers\"]\n    P3[\"Phase 3\\nMove Quartz host\\nto worker\"]\n    P0 --> P1 --> P2 --> P3\n```\n\nTarget architecture:\n\n```mermaid\nflowchart TB\n    subgraph api[\"API pods (EHRv2.Backend)\"]\n        H[\"Request handler\"] --> DB[(\"SQL Server\\n+ outbox tables\")]\n        H -->|\"synchronous, in-process\"| OS[\"OpenSearch reindex\"]\n    end\n    DB -->|\"outbox relay, after commit\"| MQ[[\"RabbitMQ\\n(StatefulSet, durable queues)\"]]\n    MQ -->|\"at-least-once + inbox dedup\"| W\n    subgraph worker[\"Worker pods (EHRv2.Worker)\"]\n        W[\"Wolverine consumers:\\naudit, comms history,\\nworkflow, inventory\"]\n        Q[\"Quartz scheduler\\n(recurring + on-demand jobs)\"]\n    end\n    W --> DB2[(\"SQL Server\")]\n    Q --> DB2\n```\n\n## Consequences\n\n**Positive**\n\n- Side effects survive pod restarts and deploys: committed data implies the event will be delivered, retried, and dead-lettered on repeated failure — observable instead of silently lost.\n- HTTP latency stops paying for audit/history/workflow work; job spikes no longer contend with interactive traffic.\n- The worker scales independently (later: KEDA on queue depth — nothing autoscales today, so not phase 1).\n- Whole messaging stack is MIT-licensed and portable across clouds: AMQP + Wolverine run unchanged on any Kubernetes, and managed vanilla-RabbitMQ hosting remains a drop-in option.\n- One library covers distributed *and* in-process messaging once [[0010-replace-mediatr-with-wolverine|0010]] lands.\n\n**Negative**\n\n- **At-least-once means duplicates.** Every handler that moves must be audited for idempotency before it moves — this is the bulk of the migration work, not the plumbing.\n- **Eventual consistency becomes user-visible** wherever moved handler output is read immediately after the action (e.g. organization history views reading `OrganizationLog`). Each moved handler group needs a UX check.\n- New stateful service in-cluster with no HA on the current single-node pool; memory headroom must be verified (backend already requests 2G/limits 4G, OpenSearch holds 2G).\n- Wolverine has a smaller community than MassTransit; mitigations are its strong docs, JasperFx paid support as an option, and conventional handler code that doesn't lock us in deeply.\n- Cross-handler ordering guarantees are lost; handlers needing order must encode it (versioned documents, last-write-wins checks) rather than assume it.\n\n**Follow-ups**\n\n- ADR for the cloud-agnostic strategy at large: messaging is the easy piece; the sticker Azure couplings are the config pipeline (Azure Table Storage), blob storage, Communication Services email, Maps, and App Insights. Without a priority order, the bus ends up portable while the app still can't boot off Azure.\n- Decide the SignalR-from-worker question (Redis backplane) when the first real-time handler is considered for the move.\n- Quorum-queue RabbitMQ cluster + KEDA when the node pool grows.\n","date":"2026-06-10T00:00:00Z","format":"Markdown","id":"9","status":"Accepted","title":"Eventual consistency: background jobs and notifications move to a worker service over RabbitMQ, using Wolverine"}],"sections":[{"content":"# -","filename":"compass.md","format":"Markdown","order":1,"title":""}]},"group":"OP Solutions","id":"15","name":"Compass","properties":{"structurizr.dsl.identifier":"ss_ops_compass"},"relationships":[{"description":"TBD","destinationId":"5","id":"69","sourceId":"15","tags":"Relationship,ContextLevel"},{"description":"TBD (LLPR)","destinationId":"6","id":"70","sourceId":"15","tags":"Relationship,ContextLevel"},{"description":"orders parts through punchout","destinationId":"7","id":"71","sourceId":"15","tags":"Relationship,ContextLevel"},{"description":"facilitates customer support","destinationId":"14","id":"78","sourceId":"15","tags":"Relationship,ContextLevel","technology":"Freshdesk widget"}],"tags":"Element,Software System,Internal System"}]},"name":"OP Solutions's Compass Context","properties":{"structurizr.inspection.info":"0","structurizr.inspection.ignore":"0","structurizr.inspection.error":"29","structurizr.inspection.warning":"0"},"views":{"componentViews":[{"automaticLayout":{"applied":false,"edgeSeparation":0,"implementation":"Graphviz","nodeSeparation":300,"rankDirection":"TopBottom","rankSeparation":300,"vertices":false},"containerId":"18","description":"Describes the Compass' backend API","externalContainerBoundariesVisible":false,"key":"component_compass_api","name":"Component View: Compass - Compass API","order":3},{"automaticLayout":{"applied":false,"edgeSeparation":0,"implementation":"Graphviz","nodeSeparation":300,"rankDirection":"TopBottom","rankSeparation":300,"vertices":false},"containerId":"16","description":"Describes the Compass' frontend application","externalContainerBoundariesVisible":false,"key":"component_compass_frontend","name":"Component View: Compass - SPA UI","order":4},{"automaticLayout":{"applied":false,"edgeSeparation":0,"implementation":"Graphviz","nodeSeparation":300,"rankDirection":"TopBottom","rankSeparation":300,"vertices":false},"containerId":"17","description":"Describes the Compass' database","externalContainerBoundariesVisible":false,"key":"component_compass_db","name":"Component View: Compass - Database","order":5},{"automaticLayout":{"applied":false,"edgeSeparation":0,"implementation":"Graphviz","nodeSeparation":300,"rankDirection":"TopBottom","rankSeparation":300,"vertices":false},"containerId":"21","description":"Describes the Compass' search engine","externalContainerBoundariesVisible":false,"key":"component_compass_search","name":"Component View: Compass - Search engine","order":6}],"configuration":{"branding":{},"styles":{"elements":[{"background":"#009688","tag":"External Person"},{"background":"#4db6ac","tag":"External System"},{"background":"#03a9f4","tag":"Internal Person"},{"background":"#4fc3f7","tag":"Internal System"},{"background":"#8bc34a","tag":"OP Clinic Person"},{"shape":"Cylinder","tag":"Storage"},{"shape":"WebBrowser","tag":"UI"},{"shape":"Robot","tag":"Worker"}],"relationships":[{"color":"#888888","dashed":false,"tag":"Relationship","thickness":1}]},"terminology":{},"themes":["https://static.structurizr.com/themes/default/theme.json","https://static.structurizr.com/themes/microsoft-azure-2023.01.24/theme.json","https://static.structurizr.com/themes/kubernetes-v0.3/theme.json"]},"containerViews":[{"automaticLayout":{"applied":false,"edgeSeparation":0,"implementation":"Graphviz","nodeSeparation":300,"rankDirection":"TopBottom","rankSeparation":300,"vertices":false},"description":"Describes the Compass system components","elements":[{"id":"10","x":0,"y":0},{"id":"16","x":0,"y":0},{"id":"17","x":0,"y":0},{"id":"18","x":0,"y":0},{"id":"19","x":0,"y":0},{"id":"20","x":0,"y":0},{"id":"21","x":0,"y":0},{"id":"22","x":0,"y":0}],"externalSoftwareSystemBoundariesVisible":false,"key":"container_compass","name":"Container View: Compass","order":2,"relationships":[{"id":"23"},{"id":"25"},{"id":"26"},{"id":"27"},{"id":"28"},{"id":"29"},{"id":"30"},{"id":"31"}],"softwareSystemId":"15"}],"deploymentViews":[{"automaticLayout":{"applied":false,"edgeSeparation":0,"implementation":"Graphviz","nodeSeparation":50,"rankDirection":"TopBottom","rankSeparation":50,"vertices":false},"elements":[{"id":"32","x":0,"y":0},{"id":"33","x":0,"y":0},{"id":"34","x":0,"y":0},{"id":"35","x":0,"y":0},{"id":"36","x":0,"y":0},{"id":"37","x":0,"y":0},{"id":"38","x":0,"y":0},{"id":"39","x":0,"y":0},{"id":"40","x":0,"y":0},{"id":"41","x":0,"y":0},{"id":"42","x":0,"y":0},{"id":"43","x":0,"y":0},{"id":"44","x":0,"y":0},{"id":"47","x":0,"y":0},{"id":"48","x":0,"y":0},{"id":"50","x":0,"y":0},{"id":"51","x":0,"y":0},{"id":"53","x":0,"y":0},{"id":"54","x":0,"y":0},{"id":"56","x":0,"y":0},{"id":"58","x":0,"y":0},{"id":"59","x":0,"y":0}],"environment":"Production","key":"azureDeployment","name":"Deployment View: Production","order":7,"relationships":[{"id":"61"},{"id":"62"},{"id":"63"},{"id":"64"},{"id":"65"},{"id":"66"},{"id":"67"},{"id":"68"}]}],"systemLandscapeViews":[{"automaticLayout":{"applied":false,"edgeSeparation":0,"implementation":"Graphviz","nodeSeparation":100,"rankDirection":"TopBottom","rankSeparation":400,"vertices":false},"description":"Describes the Compass system landscape","elements":[{"id":"2","x":0,"y":0},{"id":"3","x":0,"y":0},{"id":"4","x":0,"y":0},{"id":"5","x":0,"y":0},{"id":"6","x":0,"y":0},{"id":"7","x":0,"y":0},{"id":"8","x":0,"y":0},{"id":"9","x":0,"y":0},{"id":"11","x":0,"y":0},{"id":"12","x":0,"y":0},{"id":"13","x":0,"y":0},{"id":"14","x":0,"y":0},{"id":"15","x":0,"y":0}],"enterpriseBoundaryVisible":true,"key":"ops_compass","name":"System Landscape View","order":1,"relationships":[{"id":"69"},{"id":"70"},{"id":"71"},{"id":"73"},{"id":"74"},{"id":"75"},{"id":"76"},{"id":"77"},{"id":"78"},{"id":"79"},{"id":"80"},{"id":"81"}]}]}}