docker-compose.yml
This page is part of the documentation for Orleans.Lattice 9.9.0 (release line 9.9), built 2026-10-04. It is also published as markdown, with every table and list, at docker-compose-yml.md, and llms.txt lists every page.Part of MultiSiteManufacturing source.
# MultiSiteManufacturing - Docker Compose topology (M14 + Traefik).
#
# Two Orleans clusters, each with its own Azurite and its own internal
# cluster network. There is NO shared silo-to-silo network: a US
# silo cannot resolve or open a TCP connection to an EU silo,
# and vice versa. The only application route between the two clusters is
# the Traefik reverse proxy in front of each cluster, which is multi-homed
# onto both cluster networks. The shared backup Azurite account and
# Prometheus are also multi-homed, but only for backup storage and metrics:
#
# traefik-us : us-net + eu-net
# traefik-eu : eu-net + us-net
# azurite-backup : us-net + eu-net
# prometheus : us-net + eu-net + obs-net
#
# That means a US silo reaches the EU cluster by POSTing to
# `traefik-eu` (reachable on us-net because traefik-eu
# is attached to us-net) and Traefik then forwards the request onto
# eu-net to one of the EU silos. No silo-to-silo traffic crosses.
#
# Each Traefik provides its cluster's single public application endpoint,
# sticky-session load balancing across the cluster's two silos for the Blazor
# UI (SignalR must pin to one silo per browser tab). Cross-cluster
# replication uses Orleans.Lattice.Replication's gRPC push transport
# pointed at the peer Traefik; that traffic uses Traefik's default
# (round-robin) router for the gRPC service path.
#
# Only three host ports are published:
# 5001 -> traefik-us -> Blazor UI + replication gRPC
# 5002 -> traefik-eu -> Blazor UI + replication gRPC
# 3000 -> grafana -> observability dashboards
#
# Silo HTTP (:8080), Orleans silo (:11111), and gateway (:30000) ports
# are `expose`-d only on their internal networks - nothing silo-side
# is reachable from the host.
#
# Networks (full membership):
# us-net: azurite-us, azurite-backup, silo-us-{a,b}, traefik-us,
# traefik-eu (as the cross-cluster ingress), prometheus
# eu-net: azurite-eu, azurite-backup, silo-eu-{a,b}, traefik-eu,
# traefik-us (as the cross-cluster ingress), prometheus
# obs-net: prometheus, grafana
#
# azurite-backup is the ONE shared storage account both clusters can reach; it
# backs the backup/restore sink so a backup captured on one cluster is
# restorable from any peer (the prerequisite for a coordinated multi-cluster
# restore). It is the only non-Traefik application dependency multi-homed onto
# both cluster nets; Prometheus is multi-homed for scraping only. Silos still
# cannot reach a peer cluster's silos directly.
#
# Cluster isolation properties:
# silo-us-a -> silo-us-b : direct (us-net) OK
# silo-us-* -> traefik-eu : direct (us-net) OK
# silo-us-* -> silo-eu-* : NO SHARED NETWORK - blocked OK
#
# Replication path in this topology:
# silo-us-X --(us-net)--> traefik-eu --(eu-net)--> silo-eu-{a|b}
#
# Each silo points its package-shipped gRPC push transport
# (`PackageReplication__PeerGrpcEndpoint`) at the peer's Traefik, which
# is resolvable on the local cluster network because the peer Traefik
# is multi-homed there. Traefik load balancing routes around unhealthy
# containers: each Traefik health-checks its silos every 5s (4s timeout,
# traefik/us.yml and traefik/eu.yml) and takes a silo that fails the check
# out of rotation.
#
# Simulate a cross-cluster partition (Tier 5):
# # Sever us -> eu (remove peer Traefik from local cluster net):
# docker network disconnect msmfg_us-net msmfg-traefik-eu
# docker network disconnect msmfg_eu-net msmfg-traefik-us
# # Restore:
# docker network connect msmfg_us-net msmfg-traefik-eu
# docker network connect msmfg_eu-net msmfg-traefik-us
#
# Disconnecting the peer Traefik from the local cluster net removes
# the only route from local silos to the peer cluster; the package's
# gRPC push transport fails, exponential backoff engages, and on
# reconnect the WAL drains in HLC order. That is a genuine
# transport-level partition, not the hash-filter sim in
# FederationRouter.IsDroppedByPartitionAsync (Tier 4) - the two Tier
# types coexist; Tier 5 is network-level, Tier 4 stays as a fast-path
# simulation that requires no docker interaction.
name: msmfg
networks:
us-net:
driver: bridge
eu-net:
driver: bridge
obs-net:
driver: bridge
volumes:
azurite-us-data:
azurite-eu-data:
azurite-backup-data:
prometheus-data:
grafana-data:
x-silo-common: &silo-common
build:
context: ../../
dockerfile: samples/MultiSiteManufacturing/Dockerfile
image: msmfg-host:dev
restart: unless-stopped
# Optional state-API auth credentials (issue #886) and backup toggle (issue
# #1131). run.ps1 -Username/-Password writes a git-ignored .env next to this
# compose file containing EXPLORER_STATE_AUTH=true plus one
# LATTICE_STATE_USER_<user>=pbkdf2-sha256$... line (the salted hash, never the
# plaintext); run.ps1 -Backup adds LATTICE_BACKUP_ENABLED=true. env_file
# injects whatever it finds verbatim into every silo container, which is the
# only mechanism that carries the dynamically-named per-user credential var
# across without Compose having to template a variable NAME. `required: false`
# keeps an un-parameterised `docker compose up` (no .env present) working
# unchanged: with no file, EXPLORER_STATE_AUTH and LATTICE_BACKUP_ENABLED are
# unset/false and the host leaves both subsystems disabled.
env_file:
- path: .env
required: false
environment:
# Orleans + ASP.NET Core bind configuration. ASPNETCORE_URLS must
# bind 0.0.0.0 (via "+") so host port publishing reaches the app;
# localhost inside a container only binds the loopback interface.
ASPNETCORE_URLS: "http://+:8080"
ASPNETCORE_ENVIRONMENT: "Production"
# Each container has its own IP, so all silos can share fixed
# Orleans ports - no A/B split needed inside the container. The
# appsettings overlay still provides per-A/B ports for the legacy
# localhost run.ps1 path; we override them here.
Cluster__SiloPortA: "11111"
Cluster__SiloPortB: "11111"
Cluster__GatewayPortA: "30000"
Cluster__GatewayPortB: "30000"
expose:
- "8080"
- "8081"
- "11111"
- "30000"
# ---- Traefik common config ------------------------------------------
# Use the file provider, not the docker provider. Docker Desktop on
# Windows frequently returns "Error response from daemon" on every
# docker-API call from inside a container, which leaves the docker
# provider with zero discovered routers and makes Traefik answer 404
# for every request. The file provider is trivial, socket-free, and
# just as capable for a two-backend sticky-LB setup.
x-traefik-common: &traefik-common
image: traefik:v3.1
restart: unless-stopped
services:
# ----- Shared backup store --------------------------------------------
# A single Azurite instance, multi-homed onto BOTH cluster networks, that
# backs the causally-consistent backup/restore sink. Unlike the per-cluster
# azurite-us / azurite-eu (each reachable from only one cluster), this
# account is reachable from every silo in both clusters, so a backup
# captured on one cluster is resolvable and restorable from any peer. That
# shared visibility is what makes a coordinated multi-cluster restore of a
# replicated tree work: the peer participant in the restore saga reads the
# manifest and artifacts from this same account. Only the blob service is
# used (the backup sink is blob-only); queue/table ports are exposed for
# parity but unused.
azurite-backup:
image: mcr.microsoft.com/azure-storage/azurite:latest
container_name: msmfg-azurite-backup
restart: unless-stopped
command:
- "azurite"
- "--blobHost"
- "0.0.0.0"
- "--queueHost"
- "0.0.0.0"
- "--tableHost"
- "0.0.0.0"
- "--location"
- "/data"
- "--skipApiVersionCheck"
volumes:
- azurite-backup-data:/data
networks:
# Multi-homed: every silo in either cluster can reach the shared backup
# account, which is the whole point - a cross-cluster restore reads the
# backup from here regardless of which cluster captured it.
- us-net
- eu-net
healthcheck:
# Probe the blob port (10000) - the only service the backup sink uses.
test: ["CMD-SHELL", "node -e \"require('net').createConnection(10000,'127.0.0.1').on('connect',()=>process.exit(0)).on('error',()=>process.exit(1))\""]
interval: 5s
timeout: 3s
retries: 20
start_period: 5s
# ----- US cluster -----------------------------------------------------
azurite-us:
image: mcr.microsoft.com/azure-storage/azurite:latest
container_name: msmfg-azurite-us
restart: unless-stopped
command:
- "azurite"
- "--blobHost"
- "0.0.0.0"
- "--queueHost"
- "0.0.0.0"
- "--tableHost"
- "0.0.0.0"
- "--location"
- "/data"
- "--skipApiVersionCheck"
volumes:
- azurite-us-data:/data
networks:
- us-net
healthcheck:
# Azurite ships on a Node base image so the healthcheck uses node
# to probe the Table port - no curl/nc available in the image.
test: ["CMD-SHELL", "node -e \"require('net').createConnection(10002,'127.0.0.1').on('connect',()=>process.exit(0)).on('error',()=>process.exit(1))\""]
interval: 5s
timeout: 3s
retries: 20
start_period: 5s
traefik-us:
<<: *traefik-common
container_name: msmfg-traefik-us
command:
- "--providers.file.filename=/etc/traefik/dynamic.yml"
- "--providers.file.watch=true"
- "--entrypoints.web.address=:80"
- "--log.level=INFO"
volumes:
- ./traefik/us.yml:/etc/traefik/dynamic.yml:ro
ports:
- "5001:80"
networks:
# Multi-homed: us-net reaches its own backend silos; eu-net
# makes this Traefik the cross-cluster ingress for EU silos
# (they open replication gRPC streams here). No `wan` - the two
# cluster nets are the only bridges, and only Traefiks span both.
- us-net
- eu-net
silo-us-a:
<<: *silo-common
container_name: msmfg-silo-us-a
depends_on:
azurite-us:
condition: service_healthy
azurite-backup:
condition: service_healthy
environment:
ASPNETCORE_URLS: "http://+:8080"
ASPNETCORE_ENVIRONMENT: "Production"
Cluster__SiloPortA: "11111"
Cluster__SiloPortB: "11111"
Cluster__GatewayPortA: "30000"
Cluster__GatewayPortB: "30000"
CLUSTER_NAME: "us"
SILO_ID: "a"
Seeder__Enabled: "true"
ConnectionStrings__AzureTableStorage: "DefaultEndpointsProtocol=http;AccountName=devstoreaccount1;AccountKey=Eby8vdM02xNOcqFlqUwJPLlmEtlCDXJ1OUzFT50uSRZ6IFsuFq2UVErCz4I6tq/K1SZFPTOtr/KBHBeksoGMGw==;BlobEndpoint=http://azurite-us:10000/devstoreaccount1;QueueEndpoint=http://azurite-us:10001/devstoreaccount1;TableEndpoint=http://azurite-us:10002/devstoreaccount1;"
# Shared backup sink account (azurite-backup), reachable from BOTH
# clusters. The backup subsystem writes content-addressed manifests plus per-capture
# artifacts here so a backup captured on either cluster is restorable
# from any peer - the prerequisite for a coordinated multi-cluster
# restore of a replicated tree.
ConnectionStrings__BackupBlobStorage: "DefaultEndpointsProtocol=http;AccountName=devstoreaccount1;AccountKey=Eby8vdM02xNOcqFlqUwJPLlmEtlCDXJ1OUzFT50uSRZ6IFsuFq2UVErCz4I6tq/K1SZFPTOtr/KBHBeksoGMGw==;BlobEndpoint=http://azurite-backup:10000/devstoreaccount1;"
# Cross-cluster replication peer. The package's gRPC push
# transport sends unary batches to the peer's Traefik, which is
# resolvable on us-net because it is multi-homed there; US silos
# have NO route to EU silos directly (no shared network).
# Traefik load balancing handles per-silo failover on the
# peer side.
PackageReplication__PeerClusterId: "eu"
PackageReplication__PeerGrpcEndpoint: "http://traefik-eu:80"
# Shared replication secret. Every silo in both clusters must
# carry the same value so the package's auth interceptor
# accepts inbound batches. This value is sample-only and is
# checked into the compose file deliberately - production
# deployments must source the secret from a secret store
# (env-from a Kubernetes Secret, Docker secrets, etc.).
LATTICE_REPLICATION_SECRET: "msmfg-sample-shared-secret-do-not-use-in-production"
command: ["--cluster", "us", "--silo-id", "a"]
networks:
- us-net
silo-us-b:
<<: *silo-common
container_name: msmfg-silo-us-b
depends_on:
azurite-us:
condition: service_healthy
azurite-backup:
condition: service_healthy
silo-us-a:
condition: service_started
environment:
ASPNETCORE_URLS: "http://+:8080"
ASPNETCORE_ENVIRONMENT: "Production"
Cluster__SiloPortA: "11111"
Cluster__SiloPortB: "11111"
Cluster__GatewayPortA: "30000"
Cluster__GatewayPortB: "30000"
CLUSTER_NAME: "us"
SILO_ID: "b"
Seeder__Enabled: "false"
ConnectionStrings__AzureTableStorage: "DefaultEndpointsProtocol=http;AccountName=devstoreaccount1;AccountKey=Eby8vdM02xNOcqFlqUwJPLlmEtlCDXJ1OUzFT50uSRZ6IFsuFq2UVErCz4I6tq/K1SZFPTOtr/KBHBeksoGMGw==;BlobEndpoint=http://azurite-us:10000/devstoreaccount1;QueueEndpoint=http://azurite-us:10001/devstoreaccount1;TableEndpoint=http://azurite-us:10002/devstoreaccount1;"
ConnectionStrings__BackupBlobStorage: "DefaultEndpointsProtocol=http;AccountName=devstoreaccount1;AccountKey=Eby8vdM02xNOcqFlqUwJPLlmEtlCDXJ1OUzFT50uSRZ6IFsuFq2UVErCz4I6tq/K1SZFPTOtr/KBHBeksoGMGw==;BlobEndpoint=http://azurite-backup:10000/devstoreaccount1;"
PackageReplication__PeerClusterId: "eu"
PackageReplication__PeerGrpcEndpoint: "http://traefik-eu:80"
LATTICE_REPLICATION_SECRET: "msmfg-sample-shared-secret-do-not-use-in-production"
command: ["--cluster", "us", "--silo-id", "b"]
networks:
- us-net
# ----- EU cluster -----------------------------------------------------
azurite-eu:
image: mcr.microsoft.com/azure-storage/azurite:latest
container_name: msmfg-azurite-eu
restart: unless-stopped
command:
- "azurite"
- "--blobHost"
- "0.0.0.0"
- "--queueHost"
- "0.0.0.0"
- "--tableHost"
- "0.0.0.0"
- "--location"
- "/data"
- "--skipApiVersionCheck"
volumes:
- azurite-eu-data:/data
networks:
- eu-net
healthcheck:
test: ["CMD-SHELL", "node -e \"require('net').createConnection(10002,'127.0.0.1').on('connect',()=>process.exit(0)).on('error',()=>process.exit(1))\""]
interval: 5s
timeout: 3s
retries: 20
start_period: 5s
traefik-eu:
<<: *traefik-common
container_name: msmfg-traefik-eu
command:
- "--providers.file.filename=/etc/traefik/dynamic.yml"
- "--providers.file.watch=true"
- "--entrypoints.web.address=:80"
- "--log.level=INFO"
volumes:
- ./traefik/eu.yml:/etc/traefik/dynamic.yml:ro
ports:
- "5002:80"
networks:
# Multi-homed: eu-net reaches its own backend silos;
# us-net makes this Traefik the cross-cluster ingress for
# US silos (they open replication gRPC streams here).
- eu-net
- us-net
silo-eu-a:
<<: *silo-common
container_name: msmfg-silo-eu-a
depends_on:
azurite-eu:
condition: service_healthy
azurite-backup:
condition: service_healthy
environment:
ASPNETCORE_URLS: "http://+:8080"
ASPNETCORE_ENVIRONMENT: "Production"
Cluster__SiloPortA: "11111"
Cluster__SiloPortB: "11111"
Cluster__GatewayPortA: "30000"
Cluster__GatewayPortB: "30000"
CLUSTER_NAME: "eu"
SILO_ID: "a"
Seeder__Enabled: "false"
ConnectionStrings__AzureTableStorage: "DefaultEndpointsProtocol=http;AccountName=devstoreaccount1;AccountKey=Eby8vdM02xNOcqFlqUwJPLlmEtlCDXJ1OUzFT50uSRZ6IFsuFq2UVErCz4I6tq/K1SZFPTOtr/KBHBeksoGMGw==;BlobEndpoint=http://azurite-eu:10000/devstoreaccount1;QueueEndpoint=http://azurite-eu:10001/devstoreaccount1;TableEndpoint=http://azurite-eu:10002/devstoreaccount1;"
ConnectionStrings__BackupBlobStorage: "DefaultEndpointsProtocol=http;AccountName=devstoreaccount1;AccountKey=Eby8vdM02xNOcqFlqUwJPLlmEtlCDXJ1OUzFT50uSRZ6IFsuFq2UVErCz4I6tq/K1SZFPTOtr/KBHBeksoGMGw==;BlobEndpoint=http://azurite-backup:10000/devstoreaccount1;"
PackageReplication__PeerClusterId: "us"
PackageReplication__PeerGrpcEndpoint: "http://traefik-us:80"
LATTICE_REPLICATION_SECRET: "msmfg-sample-shared-secret-do-not-use-in-production"
command: ["--cluster", "eu", "--silo-id", "a"]
networks:
- eu-net
silo-eu-b:
<<: *silo-common
container_name: msmfg-silo-eu-b
depends_on:
azurite-eu:
condition: service_healthy
azurite-backup:
condition: service_healthy
silo-eu-a:
condition: service_started
environment:
ASPNETCORE_URLS: "http://+:8080"
ASPNETCORE_ENVIRONMENT: "Production"
Cluster__SiloPortA: "11111"
Cluster__SiloPortB: "11111"
Cluster__GatewayPortA: "30000"
Cluster__GatewayPortB: "30000"
CLUSTER_NAME: "eu"
SILO_ID: "b"
Seeder__Enabled: "false"
ConnectionStrings__AzureTableStorage: "DefaultEndpointsProtocol=http;AccountName=devstoreaccount1;AccountKey=Eby8vdM02xNOcqFlqUwJPLlmEtlCDXJ1OUzFT50uSRZ6IFsuFq2UVErCz4I6tq/K1SZFPTOtr/KBHBeksoGMGw==;BlobEndpoint=http://azurite-eu:10000/devstoreaccount1;QueueEndpoint=http://azurite-eu:10001/devstoreaccount1;TableEndpoint=http://azurite-eu:10002/devstoreaccount1;"
ConnectionStrings__BackupBlobStorage: "DefaultEndpointsProtocol=http;AccountName=devstoreaccount1;AccountKey=Eby8vdM02xNOcqFlqUwJPLlmEtlCDXJ1OUzFT50uSRZ6IFsuFq2UVErCz4I6tq/K1SZFPTOtr/KBHBeksoGMGw==;BlobEndpoint=http://azurite-backup:10000/devstoreaccount1;"
PackageReplication__PeerClusterId: "us"
PackageReplication__PeerGrpcEndpoint: "http://traefik-us:80"
LATTICE_REPLICATION_SECRET: "msmfg-sample-shared-secret-do-not-use-in-production"
command: ["--cluster", "eu", "--silo-id", "b"]
networks:
- eu-net
# ----- Observability ---------------------------------------------------
# Single Prometheus + Grafana pair giving cross-cluster visibility into
# both regions. Prometheus is multi-homed onto us-net + eu-net so it
# can scrape every silo's /metrics endpoint (exposed by the
# OpenTelemetry Prometheus exporter wired in Program.cs). Grafana sits
# on obs-net + the prometheus-bridging nets so it can query Prometheus.
#
# The Grafana dashboards - every JSON in src/lattice.dashboards/Grafana/, the
# Orleans.Lattice.Dashboards package's own folder - are served verbatim via a
# read-only bind mount; the package's drift-guard test ensures those JSONs
# stay in sync with the live meter instruments.
#
# Grafana UI is published on host port 3000 with anonymous Viewer
# access enabled - this is a sample, not a production deployment.
prometheus:
image: prom/prometheus:v2.55.1
container_name: msmfg-prometheus
restart: unless-stopped
command:
- "--config.file=/etc/prometheus/prometheus.yml"
- "--storage.tsdb.path=/prometheus"
- "--storage.tsdb.retention.time=2h"
- "--web.enable-lifecycle"
volumes:
- ./observability/prometheus.yml:/etc/prometheus/prometheus.yml:ro
- prometheus-data:/prometheus
networks:
# Multi-homed: scrape silos in both clusters, plus reachable from
# Grafana on obs-net.
- us-net
- eu-net
- obs-net
grafana:
image: grafana/grafana:11.3.0
container_name: msmfg-grafana
restart: unless-stopped
depends_on:
- prometheus
environment:
GF_AUTH_ANONYMOUS_ENABLED: "true"
GF_AUTH_ANONYMOUS_ORG_ROLE: "Viewer"
GF_AUTH_DISABLE_LOGIN_FORM: "false"
GF_SECURITY_ADMIN_USER: "admin"
GF_SECURITY_ADMIN_PASSWORD: "admin"
volumes:
- grafana-data:/var/lib/grafana
- ./observability/grafana/provisioning:/etc/grafana/provisioning:ro
# Bind-mount the dashboards JSON from the Orleans.Lattice.Dashboards
# package source tree directly. The path is relative to the compose
# file (samples/MultiSiteManufacturing/) so it walks up to the repo
# root and back down into src/lattice.dashboards/Grafana/.
- ../../src/lattice.dashboards/Grafana:/var/lib/grafana/dashboards/orleans-lattice:ro
ports:
- "3000:3000"
networks:
- obs-net