Table of Contents

docker-compose.yml

This page is part of the documentation for Orleans.Lattice 9.9.0 (release line 9.9), built 2026-10-04. It is also published as markdown, with every table and list, at docker-compose-yml.md, and llms.txt lists every page.

Part of MultiSiteManufacturing source.

# MultiSiteManufacturing - Docker Compose topology (M14 + Traefik).
#
# Two Orleans clusters, each with its own Azurite and its own internal
# cluster network. There is NO shared silo-to-silo network: a US
# silo cannot resolve or open a TCP connection to an EU silo,
# and vice versa. The only application route between the two clusters is
# the Traefik reverse proxy in front of each cluster, which is multi-homed
# onto both cluster networks. The shared backup Azurite account and
# Prometheus are also multi-homed, but only for backup storage and metrics:
#
#   traefik-us      : us-net + eu-net
#   traefik-eu      : eu-net + us-net
#   azurite-backup  : us-net + eu-net
#   prometheus      : us-net + eu-net + obs-net
#
# That means a US silo reaches the EU cluster by POSTing to
# `traefik-eu` (reachable on us-net because traefik-eu
# is attached to us-net) and Traefik then forwards the request onto
# eu-net to one of the EU silos. No silo-to-silo traffic crosses.
#
# Each Traefik provides its cluster's single public application endpoint,
# sticky-session load balancing across the cluster's two silos for the Blazor
# UI (SignalR must pin to one silo per browser tab). Cross-cluster
# replication uses Orleans.Lattice.Replication's gRPC push transport
# pointed at the peer Traefik; that traffic uses Traefik's default
# (round-robin) router for the gRPC service path.
#
# Only three host ports are published:
#   5001 -> traefik-us -> Blazor UI + replication gRPC
#   5002 -> traefik-eu -> Blazor UI + replication gRPC
#   3000 -> grafana -> observability dashboards
#
# Silo HTTP (:8080), Orleans silo (:11111), and gateway (:30000) ports
# are `expose`-d only on their internal networks - nothing silo-side
# is reachable from the host.
#
# Networks (full membership):
#   us-net: azurite-us, azurite-backup, silo-us-{a,b}, traefik-us,
#           traefik-eu (as the cross-cluster ingress), prometheus
#   eu-net: azurite-eu, azurite-backup, silo-eu-{a,b}, traefik-eu,
#           traefik-us (as the cross-cluster ingress), prometheus
#   obs-net: prometheus, grafana
#
# azurite-backup is the ONE shared storage account both clusters can reach; it
# backs the backup/restore sink so a backup captured on one cluster is
# restorable from any peer (the prerequisite for a coordinated multi-cluster
# restore). It is the only non-Traefik application dependency multi-homed onto
# both cluster nets; Prometheus is multi-homed for scraping only. Silos still
# cannot reach a peer cluster's silos directly.
#
# Cluster isolation properties:
#   silo-us-a  -> silo-us-b     : direct (us-net)             OK
#   silo-us-*  -> traefik-eu    : direct (us-net)             OK
#   silo-us-*  -> silo-eu-*     : NO SHARED NETWORK - blocked OK
#
# Replication path in this topology:
#   silo-us-X  --(us-net)--> traefik-eu --(eu-net)--> silo-eu-{a|b}
#
# Each silo points its package-shipped gRPC push transport
# (`PackageReplication__PeerGrpcEndpoint`) at the peer's Traefik, which
# is resolvable on the local cluster network because the peer Traefik
# is multi-homed there. Traefik load balancing routes around unhealthy
# containers: each Traefik health-checks its silos every 5s (4s timeout,
# traefik/us.yml and traefik/eu.yml) and takes a silo that fails the check
# out of rotation.
#
# Simulate a cross-cluster partition (Tier 5):
#   # Sever us -> eu (remove peer Traefik from local cluster net):
#   docker network disconnect msmfg_us-net msmfg-traefik-eu
#   docker network disconnect msmfg_eu-net msmfg-traefik-us
#   # Restore:
#   docker network connect    msmfg_us-net msmfg-traefik-eu
#   docker network connect    msmfg_eu-net msmfg-traefik-us
#
# Disconnecting the peer Traefik from the local cluster net removes
# the only route from local silos to the peer cluster; the package's
# gRPC push transport fails, exponential backoff engages, and on
# reconnect the WAL drains in HLC order. That is a genuine
# transport-level partition, not the hash-filter sim in
# FederationRouter.IsDroppedByPartitionAsync (Tier 4) - the two Tier
# types coexist; Tier 5 is network-level, Tier 4 stays as a fast-path
# simulation that requires no docker interaction.

name: msmfg

networks:
  us-net:
    driver: bridge
  eu-net:
    driver: bridge
  obs-net:
    driver: bridge

volumes:
  azurite-us-data:
  azurite-eu-data:
  azurite-backup-data:
  prometheus-data:
  grafana-data:

x-silo-common: &silo-common
  build:
    context: ../../
    dockerfile: samples/MultiSiteManufacturing/Dockerfile
  image: msmfg-host:dev
  restart: unless-stopped
  # Optional state-API auth credentials (issue #886) and backup toggle (issue
  # #1131). run.ps1 -Username/-Password writes a git-ignored .env next to this
  # compose file containing EXPLORER_STATE_AUTH=true plus one
  # LATTICE_STATE_USER_<user>=pbkdf2-sha256$... line (the salted hash, never the
  # plaintext); run.ps1 -Backup adds LATTICE_BACKUP_ENABLED=true. env_file
  # injects whatever it finds verbatim into every silo container, which is the
  # only mechanism that carries the dynamically-named per-user credential var
  # across without Compose having to template a variable NAME. `required: false`
  # keeps an un-parameterised `docker compose up` (no .env present) working
  # unchanged: with no file, EXPLORER_STATE_AUTH and LATTICE_BACKUP_ENABLED are
  # unset/false and the host leaves both subsystems disabled.
  env_file:
    - path: .env
      required: false
  environment:
    # Orleans + ASP.NET Core bind configuration. ASPNETCORE_URLS must
    # bind 0.0.0.0 (via "+") so host port publishing reaches the app;
    # localhost inside a container only binds the loopback interface.
    ASPNETCORE_URLS: "http://+:8080"
    ASPNETCORE_ENVIRONMENT: "Production"
    # Each container has its own IP, so all silos can share fixed
    # Orleans ports - no A/B split needed inside the container. The
    # appsettings overlay still provides per-A/B ports for the legacy
    # localhost run.ps1 path; we override them here.
    Cluster__SiloPortA: "11111"
    Cluster__SiloPortB: "11111"
    Cluster__GatewayPortA: "30000"
    Cluster__GatewayPortB: "30000"
  expose:
    - "8080"
    - "8081"
    - "11111"
    - "30000"

# ---- Traefik common config ------------------------------------------
# Use the file provider, not the docker provider. Docker Desktop on
# Windows frequently returns "Error response from daemon" on every
# docker-API call from inside a container, which leaves the docker
# provider with zero discovered routers and makes Traefik answer 404
# for every request. The file provider is trivial, socket-free, and
# just as capable for a two-backend sticky-LB setup.
x-traefik-common: &traefik-common
  image: traefik:v3.1
  restart: unless-stopped

services:
  # ----- Shared backup store --------------------------------------------
  # A single Azurite instance, multi-homed onto BOTH cluster networks, that
  # backs the causally-consistent backup/restore sink. Unlike the per-cluster
  # azurite-us / azurite-eu (each reachable from only one cluster), this
  # account is reachable from every silo in both clusters, so a backup
  # captured on one cluster is resolvable and restorable from any peer. That
  # shared visibility is what makes a coordinated multi-cluster restore of a
  # replicated tree work: the peer participant in the restore saga reads the
  # manifest and artifacts from this same account. Only the blob service is
  # used (the backup sink is blob-only); queue/table ports are exposed for
  # parity but unused.
  azurite-backup:
    image: mcr.microsoft.com/azure-storage/azurite:latest
    container_name: msmfg-azurite-backup
    restart: unless-stopped
    command:
      - "azurite"
      - "--blobHost"
      - "0.0.0.0"
      - "--queueHost"
      - "0.0.0.0"
      - "--tableHost"
      - "0.0.0.0"
      - "--location"
      - "/data"
      - "--skipApiVersionCheck"
    volumes:
      - azurite-backup-data:/data
    networks:
      # Multi-homed: every silo in either cluster can reach the shared backup
      # account, which is the whole point - a cross-cluster restore reads the
      # backup from here regardless of which cluster captured it.
      - us-net
      - eu-net
    healthcheck:
      # Probe the blob port (10000) - the only service the backup sink uses.
      test: ["CMD-SHELL", "node -e \"require('net').createConnection(10000,'127.0.0.1').on('connect',()=>process.exit(0)).on('error',()=>process.exit(1))\""]
      interval: 5s
      timeout: 3s
      retries: 20
      start_period: 5s

  # ----- US cluster -----------------------------------------------------
  azurite-us:
    image: mcr.microsoft.com/azure-storage/azurite:latest
    container_name: msmfg-azurite-us
    restart: unless-stopped
    command:
      - "azurite"
      - "--blobHost"
      - "0.0.0.0"
      - "--queueHost"
      - "0.0.0.0"
      - "--tableHost"
      - "0.0.0.0"
      - "--location"
      - "/data"
      - "--skipApiVersionCheck"
    volumes:
      - azurite-us-data:/data
    networks:
      - us-net
    healthcheck:
      # Azurite ships on a Node base image so the healthcheck uses node
      # to probe the Table port - no curl/nc available in the image.
      test: ["CMD-SHELL", "node -e \"require('net').createConnection(10002,'127.0.0.1').on('connect',()=>process.exit(0)).on('error',()=>process.exit(1))\""]
      interval: 5s
      timeout: 3s
      retries: 20
      start_period: 5s

  traefik-us:
    <<: *traefik-common
    container_name: msmfg-traefik-us
    command:
      - "--providers.file.filename=/etc/traefik/dynamic.yml"
      - "--providers.file.watch=true"
      - "--entrypoints.web.address=:80"
      - "--log.level=INFO"
    volumes:
      - ./traefik/us.yml:/etc/traefik/dynamic.yml:ro
    ports:
      - "5001:80"
    networks:
      # Multi-homed: us-net reaches its own backend silos; eu-net
      # makes this Traefik the cross-cluster ingress for EU silos
      # (they open replication gRPC streams here). No `wan` - the two
      # cluster nets are the only bridges, and only Traefiks span both.
      - us-net
      - eu-net

  silo-us-a:
    <<: *silo-common
    container_name: msmfg-silo-us-a
    depends_on:
      azurite-us:
        condition: service_healthy
      azurite-backup:
        condition: service_healthy
    environment:
      ASPNETCORE_URLS: "http://+:8080"
      ASPNETCORE_ENVIRONMENT: "Production"
      Cluster__SiloPortA: "11111"
      Cluster__SiloPortB: "11111"
      Cluster__GatewayPortA: "30000"
      Cluster__GatewayPortB: "30000"
      CLUSTER_NAME: "us"
      SILO_ID: "a"
      Seeder__Enabled: "true"
      ConnectionStrings__AzureTableStorage: "DefaultEndpointsProtocol=http;AccountName=devstoreaccount1;AccountKey=Eby8vdM02xNOcqFlqUwJPLlmEtlCDXJ1OUzFT50uSRZ6IFsuFq2UVErCz4I6tq/K1SZFPTOtr/KBHBeksoGMGw==;BlobEndpoint=http://azurite-us:10000/devstoreaccount1;QueueEndpoint=http://azurite-us:10001/devstoreaccount1;TableEndpoint=http://azurite-us:10002/devstoreaccount1;"
      # Shared backup sink account (azurite-backup), reachable from BOTH
      # clusters. The backup subsystem writes content-addressed manifests plus per-capture
      # artifacts here so a backup captured on either cluster is restorable
      # from any peer - the prerequisite for a coordinated multi-cluster
      # restore of a replicated tree.
      ConnectionStrings__BackupBlobStorage: "DefaultEndpointsProtocol=http;AccountName=devstoreaccount1;AccountKey=Eby8vdM02xNOcqFlqUwJPLlmEtlCDXJ1OUzFT50uSRZ6IFsuFq2UVErCz4I6tq/K1SZFPTOtr/KBHBeksoGMGw==;BlobEndpoint=http://azurite-backup:10000/devstoreaccount1;"
      # Cross-cluster replication peer. The package's gRPC push
      # transport sends unary batches to the peer's Traefik, which is
      # resolvable on us-net because it is multi-homed there; US silos
      # have NO route to EU silos directly (no shared network).
      # Traefik load balancing handles per-silo failover on the
      # peer side.
      PackageReplication__PeerClusterId: "eu"
      PackageReplication__PeerGrpcEndpoint: "http://traefik-eu:80"
      # Shared replication secret. Every silo in both clusters must
      # carry the same value so the package's auth interceptor
      # accepts inbound batches. This value is sample-only and is
      # checked into the compose file deliberately - production
      # deployments must source the secret from a secret store
      # (env-from a Kubernetes Secret, Docker secrets, etc.).
      LATTICE_REPLICATION_SECRET: "msmfg-sample-shared-secret-do-not-use-in-production"
    command: ["--cluster", "us", "--silo-id", "a"]
    networks:
      - us-net

  silo-us-b:
    <<: *silo-common
    container_name: msmfg-silo-us-b
    depends_on:
      azurite-us:
        condition: service_healthy
      azurite-backup:
        condition: service_healthy
      silo-us-a:
        condition: service_started
    environment:
      ASPNETCORE_URLS: "http://+:8080"
      ASPNETCORE_ENVIRONMENT: "Production"
      Cluster__SiloPortA: "11111"
      Cluster__SiloPortB: "11111"
      Cluster__GatewayPortA: "30000"
      Cluster__GatewayPortB: "30000"
      CLUSTER_NAME: "us"
      SILO_ID: "b"
      Seeder__Enabled: "false"
      ConnectionStrings__AzureTableStorage: "DefaultEndpointsProtocol=http;AccountName=devstoreaccount1;AccountKey=Eby8vdM02xNOcqFlqUwJPLlmEtlCDXJ1OUzFT50uSRZ6IFsuFq2UVErCz4I6tq/K1SZFPTOtr/KBHBeksoGMGw==;BlobEndpoint=http://azurite-us:10000/devstoreaccount1;QueueEndpoint=http://azurite-us:10001/devstoreaccount1;TableEndpoint=http://azurite-us:10002/devstoreaccount1;"
      ConnectionStrings__BackupBlobStorage: "DefaultEndpointsProtocol=http;AccountName=devstoreaccount1;AccountKey=Eby8vdM02xNOcqFlqUwJPLlmEtlCDXJ1OUzFT50uSRZ6IFsuFq2UVErCz4I6tq/K1SZFPTOtr/KBHBeksoGMGw==;BlobEndpoint=http://azurite-backup:10000/devstoreaccount1;"
      PackageReplication__PeerClusterId: "eu"
      PackageReplication__PeerGrpcEndpoint: "http://traefik-eu:80"
      LATTICE_REPLICATION_SECRET: "msmfg-sample-shared-secret-do-not-use-in-production"
    command: ["--cluster", "us", "--silo-id", "b"]
    networks:
      - us-net

  # ----- EU cluster -----------------------------------------------------
  azurite-eu:
    image: mcr.microsoft.com/azure-storage/azurite:latest
    container_name: msmfg-azurite-eu
    restart: unless-stopped
    command:
      - "azurite"
      - "--blobHost"
      - "0.0.0.0"
      - "--queueHost"
      - "0.0.0.0"
      - "--tableHost"
      - "0.0.0.0"
      - "--location"
      - "/data"
      - "--skipApiVersionCheck"
    volumes:
      - azurite-eu-data:/data
    networks:
      - eu-net
    healthcheck:
      test: ["CMD-SHELL", "node -e \"require('net').createConnection(10002,'127.0.0.1').on('connect',()=>process.exit(0)).on('error',()=>process.exit(1))\""]
      interval: 5s
      timeout: 3s
      retries: 20
      start_period: 5s

  traefik-eu:
    <<: *traefik-common
    container_name: msmfg-traefik-eu
    command:
      - "--providers.file.filename=/etc/traefik/dynamic.yml"
      - "--providers.file.watch=true"
      - "--entrypoints.web.address=:80"
      - "--log.level=INFO"
    volumes:
      - ./traefik/eu.yml:/etc/traefik/dynamic.yml:ro
    ports:
      - "5002:80"
    networks:
      # Multi-homed: eu-net reaches its own backend silos;
      # us-net makes this Traefik the cross-cluster ingress for
      # US silos (they open replication gRPC streams here).
      - eu-net
      - us-net

  silo-eu-a:
    <<: *silo-common
    container_name: msmfg-silo-eu-a
    depends_on:
      azurite-eu:
        condition: service_healthy
      azurite-backup:
        condition: service_healthy
    environment:
      ASPNETCORE_URLS: "http://+:8080"
      ASPNETCORE_ENVIRONMENT: "Production"
      Cluster__SiloPortA: "11111"
      Cluster__SiloPortB: "11111"
      Cluster__GatewayPortA: "30000"
      Cluster__GatewayPortB: "30000"
      CLUSTER_NAME: "eu"
      SILO_ID: "a"
      Seeder__Enabled: "false"
      ConnectionStrings__AzureTableStorage: "DefaultEndpointsProtocol=http;AccountName=devstoreaccount1;AccountKey=Eby8vdM02xNOcqFlqUwJPLlmEtlCDXJ1OUzFT50uSRZ6IFsuFq2UVErCz4I6tq/K1SZFPTOtr/KBHBeksoGMGw==;BlobEndpoint=http://azurite-eu:10000/devstoreaccount1;QueueEndpoint=http://azurite-eu:10001/devstoreaccount1;TableEndpoint=http://azurite-eu:10002/devstoreaccount1;"
      ConnectionStrings__BackupBlobStorage: "DefaultEndpointsProtocol=http;AccountName=devstoreaccount1;AccountKey=Eby8vdM02xNOcqFlqUwJPLlmEtlCDXJ1OUzFT50uSRZ6IFsuFq2UVErCz4I6tq/K1SZFPTOtr/KBHBeksoGMGw==;BlobEndpoint=http://azurite-backup:10000/devstoreaccount1;"
      PackageReplication__PeerClusterId: "us"
      PackageReplication__PeerGrpcEndpoint: "http://traefik-us:80"
      LATTICE_REPLICATION_SECRET: "msmfg-sample-shared-secret-do-not-use-in-production"
    command: ["--cluster", "eu", "--silo-id", "a"]
    networks:
      - eu-net

  silo-eu-b:
    <<: *silo-common
    container_name: msmfg-silo-eu-b
    depends_on:
      azurite-eu:
        condition: service_healthy
      azurite-backup:
        condition: service_healthy
      silo-eu-a:
        condition: service_started
    environment:
      ASPNETCORE_URLS: "http://+:8080"
      ASPNETCORE_ENVIRONMENT: "Production"
      Cluster__SiloPortA: "11111"
      Cluster__SiloPortB: "11111"
      Cluster__GatewayPortA: "30000"
      Cluster__GatewayPortB: "30000"
      CLUSTER_NAME: "eu"
      SILO_ID: "b"
      Seeder__Enabled: "false"
      ConnectionStrings__AzureTableStorage: "DefaultEndpointsProtocol=http;AccountName=devstoreaccount1;AccountKey=Eby8vdM02xNOcqFlqUwJPLlmEtlCDXJ1OUzFT50uSRZ6IFsuFq2UVErCz4I6tq/K1SZFPTOtr/KBHBeksoGMGw==;BlobEndpoint=http://azurite-eu:10000/devstoreaccount1;QueueEndpoint=http://azurite-eu:10001/devstoreaccount1;TableEndpoint=http://azurite-eu:10002/devstoreaccount1;"
      ConnectionStrings__BackupBlobStorage: "DefaultEndpointsProtocol=http;AccountName=devstoreaccount1;AccountKey=Eby8vdM02xNOcqFlqUwJPLlmEtlCDXJ1OUzFT50uSRZ6IFsuFq2UVErCz4I6tq/K1SZFPTOtr/KBHBeksoGMGw==;BlobEndpoint=http://azurite-backup:10000/devstoreaccount1;"
      PackageReplication__PeerClusterId: "us"
      PackageReplication__PeerGrpcEndpoint: "http://traefik-us:80"
      LATTICE_REPLICATION_SECRET: "msmfg-sample-shared-secret-do-not-use-in-production"
    command: ["--cluster", "eu", "--silo-id", "b"]
    networks:
      - eu-net

  # ----- Observability ---------------------------------------------------
  # Single Prometheus + Grafana pair giving cross-cluster visibility into
  # both regions. Prometheus is multi-homed onto us-net + eu-net so it
  # can scrape every silo's /metrics endpoint (exposed by the
  # OpenTelemetry Prometheus exporter wired in Program.cs). Grafana sits
  # on obs-net + the prometheus-bridging nets so it can query Prometheus.
  #
  # The Grafana dashboards - every JSON in src/lattice.dashboards/Grafana/, the
  # Orleans.Lattice.Dashboards package's own folder - are served verbatim via a
  # read-only bind mount; the package's drift-guard test ensures those JSONs
  # stay in sync with the live meter instruments.
  #
  # Grafana UI is published on host port 3000 with anonymous Viewer
  # access enabled - this is a sample, not a production deployment.
  prometheus:
    image: prom/prometheus:v2.55.1
    container_name: msmfg-prometheus
    restart: unless-stopped
    command:
      - "--config.file=/etc/prometheus/prometheus.yml"
      - "--storage.tsdb.path=/prometheus"
      - "--storage.tsdb.retention.time=2h"
      - "--web.enable-lifecycle"
    volumes:
      - ./observability/prometheus.yml:/etc/prometheus/prometheus.yml:ro
      - prometheus-data:/prometheus
    networks:
      # Multi-homed: scrape silos in both clusters, plus reachable from
      # Grafana on obs-net.
      - us-net
      - eu-net
      - obs-net

  grafana:
    image: grafana/grafana:11.3.0
    container_name: msmfg-grafana
    restart: unless-stopped
    depends_on:
      - prometheus
    environment:
      GF_AUTH_ANONYMOUS_ENABLED: "true"
      GF_AUTH_ANONYMOUS_ORG_ROLE: "Viewer"
      GF_AUTH_DISABLE_LOGIN_FORM: "false"
      GF_SECURITY_ADMIN_USER: "admin"
      GF_SECURITY_ADMIN_PASSWORD: "admin"
    volumes:
      - grafana-data:/var/lib/grafana
      - ./observability/grafana/provisioning:/etc/grafana/provisioning:ro
      # Bind-mount the dashboards JSON from the Orleans.Lattice.Dashboards
      # package source tree directly. The path is relative to the compose
      # file (samples/MultiSiteManufacturing/) so it walks up to the repo
      # root and back down into src/lattice.dashboards/Grafana/.
      - ../../src/lattice.dashboards/Grafana:/var/lib/grafana/dashboards/orleans-lattice:ro
    ports:
      - "3000:3000"
    networks:
      - obs-net