From 05280513f622ddebb9afe086d7177f7332ec0bf2 Mon Sep 17 00:00:00 2001 From: Andre da Silva Date: Mon, 31 Aug 2026 21:44:15 +0000 Subject: [PATCH 1/2] docker: make the shard count configurable, defaulting to conway's 8 The compose stack hardcoded four shards while testnet-conway runs eight (linera-infra argo/app-values/validator/base-values.yaml numShards: 8), so an external validator following our own docs ran half our capacity. Shard assignment is hash(validator_public_key, chain_id) % num_shards and lives in ValidatorInternalNetworkConfig, so the count is per-validator capacity, not something the network agrees on - which is why this was a sizing gap rather than a correctness one. Compose cannot template a variable number of services, so define eight and put shard-4..7 behind per-shard profiles. Shards 0..3 carry no profile: a deployment that predates this has no COMPOSE_PROFILES in .env and keeps exactly the four it already ran. The trap is that server.json holds the shard list and generate_validator_keys deliberately never regenerates it, because that would rotate the signing key and drop the validator from the committee. A re-run that took the new default would therefore start shards indexing past the end of that list, and ValidatorInternalNetworkPreConfig::shard is a plain Vec index - every one of them panics on boot. So the resolution order is: an existing server.json pins the count, an explicit --num-shards that disagrees is refused, and 8 applies only to a fresh deployment. Missing jq is fatal there rather than a silent fallthrough, for the same reason. Prometheus and Alloy now discover shards through the Docker daemon and take the index from each container's linera.shard label, so neither names a shard and neither alerts on shards a smaller deployment does not run. Without that the LineraValidatorDown alert fixed in #38 would fire on every phantom target. The hardware budget is unchanged: 8 x 6 GiB is the same 48 GiB as 4 x 12 GiB, so the reference box stays 16 cores / 128 GB. Note helm/linera-validator still defaults shards.replicas to 4. Same drift, different deployment path, and changing it needs the same care about existing StatefulSets - left alone here. --- docker/.env.production.template | 15 +- docker/alloy-config.river | 62 ++-- docker/docker-compose.local-monitoring.yaml | 2 + docker/docker-compose.remote-scylla.yaml | 13 + docker/docker-compose.yaml | 324 ++++++++++++++++++++ docker/prometheus.yaml | 37 ++- docs/DOCKER-COMPOSE.md | 31 +- docs/HARDWARE.md | 12 +- scripts/deploy-validator.sh | 78 ++++- tests/deploy-validator-test.sh | 61 +++- 10 files changed, 568 insertions(+), 67 deletions(-) diff --git a/docker/.env.production.template b/docker/.env.production.template index 511b0bf..70d55c7 100644 --- a/docker/.env.production.template +++ b/docker/.env.production.template @@ -3,7 +3,12 @@ # Normally written for you by scripts/deploy-validator.sh, which fills in # DOMAIN, ACME_EMAIL, GENESIS_URL, GENESIS_BUCKET, GENESIS_PATH_PREFIX, # VALIDATOR_KEY, VALIDATOR_NAME, HOSTNAME, LINERA_VALIDATOR_IMAGE, -# LINERA_CLIENT_IMAGE and NUM_SHARDS. +# LINERA_CLIENT_IMAGE, NUM_SHARDS and COMPOSE_PROFILES. +# +# COMPOSE_PROFILES enables shard-4..7; shard-0..3 always run. Docker Compose +# reads it from this file, so editing it changes how many shards start - which +# will not match the shard list in server.json. Use --num-shards at deploy time +# instead; see docs/DOCKER-COMPOSE.md#changing-the-shard-count. # See https://docs.infra.linera.net/QUICKSTART/. This template is the # reference for manual setup and for tuning an existing deployment. # @@ -189,6 +194,14 @@ HOSTNAME=your-domain.example.com #LIMIT_MEM_SHARD_2=6G #LIMIT_CPUS_SHARD_3=2.00 #LIMIT_MEM_SHARD_3=6G +#LIMIT_CPUS_SHARD_4=2.00 +#LIMIT_MEM_SHARD_4=6G +#LIMIT_CPUS_SHARD_5=2.00 +#LIMIT_MEM_SHARD_5=6G +#LIMIT_CPUS_SHARD_6=2.00 +#LIMIT_MEM_SHARD_6=6G +#LIMIT_CPUS_SHARD_7=2.00 +#LIMIT_MEM_SHARD_7=6G # Observability stack #LIMIT_CPUS_ALLOY=0.50 diff --git a/docker/alloy-config.river b/docker/alloy-config.river index 6e291b7..19afd12 100644 --- a/docker/alloy-config.river +++ b/docker/alloy-config.river @@ -17,34 +17,44 @@ prometheus.scrape "proxy_metrics" { scrape_timeout = "10s" } +// Discover the shards instead of listing them, so the shard count is +// configurable without editing this file. Mirrors the docker_sd_configs block +// in prometheus.yaml — the two scrape paths must agree on job and shard. +discovery.relabel "shard_metrics" { + targets = discovery.docker.containers.targets + + // Every shard service carries linera.shard; nothing else does. + rule { + source_labels = ["__meta_docker_container_label_linera_shard"] + regex = ".+" + action = "keep" + } + + rule { + source_labels = ["__meta_docker_network_ip"] + target_label = "__address__" + replacement = "$1:21100" + } + + rule { + source_labels = ["__meta_docker_container_label_linera_shard"] + target_label = "shard" + } + + rule { + target_label = "job" + replacement = "linera-shard" + } + + rule { + target_label = "instance" + replacement = env("HOSTNAME") + } +} + // Scrape metrics from all shard services prometheus.scrape "shard_metrics" { - targets = [ - { - __address__ = "docker-shard-1:21100", - job = "linera-shard", - shard = "0", - instance = env("HOSTNAME"), - }, - { - __address__ = "docker-shard-2:21100", - job = "linera-shard", - shard = "1", - instance = env("HOSTNAME"), - }, - { - __address__ = "docker-shard-3:21100", - job = "linera-shard", - shard = "2", - instance = env("HOSTNAME"), - }, - { - __address__ = "docker-shard-4:21100", - job = "linera-shard", - shard = "3", - instance = env("HOSTNAME"), - }, - ] + targets = discovery.relabel.shard_metrics.output forward_to = [otelcol.receiver.prometheus.default.receiver] diff --git a/docker/docker-compose.local-monitoring.yaml b/docker/docker-compose.local-monitoring.yaml index 33a5eba..8a098ec 100644 --- a/docker/docker-compose.local-monitoring.yaml +++ b/docker/docker-compose.local-monitoring.yaml @@ -140,6 +140,8 @@ services: # Prometheus fails to start on a missing rule file. - ./recording.rules.yaml:/etc/prometheus/recording.rules.yaml:ro - ${PROMETHEUS_DATA_DIR:-./data/prometheus}:/prometheus + # Read-only: docker_sd_configs discovers the shards through it. + - /var/run/docker.sock:/var/run/docker.sock:ro ports: - "${PROMETHEUS_PORT:-9090}:9090" deploy: diff --git a/docker/docker-compose.remote-scylla.yaml b/docker/docker-compose.remote-scylla.yaml index 3101304..38745ed 100644 --- a/docker/docker-compose.remote-scylla.yaml +++ b/docker/docker-compose.remote-scylla.yaml @@ -46,3 +46,16 @@ services: shard-3: extra_hosts: - "scylla=${SCYLLA_HOST:?set SCYLLA_HOST to the address of your ScyllaDB host}" + + shard-4: + extra_hosts: + - "scylla=${SCYLLA_HOST:?set SCYLLA_HOST to the address of your ScyllaDB host}" + shard-5: + extra_hosts: + - "scylla=${SCYLLA_HOST:?set SCYLLA_HOST to the address of your ScyllaDB host}" + shard-6: + extra_hosts: + - "scylla=${SCYLLA_HOST:?set SCYLLA_HOST to the address of your ScyllaDB host}" + shard-7: + extra_hosts: + - "scylla=${SCYLLA_HOST:?set SCYLLA_HOST to the address of your ScyllaDB host}" diff --git a/docker/docker-compose.yaml b/docker/docker-compose.yaml index a0433b7..abbb53b 100644 --- a/docker/docker-compose.yaml +++ b/docker/docker-compose.yaml @@ -380,6 +380,7 @@ services: - .:/config labels: com.centurylinklabs.watchtower.enable: "true" + linera.shard: "0" # Same probe and timings as the chart's shard livenessProbe. healthcheck: test: ["CMD-SHELL", "bash -c 'exec 3<>/dev/tcp/localhost/19100' || exit 1"] @@ -457,6 +458,7 @@ services: - .:/config labels: com.centurylinklabs.watchtower.enable: "true" + linera.shard: "1" # Same probe and timings as the chart's shard livenessProbe. healthcheck: test: ["CMD-SHELL", "bash -c 'exec 3<>/dev/tcp/localhost/19100' || exit 1"] @@ -534,6 +536,7 @@ services: - .:/config labels: com.centurylinklabs.watchtower.enable: "true" + linera.shard: "2" # Same probe and timings as the chart's shard livenessProbe. healthcheck: test: ["CMD-SHELL", "bash -c 'exec 3<>/dev/tcp/localhost/19100' || exit 1"] @@ -611,6 +614,7 @@ services: - .:/config labels: com.centurylinklabs.watchtower.enable: "true" + linera.shard: "3" # Same probe and timings as the chart's shard livenessProbe. healthcheck: test: ["CMD-SHELL", "bash -c 'exec 3<>/dev/tcp/localhost/19100' || exit 1"] @@ -631,6 +635,326 @@ services: cpus: "${LIMIT_CPUS_SHARD_3:-2.00}" memory: ${LIMIT_MEM_SHARD_3:-6G} + shard-4: + image: >- + ${LINERA_VALIDATOR_IMAGE:-us-docker.pkg.dev/linera-io-dev/linera-public-registry/linera-validator:testnet_conway_release} + container_name: docker-shard-5 + hostname: docker-shard-5 + restart: unless-stopped + logging: *default-logging + cpuset: "${WORKLOAD_CPUSET:-}" + command: + - /linera-server + - run + - --storage + - scylladb:tcp:scylla:${SCYLLA_PORT:-9042} + - --server + - /config/server.json + - --shard + - "4" + - --storage-replication-factor + - "${STORAGE_REPLICATION_FACTOR:-1}" + # Storage (LRU) caches — parity with the k8s chart's shards.cli. + - --storage-max-cache-size + - "${LINERA_STORAGE_MAX_CACHE_SIZE:-1000000000}" + - --storage-max-cache-entries + - "${LINERA_STORAGE_MAX_CACHE_ENTRIES:-500000}" + - --storage-max-value-entry-size + - "${LINERA_STORAGE_MAX_VALUE_ENTRY_SIZE:-1000000}" + - --storage-max-find-keys-entry-size + - "${LINERA_STORAGE_MAX_FIND_KEYS_ENTRY_SIZE:-1000000}" + - --storage-max-find-key-values-entry-size + - "${LINERA_STORAGE_MAX_FIND_KEY_VALUES_ENTRY_SIZE:-1000000}" + - --storage-max-cache-value-size + - "${LINERA_STORAGE_MAX_CACHE_VALUE_SIZE:-500000000}" + - --storage-max-cache-find-keys-size + - "${LINERA_STORAGE_MAX_CACHE_FIND_KEYS_SIZE:-200000000}" + - --storage-max-cache-find-key-values-size + - "${LINERA_STORAGE_MAX_CACHE_FIND_KEY_VALUES_SIZE:-10000000}" + # In-memory caches — parity with the k8s chart's shards.cli. + - --blob-cache-size + - "${LINERA_BLOB_CACHE_SIZE:-1000}" + - --confirmed-block-cache-size + - "${LINERA_CONFIRMED_BLOCK_CACHE_SIZE:-10000}" + - --certificate-cache-size + - "${LINERA_CERTIFICATE_CACHE_SIZE:-5000}" + - --certificate-raw-cache-size + - "${LINERA_CERTIFICATE_RAW_CACHE_SIZE:-50000}" + - --event-cache-size + - "${LINERA_EVENT_CACHE_SIZE:-20000}" + - --execution-state-cache-size + - "${LINERA_EXECUTION_STATE_CACHE_SIZE:-20000}" + - --block-cache-size + - "${LINERA_BLOCK_CACHE_SIZE:-20000}" + - --chain-worker-ttl-ms + - "${LINERA_CHAIN_WORKER_TTL_MS:-300000}" + profiles: + - shard-4 + volumes: + - .:/config + labels: + com.centurylinklabs.watchtower.enable: "true" + linera.shard: "4" + # Same probe and timings as the chart's shard livenessProbe. + healthcheck: + test: ["CMD-SHELL", "bash -c 'exec 3<>/dev/tcp/localhost/19100' || exit 1"] + interval: 30s + timeout: 10s + retries: 3 + start_period: 60s + depends_on: + shard-init: + condition: service_completed_successfully + ulimits: + nofile: + soft: ${ULIMIT_NOFILE:-524288} + hard: ${ULIMIT_NOFILE:-524288} + deploy: + resources: + limits: + cpus: "${LIMIT_CPUS_SHARD_4:-2.00}" + memory: ${LIMIT_MEM_SHARD_4:-6G} + + shard-5: + image: >- + ${LINERA_VALIDATOR_IMAGE:-us-docker.pkg.dev/linera-io-dev/linera-public-registry/linera-validator:testnet_conway_release} + container_name: docker-shard-6 + hostname: docker-shard-6 + restart: unless-stopped + logging: *default-logging + cpuset: "${WORKLOAD_CPUSET:-}" + command: + - /linera-server + - run + - --storage + - scylladb:tcp:scylla:${SCYLLA_PORT:-9042} + - --server + - /config/server.json + - --shard + - "5" + - --storage-replication-factor + - "${STORAGE_REPLICATION_FACTOR:-1}" + # Storage (LRU) caches — parity with the k8s chart's shards.cli. + - --storage-max-cache-size + - "${LINERA_STORAGE_MAX_CACHE_SIZE:-1000000000}" + - --storage-max-cache-entries + - "${LINERA_STORAGE_MAX_CACHE_ENTRIES:-500000}" + - --storage-max-value-entry-size + - "${LINERA_STORAGE_MAX_VALUE_ENTRY_SIZE:-1000000}" + - --storage-max-find-keys-entry-size + - "${LINERA_STORAGE_MAX_FIND_KEYS_ENTRY_SIZE:-1000000}" + - --storage-max-find-key-values-entry-size + - "${LINERA_STORAGE_MAX_FIND_KEY_VALUES_ENTRY_SIZE:-1000000}" + - --storage-max-cache-value-size + - "${LINERA_STORAGE_MAX_CACHE_VALUE_SIZE:-500000000}" + - --storage-max-cache-find-keys-size + - "${LINERA_STORAGE_MAX_CACHE_FIND_KEYS_SIZE:-200000000}" + - --storage-max-cache-find-key-values-size + - "${LINERA_STORAGE_MAX_CACHE_FIND_KEY_VALUES_SIZE:-10000000}" + # In-memory caches — parity with the k8s chart's shards.cli. + - --blob-cache-size + - "${LINERA_BLOB_CACHE_SIZE:-1000}" + - --confirmed-block-cache-size + - "${LINERA_CONFIRMED_BLOCK_CACHE_SIZE:-10000}" + - --certificate-cache-size + - "${LINERA_CERTIFICATE_CACHE_SIZE:-5000}" + - --certificate-raw-cache-size + - "${LINERA_CERTIFICATE_RAW_CACHE_SIZE:-50000}" + - --event-cache-size + - "${LINERA_EVENT_CACHE_SIZE:-20000}" + - --execution-state-cache-size + - "${LINERA_EXECUTION_STATE_CACHE_SIZE:-20000}" + - --block-cache-size + - "${LINERA_BLOCK_CACHE_SIZE:-20000}" + - --chain-worker-ttl-ms + - "${LINERA_CHAIN_WORKER_TTL_MS:-300000}" + profiles: + - shard-5 + volumes: + - .:/config + labels: + com.centurylinklabs.watchtower.enable: "true" + linera.shard: "5" + # Same probe and timings as the chart's shard livenessProbe. + healthcheck: + test: ["CMD-SHELL", "bash -c 'exec 3<>/dev/tcp/localhost/19100' || exit 1"] + interval: 30s + timeout: 10s + retries: 3 + start_period: 60s + depends_on: + shard-init: + condition: service_completed_successfully + ulimits: + nofile: + soft: ${ULIMIT_NOFILE:-524288} + hard: ${ULIMIT_NOFILE:-524288} + deploy: + resources: + limits: + cpus: "${LIMIT_CPUS_SHARD_5:-2.00}" + memory: ${LIMIT_MEM_SHARD_5:-6G} + + shard-6: + image: >- + ${LINERA_VALIDATOR_IMAGE:-us-docker.pkg.dev/linera-io-dev/linera-public-registry/linera-validator:testnet_conway_release} + container_name: docker-shard-7 + hostname: docker-shard-7 + restart: unless-stopped + logging: *default-logging + cpuset: "${WORKLOAD_CPUSET:-}" + command: + - /linera-server + - run + - --storage + - scylladb:tcp:scylla:${SCYLLA_PORT:-9042} + - --server + - /config/server.json + - --shard + - "6" + - --storage-replication-factor + - "${STORAGE_REPLICATION_FACTOR:-1}" + # Storage (LRU) caches — parity with the k8s chart's shards.cli. + - --storage-max-cache-size + - "${LINERA_STORAGE_MAX_CACHE_SIZE:-1000000000}" + - --storage-max-cache-entries + - "${LINERA_STORAGE_MAX_CACHE_ENTRIES:-500000}" + - --storage-max-value-entry-size + - "${LINERA_STORAGE_MAX_VALUE_ENTRY_SIZE:-1000000}" + - --storage-max-find-keys-entry-size + - "${LINERA_STORAGE_MAX_FIND_KEYS_ENTRY_SIZE:-1000000}" + - --storage-max-find-key-values-entry-size + - "${LINERA_STORAGE_MAX_FIND_KEY_VALUES_ENTRY_SIZE:-1000000}" + - --storage-max-cache-value-size + - "${LINERA_STORAGE_MAX_CACHE_VALUE_SIZE:-500000000}" + - --storage-max-cache-find-keys-size + - "${LINERA_STORAGE_MAX_CACHE_FIND_KEYS_SIZE:-200000000}" + - --storage-max-cache-find-key-values-size + - "${LINERA_STORAGE_MAX_CACHE_FIND_KEY_VALUES_SIZE:-10000000}" + # In-memory caches — parity with the k8s chart's shards.cli. + - --blob-cache-size + - "${LINERA_BLOB_CACHE_SIZE:-1000}" + - --confirmed-block-cache-size + - "${LINERA_CONFIRMED_BLOCK_CACHE_SIZE:-10000}" + - --certificate-cache-size + - "${LINERA_CERTIFICATE_CACHE_SIZE:-5000}" + - --certificate-raw-cache-size + - "${LINERA_CERTIFICATE_RAW_CACHE_SIZE:-50000}" + - --event-cache-size + - "${LINERA_EVENT_CACHE_SIZE:-20000}" + - --execution-state-cache-size + - "${LINERA_EXECUTION_STATE_CACHE_SIZE:-20000}" + - --block-cache-size + - "${LINERA_BLOCK_CACHE_SIZE:-20000}" + - --chain-worker-ttl-ms + - "${LINERA_CHAIN_WORKER_TTL_MS:-300000}" + profiles: + - shard-6 + volumes: + - .:/config + labels: + com.centurylinklabs.watchtower.enable: "true" + linera.shard: "6" + # Same probe and timings as the chart's shard livenessProbe. + healthcheck: + test: ["CMD-SHELL", "bash -c 'exec 3<>/dev/tcp/localhost/19100' || exit 1"] + interval: 30s + timeout: 10s + retries: 3 + start_period: 60s + depends_on: + shard-init: + condition: service_completed_successfully + ulimits: + nofile: + soft: ${ULIMIT_NOFILE:-524288} + hard: ${ULIMIT_NOFILE:-524288} + deploy: + resources: + limits: + cpus: "${LIMIT_CPUS_SHARD_6:-2.00}" + memory: ${LIMIT_MEM_SHARD_6:-6G} + + shard-7: + image: >- + ${LINERA_VALIDATOR_IMAGE:-us-docker.pkg.dev/linera-io-dev/linera-public-registry/linera-validator:testnet_conway_release} + container_name: docker-shard-8 + hostname: docker-shard-8 + restart: unless-stopped + logging: *default-logging + cpuset: "${WORKLOAD_CPUSET:-}" + command: + - /linera-server + - run + - --storage + - scylladb:tcp:scylla:${SCYLLA_PORT:-9042} + - --server + - /config/server.json + - --shard + - "7" + - --storage-replication-factor + - "${STORAGE_REPLICATION_FACTOR:-1}" + # Storage (LRU) caches — parity with the k8s chart's shards.cli. + - --storage-max-cache-size + - "${LINERA_STORAGE_MAX_CACHE_SIZE:-1000000000}" + - --storage-max-cache-entries + - "${LINERA_STORAGE_MAX_CACHE_ENTRIES:-500000}" + - --storage-max-value-entry-size + - "${LINERA_STORAGE_MAX_VALUE_ENTRY_SIZE:-1000000}" + - --storage-max-find-keys-entry-size + - "${LINERA_STORAGE_MAX_FIND_KEYS_ENTRY_SIZE:-1000000}" + - --storage-max-find-key-values-entry-size + - "${LINERA_STORAGE_MAX_FIND_KEY_VALUES_ENTRY_SIZE:-1000000}" + - --storage-max-cache-value-size + - "${LINERA_STORAGE_MAX_CACHE_VALUE_SIZE:-500000000}" + - --storage-max-cache-find-keys-size + - "${LINERA_STORAGE_MAX_CACHE_FIND_KEYS_SIZE:-200000000}" + - --storage-max-cache-find-key-values-size + - "${LINERA_STORAGE_MAX_CACHE_FIND_KEY_VALUES_SIZE:-10000000}" + # In-memory caches — parity with the k8s chart's shards.cli. + - --blob-cache-size + - "${LINERA_BLOB_CACHE_SIZE:-1000}" + - --confirmed-block-cache-size + - "${LINERA_CONFIRMED_BLOCK_CACHE_SIZE:-10000}" + - --certificate-cache-size + - "${LINERA_CERTIFICATE_CACHE_SIZE:-5000}" + - --certificate-raw-cache-size + - "${LINERA_CERTIFICATE_RAW_CACHE_SIZE:-50000}" + - --event-cache-size + - "${LINERA_EVENT_CACHE_SIZE:-20000}" + - --execution-state-cache-size + - "${LINERA_EXECUTION_STATE_CACHE_SIZE:-20000}" + - --block-cache-size + - "${LINERA_BLOCK_CACHE_SIZE:-20000}" + - --chain-worker-ttl-ms + - "${LINERA_CHAIN_WORKER_TTL_MS:-300000}" + profiles: + - shard-7 + volumes: + - .:/config + labels: + com.centurylinklabs.watchtower.enable: "true" + linera.shard: "7" + # Same probe and timings as the chart's shard livenessProbe. + healthcheck: + test: ["CMD-SHELL", "bash -c 'exec 3<>/dev/tcp/localhost/19100' || exit 1"] + interval: 30s + timeout: 10s + retries: 3 + start_period: 60s + depends_on: + shard-init: + condition: service_completed_successfully + ulimits: + nofile: + soft: ${ULIMIT_NOFILE:-524288} + hard: ${ULIMIT_NOFILE:-524288} + deploy: + resources: + limits: + cpus: "${LIMIT_CPUS_SHARD_7:-2.00}" + memory: ${LIMIT_MEM_SHARD_7:-6G} + # --------------------------------------------------------------------------- # Watchtower — periodically polls the registry and pulls new images for # any container labelled `com.centurylinklabs.watchtower.enable=true`. diff --git a/docker/prometheus.yaml b/docker/prometheus.yaml index 1a0ec75..e6f449c 100644 --- a/docker/prometheus.yaml +++ b/docker/prometheus.yaml @@ -10,8 +10,8 @@ rule_files: # Job names and the `shard` label match alloy-config.river exactly, so the # dashboards work whichever of the two scrape paths you run. `shard` is not -# emitted by the binaries — it exists only because the scrape config attaches -# it, and the per-shard panels group by it. +# emitted by the binaries — both paths relabel it from each shard container's +# linera.shard label, and the per-shard panels group by it. scrape_configs: # Alloy's own metrics (exposes a "targets" page at :12345). - job_name: 'alloy' @@ -35,18 +35,23 @@ scrape_configs: static_configs: - targets: ['scylla:9180'] - # Linera shards. + # Linera shards, discovered from the Docker daemon so the shard count is + # configurable without editing this file. A static list would alert on the + # shards a smaller deployment does not run. - job_name: 'linera-shard' - static_configs: - - targets: ['docker-shard-1:21100'] - labels: - shard: '0' - - targets: ['docker-shard-2:21100'] - labels: - shard: '1' - - targets: ['docker-shard-3:21100'] - labels: - shard: '2' - - targets: ['docker-shard-4:21100'] - labels: - shard: '3' + docker_sd_configs: + - host: unix:///var/run/docker.sock + port: 21100 + relabel_configs: + # Every shard service carries linera.shard; nothing else does. + - source_labels: [__meta_docker_container_label_linera_shard] + regex: .+ + action: keep + # The image declares no EXPOSE, so `port` above already yields one + # target per container. Pin the address anyway: an EXPOSE added + # upstream would otherwise produce a target per exposed port. + - source_labels: [__meta_docker_network_ip] + target_label: __address__ + replacement: '$1:21100' + - source_labels: [__meta_docker_container_label_linera_shard] + target_label: shard diff --git a/docs/DOCKER-COMPOSE.md b/docs/DOCKER-COMPOSE.md index facf746..bf04d94 100644 --- a/docs/DOCKER-COMPOSE.md +++ b/docs/DOCKER-COMPOSE.md @@ -54,7 +54,7 @@ Options: linera-proxy, and generates the validator key). --client-image REF Override the linera-client image (runs the `linera` CLI). --xfs-path PATH Bind-mount this XFS dir for ScyllaDB data. ---num-shards N Number of shards (must match docker-compose.yaml services). +--num-shards N Number of shards, 4-8 (default: 8). Fixed after the first deploy. --dry-run Print what would happen, change nothing. ``` @@ -341,6 +341,35 @@ disks, and NICs with the shards and proxy. The `SCYLLA_CPUSET` / hard limit of single-host co-tenancy. For full isolation and HA, use the Helm path with a real ScyllaDB cluster. +## Changing the shard count + +`--num-shards` takes 4 to 8 and defaults to 8, matching testnet-conway. Shard +assignment is internal to each validator — `hash(validator_public_key, chain_id) +% num_shards` — so the count is a capacity choice, not something the network +agrees on. Different validators can and do run different numbers. + +`deploy-validator.sh` writes both `NUM_SHARDS` and the matching +`COMPOSE_PROFILES` into `.env`. Shards 0–3 have no profile and always run; 4–7 +are enabled by their profile, so a deployment that predates this and has no +`COMPOSE_PROFILES` keeps the four it already had. + +**The count is fixed once you deploy.** It lives in `server.json`, which also +holds your signing key and is never regenerated — rotating that key would drop +your validator from the committee. So `deploy-validator.sh` reads the count back +out of `server.json` on every re-run and refuses a `--num-shards` that +disagrees, rather than starting shards that index past the end of the list +`server.json` still holds. Those shards panic on boot. + +Growing an existing validator therefore means regenerating the network half of +`server.json` while preserving `validator_secret`, then restarting the stack. +There is no supported automation for it yet — [open an +issue](https://github.com/linera-io/linera-artifacts/issues) if you need it and +we will document the procedure. + +Nothing else has to change: Prometheus and Alloy discover shards through the +Docker daemon and read the index off each container's `linera.shard` label, so +neither `prometheus.yaml` nor `alloy-config.river` names a shard. + ## Upgrading .env safely When a new release adds configuration variables, merge them in without diff --git a/docs/HARDWARE.md b/docs/HARDWARE.md index f30ec6c..3c16564 100644 --- a/docs/HARDWARE.md +++ b/docs/HARDWARE.md @@ -96,7 +96,7 @@ The compose stack runs everything the two reference machines run — on one kernel: - ScyllaDB (the dominant resource consumer) -- 4 `linera-server` shards +- 8 `linera-server` shards (the testnet-conway count; see below) - `linera-proxy` - Caddy - Watchtower + cAdvisor + (optionally) a monitoring stack @@ -108,7 +108,7 @@ the shards, you need the sum of the two reference machines: | Service | CPU limit | Memory limit | Reference equivalent | |-------------|------------|----------------------------|------------------------------------| | ScyllaDB | 7 | 51 GiB | 7c / 51 GiB pod on a dedicated node | -| Shards × 4 | 2 each | 12 GiB each (48 GiB total) | shards on the workload node | +| Shards × 8 | 2 each | 6 GiB each (48 GiB total) | shards on the workload node | | Proxy | 1.5 | 6 GiB | proxy on the workload node | | Caddy (web) | 1 | 3 GiB | — | | Alloy | 0.5 | 1 GiB | — | @@ -119,6 +119,14 @@ Set `LIMIT_CPUS_SCYLLA=7` and `LIMIT_MEM_SCYLLA=51G` in `.env` to get this — the shipped defaults are conservative and **must** be raised to these values on a properly sized host. +Shard count is set once, at deploy time, with `--num-shards` (4–8, +default 8 to match testnet-conway). It is the **total shard memory** +that has to fit, not the per-shard figure: 8 × 6 GiB and 4 × 12 GiB both +come to the same 48 GiB, so a smaller count buys a bigger per-shard +cache rather than a smaller host. Changing the count afterwards is a +resharding, not a config edit — see +[Changing the shard count](DOCKER-COMPOSE.md#changing-the-shard-count). + Two things to know about this table: - **Memory must add up; CPU may oversubscribe.** The memory column diff --git a/scripts/deploy-validator.sh b/scripts/deploy-validator.sh index 3238c4f..da74e30 100755 --- a/scripts/deploy-validator.sh +++ b/scripts/deploy-validator.sh @@ -23,9 +23,10 @@ # --client-image REF Override the linera-client image reference. Runs the # `linera` CLI that shard-init invokes. # --xfs-path PATH Bind-mount this XFS directory as the ScyllaDB data dir. -# --num-shards N Number of shards in validator-config.toml (default: 4). -# Must match the number of shard-N services in -# docker-compose.yaml. +# --num-shards N Number of shards, 4..8 (default: 8, matching +# testnet-conway). Ignored when server.json already +# exists: that file pins the count, and changing it +# needs a resharding, not a re-run. # --with-alloy Add Grafana Alloy + cAdvisor; push metrics/ # logs/traces to remote endpoints you set in # .env (PROMETHEUS_OTLP_URL etc.). What Linera @@ -58,7 +59,10 @@ readonly DEFAULT_CLIENT_IMAGE_NAME="linera-client" readonly DEFAULT_IMAGE_TAG="testnet_conway_release" readonly DEFAULT_GENESIS_BUCKET="https://storage.googleapis.com/linera-io-dev-public" readonly DEFAULT_NETWORK="testnet-conway" -readonly DEFAULT_NUM_SHARDS=4 +readonly DEFAULT_NUM_SHARDS=8 +# shard-0..3 have no compose profile, so they always run and cannot be +# switched off. Anything above that is opt-in via COMPOSE_PROFILES. +readonly MIN_NUM_SHARDS=4 readonly RED='\033[0;31m' readonly GREEN='\033[0;32m' @@ -113,14 +117,47 @@ validate_num_shards() { [[ "$requested" =~ ^[1-9][0-9]*$ ]] || die "--num-shards must be a positive integer, got: ${requested}" [[ -f "$compose" ]] || die "Missing ${compose}" local available - available="$(grep -cE '^[[:space:]]+hostname: docker-shard-[0-9]+$' "$compose")" - if [[ "$requested" -ne "$available" ]]; then - die "--num-shards is ${requested} but docker-compose.yaml defines ${available} shard services. - They must match, or the validator will address shards that do not exist. - Either pass --num-shards ${available}, or add/remove shard services in ${compose}." + available="$(compose_shard_services)" + if (( requested < MIN_NUM_SHARDS || requested > available )); then + die "--num-shards is ${requested}; docker-compose.yaml supports ${MIN_NUM_SHARDS}..${available}. + shard-0..$((MIN_NUM_SHARDS - 1)) always run; the rest are enabled through COMPOSE_PROFILES. + Add more shard-N services in ${compose} to raise the ceiling." fi } +compose_shard_services() { + grep -cE '^[[:space:]]+hostname: docker-shard-[0-9]+$' "${COMPOSE_DIR}/docker-compose.yaml" +} + +# The shard list lives in server.json, and generate_validator_keys refuses to +# regenerate that file so the signing key survives a re-run. Changing the count +# on an existing validator therefore leaves server.json behind, and a shard +# started with an index past its end panics on boot. Treat what is already +# deployed as authoritative. +shards_in_server_json() { + local f="${COMPOSE_DIR}/server.json" + [[ -f "$f" ]] || return 0 + # Refuse rather than fall through to the default: guessing here migrates a + # running validator to a shard count its server.json cannot serve. + command -v jq >/dev/null 2>&1 \ + || die "jq is required to read the shard count out of ${f}. Install jq and re-run." + # `empty` rather than 0, so "no shard list" stays distinguishable from + # "zero shards" and falls through to the default. + jq -r 'if .internal_network.shards then .internal_network.shards | length else empty end' \ + "$f" 2>/dev/null || true +} + +# COMPOSE_PROFILES is read from .env by docker compose itself. shard-0..3 carry +# no profile, so only the extras above MIN_NUM_SHARDS are listed. +shard_profiles_for() { + local num_shards="$1" profiles=() i + for (( i = MIN_NUM_SHARDS; i < num_shards; i++ )); do + profiles+=("shard-${i}") + done + local IFS=, + echo "${profiles[*]}" +} + # A validator opens many short-lived gRPC/DNS connections. A low # netfilter conntrack limit fills up and the kernel silently drops # packets (including DNS), surfacing as cross-chain failures. Heavily @@ -318,6 +355,7 @@ build_env() { set_or_replace LINERA_VALIDATOR_IMAGE "$validator_image" set_or_replace LINERA_CLIENT_IMAGE "$client_image" set_or_replace NUM_SHARDS "$num_shards" + set_or_replace COMPOSE_PROFILES "$(shard_profiles_for "$num_shards")" # Refresh the deployment metadata block, stripping the previous one by the # lines it owns. Deleting through to end-of-file instead would take any @@ -346,7 +384,7 @@ main() { echo " Use LINERA_VALIDATOR_IMAGE and LINERA_CLIENT_IMAGE instead." >&2 exit 1 fi - local num_shards="${NUM_SHARDS:-$DEFAULT_NUM_SHARDS}" + local num_shards="${NUM_SHARDS:-}" local xfs_path="${SCYLLA_XFS_PATH:-}" local with_alloy="${WITH_ALLOY:-0}" local with_local_monitoring="${WITH_LOCAL_MONITORING:-0}" @@ -383,6 +421,26 @@ main() { [[ -n "$email" ]] || { usage; die "email is required"; } validate_host "$host" validate_email "$email" + + # An existing server.json pins the count: the default must never migrate a + # running validator, and an explicit request that disagrees is refused + # rather than applied, because applying it panics every shard past the end + # of the list server.json still holds. + local deployed_shards + deployed_shards="$(shards_in_server_json)" + if [[ -n "$deployed_shards" ]]; then + if [[ -z "$num_shards" ]]; then + num_shards="$deployed_shards" + log INFO "Keeping the ${deployed_shards} shards already in server.json." + elif [[ "$num_shards" -ne "$deployed_shards" ]]; then + die "Requested ${num_shards} shards but server.json describes ${deployed_shards}. + Changing the count needs server.json regenerated, which would rotate your + signing key and drop this validator from the committee. + Re-run without --num-shards to keep ${deployed_shards}, or see + docs/DOCKER-COMPOSE.md for the resharding procedure." + fi + fi + [[ -n "$num_shards" ]] || num_shards="$DEFAULT_NUM_SHARDS" validate_num_shards "$num_shards" require_cmd docker diff --git a/tests/deploy-validator-test.sh b/tests/deploy-validator-test.sh index f1c3dea..fce633a 100755 --- a/tests/deploy-validator-test.sh +++ b/tests/deploy-validator-test.sh @@ -83,7 +83,7 @@ assert_eq "DOMAIN" "v.example.com" "$(env_value "$d" DOMAIN)" assert_eq "ACME_EMAIL" "ops@example.com" "$(env_value "$d" ACME_EMAIL)" assert_eq "VALIDATOR_NAME" "v.example.com" "$(env_value "$d" VALIDATOR_NAME)" assert_eq "HOSTNAME" "v.example.com" "$(env_value "$d" HOSTNAME)" -assert_eq "NUM_SHARDS" "4" "$(env_value "$d" NUM_SHARDS)" +assert_eq "NUM_SHARDS" "8" "$(env_value "$d" NUM_SHARDS)" assert_eq "VALIDATOR_KEY" "02aabb,00ccdd" "$(env_value "$d" VALIDATOR_KEY)" assert_eq "LINERA_VALIDATOR_IMAGE" "${REGISTRY}/linera-validator:${DEFAULT_TAG}" \ "$(env_value "$d" LINERA_VALIDATOR_IMAGE)" @@ -111,16 +111,19 @@ rm -rf "$d" # --num-shards used to be accepted unchecked, writing a config that addressed # containers the stack never creates. -start_case "--num-shards must match the compose shard count" +start_case "--num-shards is held to what the compose stack can serve" d="$(new_sandbox)" compose_shards="$(grep -cE '^[[:space:]]+hostname: docker-shard-[0-9]+$' "$d/docker/docker-compose.yaml")" -if run_deploy "$d" v.example.com ops@example.com --skip-genesis --num-shards $((compose_shards + 4)); then - fail "too many shards was accepted" +if run_deploy "$d" v.example.com ops@example.com --skip-genesis --num-shards $((compose_shards + 1)); then + fail "more shards than compose defines was accepted" else - assert_contains "error names both counts" "docker-compose.yaml defines ${compose_shards}" "$d/stderr.log" + assert_contains "error names the range" "supports 4..${compose_shards}" "$d/stderr.log" fi -if run_deploy "$d" v.example.com ops@example.com --skip-genesis --num-shards $((compose_shards - 1)); then - fail "too few shards was accepted" +# shard-0..3 carry no profile, so they always run and a smaller count is a lie. +if run_deploy "$d" v.example.com ops@example.com --skip-genesis --num-shards 3; then + fail "fewer shards than always run was accepted" +else + assert_contains "error names the floor" "supports 4..${compose_shards}" "$d/stderr.log" fi if run_deploy "$d" v.example.com ops@example.com --skip-genesis --num-shards 0; then fail "zero shards was accepted" @@ -128,9 +131,12 @@ fi if run_deploy "$d" v.example.com ops@example.com --skip-genesis --num-shards abc; then fail "non-numeric shard count was accepted" fi -if ! run_deploy "$d" v.example.com ops@example.com --skip-genesis --num-shards "$compose_shards"; then - fail "the matching shard count was rejected" -fi +# Anything inside the range is fine, including a count below the default. +for n in 4 6 "$compose_shards"; do + if ! run_deploy "$d" v.example.com ops@example.com --skip-genesis --num-shards "$n"; then + fail "--num-shards ${n} was rejected" + fi +done rm -rf "$d" # --- image selection ------------------------------------------------------- @@ -214,7 +220,7 @@ printf 'DOMAIN=old.example.com\nMY_CUSTOM_TUNING=keepme\n' > "$d/docker/.env" run_deploy "$d" v.example.com ops@example.com --skip-genesis run_deploy "$d" v.example.com ops@example.com --skip-genesis assert_eq "custom value survives" "keepme" "$(env_value "$d" MY_CUSTOM_TUNING)" -assert_eq "NUM_SHARDS written" "4" "$(env_value "$d" NUM_SHARDS)" +assert_eq "NUM_SHARDS written" "8" "$(env_value "$d" NUM_SHARDS)" assert_eq "validator image written" "${REGISTRY}/linera-validator:${DEFAULT_TAG}" \ "$(env_value "$d" LINERA_VALIDATOR_IMAGE)" rm -rf "$d" @@ -244,6 +250,39 @@ if run_deploy "$d" 'not a host' ops@example.com --skip-genesis; then fail "inval if run_deploy "$d" v.example.com 'not-an-email' --skip-genesis; then fail "invalid email accepted"; fi rm -rf "$d" +# --- an already-deployed validator keeps its shard count ------------------- +# The default moved from 4 to 8. server.json holds the shard list and is never +# regenerated (that would rotate the signing key), so a re-run that migrated an +# existing validator to 8 would start shards indexing past the end of that list +# and panic every one of them on boot. +start_case "a re-run never changes the shard count in server.json" +d="$(new_sandbox)" +jq '.internal_network = {shards: [range(1;5) | {host: ("docker-shard-" + (. | tostring)), port: 19100}]}' \ + "$d/docker/server.json" > "$d/server.tmp" && mv "$d/server.tmp" "$d/docker/server.json" +run_deploy "$d" v.example.com ops@example.com --skip-genesis +assert_eq "NUM_SHARDS pinned by server.json" "4" "$(env_value "$d" NUM_SHARDS)" +assert_eq "no extra shard profiles enabled" "" "$(env_value "$d" COMPOSE_PROFILES)" +assert_eq "validator-config.toml stays at 4" "4" \ + "$(grep -c '^\[\[shards\]\]' "$d/docker/validator-config.toml")" +# Asking for a different count is refused, not silently applied. +if run_deploy "$d" v.example.com ops@example.com --skip-genesis --num-shards 8; then + fail "--num-shards 8 was accepted against a 4-shard server.json" +else + assert_contains "mismatch is explained" "server.json describes 4" "$d/stderr.log" +fi +assert_eq "NUM_SHARDS unchanged after refusal" "4" "$(env_value "$d" NUM_SHARDS)" +rm -rf "$d" + +# --- a fresh deployment gets the testnet-conway shard count ---------------- +start_case "a fresh deployment defaults to 8 shards and enables their profiles" +d="$(new_sandbox)" +run_deploy "$d" v.example.com ops@example.com --skip-genesis +assert_eq "NUM_SHARDS" "8" "$(env_value "$d" NUM_SHARDS)" +assert_eq "profiles for shards 4..7" "shard-4,shard-5,shard-6,shard-7" \ + "$(env_value "$d" COMPOSE_PROFILES)" +rm -rf "$d" + + echo if [ "$failures" -eq 0 ]; then echo "all deploy-validator tests passed" From 2dfb9b37aacfa368c7fb9d850b4b0531746df101 Mon Sep 17 00:00:00 2001 From: Andre da Silva Date: Mon, 31 Aug 2026 22:10:36 +0000 Subject: [PATCH 2/2] deploy script: never guess the shard count when server.json is unreadable The jq guard checked that jq exists, then ran it with 2>/dev/null || true. A jq that is present but fails - wrong version, broken install, anything non-zero - therefore produced an empty count, fell through to DEFAULT_NUM_SHARDS, and migrated a running 4-shard validator to 8. Exit 0, no warning. That is exactly the migration the pinning exists to prevent. Found by running the script with a jq stub that exits 127: NUM_SHARDS went from 4 to 8 silently. Now the presence check and the parse are separate, neither swallows failure, and an unreadable server.json aborts. Regression test covers the parse failure; the jq-absent branch was verified by hand against a PATH with no jq on it. --- scripts/deploy-validator.sh | 25 +++++++++++++++---------- tests/deploy-validator-test.sh | 15 +++++++++++++++ 2 files changed, 30 insertions(+), 10 deletions(-) diff --git a/scripts/deploy-validator.sh b/scripts/deploy-validator.sh index da74e30..f00ff49 100755 --- a/scripts/deploy-validator.sh +++ b/scripts/deploy-validator.sh @@ -135,16 +135,12 @@ compose_shard_services() { # started with an index past its end panics on boot. Treat what is already # deployed as authoritative. shards_in_server_json() { - local f="${COMPOSE_DIR}/server.json" - [[ -f "$f" ]] || return 0 - # Refuse rather than fall through to the default: guessing here migrates a - # running validator to a shard count its server.json cannot serve. - command -v jq >/dev/null 2>&1 \ - || die "jq is required to read the shard count out of ${f}. Install jq and re-run." # `empty` rather than 0, so "no shard list" stays distinguishable from - # "zero shards" and falls through to the default. + # "zero shards". Errors are NOT swallowed: a failure here must reach the + # caller, because falling through to the default migrates a running + # validator to a shard count its server.json cannot serve. jq -r 'if .internal_network.shards then .internal_network.shards | length else empty end' \ - "$f" 2>/dev/null || true + "${COMPOSE_DIR}/server.json" } # COMPOSE_PROFILES is read from .env by docker compose itself. shard-0..3 carry @@ -426,8 +422,17 @@ main() { # running validator, and an explicit request that disagrees is refused # rather than applied, because applying it panics every shard past the end # of the list server.json still holds. - local deployed_shards - deployed_shards="$(shards_in_server_json)" + local deployed_shards="" + if [[ -f "${COMPOSE_DIR}/server.json" ]]; then + command -v jq >/dev/null 2>&1 \ + || die "jq is required to read the shard count out of ${COMPOSE_DIR}/server.json. + Without it this script cannot tell how many shards you already run. + Install jq and re-run." + deployed_shards="$(shards_in_server_json)" \ + || die "Could not read the shard list from ${COMPOSE_DIR}/server.json. + It should be the file linera-server generate wrote. Refusing to guess the + shard count: guessing wrong starts shards that panic on boot." + fi if [[ -n "$deployed_shards" ]]; then if [[ -z "$num_shards" ]]; then num_shards="$deployed_shards" diff --git a/tests/deploy-validator-test.sh b/tests/deploy-validator-test.sh index fce633a..dba273e 100755 --- a/tests/deploy-validator-test.sh +++ b/tests/deploy-validator-test.sh @@ -273,6 +273,21 @@ fi assert_eq "NUM_SHARDS unchanged after refusal" "4" "$(env_value "$d" NUM_SHARDS)" rm -rf "$d" +# --- an unreadable server.json is fatal, never a fallthrough --------------- +# Swallowing this error silently defaults to DEFAULT_NUM_SHARDS, which is the +# migration the case above exists to prevent — so the failure has to be loud. +start_case "a server.json that cannot be read refuses rather than guesses" +d="$(new_sandbox)" +printf '%s\n' 'NUM_SHARDS=4' > "$d/docker/.env" +printf 'not json at all\n' > "$d/docker/server.json" +if run_deploy "$d" v.example.com ops@example.com --skip-genesis; then + fail "an unparseable server.json was accepted" +else + assert_contains "refusal is explained" "Could not read the shard list" "$d/stderr.log" +fi +assert_eq "NUM_SHARDS not migrated" "4" "$(env_value "$d" NUM_SHARDS)" +rm -rf "$d" + # --- a fresh deployment gets the testnet-conway shard count ---------------- start_case "a fresh deployment defaults to 8 shards and enables their profiles" d="$(new_sandbox)"