Compare commits

..

16 Commits

Author SHA1 Message Date
Alex Auvolat dbe457d3fa Improve clarity 2021-10-21 17:43:42 +02:00
Alex Auvolat bff5333c62 Move things around, improvements to CLI 2021-10-21 17:29:27 +02:00
Alex Auvolat b7eccf5264 update cargo.nix 2021-10-21 14:10:11 +02:00
Alex Auvolat 2fa65554f1 Persist peer list to file 2021-10-21 13:45:52 +02:00
Alex Auvolat 1dc7cc7936 update netapp to fix connection bug 2021-10-21 12:34:05 +02:00
Alex Auvolat 5e986c443f Discovery via consul 2021-10-21 00:29:50 +02:00
Alex Auvolat 6455347955 Update documentation 2021-10-20 16:10:18 +02:00
Alex Auvolat c18920f8d1 run cargo2nix 2021-10-20 00:00:28 +02:00
Alex Auvolat b2d24b2b5b bump garage version to 0.4.0 2021-10-19 23:59:05 +02:00
Alex Auvolat f1a68f6b57 Fix clippy 2021-10-19 23:53:49 +02:00
Alex Auvolat d9c52e9a9c Update documentation: real_world.md 2021-10-19 23:40:21 +02:00
Alex Auvolat 550ce7db2a Update doc: quick_start 2021-10-19 23:39:47 +02:00
Alex Auvolat 12190efd41 Everything works, actually! 2021-10-19 23:39:47 +02:00
Alex Auvolat a8ae78af0a Adapt tests to new syntax with public keys 2021-10-19 23:39:45 +02:00
Alex Auvolat 65070f3c05 Improvements to CLI and various fixes for netapp version 2021-10-19 23:38:42 +02:00
Alex Auvolat e6da0dc900 First port of Garage to Netapp 2021-10-19 23:38:38 +02:00
9 changed files with 434 additions and 585 deletions
+210 -210
View File
@@ -276,115 +276,115 @@ trigger:
node: node:
nix: 1 nix: 1
# --- ---
# kind: pipeline kind: pipeline
# type: docker type: docker
# name: release-linux-i686 name: release-linux-i686
#
# volumes: volumes:
# - name: nix_store - name: nix_store
# host: host:
# path: /var/lib/drone/nix path: /var/lib/drone/nix
# - name: nix_config - name: nix_config
# temp: {} temp: {}
#
# environment: environment:
# TARGET: i686-unknown-linux-musl TARGET: i686-unknown-linux-musl
#
# steps: steps:
# - name: setup nix - name: setup nix
# image: nixpkgs/nix:nixos-21.05 image: nixpkgs/nix:nixos-21.05
# volumes: volumes:
# - name: nix_store - name: nix_store
# path: /nix path: /nix
# - name: nix_config - name: nix_config
# path: /etc/nix path: /etc/nix
# commands: commands:
# - cp nix/nix.conf /etc/nix/nix.conf - cp nix/nix.conf /etc/nix/nix.conf
# - nix-build --no-build-output --no-out-link shell.nix -A inputDerivation - nix-build --no-build-output --no-out-link shell.nix -A inputDerivation
#
# - name: build - name: build
# image: nixpkgs/nix:nixos-21.05 image: nixpkgs/nix:nixos-21.05
# volumes: volumes:
# - name: nix_store - name: nix_store
# path: /nix path: /nix
# - name: nix_config - name: nix_config
# path: /etc/nix path: /etc/nix
# commands: commands:
# - nix-build --no-build-output --argstr target $TARGET --arg release true --argstr git_version $DRONE_COMMIT - nix-build --no-build-output --argstr target $TARGET --arg release true --argstr git_version $DRONE_COMMIT
#
# - name: integration - name: integration
# image: nixpkgs/nix:nixos-21.05 image: nixpkgs/nix:nixos-21.05
# volumes: volumes:
# - name: nix_store - name: nix_store
# path: /nix path: /nix
# - name: nix_config - name: nix_config
# path: /etc/nix path: /etc/nix
# commands: commands:
# - nix-shell --run ./script/test-smoke.sh || (cat /tmp/garage.log; false) - nix-shell --run ./script/test-smoke.sh || (cat /tmp/garage.log; false)
#
# - name: update cache - name: update cache
# image: nixpkgs/nix:nixos-21.05 image: nixpkgs/nix:nixos-21.05
# environment: environment:
# AWS_ACCESS_KEY_ID: AWS_ACCESS_KEY_ID:
# from_secret: cache_aws_access_key_id from_secret: cache_aws_access_key_id
# AWS_SECRET_ACCESS_KEY: AWS_SECRET_ACCESS_KEY:
# from_secret: cache_aws_secret_access_key from_secret: cache_aws_secret_access_key
# NIX_PRIV_KEY: NIX_PRIV_KEY:
# from_secret: nix_priv_key from_secret: nix_priv_key
# volumes: volumes:
# - name: nix_store - name: nix_store
# path: /nix path: /nix
# - name: nix_config - name: nix_config
# path: /etc/nix path: /etc/nix
# commands: commands:
# - (umask 377 && echo $NIX_PRIV_KEY > /etc/nix/signing-key.sec) - (umask 377 && echo $NIX_PRIV_KEY > /etc/nix/signing-key.sec)
# - | - |
# nix copy --to 's3://nix?endpoint=garage.deuxfleurs.fr&region=garage&secret-key=/etc/nix/signing-key.sec' \ nix copy --to 's3://nix?endpoint=garage.deuxfleurs.fr&region=garage&secret-key=/etc/nix/signing-key.sec' \
# $(nix-store -qR --include-outputs \ $(nix-store -qR --include-outputs \
# $(nix-instantiate --argstr target $TARGET --arg release true)) $(nix-instantiate --argstr target $TARGET --arg release true))
#
# - name: push static binary - name: push static binary
# image: nixpkgs/nix:nixos-21.05 image: nixpkgs/nix:nixos-21.05
# volumes: volumes:
# - name: nix_store - name: nix_store
# path: /nix path: /nix
# - name: nix_config - name: nix_config
# path: /etc/nix path: /etc/nix
# environment: environment:
# AWS_ACCESS_KEY_ID: AWS_ACCESS_KEY_ID:
# from_secret: garagehq_aws_access_key_id from_secret: garagehq_aws_access_key_id
# AWS_SECRET_ACCESS_KEY: AWS_SECRET_ACCESS_KEY:
# from_secret: garagehq_aws_secret_access_key from_secret: garagehq_aws_secret_access_key
# commands: commands:
# - nix-shell --arg rust false --arg integration false --run "to_s3" - nix-shell --arg rust false --arg integration false --run "to_s3"
#
# - name: docker build and publish - name: docker build and publish
# image: nixpkgs/nix:nixos-21.05 image: nixpkgs/nix:nixos-21.05
# volumes: volumes:
# - name: nix_store - name: nix_store
# path: /nix path: /nix
# - name: nix_config - name: nix_config
# path: /etc/nix path: /etc/nix
# environment: environment:
# DOCKER_AUTH: DOCKER_AUTH:
# from_secret: docker_auth from_secret: docker_auth
# DOCKER_PLATFORM: "linux/386" DOCKER_PLATFORM: "linux/386"
# CONTAINER_NAME: "dxflrs/386_garage" CONTAINER_NAME: "dxflrs/386_garage"
# HOME: "/kaniko" HOME: "/kaniko"
# commands: commands:
# - mkdir -p /kaniko/.docker - mkdir -p /kaniko/.docker
# - echo $DOCKER_AUTH > /kaniko/.docker/config.json - echo $DOCKER_AUTH > /kaniko/.docker/config.json
# - export CONTAINER_TAG=${DRONE_TAG:-$DRONE_COMMIT} - export CONTAINER_TAG=${DRONE_TAG:-$DRONE_COMMIT}
# - nix-shell --arg rust false --arg integration false --run "to_docker" - nix-shell --arg rust false --arg integration false --run "to_docker"
#
# trigger: trigger:
# event: event:
# - promote - promote
# - cron - cron
#
# node: node:
# nix: 1 nix: 1
--- ---
kind: pipeline kind: pipeline
@@ -486,105 +486,105 @@ trigger:
node: node:
nix: 1 nix: 1
# --- ---
# kind: pipeline kind: pipeline
# type: docker type: docker
# name: release-linux-armv6l name: release-linux-armv6l
#
# volumes: volumes:
# - name: nix_store - name: nix_store
# host: host:
# path: /var/lib/drone/nix path: /var/lib/drone/nix
# - name: nix_config - name: nix_config
# temp: {} temp: {}
#
# environment: environment:
# TARGET: armv6l-unknown-linux-musleabihf TARGET: armv6l-unknown-linux-musleabihf
#
# steps: steps:
# - name: setup nix - name: setup nix
# image: nixpkgs/nix:nixos-21.05 image: nixpkgs/nix:nixos-21.05
# volumes: volumes:
# - name: nix_store - name: nix_store
# path: /nix path: /nix
# - name: nix_config - name: nix_config
# path: /etc/nix path: /etc/nix
# commands: commands:
# - cp nix/nix.conf /etc/nix/nix.conf - cp nix/nix.conf /etc/nix/nix.conf
# - nix-build --no-build-output --no-out-link --arg rust false --arg integration false -A inputDerivation - nix-build --no-build-output --no-out-link --arg rust false --arg integration false -A inputDerivation
#
# - name: build - name: build
# image: nixpkgs/nix:nixos-21.05 image: nixpkgs/nix:nixos-21.05
# volumes: volumes:
# - name: nix_store - name: nix_store
# path: /nix path: /nix
# - name: nix_config - name: nix_config
# path: /etc/nix path: /etc/nix
# commands: commands:
# - nix-build --no-build-output --argstr target $TARGET --arg release true --argstr git_version $DRONE_COMMIT - nix-build --no-build-output --argstr target $TARGET --arg release true --argstr git_version $DRONE_COMMIT
#
# - name: update cache - name: update cache
# image: nixpkgs/nix:nixos-21.05 image: nixpkgs/nix:nixos-21.05
# environment: environment:
# AWS_ACCESS_KEY_ID: AWS_ACCESS_KEY_ID:
# from_secret: cache_aws_access_key_id from_secret: cache_aws_access_key_id
# AWS_SECRET_ACCESS_KEY: AWS_SECRET_ACCESS_KEY:
# from_secret: cache_aws_secret_access_key from_secret: cache_aws_secret_access_key
# NIX_PRIV_KEY: NIX_PRIV_KEY:
# from_secret: nix_priv_key from_secret: nix_priv_key
# volumes: volumes:
# - name: nix_store - name: nix_store
# path: /nix path: /nix
# - name: nix_config - name: nix_config
# path: /etc/nix path: /etc/nix
# commands: commands:
# - (umask 377 && echo $NIX_PRIV_KEY > /etc/nix/signing-key.sec) - (umask 377 && echo $NIX_PRIV_KEY > /etc/nix/signing-key.sec)
# - | - |
# nix copy --to 's3://nix?endpoint=garage.deuxfleurs.fr&region=garage&secret-key=/etc/nix/signing-key.sec' \ nix copy --to 's3://nix?endpoint=garage.deuxfleurs.fr&region=garage&secret-key=/etc/nix/signing-key.sec' \
# $(nix-store -qR --include-outputs \ $(nix-store -qR --include-outputs \
# $(nix-instantiate --argstr target $TARGET --arg release true)) $(nix-instantiate --argstr target $TARGET --arg release true))
#
# - name: push static binary - name: push static binary
# image: nixpkgs/nix:nixos-21.05 image: nixpkgs/nix:nixos-21.05
# volumes: volumes:
# - name: nix_store - name: nix_store
# path: /nix path: /nix
# - name: nix_config - name: nix_config
# path: /etc/nix path: /etc/nix
# environment: environment:
# AWS_ACCESS_KEY_ID: AWS_ACCESS_KEY_ID:
# from_secret: garagehq_aws_access_key_id from_secret: garagehq_aws_access_key_id
# AWS_SECRET_ACCESS_KEY: AWS_SECRET_ACCESS_KEY:
# from_secret: garagehq_aws_secret_access_key from_secret: garagehq_aws_secret_access_key
# commands: commands:
# - nix-shell --arg integration false --arg rust false --run "to_s3" - nix-shell --arg integration false --arg rust false --run "to_s3"
#
# - name: docker build and publish - name: docker build and publish
# image: nixpkgs/nix:nixos-21.05 image: nixpkgs/nix:nixos-21.05
# volumes: volumes:
# - name: nix_store - name: nix_store
# path: /nix path: /nix
# - name: nix_config - name: nix_config
# path: /etc/nix path: /etc/nix
# environment: environment:
# DOCKER_AUTH: DOCKER_AUTH:
# from_secret: docker_auth from_secret: docker_auth
# DOCKER_PLATFORM: "linux/arm" DOCKER_PLATFORM: "linux/arm"
# CONTAINER_NAME: "dxflrs/arm_garage" CONTAINER_NAME: "dxflrs/arm_garage"
# HOME: "/kaniko" HOME: "/kaniko"
# commands: commands:
# - mkdir -p /kaniko/.docker - mkdir -p /kaniko/.docker
# - echo $DOCKER_AUTH > /kaniko/.docker/config.json - echo $DOCKER_AUTH > /kaniko/.docker/config.json
# - export CONTAINER_TAG=${DRONE_TAG:-$DRONE_COMMIT} - export CONTAINER_TAG=${DRONE_TAG:-$DRONE_COMMIT}
# - nix-shell --arg rust false --arg integration false --run "to_docker" - nix-shell --arg rust false --arg integration false --run "to_docker"
#
# trigger: trigger:
# event: event:
# - promote - promote
# - cron - cron
#
# node: node:
# nix: 1 nix: 1
--- ---
kind: pipeline kind: pipeline
@@ -613,9 +613,9 @@ steps:
depends_on: depends_on:
- release-linux-x86_64 - release-linux-x86_64
#- release-linux-i686 - release-linux-i686
- release-linux-aarch64 - release-linux-aarch64
#- release-linux-armv6l - release-linux-armv6l
trigger: trigger:
event: event:
Generated
+1 -1
View File
@@ -874,7 +874,7 @@ dependencies = [
[[package]] [[package]]
name = "netapp" name = "netapp"
version = "0.3.0" version = "0.3.0"
source = "git+https://git.deuxfleurs.fr/lx/netapp#c20d36892bcccae580603249706ba60d54a46d7f" source = "git+https://git.deuxfleurs.fr/lx/netapp#57327f10e2536a89004f3a1def83ed16243c1a3e"
dependencies = [ dependencies = [
"arc-swap", "arc-swap",
"async-trait", "async-trait",
+2 -2
View File
@@ -246,7 +246,7 @@ in
registry = "registry+https://github.com/rust-lang/crates.io-index"; registry = "registry+https://github.com/rust-lang/crates.io-index";
src = fetchCratesIo { inherit name version; sha256 = "95059428f66df56b63431fdb4e1947ed2190586af5c5a8a8b71122bdf5a7f469"; }; src = fetchCratesIo { inherit name version; sha256 = "95059428f66df56b63431fdb4e1947ed2190586af5c5a8a8b71122bdf5a7f469"; };
dependencies = { dependencies = {
${ if hostPlatform.config == "aarch64-apple-darwin" || hostPlatform.parsed.cpu.name == "aarch64" && hostPlatform.parsed.kernel.name == "linux" then "libc" else null } = rustPackages."registry+https://github.com/rust-lang/crates.io-index".libc."0.2.103" { inherit profileName; }; ${ if hostPlatform.parsed.cpu.name == "aarch64" && hostPlatform.parsed.kernel.name == "linux" || hostPlatform.config == "aarch64-apple-darwin" then "libc" else null } = rustPackages."registry+https://github.com/rust-lang/crates.io-index".libc."0.2.103" { inherit profileName; };
}; };
}); });
@@ -1236,7 +1236,7 @@ in
url = https://git.deuxfleurs.fr/lx/netapp; url = https://git.deuxfleurs.fr/lx/netapp;
name = "netapp"; name = "netapp";
version = "0.3.0"; version = "0.3.0";
rev = "9b64c27da68f7ac9049e02e26da918e871a63f07";}; rev = "57327f10e2536a89004f3a1def83ed16243c1a3e";};
features = builtins.concatLists [ features = builtins.concatLists [
[ "default" ] [ "default" ]
]; ];
-1
View File
@@ -30,4 +30,3 @@
- [Working Documents](./working_documents/index.md) - [Working Documents](./working_documents/index.md)
- [Load Balancing Data](./working_documents/load_balancing.md) - [Load Balancing Data](./working_documents/load_balancing.md)
- [Migrating from 0.3 to 0.4](./working_documents/migration_04.md)
+39 -46
View File
@@ -11,7 +11,7 @@ to get familiar with Garage's command line and usage patterns.
## Prerequisites ## Prerequisites
To run a real-world deployment, make sure the following conditions are met: To run a real-world deployment, make sure you the following conditions are met:
- You have at least three machines with sufficient storage space available. - You have at least three machines with sufficient storage space available.
@@ -52,6 +52,7 @@ For example:
sudo docker pull lxpz/garage_amd64:v0.4.0 sudo docker pull lxpz/garage_amd64:v0.4.0
``` ```
## Deploying and configuring Garage ## Deploying and configuring Garage
On each machine, we will have a similar setup, On each machine, we will have a similar setup,
@@ -78,6 +79,10 @@ rpc_bind_addr = "[::]:3901"
rpc_public_addr = "<this node's public IP>:3901" rpc_public_addr = "<this node's public IP>:3901"
rpc_secret = "<RPC secret>" rpc_secret = "<RPC secret>"
bootstrap_peers = [
# We will fill this in later
]
[s3_api] [s3_api]
s3_region = "garage" s3_region = "garage"
api_bind_addr = "[::]:3900" api_bind_addr = "[::]:3900"
@@ -97,6 +102,33 @@ Check the following for your configuration files:
- Make sure `rpc_secret` is the same value on all nodes. It should be a 32-bytes hex-encoded secret key. - Make sure `rpc_secret` is the same value on all nodes. It should be a 32-bytes hex-encoded secret key.
You can generate such a key with `openssl rand -hex 32`. You can generate such a key with `openssl rand -hex 32`.
You will now have to run `garage node-id` on all nodes to generate node keys.
This will print keys as follows:
```bash
Mercury$ garage node-id
563e1ac825ee3323aa441e72c26d1030d6d4414aeb3dd25287c531e7fc2bc95d@[fc00:1::1]:3901
Venus$ garage node-id
86f0f26ae4afbd59aaf9cfb059eefac844951efd5b8caeec0d53f4ed6c85f332@[fc00:1::2]:3901
etc.
```
You can then add these nodes to the `bootstrap_peers` list of at least one of your nodes:
```toml
bootstrap_peers = [
"563e1ac825ee3323aa441e72c26d1030d6d4414aeb3dd25287c531e7fc2bc95d@[fc00:1::1]:3901",
"86f0f26ae4afbd59aaf9cfb059eefac844951efd5b8caeec0d53f4ed6c85f332@[fc00:1::2]:3901",
...
]
```
Check the [configuration file reference documentation](../reference_manual/configuration.md)
to learn more about all available configuration options.
## Starting Garage using Docker ## Starting Garage using Docker
On each machine, you can run the daemon with: On each machine, you can run the daemon with:
@@ -122,6 +154,7 @@ but please check the relase notes before doing so!
To upgrade, simply stop and remove this container and To upgrade, simply stop and remove this container and
start again the command with a new version of Garage. start again the command with a new version of Garage.
## Controling the daemon ## Controling the daemon
The `garage` binary has two purposes: The `garage` binary has two purposes:
@@ -133,51 +166,11 @@ If your configuration file is at `/etc/garage.toml`, the `garage` binary should
You can test your `garage` CLI utility by running a simple command such as: You can test your `garage` CLI utility by running a simple command such as:
```bash ```
garage status garage status
``` ```
At this point, nodes are not yet talking to one another. You should get something like that as result:
Your output should therefore look like follows:
```
Mercury$ garage node-id
==== HEALTHY NODES ====
ID Hostname Address Tag Zone Capacity
563e1ac825ee3323… Mercury [fc00:1::1]:3901 NO ROLE ASSIGNED
```
## Connecting nodes together
When your Garage nodes first start, they will generate a local node identifier
(based on a public/private key pair).
To obtain the node identifier of a node, once it is generated,
run `garage node-id`.
This will print keys as follows:
```bash
Mercury$ garage node-id
563e1ac825ee3323aa441e72c26d1030d6d4414aeb3dd25287c531e7fc2bc95d@[fc00:1::1]:3901
Venus$ garage node-id
86f0f26ae4afbd59aaf9cfb059eefac844951efd5b8caeec0d53f4ed6c85f332@[fc00:1::2]:3901
etc.
```
You can then instruct nodes to connect to one another as follows:
```bash
# Instruct Venus to connect to Mercury (this will establish communication both ways)
Venus$ garage node connect 563e1ac825ee3323aa441e72c26d1030d6d4414aeb3dd25287c531e7fc2bc95d@[fc00:1::1]:3901
```
You don't nead to instruct all node to connect to all other nodes:
nodes will discover one another transitively.
Now if your run `garage status` on any node, you should have an output that looks as follows:
``` ```
==== HEALTHY NODES ==== ==== HEALTHY NODES ====
@@ -188,13 +181,13 @@ ID Hostname Address Tag Zone Capa
212f7572f0c89da9… Mars [fc00:F::1]:3901 NO ROLE ASSIGNED 212f7572f0c89da9… Mars [fc00:F::1]:3901 NO ROLE ASSIGNED
``` ```
## Giving roles to nodes
## Configuring a cluster
We will now inform Garage of the disk space available on each node of the cluster We will now inform Garage of the disk space available on each node of the cluster
as well as the zone (e.g. datacenter) in which each machine is located. as well as the zone (e.g. datacenter) in which each machine is located.
For our example, we will suppose we have the following infrastructure For our example, we will suppose we have the following infrastructure (Capacity, Identifier and Datacenter are specific values to Garage described in the following):
(Capacity, Identifier and Zone are specific values to Garage described in the following):
| Location | Name | Disk Space | `Capacity` | `Identifier` | `Zone` | | Location | Name | Disk Space | `Capacity` | `Identifier` | `Zone` |
|----------|---------|------------|------------|--------------|--------------| |----------|---------|------------|------------|--------------|--------------|
@@ -1,61 +0,0 @@
# Migrating from 0.3 to 0.4
**Migrating from 0.3 to 0.4 is unsupported. This document is only intended to document the process internally for the Deuxfleurs cluster where we have to do it. Do not try it yourself, you will lose your data and we will not help you.**
**Migrating from 0.2 to 0.4 will break everything for sure. Never try it.**
The internal data format of Garage hasn't changed much between 0.3 and 0.4.
The Sled database is still the same, and the data directory as well.
The following has changed, all in the meta directory:
- `node_id` in 0.3 contains the identifier of the current node. In 0.4, this file does nothing and should be deleted. It is replaced by `node_key` (the secret key) and `node_key.pub` (the associated public key). A node's identifier on the ring is its public key.
- `peer_info` in 0.3 contains the list of peers saved automatically by Garage. The format has changed and it is now stored in `peer_list` (`peer_info` should be deleted).
When migrating, all node identifiers will change. This also means that the affectation of data partitions on the ring will change, and lots of data will have to be rebalanced.
- If your cluster has only 3 nodes, all nodes store everything, therefore nothing has to be rebalanced.
- If your cluster has only 4 nodes, for any partition there will always be at least 2 nodes that stored data before that still store it after. Therefore the migration should in theory be transparent and Garage should continue to work during the rebalance.
- If your cluster has 5 or more nodes, data will disappear during the migration. Do not migrate (fortunately we don't have this scenario at Deuxfleurs), or if you do, make Garage unavailable until things stabilize (disable web and api access).
The migration steps are as follows:
1. Prepare a new configuration file for 0.4. For each node, point to the same meta and data directories as Garage 0.3. Basically, the things that change are the following:
- No more `rpc_tls` section
- You have to generate a shared `rpc_secret` and put it in all config files
- `bootstrap_nodes` has a different syntax as it has to contain node keys. Leave it empty and use `garage node-id` and `garage node connect` instead (new features of 0.4)
- put the publicly accessible RPC address of your node in `rpc_public_addr` if possible (its optional but recommended)
- If you are using Consul, change the `consul_service_name` to NOT be the name advertised by Nomad. Now Garage is responsible for advertising its own service itself.
2. Disable api and web access for some time, do `garage repair --all --yes tables` and `garage repair --all --yes blocks`, check the logs and check that all data seems to be synced correctly between nodes.
3. Save somewhere the output of `garage status`. We will need this to remember how to reconfigure nodes in 0.4.
4. Turn off Garage 0.3
5. Backup metadata folders if you can (i.e. if you have space to do it somewhere). Backuping data folders could also be usefull but that's much harder to do. If your filesystem supports snapshots, this could be a good time to use them.
6. Turn on Garage 0.4
7. At this point, running `garage status` should indicate that all nodes of the previous cluster are "unavailable". The nodes have new identifiers that should appear in healthy nodes once they can talk to one another (use `garage node connect` if necessary`). They should have NO ROLE ASSIGNED at the moment.
8. Prepare a script with several `garage node configure` commands that replace each of the v0.3 node ID with the corresponding v0.4 node ID, with the same zone/tag/capacity. For example if your node `drosera` had identifier `c24e` before and now has identifier `789a`, and it was configured with capacity `2` in zone `dc1`, put the following command in your script:
```bash
garage node configure 789a -z dc1 -c 2 -t drosera --replace c24e
```
9. Run your reconfiguration script. Check that the new output of `garage status` contains the correct node IDs with the correct values for capacity and zone. Old nodes should no longer be mentioned.
10. If your cluster has 4 nodes or less, and you are feeling adventurous, you can reenable Web and API access now. Things will probably work.
11. Garage might already be resyncing stuff. Issue a `garage repair --all --yes tables` and `garage repair --all --yes blocks` to force it to do so.
12. Wait for resyncing activity to stop in the logs. Do steps 11 and 12 two or three times, until you see that when you issue the repair commands, nothing gets resynced any longer.
13. Your upgraded cluster should be in a working state. Re-enable API and Web access and check that everything went well.
+2 -5
View File
@@ -93,17 +93,14 @@ pub async fn cmd_status(rpc_cli: &Endpoint<SystemRpc, ()>, rpc_host: NodeID) ->
for adv in status.iter().filter(|adv| !adv.is_up) { for adv in status.iter().filter(|adv| !adv.is_up) {
if let Some(cfg) = config.members.get(&adv.id) { if let Some(cfg) = config.members.get(&adv.id) {
failed_nodes.push(format!( failed_nodes.push(format!(
"{id:?}\t{host}\t{addr}\t[{tag}]\t{zone}\t{capacity}\t{last_seen}", "{id:?}\t{host}\t{addr}\t[{tag}]\t{zone}\t{capacity}\t{last_seen}s ago",
id = adv.id, id = adv.id,
host = adv.status.hostname, host = adv.status.hostname,
addr = adv.addr, addr = adv.addr,
tag = cfg.tag, tag = cfg.tag,
zone = cfg.zone, zone = cfg.zone,
capacity = cfg.capacity_string(), capacity = cfg.capacity_string(),
last_seen = adv last_seen = "FIXME", // FIXME was (now_msec() - adv.last_seen) / 1000,
.last_seen_secs_ago
.map(|s| format!("{}s ago", s))
.unwrap_or_else(|| "never seen".into()),
)); ));
} }
} }
+179 -256
View File
@@ -27,7 +27,7 @@ use crate::garage::Garage;
/// Size under which data will be stored inlined in database instead of as files /// Size under which data will be stored inlined in database instead of as files
pub const INLINE_THRESHOLD: usize = 3072; pub const INLINE_THRESHOLD: usize = 3072;
pub const BACKGROUND_WORKERS: u64 = 2; pub const BACKGROUND_WORKERS: u64 = 1;
const BLOCK_RW_TIMEOUT: Duration = Duration::from_secs(42); const BLOCK_RW_TIMEOUT: Duration = Duration::from_secs(42);
const BLOCK_GC_TIMEOUT: Duration = Duration::from_secs(60); const BLOCK_GC_TIMEOUT: Duration = Duration::from_secs(60);
@@ -70,8 +70,8 @@ pub struct BlockManager {
pub replication: TableShardedReplication, pub replication: TableShardedReplication,
/// Directory in which block are stored /// Directory in which block are stored
pub data_dir: PathBuf, pub data_dir: PathBuf,
/// Lock to prevent concurrent edition of the directory
mutation_lock: Mutex<BlockManagerLocked>, pub data_dir_lock: Mutex<()>,
rc: sled::Tree, rc: sled::Tree,
@@ -83,11 +83,6 @@ pub struct BlockManager {
pub(crate) garage: ArcSwapOption<Garage>, pub(crate) garage: ArcSwapOption<Garage>,
} }
// This custom struct contains functions that must only be ran
// when the lock is held. We ensure that it is the case by storing
// it INSIDE a Mutex.
struct BlockManagerLocked();
impl BlockManager { impl BlockManager {
pub fn new( pub fn new(
db: &sled::Db, db: &sled::Db,
@@ -107,12 +102,10 @@ impl BlockManager {
.netapp .netapp
.endpoint("garage_model/block.rs/Rpc".to_string()); .endpoint("garage_model/block.rs/Rpc".to_string());
let manager_locked = BlockManagerLocked();
let block_manager = Arc::new(Self { let block_manager = Arc::new(Self {
replication, replication,
data_dir, data_dir,
mutation_lock: Mutex::new(manager_locked), data_dir_lock: Mutex::new(()),
rc, rc,
resync_queue, resync_queue,
resync_notify: Notify::new(), resync_notify: Notify::new(),
@@ -125,95 +118,99 @@ impl BlockManager {
block_manager block_manager
} }
// ---- Public interface ---- pub fn spawn_background_worker(self: Arc<Self>) {
// Launch 2 simultaneous workers for background resync loop preprocessing <= TODO actually this
/// Ask nodes that might have a block for it // launches only one worker with current value of BACKGROUND_WORKERS
pub async fn rpc_get_block(&self, hash: &Hash) -> Result<Vec<u8>, Error> { for i in 0..BACKGROUND_WORKERS {
let who = self.replication.read_nodes(&hash); let bm2 = self.clone();
let resps = self let background = self.system.background.clone();
.system tokio::spawn(async move {
.rpc tokio::time::sleep(Duration::from_secs(10 * (i + 1))).await;
.try_call_many( background.spawn_worker(format!("block resync worker {}", i), move |must_exit| {
&self.endpoint, bm2.resync_loop(must_exit)
&who[..], });
BlockRpc::GetBlock(*hash), });
RequestStrategy::with_priority(PRIO_NORMAL)
.with_quorum(1)
.with_timeout(BLOCK_RW_TIMEOUT)
.interrupt_after_quorum(true),
)
.await?;
for resp in resps {
if let BlockRpc::PutBlock(msg) = resp {
return Ok(msg.data);
}
} }
Err(Error::Message(format!(
"Unable to read block {:?}: no valid blocks returned",
hash
)))
} }
/// Send block to nodes that should have it /// Write a block to disk
pub async fn rpc_put_block(&self, hash: Hash, data: Vec<u8>) -> Result<(), Error> { async fn write_block(&self, hash: &Hash, data: &[u8]) -> Result<BlockRpc, Error> {
let who = self.replication.write_nodes(&hash); let _lock = self.data_dir_lock.lock().await;
self.system
.rpc
.try_call_many(
&self.endpoint,
&who[..],
BlockRpc::PutBlock(PutBlockMessage { hash, data }),
RequestStrategy::with_priority(PRIO_NORMAL)
.with_quorum(self.replication.write_quorum())
.with_timeout(BLOCK_RW_TIMEOUT),
)
.await?;
Ok(())
}
/// Launch the repair procedure on the data store let mut path = self.block_dir(hash);
/// fs::create_dir_all(&path).await?;
/// This will list all blocks locally present, as well as those
/// that are required because of refcount > 0, and will try path.push(hex::encode(hash));
/// to fix any mismatch between the two. if fs::metadata(&path).await.is_ok() {
pub async fn repair_data_store(&self, must_exit: &watch::Receiver<bool>) -> Result<(), Error> { return Ok(BlockRpc::Ok);
// 1. Repair blocks from RC table
let garage = self.garage.load_full().unwrap();
let mut last_hash = None;
for (i, entry) in garage.block_ref_table.data.store.iter().enumerate() {
let (_k, v_bytes) = entry?;
let block_ref = rmp_serde::decode::from_read_ref::<_, BlockRef>(v_bytes.as_ref())?;
if Some(&block_ref.block) == last_hash.as_ref() {
continue;
}
if !block_ref.deleted.get() {
last_hash = Some(block_ref.block);
self.put_to_resync(&block_ref.block, Duration::from_secs(0))?;
}
if i & 0xFF == 0 && *must_exit.borrow() {
return Ok(());
}
} }
// 2. Repair blocks actually on disk let mut f = fs::File::create(path).await?;
self.repair_aux_read_dir_rec(&self.data_dir, must_exit) f.write_all(data).await?;
.await?; drop(f);
Ok(()) Ok(BlockRpc::Ok)
} }
/// Get lenght of resync queue /// Read block from disk, verifying it's integrity
pub fn resync_queue_len(&self) -> usize { async fn read_block(&self, hash: &Hash) -> Result<BlockRpc, Error> {
self.resync_queue.len() let path = self.block_path(hash);
let mut f = match fs::File::open(&path).await {
Ok(f) => f,
Err(e) => {
// Not found but maybe we should have had it ??
self.put_to_resync(hash, Duration::from_millis(0))?;
return Err(Into::into(e));
}
};
let mut data = vec![];
f.read_to_end(&mut data).await?;
drop(f);
if blake2sum(&data[..]) != *hash {
let _lock = self.data_dir_lock.lock().await;
warn!(
"Block {:?} is corrupted. Renaming to .corrupted and resyncing.",
hash
);
let mut path2 = path.clone();
path2.set_extension(".corrupted");
fs::rename(path, path2).await?;
self.put_to_resync(&hash, Duration::from_millis(0))?;
return Err(Error::CorruptData(*hash));
}
Ok(BlockRpc::PutBlock(PutBlockMessage { hash: *hash, data }))
} }
/// Get number of items in the refcount table /// Check if this node should have a block, but don't actually have it
pub fn rc_len(&self) -> usize { async fn need_block(&self, hash: &Hash) -> Result<bool, Error> {
self.rc.len() let needed = self
.rc
.get(hash.as_ref())?
.map(|x| u64_from_be_bytes(x) > 0)
.unwrap_or(false);
if needed {
let path = self.block_path(hash);
let exists = fs::metadata(&path).await.is_ok();
Ok(!exists)
} else {
Ok(false)
}
} }
//// ----- Managing the reference counter ---- fn block_dir(&self, hash: &Hash) -> PathBuf {
let mut path = self.data_dir.clone();
path.push(hex::encode(&hash.as_slice()[0..1]));
path.push(hex::encode(&hash.as_slice()[1..2]));
path
}
fn block_path(&self, hash: &Hash) -> PathBuf {
let mut path = self.block_dir(hash);
path.push(hex::encode(hash.as_ref()));
path
}
/// Increment the number of time a block is used, putting it to resynchronization if it is /// Increment the number of time a block is used, putting it to resynchronization if it is
/// required, but not known /// required, but not known
@@ -245,96 +242,6 @@ impl BlockManager {
Ok(()) Ok(())
} }
/// Read a block's reference count
pub fn get_block_rc(&self, hash: &Hash) -> Result<u64, Error> {
Ok(self
.rc
.get(hash.as_ref())?
.map(u64_from_be_bytes)
.unwrap_or(0))
}
// ---- Reading and writing blocks locally ----
/// Write a block to disk
async fn write_block(&self, hash: &Hash, data: &[u8]) -> Result<BlockRpc, Error> {
self.mutation_lock
.lock()
.await
.write_block(hash, data, self)
.await
}
/// Read block from disk, verifying it's integrity
async fn read_block(&self, hash: &Hash) -> Result<BlockRpc, Error> {
let path = self.block_path(hash);
let mut f = match fs::File::open(&path).await {
Ok(f) => f,
Err(e) => {
// Not found but maybe we should have had it ??
self.put_to_resync(hash, Duration::from_millis(0))?;
return Err(Into::into(e));
}
};
let mut data = vec![];
f.read_to_end(&mut data).await?;
drop(f);
if blake2sum(&data[..]) != *hash {
self.mutation_lock
.lock()
.await
.move_block_to_corrupted(hash, self)
.await?;
return Err(Error::CorruptData(*hash));
}
Ok(BlockRpc::PutBlock(PutBlockMessage { hash: *hash, data }))
}
/// Check if this node should have a block, but don't actually have it
async fn need_block(&self, hash: &Hash) -> Result<bool, Error> {
let BlockStatus { exists, needed } = self
.mutation_lock
.lock()
.await
.check_block_status(hash, self)
.await?;
Ok(needed && !exists)
}
/// Utility: gives the path of the directory in which a block should be found
fn block_dir(&self, hash: &Hash) -> PathBuf {
let mut path = self.data_dir.clone();
path.push(hex::encode(&hash.as_slice()[0..1]));
path.push(hex::encode(&hash.as_slice()[1..2]));
path
}
/// Utility: give the full path where a block should be found
fn block_path(&self, hash: &Hash) -> PathBuf {
let mut path = self.block_dir(hash);
path.push(hex::encode(hash.as_ref()));
path
}
// ---- Resync loop ----
pub fn spawn_background_worker(self: Arc<Self>) {
// Launch 2 simultaneous workers for background resync loop preprocessing
for i in 0..BACKGROUND_WORKERS {
let bm2 = self.clone();
let background = self.system.background.clone();
tokio::spawn(async move {
tokio::time::sleep(Duration::from_secs(10 * (i + 1))).await;
background.spawn_worker(format!("block resync worker {}", i), move |must_exit| {
bm2.resync_loop(must_exit)
});
});
}
}
fn put_to_resync(&self, hash: &Hash, delay: Duration) -> Result<(), Error> { fn put_to_resync(&self, hash: &Hash, delay: Duration) -> Result<(), Error> {
let when = now_msec() + delay.as_millis() as u64; let when = now_msec() + delay.as_millis() as u64;
trace!("Put resync_queue: {} {:?}", when, hash); trace!("Put resync_queue: {} {:?}", when, hash);
@@ -389,12 +296,16 @@ impl BlockManager {
} }
async fn resync_block(&self, hash: &Hash) -> Result<(), Error> { async fn resync_block(&self, hash: &Hash) -> Result<(), Error> {
let BlockStatus { exists, needed } = self let lock = self.data_dir_lock.lock().await;
.mutation_lock
.lock() let path = self.block_path(hash);
.await
.check_block_status(hash, self) let exists = fs::metadata(&path).await.is_ok();
.await?; let needed = self
.rc
.get(hash.as_ref())?
.map(|x| u64_from_be_bytes(x) > 0)
.unwrap_or(false);
if exists != needed { if exists != needed {
info!( info!(
@@ -467,14 +378,12 @@ impl BlockManager {
who.len() who.len()
); );
self.mutation_lock fs::remove_file(path).await?;
.lock()
.await
.delete_if_unneeded(hash, self)
.await?;
} }
if needed && !exists { if needed && !exists {
drop(lock);
// TODO find a way to not do this if they are sending it to us // TODO find a way to not do this if they are sending it to us
// Let's suppose this isn't an issue for now with the BLOCK_RW_TIMEOUT delay // Let's suppose this isn't an issue for now with the BLOCK_RW_TIMEOUT delay
// between the RC being incremented and this part being called. // between the RC being incremented and this part being called.
@@ -485,6 +394,79 @@ impl BlockManager {
Ok(()) Ok(())
} }
/// Ask nodes that might have a block for it
pub async fn rpc_get_block(&self, hash: &Hash) -> Result<Vec<u8>, Error> {
let who = self.replication.read_nodes(&hash);
let resps = self
.system
.rpc
.try_call_many(
&self.endpoint,
&who[..],
BlockRpc::GetBlock(*hash),
RequestStrategy::with_priority(PRIO_NORMAL)
.with_quorum(1)
.with_timeout(BLOCK_RW_TIMEOUT)
.interrupt_after_quorum(true),
)
.await?;
for resp in resps {
if let BlockRpc::PutBlock(msg) = resp {
return Ok(msg.data);
}
}
Err(Error::Message(format!(
"Unable to read block {:?}: no valid blocks returned",
hash
)))
}
/// Send block to nodes that should have it
pub async fn rpc_put_block(&self, hash: Hash, data: Vec<u8>) -> Result<(), Error> {
let who = self.replication.write_nodes(&hash);
self.system
.rpc
.try_call_many(
&self.endpoint,
&who[..],
BlockRpc::PutBlock(PutBlockMessage { hash, data }),
RequestStrategy::with_priority(PRIO_NORMAL)
.with_quorum(self.replication.write_quorum())
.with_timeout(BLOCK_RW_TIMEOUT),
)
.await?;
Ok(())
}
pub async fn repair_data_store(&self, must_exit: &watch::Receiver<bool>) -> Result<(), Error> {
// 1. Repair blocks from RC table
let garage = self.garage.load_full().unwrap();
let mut last_hash = None;
let mut i = 0usize;
for entry in garage.block_ref_table.data.store.iter() {
let (_k, v_bytes) = entry?;
let block_ref = rmp_serde::decode::from_read_ref::<_, BlockRef>(v_bytes.as_ref())?;
if Some(&block_ref.block) == last_hash.as_ref() {
continue;
}
if !block_ref.deleted.get() {
last_hash = Some(block_ref.block);
self.put_to_resync(&block_ref.block, Duration::from_secs(0))?;
}
i += 1;
if i & 0xFF == 0 && *must_exit.borrow() {
return Ok(());
}
}
// 2. Repair blocks actually on disk
self.repair_aux_read_dir_rec(&self.data_dir, must_exit)
.await?;
Ok(())
}
fn repair_aux_read_dir_rec<'a>( fn repair_aux_read_dir_rec<'a>(
&'a self, &'a self,
path: &'a Path, path: &'a Path,
@@ -529,6 +511,15 @@ impl BlockManager {
} }
.boxed() .boxed()
} }
/// Get lenght of resync queue
pub fn resync_queue_len(&self) -> usize {
self.resync_queue.len()
}
pub fn rc_len(&self) -> usize {
self.rc.len()
}
} }
#[async_trait] #[async_trait]
@@ -547,74 +538,6 @@ impl EndpointHandler<BlockRpc> for BlockManager {
} }
} }
struct BlockStatus {
exists: bool,
needed: bool,
}
impl BlockManagerLocked {
async fn check_block_status(
&self,
hash: &Hash,
mgr: &BlockManager,
) -> Result<BlockStatus, Error> {
let path = mgr.block_path(hash);
let exists = fs::metadata(&path).await.is_ok();
let needed = mgr.get_block_rc(hash)? > 0;
Ok(BlockStatus { exists, needed })
}
async fn write_block(
&self,
hash: &Hash,
data: &[u8],
mgr: &BlockManager,
) -> Result<BlockRpc, Error> {
let mut path = mgr.block_dir(hash);
fs::create_dir_all(&path).await?;
path.push(hex::encode(hash));
if fs::metadata(&path).await.is_ok() {
return Ok(BlockRpc::Ok);
}
let mut path2 = path.clone();
path2.set_extension("tmp");
let mut f = fs::File::create(&path2).await?;
f.write_all(data).await?;
drop(f);
fs::rename(path2, path).await?;
Ok(BlockRpc::Ok)
}
async fn move_block_to_corrupted(&self, hash: &Hash, mgr: &BlockManager) -> Result<(), Error> {
warn!(
"Block {:?} is corrupted. Renaming to .corrupted and resyncing.",
hash
);
let path = mgr.block_path(hash);
let mut path2 = path.clone();
path2.set_extension(".corrupted");
fs::rename(path, path2).await?;
mgr.put_to_resync(&hash, Duration::from_millis(0))?;
Ok(())
}
async fn delete_if_unneeded(&self, hash: &Hash, mgr: &BlockManager) -> Result<(), Error> {
let BlockStatus { exists, needed } = self.check_block_status(hash, mgr).await?;
if exists && !needed {
let path = mgr.block_path(hash);
fs::remove_file(path).await?;
}
Ok(())
}
}
fn u64_from_be_bytes<T: AsRef<[u8]>>(bytes: T) -> u64 { fn u64_from_be_bytes<T: AsRef<[u8]>>(bytes: T) -> u64 {
assert!(bytes.as_ref().len() == 8); assert!(bytes.as_ref().len() == 8);
let mut x8 = [0u8; 8]; let mut x8 = [0u8; 8];
+1 -3
View File
@@ -4,7 +4,7 @@ use std::io::{Read, Write};
use std::net::SocketAddr; use std::net::SocketAddr;
use std::path::Path; use std::path::Path;
use std::sync::{Arc, RwLock}; use std::sync::{Arc, RwLock};
use std::time::{Duration, Instant}; use std::time::Duration;
use arc_swap::ArcSwap; use arc_swap::ArcSwap;
use async_trait::async_trait; use async_trait::async_trait;
@@ -112,7 +112,6 @@ pub struct KnownNodeInfo {
pub id: Uuid, pub id: Uuid,
pub addr: SocketAddr, pub addr: SocketAddr,
pub is_up: bool, pub is_up: bool,
pub last_seen_secs_ago: Option<u64>,
pub status: NodeStatus, pub status: NodeStatus,
} }
@@ -355,7 +354,6 @@ impl System {
id: n.id.into(), id: n.id.into(),
addr: n.addr, addr: n.addr,
is_up: n.is_up(), is_up: n.is_up(),
last_seen_secs_ago: n.last_seen.map(|t| (Instant::now() - t).as_secs()),
status: node_status status: node_status
.get(&n.id.into()) .get(&n.id.into())
.cloned() .cloned()