forked from cerc-io/stack-orchestrator
Merge commit '481e9d239247c01604ed9e11160abc94e9dd9eb4' as 'agave-stack'
This commit is contained in:
Executable
+234
@@ -0,0 +1,234 @@
|
||||
#!/bin/bash
|
||||
|
||||
set -Eeuo pipefail
|
||||
|
||||
export PATH=/sbin:/bin:/usr/sbin:/usr/bin:/usr/local/sbin:/usr/local/bin
|
||||
export XDG_RUNTIME_DIR="/run/user/$(id -u)"
|
||||
mkdir -p "$XDG_RUNTIME_DIR"
|
||||
|
||||
# optional suffix from command-line, prepend dash if non-empty
|
||||
SUFFIX="${1:-}"
|
||||
SUFFIX="${SUFFIX:+-$SUFFIX}"
|
||||
|
||||
# define variables
|
||||
DATASET="biscayne/DATA/deployments"
|
||||
DEPLOYMENT_DIR="/srv/deployments/agave"
|
||||
LOG_FILE="$HOME/.backlog_history"
|
||||
ZFS_HOLD="backlog:pending"
|
||||
SERVICE_STOP_TIMEOUT="300"
|
||||
SNAPSHOT_RETENTION="6"
|
||||
SNAPSHOT_PREFIX="backlog"
|
||||
SNAPSHOT_TAG="$(date +%Y%m%d)${SUFFIX}"
|
||||
SNAPSHOT="${DATASET}@${SNAPSHOT_PREFIX}-${SNAPSHOT_TAG}"
|
||||
|
||||
# remote replication targets
|
||||
REMOTES=(
|
||||
"mysterio:edith/DATA/backlog/biscayne-main"
|
||||
"ardham:batterywharf/DATA/backlog/biscayne-main"
|
||||
)
|
||||
|
||||
# log functions
|
||||
log() {
|
||||
local time_fmt
|
||||
time_fmt=$(date -u +"%Y-%m-%dT%H:%M:%SZ")
|
||||
echo "[$time_fmt] $1" >> "$LOG_FILE"
|
||||
}
|
||||
|
||||
log_close() {
|
||||
local end_time duration
|
||||
end_time=$(date +%s)
|
||||
duration=$((end_time - start_time))
|
||||
log "Backlog completed in ${duration}s"
|
||||
echo "" >> "$LOG_FILE"
|
||||
}
|
||||
|
||||
# service controls
|
||||
services() {
|
||||
local action="$1"
|
||||
|
||||
case "$action" in
|
||||
stop)
|
||||
log "Stopping agave deployment..."
|
||||
laconic-so deployment --dir "$DEPLOYMENT_DIR" stop
|
||||
|
||||
log "Waiting for services to fully stop..."
|
||||
local deadline=$(( $(date +%s) + SERVICE_STOP_TIMEOUT ))
|
||||
while true; do
|
||||
local running
|
||||
running=$(docker ps --filter "label=com.docker.compose.project.working_dir=$DEPLOYMENT_DIR" -q 2>/dev/null | wc -l)
|
||||
if [[ "$running" -eq 0 ]]; then
|
||||
break
|
||||
fi
|
||||
if (( $(date +%s) >= deadline )); then
|
||||
log "WARNING: Timeout waiting for services to stop; continuing."
|
||||
break
|
||||
fi
|
||||
sleep 0.2
|
||||
done
|
||||
;;
|
||||
start)
|
||||
log "Starting agave deployment..."
|
||||
laconic-so deployment --dir "$DEPLOYMENT_DIR" start
|
||||
;;
|
||||
*)
|
||||
log "ERROR: Unknown action '$action' in services()"
|
||||
exit 2
|
||||
;;
|
||||
esac
|
||||
}
|
||||
|
||||
# send a snapshot to one remote
|
||||
# args: snap remote_host remote_dataset
|
||||
snapshot_send_one() {
|
||||
local snap="$1" remote_host="$2" remote_dataset="$3"
|
||||
|
||||
log "Checking remote snapshots on $remote_host..."
|
||||
|
||||
local -a local_snaps remote_snaps
|
||||
mapfile -t local_snaps < <(zfs list -H -t snapshot -o name -s creation -d1 "$DATASET" | grep -F "${DATASET}@${SNAPSHOT_PREFIX}-")
|
||||
mapfile -t remote_snaps < <(ssh "$remote_host" zfs list -H -t snapshot -o name -s creation "$remote_dataset" | grep -F "${remote_dataset}@${SNAPSHOT_PREFIX}-" || true)
|
||||
|
||||
# find latest common snapshot
|
||||
local base=""
|
||||
local local_snap remote_snap remote_check
|
||||
for local_snap in "${local_snaps[@]}"; do
|
||||
remote_snap="${local_snap/$DATASET/$remote_dataset}"
|
||||
for remote_check in "${remote_snaps[@]}"; do
|
||||
if [[ "$remote_check" == "$remote_snap" ]]; then
|
||||
base="$local_snap"
|
||||
break
|
||||
fi
|
||||
done
|
||||
done
|
||||
|
||||
if [[ -z "$base" && ${#remote_snaps[@]} -eq 0 ]]; then
|
||||
log "No remote snapshots found on $remote_host — sending full snapshot."
|
||||
if zfs send "$snap" | ssh "$remote_host" zfs receive -sF "$remote_dataset"; then
|
||||
log "Full send to $remote_host succeeded."
|
||||
return 0
|
||||
else
|
||||
log "ERROR: Full send to $remote_host failed."
|
||||
return 1
|
||||
fi
|
||||
elif [[ -n "$base" ]]; then
|
||||
log "Common base snapshot $base found — sending incremental to $remote_host."
|
||||
if zfs send -i "$base" "$snap" | ssh "$remote_host" zfs receive -sF "$remote_dataset"; then
|
||||
log "Incremental send to $remote_host succeeded."
|
||||
return 0
|
||||
else
|
||||
log "ERROR: Incremental send to $remote_host failed."
|
||||
return 1
|
||||
fi
|
||||
else
|
||||
log "STALE DESTINATION: $remote_host has snapshots but no common base with local — skipping."
|
||||
return 1
|
||||
fi
|
||||
}
|
||||
|
||||
# send snapshot to all remotes
|
||||
snapshot_send() {
|
||||
local snap="$1"
|
||||
local failure_count=0
|
||||
|
||||
set +e
|
||||
local entry remote_host remote_dataset
|
||||
for entry in "${REMOTES[@]}"; do
|
||||
remote_host="${entry%%:*}"
|
||||
remote_dataset="${entry#*:}"
|
||||
if ! snapshot_send_one "$snap" "$remote_host" "$remote_dataset"; then
|
||||
failure_count=$((failure_count + 1))
|
||||
fi
|
||||
done
|
||||
set -e
|
||||
|
||||
if [[ "$failure_count" -gt 0 ]]; then
|
||||
log "WARNING: $failure_count destination(s) failed or are out of sync."
|
||||
return 1
|
||||
fi
|
||||
return 0
|
||||
}
|
||||
|
||||
# snapshot management
|
||||
snapshot() {
|
||||
local action="$1"
|
||||
|
||||
case "$action" in
|
||||
create)
|
||||
log "Creating snapshot: $SNAPSHOT"
|
||||
zfs snapshot "$SNAPSHOT"
|
||||
zfs hold "$ZFS_HOLD" "$SNAPSHOT" || log "ERROR: Failed to hold $SNAPSHOT"
|
||||
;;
|
||||
send)
|
||||
log "Sending snapshot $SNAPSHOT..."
|
||||
if snapshot_send "$SNAPSHOT"; then
|
||||
log "Snapshot send completed. Releasing hold."
|
||||
zfs release "$ZFS_HOLD" "$SNAPSHOT" || log "ERROR: Failed to release hold on $SNAPSHOT"
|
||||
else
|
||||
log "WARNING: Snapshot send encountered errors. Hold retained on $SNAPSHOT."
|
||||
fi
|
||||
;;
|
||||
prune)
|
||||
if [[ "$SNAPSHOT_RETENTION" -gt 0 ]]; then
|
||||
log "Pruning old snapshots in $DATASET (retaining $SNAPSHOT_RETENTION destroyable snapshots)..."
|
||||
|
||||
local -a all_snaps destroyable
|
||||
mapfile -t all_snaps < <(zfs list -H -t snapshot -o name -s creation -d1 "$DATASET" | grep -F "${DATASET}@${SNAPSHOT_PREFIX}-")
|
||||
|
||||
destroyable=()
|
||||
for snap in "${all_snaps[@]}"; do
|
||||
if zfs destroy -n -- "$snap" &>/dev/null; then
|
||||
destroyable+=("$snap")
|
||||
else
|
||||
log "Skipping $snap — snapshot not destroyable (likely held)"
|
||||
fi
|
||||
done
|
||||
|
||||
local count to_destroy
|
||||
count="${#destroyable[@]}"
|
||||
to_destroy=$((count - SNAPSHOT_RETENTION))
|
||||
|
||||
if [[ "$to_destroy" -le 0 ]]; then
|
||||
log "Nothing to prune — only $count destroyable snapshots exist"
|
||||
else
|
||||
local i
|
||||
for (( i=0; i<to_destroy; i++ )); do
|
||||
snap="${destroyable[$i]}"
|
||||
log "Destroying snapshot: $snap"
|
||||
if ! zfs destroy -- "$snap"; then
|
||||
log "WARNING: Failed to destroy $snap despite earlier check"
|
||||
fi
|
||||
done
|
||||
fi
|
||||
else
|
||||
log "Skipping pruning — retention is set to $SNAPSHOT_RETENTION"
|
||||
fi
|
||||
;;
|
||||
*)
|
||||
log "ERROR: Snapshot unknown action: $action"
|
||||
exit 2
|
||||
;;
|
||||
esac
|
||||
}
|
||||
|
||||
# open logging and begin execution
|
||||
mkdir -p "$(dirname -- "$LOG_FILE")"
|
||||
|
||||
start_time=$(date +%s)
|
||||
exec >> "$LOG_FILE" 2>&1
|
||||
trap 'log_close' EXIT
|
||||
trap 'rc=$?; log "ERROR: command failed at line $LINENO (exit $rc)"; exit $rc' ERR
|
||||
|
||||
log "Backlog Started"
|
||||
|
||||
if zfs list -H -t snapshot -o name -d1 "$DATASET" | grep -qxF "$SNAPSHOT"; then
|
||||
log "WARNING: Snapshot $SNAPSHOT already exists. Exiting."
|
||||
exit 1
|
||||
fi
|
||||
|
||||
services stop
|
||||
snapshot create
|
||||
services start
|
||||
snapshot send
|
||||
snapshot prune
|
||||
|
||||
# end
|
||||
Executable
+280
@@ -0,0 +1,280 @@
|
||||
#!/usr/bin/env python3
|
||||
"""Biscayne agave validator status check.
|
||||
|
||||
Collects and displays key health metrics:
|
||||
- Slot position (local vs mainnet, gap, replay rate)
|
||||
- Pod status (running, restarts, age)
|
||||
- Memory usage (cgroup current vs limit, % used)
|
||||
- OOM kills (recent dmesg entries)
|
||||
- Shred relay (packets/sec on port 9100, shred-unwrap.py alive)
|
||||
- Validator process state (from logs)
|
||||
"""
|
||||
|
||||
import json
|
||||
import subprocess
|
||||
import sys
|
||||
import time
|
||||
|
||||
NAMESPACE = "laconic-laconic-70ce4c4b47e23b85"
|
||||
DEPLOYMENT = "laconic-70ce4c4b47e23b85-deployment"
|
||||
KIND_NODE = "laconic-70ce4c4b47e23b85-control-plane"
|
||||
SSH = "rix@biscayne.vaasl.io"
|
||||
MAINNET_RPC = "https://api.mainnet-beta.solana.com"
|
||||
LOCAL_RPC = "http://127.0.0.1:8899"
|
||||
|
||||
|
||||
def ssh(cmd: str, timeout: int = 10) -> str:
|
||||
try:
|
||||
r = subprocess.run(
|
||||
["ssh", SSH, cmd],
|
||||
capture_output=True, text=True, timeout=timeout,
|
||||
)
|
||||
return r.stdout.strip() + r.stderr.strip()
|
||||
except subprocess.TimeoutExpired:
|
||||
return "<timeout>"
|
||||
|
||||
|
||||
def local(cmd: str, timeout: int = 10) -> str:
|
||||
try:
|
||||
r = subprocess.run(
|
||||
cmd, shell=True, capture_output=True, text=True, timeout=timeout,
|
||||
)
|
||||
return r.stdout.strip()
|
||||
except subprocess.TimeoutExpired:
|
||||
return "<timeout>"
|
||||
|
||||
|
||||
def rpc_call(method: str, url: str = LOCAL_RPC, remote: bool = True, params: list | None = None) -> dict | None:
|
||||
payload = json.dumps({"jsonrpc": "2.0", "id": 1, "method": method, "params": params or []})
|
||||
cmd = f"curl -s {url} -X POST -H 'Content-Type: application/json' -d '{payload}'"
|
||||
raw = ssh(cmd) if remote else local(cmd)
|
||||
try:
|
||||
return json.loads(raw)
|
||||
except (json.JSONDecodeError, TypeError):
|
||||
return None
|
||||
|
||||
|
||||
def get_slots() -> tuple[int | None, int | None]:
|
||||
local_resp = rpc_call("getSlot")
|
||||
mainnet_resp = rpc_call("getSlot", MAINNET_RPC, remote=False)
|
||||
local_slot = local_resp.get("result") if local_resp else None
|
||||
mainnet_slot = mainnet_resp.get("result") if mainnet_resp else None
|
||||
return local_slot, mainnet_slot
|
||||
|
||||
|
||||
def get_health() -> str:
|
||||
resp = rpc_call("getHealth")
|
||||
if not resp:
|
||||
return "unreachable"
|
||||
if "result" in resp and resp["result"] == "ok":
|
||||
return "healthy"
|
||||
err = resp.get("error", {})
|
||||
msg = err.get("message", "unknown")
|
||||
behind = err.get("data", {}).get("numSlotsBehind")
|
||||
if behind is not None:
|
||||
return f"behind {behind:,} slots"
|
||||
return msg
|
||||
|
||||
|
||||
def get_pod_status() -> str:
|
||||
cmd = f"kubectl -n {NAMESPACE} get pods -o json"
|
||||
raw = ssh(cmd, timeout=15)
|
||||
try:
|
||||
data = json.loads(raw)
|
||||
except (json.JSONDecodeError, TypeError):
|
||||
return "unknown"
|
||||
items = data.get("items", [])
|
||||
if not items:
|
||||
return "no pods"
|
||||
pod = items[0]
|
||||
name = pod["metadata"]["name"].split("-")[-1]
|
||||
phase = pod["status"].get("phase", "?")
|
||||
containers = pod["status"].get("containerStatuses", [])
|
||||
restarts = sum(c.get("restartCount", 0) for c in containers)
|
||||
ready = sum(1 for c in containers if c.get("ready"))
|
||||
total = len(containers)
|
||||
age = pod["metadata"].get("creationTimestamp", "?")
|
||||
return f"{ready}/{total} {phase} restarts={restarts} pod=..{name} created={age}"
|
||||
|
||||
|
||||
def get_memory() -> str:
|
||||
cmd = (
|
||||
f"docker exec {KIND_NODE} bash -c '"
|
||||
"find /sys/fs/cgroup -name memory.current -path \"*burstable*\" 2>/dev/null | head -1 | "
|
||||
"while read f; do "
|
||||
" dir=$(dirname $f); "
|
||||
" cur=$(cat $f); "
|
||||
" max=$(cat $dir/memory.max 2>/dev/null || echo unknown); "
|
||||
" echo $cur $max; "
|
||||
"done'"
|
||||
)
|
||||
raw = ssh(cmd, timeout=10)
|
||||
try:
|
||||
parts = raw.split()
|
||||
current = int(parts[0])
|
||||
limit_str = parts[1]
|
||||
cur_gb = current / (1024**3)
|
||||
if limit_str == "max":
|
||||
return f"{cur_gb:.0f}GB / unlimited"
|
||||
limit = int(limit_str)
|
||||
lim_gb = limit / (1024**3)
|
||||
pct = (current / limit) * 100
|
||||
return f"{cur_gb:.0f}GB / {lim_gb:.0f}GB ({pct:.0f}%)"
|
||||
except (IndexError, ValueError):
|
||||
return raw or "unknown"
|
||||
|
||||
|
||||
def get_oom_kills() -> str:
|
||||
raw = ssh("sudo dmesg | grep -c 'oom-kill' || echo 0")
|
||||
try:
|
||||
count = int(raw.strip())
|
||||
except ValueError:
|
||||
return "check failed"
|
||||
if count == 0:
|
||||
return "none"
|
||||
# Get kernel uptime-relative timestamp and convert to UTC
|
||||
# dmesg timestamps are seconds since boot; combine with boot time
|
||||
raw = ssh(
|
||||
"BOOT=$(date -d \"$(uptime -s)\" +%s); "
|
||||
"KERN_TS=$(sudo dmesg | grep 'oom-kill' | tail -1 | "
|
||||
" sed 's/\\[\\s*\\([0-9.]*\\)\\].*/\\1/'); "
|
||||
"echo $BOOT $KERN_TS"
|
||||
)
|
||||
try:
|
||||
parts = raw.split()
|
||||
boot_epoch = int(parts[0])
|
||||
kern_secs = float(parts[1])
|
||||
oom_epoch = boot_epoch + int(kern_secs)
|
||||
from datetime import datetime, timezone
|
||||
oom_utc = datetime.fromtimestamp(oom_epoch, tz=timezone.utc).strftime("%Y-%m-%d %H:%M:%S UTC")
|
||||
return f"{count} total (last: {oom_utc})"
|
||||
except (IndexError, ValueError):
|
||||
return f"{count} total (timestamp parse failed)"
|
||||
|
||||
|
||||
def get_relay_rate() -> str:
|
||||
# Two samples 3s apart from /proc/net/snmp
|
||||
cmd = (
|
||||
"T0=$(cat /proc/net/snmp | grep '^Udp:' | tail -1 | awk '{print $2}'); "
|
||||
"sleep 3; "
|
||||
"T1=$(cat /proc/net/snmp | grep '^Udp:' | tail -1 | awk '{print $2}'); "
|
||||
"echo $T0 $T1"
|
||||
)
|
||||
raw = ssh(cmd, timeout=15)
|
||||
try:
|
||||
parts = raw.split()
|
||||
t0, t1 = int(parts[0]), int(parts[1])
|
||||
rate = (t1 - t0) / 3
|
||||
return f"{rate:,.0f} UDP dgrams/sec (all ports)"
|
||||
except (IndexError, ValueError):
|
||||
return raw or "unknown"
|
||||
|
||||
|
||||
def get_shreds_per_sec() -> str:
|
||||
"""Count UDP packets on TVU port 9000 over 3 seconds using tcpdump."""
|
||||
cmd = "sudo timeout 3 tcpdump -i any udp dst port 9000 -q 2>&1 | grep -oP '\\d+(?= packets captured)'"
|
||||
raw = ssh(cmd, timeout=15)
|
||||
try:
|
||||
count = int(raw.strip())
|
||||
rate = count / 3
|
||||
return f"{rate:,.0f} shreds/sec ({count:,} in 3s)"
|
||||
except (ValueError, TypeError):
|
||||
return raw or "unknown"
|
||||
|
||||
|
||||
def get_unwrap_status() -> str:
|
||||
raw = ssh("ps -p $(pgrep -f shred-unwrap | head -1) -o pid,etime,rss --no-headers 2>/dev/null || echo dead")
|
||||
if "dead" in raw or not raw.strip():
|
||||
return "NOT RUNNING"
|
||||
parts = raw.split()
|
||||
if len(parts) >= 3:
|
||||
pid, etime, rss_kb = parts[0], parts[1], parts[2]
|
||||
rss_mb = int(rss_kb) / 1024
|
||||
return f"pid={pid} uptime={etime} rss={rss_mb:.0f}MB"
|
||||
return raw
|
||||
|
||||
|
||||
def get_replay_rate() -> tuple[float | None, int | None, int | None]:
|
||||
"""Sample processed slot twice over 10s to measure replay rate."""
|
||||
params = [{"commitment": "processed"}]
|
||||
r0 = rpc_call("getSlot", params=params)
|
||||
s0 = r0.get("result") if r0 else None
|
||||
if s0 is None:
|
||||
return None, None, None
|
||||
t0 = time.monotonic()
|
||||
time.sleep(10)
|
||||
r1 = rpc_call("getSlot", params=params)
|
||||
s1 = r1.get("result") if r1 else None
|
||||
if s1 is None:
|
||||
return None, s0, None
|
||||
dt = time.monotonic() - t0
|
||||
rate = (s1 - s0) / dt if s1 != s0 else 0
|
||||
return rate, s0, s1
|
||||
|
||||
|
||||
def main() -> None:
|
||||
print("=" * 60)
|
||||
print(" BISCAYNE VALIDATOR STATUS")
|
||||
print("=" * 60)
|
||||
|
||||
# Health + slots
|
||||
print("\n--- RPC ---")
|
||||
health = get_health()
|
||||
local_slot, mainnet_slot = get_slots()
|
||||
print(f" Health: {health}")
|
||||
if local_slot is not None:
|
||||
print(f" Local slot: {local_slot:,}")
|
||||
else:
|
||||
print(" Local slot: unreachable")
|
||||
if mainnet_slot is not None:
|
||||
print(f" Mainnet slot: {mainnet_slot:,}")
|
||||
if local_slot and mainnet_slot:
|
||||
gap = mainnet_slot - local_slot
|
||||
print(f" Gap: {gap:,} slots")
|
||||
|
||||
# Replay rate (10s sample)
|
||||
print("\n--- Replay ---")
|
||||
print(" Sampling replay rate (10s)...", end="", flush=True)
|
||||
rate, s0, s1 = get_replay_rate()
|
||||
if rate is not None:
|
||||
print(f"\r Replay rate: {rate:.1f} slots/sec ({s0:,} → {s1:,})")
|
||||
net = rate - 2.5
|
||||
if net > 0:
|
||||
print(f" Net catchup: +{net:.1f} slots/sec (gaining)")
|
||||
elif net < 0:
|
||||
print(f" Net catchup: {net:.1f} slots/sec (falling behind)")
|
||||
else:
|
||||
print(" Net catchup: 0 (keeping pace)")
|
||||
else:
|
||||
print("\r Replay rate: could not measure")
|
||||
|
||||
# Pod
|
||||
print("\n--- Pod ---")
|
||||
pod = get_pod_status()
|
||||
print(f" {pod}")
|
||||
|
||||
# Memory
|
||||
print("\n--- Memory ---")
|
||||
mem = get_memory()
|
||||
print(f" Cgroup: {mem}")
|
||||
|
||||
# OOM
|
||||
oom = get_oom_kills()
|
||||
print(f" OOM kills: {oom}")
|
||||
|
||||
# Relay
|
||||
print("\n--- Shred Relay ---")
|
||||
unwrap = get_unwrap_status()
|
||||
print(f" shred-unwrap: {unwrap}")
|
||||
print(" Measuring shred rate (3s)...", end="", flush=True)
|
||||
shreds = get_shreds_per_sec()
|
||||
print(f"\r TVU shreds: {shreds} ")
|
||||
print(" Measuring UDP rate (3s)...", end="", flush=True)
|
||||
relay = get_relay_rate()
|
||||
print(f"\r UDP inbound: {relay} ")
|
||||
|
||||
print("\n" + "=" * 60)
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
Executable
+546
@@ -0,0 +1,546 @@
|
||||
#!/usr/bin/env python3
|
||||
"""Download Solana snapshots using aria2c for parallel multi-connection downloads.
|
||||
|
||||
Discovers snapshot sources by querying getClusterNodes for all RPCs in the
|
||||
cluster, probing each for available snapshots, benchmarking download speed,
|
||||
and downloading from the fastest source using aria2c (16 connections by default).
|
||||
|
||||
Based on the discovery approach from etcusr/solana-snapshot-finder but replaces
|
||||
the single-connection wget download with aria2c parallel chunked downloads.
|
||||
|
||||
Usage:
|
||||
# Download to /srv/solana/snapshots (mainnet, 16 connections)
|
||||
./snapshot-download.py -o /srv/solana/snapshots
|
||||
|
||||
# Dry run — find best source, print URL
|
||||
./snapshot-download.py --dry-run
|
||||
|
||||
# Custom RPC for cluster node discovery + 32 connections
|
||||
./snapshot-download.py -r https://api.mainnet-beta.solana.com -n 32
|
||||
|
||||
# Testnet
|
||||
./snapshot-download.py -c testnet -o /data/snapshots
|
||||
|
||||
Requirements:
|
||||
- aria2c (apt install aria2)
|
||||
- python3 >= 3.10 (stdlib only, no pip dependencies)
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import argparse
|
||||
import concurrent.futures
|
||||
import json
|
||||
import logging
|
||||
import os
|
||||
import re
|
||||
import shutil
|
||||
import subprocess
|
||||
import sys
|
||||
import time
|
||||
import urllib.error
|
||||
import urllib.request
|
||||
from dataclasses import dataclass, field
|
||||
from http.client import HTTPResponse
|
||||
from pathlib import Path
|
||||
from typing import NoReturn
|
||||
from urllib.request import Request
|
||||
|
||||
log: logging.Logger = logging.getLogger("snapshot-download")
|
||||
|
||||
CLUSTER_RPC: dict[str, str] = {
|
||||
"mainnet-beta": "https://api.mainnet-beta.solana.com",
|
||||
"testnet": "https://api.testnet.solana.com",
|
||||
"devnet": "https://api.devnet.solana.com",
|
||||
}
|
||||
|
||||
# Snapshot filenames:
|
||||
# snapshot-<slot>-<hash>.tar.zst
|
||||
# incremental-snapshot-<base_slot>-<slot>-<hash>.tar.zst
|
||||
FULL_SNAP_RE: re.Pattern[str] = re.compile(
|
||||
r"^snapshot-(\d+)-([A-Za-z0-9]+)\.tar\.(zst|bz2)$"
|
||||
)
|
||||
INCR_SNAP_RE: re.Pattern[str] = re.compile(
|
||||
r"^incremental-snapshot-(\d+)-(\d+)-([A-Za-z0-9]+)\.tar\.(zst|bz2)$"
|
||||
)
|
||||
|
||||
|
||||
@dataclass
|
||||
class SnapshotSource:
|
||||
"""A snapshot file available from a specific RPC node."""
|
||||
|
||||
rpc_address: str
|
||||
# Full redirect paths as returned by the server (e.g. /snapshot-123-hash.tar.zst)
|
||||
file_paths: list[str] = field(default_factory=list)
|
||||
slots_diff: int = 0
|
||||
latency_ms: float = 0.0
|
||||
download_speed: float = 0.0 # bytes/sec
|
||||
|
||||
|
||||
# -- JSON-RPC helpers ----------------------------------------------------------
|
||||
|
||||
|
||||
class _NoRedirectHandler(urllib.request.HTTPRedirectHandler):
|
||||
"""Handler that captures redirect Location instead of following it."""
|
||||
|
||||
def redirect_request(
|
||||
self,
|
||||
req: Request,
|
||||
fp: HTTPResponse,
|
||||
code: int,
|
||||
msg: str,
|
||||
headers: dict[str, str], # type: ignore[override]
|
||||
newurl: str,
|
||||
) -> None:
|
||||
return None
|
||||
|
||||
|
||||
def rpc_post(url: str, method: str, params: list[object] | None = None,
|
||||
timeout: int = 25) -> object | None:
|
||||
"""JSON-RPC POST. Returns parsed 'result' field or None on error."""
|
||||
payload: bytes = json.dumps({
|
||||
"jsonrpc": "2.0", "id": 1,
|
||||
"method": method, "params": params or [],
|
||||
}).encode()
|
||||
req = Request(url, data=payload,
|
||||
headers={"Content-Type": "application/json"})
|
||||
try:
|
||||
with urllib.request.urlopen(req, timeout=timeout) as resp:
|
||||
data: dict[str, object] = json.loads(resp.read())
|
||||
return data.get("result")
|
||||
except (urllib.error.URLError, json.JSONDecodeError, OSError, TimeoutError) as e:
|
||||
log.debug("rpc_post %s %s failed: %s", url, method, e)
|
||||
return None
|
||||
|
||||
|
||||
def head_no_follow(url: str, timeout: float = 3) -> tuple[str | None, float]:
|
||||
"""HEAD request without following redirects.
|
||||
|
||||
Returns (Location header value, latency_sec) if the server returned a
|
||||
3xx redirect. Returns (None, 0.0) on any error or non-redirect response.
|
||||
"""
|
||||
opener: urllib.request.OpenerDirector = urllib.request.build_opener(_NoRedirectHandler)
|
||||
req = Request(url, method="HEAD")
|
||||
try:
|
||||
start: float = time.monotonic()
|
||||
resp: HTTPResponse = opener.open(req, timeout=timeout) # type: ignore[assignment]
|
||||
latency: float = time.monotonic() - start
|
||||
# Non-redirect (2xx) — server didn't redirect, not useful for discovery
|
||||
location: str | None = resp.headers.get("Location")
|
||||
resp.close()
|
||||
return location, latency
|
||||
except urllib.error.HTTPError as e:
|
||||
# 3xx redirects raise HTTPError with the redirect info
|
||||
latency = time.monotonic() - start # type: ignore[possibly-undefined]
|
||||
location = e.headers.get("Location")
|
||||
if location and 300 <= e.code < 400:
|
||||
return location, latency
|
||||
return None, 0.0
|
||||
except (urllib.error.URLError, OSError, TimeoutError):
|
||||
return None, 0.0
|
||||
|
||||
|
||||
# -- Discovery -----------------------------------------------------------------
|
||||
|
||||
|
||||
def get_current_slot(rpc_url: str) -> int | None:
|
||||
"""Get current slot from RPC."""
|
||||
result: object | None = rpc_post(rpc_url, "getSlot")
|
||||
if isinstance(result, int):
|
||||
return result
|
||||
return None
|
||||
|
||||
|
||||
def get_cluster_rpc_nodes(rpc_url: str, version_filter: str | None = None) -> list[str]:
|
||||
"""Get all RPC node addresses from getClusterNodes."""
|
||||
result: object | None = rpc_post(rpc_url, "getClusterNodes")
|
||||
if not isinstance(result, list):
|
||||
return []
|
||||
|
||||
rpc_addrs: list[str] = []
|
||||
for node in result:
|
||||
if not isinstance(node, dict):
|
||||
continue
|
||||
if version_filter is not None:
|
||||
node_version: str | None = node.get("version")
|
||||
if node_version and not node_version.startswith(version_filter):
|
||||
continue
|
||||
rpc: str | None = node.get("rpc")
|
||||
if rpc:
|
||||
rpc_addrs.append(rpc)
|
||||
return list(set(rpc_addrs))
|
||||
|
||||
|
||||
def _parse_snapshot_filename(location: str) -> tuple[str, str | None]:
|
||||
"""Extract filename and full redirect path from Location header.
|
||||
|
||||
Returns (filename, full_path). full_path includes any path prefix
|
||||
the server returned (e.g. '/snapshots/snapshot-123-hash.tar.zst').
|
||||
"""
|
||||
# Location may be absolute URL or relative path
|
||||
if location.startswith("http://") or location.startswith("https://"):
|
||||
# Absolute URL — extract path
|
||||
from urllib.parse import urlparse
|
||||
path: str = urlparse(location).path
|
||||
else:
|
||||
path = location
|
||||
|
||||
filename: str = path.rsplit("/", 1)[-1]
|
||||
return filename, path
|
||||
|
||||
|
||||
def probe_rpc_snapshot(
|
||||
rpc_address: str,
|
||||
current_slot: int,
|
||||
max_age_slots: int,
|
||||
max_latency_ms: float,
|
||||
) -> SnapshotSource | None:
|
||||
"""Probe a single RPC node for available snapshots.
|
||||
|
||||
Probes for full snapshot first (required), then incremental. Records all
|
||||
available files. Which files to actually download is decided at download
|
||||
time based on what already exists locally — not here.
|
||||
|
||||
Based on the discovery approach from etcusr/solana-snapshot-finder.
|
||||
"""
|
||||
full_url: str = f"http://{rpc_address}/snapshot.tar.bz2"
|
||||
|
||||
# Full snapshot is required — every source must have one
|
||||
full_location, full_latency = head_no_follow(full_url, timeout=2)
|
||||
if not full_location:
|
||||
return None
|
||||
|
||||
latency_ms: float = full_latency * 1000
|
||||
if latency_ms > max_latency_ms:
|
||||
return None
|
||||
|
||||
full_filename, full_path = _parse_snapshot_filename(full_location)
|
||||
fm: re.Match[str] | None = FULL_SNAP_RE.match(full_filename)
|
||||
if not fm:
|
||||
return None
|
||||
|
||||
full_snap_slot: int = int(fm.group(1))
|
||||
slots_diff: int = current_slot - full_snap_slot
|
||||
|
||||
if slots_diff > max_age_slots or slots_diff < -100:
|
||||
return None
|
||||
|
||||
file_paths: list[str] = [full_path]
|
||||
|
||||
# Also check for incremental snapshot
|
||||
inc_url: str = f"http://{rpc_address}/incremental-snapshot.tar.bz2"
|
||||
inc_location, _ = head_no_follow(inc_url, timeout=2)
|
||||
if inc_location:
|
||||
inc_filename, inc_path = _parse_snapshot_filename(inc_location)
|
||||
m: re.Match[str] | None = INCR_SNAP_RE.match(inc_filename)
|
||||
if m:
|
||||
inc_base_slot: int = int(m.group(1))
|
||||
# Incremental must be based on this source's full snapshot
|
||||
if inc_base_slot == full_snap_slot:
|
||||
file_paths.append(inc_path)
|
||||
|
||||
return SnapshotSource(
|
||||
rpc_address=rpc_address,
|
||||
file_paths=file_paths,
|
||||
slots_diff=slots_diff,
|
||||
latency_ms=latency_ms,
|
||||
)
|
||||
|
||||
|
||||
def discover_sources(
|
||||
rpc_url: str,
|
||||
current_slot: int,
|
||||
max_age_slots: int,
|
||||
max_latency_ms: float,
|
||||
threads: int,
|
||||
version_filter: str | None,
|
||||
) -> list[SnapshotSource]:
|
||||
"""Discover all snapshot sources from the cluster."""
|
||||
rpc_nodes: list[str] = get_cluster_rpc_nodes(rpc_url, version_filter)
|
||||
if not rpc_nodes:
|
||||
log.error("No RPC nodes found via getClusterNodes")
|
||||
return []
|
||||
|
||||
log.info("Found %d RPC nodes, probing for snapshots...", len(rpc_nodes))
|
||||
|
||||
sources: list[SnapshotSource] = []
|
||||
with concurrent.futures.ThreadPoolExecutor(max_workers=threads) as pool:
|
||||
futures: dict[concurrent.futures.Future[SnapshotSource | None], str] = {
|
||||
pool.submit(
|
||||
probe_rpc_snapshot, addr, current_slot,
|
||||
max_age_slots, max_latency_ms,
|
||||
): addr
|
||||
for addr in rpc_nodes
|
||||
}
|
||||
done: int = 0
|
||||
for future in concurrent.futures.as_completed(futures):
|
||||
done += 1
|
||||
if done % 200 == 0:
|
||||
log.info(" probed %d/%d nodes, %d sources found",
|
||||
done, len(rpc_nodes), len(sources))
|
||||
try:
|
||||
result: SnapshotSource | None = future.result()
|
||||
except (urllib.error.URLError, OSError, TimeoutError) as e:
|
||||
log.debug("Probe failed for %s: %s", futures[future], e)
|
||||
continue
|
||||
if result:
|
||||
sources.append(result)
|
||||
|
||||
log.info("Found %d RPC nodes with suitable snapshots", len(sources))
|
||||
return sources
|
||||
|
||||
|
||||
# -- Speed benchmark -----------------------------------------------------------
|
||||
|
||||
|
||||
def measure_speed(rpc_address: str, measure_time: int = 7) -> float:
|
||||
"""Measure download speed from an RPC node. Returns bytes/sec."""
|
||||
url: str = f"http://{rpc_address}/snapshot.tar.bz2"
|
||||
req = Request(url)
|
||||
try:
|
||||
with urllib.request.urlopen(req, timeout=measure_time + 5) as resp:
|
||||
start: float = time.monotonic()
|
||||
total: int = 0
|
||||
while True:
|
||||
elapsed: float = time.monotonic() - start
|
||||
if elapsed >= measure_time:
|
||||
break
|
||||
chunk: bytes = resp.read(81920)
|
||||
if not chunk:
|
||||
break
|
||||
total += len(chunk)
|
||||
elapsed = time.monotonic() - start
|
||||
if elapsed <= 0:
|
||||
return 0.0
|
||||
return total / elapsed
|
||||
except (urllib.error.URLError, OSError, TimeoutError):
|
||||
return 0.0
|
||||
|
||||
|
||||
# -- Download ------------------------------------------------------------------
|
||||
|
||||
|
||||
def download_aria2c(
|
||||
urls: list[str],
|
||||
output_dir: str,
|
||||
filename: str,
|
||||
connections: int = 16,
|
||||
) -> bool:
|
||||
"""Download a file using aria2c with parallel connections.
|
||||
|
||||
When multiple URLs are provided, aria2c treats them as mirrors of the
|
||||
same file and distributes chunks across all of them.
|
||||
"""
|
||||
num_mirrors: int = len(urls)
|
||||
total_splits: int = max(connections, connections * num_mirrors)
|
||||
cmd: list[str] = [
|
||||
"aria2c",
|
||||
"--file-allocation=none",
|
||||
"--continue=true",
|
||||
f"--max-connection-per-server={connections}",
|
||||
f"--split={total_splits}",
|
||||
"--min-split-size=50M",
|
||||
# aria2c retries individual chunk connections on transient network
|
||||
# errors (TCP reset, timeout). This is transport-level retry analogous
|
||||
# to TCP retransmit, not application-level retry of a failed operation.
|
||||
"--max-tries=5",
|
||||
"--retry-wait=5",
|
||||
"--timeout=60",
|
||||
"--connect-timeout=10",
|
||||
"--summary-interval=10",
|
||||
"--console-log-level=notice",
|
||||
f"--dir={output_dir}",
|
||||
f"--out={filename}",
|
||||
"--auto-file-renaming=false",
|
||||
"--allow-overwrite=true",
|
||||
*urls,
|
||||
]
|
||||
|
||||
log.info("Downloading %s", filename)
|
||||
log.info(" aria2c: %d connections × %d mirrors (%d splits)",
|
||||
connections, num_mirrors, total_splits)
|
||||
|
||||
start: float = time.monotonic()
|
||||
result: subprocess.CompletedProcess[bytes] = subprocess.run(cmd)
|
||||
elapsed: float = time.monotonic() - start
|
||||
|
||||
if result.returncode != 0:
|
||||
log.error("aria2c failed with exit code %d", result.returncode)
|
||||
return False
|
||||
|
||||
filepath: Path = Path(output_dir) / filename
|
||||
if not filepath.exists():
|
||||
log.error("aria2c reported success but %s does not exist", filepath)
|
||||
return False
|
||||
|
||||
size_bytes: int = filepath.stat().st_size
|
||||
size_gb: float = size_bytes / (1024 ** 3)
|
||||
avg_mb: float = size_bytes / elapsed / (1024 ** 2) if elapsed > 0 else 0
|
||||
log.info(" Done: %.1f GB in %.0fs (%.1f MiB/s avg)", size_gb, elapsed, avg_mb)
|
||||
return True
|
||||
|
||||
|
||||
# -- Main ----------------------------------------------------------------------
|
||||
|
||||
|
||||
def main() -> int:
|
||||
p: argparse.ArgumentParser = argparse.ArgumentParser(
|
||||
description="Download Solana snapshots with aria2c parallel downloads",
|
||||
)
|
||||
p.add_argument("-o", "--output", default="/srv/solana/snapshots",
|
||||
help="Snapshot output directory (default: /srv/solana/snapshots)")
|
||||
p.add_argument("-c", "--cluster", default="mainnet-beta",
|
||||
choices=list(CLUSTER_RPC),
|
||||
help="Solana cluster (default: mainnet-beta)")
|
||||
p.add_argument("-r", "--rpc", default=None,
|
||||
help="RPC URL for cluster discovery (default: public RPC)")
|
||||
p.add_argument("-n", "--connections", type=int, default=16,
|
||||
help="aria2c connections per download (default: 16)")
|
||||
p.add_argument("-t", "--threads", type=int, default=500,
|
||||
help="Threads for parallel RPC probing (default: 500)")
|
||||
p.add_argument("--max-snapshot-age", type=int, default=1300,
|
||||
help="Max snapshot age in slots (default: 1300)")
|
||||
p.add_argument("--max-latency", type=float, default=100,
|
||||
help="Max RPC probe latency in ms (default: 100)")
|
||||
p.add_argument("--min-download-speed", type=int, default=20,
|
||||
help="Min download speed in MiB/s (default: 20)")
|
||||
p.add_argument("--measurement-time", type=int, default=7,
|
||||
help="Speed measurement duration in seconds (default: 7)")
|
||||
p.add_argument("--max-speed-checks", type=int, default=15,
|
||||
help="Max nodes to benchmark before giving up (default: 15)")
|
||||
p.add_argument("--version", default=None,
|
||||
help="Filter nodes by version prefix (e.g. '2.2')")
|
||||
p.add_argument("--full-only", action="store_true",
|
||||
help="Download only full snapshot, skip incremental")
|
||||
p.add_argument("--dry-run", action="store_true",
|
||||
help="Find best source and print URL, don't download")
|
||||
p.add_argument("-v", "--verbose", action="store_true")
|
||||
args: argparse.Namespace = p.parse_args()
|
||||
|
||||
logging.basicConfig(
|
||||
level=logging.DEBUG if args.verbose else logging.INFO,
|
||||
format="%(asctime)s %(levelname)s %(message)s",
|
||||
datefmt="%H:%M:%S",
|
||||
)
|
||||
|
||||
rpc_url: str = args.rpc or CLUSTER_RPC[args.cluster]
|
||||
|
||||
# aria2c is required for actual downloads (not dry-run)
|
||||
if not args.dry_run and not shutil.which("aria2c"):
|
||||
log.error("aria2c not found. Install with: apt install aria2")
|
||||
return 1
|
||||
|
||||
# Get current slot
|
||||
log.info("Cluster: %s | RPC: %s", args.cluster, rpc_url)
|
||||
current_slot: int | None = get_current_slot(rpc_url)
|
||||
if current_slot is None:
|
||||
log.error("Cannot get current slot from %s", rpc_url)
|
||||
return 1
|
||||
log.info("Current slot: %d", current_slot)
|
||||
|
||||
# Discover sources
|
||||
sources: list[SnapshotSource] = discover_sources(
|
||||
rpc_url, current_slot,
|
||||
max_age_slots=args.max_snapshot_age,
|
||||
max_latency_ms=args.max_latency,
|
||||
threads=args.threads,
|
||||
version_filter=args.version,
|
||||
)
|
||||
if not sources:
|
||||
log.error("No snapshot sources found")
|
||||
return 1
|
||||
|
||||
# Sort by latency (lowest first) for speed benchmarking
|
||||
sources.sort(key=lambda s: s.latency_ms)
|
||||
|
||||
# Benchmark top candidates — all speeds in MiB/s (binary, 1 MiB = 1048576 bytes)
|
||||
log.info("Benchmarking download speed on top %d sources...", args.max_speed_checks)
|
||||
fast_sources: list[SnapshotSource] = []
|
||||
checked: int = 0
|
||||
min_speed_bytes: int = args.min_download_speed * 1024 * 1024 # MiB to bytes
|
||||
|
||||
for source in sources:
|
||||
if checked >= args.max_speed_checks:
|
||||
break
|
||||
checked += 1
|
||||
|
||||
speed: float = measure_speed(source.rpc_address, args.measurement_time)
|
||||
source.download_speed = speed
|
||||
speed_mib: float = speed / (1024 ** 2)
|
||||
|
||||
if speed < min_speed_bytes:
|
||||
log.info(" %s: %.1f MiB/s (too slow, need >=%d MiB/s)",
|
||||
source.rpc_address, speed_mib, args.min_download_speed)
|
||||
continue
|
||||
|
||||
log.info(" %s: %.1f MiB/s (latency: %.0fms, age: %d slots)",
|
||||
source.rpc_address, speed_mib,
|
||||
source.latency_ms, source.slots_diff)
|
||||
fast_sources.append(source)
|
||||
|
||||
if not fast_sources:
|
||||
log.error("No source met minimum speed requirement (%d MiB/s)",
|
||||
args.min_download_speed)
|
||||
log.info("Try: --min-download-speed 10")
|
||||
return 1
|
||||
|
||||
# Use the fastest source as primary, collect mirrors for each file
|
||||
best: SnapshotSource = fast_sources[0]
|
||||
file_paths: list[str] = best.file_paths
|
||||
if args.full_only:
|
||||
file_paths = [fp for fp in file_paths
|
||||
if fp.rsplit("/", 1)[-1].startswith("snapshot-")]
|
||||
|
||||
# Build mirror URL lists: for each file, collect URLs from all fast sources
|
||||
# that serve the same filename
|
||||
download_plan: list[tuple[str, list[str]]] = []
|
||||
for fp in file_paths:
|
||||
filename: str = fp.rsplit("/", 1)[-1]
|
||||
mirror_urls: list[str] = [f"http://{best.rpc_address}{fp}"]
|
||||
for other in fast_sources[1:]:
|
||||
for other_fp in other.file_paths:
|
||||
if other_fp.rsplit("/", 1)[-1] == filename:
|
||||
mirror_urls.append(f"http://{other.rpc_address}{other_fp}")
|
||||
break
|
||||
download_plan.append((filename, mirror_urls))
|
||||
|
||||
speed_mib: float = best.download_speed / (1024 ** 2)
|
||||
log.info("Best source: %s (%.1f MiB/s), %d mirrors total",
|
||||
best.rpc_address, speed_mib, len(fast_sources))
|
||||
for filename, mirror_urls in download_plan:
|
||||
log.info(" %s (%d mirrors)", filename, len(mirror_urls))
|
||||
for url in mirror_urls:
|
||||
log.info(" %s", url)
|
||||
|
||||
if args.dry_run:
|
||||
for _, mirror_urls in download_plan:
|
||||
for url in mirror_urls:
|
||||
print(url)
|
||||
return 0
|
||||
|
||||
# Download — skip files that already exist locally
|
||||
os.makedirs(args.output, exist_ok=True)
|
||||
total_start: float = time.monotonic()
|
||||
|
||||
for filename, mirror_urls in download_plan:
|
||||
filepath: Path = Path(args.output) / filename
|
||||
if filepath.exists() and filepath.stat().st_size > 0:
|
||||
log.info("Skipping %s (already exists: %.1f GB)",
|
||||
filename, filepath.stat().st_size / (1024 ** 3))
|
||||
continue
|
||||
if not download_aria2c(mirror_urls, args.output, filename, args.connections):
|
||||
log.error("Failed to download %s", filename)
|
||||
return 1
|
||||
|
||||
total_elapsed: float = time.monotonic() - total_start
|
||||
log.info("All downloads complete in %.0fs", total_elapsed)
|
||||
for filename, _ in download_plan:
|
||||
fp: Path = Path(args.output) / filename
|
||||
if fp.exists():
|
||||
log.info(" %s (%.1f GB)", fp.name, fp.stat().st_size / (1024 ** 3))
|
||||
|
||||
return 0
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
sys.exit(main())
|
||||
@@ -0,0 +1,109 @@
|
||||
# ZFS Setup for Biscayne
|
||||
|
||||
## Current State
|
||||
|
||||
```
|
||||
biscayne none (pool root)
|
||||
biscayne/DATA none
|
||||
biscayne/DATA/home /home 42G
|
||||
biscayne/DATA/home/solana /home/solana 2.9G
|
||||
biscayne/DATA/srv /srv 712G
|
||||
biscayne/DATA/srv/backups /srv/backups 208G
|
||||
biscayne/DATA/volumes/solana (zvol, 4T) → block-mounted at /srv/solana
|
||||
```
|
||||
|
||||
Docker root: `/var/lib/docker` on root filesystem (`/dev/md0`, 439G).
|
||||
|
||||
## Target State
|
||||
|
||||
```
|
||||
biscayne/DATA/deployments /srv/deployments ← laconic-so deployment dirs (snapshotted)
|
||||
biscayne/DATA/var/docker /var/lib/docker ← docker storage on ZFS
|
||||
biscayne/DATA/volumes/solana (zvol, 4T) ← bulk solana data (not backed up)
|
||||
```
|
||||
|
||||
## Steps
|
||||
|
||||
### 1. Create deployments dataset
|
||||
|
||||
```bash
|
||||
zfs create -o mountpoint=/srv/deployments biscayne/DATA/deployments
|
||||
```
|
||||
|
||||
### 2. Move docker onto ZFS
|
||||
|
||||
Stop docker and all containers first:
|
||||
|
||||
```bash
|
||||
systemctl stop docker.socket docker.service
|
||||
```
|
||||
|
||||
Create the dataset:
|
||||
|
||||
```bash
|
||||
zfs create -o mountpoint=/var/lib/docker biscayne/DATA/var
|
||||
zfs create biscayne/DATA/var/docker
|
||||
```
|
||||
|
||||
Copy existing docker data (if any worth keeping):
|
||||
|
||||
```bash
|
||||
rsync -aHAX /var/lib/docker.bak/ /var/lib/docker/
|
||||
```
|
||||
|
||||
Or just start fresh — the only running containers are telegraf/influxdb monitoring
|
||||
which can be recreated.
|
||||
|
||||
Start docker:
|
||||
|
||||
```bash
|
||||
systemctl start docker.service
|
||||
```
|
||||
|
||||
### 3. Grant ZFS permissions to the backup user
|
||||
|
||||
```bash
|
||||
zfs allow -u <backup-user> destroy,snapshot,send,hold,release,mount biscayne/DATA/deployments
|
||||
```
|
||||
|
||||
### 4. Create remote receiving datasets
|
||||
|
||||
On mysterio:
|
||||
|
||||
```bash
|
||||
zfs create -p edith/DATA/backlog/biscayne-main
|
||||
```
|
||||
|
||||
On ardham:
|
||||
|
||||
```bash
|
||||
zfs create -p batterywharf/DATA/backlog/biscayne-main
|
||||
```
|
||||
|
||||
These will fail until SSH keys and network access are configured for biscayne
|
||||
to reach these hosts. The backup script handles this gracefully.
|
||||
|
||||
### 5. Install backlog.sh and crontab
|
||||
|
||||
```bash
|
||||
mkdir -p ~/.local/bin
|
||||
cp scripts/backlog.sh ~/.local/bin/backlog.sh
|
||||
chmod +x ~/.local/bin/backlog.sh
|
||||
crontab -e
|
||||
# Add: 01 0 * * * /home/<user>/.local/bin/backlog.sh
|
||||
```
|
||||
|
||||
## Volume Layout
|
||||
|
||||
laconic-so deployment at `/srv/deployments/agave/`:
|
||||
|
||||
| Volume | Location | Backed up |
|
||||
|---|---|---|
|
||||
| validator-config | `/srv/deployments/agave/data/validator-config/` | Yes (ZFS snapshot) |
|
||||
| doublezero-config | `/srv/deployments/agave/data/doublezero-config/` | Yes (ZFS snapshot) |
|
||||
| validator-ledger | `/srv/solana/ledger/` (zvol) | No (rebuildable) |
|
||||
| validator-accounts | `/srv/solana/accounts/` (zvol) | No (rebuildable) |
|
||||
| validator-snapshots | `/srv/solana/snapshots/` (zvol) | No (rebuildable) |
|
||||
|
||||
The laconic-so spec.yml must map the heavy volumes to zvol paths and the small
|
||||
config volumes to the deployment directory.
|
||||
Reference in New Issue
Block a user