Compare commits

...
Author SHA1 Message Date
nabarun 320517b6b0 Add URL to dashboard panel label 2025-04-08 12:21:50 +05:30
nabarun 159402921d Use alias for blackbox targets in dashboard 2025-04-07 14:23:14 +05:30
nabarun fb0138e975 Add alerts for testnet services 2025-04-04 18:06:53 +05:30
nabarun 27412519b4 Add readme for monitoring testnet services 2025-04-02 18:54:06 +05:30
prathamesh 873a6d472c Update webapp deployment flow for supporting custom domains (#963)
Part of https://www.notion.so/Support-custom-domains-in-deploy-laconic-com-18aa6b22d4728067a44ae27090c02ce5 and cerc-io/snowballtools-base#47

- Set `value` (IP address for `A` resource) in the DNS records
- Update `deploy-webapp-from-registry` command with an option to pass k8s cluster IP address (only required with fqdn policy `allow`)

Reviewed-on: cerc-io/stack-orchestrator#963
Reviewed-by: ashwin <ashwin@noreply.git.vdb.to>
Co-authored-by: Prathamesh Musale <prathamesh.musale0@gmail.com>
Co-committed-by: Prathamesh Musale <prathamesh.musale0@gmail.com>
2025-02-04 13:26:31 +00:00
9 changed files with 307 additions and 13 deletions
@@ -1,7 +1,8 @@
modules:
http_2xx:
prober: http
timeout: 5s
timeout: 15s
http:
valid_status_codes: [] #default to 2xx
method: GET
preferred_ip_protocol: ip4
@@ -133,10 +133,13 @@
"type": "prometheus",
"uid": "PBFA97CFB590B2093"
},
"expr": "probe_success{instance=~\"$target\"}",
"format": "time_series",
"instant": true,
"refId": "A"
}
],
"title": "$target status",
"title": "$target ($url)",
"type": "row"
},
{
@@ -1057,6 +1060,29 @@
"tagsQuery": "",
"type": "query",
"useTags": false
},
{
"current": {
"selected": false,
"text": "",
"value": ""
},
"datasource": {
"type": "prometheus",
"uid": "PBFA97CFB590B2093"
},
"definition": "label_values(probe_success{instance=~\"$target\"}, url)",
"hide": 2,
"includeAll": false,
"multi": false,
"name": "url",
"options": [],
"query": "label_values(probe_success{instance=~\"$target\"}, url)",
"refresh": 2,
"regex": "",
"skipUrlSync": false,
"sort": 0,
"type": "query"
}
]
},
@@ -8,6 +8,7 @@ policies:
group_by:
- grafana_folder
- alertname
- instance
routes:
- receiver: SlackNotifier
object_matchers:
@@ -25,20 +25,34 @@ scrape_configs:
module: [http_2xx]
static_configs:
# Add URLs to be monitored below
- targets:
# - https://github.com
# - targets: ["https://github.com"]
# labels:
# alias: "GitHub"
# url: "https://github.com"
relabel_configs:
# Forward the original target URL as the 'target' parameter.
- source_labels: [__address__]
regex: (.*)(:80)?
target_label: __param_target
- source_labels: [__param_target]
regex: (.*)
# Use the custom alias if defined for the 'instance' label.
- source_labels: [alias]
target_label: instance
replacement: ${1}
- source_labels: []
regex: .*
target_label: __address__
action: replace
# Preserve the URL label
- source_labels: [url]
target_label: url
action: replace
# If no alias is set, fall back to the target URL.
- source_labels: [instance]
regex: ^$
target_label: instance
replacement: ${__param_target}
# Finally, tell Prometheus to scrape the blackbox_exporter.
- target_label: __address__
replacement: blackbox:9115
# Drop the original alias label as it's now redundant with instance
- action: labeldrop
regex: ^alias$
- job_name: chain_heads
scrape_interval: 10s
@@ -0,0 +1,64 @@
apiVersion: 1
groups:
- orgId: 1
name: testnet
folder: TestnetAlerts
interval: 30s
rules:
- uid: endpoint_down
title: endpoint_down
condition: condition
data:
- refId: probe_success
relativeTimeRange:
from: 600
to: 0
datasourceUid: PBFA97CFB590B2093
model:
datasource:
type: prometheus
uid: PBFA97CFB590B2093
editorMode: code
expr: probe_success{job="blackbox"}
instant: true
intervalMs: 1000
legendFormat: __auto
maxDataPoints: 43200
range: false
refId: probe_success
- refId: condition
relativeTimeRange:
from: 600
to: 0
datasourceUid: __expr__
model:
conditions:
- evaluator:
params:
- 0
- 0
type: eq
operator:
type: and
query:
params: []
reducer:
params: []
type: avg
type: query
datasource:
name: Expression
type: __expr__
uid: __expr__
expression: ${probe_success} == 0
intervalMs: 1000
maxDataPoints: 43200
refId: condition
type: math
noDataState: Alerting
execErrState: Alerting
for: 5m
annotations:
summary: Endpoint {{ $labels.instance }} is down
isPaused: false
@@ -0,0 +1,170 @@
# Monitoring Testnet
Instructions to setup and run monitoring stack for testnet services
## Create a deployment
Create a spec file for the deployment, which will map the stack's ports and volumes to the host:
```bash
laconic-so --stack monitoring deploy init --output monitoring-testnet-spec.yml
```
### Ports
Edit `network` in spec file to map container ports to same ports in host:
```
...
network:
ports:
prometheus:
- '9090:9090'
grafana:
- '3000:3000'
...
```
---
Once you've made any needed changes to the spec file, create a deployment from it:
```bash
laconic-so --stack monitoring deploy create --spec-file monitoring-testnet-spec.yml --deployment-dir monitoring-testnet-deployment
```
## Configure
### Prometheus scrape config
- Setup the following scrape configs in prometheus config file (`monitoring-testnet-deployment/config/monitoring/prometheus/prometheus.yml`) in the deployment folder:
```yml
...
- job_name: 'blackbox'
...
static_configs:
- targets: ["https://wallet.laconic.com"]
labels:
alias: "Wallet App"
url: "https://wallet.laconic.com"
- targets: ["https://laconicd-sapo.laconic.com"]
labels:
alias: "Node laconicd"
url: "https://laconicd-sapo.laconic.com"
- targets: ["https://console-sapo.laconic.com"]
labels:
alias: "Console App"
url: "https://console-sapo.laconic.com"
- targets: ["https://fixturenet-eth.laconic.com"]
labels:
alias: "Fixturenet ETH"
url: "https://fixturenet-eth.laconic.com"
- targets: ["https://deploy.laconic.com"]
labels:
alias: "Deploy App"
url: "https://deploy.laconic.com"
- targets: ["https://deploy-backend.laconic.com/staging/version"]
labels:
alias: "Deploy Backend"
url: "https://deploy-backend.laconic.com/staging/version"
- targets: ["https://container-registry.apps.vaasl.io"]
labels:
alias: "Container Registry"
url: "https://container-registry.apps.vaasl.io"
- targets: ["https://webapp-deployer-api.apps.vaasl.io"]
labels:
alias: "Webapp Deployer API"
url: "https://webapp-deployer-api.apps.vaasl.io"
- targets: ["https://webapp-deployer-ui.apps.vaasl.io"]
labels:
alias: "Webapp Deployer UI"
url: "https://webapp-deployer-ui.apps.vaasl.io"
...
- job_name: laconicd
...
static_configs:
- targets: ['LACONICD_REST_HOST:LACONICD_REST_PORT']
# Example: 'host.docker.internal:3317'
```
- Remove docker compose services which are not required in `monitoring-testnet-deployment/compose/docker-compose-prom-server.yml`
- `ethereum-chain-head-exporter`
- `filecoin-chain-head-exporter`
- `graph-node-upstream-head-exporter`
- `postgres-exporter`
### Grafana dashboards
Remove some of the existing dashboards which are not required in monitoring testnet
```
cd monitoring-testnet-deployment/config/monitoring/grafana/dashboards
rm postgres-dashboard.json subgraphs-dashboard.json watcher-dashboard.json
cd -
```
<!-- TODO: Check node-exporter-full.json, nodejs-app-dashboard.json -->
### Grafana alerts config
Place the pre-configured alerts rules in Grafana provisioning directory:
```bash
# watcher alert rules
cp monitoring-testnet-deployment/config/monitoring/testnet-alert-rules.yml monitoring-testnet-deployment/config/monitoring/grafana/provisioning/alerting/
```
Update the alerting contact points config (`monitoring-testnet-deployment/config/monitoring/grafana/provisioning/alerting/contactpoints.yml`) with desired contact points
Add corresponding routes to the notification policies config (`monitoring-testnet-deployment/config/monitoring/grafana/provisioning/alerting/policies.yml`) with appropriate object-matchers:
```yml
...
routes:
- receiver: SlackNotifier
object_matchers:
# Add matchers below
- ['grafana_folder', '=~', 'TestnetAlerts']
```
### Env
Set the following env variables in the deployment env config file (`monitoring-testnet-deployment/config.env`):
```bash
# Grafana server host URL to be used
# (Optional, default: http://localhost:3000)
GF_SERVER_ROOT_URL=
```
## Start the stack
Start the deployment:
```bash
laconic-so deployment --dir monitoring-testnet-deployment start
```
* List and check the health status of all the containers using `docker ps` and wait for them to be `healthy`
* Grafana should now be visible at http://localhost:3000 with configured dashboards
## Clean up
To stop monitoring services running in the background, while preserving data:
```bash
# Only stop the docker containers
laconic-so deployment --dir monitoring-watchers-deployment stop
# Run 'start' to restart the deployment
```
To stop monitoring services and also delete data:
```bash
# Stop the docker containers
laconic-so deployment --dir monitoring-watchers-deployment stop --delete-volumes
# Remove deployment directory (deployment will have to be recreated for a re-run)
rm -rf monitoring-watchers-deployment
```
@@ -44,9 +44,12 @@ Add the following scrape configs to prometheus config file (`monitoring-watchers
- job_name: 'blackbox'
...
static_configs:
- targets:
- <AZIMUTH_GATEWAY_GQL_ENDPOINT>
- <LACONICD_GQL_ENDPOINT>
- targets: ["<AZIMUTH_GATEWAY_GQL_ENDPOINT>"]
labels:
alias: "Azimuth Watcher"
- targets: ["<LACONICD_GQL_ENDPOINT>"]
labels:
alias: "Node (laconicd)"
...
- job_name: laconicd
static_configs:
@@ -54,6 +54,7 @@ def process_app_deployment_request(
deployment_record_namespace,
dns_record_namespace,
default_dns_suffix,
dns_value,
deployment_parent_dir,
kube_config,
image_registry,
@@ -251,6 +252,7 @@ def process_app_deployment_request(
dns_record,
dns_lrn,
deployment_dir,
dns_value,
app_deployment_request,
webapp_deployer_record,
logger,
@@ -304,6 +306,7 @@ def dump_known_requests(filename, requests, status="SEEN"):
help="How to handle requests with an FQDN: prohibit, allow, preexisting",
default="prohibit",
)
@click.option("--ip", help="IP address of the k8s deployment (to be set in DNS record)", default=None)
@click.option("--record-namespace-dns", help="eg, lrn://laconic/dns", required=True)
@click.option(
"--record-namespace-deployments",
@@ -381,6 +384,7 @@ def command( # noqa: C901
only_update_state,
dns_suffix,
fqdn_policy,
ip,
record_namespace_dns,
record_namespace_deployments,
dry_run,
@@ -429,6 +433,13 @@ def command( # noqa: C901
)
sys.exit(2)
if fqdn_policy == "allow" and not ip:
print(
"--ip is required with 'allow' fqdn-policy",
file=sys.stderr,
)
sys.exit(2)
tempdir = tempfile.mkdtemp()
gpg = gnupg.GPG(gnupghome=tempdir)
@@ -665,6 +676,7 @@ def command( # noqa: C901
record_namespace_deployments,
record_namespace_dns,
dns_suffix,
ip,
os.path.abspath(deployment_parent_dir),
kube_config,
image_registry,
+3
View File
@@ -689,6 +689,7 @@ def publish_deployment(
dns_record,
dns_lrn,
deployment_dir,
dns_value=None,
app_deployment_request=None,
webapp_deployer_record=None,
logger=None,
@@ -721,6 +722,8 @@ def publish_deployment(
}
if app_deployment_request:
new_dns_record["record"]["request"] = app_deployment_request.id
if dns_value:
new_dns_record["record"]["value"] = dns_value
if logger:
logger.log("Publishing DnsRecord.")