Compare commits

..
Author SHA1 Message Date
nabarun 320517b6b0 Add URL to dashboard panel label 2025-04-08 12:21:50 +05:30
nabarun 159402921d Use alias for blackbox targets in dashboard 2025-04-07 14:23:14 +05:30
nabarun fb0138e975 Add alerts for testnet services 2025-04-04 18:06:53 +05:30
nabarun 27412519b4 Add readme for monitoring testnet services 2025-04-02 18:54:06 +05:30
prathamesh 873a6d472c Update webapp deployment flow for supporting custom domains (#963)
Part of https://www.notion.so/Support-custom-domains-in-deploy-laconic-com-18aa6b22d4728067a44ae27090c02ce5 and cerc-io/snowballtools-base#47

- Set `value` (IP address for `A` resource) in the DNS records
- Update `deploy-webapp-from-registry` command with an option to pass k8s cluster IP address (only required with fqdn policy `allow`)

Reviewed-on: cerc-io/stack-orchestrator#963
Reviewed-by: ashwin <ashwin@noreply.git.vdb.to>
Co-authored-by: Prathamesh Musale <prathamesh.musale0@gmail.com>
Co-committed-by: Prathamesh Musale <prathamesh.musale0@gmail.com>
2025-02-04 13:26:31 +00:00
prathamesh 39df4683ac Allow payment reuse for same app LRN (#961)
Part of [Service provider auctions for web deployments](https://www.notion.so/Service-provider-auctions-for-web-deployments-104a6b22d47280dbad51d28aa3a91d75)

Reviewed-on: cerc-io/stack-orchestrator#961
Reviewed-by: ashwin <ashwin@noreply.git.vdb.to>
Co-authored-by: Prathamesh Musale <prathamesh.musale0@gmail.com>
Co-committed-by: Prathamesh Musale <prathamesh.musale0@gmail.com>
2024-10-29 11:30:03 +00:00
prathamesh 23ca4c4341 Allow payment reuse for application redeployment (#960)
Part of [Service provider auctions for web deployments](https://www.notion.so/Service-provider-auctions-for-web-deployments-104a6b22d47280dbad51d28aa3a91d75)

Reviewed-on: cerc-io/stack-orchestrator#960
Reviewed-by: ashwin <ashwin@noreply.git.vdb.to>
Co-authored-by: Prathamesh Musale <prathamesh.musale0@gmail.com>
Co-committed-by: Prathamesh Musale <prathamesh.musale0@gmail.com>
2024-10-29 06:51:48 +00:00
prathamesh f64ef5d128 Use file existence for registry mutex (#959)
Part of [Service provider auctions for web deployments](https://www.notion.so/Service-provider-auctions-for-web-deployments-104a6b22d47280dbad51d28aa3a91d75)

Reviewed-on: cerc-io/stack-orchestrator#959
Reviewed-by: ashwin <ashwin@noreply.git.vdb.to>
Co-authored-by: Prathamesh Musale <prathamesh.musale0@gmail.com>
Co-committed-by: Prathamesh Musale <prathamesh.musale0@gmail.com>
2024-10-29 04:05:35 +00:00
10 changed files with 379 additions and 32 deletions
@@ -1,7 +1,8 @@
modules:
http_2xx:
prober: http
timeout: 5s
timeout: 15s
http:
valid_status_codes: [] #default to 2xx
method: GET
preferred_ip_protocol: ip4
@@ -133,10 +133,13 @@
"type": "prometheus",
"uid": "PBFA97CFB590B2093"
},
"expr": "probe_success{instance=~\"$target\"}",
"format": "time_series",
"instant": true,
"refId": "A"
}
],
"title": "$target status",
"title": "$target ($url)",
"type": "row"
},
{
@@ -1057,6 +1060,29 @@
"tagsQuery": "",
"type": "query",
"useTags": false
},
{
"current": {
"selected": false,
"text": "",
"value": ""
},
"datasource": {
"type": "prometheus",
"uid": "PBFA97CFB590B2093"
},
"definition": "label_values(probe_success{instance=~\"$target\"}, url)",
"hide": 2,
"includeAll": false,
"multi": false,
"name": "url",
"options": [],
"query": "label_values(probe_success{instance=~\"$target\"}, url)",
"refresh": 2,
"regex": "",
"skipUrlSync": false,
"sort": 0,
"type": "query"
}
]
},
@@ -8,6 +8,7 @@ policies:
group_by:
- grafana_folder
- alertname
- instance
routes:
- receiver: SlackNotifier
object_matchers:
@@ -25,20 +25,34 @@ scrape_configs:
module: [http_2xx]
static_configs:
# Add URLs to be monitored below
- targets:
# - https://github.com
# - targets: ["https://github.com"]
# labels:
# alias: "GitHub"
# url: "https://github.com"
relabel_configs:
# Forward the original target URL as the 'target' parameter.
- source_labels: [__address__]
regex: (.*)(:80)?
target_label: __param_target
- source_labels: [__param_target]
regex: (.*)
# Use the custom alias if defined for the 'instance' label.
- source_labels: [alias]
target_label: instance
replacement: ${1}
- source_labels: []
regex: .*
target_label: __address__
action: replace
# Preserve the URL label
- source_labels: [url]
target_label: url
action: replace
# If no alias is set, fall back to the target URL.
- source_labels: [instance]
regex: ^$
target_label: instance
replacement: ${__param_target}
# Finally, tell Prometheus to scrape the blackbox_exporter.
- target_label: __address__
replacement: blackbox:9115
# Drop the original alias label as it's now redundant with instance
- action: labeldrop
regex: ^alias$
- job_name: chain_heads
scrape_interval: 10s
@@ -0,0 +1,64 @@
apiVersion: 1
groups:
- orgId: 1
name: testnet
folder: TestnetAlerts
interval: 30s
rules:
- uid: endpoint_down
title: endpoint_down
condition: condition
data:
- refId: probe_success
relativeTimeRange:
from: 600
to: 0
datasourceUid: PBFA97CFB590B2093
model:
datasource:
type: prometheus
uid: PBFA97CFB590B2093
editorMode: code
expr: probe_success{job="blackbox"}
instant: true
intervalMs: 1000
legendFormat: __auto
maxDataPoints: 43200
range: false
refId: probe_success
- refId: condition
relativeTimeRange:
from: 600
to: 0
datasourceUid: __expr__
model:
conditions:
- evaluator:
params:
- 0
- 0
type: eq
operator:
type: and
query:
params: []
reducer:
params: []
type: avg
type: query
datasource:
name: Expression
type: __expr__
uid: __expr__
expression: ${probe_success} == 0
intervalMs: 1000
maxDataPoints: 43200
refId: condition
type: math
noDataState: Alerting
execErrState: Alerting
for: 5m
annotations:
summary: Endpoint {{ $labels.instance }} is down
isPaused: false
@@ -0,0 +1,170 @@
# Monitoring Testnet
Instructions to setup and run monitoring stack for testnet services
## Create a deployment
Create a spec file for the deployment, which will map the stack's ports and volumes to the host:
```bash
laconic-so --stack monitoring deploy init --output monitoring-testnet-spec.yml
```
### Ports
Edit `network` in spec file to map container ports to same ports in host:
```
...
network:
ports:
prometheus:
- '9090:9090'
grafana:
- '3000:3000'
...
```
---
Once you've made any needed changes to the spec file, create a deployment from it:
```bash
laconic-so --stack monitoring deploy create --spec-file monitoring-testnet-spec.yml --deployment-dir monitoring-testnet-deployment
```
## Configure
### Prometheus scrape config
- Setup the following scrape configs in prometheus config file (`monitoring-testnet-deployment/config/monitoring/prometheus/prometheus.yml`) in the deployment folder:
```yml
...
- job_name: 'blackbox'
...
static_configs:
- targets: ["https://wallet.laconic.com"]
labels:
alias: "Wallet App"
url: "https://wallet.laconic.com"
- targets: ["https://laconicd-sapo.laconic.com"]
labels:
alias: "Node laconicd"
url: "https://laconicd-sapo.laconic.com"
- targets: ["https://console-sapo.laconic.com"]
labels:
alias: "Console App"
url: "https://console-sapo.laconic.com"
- targets: ["https://fixturenet-eth.laconic.com"]
labels:
alias: "Fixturenet ETH"
url: "https://fixturenet-eth.laconic.com"
- targets: ["https://deploy.laconic.com"]
labels:
alias: "Deploy App"
url: "https://deploy.laconic.com"
- targets: ["https://deploy-backend.laconic.com/staging/version"]
labels:
alias: "Deploy Backend"
url: "https://deploy-backend.laconic.com/staging/version"
- targets: ["https://container-registry.apps.vaasl.io"]
labels:
alias: "Container Registry"
url: "https://container-registry.apps.vaasl.io"
- targets: ["https://webapp-deployer-api.apps.vaasl.io"]
labels:
alias: "Webapp Deployer API"
url: "https://webapp-deployer-api.apps.vaasl.io"
- targets: ["https://webapp-deployer-ui.apps.vaasl.io"]
labels:
alias: "Webapp Deployer UI"
url: "https://webapp-deployer-ui.apps.vaasl.io"
...
- job_name: laconicd
...
static_configs:
- targets: ['LACONICD_REST_HOST:LACONICD_REST_PORT']
# Example: 'host.docker.internal:3317'
```
- Remove docker compose services which are not required in `monitoring-testnet-deployment/compose/docker-compose-prom-server.yml`
- `ethereum-chain-head-exporter`
- `filecoin-chain-head-exporter`
- `graph-node-upstream-head-exporter`
- `postgres-exporter`
### Grafana dashboards
Remove some of the existing dashboards which are not required in monitoring testnet
```
cd monitoring-testnet-deployment/config/monitoring/grafana/dashboards
rm postgres-dashboard.json subgraphs-dashboard.json watcher-dashboard.json
cd -
```
<!-- TODO: Check node-exporter-full.json, nodejs-app-dashboard.json -->
### Grafana alerts config
Place the pre-configured alerts rules in Grafana provisioning directory:
```bash
# watcher alert rules
cp monitoring-testnet-deployment/config/monitoring/testnet-alert-rules.yml monitoring-testnet-deployment/config/monitoring/grafana/provisioning/alerting/
```
Update the alerting contact points config (`monitoring-testnet-deployment/config/monitoring/grafana/provisioning/alerting/contactpoints.yml`) with desired contact points
Add corresponding routes to the notification policies config (`monitoring-testnet-deployment/config/monitoring/grafana/provisioning/alerting/policies.yml`) with appropriate object-matchers:
```yml
...
routes:
- receiver: SlackNotifier
object_matchers:
# Add matchers below
- ['grafana_folder', '=~', 'TestnetAlerts']
```
### Env
Set the following env variables in the deployment env config file (`monitoring-testnet-deployment/config.env`):
```bash
# Grafana server host URL to be used
# (Optional, default: http://localhost:3000)
GF_SERVER_ROOT_URL=
```
## Start the stack
Start the deployment:
```bash
laconic-so deployment --dir monitoring-testnet-deployment start
```
* List and check the health status of all the containers using `docker ps` and wait for them to be `healthy`
* Grafana should now be visible at http://localhost:3000 with configured dashboards
## Clean up
To stop monitoring services running in the background, while preserving data:
```bash
# Only stop the docker containers
laconic-so deployment --dir monitoring-watchers-deployment stop
# Run 'start' to restart the deployment
```
To stop monitoring services and also delete data:
```bash
# Stop the docker containers
laconic-so deployment --dir monitoring-watchers-deployment stop --delete-volumes
# Remove deployment directory (deployment will have to be recreated for a re-run)
rm -rf monitoring-watchers-deployment
```
@@ -44,9 +44,12 @@ Add the following scrape configs to prometheus config file (`monitoring-watchers
- job_name: 'blackbox'
...
static_configs:
- targets:
- <AZIMUTH_GATEWAY_GQL_ENDPOINT>
- <LACONICD_GQL_ENDPOINT>
- targets: ["<AZIMUTH_GATEWAY_GQL_ENDPOINT>"]
labels:
alias: "Azimuth Watcher"
- targets: ["<LACONICD_GQL_ENDPOINT>"]
labels:
alias: "Node (laconicd)"
...
- job_name: laconicd
static_configs:
@@ -54,6 +54,7 @@ def process_app_deployment_request(
deployment_record_namespace,
dns_record_namespace,
default_dns_suffix,
dns_value,
deployment_parent_dir,
kube_config,
image_registry,
@@ -251,6 +252,7 @@ def process_app_deployment_request(
dns_record,
dns_lrn,
deployment_dir,
dns_value,
app_deployment_request,
webapp_deployer_record,
logger,
@@ -304,6 +306,7 @@ def dump_known_requests(filename, requests, status="SEEN"):
help="How to handle requests with an FQDN: prohibit, allow, preexisting",
default="prohibit",
)
@click.option("--ip", help="IP address of the k8s deployment (to be set in DNS record)", default=None)
@click.option("--record-namespace-dns", help="eg, lrn://laconic/dns", required=True)
@click.option(
"--record-namespace-deployments",
@@ -381,6 +384,7 @@ def command( # noqa: C901
only_update_state,
dns_suffix,
fqdn_policy,
ip,
record_namespace_dns,
record_namespace_deployments,
dry_run,
@@ -429,6 +433,13 @@ def command( # noqa: C901
)
sys.exit(2)
if fqdn_policy == "allow" and not ip:
print(
"--ip is required with 'allow' fqdn-policy",
file=sys.stderr,
)
sys.exit(2)
tempdir = tempfile.mkdtemp()
gpg = gnupg.GPG(gnupghome=tempdir)
@@ -665,6 +676,7 @@ def command( # noqa: C901
record_namespace_deployments,
record_namespace_dns,
dns_suffix,
ip,
os.path.abspath(deployment_parent_dir),
kube_config,
image_registry,
@@ -1,8 +1,59 @@
import fcntl
from functools import wraps
import os
import time
# Define default file path for the lock
DEFAULT_LOCK_FILE_PATH = "/tmp/registry_mutex_lock_file"
LOCK_TIMEOUT = 30
LOCK_RETRY_INTERVAL = 3
def acquire_lock(client, lock_file_path, timeout):
# Lock alreay acquired by the current client
if client.mutex_lock_acquired:
return
while True:
try:
# Check if lock file exists and is potentially stale
if os.path.exists(lock_file_path):
with open(lock_file_path, 'r') as lock_file:
timestamp = float(lock_file.read().strip())
# If lock is stale, remove the lock file
if time.time() - timestamp > timeout:
print(f"Stale lock detected, removing lock file {lock_file_path}")
os.remove(lock_file_path)
else:
print(f"Lock file {lock_file_path} exists and is recent, waiting...")
time.sleep(LOCK_RETRY_INTERVAL)
continue
# Try to create a new lock file with the current timestamp
fd = os.open(lock_file_path, os.O_CREAT | os.O_EXCL | os.O_RDWR)
with os.fdopen(fd, 'w') as lock_file:
lock_file.write(str(time.time()))
client.mutex_lock_acquired = True
print(f"Registry lock acquired, {lock_file_path}")
# Lock successfully acquired
return
except FileExistsError:
print(f"Lock file {lock_file_path} exists, waiting...")
time.sleep(LOCK_RETRY_INTERVAL)
def release_lock(client, lock_file_path):
try:
os.remove(lock_file_path)
client.mutex_lock_acquired = False
print(f"Registry lock released, {lock_file_path}")
except FileNotFoundError:
# Lock file already removed
pass
def registry_mutex():
@@ -13,18 +64,13 @@ def registry_mutex():
if self.mutex_lock_file:
lock_file_path = self.mutex_lock_file
with open(lock_file_path, 'w') as lock_file:
try:
# Try to acquire the lock
fcntl.flock(lock_file, fcntl.LOCK_EX)
# Call the actual function
result = func(self, *args, **kwargs)
finally:
# Always release the lock
fcntl.flock(lock_file, fcntl.LOCK_UN)
return result
# Acquire the lock before running the function
acquire_lock(self, lock_file_path, LOCK_TIMEOUT)
try:
return func(self, *args, **kwargs)
finally:
# Release the lock after the function completes
release_lock(self, lock_file_path)
return wrapper
+16 -6
View File
@@ -117,7 +117,6 @@ class LaconicRegistryClient:
def __init__(self, config_file, log_file=None, mutex_lock_file=None):
self.config_file = config_file
self.log_file = log_file
self.mutex_lock_file = mutex_lock_file
self.cache = AttrDict(
{
"name_or_id": {},
@@ -126,6 +125,9 @@ class LaconicRegistryClient:
}
)
self.mutex_lock_file = mutex_lock_file
self.mutex_lock_acquired = False
def whoami(self, refresh=False):
if not refresh and "whoami" in self.cache:
return self.cache["whoami"]
@@ -687,6 +689,7 @@ def publish_deployment(
dns_record,
dns_lrn,
deployment_dir,
dns_value=None,
app_deployment_request=None,
webapp_deployer_record=None,
logger=None,
@@ -719,6 +722,8 @@ def publish_deployment(
}
if app_deployment_request:
new_dns_record["record"]["request"] = app_deployment_request.id
if dns_value:
new_dns_record["record"]["value"] = dns_value
if logger:
logger.log("Publishing DnsRecord.")
@@ -844,16 +849,21 @@ def confirm_payment(laconic: LaconicRegistryClient, record, payment_address, min
)
return False
# Check if the payment was already used on a
# Check if the payment was already used on a deployment
used = laconic.app_deployments(
{"deployer": payment_address, "payment": tx.hash}, all=True
{"deployer": record.attributes.deployer, "payment": tx.hash}, all=True
)
if len(used):
logger.log(f"{record.id}: payment {tx.hash} already used on deployment {used}")
return False
# Fetch the app name from request record
used_request = laconic.get_record(used[0].attributes.request, require=True)
# Check that payment was used for deployment of same application
if record.attributes.application != used_request.attributes.application:
logger.log(f"{record.id}: payment {tx.hash} already used on a different application deployment {used}")
return False
used = laconic.app_deployment_removals(
{"deployer": payment_address, "payment": tx.hash}, all=True
{"deployer": record.attributes.deployer, "payment": tx.hash}, all=True
)
if len(used):
logger.log(