Prometheus Monitoring Skill
Query, analyze, and manage Prometheus metrics using the HTTP API and PromQL.
MANDATORY: Discovery-First Pattern
Always discover available metrics and labels before writing PromQL. Never guess metric names.
Phase 1: Discovery
#!/bin/bash
prom_api() {
curl -s "${PROMETHEUS_URL}/api/v1/$1"
}
echo "=== Prometheus Build Info ==="
prom_api "status/buildinfo" | jq '{version: .data.version, goVersion: .data.goVersion}'
echo ""
echo "=== Active Targets ==="
prom_api "targets" | jq -r '.data.activeTargets[] | "\(.health)\t\(.scrapePool)\t\(.labels.instance // .scrapeUrl)"' \
| sort | head -30
echo ""
echo "=== Failed Targets ==="
prom_api "targets" | jq -r '.data.activeTargets[] | select(.health != "up") | "\(.health)\t\(.scrapePool)\t\(.lastError)"' | head -10
echo ""
echo "=== Available Metric Namespaces (sampling) ==="
prom_api "label/__name__/values" | jq -r '.data[]' | cut -d_ -f1 | sort -u | head -30
echo ""
echo "=== Alerting Rules ==="
prom_api "rules" | jq -r '.data.groups[].rules[] | select(.type=="alerting") | "\(.state)\t\(.name)\t\(.labels.severity // "none")"' | sort | head -20
Phase 2: PromQL Query
Only reference metrics confirmed in Phase 1 discovery.
PromQL Patterns
Helper Function
#!/bin/bash
prom_api() {
local endpoint="$1"
curl -s "${PROMETHEUS_URL}/api/v1/${endpoint}"
}
# Instant query
prom_query() {
local query="$1"
local time="${2:-}" # Optional: unix timestamp
local params="query=$(python3 -c "import urllib.parse; print(urllib.parse.quote('${query}'))")"
[ -n "$time" ] && params="${params}&time=${time}"
prom_api "query?${params}"
}
# Range query
prom_range() {
local query="$1"
local start="$2"
local end="$3"
local step="${4:-60}" # seconds
local params="query=$(python3 -c "import urllib.parse; print(urllib.parse.quote('${query}'))")"
prom_api "query_range?${params}&start=${start}&end=${end}&step=${step}"
}
# Get label values
prom_labels() {
local metric="$1"
local label="$2"
prom_api "label/${label}/values?match[]=${metric}"
}
Common PromQL Patterns
Infrastructure Metrics
#!/bin/bash
NOW=$(date +%s) - 3600))
echo "=== CPU Usage by Instance ==="
prom_query 'sort_desc(100 - (avg by (instance) (irate(node_cpu_seconds_total{mode="idle"}[5m])) * 100))' \
| jq -r '.data.result[] | "\(.metric.instance)\t\(.value[1] | tonumber | . * 10 | round / 10)%"' \
| column -t | head -15
echo ""
echo "=== Memory Usage by Instance ==="
prom_query 'sort_desc(100 - (node_memory_MemAvailable_bytes / node_memory_MemTotal_bytes * 100))' \
| jq -r '.data.result[] | "\(.metric.instance)\t\(.value[1] | tonumber | . * 10 | round / 10)%"' \
| column -t | head -15
echo ""
echo "=== Disk Usage (>70%) ==="
prom_query '(node_filesystem_avail_bytes{fstype!~"tmpfs|squashfs"} / node_filesystem_size_bytes{fstype!~"tmpfs|squashfs"} * 100) < 30' \
| jq -r '.data.result[] | "\(.metric.instance)\t\(.metric.mountpoint)\t\(100 - (.value[1] | tonumber) | . * 10 | round / 10)% used"' \
| column -t | head -15
echo ""
echo "=== Network Traffic (top 10) ==="
prom_query 'sort_desc(sum by (instance) (irate(node_network_receive_bytes_total[5m])) + sum by (instance) (irate(node_network_transmit_bytes_total[5m])))' \
| jq -r '.data.result[] | "\(.metric.instance)\t\(.value[1] | tonumber / 1024 / 1024 | . * 10 | round / 10) MB/s"' \
| column -t | head -10
Application / HTTP Metrics
#!/bin/bash
echo "=== HTTP Request Rate by Service ==="
prom_query 'sort_desc(sum by (job) (rate(http_requests_total[5m])))' \
| jq -r '.data.result[] | "\(.metric.job)\t\(.value[1] | tonumber | . * 100 | round / 100) req/s"' \
| column -t | head -15
echo ""
echo "=== HTTP Error Rate (5xx) ==="
prom_query 'sort_desc(sum by (job) (rate(http_requests_total{status=~"5.."}[5m])) / sum by (job) (rate(http_requests_total[5m])) * 100)' \
| jq -r '.data.result[] | "\(.metric.job)\t\(.value[1] | tonumber | . * 100 | round / 100)% errors"' \
| column -t | head -10
echo ""
echo "=== P99 Latency by Service ==="
prom_query 'histogram_quantile(0.99, sum by (job, le) (rate(http_request_duration_seconds_bucket[5m])))' \
| jq -r '.data.result[] | "\(.metric.job)\t\(.value[1] | tonumber * 1000 | . * 10 | round / 10)ms"' \
| column -t | head -10
echo ""
echo "=== Apdex Score (target 300ms) ==="
prom_query 'sum by (job) (rate(http_request_duration_seconds_bucket{le="0.3"}[5m])) / sum by (job) (rate(http_request_duration_seconds_count[5m]))' \
| jq -r '.data.result[] | "\(.metric.job)\t\(.value[1] | tonumber | . * 1000 | round / 1000)"' \
| column -t | head -10
Alerting Rules Analysis
#!/bin/bash
echo "=== All Alerting Rules ==="
prom_api "rules" | jq -r '
.data.groups[] as $group |
$group.rules[] |
select(.type == "alerting") |
"\($group.name)\t\(.name)\t\(.state)\t\(.labels.severity // "none")"
' | column -t | sort -k3
echo ""
echo "=== Firing Alerts ==="
prom_api "alerts" | jq -r '
.data.alerts[] |
select(.state == "firing") |
"\(.labels.severity // "none")\t\(.labels.alertname)\t\(.labels.instance // "")\t\(.activeAt[0:16])"
' | sort | column -t
echo ""
echo "=== Pending Alerts (about to fire) ==="
prom_api "alerts" | jq -r '
.data.alerts[] |
select(.state == "pending") |
"\(.labels.alertname)\t\(.labels.instance // "")\t\(.activeAt[0:16])"
' | column -t | head -10
TSDB Storage Analysis
#!/bin/bash
echo "=== TSDB Head Stats ==="
prom_api "tsdb/head_stats" | jq '.data | {
numSeries: .numSeries,
numLabelPairs: .numLabelPairs,
chunkCount: .chunkCount,
minTime_epoch: .minTime,
maxTime_epoch: .maxTime
}'
echo ""
echo "=== TSDB Block Stats ==="
prom_api "tsdb/blocks" | jq '.data | length | "Total blocks: \(.)"'
echo ""
echo "=== Cardinality Analysis (top label values) ==="
# High cardinality labels cause memory issues
prom_query 'topk(10, count by (__name__) ({__name__=~".+"}))' \
| jq -r '.data.result[] | "\(.metric.__name__)\t\(.value[1])"' \
| sort -t$'\t' -k2 -rn | head -10
Scrape Target Health
#!/bin/bash
echo "=== Scrape Target Summary ==="
prom_api "targets" | jq '
.data.activeTargets |
{
total: length,
up: [.[] | select(.health == "up")] | length,
down: [.[] | select(.health != "up")] | length
}'
echo ""
echo "=== Down Targets ==="
prom_api "targets" | jq -r '
.data.activeTargets[] |
select(.health != "up") |
"\(.scrapePool)\t\(.labels.instance // .scrapeUrl)\t\(.lastError // "unknown")"
' | column -t
echo ""
echo "=== Scrape Duration (slowest targets) ==="
prom_query 'sort_desc(scrape_duration_seconds)' \
| jq -r '.data.result[] | "\(.metric.instance)\t\(.value[1] | tonumber | . * 1000 | round)ms"' \
| column -t | head -10
Recording Rules Check
#!/bin/bash
echo "=== Recording Rules ==="
prom_api "rules" | jq -r '
.data.groups[] as $group |
$group.rules[] |
select(.type == "recording") |
"\($group.name)\t\(.name)\t\(.health)"
' | column -t | head -20
echo ""
echo "=== Rule Evaluation Errors ==="
prom_api "rules" | jq -r '
.data.groups[].rules[] |
select(.lastError != null and .lastError != "") |
"\(.name): \(.lastError)"
' | head -10
Output Format
Present results as a structured report:
Monitoring Prometheus Report
════════════════════════════
Resources discovered: [count]
Resource Status Key Metric Issues
──────────────────────────────────────────────
[name] [ok/warn] [value] [findings]
Summary: [total] resources | [ok] healthy | [warn] warnings | [crit] critical
Action Items: [list of prioritized findings]
Target ≤50 lines of output. Use tables for multi-resource comparisons.
Anti-Hallucination Rules
- NEVER assume resource names — always discover via CLI/API in Phase 1 before referencing in Phase 2.
- NEVER fabricate metric names or dimensions — verify against the service documentation or
--helpoutput. - NEVER mix CLI commands between service versions — confirm which version/API you are targeting.
- ALWAYS use the discovery → verify → analyze chain — every resource referenced must have been discovered first.
- ALWAYS handle empty results gracefully — an empty response is valid data, not an error to retry.
Counter-Rationalizations
| Shortcut | Counter | Why |
|---|---|---|
| "I'll skip discovery and check known resources" | Always run Phase 1 discovery first | Resource names change, new resources appear — assumed names cause errors |
| "The user only asked for a quick check" | Follow the full discovery → analysis flow | Quick checks miss critical issues; structured analysis catches silent failures |
| "Default configuration is probably fine" | Audit configuration explicitly | Defaults often leave logging, security, and optimization features disabled |
| "Metrics aren't needed for this" | Always check relevant metrics when available | API/CLI responses show current state; metrics reveal trends and intermittent issues |
| "I don't have access to that" | Try the command and report the actual error | Assumed permission failures prevent useful investigation; actual errors are informative |
Common Pitfalls
- Metric name guessing: Use
label/__name__/valuesto list actual metric names — never guess namespaces - Rate vs irate:
rate()smooths over the window;irate()uses last 2 samples — preferrate()for dashboards,irate()for alerting - Range vector requirement:
rate(),irate(),increase()require range vectors ([5m]) — instant vectors cause parse errors [5m]window too small: If scrape interval is 60s,[5m]gives only 5 samples — use at least 4x scrape interval- Histogram quantiles:
histogram_quantile()needs_bucketmetric withlelabel — verify bucket metric exists first - Counter resets:
rate()handles counter resets automatically;delta()does not — userate()for counters - High cardinality queries:
count by ()across all metrics can be slow — limit with label matchers - Step alignment: Range query
stepshould be ≥ scrape interval to avoid gaps - Alertmanager vs Prometheus alerts: Alerts fire in Prometheus, route via Alertmanager — check both for full picture