Skip to content
Closed
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
8 changes: 6 additions & 2 deletions charmcraft.yaml
Original file line number Diff line number Diff line change
Expand Up @@ -11,8 +11,8 @@ parts:
build-packages: [git]

platforms:
ubuntu@20.04:amd64:
ubuntu@22.04:amd64:
# ubuntu@20.04:amd64:
# ubuntu@22.04:amd64:
ubuntu@24.04:amd64:

Copy link
Copy Markdown
Contributor

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

Did you comment these for quick testing?

Copy link
Copy Markdown
Member Author

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

Yes, I'll remove it after another round of testing

ubuntu@20.04:arm64:
ubuntu@22.04:arm64:
Expand Down Expand Up @@ -41,3 +41,7 @@ actions:
Use the re-detected list of hardware tools as the new enable-list to reconfigure
and restart the exporter.
default: false

charm-libs:
- lib: grafana-agent.cos_agent
version: "0"
2 changes: 1 addition & 1 deletion config.yaml
Original file line number Diff line number Diff line change
Expand Up @@ -51,7 +51,7 @@ options:
description: |
Timeout for collectors' shell commands in seconds. Changing this will also change
the scrape_timeout config option for prometheus for the cos-agent relation with
grafana-agent.
opentelemetry or grafana-agent.
This value is also used for the redfish client's timeout parameter.
The value of this timeout should not be greater than prometheus scrape_interval (which
is 60 seconds by default), as it greater would cause the scrape_timeout to be
Expand Down
12 changes: 6 additions & 6 deletions dev-environment.md
Original file line number Diff line number Diff line change
Expand Up @@ -99,14 +99,14 @@ juju add-machine ssh:ubuntu@$BR0_ADDR
juju switch hw-obs
juju deploy ubuntu --to 0
juju deploy hardware-observer
juju deploy grafana-agent
juju deploy opentelemetry-collector --channel=2/stable
```

Add necessary relations:
```
juju relate hardware-observer ubuntu
juju relate grafana-agent ubuntu
juju relate grafana-agent hardware-observer
juju relate opentelemetry-collector ubuntu
juju relate opentelemetry-collector hardware-observer
```


Expand Down Expand Up @@ -181,9 +181,9 @@ juju deploy cos-lite --trust --overlay ./offers-overlay.yaml
Switch back to your physical model and add the relations to COS:
```
juju switch hw-obs
juju relate grafana-agent k8s-controller:cos.prometheus-receive-remote-write
juju relate grafana-agent k8s-controller:cos.grafana-dashboards
juju relate grafana-agent k8s-controller:cos.loki-logging
juju relate opentelemetry-collector k8s-controller:cos.prometheus-receive-remote-write
juju relate opentelemetry-collector k8s-controller:cos.grafana-dashboards
juju relate opentelemetry-collector k8s-controller:cos.loki-logging
```


Expand Down
81 changes: 62 additions & 19 deletions lib/charms/grafana_agent/v0/cos_agent.py
Original file line number Diff line number Diff line change
Expand Up @@ -211,7 +211,9 @@ def __init__(self, *args):
```
"""

import copy
import enum
import hashlib
import json
import logging
import socket
Expand Down Expand Up @@ -254,7 +256,7 @@ class _MetricsEndpointDict(TypedDict):

LIBID = "dc15fa84cef84ce58155fb84f6c6213a"
LIBAPI = 0
LIBPATCH = 22
LIBPATCH = 25

PYDEPS = ["cosl >= 0.0.50", "pydantic"]

Expand All @@ -264,12 +266,6 @@ class _MetricsEndpointDict(TypedDict):
logger = logging.getLogger(__name__)
SnapEndpoint = namedtuple("SnapEndpoint", "owner, name")

# Note: MutableMapping is imported from the typing module and not collections.abc
# because subscripting collections.abc.MutableMapping was added in python 3.9, but
# most of our charms are based on 20.04, which has python 3.8.

_RawDatabag = MutableMapping[str, str]


class TransportProtocolType(str, enum.Enum):
"""Receiver Type."""
Expand Down Expand Up @@ -305,6 +301,22 @@ class TransportProtocolType(str, enum.Enum):
ReceiverProtocol = Literal["otlp_grpc", "otlp_http", "zipkin", "jaeger_thrift_http", "jaeger_grpc"]


def _dedupe_list(items: List[Dict[str, Any]]) -> List[Dict[str, Any]]:
"""Deduplicate items in the list via object identity."""
unique_items = []
for item in items:
if item not in unique_items:
unique_items.append(item)
return unique_items


def _dict_hash_except_key(scrape_config: Dict[str, Any], key: Optional[str]):
"""Get a hash of the scrape_config dict, except for the specified key."""
cfg_for_hash = {k: v for k, v in scrape_config.items() if k != key}
serialized = json.dumps(cfg_for_hash, sort_keys=True)
return hashlib.blake2b(serialized.encode(), digest_size=4).hexdigest()


class TracingError(Exception):
"""Base class for custom errors raised by tracing."""

Expand Down Expand Up @@ -619,7 +631,8 @@ def __init__(
refresh_events: Optional[List] = None,
tracing_protocols: Optional[List[str]] = None,
*,
scrape_configs: Optional[Union[List[dict], Callable]] = None,
scrape_configs: Optional[Union[List[dict], Callable[[], List[Dict[str, Any]]]]] = None,
extra_alert_groups: Optional[Callable[[], Dict[str, Any]]] = None,
):
"""Create a COSAgentProvider instance.

Expand All @@ -640,6 +653,9 @@ def __init__(
scrape_configs: List of standard scrape_configs dicts or a callable
that returns the list in case the configs need to be generated dynamically.
The contents of this list will be merged with the contents of `metrics_endpoints`.
extra_alert_groups: A callable that returns a dict of alert rule groups in case the
alerts need to be generated dynamically. The contents of this dict will be merged
with generic and bundled alert rules.
"""
super().__init__(charm, relation_name)
dashboard_dirs = dashboard_dirs or ["./src/grafana_dashboards"]
Expand All @@ -648,6 +664,7 @@ def __init__(
self._relation_name = relation_name
self._metrics_endpoints = metrics_endpoints or []
self._scrape_configs = scrape_configs or []
self._extra_alert_groups = extra_alert_groups or {}
self._metrics_rules = metrics_rules_dir
self._logs_rules = logs_rules_dir
self._recursive = recurse_rules_dirs
Expand Down Expand Up @@ -689,12 +706,34 @@ def _on_refresh(self, event):
) as e:
logger.error("Invalid relation data provided: %s", e)

def _deterministic_scrape_configs(
self, scrape_configs: List[Dict[str, Any]]
) -> List[Dict[str, Any]]:
"""Get deterministic scrape_configs with stable job names.

For stability across serializations, compute a short per-config hash
and append it to the existing job name (or 'default'). Keep the app
name as a prefix: <app>_<job_or_default>_<8hex-hash>.

Hash the whole scrape_config (except any existing job_name) so the
suffix is sensitive to all stable fields. Use deterministic JSON
serialization.
"""
local_scrape_configs = copy.deepcopy(scrape_configs)
for scrape_config in local_scrape_configs:
name = scrape_config.get("job_name", "default")
short_id = _dict_hash_except_key(scrape_config, "job_name")
scrape_config["job_name"] = f"{self._charm.app.name}_{name}_{short_id}"

return sorted(local_scrape_configs, key=lambda c: c.get("job_name", ""))

@property
def _scrape_jobs(self) -> List[Dict]:
"""Return a prometheus_scrape-like data structure for jobs.
"""Return a list of scrape_configs.

https://prometheus.io/docs/prometheus/latest/configuration/configuration/#scrape_config
"""
# Optionally allow the charm to set the scrape_configs
if callable(self._scrape_configs):
scrape_configs = self._scrape_configs()
else:
Expand All @@ -712,26 +751,30 @@ def _scrape_jobs(self) -> List[Dict]:

scrape_configs = scrape_configs or []

# Augment job name to include the app name and a unique id (index)
for idx, scrape_config in enumerate(scrape_configs):
scrape_config["job_name"] = "_".join(
[self._charm.app.name, str(idx), scrape_config.get("job_name", "default")]
)

return scrape_configs
return self._deterministic_scrape_configs(scrape_configs)

@property
def _metrics_alert_rules(self) -> Dict:
"""Use (for now) the prometheus_scrape AlertRules to initialize this."""
"""Return a dict of alert rule groups."""
# Optionally allow the charm to add the metrics_alert_rules
if callable(self._extra_alert_groups):
rules = self._extra_alert_groups()
else:
rules = {"groups": []}

alert_rules = AlertRules(
query_type="promql", topology=JujuTopology.from_charm(self._charm)
)
alert_rules.add_path(self._metrics_rules, recursive=self._recursive)
alert_rules.add(
generic_alert_groups.application_rules,
copy.deepcopy(generic_alert_groups.application_rules),
group_name_prefix=JujuTopology.from_charm(self._charm).identifier,
)
return alert_rules.as_dict()

# NOTE: The charm could supply rules we implement in this method, so we deduplicate
rules["groups"] = _dedupe_list(rules["groups"] + alert_rules.as_dict()["groups"])

return rules

@property
def _log_alert_rules(self) -> Dict:
Expand Down
12 changes: 6 additions & 6 deletions tests/functional/bundle.yaml.j2
Original file line number Diff line number Diff line change
@@ -1,7 +1,7 @@
# Test basic deployment:
# ubuntu:juju-info <-> grafana-agent:juju-info
# ubuntu:juju-info <-> opentelemetry-collector:juju-info
# ubuntu:juju-info <-> hardware-observer:general-info
# grafana-agent:cos-agent <-> hardware-observer:cos-agent
# opentelemetry-collector:cos-agent <-> hardware-observer:cos-agent

default-base: {{ base }}

Expand All @@ -14,16 +14,16 @@ applications:
num_units: 1
to:
- "0"
grafana-agent:
charm: grafana-agent
channel: 1/stable
opentelemetry-collector:
charm: opentelemetry-collector
channel: 2/stable
hardware-observer:
charm: {{ charm }}
options:
redfish-disable: {{ redfish_disable }}

relations:
- - grafana-agent:juju-info
- - opentelemetry-collector:juju-info
- ubuntu:juju-info
- - hardware-observer:general-info
- ubuntu:juju-info
14 changes: 6 additions & 8 deletions tests/functional/test_charm.py
Original file line number Diff line number Diff line change
Expand Up @@ -34,7 +34,7 @@
METADATA = yaml.safe_load(Path("./metadata.yaml").read_text())
APP_NAME = METADATA["name"]
PRINCIPAL_APP_NAME = "ubuntu"
GRAFANA_AGENT_APP_NAME = "grafana-agent"
OTEL_APP_NAME = "opentelemetry-collector"

TIMEOUT = 600

Expand Down Expand Up @@ -85,7 +85,7 @@ async def test_build_and_deploy( # noqa: C901, function is too complex
timeout=TIMEOUT,
)
await ops_test.model.wait_for_idle(
apps=[GRAFANA_AGENT_APP_NAME],
apps=[OTEL_APP_NAME],
status="blocked",
timeout=TIMEOUT,
)
Expand All @@ -101,8 +101,8 @@ async def test_build_and_deploy( # noqa: C901, function is too complex
else:
assert unit.workload_status_message == AppStatus.MISSING_RELATION

for unit in ops_test.model.applications[GRAFANA_AGENT_APP_NAME].units:
messages = ["Missing", "grafana-cloud-config", "logging-consumer", "send-remote-write"]
for unit in ops_test.model.applications[OTEL_APP_NAME].units:
messages = ["cloud-config", "send-loki-logs", "send-remote-write"]
for msg in messages:
assert msg in unit.workload_status_message

Expand Down Expand Up @@ -144,16 +144,14 @@ async def test_required_resources(ops_test: OpsTest, required_resources):

@pytest.mark.abort_on_fail
async def test_cos_agent_relation(ops_test: OpsTest, provided_collectors):
"""Test adding relation with grafana-agent."""
"""Test adding relation with opentelemetry-collector."""
redfish_present = True if "redfish" in provided_collectors else False

# Add cos-agent relation
logging.info("Adding cos-agent relation.")
status = "blocked" if redfish_present else "active"
await asyncio.gather(
ops_test.model.add_relation(
f"{APP_NAME}:cos-agent", f"{GRAFANA_AGENT_APP_NAME}:cos-agent"
),
ops_test.model.add_relation(f"{APP_NAME}:cos-agent", f"{OTEL_APP_NAME}:cos-agent"),
ops_test.model.wait_for_idle(
apps=[APP_NAME],
status=status,
Expand Down
14 changes: 7 additions & 7 deletions tests/integration/test_cos_integration.py
Original file line number Diff line number Diff line change
Expand Up @@ -28,7 +28,7 @@ async def test_setup_and_deploy(base, channel, lxd_ctl, k8s_ctl, lxd_model, k8s_
await _add_cross_controller_relations(k8s_ctl, lxd_ctl, k8s_model, lxd_model)

# This verifies that the cross-controller relation with COS is successful
assert lxd_model.applications["grafana-agent"].status == "active"
assert lxd_model.applications["opentelemetry-collector"].status == "active"


async def test_alerts(ops_test: OpsTest, lxd_model, k8s_model):
Expand Down Expand Up @@ -155,19 +155,19 @@ async def _deploy_hardware_observer(base, channel, model):
model.deploy("ubuntu", num_units=1, base=base, channel=channel),
# Hardware Observer
model.deploy("hardware-observer", base=base, num_units=0, channel=channel),
# Grafana Agent
model.deploy("grafana-agent", num_units=0, base=base, channel=channel),
# OpenTelemetry Collector
model.deploy("opentelemetry-collector", num_units=0, base=base, channel=channel),
)

await model.add_relation("ubuntu:juju-info", "hardware-observer:general-info")
await model.add_relation("hardware-observer:cos-agent", "grafana-agent:cos-agent")
await model.add_relation("ubuntu:juju-info", "grafana-agent:juju-info")
await model.add_relation("hardware-observer:cos-agent", "opentelemetry-collector:cos-agent")
await model.add_relation("ubuntu:juju-info", "opentelemetry-collector:juju-info")

await model.block_until(lambda: model.applications["hardware-observer"].status == "active")


async def _add_cross_controller_relations(k8s_ctl, lxd_ctl, k8s_model, lxd_model):
"""Add relations between Grafana Agent and COS."""
"""Add relations between OpenTelemetry Collector and COS."""
cos_saas_names = ["prometheus-receive-remote-write", "loki-logging", "grafana-dashboards"]
for saas in cos_saas_names:
# Using juju cli since Model.consume() from libjuju causes error.
Expand All @@ -180,7 +180,7 @@ async def _add_cross_controller_relations(k8s_ctl, lxd_ctl, k8s_model, lxd_model
f"{k8s_ctl.controller_name}:admin/{k8s_model.name}.{saas}",
]
subprocess.run(cmd, check=True, stdout=subprocess.PIPE, stderr=subprocess.STDOUT)
await lxd_model.add_relation("grafana-agent", saas),
await lxd_model.add_relation("opentelemetry-collector", saas),

# `idle_period` needs to be greater than the scrape interval to make sure metrics ingested.
await asyncio.gather(
Expand Down
14 changes: 7 additions & 7 deletions tests/manual/etc/4_deploy_grafana_agent/main.tf
Original file line number Diff line number Diff line change
Expand Up @@ -9,19 +9,19 @@ terraform {

provider "juju" {}

module "grafana-agent" {
module "opentelemetry-collector" {
source = "git::https://github.com/canonical/snap-openstack.git//sunbeam-python/sunbeam/features/observability/etc/deploy-grafana-agent"

grafana-agent-base = var.grafana_agent_base
grafana-agent-channel = "1/stable" # Can move back to latest/stable when the charm is updated
opentelemetry-collector-base = var.otel_base
opentelemetry-collector-channel = "2/stable"
principal-application-model = var.machine_model
receive-remote-write-offer-url = var.receive-remote-write-offer-url
grafana-dashboard-offer-url = var.grafana-dashboard-offer-url
logging-offer-url = var.loki-logging-offer-url

}

resource "juju_integration" "ubuntu-to-grafana-agent" {
resource "juju_integration" "ubuntu-to-opentelemetry-collector" {
model = var.machine_model

application {
Expand All @@ -30,12 +30,12 @@ resource "juju_integration" "ubuntu-to-grafana-agent" {
}

application {
name = "grafana-agent"
name = "opentelemetry-collector"
endpoint = "juju-info"
}
}

resource "juju_integration" "hardware-observer-to-grafana-agent" {
resource "juju_integration" "hardware-observer-to-opentelemetry-collector" {
model = var.machine_model

application {
Expand All @@ -44,7 +44,7 @@ resource "juju_integration" "hardware-observer-to-grafana-agent" {
}

application {
name = "grafana-agent"
name = "opentelemetry-collector"
endpoint = "cos-agent"
}
}
Loading
Loading