Files
CloudOps/platform/modules/kubernetes.py
root a8f996422c Rewrite Application Sites page to reflect real k8s state
modules/sites.py was still a fully Docker-era hardcoded registry:
container names (odoo-clean-odoo-1, frappe-erpnext, nextcloud-app,
mautic-app, n8n-app) that no longer exist post-migration, checked via
`docker inspect` over SSH, plus hardcoded domain/port fields -
nextcloud and mautic even had domain: None, silently falling back to
dead Docker host-port URLs while their real Ingress domains
(next.cloud.nav.ovh, mautics.nav.ovh) sat unused.

Now sources everything live from the cluster:
- Domain + TLS from the actual Ingress object per app (via the new
  get_ingress_info() in modules/kubernetes.py), not a static guess.
  Odoo/Nextcloud's Ingress lives in `default` (see prior commit for the
  RBAC this needed); n8n/mautic/erpnext's lives in their own namespace.
- App/DB/cache/worker/scheduler status from real Deployment state
  (list_deployments_for_namespace(), new in modules/kubernetes.py)
  instead of `docker inspect`.
- Health probe now hits the real domain directly from the pod (it has
  normal internet egress) instead of SSH-ing back out to the host to
  curl a public HTTPS URL, which is what the RUNNING_ON_MAIN_SERVER
  branch would have done for a pod whose hostname never matches the
  main-server hostname check.

Same API/template contract as before (get_sites_list/get_site_health
return the same field shapes) - templates/pages/sites.html needs no
changes.

Co-Authored-By: Claude Sonnet 5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_017avLHFqkiti3g62Anq9sVA
2026-08-21 01:43:38 +02:00

292 lines
9.7 KiB
Python

# modules/kubernetes.py — read-only cluster view via the in-cluster ServiceAccount
#
# Auth: management-platform-viewer-sa (get/list/watch on pods, services,
# deployments.apps — no secrets, no write, no exec/log), bound per-namespace
# across the 8 app namespaces below. There is no cluster-wide RoleBinding,
# so every list call below is scoped to one namespace at a time.
from datetime import datetime, timezone
try:
from kubernetes import client, config as k8s_config
from kubernetes.client.rest import ApiException
_K8S_IMPORT_OK = True
except ImportError:
_K8S_IMPORT_OK = False
NAMESPACES = [
'n8n', 'odoo', 'mautic', 'erpnext', 'nextcloud',
'jenkins-agents', 'jenkins', 'management-platform',
]
_core_v1 = None
_apps_v1 = None
_custom_objects = None
_networking_v1 = None
K8S_AVAILABLE = False
def _init_client():
global _core_v1, _apps_v1, _custom_objects, _networking_v1, K8S_AVAILABLE
if _core_v1 is not None or not _K8S_IMPORT_OK:
return
try:
k8s_config.load_incluster_config()
_core_v1 = client.CoreV1Api()
_apps_v1 = client.AppsV1Api()
_custom_objects = client.CustomObjectsApi()
_networking_v1 = client.NetworkingV1Api()
K8S_AVAILABLE = True
except Exception as e:
print(f"[kubernetes] in-cluster config unavailable: {e}")
# ────────────────────────────────────────────────────────────────
# HELPERS
# ────────────────────────────────────────────────────────────────
def _age_str(dt):
if not dt:
return ''
seconds = int((datetime.now(timezone.utc) - dt).total_seconds())
if seconds < 60:
return f'{seconds}s'
minutes = seconds // 60
if minutes < 60:
return f'{minutes}m'
hours = minutes // 60
if hours < 24:
return f'{hours}h {minutes % 60}m'
days = hours // 24
return f'{days}d {hours % 24}h'
_CRASH_REASONS = ('CrashLoopBackOff', 'ImagePullBackOff', 'ErrImagePull', 'Error')
def _pod_health(pod):
statuses = pod.status.container_statuses or []
for cs in statuses:
waiting = cs.state.waiting if cs.state else None
if waiting and waiting.reason in _CRASH_REASONS:
return 'failed'
phase = pod.status.phase or 'Unknown'
if phase == 'Failed':
return 'failed'
if phase == 'Running':
all_ready = all(cs.ready for cs in statuses) if statuses else True
return 'healthy' if all_ready else 'degraded'
return 'degraded' # Pending, Unknown, etc.
def _ready_count(pod):
statuses = pod.status.container_statuses or []
ready = sum(1 for cs in statuses if cs.ready)
return f'{ready}/{len(statuses)}'
def _restart_count(pod):
statuses = pod.status.container_statuses or []
return sum(cs.restart_count for cs in statuses)
def _deploy_health(desired, ready):
if desired == 0:
return 'degraded'
if ready == desired:
return 'healthy'
if ready == 0:
return 'failed'
return 'degraded'
def _parse_cpu(v):
v = (v or '0').strip()
try:
if v.endswith('n'):
return int(v[:-1]) / 1_000_000
if v.endswith('u'):
return int(v[:-1]) / 1_000
if v.endswith('m'):
return int(v[:-1])
return float(v) * 1000
except ValueError:
return 0
_MEM_UNITS = {'Ki': 1 / 1024, 'Mi': 1, 'Gi': 1024, 'K': 1 / 1024, 'M': 1, 'G': 1024}
def _parse_mem_mib(v):
v = (v or '0').strip()
for suffix, factor in _MEM_UNITS.items():
if v.endswith(suffix):
try:
return float(v[:-len(suffix)]) * factor
except ValueError:
return 0
try:
return float(v) / (1024 * 1024)
except ValueError:
return 0
# ────────────────────────────────────────────────────────────────
# PODS / DEPLOYMENTS
# ────────────────────────────────────────────────────────────────
def list_pods():
_init_client()
pods = []
if not K8S_AVAILABLE:
return pods
for ns in NAMESPACES:
try:
resp = _core_v1.list_namespaced_pod(ns)
except ApiException as e:
print(f"[kubernetes] list pods failed for {ns}: {e}")
continue
for pod in resp.items:
pods.append({
'name': pod.metadata.name,
'namespace': ns,
'phase': pod.status.phase or 'Unknown',
'health': _pod_health(pod),
'ready': _ready_count(pod),
'restarts': _restart_count(pod),
'age': _age_str(pod.metadata.creation_timestamp),
})
return pods
def list_deployments():
_init_client()
deployments = []
if not K8S_AVAILABLE:
return deployments
for ns in NAMESPACES:
try:
resp = _apps_v1.list_namespaced_deployment(ns)
except ApiException as e:
print(f"[kubernetes] list deployments failed for {ns}: {e}")
continue
for dep in resp.items:
desired = dep.spec.replicas or 0
ready = dep.status.ready_replicas or 0
deployments.append({
'name': dep.metadata.name,
'namespace': ns,
'ready': ready,
'desired': desired,
'health': _deploy_health(desired, ready),
'age': _age_str(dep.metadata.creation_timestamp),
})
return deployments
def get_ingress_info(name, namespace):
"""Real domain/TLS for one Ingress, or None if unavailable/not found.
Used by modules/sites.py instead of the old hardcoded Docker-era
domain fields — reflects actual live k8s Ingress state."""
_init_client()
if not K8S_AVAILABLE:
return None
try:
ing = _networking_v1.read_namespaced_ingress(name, namespace)
except Exception:
return None
host = None
if ing.spec and ing.spec.rules:
for rule in ing.spec.rules:
if rule.host:
host = rule.host
break
tls = bool(ing.spec and ing.spec.tls)
return {'host': host, 'tls': tls}
def list_deployments_for_namespace(ns):
"""Targeted single-namespace deployment list (vs. list_deployments()'s
fixed NAMESPACES sweep) — used by modules/sites.py per app."""
_init_client()
if not K8S_AVAILABLE:
return []
try:
resp = _apps_v1.list_namespaced_deployment(ns)
except Exception as e:
print(f"[kubernetes] list deployments failed for {ns}: {e}")
return []
out = []
for dep in resp.items:
desired = dep.spec.replicas or 0
ready = dep.status.ready_replicas or 0
containers = dep.spec.template.spec.containers or []
image = containers[0].image if containers else ''
out.append({
'name': dep.metadata.name,
'desired': desired,
'ready': ready,
'health': _deploy_health(desired, ready),
'image': image,
})
return out
def get_pod_metrics():
"""Best-effort CPU/mem per pod via metrics-server. Returns {} if it's not
installed or unreachable — callers must treat an empty dict as 'no data',
never as an error."""
_init_client()
metrics = {}
if not K8S_AVAILABLE:
return metrics
for ns in NAMESPACES:
try:
resp = _custom_objects.list_namespaced_custom_object(
'metrics.k8s.io', 'v1beta1', ns, 'pods'
)
except Exception:
continue
for item in resp.get('items', []):
name = item.get('metadata', {}).get('name')
containers = item.get('containers', [])
cpu = sum(_parse_cpu(c.get('usage', {}).get('cpu')) for c in containers)
mem = sum(_parse_mem_mib(c.get('usage', {}).get('memory')) for c in containers)
metrics[f'{ns}/{name}'] = {'cpu_millicores': round(cpu), 'mem_mib': round(mem)}
return metrics
def get_cluster_overview():
"""Single call powering the Cluster page: pods + deployments grouped by
namespace, plus summary counts. Mirrors the shape of get_all_stats() /
get_sites_list() in the docker-facing modules."""
pods = list_pods()
deployments = list_deployments()
metrics = get_pod_metrics()
for pod in pods:
key = f"{pod['namespace']}/{pod['name']}"
if key in metrics:
pod['metrics'] = metrics[key]
by_namespace = {ns: {'pods': [], 'deployments': []} for ns in NAMESPACES}
for pod in pods:
by_namespace[pod['namespace']]['pods'].append(pod)
for dep in deployments:
by_namespace[dep['namespace']]['deployments'].append(dep)
summary = {
'total_pods': len(pods),
'healthy': sum(1 for p in pods if p['health'] == 'healthy'),
'degraded': sum(1 for p in pods if p['health'] == 'degraded'),
'failed': sum(1 for p in pods if p['health'] == 'failed'),
'total_deployments': len(deployments),
'deployments_healthy': sum(1 for d in deployments if d['health'] == 'healthy'),
}
return {
'available': K8S_AVAILABLE,
'namespaces': by_namespace,
'summary': summary,
'metrics_available': bool(metrics),
}