Add Deck dashboard with Athena telemetry history
This commit is contained in:
+46
-7
@@ -3,8 +3,14 @@ import csv
|
||||
import json
|
||||
import time
|
||||
import subprocess
|
||||
import os
|
||||
import shutil
|
||||
from pathlib import Path
|
||||
|
||||
PROC = Path(os.environ.get('DECK_HOST_PROC', '/proc'))
|
||||
DATA_DISK = Path(os.environ.get('DECK_HOST_DATA_DISK', '/var/lib/deck'))
|
||||
_network_previous = None
|
||||
|
||||
|
||||
def read(path):
|
||||
try:
|
||||
@@ -21,11 +27,12 @@ def number(value):
|
||||
|
||||
|
||||
def ticks():
|
||||
values = list(map(int, read('/proc/stat').splitlines()[0].split()[1:9]))
|
||||
values = list(map(int, read(PROC/'stat').splitlines()[0].split()[1:9]))
|
||||
return sum(values), sum(values[3:5])
|
||||
|
||||
|
||||
def collect():
|
||||
global _network_previous
|
||||
errors = []
|
||||
cpu = {'name': None, 'percent': None, 'temperature_c': None}
|
||||
try:
|
||||
@@ -35,7 +42,7 @@ def collect():
|
||||
cpu['percent'] = round(100 * (1 - (bi-ai)/(b-a)), 1) if b > a else None
|
||||
except (ValueError, IndexError):
|
||||
errors.append('CPU-Auslastung nicht verfügbar')
|
||||
for line in read('/proc/cpuinfo').splitlines():
|
||||
for line in read(PROC/'cpuinfo').splitlines():
|
||||
if line.startswith('model name'):
|
||||
cpu['name'] = line.split(':', 1)[1].strip()
|
||||
break
|
||||
@@ -45,7 +52,7 @@ def collect():
|
||||
cpu['temperature_c'] = value / 1000 if value is not None else None
|
||||
break
|
||||
mem = {}
|
||||
for line in read('/proc/meminfo').splitlines():
|
||||
for line in read(PROC/'meminfo').splitlines():
|
||||
parts = line.split()
|
||||
if parts[0] in ('MemTotal:', 'MemAvailable:'):
|
||||
mem[parts[0][:-1]] = int(parts[1]) * 1024
|
||||
@@ -53,13 +60,45 @@ def collect():
|
||||
ram = {'total_bytes': total, 'used_bytes': total-available if total is not None and available is not None else None}
|
||||
gpus = []
|
||||
try:
|
||||
result = subprocess.run(['nvidia-smi', '--query-gpu=index,name,uuid,memory.total,memory.used,utilization.gpu,temperature.gpu', '--format=csv,noheader,nounits'], capture_output=True, text=True, timeout=4, check=True)
|
||||
result = subprocess.run(['nvidia-smi', '--query-gpu=index,name,uuid,memory.total,memory.used,utilization.gpu,temperature.gpu,utilization.memory,memory.free,power.draw,power.limit,clocks.current.graphics,clocks.current.memory,fan.speed,pstate', '--format=csv,noheader,nounits'], capture_output=True, text=True, timeout=4, check=True)
|
||||
for row in csv.reader(result.stdout.splitlines(), skipinitialspace=True):
|
||||
index, name, uuid, total, used, usage, temp = row
|
||||
gpus.append(dict(index=int(index), name=name, uuid=uuid, total_mib=number(total), used_mib=number(used), percent=number(usage), temperature_c=number(temp)))
|
||||
if len(row)!=15:continue
|
||||
index, name, uuid, total, used, usage, temp, memory_usage, free, power, power_limit, graphics_clock, memory_clock, fan, pstate = row
|
||||
gpus.append(dict(index=int(index), name=name, uuid=uuid, total_mib=number(total), used_mib=number(used), percent=number(usage), temperature_c=number(temp),memory_controller_percent=number(memory_usage),free_mib=number(free),power_w=number(power),power_limit_w=number(power_limit),graphics_clock_mhz=number(graphics_clock),memory_clock_mhz=number(memory_clock),fan_percent=number(fan),pstate=pstate))
|
||||
except (OSError, subprocess.SubprocessError, ValueError):
|
||||
errors.append('GPU-Messwerte nicht verfügbar')
|
||||
return dict(source='Debian-Server · /proc + hwmon + nvidia-smi', sampled_at=time.time(), cpu=cpu, ram=ram, gpus=gpus, errors=errors)
|
||||
gpu_processes=[]
|
||||
try:
|
||||
result=subprocess.run(['nvidia-smi','--query-compute-apps=gpu_uuid,pid,process_name,used_memory','--format=csv,noheader,nounits'],capture_output=True,text=True,timeout=4,check=True)
|
||||
for row in csv.reader(result.stdout.splitlines(),skipinitialspace=True):
|
||||
if len(row)==4:gpu_processes.append(dict(gpu_uuid=row[0],pid=number(row[1]),name=row[2],memory_mib=number(row[3])))
|
||||
except (OSError,subprocess.SubprocessError):pass
|
||||
try:load=[float(x) for x in read(PROC/'loadavg').split()[:3]]
|
||||
except ValueError:load=[]
|
||||
try:uptime=float(read(PROC/'uptime').split()[0])
|
||||
except (ValueError,IndexError):uptime=None
|
||||
try:
|
||||
disk=shutil.disk_usage(DATA_DISK)
|
||||
disk_data=dict(total=disk.total,used=disk.used,free=disk.free)
|
||||
except OSError:disk_data={}
|
||||
rx=tx=interfaces=0
|
||||
# A bind-mounted /proc/net is a self/net symlink and would show the
|
||||
# container namespace. PID 1 in the host proc mount has the host network.
|
||||
host_net=PROC/'1/net/dev' if (PROC/'1/net/dev').exists() else PROC/'net/dev'
|
||||
for line in read(host_net).splitlines()[2:]:
|
||||
if ':' not in line:continue
|
||||
name,raw=line.split(':',1)
|
||||
if name.strip()=='lo':continue
|
||||
parts=raw.split()
|
||||
if len(parts)<9:continue
|
||||
try:rx+=int(parts[0]);tx+=int(parts[8]);interfaces+=1
|
||||
except ValueError:pass
|
||||
now=time.monotonic();rates=dict(rx_bytes_per_second=None,tx_bytes_per_second=None)
|
||||
if _network_previous and now>_network_previous[0]:
|
||||
elapsed=now-_network_previous[0]
|
||||
rates=dict(rx_bytes_per_second=round(max(0,rx-_network_previous[1])/elapsed,1),tx_bytes_per_second=round(max(0,tx-_network_previous[2])/elapsed,1))
|
||||
_network_previous=(now,rx,tx)
|
||||
return dict(source='Debian-Server · /proc + hwmon + nvidia-smi', sampled_at=time.time(), cpu={**cpu,'logical_cpus':os.cpu_count(),'load':load,'host_uptime_seconds':uptime,'disk_data':disk_data,'network':dict(interfaces=interfaces,rx_bytes=rx,tx_bytes=tx,**rates)}, ram=ram, gpus=gpus,gpu_processes=gpu_processes, errors=errors)
|
||||
|
||||
|
||||
if __name__ == '__main__':
|
||||
|
||||
Reference in New Issue
Block a user