How keilhau.org is made. Everything runs on node1, a small server at home. A cron job regenerates the site every hour and uploads it.

Overview

 loggers (every minute)            hourly: make
 ----------------------            ------------------------------------------
 smaread.service   -> log.smaread  graphs.py      logs -> diagramms/*.png
 opendtu.service   -> log.dtu                           -> stats*.adoc
 pingcheck.service -> log.ping     generate-docs  *.adoc, *.txt -> *.html
 sysstat.service   -> log.sysstat  lftp           upload

The graphs are drawn with matplotlib by rrdplot.py, a small module that imitates rrdtool: its layout, legend and fonts, dotted grid, one consolidated value per pixel column and gaps where there is no data. The logs stay plain text files, there is no round robin database.

The pages are written in AsciiDoc and rendered with Asciidoctor and a plain black and white stylesheet. The stats pages are rendered once per period (24 hours, 7 days, 30 days, 365 days).

crontab

# /etc/crontab: system-wide crontab
# Unlike any other crontab you don't have to run the `crontab'
# command to install the new version when you edit this file
# and files in /etc/cron.d. These files also have username fields,
# that none of the other crontabs do.

SHELL=/bin/sh
PATH=/usr/local/sbin:/usr/local/bin:/sbin:/bin:/usr/sbin:/usr/bin

# m h dom mon dow user	command
17 *	* * *	root    cd / && run-parts --report /etc/cron.hourly
25 6	* * *	root	test -x /usr/sbin/anacron || ( cd / && run-parts --report /etc/cron.daily )
47 6	* * 7	root	test -x /usr/sbin/anacron || ( cd / && run-parts --report /etc/cron.weekly )
52 6	1 * *	root	test -x /usr/sbin/anacron || ( cd / && run-parts --report /etc/cron.monthly )
#

# update keilhau.org website every hour
0 * * * * timo /home/timo/databox/DEVelop/keilhau.org/content/./make

Services

The loggers run as systemd services.

[Unit]
Description=keilhau.org system stats logger
After=local-fs.target

[Service]
User=timo
ExecStart=/usr/bin/python3 /home/timo/databox/DEVelop/keilhau.org/content/sysstat.py
Restart=always
RestartSec=30

[Install]
WantedBy=multi-user.target
sudo cp -av ../systemctl/*.service /etc/systemd/system
sudo systemctl daemon-reload
sudo systemctl enable --now sysstat.service pingcheck.service opendtu.service

Scripts

make

Called by cron.

#!/bin/bash
# make does it all, called hourly by cron

dir=$(dirname "$(realpath "$0")")

(cd "$dir" && ./update_server_info)
(cd "$dir" && ./graphs.py)
(cd "$dir" && ./generate-docs)
(cd "$dir" && lftp -f ftp-upload)

generate-docs

#!/bin/bash
# generates keilhau.org with asciidoctor
#   *.adoc  pages in current asciidoc syntax
#   *.txt   older notes in legacy asciidoc syntax, rendered in compat mode
# the stats pages are rendered once per period: pv.html, pv-7d.html, ...

cd "$(dirname "$(realpath "$0")")" || exit 1

opts=(-a stylesheet=site.css -a stylesdir=. -a docinfodir=.)
stats=" pv ping server-status temperature "

# links to the other periods of a stats page, the current one in bold
periods(){
	local page=$1 cur=$2 out="" p file
	for p in 24h 7d 30d 365d; do
		file=$page-$p.html
		[[ $p == 24h ]] && file=$page.html
		label=${p/h/ hours}; label=${label/d/ days}
		if [[ $p == "$cur" ]]; then
			out+="link:$file[$label,role=current] "
		else
			out+="link:$file[$label] "
		fi
	done
	echo "$out"
}

for f in *.adoc; do
	page=${f%.adoc}
	[[ $page == stats* ]] && continue	# generated includes, see graphs.py
	if [[ $stats == *" $page "* ]]; then
		for p in 24h 7d 30d 365d; do
			out=$page-$p.html
			[[ $p == 24h ]] && out=$page.html
			asciidoctor "${opts[@]}" -a period=$p -a periods="$(periods "$page" $p)" \
				-o "$out" "$f"
		done
	else
		asciidoctor "${opts[@]}" "$f"
	fi
done

for f in *.txt; do
	asciidoctor "${opts[@]}" -a compat-mode "$f"
done

sysstat.py

System stats logger, replaces dstat and the USB thermometer.

#!/usr/bin/python3
# collects basic system stats once a minute and appends them to a csv log
# replaces the dstat and temper loggers; reads /proc and /sys only, no root needed

import glob
import os
import sys
import time
from datetime import datetime

LOG = "/home/timo/logs/log.sysstat"
IFACE = "enp1s0"
DISKS = ["sda", "sdb"]              # block devices for i/o and temperature
MOUNTS = ["/", "/mnt/ssd"]          # filesystems for usage in percent
INTERVAL = 60

COLUMNS = ["Timestamp", "Load1", "Load5", "Load15",
           "CpuUser", "CpuSys", "CpuIowait", "CpuIrq", "CpuIdle",
           "MemUsed", "MemBuffers", "MemCached", "MemFree",
           "NetRx", "NetTx", "DiskRead", "DiskWrite",
           "TempCpu", "TempBoard"] \
    + ["Temp_" + d for d in DISKS] \
    + ["Usage_" + (m.strip("/").replace("/", "_") or "root") for m in MOUNTS]


def cpu_times():
    with open("/proc/stat") as f:
        v = [int(x) for x in f.readline().split()[1:]]
    # user nice system idle iowait irq softirq steal
    return {"user": v[0] + v[1], "sys": v[2], "idle": v[3], "iowait": v[4],
            "irq": v[5] + v[6] + v[7]}


def meminfo():
    m = {}
    with open("/proc/meminfo") as f:
        for line in f:
            k, v = line.split(":")
            m[k] = int(v.split()[0]) * 1024
    cached = m["Cached"] + m.get("SReclaimable", 0)
    used = m["MemTotal"] - m["MemFree"] - m["Buffers"] - cached
    return used, m["Buffers"], cached, m["MemFree"]


def net_bytes():
    with open("/proc/net/dev") as f:
        for line in f:
            if line.strip().startswith(IFACE + ":"):
                v = line.split(":")[1].split()
                return int(v[0]), int(v[8])
    return 0, 0


def disk_bytes():
    rd = wr = 0
    with open("/proc/diskstats") as f:
        for line in f:
            v = line.split()
            if v[2] in DISKS:
                rd += int(v[5]) * 512
                wr += int(v[9]) * 512
    return rd, wr


def read_temp(path):
    try:
        with open(path) as f:
            return int(f.read()) / 1000.0
    except (OSError, ValueError):
        return None


def temperatures():
    cpu = board = None
    disks = {d: None for d in DISKS}
    for hw in glob.glob("/sys/class/hwmon/hwmon*"):
        try:
            name = open(hw + "/name").read().strip()
        except OSError:
            continue
        if name == "coretemp":
            # temp1 is the package sensor, fall back to the hottest core
            temps = [read_temp(p) for p in glob.glob(hw + "/temp*_input")]
            temps = [t for t in temps if t is not None]
            cpu = read_temp(hw + "/temp1_input") or (max(temps) if temps else None)
        elif name == "acpitz":
            board = read_temp(hw + "/temp1_input")
        elif name == "drivetemp":
            # needs the drivetemp kernel module: modprobe drivetemp
            blocks = glob.glob(hw + "/device/block/*")
            if blocks and os.path.basename(blocks[0]) in disks:
                disks[os.path.basename(blocks[0])] = read_temp(hw + "/temp1_input")
    return cpu, board, [disks[d] for d in DISKS]


def usage():
    res = []
    for m in MOUNTS:
        try:
            s = os.statvfs(m)
            res.append(100.0 * (s.f_blocks - s.f_bfree) / s.f_blocks)
        except OSError:
            res.append(None)
    return res


def fmt(v):
    if v is None:
        return ""
    if isinstance(v, float):
        return "%.2f" % v
    return str(v)


def main():
    new = not os.path.exists(LOG)
    out = open(LOG, "a")
    if new:
        out.write(",".join(COLUMNS) + "\n")
        out.flush()

    prev_t = time.monotonic()
    prev_cpu, prev_net, prev_disk = cpu_times(), net_bytes(), disk_bytes()
    while True:
        # sleep until the next full minute so samples line up with the other logs
        time.sleep(INTERVAL - time.time() % INTERVAL)
        try:
            now = time.monotonic()
            dt = now - prev_t
            cpu, net, disk = cpu_times(), net_bytes(), disk_bytes()

            d = {k: cpu[k] - prev_cpu[k] for k in cpu}
            total = sum(d.values()) or 1
            pct = {k: 100.0 * v / total for k, v in d.items()}
            # counters can reset (interface down, reboot); clamp to zero
            rate = lambda a, b: max(0.0, (a - b) / dt)

            load = open("/proc/loadavg").read().split()[:3]
            temp_cpu, temp_board, temp_disks = temperatures()
            row = [datetime.now().strftime("%Y-%m-%d %H:%M:%S")] + load \
                + [pct["user"], pct["sys"], pct["iowait"], pct["irq"], pct["idle"]] \
                + list(meminfo()) \
                + [rate(net[0], prev_net[0]), rate(net[1], prev_net[1]),
                   rate(disk[0], prev_disk[0]), rate(disk[1], prev_disk[1]),
                   temp_cpu, temp_board] + temp_disks + usage()
            out.write(",".join(fmt(v) for v in row) + "\n")
            out.flush()
            prev_t, prev_cpu, prev_net, prev_disk = now, cpu, net, disk
        except Exception as e:
            print("sysstat: %s" % e, file=sys.stderr)


if __name__ == "__main__":
    main()

opendtu.py

Micro inverter logger, reads the OpenDTU web API.

#!/usr/bin/python3
# logs the hoymiles micro inverter via the OpenDTU web api once a minute
# replaces start_opendtu (log.opendtu), which only logged a few values
#
#   log.dtu          all live values, csv with header
#   log.dtu-events   the inverter event log (start, stop, grid faults, ...)
#
# rows with Online=0 mean the DTU itself was not reachable

import csv
import json
import os
import sys
import time
import urllib.request
from datetime import datetime, timedelta

DTU = "http://10.0.0.245"
LOG = "/home/timo/logs/log.dtu"
EVENTS = "/home/timo/logs/log.dtu-events"
INTERVAL = 60
EVENT_INTERVAL = 600        # the event log changes rarely

# (column, section, field) in the livedata of the first inverter
LIVE = [("AcPower", "AC", "Power"), ("AcVoltage", "AC", "Voltage"),
        ("AcCurrent", "AC", "Current"), ("AcFrequency", "AC", "Frequency"),
        ("PowerFactor", "AC", "PowerFactor"), ("ReactivePower", "AC", "ReactivePower"),
        ("Efficiency", "AC", "Efficiency"),
        ("DcPower", "DC", "Power"), ("DcVoltage", "DC", "Voltage"),
        ("DcCurrent", "DC", "Current"), ("Irradiation", "DC", "Irradiation"),
        ("YieldDay", "AC", "YieldDay"), ("YieldTotal", "AC", "YieldTotal"),
        ("Temperature", "INV", "Temperature")]

COLUMNS = ["Timestamp", "Online", "Reachable", "Producing", "DataAge", "LimitRel"] \
    + [c for c, _, _ in LIVE] + ["Rssi", "DtuUptime", "DtuHeapFree", "Events"]


def get(path):
    with urllib.request.urlopen(DTU + path, timeout=10) as r:
        return json.load(r)


def sample(serial):
    live = get("/api/livedata/status?inv=" + serial)["inverters"][0]
    system = get("/api/system/status")
    net = get("/api/network/status")
    row = [1, int(live["reachable"]), int(live["producing"]), live["data_age"],
           live.get("limit_relative")]
    for _, sec, field in LIVE:
        try:
            row.append(live[sec]["0"][field]["v"])
        except KeyError:
            row.append(None)
    row += [net.get("sta_rssi"), system.get("uptime"),
            system["heap_total"] - system["heap_used"] if "heap_total" in system else None,
            live.get("events")]
    return row


def fmt(v):
    if v is None:
        return ""
    if isinstance(v, float):
        return ("%.3f" % v).rstrip("0").rstrip(".")
    return str(v)


def known_events():
    """(day, id, start) of the events already logged"""
    seen = set()
    if os.path.exists(EVENTS):
        with open(EVENTS) as f:
            for r in csv.DictReader(f):
                seen.add((r["Timestamp"][:10], r["Id"], r["Timestamp"][11:]))
    return seen


def log_events(serial, seen):
    """the inverter keeps the events of the current day, start and end in
    seconds since midnight; append the ones not logged yet"""
    ev = get("/api/eventlog/status?inv=" + serial).get("events", [])
    now = datetime.now()
    midnight = now.replace(hour=0, minute=0, second=0, microsecond=0)
    new = not os.path.exists(EVENTS)
    with open(EVENTS, "a") as f:
        if new:
            f.write("Timestamp,End,Id,Message\n")
        for e in ev:
            start = midnight + timedelta(seconds=e["start_time"])
            if start > now + timedelta(minutes=5):     # left over from yesterday
                continue
            end = midnight + timedelta(seconds=e["end_time"])
            key = (start.strftime("%Y-%m-%d"), str(e["message_id"]), start.strftime("%H:%M:%S"))
            if key in seen:
                continue
            seen.add(key)
            f.write("%s,%s,%d,%s\n" % (start.strftime("%Y-%m-%d %H:%M:%S"),
                                        end.strftime("%H:%M:%S"), e["message_id"],
                                        e["message"].replace(",", ";")))


def main():
    new = not os.path.exists(LOG)
    out = open(LOG, "a")
    if new:
        out.write(",".join(COLUMNS) + "\n")
        out.flush()

    serial, seen, last_events = None, known_events(), 0
    while True:
        # sleep until the next full minute so samples line up with the other logs
        time.sleep(INTERVAL - time.time() % INTERVAL)
        ts = datetime.now().strftime("%Y-%m-%d %H:%M:%S")
        try:
            if serial is None:
                serial = get("/api/livedata/status")["inverters"][0]["serial"]
            row = sample(serial)
        except Exception as e:
            print("opendtu: %s" % e, file=sys.stderr)
            row = [0]
        out.write(",".join([ts] + [fmt(v) for v in row]) + "\n")
        out.flush()
        if serial and row[0] and time.time() - last_events >= EVENT_INTERVAL:
            try:
                log_events(serial, seen)
                last_events = time.time()
            except Exception as e:
                print("opendtu events: %s" % e, file=sys.stderr)


if __name__ == "__main__":
    main()

start_pingcheck

#!/bin/bash

host="8.8.8.8"
pingCmd="ping -c1 $host"
filter="| grep min | sed 's/[^0-9.]*\([0-9.]*\).*/\1/'"
dir=$(dirname $0)
log="/home/timo/logs/log.ping"
zzz=60

while [ 1 ]; do 
	ms=$(eval "$pingCmd $filter")
	sleep $zzz
	date=$(date "+%Y/%m/%d %H:%M:%S")
	# if ms == "" the ping to host was lost
	if [[ $ms == "" ]]; then
		echo "$date 0.0" >> $log
		continue
	fi
	echo "$date $ms" >> $log
done