Geomys Tuscolo CT Log Server Public Configuration

This is the live public configuration for the Geomys Tuscolo CT Log Server, a Sunlight instance.

See also the public playbooks.

/etc/sunlight/sunlight.yaml

homehtml: /etc/sunlight/home.html
listen:
  - "185.230.223.193:443"
  - "[2a0c:2f07:c1::c1]:443"
acme:
  hosts:
    - tuscolo.sunlight.geomys.org
  cache: /var/db/sunlight/autocert/
checkpoints: /tank/shared/checkpoints.db
logs:
  - shortname: tuscolo2026h2
    inception: 2025-04-27
    period: 200
    submissionprefix: https://tuscolo2026h2.sunlight.geomys.org
    monitoringprefix: https://tuscolo2026h2.skylight.geomys.org
    secret: /tank/enc/tuscolo2026h2.seed.bin
    cache: /tank/caches/tuscolo2026h2/cache.db
    poolsize: 150
    localdirectory: /tank/logs/tuscolo2026h2/data
    notafterstart: 2026-07-01T00:00:00Z
    notafterlimit: 2027-01-01T00:00:00Z
  - shortname: tuscolo2027h1
    inception: 2025-06-02
    period: 200
    submissionprefix: https://tuscolo2027h1.sunlight.geomys.org
    monitoringprefix: https://tuscolo2027h1.skylight.geomys.org
    secret: /tank/enc/tuscolo2027h1.seed.bin
    cache: /tank/caches/tuscolo2027h1/cache.db
    poolsize: 150
    localdirectory: /tank/logs/tuscolo2027h1/data
    notafterstart: 2027-01-01T00:00:00Z
    notafterlimit: 2027-07-01T00:00:00Z
  - shortname: tuscolo2027h2
    inception: 2025-06-02
    period: 200
    submissionprefix: https://tuscolo2027h2.sunlight.geomys.org
    monitoringprefix: https://tuscolo2027h2.skylight.geomys.org
    secret: /tank/enc/tuscolo2027h2.seed.bin
    cache: /tank/caches/tuscolo2027h2/cache.db
    poolsize: 150
    localdirectory: /tank/logs/tuscolo2027h2/data
    notafterstart: 2027-07-01T00:00:00Z
    notafterlimit: 2028-01-01T00:00:00Z
  - shortname: tuscolo2028h1
    inception: 2026-07-28
    period: 200
    submissionprefix: https://tuscolo2028h1.sunlight.geomys.org
    monitoringprefix: https://tuscolo2028h1.skylight.geomys.org
    secret: /tank/enc/tuscolo2028h1.seed.bin
    cache: /tank/caches/tuscolo2028h1/cache.db
    poolsize: 150
    localdirectory: /tank/logs/tuscolo2028h1/data
    notafterstart: 2028-01-01T00:00:00Z
    notafterlimit: 2028-07-01T00:00:00Z
  - shortname: tuscolo2028h2
    inception: 2026-07-28
    period: 200
    submissionprefix: https://tuscolo2028h2.sunlight.geomys.org
    monitoringprefix: https://tuscolo2028h2.skylight.geomys.org
    secret: /tank/enc/tuscolo2028h2.seed.bin
    cache: /tank/caches/tuscolo2028h2/cache.db
    poolsize: 150
    localdirectory: /tank/logs/tuscolo2028h2/data
    notafterstart: 2028-07-01T00:00:00Z
    notafterlimit: 2029-01-01T00:00:00Z

/etc/sunlight/sunlight-staging.yaml

listen:
  - "185.230.223.195:443"
  - "[2a0c:2f07:c1::c3]:443"
acme:
  hosts:
    - navigli.sunlight.geomys.org
  cache: /var/db/sunlight/autocert-staging/
checkpoints: /tank/shared/checkpoints.db
witness:
  name: witness.navigli.sunlight.geomys.org
  mirrorname: oid/1.3.6.1.4.1.66252.128.0
  submissionprefix: https://witness.navigli.sunlight.geomys.org
  monitoringprefix: https://witness.navigli.skylight.geomys.org
  secret: /tank/enc/navigli-witness.seed.bin
  localdirectory: /tank/witness-staging
  loglists:
    - "https://testing.witness-network.org/log-list.1"
    - "https://staging.witness-network.org/log-list-10qps-4klogs.1"
    - "https://staging.witness-network.org/log-list-100qps-40klogs.1"
    - "https://uptime.geomys.org/witness/log-list"
  mirrorloglists:
    - "https://github.com/geomys/magnolia/raw/refs/heads/main/staging/mirror.log-list.txt"
logs:
  - shortname: navigli2026h2
    inception: 2025-05-03
    period: 200
    submissionprefix: https://navigli2026h2.sunlight.geomys.org
    monitoringprefix: https://navigli2026h2.skylight.geomys.org
    ccadbroots: testing
    extraroots: /etc/sunlight/extra-roots-staging.pem
    secret: /tank/enc/navigli2026h2.seed.bin
    cache: /tank/caches/navigli2026h2/cache.db
    poolsize: 150
    localdirectory: /tank/logs/navigli2026h2/data
    notafterstart: 2026-07-01T00:00:00Z
    notafterlimit: 2027-01-01T00:00:00Z
  - shortname: navigli2027h1
    inception: 2025-06-02
    period: 200
    submissionprefix: https://navigli2027h1.sunlight.geomys.org
    monitoringprefix: https://navigli2027h1.skylight.geomys.org
    ccadbroots: testing
    extraroots: /etc/sunlight/extra-roots-staging.pem
    secret: /tank/enc/navigli2027h1.seed.bin
    cache: /tank/caches/navigli2027h1/cache.db
    poolsize: 150
    localdirectory: /tank/logs/navigli2027h1/data
    notafterstart: 2027-01-01T00:00:00Z
    notafterlimit: 2027-07-01T00:00:00Z
  - shortname: navigli2027h2
    inception: 2025-06-02
    period: 200
    submissionprefix: https://navigli2027h2.sunlight.geomys.org
    monitoringprefix: https://navigli2027h2.skylight.geomys.org
    ccadbroots: testing
    extraroots: /etc/sunlight/extra-roots-staging.pem
    secret: /tank/enc/navigli2027h2.seed.bin
    cache: /tank/caches/navigli2027h2/cache.db
    poolsize: 150
    localdirectory: /tank/logs/navigli2027h2/data
    notafterstart: 2027-07-01T00:00:00Z
    notafterlimit: 2028-01-01T00:00:00Z
  - shortname: navigli2028h1
    inception: 2026-07-28
    period: 200
    submissionprefix: https://navigli2028h1.sunlight.geomys.org
    monitoringprefix: https://navigli2028h1.skylight.geomys.org
    ccadbroots: testing
    extraroots: /etc/sunlight/extra-roots-staging.pem
    secret: /tank/enc/navigli2028h1.seed.bin
    cache: /tank/caches/navigli2028h1/cache.db
    poolsize: 150
    localdirectory: /tank/logs/navigli2028h1/data
    notafterstart: 2028-01-01T00:00:00Z
    notafterlimit: 2028-07-01T00:00:00Z
  - shortname: navigli2028h2
    inception: 2026-07-28
    period: 200
    submissionprefix: https://navigli2028h2.sunlight.geomys.org
    monitoringprefix: https://navigli2028h2.skylight.geomys.org
    ccadbroots: testing
    extraroots: /etc/sunlight/extra-roots-staging.pem
    secret: /tank/enc/navigli2028h2.seed.bin
    cache: /tank/caches/navigli2028h2/cache.db
    poolsize: 150
    localdirectory: /tank/logs/navigli2028h2/data
    notafterstart: 2028-07-01T00:00:00Z
    notafterlimit: 2029-01-01T00:00:00Z

/etc/sunlight/skylight.yaml

listen:
  - "185.230.223.194:443"
  - "[2a0c:2f07:c1::c2]:443"
acme:
  hosts:
    - tuscolo.skylight.geomys.org
  cache: /var/db/sunlight/skylight/
logsjsonprefix: https://tuscolo.skylight.geomys.org
homeredirect: https://tuscolo.sunlight.geomys.org
operatorname: Geomys

witnesses:
  - monitoringprefix: https://witness-golf.skylight.geomys.org
    localdirectory: /tank/witness-golf
    staging: true # TODO: remove after it's stable
  - monitoringprefix: https://witness.navigli.skylight.geomys.org
    localdirectory: /tank/witness-staging
    staging: true

logs:
  - shortname: tuscolo2025h2
    monitoringprefix: https://tuscolo2025h2.skylight.geomys.org
    localdirectory: /tank/logs/tuscolo2025h2/data
  - shortname: tuscolo2026h1
    monitoringprefix: https://tuscolo2026h1.skylight.geomys.org
    localdirectory: /tank/logs/tuscolo2026h1/data
  - shortname: tuscolo2026h2
    monitoringprefix: https://tuscolo2026h2.skylight.geomys.org
    localdirectory: /tank/logs/tuscolo2026h2/data
  - shortname: tuscolo2027h1
    monitoringprefix: https://tuscolo2027h1.skylight.geomys.org
    localdirectory: /tank/logs/tuscolo2027h1/data
  - shortname: tuscolo2027h2
    monitoringprefix: https://tuscolo2027h2.skylight.geomys.org
    localdirectory: /tank/logs/tuscolo2027h2/data
  - shortname: tuscolo2028h1
    monitoringprefix: https://tuscolo2028h1.skylight.geomys.org
    localdirectory: /tank/logs/tuscolo2028h1/data
  - shortname: tuscolo2028h2
    monitoringprefix: https://tuscolo2028h2.skylight.geomys.org
    localdirectory: /tank/logs/tuscolo2028h2/data

  - shortname: navigli2026h1
    monitoringprefix: https://navigli2026h1.skylight.geomys.org
    localdirectory: /tank/logs/navigli2026h1/data
    staging: true
  - shortname: navigli2026h2
    monitoringprefix: https://navigli2026h2.skylight.geomys.org
    localdirectory: /tank/logs/navigli2026h2/data
    staging: true
  - shortname: navigli2027h1
    monitoringprefix: https://navigli2027h1.skylight.geomys.org
    localdirectory: /tank/logs/navigli2027h1/data
    staging: true
  - shortname: navigli2027h2
    monitoringprefix: https://navigli2027h2.skylight.geomys.org
    localdirectory: /tank/logs/navigli2027h2/data
    staging: true
  - shortname: navigli2028h1
    monitoringprefix: https://navigli2028h1.skylight.geomys.org
    localdirectory: /tank/logs/navigli2028h1/data
    staging: true
  - shortname: navigli2028h2
    monitoringprefix: https://navigli2028h2.skylight.geomys.org
    localdirectory: /tank/logs/navigli2028h2/data
    staging: true

/usr/local/bin/debug

#!/bin/bash
set -euo pipefail

unit_flag="skylight"

display_help() {
    echo "Usage: debug [-u unit] {useragents|useragents-bytes|ips|ips-bytes|sourcelimit|keylog={on|off}|logs={on|off}|port}"
}

while getopts "u:h" opt; do
    case ${opt} in
        u )
            unit_flag=$OPTARG
            ;;
        h )
            display_help >&2
            exit 0
            ;;
        \? )
            echo "Invalid option: -$OPTARG" >&2
            display_help >&2
            exit 1
            ;;
        : )
            echo "Option -$OPTARG requires an argument" >&2
            display_help >&2
            exit 1
            ;;
    esac
done

shift $((OPTIND - 1))

if [ "$#" -ne 1 ]; then
    echo "Exactly one positional argument is required" >&2
    display_help >&2
    exit 1
fi

PID=$(systemctl show "$unit_flag" --property MainPID | cut -d'=' -f2)
if [ -z "$PID" ] || [ "$PID" = "0" ]; then
    echo "Unit $unit_flag is not running" >&2
    exit 1
fi
PORT=$(ss -tulnp | grep "pid=$PID," | awk '{print $5}' | grep 127.0.0.1) || true
if [ -z "$PORT" ]; then
    echo "No port found for unit $unit_flag" >&2
    exit 1
fi

case $1 in
    useragents )
        curl -s "$PORT/debug/heavyhitter/useragents"
        ;;
    useragents-bytes )
        curl -s "$PORT/debug/heavyhitter/useragents-bytes"
        ;;
    ips-bytes )
        curl -s "$PORT/debug/heavyhitter/ips-bytes"
        ;;
    ips )
        curl -s "$PORT/debug/heavyhitter/ips"
        ;;
    sourcelimit )
        curl -s "$PORT/debug/sourcelimit"
        ;;
    keylog=on )
        curl -s -X POST "$PORT/debug/keylog/on"
        ;;
    keylog=off )
        curl -s -X POST "$PORT/debug/keylog/off"
        ;;
    logs=on )
        curl -s -X POST "$PORT/debug/logs/on"
        ;;
    logs=off )
        curl -s -X POST "$PORT/debug/logs/off"
        ;;
    port )
    	echo "$PORT"
    	;;
    * )
        echo "Invalid argument: $1" >&2
        display_help >&2
        exit 1
        ;;
esac

/etc/systemd/system/sunlight.service

[Unit]
Description=Sunlight Certificate Transparency Log
After=network-online.target tank-enc.mount
Wants=network-online.target
StartLimitIntervalSec=0

[Service]
ExecStart=/usr/local/bin/sunlight -c /etc/sunlight/sunlight.yaml
ExecReload=kill -HUP $MAINPID
StandardOutput=append:/var/log/sunlight.jsonl
StandardError=journal
Restart=always
RestartSteps=20
RestartSec=1ms
RestartMaxDelaySec=10s
LimitMEMLOCK=256M
MemorySwapMax=0

[Install]
WantedBy=tank-enc.mount

/etc/systemd/system/sunlight-staging.service

[Unit]
Description=Sunlight Certificate Transparency Log (staging)
After=network-online.target tank-enc.mount
Wants=network-online.target
StartLimitIntervalSec=0

[Service]
ExecStart=/usr/local/bin/sunlight-staging -c /etc/sunlight/sunlight-staging.yaml
ExecReload=kill -HUP $MAINPID
StandardOutput=append:/var/log/sunlight-staging.jsonl
StandardError=journal
Restart=always
RestartSteps=20
RestartSec=1ms
RestartMaxDelaySec=10s
MemoryMax=8G
MemorySwapMax=0
RuntimeMaxSec=1d
LimitMEMLOCK=256M

[Install]
WantedBy=tank-enc.mount

/etc/systemd/system/skylight.service

[Unit]
Description=Sunlight Certificate Transparency Log (read path)
After=network-online.target
Wants=network-online.target
StartLimitIntervalSec=0

[Service]
ExecStart=/usr/local/bin/skylight -c /etc/sunlight/skylight.yaml
StandardOutput=append:/var/log/skylight.jsonl
StandardError=journal
Restart=always
RestartSteps=20
RestartSec=1ms
RestartMaxDelaySec=10s
LimitMEMLOCK=256M
MemorySwapMax=0

[Install]
WantedBy=multi-user.target

/etc/systemd/system/partial-aftersun.service

[Unit]
Description=Clean up partial tiles

[Service]
Type=oneshot
ExecStart=/usr/local/bin/partial-aftersun -c /etc/sunlight/sunlight.yaml -metrics /var/lib/prometheus/node-exporter/partial-aftersun.prom
ExecStartPost=/usr/bin/curl --retry 3 --retry-delay 1 -m 15 https://sm.hetrixtools.net/hb/?s=a4f010ea1bd8d93598fc96f94000190f
StandardOutput=append:/var/log/partial-aftersun.jsonl
StandardError=journal

/etc/systemd/system/partial-aftersun.timer

[Unit]
Description=Periodically run partial tiles cleanup while Sunlight is running
RefuseManualStart=yes
PartOf=sunlight.service

[Timer]
OnActiveSec=5s
OnUnitActiveSec=5m

[Install]
WantedBy=sunlight.service

/etc/systemd/system/partial-aftersun-staging.service

[Unit]
Description=Clean up partial tiles (staging)

[Service]
Type=oneshot
ExecStart=/usr/local/bin/partial-aftersun -c /etc/sunlight/sunlight-staging.yaml -metrics /var/lib/prometheus/node-exporter/partial-aftersun-staging.prom
ExecStartPost=/usr/bin/curl --retry 3 --retry-delay 1 -m 15 https://sm.hetrixtools.net/hb/?s=590ce0eb9954e649e6f2be05a4c651bd
StandardOutput=append:/var/log/partial-aftersun-staging.jsonl
StandardError=journal

/etc/systemd/system/partial-aftersun-staging.timer

[Unit]
Description=Periodically run partial tiles cleanup while Sunlight is running (staging)
RefuseManualStart=yes
PartOf=sunlight-staging.service

[Timer]
OnActiveSec=5s
OnUnitActiveSec=5m

[Install]
WantedBy=sunlight-staging.service

/etc/systemd/system/public-config.service

[Unit]
Description=https://github.com/FiloSottile/mostly-harmless/tree/main/public-config
After=network-online.target
Wants=network-online.target
Before=caddy.service
StartLimitIntervalSec=0

[Service]
ExecStart=/usr/local/bin/public-config -name tuscolo
DynamicUser=true
Restart=always
RestartSteps=20
RestartSec=1ms
RestartMaxDelaySec=10s

[Install]
WantedBy=multi-user.target

/etc/logrotate.d/jsonl

/var/log/*.jsonl {
    daily
    rotate 10
    copytruncate
    compress
    delaycompress
    notifempty
    missingok
}

/etc/caddy/Caddyfile

skylight.geomys.org:443 {
	bind 185.230.223.196 [2a0c:2f07:c1::c4]
	redir https://tuscolo.skylight.geomys.org{uri} permanent
}

config.tuscolo.sunlight.geomys.org:443 {
	bind 185.230.223.196 [2a0c:2f07:c1::c4]
	reverse_proxy localhost:8080
}

config.sunlight.geomys.org:443 {
	bind 185.230.223.196 [2a0c:2f07:c1::c4]
	redir https://config.tuscolo.sunlight.geomys.org{uri} permanent
}

stats.tuscolo.sunlight.geomys.org:443 {
	bind 185.230.223.196 [2a0c:2f07:c1::c4]
	root * /var/www/heliograph
	file_server
	encode zstd gzip
	header Cache-Control "public, max-age=60"
}

stats.sunlight.geomys.org:443 {
	bind 185.230.223.196 [2a0c:2f07:c1::c4]
	redir https://stats.tuscolo.sunlight.geomys.org{uri} permanent
}

stats.trastevere.sunlight.geomys.org:443 {
	bind 185.230.223.196 [2a0c:2f07:c1::c4]
	root * /var/www/heliograph-trastevere
	file_server
	encode zstd gzip
	header Cache-Control "public, max-age=60"
}

keyserver.geomys.org:443 {
	bind 185.230.223.196 [2a0c:2f07:c1::c4]
	reverse_proxy localhost:13889
}

pkg.geomys.dev:443 {
	bind 185.230.223.196 [2a0c:2f07:c1::c4]
	reverse_proxy localhost:8081
}

plc.geomys.org:443 {
	bind 185.230.223.196 [2a0c:2f07:c1::c4]
	reverse_proxy localhost:6780
}

/etc/systemd/system/caddy.service.d/override.conf

[Unit]
StartLimitIntervalSec=0

[Service]
Restart=always
RestartSteps=20
RestartSec=1ms
RestartMaxDelaySec=10s

/usr/local/bin/zfs-textfile

#!/bin/sh
set -eu
out=/var/lib/prometheus/node-exporter/zfs.prom
tmp=$(mktemp "$out.XXXXXX")
trap 'rm -f "$tmp"' EXIT
{
  echo '# TYPE zfs_dataset_referenced_bytes gauge'
  echo '# TYPE zfs_dataset_logicalreferenced_bytes gauge'
  echo '# TYPE zfs_dataset_available_bytes gauge'
  zfs list -Hp -r -o name,referenced,logicalreferenced,available tank/logs tank/caches \
    | awk '{
        printf "zfs_dataset_referenced_bytes{dataset=\"%s\"} %s\n",        $1, $2
        printf "zfs_dataset_logicalreferenced_bytes{dataset=\"%s\"} %s\n", $1, $3
        printf "zfs_dataset_available_bytes{dataset=\"%s\"} %s\n",         $1, $4
      }'
  echo '# HELP zfs_txg_last_sync_seconds Sync duration of the most recent committed transaction group.'
  echo '# TYPE zfs_txg_last_sync_seconds gauge'
  echo '# HELP zfs_txg_max_sync_seconds Maximum txg sync duration over the kernel history ring.'
  echo '# TYPE zfs_txg_max_sync_seconds gauge'
  echo '# HELP zfs_txg_last_dirty_bytes Dirty data in the most recent committed transaction group.'
  echo '# TYPE zfs_txg_last_dirty_bytes gauge'
  for txgs in /proc/spl/kstat/zfs/*/txgs; do
      pool=$(basename "$(dirname "$txgs")")
      awk -v pool="$pool" '
          $3 == "C" { last_sync = $12; if ($12 > max_sync) max_sync = $12; last_dirty = $4 }
          END {
              if (last_sync != "") {
                  printf "zfs_txg_last_sync_seconds{pool=\"%s\"} %.6f\n", pool, last_sync / 1e9
                  printf "zfs_txg_max_sync_seconds{pool=\"%s\"} %.6f\n", pool, max_sync / 1e9
                  printf "zfs_txg_last_dirty_bytes{pool=\"%s\"} %d\n", pool, last_dirty
              }
          }' "$txgs"
  done
  # Slab memory (dentries, inodes) is charged to the cgroup that created it,
  # and moves to system.slice itself when a transient unit's cgroup goes away,
  # where the ZFS shrinker can't see it. Units can vanish between the glob and
  # the read, hence the tolerant awk.
  echo '# HELP cgroup_memory_stat_bytes Selected memory.stat entries of a cgroup, including its descendants.'
  echo '# TYPE cgroup_memory_stat_bytes gauge'
  for d in /sys/fs/cgroup/system.slice /sys/fs/cgroup/system.slice/*/; do
      d=${d%/}
      awk -v cg="${d#/sys/fs/cgroup/}" '
          $1 ~ /^(anon|file|slab_reclaimable|slab_unreclaimable)$/ {
              printf "cgroup_memory_stat_bytes{cgroup=\"%s\",stat=\"%s\"} %s\n", cg, $1, $2
          }' "$d/memory.stat" 2>/dev/null || true
  done
} > "$tmp"
chmod 0644 "$tmp"
mv "$tmp" "$out"

/etc/systemd/system/zfs-textfile.service

[Unit]
Description=Write ZFS dataset metrics for node_exporter textfile collector

[Service]
Type=oneshot
ExecStart=/usr/local/bin/zfs-textfile
TimeoutStartSec=30s

/etc/systemd/system/zfs-textfile.timer

[Unit]
Description=Run zfs-textfile every minute

[Timer]
OnBootSec=30s
OnUnitActiveSec=60s
AccuracySec=1s

[Install]
WantedBy=timers.target

/etc/default/prometheus-node-exporter

# Set the command-line arguments to pass to the server.
# Due to shell escaping, to pass backslashes for regexes, you need to double
# them (\\d for \d). If running under systemd, you need to double them again
# (\\\\d to mean \d), and escape newlines too.
ARGS="--collector.vmstat.fields=^(oom_kill|pgpg|pswp|pg.*fault|pgsteal_|pgscan_|slabs_scanned|kswapd_|pginodesteal|allocstall_|drop_pagecache|drop_slab).* --collector.slabinfo --collector.sysctl --collector.sysctl.include=vm.min_free_kbytes --collector.sysctl.include=vm.watermark_scale_factor --collector.sysctl.include=vm.swappiness --collector.sysctl.include=vm.vfs_cache_pressure --collector.sysctl.include=fs.dentry-state:nr_dentry,nr_unused,age_limit,want_pages,nr_negative,dummy --collector.sysctl.include=fs.inode-nr:nr_inodes,nr_free_inodes"

/etc/systemd/system/prometheus-node-exporter.service.d/slabinfo.conf

[Service]
ExecStartPre=+/bin/chgrp prometheus /proc/slabinfo
ExecStartPre=+/bin/chmod 0440 /proc/slabinfo

/etc/systemd/system/memcg-reclaim@.service

# Drops the reclaimable memory charged to a service's cgroup. This bounds the
# dentries, and through them the ZFS znodes, dnodes and ARC metadata.
#
# It works around a combination of OpenZFS (2.3) and cgroup behaviors that can
# cause reads to stall for ~2 ms.
#
# 1. Under pressure, ARC allocations are throttled.
#
#    Every ZFS read and write allocates an ARC buffer (arc_get_data_impl),
#    whether or not the data is worth caching. If the ARC is more than a small
#    margin above its target size c, each allocation waits for an eviction pass
#    (arc_wait_for_eviction). If the pass finds nothing evictable, the waiter is
#    released anyway, having paid for a full futile scan.
#
#    The ARC is not needed for the log to perform (cache.db reads cost 0.25 ms
#    with the ARC nearly empty of data) but there is no way to opt out of the
#    wait. (direct=always would bypass it, but SQLite's page buffers are not
#    page-aligned, so it would never engage.)
#
# 2. The target ignores unevictable memory.
#
#    Under memory pressure, arc_reduce_target_size lowers c with no floor other
#    than c_min (4 GiB here), regardless of how much of the ARC is pinned by
#    open dnodes and their metadata. Once the pinned set alone exceeds c, the
#    ARC is permanently over target and every read is throttled as in (1).
#
# 3. A lot of memory is unevictable for ZFS due to cgroups.
#
#    Dentries are charged to the memory cgroup of the process that looked them
#    up (e.g. skylight's for the tiles it serves). Each dentry pins a znode,
#    which pins a dnode and ~5 KiB of ARC metadata (dnode block, dbuf, bonus
#    buffer). The only way ZFS has to release them is arc_prune -> zfs_prune ->
#    super_cache_scan, which scans the root memory cgroup only. So for ZFS this
#    memory is unevictable.
#
# 4. Memory pressure is handled by ZFS, never by the kernel.
#
#    The kernel could evict the dentries charged to all cgroups, but ZFS starts
#    shrinking its target when free plus inactive file memory drops under
#    zfs_arc_sys_free (4.25 GiB here), while kswapd only wakes under the zone
#    low watermark (~200 MiB here). The kernel's reclaim rarely runs, and when
#    it does it stops as soon as its watermark is met, a few hundred MiB later.
#
# The result is that ZFS is the only one trying to reclaim memory (4), and while
# it can't succeed (3) it keeps lowering its target (2), which throttles every
# read (1).
#
# Dropping the dentries through memory.reclaim leaves the znodes on the root
# inode LRU, where ZFS pruning (and kswapd) can dispose of them, turning the
# pinned metadata back into ordinary evictable ARC.

[Unit]
Description=Reclaim page cache and slab charged to %i.service

[Service]
Type=oneshot
ExecStart=/usr/local/bin/memcg-reclaim %i
TimeoutStartSec=5m

/etc/systemd/system/memcg-reclaim@skylight.timer

# Hourly bound pinned ZFS metadata (dentries, and through them the ZFS znodes,
# dnodes and ARC metadata) that skylight accumulates by serving tiles at roughly
# an hour of tile lookups (~0.4M dentries, ~2 GiB of ARC).
#
# The cost is that the next lookup of each tile re-reads its dnode.
#
# See memcg-reclaim@.service.

[Unit]
Description=Hourly reclaim of memory charged to skylight.service
PartOf=skylight.service

[Timer]
OnActiveSec=1h
OnUnitActiveSec=1h
AccuracySec=1m

[Install]
WantedBy=skylight.service

/usr/local/bin/memcg-reclaim

#!/bin/sh
# Usage: memcg-reclaim UNIT
#
# Drops the reclaimable memory (page cache, dentries, inodes) charged to a
# systemd service's memory cgroup. See memcg-reclaim@.service.
set -eu
cg=/sys/fs/cgroup/system.slice/$1.service

mstat() { awk -v k="$1" '$1 == k { print $2 }' "$cg/memory.stat"; }
before=$(mstat slab_reclaimable)

# Ask for every file page and reclaimable slab byte charged to the cgroup.
# The kernel reclaims what it can and then reports EAGAIN if it could not
# reach the requested amount (partially freed slabs are not page-freeable),
# so a failed write is the normal outcome and the effect is checked below.
# swappiness=0 keeps the pass away from anonymous memory, which could not
# be swapped anyway (MemorySwapMax=0).
echo "$(( $(mstat file) + before )) swappiness=0" > "$cg/memory.reclaim" 2>/dev/null || true

after=$(mstat slab_reclaimable)
echo "$1: slab_reclaimable $((before >> 20)) MiB -> $((after >> 20)) MiB;" \
    "$(awk '{ print $1 }' /proc/sys/fs/dentry-state) dentries system-wide"

# Fail loudly (node_systemd_unit_state) if the reclaim had no effect, e.g.
# because the objects are actually in use or the cgroup layout changed.
[ "$after" -lt $((256 << 20)) ]

/etc/sunlight/sunlight-golf.yaml

listen:
  - "185.230.223.197:443"
  - "[2a0c:2f07:c1::c5]:443"
acme:
  cache: /var/db/sunlight/autocert-golf/
checkpoints: /tank/shared/checkpoints.db
witness:
  name: witness-golf.sunlight.geomys.org
  monitoringprefix: https://witness-golf.skylight.geomys.org
  secret: /tank/enc/golf-witness.seed.bin
  localdirectory: /tank/witness-golf
  loglists:
    - "https://geomys-loglistfilter.exe.xyz/filter?url=https%3A%2F%2Fwitnessplz.transparency.goog%2Flog-list-10qps-10klogs.txt"
    - "https://uptime.geomys.org/witness/log-list"

/etc/systemd/system/sunlight-golf.service

[Unit]
Description=Sunlight Witness (golf)
After=network-online.target tank-enc.mount
Wants=network-online.target
StartLimitIntervalSec=0

[Service]
ExecStart=/usr/local/bin/sunlight-golf -c /etc/sunlight/sunlight-golf.yaml
ExecReload=kill -HUP $MAINPID
StandardOutput=append:/var/log/sunlight-golf.jsonl
StandardError=journal
Restart=always
RestartSteps=20
RestartSec=1ms
RestartMaxDelaySec=10s
LimitMEMLOCK=256M
MemorySwapMax=0

[Install]
WantedBy=tank-enc.mount

/etc/systemd/system/heliograph-dashboard.service

[Unit]
Description=Generate heliograph CT log dashboard
After=prometheus.service

[Service]
Type=oneshot
ExecStart=/usr/local/bin/heliograph-dashboard \
    -prometheus http://localhost:9090 \
    -title "Tuscolo CT log" \
    -log-name tuscolo \
    -o /var/www/heliograph/index.html
TimeoutStartSec=30s
ExecStartPost=/usr/bin/curl --retry 3 --retry-delay 1 -m 15 https://sm.hetrixtools.net/hb/?s=32f9e49e38582dfcf7c17c09dc3df085

ProtectSystem=strict
ReadWritePaths=/var/www/heliograph
ProtectHome=yes
NoNewPrivileges=yes
PrivateTmp=yes
PrivateDevices=yes
RestrictAddressFamilies=AF_INET AF_INET6

/etc/systemd/system/heliograph-dashboard.timer

[Unit]
Description=Refresh heliograph dashboard every minute

[Timer]
OnBootSec=20s
OnUnitActiveSec=5m
AccuracySec=5s

[Install]
WantedBy=timers.target

/etc/systemd/system/heliograph-dashboard-trastevere.service

[Unit]
Description=Generate heliograph CT log dashboard (Trastevere)
After=prometheus.service

[Service]
Type=oneshot
ExecStart=/usr/local/bin/heliograph-dashboard \
    -prometheus http://localhost:9090 \
    -title "Trastevere CT log" \
    -log-name trastevere -node-job trastevere-node \
    -network-device 'ens2f.*' -skylight-job trastevere-skylight \
    -o /var/www/heliograph-trastevere/index.html
TimeoutStartSec=30s
ExecStartPost=/usr/bin/curl --retry 3 --retry-delay 1 -m 15 https://sm.hetrixtools.net/hb/?s=3ad11212eb47a3a999add43e6d43aa62

ProtectSystem=strict
ReadWritePaths=/var/www/heliograph-trastevere
ProtectHome=yes
NoNewPrivileges=yes
PrivateTmp=yes
PrivateDevices=yes
RestrictAddressFamilies=AF_INET AF_INET6

/etc/systemd/system/heliograph-dashboard-trastevere.timer

[Unit]
Description=Refresh heliograph dashboard every minute (Trastevere)

[Timer]
OnBootSec=20s
OnUnitActiveSec=5m
AccuracySec=5s

[Install]
WantedBy=timers.target

/etc/systemd/system/age-keyserver.service

[Unit]
Description=Email-authenticated age public key server
After=network-online.target tank-enc.mount
Wants=network-online.target
StartLimitIntervalSec=0

[Service]
Type=exec
ExecStart=/usr/local/bin/age-keyserver -listen localhost:13889 -db /tank/keyserver/keyserver.sqlite3 -logdir /tank/keyserver/tlog
EnvironmentFile=/tank/enc/age-keyserver.env
StandardOutput=append:/var/log/age-keyserver.jsonl
StandardError=journal
Restart=always
RestartSteps=20
RestartSec=1ms
RestartMaxDelaySec=10s

[Install]
WantedBy=tank-enc.mount

/etc/systemd/system/pkg-geomys-dev.service

[Unit]
After=network-online.target
Wants=network-online.target
Before=caddy.service
StartLimitIntervalSec=0

[Service]
ExecStart=/usr/local/bin/pkg.geomys.dev -addr localhost:8081
DynamicUser=true
Restart=always
RestartSteps=20
RestartSec=1ms
RestartMaxDelaySec=10s

[Install]
WantedBy=multi-user.target

/etc/systemd/system/plc-replica.service

[Unit]
Description=PLC Directory Replica
Documentation=https://github.com/did-method-plc/go-didplc/tree/main/cmd/plc-replica
After=network-online.target
Wants=network-online.target
Before=caddy.service
StartLimitIntervalSec=0

[Service]
ExecStart=/usr/local/bin/plc-replica \
    --db-url "sqlite:///tank/plc/replica.db?mode=rwc&cache=shared&_journal_mode=WAL" \
    --bind localhost:6780 \
    --metrics-addr localhost:9464 \
    --num-workers 16 \
    --log-json
StandardOutput=append:/var/log/plc-replica.jsonl
StandardError=journal
Restart=always
RestartSteps=20
RestartSec=1ms
RestartMaxDelaySec=10s

MemoryMax=24G
MemorySwapMax=0
ProtectSystem=strict
ReadWritePaths=/tank/plc
ProtectHome=yes
NoNewPrivileges=yes
PrivateTmp=yes

[Install]
WantedBy=multi-user.target

/etc/prometheus/prometheus.yml

global:
  scrape_interval: 15s
scrape_configs:
  - job_name: prometheus
    static_configs:
      - targets:
        - localhost:9090
  - job_name: node
    static_configs:
      - targets:
        - localhost:9100
  - job_name: plc
    static_configs:
      - targets:
        - localhost:9464
  - job_name: tuscolo
    scheme: https
    static_configs:
      - targets:
        - tuscolo.sunlight.geomys.org
  - job_name: navigli
    scheme: https
    static_configs:
      - targets:
        - navigli.sunlight.geomys.org
  - job_name: skylight
    scheme: https
    static_configs:
      - targets:
        - tuscolo.skylight.geomys.org
  - job_name: golf
    scheme: https
    static_configs:
      - targets:
        - witness-golf.sunlight.geomys.org
  - job_name: trastevere
    scheme: https
    static_configs:
      - targets:
        - trastevere.sunlight.geomys.org
  - job_name: loreto
    scheme: https
    static_configs:
      - targets:
        - loreto.sunlight.geomys.org
  - job_name: trastevere-skylight
    scheme: https
    static_configs:
      - targets:
        - trastevere.skylight.geomys.org
  - job_name: trastevere-node
    static_configs:
      - targets:
        - 213.171.190.242:9100
  - job_name: twig
    scheme: https
    static_configs:
      - targets:
        - log.twig.ct.letsencrypt.org
  - job_name: sycamore
    scheme: https
    static_configs:
      - targets:
        - log.sycamore.ct.letsencrypt.org
  - job_name: willow
    scheme: https
    static_configs:
      - targets:
        - log.willow.ct.letsencrypt.org
  - job_name: gouda
    scheme: https
    static_configs:
      - targets:
        - gouda2027h2.log.ct.ipng.ch
  - job_name: rennet
    scheme: https
    static_configs:
      - targets:
        - rennet2027h2.log.ct.ipng.ch
  - job_name: ctuptime
    scheme: https
    metrics_path: /ctuptime/metrics
    static_configs:
      - targets: 
        - mcpherrinm.github.io
  - job_name: sctdelay
    scheme: https
    metrics_path: /chrome-sctauditing-delay/metrics.txt
    static_configs:
      - targets:
        - agwa-bot.github.io
  - job_name: sslmate
    scheme: https
    metrics_path: /ct_logs.prom
    static_configs:
      - targets:
        - feeds.sslmate.com
rule_files:
  - /etc/prometheus/rules.yml

/etc/prometheus/rules.yml

# Prometheus recording rules required by heliograph-dashboard.
#
# The add-chain counter carries issuer and root labels, so it has thousands of
# series per log. Aggregating it on the fly over the chart window is too slow,
# and gets slower every time a new CA starts submitting. These rules keep the
# aggregation incremental: each evaluation only touches the last five minutes.
#
# Install as /etc/prometheus/rules.yml and reference it from prometheus.yml:
#
#   rule_files:
#     - /etc/prometheus/rules.yml
#
# To backfill history for a freshly added rule (this Prometheus runs with
# --storage.tsdb.allow-overlapping-blocks):
#
#   promtool tsdb create-blocks-from rules --url=http://localhost:9090 \
#     --start=$(date -d '-7 days' +%s) --end=$(date +%s) \
#     --output-dir=/tmp/backfill /etc/prometheus/rules.yml
#   mv /tmp/backfill/* /tank/prometheus/
groups:
  - name: heliograph
    interval: 30s
    rules:
      - record: log:sunlight_addchain_requests:rate5m
        expr: sum by (job, log) (rate(sunlight_addchain_requests_total{error=""}[5m]))
      - record: low_priority:sunlight_addchain_requests:rate5m
        expr: sum by (job, low_priority) (rate(sunlight_addchain_requests_total{error=""}[5m]))
      # Includes failed requests, unlike the two above. The error label is a
      # bounded set of categories, so this stays at a few dozen series per log.
      - record: source_error:sunlight_addchain_requests:rate5m
        expr: sum by (job, log, source, error) (rate(sunlight_addchain_requests_total[5m]))