This is the live public configuration for the Geomys Trastevere CT Log Server, a Sunlight instance.
See also the public playbooks.
homehtml: /etc/sunlight/home.html
listen:
- "213.171.190.243:443"
- "[2001:4b7e:a002::c1]:443"
acme:
hosts:
- trastevere.sunlight.geomys.org
cache: /var/db/sunlight/autocert/
checkpoints: /tank/shared/checkpoints.db
logs:
- shortname: trastevere2026h2
inception: 2026-07-27
period: 200
submissionprefix: https://trastevere2026h2.sunlight.geomys.org
monitoringprefix: https://trastevere2026h2.skylight.geomys.org
secret: /tank/enc/trastevere2026h2.seed.bin
cache: /tank/caches/trastevere2026h2/cache.db
poolsize: 20
localdirectory: /tank/logs/trastevere2026h2/data
notafterstart: 2026-07-01T00:00:00Z
notafterlimit: 2027-01-01T00:00:00Z
- shortname: trastevere2027h1
inception: 2026-07-27
period: 200
submissionprefix: https://trastevere2027h1.sunlight.geomys.org
monitoringprefix: https://trastevere2027h1.skylight.geomys.org
secret: /tank/enc/trastevere2027h1.seed.bin
cache: /tank/caches/trastevere2027h1/cache.db
poolsize: 20
localdirectory: /tank/logs/trastevere2027h1/data
notafterstart: 2027-01-01T00:00:00Z
notafterlimit: 2027-07-01T00:00:00Z
- shortname: trastevere2027h2
inception: 2026-07-27
period: 200
submissionprefix: https://trastevere2027h2.sunlight.geomys.org
monitoringprefix: https://trastevere2027h2.skylight.geomys.org
secret: /tank/enc/trastevere2027h2.seed.bin
cache: /tank/caches/trastevere2027h2/cache.db
poolsize: 20
localdirectory: /tank/logs/trastevere2027h2/data
notafterstart: 2027-07-01T00:00:00Z
notafterlimit: 2028-01-01T00:00:00Z
- shortname: trastevere2028h1
inception: 2026-07-27
period: 200
submissionprefix: https://trastevere2028h1.sunlight.geomys.org
monitoringprefix: https://trastevere2028h1.skylight.geomys.org
secret: /tank/enc/trastevere2028h1.seed.bin
cache: /tank/caches/trastevere2028h1/cache.db
poolsize: 20
localdirectory: /tank/logs/trastevere2028h1/data
notafterstart: 2028-01-01T00:00:00Z
notafterlimit: 2028-07-01T00:00:00Z
- shortname: trastevere2028h2
inception: 2026-07-27
period: 200
submissionprefix: https://trastevere2028h2.sunlight.geomys.org
monitoringprefix: https://trastevere2028h2.skylight.geomys.org
secret: /tank/enc/trastevere2028h2.seed.bin
cache: /tank/caches/trastevere2028h2/cache.db
poolsize: 20
localdirectory: /tank/logs/trastevere2028h2/data
notafterstart: 2028-07-01T00:00:00Z
notafterlimit: 2029-01-01T00:00:00Z
listen:
- "213.171.190.245:443"
- "[2001:4b7e:a002::c3]:443"
acme:
hosts:
- loreto.sunlight.geomys.org
cache: /var/db/sunlight/autocert-staging/
checkpoints: /tank/shared/checkpoints.db
logs:
- shortname: loreto2026h2
inception: 2026-07-27
period: 200
submissionprefix: https://loreto2026h2.sunlight.geomys.org
monitoringprefix: https://loreto2026h2.skylight.geomys.org
ccadbroots: testing
extraroots: /etc/sunlight/extra-roots-staging.pem
secret: /tank/enc/loreto2026h2.seed.bin
cache: /tank/caches/loreto2026h2/cache.db
poolsize: 150
localdirectory: /tank/logs/loreto2026h2/data
notafterstart: 2026-07-01T00:00:00Z
notafterlimit: 2027-01-01T00:00:00Z
- shortname: loreto2027h1
inception: 2026-07-27
period: 200
submissionprefix: https://loreto2027h1.sunlight.geomys.org
monitoringprefix: https://loreto2027h1.skylight.geomys.org
ccadbroots: testing
extraroots: /etc/sunlight/extra-roots-staging.pem
secret: /tank/enc/loreto2027h1.seed.bin
cache: /tank/caches/loreto2027h1/cache.db
poolsize: 150
localdirectory: /tank/logs/loreto2027h1/data
notafterstart: 2027-01-01T00:00:00Z
notafterlimit: 2027-07-01T00:00:00Z
- shortname: loreto2027h2
inception: 2026-07-27
period: 200
submissionprefix: https://loreto2027h2.sunlight.geomys.org
monitoringprefix: https://loreto2027h2.skylight.geomys.org
ccadbroots: testing
extraroots: /etc/sunlight/extra-roots-staging.pem
secret: /tank/enc/loreto2027h2.seed.bin
cache: /tank/caches/loreto2027h2/cache.db
poolsize: 150
localdirectory: /tank/logs/loreto2027h2/data
notafterstart: 2027-07-01T00:00:00Z
notafterlimit: 2028-01-01T00:00:00Z
- shortname: loreto2028h1
inception: 2026-07-27
period: 200
submissionprefix: https://loreto2028h1.sunlight.geomys.org
monitoringprefix: https://loreto2028h1.skylight.geomys.org
ccadbroots: testing
extraroots: /etc/sunlight/extra-roots-staging.pem
secret: /tank/enc/loreto2028h1.seed.bin
cache: /tank/caches/loreto2028h1/cache.db
poolsize: 150
localdirectory: /tank/logs/loreto2028h1/data
notafterstart: 2028-01-01T00:00:00Z
notafterlimit: 2028-07-01T00:00:00Z
- shortname: loreto2028h2
inception: 2026-07-27
period: 200
submissionprefix: https://loreto2028h2.sunlight.geomys.org
monitoringprefix: https://loreto2028h2.skylight.geomys.org
ccadbroots: testing
extraroots: /etc/sunlight/extra-roots-staging.pem
secret: /tank/enc/loreto2028h2.seed.bin
cache: /tank/caches/loreto2028h2/cache.db
poolsize: 150
localdirectory: /tank/logs/loreto2028h2/data
notafterstart: 2028-07-01T00:00:00Z
notafterlimit: 2029-01-01T00:00:00Z
listen:
- "213.171.190.244:443"
- "[2001:4b7e:a002::c2]:443"
acme:
hosts:
- trastevere.skylight.geomys.org
cache: /var/db/sunlight/skylight/
logsjsonprefix: https://trastevere.skylight.geomys.org
homeredirect: https://trastevere.sunlight.geomys.org
operatorname: Geomys
logs:
- shortname: trastevere2026h2
monitoringprefix: https://trastevere2026h2.skylight.geomys.org
localdirectory: /tank/logs/trastevere2026h2/data
- shortname: trastevere2027h1
monitoringprefix: https://trastevere2027h1.skylight.geomys.org
localdirectory: /tank/logs/trastevere2027h1/data
- shortname: trastevere2027h2
monitoringprefix: https://trastevere2027h2.skylight.geomys.org
localdirectory: /tank/logs/trastevere2027h2/data
- shortname: trastevere2028h1
monitoringprefix: https://trastevere2028h1.skylight.geomys.org
localdirectory: /tank/logs/trastevere2028h1/data
- shortname: trastevere2028h2
monitoringprefix: https://trastevere2028h2.skylight.geomys.org
localdirectory: /tank/logs/trastevere2028h2/data
- shortname: loreto2026h2
monitoringprefix: https://loreto2026h2.skylight.geomys.org
localdirectory: /tank/logs/loreto2026h2/data
staging: true
- shortname: loreto2027h1
monitoringprefix: https://loreto2027h1.skylight.geomys.org
localdirectory: /tank/logs/loreto2027h1/data
staging: true
- shortname: loreto2027h2
monitoringprefix: https://loreto2027h2.skylight.geomys.org
localdirectory: /tank/logs/loreto2027h2/data
staging: true
- shortname: loreto2028h1
monitoringprefix: https://loreto2028h1.skylight.geomys.org
localdirectory: /tank/logs/loreto2028h1/data
staging: true
- shortname: loreto2028h2
monitoringprefix: https://loreto2028h2.skylight.geomys.org
localdirectory: /tank/logs/loreto2028h2/data
staging: true
#!/bin/bash
set -euo pipefail
unit_flag="skylight"
display_help() {
echo "Usage: debug [-u unit] {useragents|useragents-bytes|ips|ips-bytes|sourcelimit|keylog={on|off}|logs={on|off}|port}"
}
while getopts "u:h" opt; do
case ${opt} in
u )
unit_flag=$OPTARG
;;
h )
display_help >&2
exit 0
;;
\? )
echo "Invalid option: -$OPTARG" >&2
display_help >&2
exit 1
;;
: )
echo "Option -$OPTARG requires an argument" >&2
display_help >&2
exit 1
;;
esac
done
shift $((OPTIND - 1))
if [ "$#" -ne 1 ]; then
echo "Exactly one positional argument is required" >&2
display_help >&2
exit 1
fi
PID=$(systemctl show "$unit_flag" --property MainPID | cut -d'=' -f2)
if [ -z "$PID" ] || [ "$PID" = "0" ]; then
echo "Unit $unit_flag is not running" >&2
exit 1
fi
PORT=$(ss -tulnp | grep "pid=$PID," | awk '{print $5}' | grep 127.0.0.1) || true
if [ -z "$PORT" ]; then
echo "No port found for unit $unit_flag" >&2
exit 1
fi
case $1 in
useragents )
curl -s "$PORT/debug/heavyhitter/useragents"
;;
useragents-bytes )
curl -s "$PORT/debug/heavyhitter/useragents-bytes"
;;
ips-bytes )
curl -s "$PORT/debug/heavyhitter/ips-bytes"
;;
ips )
curl -s "$PORT/debug/heavyhitter/ips"
;;
sourcelimit )
curl -s "$PORT/debug/sourcelimit"
;;
keylog=on )
curl -s -X POST "$PORT/debug/keylog/on"
;;
keylog=off )
curl -s -X POST "$PORT/debug/keylog/off"
;;
logs=on )
curl -s -X POST "$PORT/debug/logs/on"
;;
logs=off )
curl -s -X POST "$PORT/debug/logs/off"
;;
port )
echo "$PORT"
;;
* )
echo "Invalid argument: $1" >&2
display_help >&2
exit 1
;;
esac
[Unit]
Description=Sunlight Certificate Transparency Log
After=network-online.target tank-enc.mount
Wants=network-online.target
StartLimitIntervalSec=0
[Service]
ExecStart=/usr/local/bin/sunlight -c /etc/sunlight/sunlight.yaml
ExecReload=kill -HUP $MAINPID
StandardOutput=append:/var/log/sunlight.jsonl
StandardError=journal
Restart=always
RestartSteps=20
RestartSec=1ms
RestartMaxDelaySec=10s
LimitMEMLOCK=256M
MemorySwapMax=0
[Install]
WantedBy=tank-enc.mount
[Unit]
Description=Sunlight Certificate Transparency Log (staging)
After=network-online.target tank-enc.mount
Wants=network-online.target
StartLimitIntervalSec=0
[Service]
ExecStart=/usr/local/bin/sunlight-staging -c /etc/sunlight/sunlight-staging.yaml
ExecReload=kill -HUP $MAINPID
StandardOutput=append:/var/log/sunlight-staging.jsonl
StandardError=journal
Restart=always
RestartSteps=20
RestartSec=1ms
RestartMaxDelaySec=10s
MemoryMax=8G
MemorySwapMax=0
RuntimeMaxSec=1d
LimitMEMLOCK=256M
[Install]
WantedBy=tank-enc.mount
[Unit]
Description=Sunlight Certificate Transparency Log (read path)
After=network-online.target
Wants=network-online.target
StartLimitIntervalSec=0
[Service]
ExecStart=/usr/local/bin/skylight -c /etc/sunlight/skylight.yaml
StandardOutput=append:/var/log/skylight.jsonl
StandardError=journal
Restart=always
RestartSteps=20
RestartSec=1ms
RestartMaxDelaySec=10s
LimitMEMLOCK=256M
MemorySwapMax=0
[Install]
WantedBy=multi-user.target
[Unit]
Description=Clean up partial tiles
[Service]
Type=oneshot
ExecStart=/usr/local/bin/partial-aftersun -c /etc/sunlight/sunlight.yaml -metrics /var/lib/prometheus/node-exporter/partial-aftersun.prom
# HetrixTools heartbeat: trastevere partial-aftersun
ExecStartPost=/usr/bin/curl --retry 3 --retry-delay 1 -m 15 https://sm.hetrixtools.net/hb/?s=fe7ca32bf25794f97c7b73c8a5e96d17
StandardOutput=append:/var/log/partial-aftersun.jsonl
StandardError=journal
[Unit]
Description=Periodically run partial tiles cleanup while Sunlight is running
RefuseManualStart=yes
PartOf=sunlight.service
[Timer]
OnActiveSec=5s
OnUnitActiveSec=5m
[Install]
WantedBy=sunlight.service
[Unit]
Description=Clean up partial tiles (staging)
[Service]
Type=oneshot
ExecStart=/usr/local/bin/partial-aftersun -c /etc/sunlight/sunlight-staging.yaml -metrics /var/lib/prometheus/node-exporter/partial-aftersun-staging.prom
# HetrixTools heartbeat: loreto partial-aftersun
ExecStartPost=/usr/bin/curl --retry 3 --retry-delay 1 -m 15 https://sm.hetrixtools.net/hb/?s=f207297f6f0214e7aeb8a5738a16c06c
StandardOutput=append:/var/log/partial-aftersun-staging.jsonl
StandardError=journal
[Unit]
Description=Periodically run partial tiles cleanup while Sunlight is running (staging)
RefuseManualStart=yes
PartOf=sunlight-staging.service
[Timer]
OnActiveSec=5s
OnUnitActiveSec=5m
[Install]
WantedBy=sunlight-staging.service
[Unit]
Description=https://github.com/FiloSottile/mostly-harmless/tree/main/public-config
After=network-online.target
Wants=network-online.target
Before=caddy.service
StartLimitIntervalSec=0
[Service]
ExecStart=/usr/local/bin/public-config -name trastevere
DynamicUser=true
Restart=always
RestartSteps=20
RestartSec=1ms
RestartMaxDelaySec=10s
[Install]
WantedBy=multi-user.target
/var/log/*.jsonl {
daily
rotate 10
copytruncate
compress
delaycompress
notifempty
missingok
}
config.trastevere.sunlight.geomys.org:443 {
bind 213.171.190.246 [2001:4b7e:a002::c4]
reverse_proxy localhost:8080
}
[Unit]
StartLimitIntervalSec=0
[Service]
Restart=always
RestartSteps=20
RestartSec=1ms
RestartMaxDelaySec=10s
#!/bin/sh
set -eu
out=/var/lib/prometheus/node-exporter/zfs.prom
tmp=$(mktemp "$out.XXXXXX")
trap 'rm -f "$tmp"' EXIT
{
echo '# TYPE zfs_dataset_referenced_bytes gauge'
echo '# TYPE zfs_dataset_logicalreferenced_bytes gauge'
echo '# TYPE zfs_dataset_available_bytes gauge'
zfs list -Hp -r -o name,referenced,logicalreferenced,available tank/logs tank/caches \
| awk '{
printf "zfs_dataset_referenced_bytes{dataset=\"%s\"} %s\n", $1, $2
printf "zfs_dataset_logicalreferenced_bytes{dataset=\"%s\"} %s\n", $1, $3
printf "zfs_dataset_available_bytes{dataset=\"%s\"} %s\n", $1, $4
}'
echo '# HELP zfs_txg_last_sync_seconds Sync duration of the most recent committed transaction group.'
echo '# TYPE zfs_txg_last_sync_seconds gauge'
echo '# HELP zfs_txg_max_sync_seconds Maximum txg sync duration over the kernel history ring.'
echo '# TYPE zfs_txg_max_sync_seconds gauge'
echo '# HELP zfs_txg_last_dirty_bytes Dirty data in the most recent committed transaction group.'
echo '# TYPE zfs_txg_last_dirty_bytes gauge'
for txgs in /proc/spl/kstat/zfs/*/txgs; do
pool=$(basename "$(dirname "$txgs")")
awk -v pool="$pool" '
$3 == "C" { last_sync = $12; if ($12 > max_sync) max_sync = $12; last_dirty = $4 }
END {
if (last_sync != "") {
printf "zfs_txg_last_sync_seconds{pool=\"%s\"} %.6f\n", pool, last_sync / 1e9
printf "zfs_txg_max_sync_seconds{pool=\"%s\"} %.6f\n", pool, max_sync / 1e9
printf "zfs_txg_last_dirty_bytes{pool=\"%s\"} %d\n", pool, last_dirty
}
}' "$txgs"
done
# Slab memory (dentries, inodes) is charged to the cgroup that created it,
# and moves to system.slice itself when a transient unit's cgroup goes away,
# where the ZFS shrinker can't see it. Units can vanish between the glob and
# the read, hence the tolerant awk.
echo '# HELP cgroup_memory_stat_bytes Selected memory.stat entries of a cgroup, including its descendants.'
echo '# TYPE cgroup_memory_stat_bytes gauge'
for d in /sys/fs/cgroup/system.slice /sys/fs/cgroup/system.slice/*/; do
d=${d%/}
awk -v cg="${d#/sys/fs/cgroup/}" '
$1 ~ /^(anon|file|slab_reclaimable|slab_unreclaimable)$/ {
printf "cgroup_memory_stat_bytes{cgroup=\"%s\",stat=\"%s\"} %s\n", cg, $1, $2
}' "$d/memory.stat" 2>/dev/null || true
done
} > "$tmp"
chmod 0644 "$tmp"
mv "$tmp" "$out"
[Unit]
Description=Write ZFS dataset metrics for node_exporter textfile collector
[Service]
Type=oneshot
ExecStart=/usr/local/bin/zfs-textfile
TimeoutStartSec=30s
[Unit]
Description=Run zfs-textfile every minute
[Timer]
OnBootSec=30s
OnUnitActiveSec=60s
AccuracySec=1s
[Install]
WantedBy=timers.target
# Set the command-line arguments to pass to the server.
# Due to shell escaping, to pass backslashes for regexes, you need to double
# them (\\d for \d). If running under systemd, you need to double them again
# (\\\\d to mean \d), and escape newlines too.
ARGS="--collector.vmstat.fields=^(oom_kill|pgpg|pswp|pg.*fault|pgsteal_|pgscan_|slabs_scanned|kswapd_|pginodesteal|allocstall_|drop_pagecache|drop_slab).* --collector.slabinfo --collector.sysctl --collector.sysctl.include=vm.min_free_kbytes --collector.sysctl.include=vm.watermark_scale_factor --collector.sysctl.include=vm.swappiness --collector.sysctl.include=vm.vfs_cache_pressure --collector.sysctl.include=fs.dentry-state:nr_dentry,nr_unused,age_limit,want_pages,nr_negative,dummy --collector.sysctl.include=fs.inode-nr:nr_inodes,nr_free_inodes"
[Service]
ExecStartPre=+/bin/chgrp prometheus /proc/slabinfo
ExecStartPre=+/bin/chmod 0440 /proc/slabinfo
# Drops the reclaimable memory charged to a service's cgroup. This bounds the
# dentries, and through them the ZFS znodes, dnodes and ARC metadata.
#
# It works around a combination of OpenZFS (2.3) and cgroup behaviors that can
# cause reads to stall for ~2 ms.
#
# 1. Under pressure, ARC allocations are throttled.
#
# Every ZFS read and write allocates an ARC buffer (arc_get_data_impl),
# whether or not the data is worth caching. If the ARC is more than a small
# margin above its target size c, each allocation waits for an eviction pass
# (arc_wait_for_eviction). If the pass finds nothing evictable, the waiter is
# released anyway, having paid for a full futile scan.
#
# The ARC is not needed for the log to perform (cache.db reads cost 0.25 ms
# with the ARC nearly empty of data) but there is no way to opt out of the
# wait. (direct=always would bypass it, but SQLite's page buffers are not
# page-aligned, so it would never engage.)
#
# 2. The target ignores unevictable memory.
#
# Under memory pressure, arc_reduce_target_size lowers c with no floor other
# than c_min (4 GiB here), regardless of how much of the ARC is pinned by
# open dnodes and their metadata. Once the pinned set alone exceeds c, the
# ARC is permanently over target and every read is throttled as in (1).
#
# 3. A lot of memory is unevictable for ZFS due to cgroups.
#
# Dentries are charged to the memory cgroup of the process that looked them
# up (e.g. skylight's for the tiles it serves). Each dentry pins a znode,
# which pins a dnode and ~5 KiB of ARC metadata (dnode block, dbuf, bonus
# buffer). The only way ZFS has to release them is arc_prune -> zfs_prune ->
# super_cache_scan, which scans the root memory cgroup only. So for ZFS this
# memory is unevictable.
#
# 4. Memory pressure is handled by ZFS, never by the kernel.
#
# The kernel could evict the dentries charged to all cgroups, but ZFS starts
# shrinking its target when free plus inactive file memory drops under
# zfs_arc_sys_free (4.25 GiB here), while kswapd only wakes under the zone
# low watermark (~200 MiB here). The kernel's reclaim rarely runs, and when
# it does it stops as soon as its watermark is met, a few hundred MiB later.
#
# The result is that ZFS is the only one trying to reclaim memory (4), and while
# it can't succeed (3) it keeps lowering its target (2), which throttles every
# read (1).
#
# Dropping the dentries through memory.reclaim leaves the znodes on the root
# inode LRU, where ZFS pruning (and kswapd) can dispose of them, turning the
# pinned metadata back into ordinary evictable ARC.
[Unit]
Description=Reclaim page cache and slab charged to %i.service
[Service]
Type=oneshot
ExecStart=/usr/local/bin/memcg-reclaim %i
TimeoutStartSec=5m
# Hourly bound pinned ZFS metadata (dentries, and through them the ZFS znodes,
# dnodes and ARC metadata) that skylight accumulates by serving tiles at roughly
# an hour of tile lookups (~0.4M dentries, ~2 GiB of ARC).
#
# The cost is that the next lookup of each tile re-reads its dnode.
#
# See memcg-reclaim@.service.
[Unit]
Description=Hourly reclaim of memory charged to skylight.service
PartOf=skylight.service
[Timer]
OnActiveSec=1h
OnUnitActiveSec=1h
AccuracySec=1m
[Install]
WantedBy=skylight.service
#!/bin/sh
# Usage: memcg-reclaim UNIT
#
# Drops the reclaimable memory (page cache, dentries, inodes) charged to a
# systemd service's memory cgroup. See memcg-reclaim@.service.
set -eu
cg=/sys/fs/cgroup/system.slice/$1.service
mstat() { awk -v k="$1" '$1 == k { print $2 }' "$cg/memory.stat"; }
before=$(mstat slab_reclaimable)
# Ask for every file page and reclaimable slab byte charged to the cgroup.
# The kernel reclaims what it can and then reports EAGAIN if it could not
# reach the requested amount (partially freed slabs are not page-freeable),
# so a failed write is the normal outcome and the effect is checked below.
# swappiness=0 keeps the pass away from anonymous memory, which could not
# be swapped anyway (MemorySwapMax=0).
echo "$(( $(mstat file) + before )) swappiness=0" > "$cg/memory.reclaim" 2>/dev/null || true
after=$(mstat slab_reclaimable)
echo "$1: slab_reclaimable $((before >> 20)) MiB -> $((after >> 20)) MiB;" \
"$(awk '{ print $1 }' /proc/sys/fs/dentry-state) dentries system-wide"
# Fail loudly (node_systemd_unit_state) if the reclaim had no effect, e.g.
# because the objects are actually in use or the cgroup layout changed.
[ "$after" -lt $((256 << 20)) ]