Files
nixos/hosts/unstable/linode/default.nix
T
emily 2c3607f17c
buildbot/nix-eval Build done. (1 warning)
buildbot/nix-build Build done.
fix: prevent HAProxy from reusing stale keep-alive conns to nginx/Nextcloud
DAVx5 (CalDAV/CardDAV) on greg's phone was intermittently failing every
sync type (CONTACTS/EVENTS/TASKS/RefreshCollectionsWorker) against
next.thehellings.com with:

  java.io.IOException: unexpected end of stream
  Caused by: java.io.EOFException: \n not found: limit=0

This is the classic OkHttp/HTTP client signature of the far end
silently closing a pooled keep-alive connection: the client reuses a
socket it still believes is open, gets zero bytes back while reading
response headers, and throws exactly this exception.

Root cause: HAProxy's 'next' backend proxies to nginx on
127.0.0.1:8080, and HAProxy defaults to end-to-end keep-alive (both
client- and server-side) unless told otherwise. nginx's
keepalive_timeout is 65s, so any HAProxy<->nginx connection idle past
that gets closed by nginx without HAProxy's knowledge. A request that
lands on that now-dead pooled connection right after gets nothing back
- surfacing to the client as a bare socket EOF while reading headers.
The frontend's existing 'option http-server-close'/'http-keep-alive'
pair only governs the client-facing side of HAProxy and does nothing
for the HAProxy->nginx leg.

Fix:
- backend next: add 'option http-server-close' so HAProxy opens a
  fresh connection to nginx per request instead of pooling/reusing
  one. The backend is localhost, so the extra TCP handshake cost is
  negligible, and this removes the whole class of stale-connection EOF
  errors.
- defaults: add 'timeout http-keep-alive 30s' to bound how long an
  idle client-facing keep-alive connection is held open. Previously
  unset, it fell back to 'timeout client' (500s) - unnecessarily long
  given maxconn is only 80, and tightens client-side connection churn
  to be more predictable too.

Diagnosed by pulling the nginx_access journal (enabled in #37/#38) for
the failing sync window and cross-referencing nginx's
services.nginx.appendHttpConfig / generated nginx.conf keepalive
settings against HAProxy's request-level defaults. Could not run
'haproxy -c'/'nginx -t' locally (no toolchain in the agent sandbox) -
recommend confirming via CI/garnix before merge, same as #38.
2026-08-09 22:21:20 -05:00

425 lines
13 KiB
Nix

{
config,
lib,
metadata,
pkgs,
pkgs',
...
}:
let
homepage = "127.0.0.1:30080";
nextcloudPort = 8080;
sshPort = 2222;
matrixServer = pkgs.writeText "matrix_server" (
builtins.toJSON {
"m.server" = "matrix.thehellings.com:443";
}
);
matrixClient = pkgs.writeText "matrix_client" (
builtins.toJSON {
"m.homeserver" = {
base_url = "https://matrix.thehellings.com";
};
"m.identity_server" = {
base_url = "https://vector.im";
};
}
);
in
{
imports = [
./hardware-configuration.nix
];
age.secrets = {
acme.file = ../../../secrets/acme.age;
nextcloudadmin = {
file = ../../../secrets/nextcloudadmin.age;
owner = "nextcloud";
};
};
environment.systemPackages = with pkgs; [
bind
graphviz
nix-du
pgloader
podman-compose
pkgs'.upgrade-pg-cluster
];
# Historical per-interface bandwidth tracking (5-min granularity, kept for
# months). This is what's actually missing when diagnosing "traffic was
# high for the past several hours" reports after the fact — journalctl
# timestamps only tell you what else was happening, not the traffic curve
# itself. `vnstat -h`/`vnstat --json h` gives an immediate confirm/deny of
# a reported window without waiting on live sampling.
services.vnstat.enable = true;
greg = {
backup.jobs = {
nextcloud-bkup = {
src = "/var/lib/nextcloud";
dest = "nextcloud-backup";
pre = lib.getExe (
pkgs.writeShellApplication {
name = "nextcloud-backup-pre";
runtimeInputs = [ config.services.nextcloud.occ ];
text = "nextcloud-occ maintenance:mode --on";
}
);
post = lib.getExe (
pkgs.writeShellApplication {
name = "nextcloud-backup-post";
runtimeInputs = [ config.services.nextcloud.occ ];
text = "nextcloud-occ maintenance:mode --off";
}
);
};
greg-postgresql-backup = {
src = config.services.postgresqlBackup.location;
dest = "linode-postgres";
};
};
gitea-runner = {
enable = true;
labels = [
"vps:host"
"blog:host"
"nixos-linode:host"
];
};
home = false;
linode.enable = true;
nebula = {
enable = true;
isLighthouse = true;
unsafeRoutes = [
{
route = "10.42.0.0/16";
via = metadata.hosts.genesis.nebulaIp;
}
];
};
tailscale.enable = true;
};
networking = {
domain = "thehellings.com";
firewall.allowedTCPPorts = [
sshPort
80
443
];
hostName = "linode";
nameservers = [
"10.157.0.2"
"100.96.198.104"
];
networkmanager.enable = lib.mkForce false;
};
programs.ssh.extraConfig = lib.strings.concatStringsSep "\n" [
"Host chronicles.shire-zebra.ts.net"
" User backup"
" IdentityFile /etc/ssh/backup_ed25519"
" StrictHostKeyChecking no"
" UserKnownHostsFile /dev/null"
];
security.acme = {
acceptTerms = true;
defaults = {
dnsPropagationCheck = false;
dnsResolver = "92.123.95.3:53,92.123.94.3:53,92.123.94.2:53,92.123.95.4:53,92.123.95.2:53";
email = "greg.hellings@gmail.com";
extraLegoRunFlags = [ "--ipv4only" ]; # Force IPv4 only
#server = "https://acme-staging-v02.api.letsencrypt.org/directory";
};
certs."thehellings.com" = {
dnsProvider = "linode";
environmentFile = config.age.secrets.acme.path;
extraDomainNames = [
"*.thehellings.com"
];
};
};
services = {
anubis = {
instances = {
git = {
enable = true;
settings = {
BIND = "/run/anubis/anubis-git/anubis.sock";
COOKIE_DOMAIN = "thehellings.com";
SERVE_ROBOTS_TXT = true;
SLOG_LEVEL = "DEBUG";
TARGET = "http://git.k3s.thehellings.lan";
};
};
};
};
haproxy = {
enable = true;
config = ''
global
nbthread 4
maxconn 80
log /dev/log local0
defaults
log global
timeout connect 500s
timeout client 500s
timeout server 1h
# HAProxy defaults to end-to-end keep-alive (client AND server side)
# unless a proxy overrides it. Bound how long an idle client-facing
# keep-alive connection is held: maxconn is only 80, and leaving
# this unset falls back to "timeout client" (500s), which is far
# longer than needed just to wait for a pipelined next request.
timeout http-keep-alive 30s
listen gitsshd
bind *:${toString sshPort}
timeout client 1h
mode tcp
server git-isaiah isaiah.thehellings.lan:32222
server git-jeremiah jeremiah.thehellings.lan:32222
server git-zeke zeke.thehellings.lan:32222
listen stats
bind 127.0.0.1:8404
stats enable
stats uri /
stats refresh 10s
frontend https
bind *:80
bind *:443 ssl crt ${config.security.acme.certs."thehellings.com".directory}/full.pem
http-request redirect scheme https unless { ssl_fc }
http-request add-header X-Forwarded-Proto https
http-response replace-header ^Set-Cookie:\ (.*) Set-Cookie \1;\ Secure
option http-server-close
option http-keep-alive
option httplog
#declare capture response len 80
#http-response capture res.hdr(Location) id 0
use_backend git if { hdr(host) -i src.thehellings.com }
use_backend git if { req_ssl_sni -i src.thehellings.com }
use_backend next if { hdr(host) -i next.thehellings.com }
use_backend next if { req_ssl_sni -i next.thehellings.com }
use_backend matrix if { hdr(host) -i matrix.thehellings.com }
use_backend matrix if { req_ssl_sni -i matrix.thehellings.com }
use_backend immich if { hdr(host) -i immich.thehellings.com }
use_backend immich if { req_ssl_sni -i immich.thehellings.com }
use_backend web if { hdr(host) -i thehellings.com }
use_backend web if { req_ssl_sni -i thehellings.com }
backend git
mode http
balance roundrobin
option accept-unsafe-violations-in-http-response
retries 3
option forwardfor
http-request set-header Host git.k3s.thehellings.lan
server git-isaiah isaiah.thehellings.lan:80
server git-jeremiah jeremiah.thehellings.lan:80
server git-zeke zeke.thehellings.lan:80
backend immich
mode http
balance roundrobin
option accept-unsafe-violations-in-http-response
retries 3
option forwardfor
server immich-proxy 127.0.0.1:${builtins.toString config.services.immich-public-proxy.port}
backend matrix
mode http
balance roundrobin
option accept-unsafe-violations-in-http-response
retries 3
option forwardfor
http-request set-header Host matrix.k3s.thehellings.lan
server git-isaiah isaiah.thehellings.lan:80
server git-jeremiah jeremiah.thehellings.lan:80
server git-zeke zeke.thehellings.lan:80
backend web
mode http
balance roundrobin
option accept-unsafe-violations-in-http-response
retries 3
option forwardfor
http-request return status 200 content-type "application/json" file ${matrixClient} hdr "cache-control" "no-cache" if { path /.well-known/matrix/client }
http-request return status 200 content-type "application/json" file ${matrixServer} hdr "cache-control" "no-cache" if { path /.well-known/matrix/server }
server web-container ${homepage}
backend next
log global
mode http
balance roundrobin
option accept-unsafe-violations-in-http-response
retries 3
option forwardfor
# nginx (the actual listener on 127.0.0.1:8080) has
# keepalive_timeout 65s and will silently close an idle backend
# socket after that. HAProxy's default mode is end-to-end
# keep-alive, so without this it will happily try to reuse a
# backend connection nginx already closed once a mobile client's
# own (longer) keep-alive idle assumption outlives 65s - producing
# exactly the "unexpected end of stream" / EOFException the
# CalDAV/CardDAV client saw. Since the backend is localhost, the
# cost of a fresh TCP connection per request is negligible, so
# just don't try to reuse them here.
option http-server-close
#http-response replace-value Location http://localhost:${builtins.toString nextcloudPort}/(.*) https://next.thehellings.com/\2
server nextcloud 127.0.0.1:${builtins.toString nextcloudPort}
'';
};
immich-public-proxy = {
enable = true;
immichUrl = "http://immich.k3s.thehellings.lan";
};
logrotate = {
enable = true;
settings = {
postgresBackup = {
enable = true;
files = "${config.services.postgresqlBackup.location}/*.gz";
};
postgresLog = {
enable = true;
files = "/var/lib/postgresql/*/log/*.log";
compress = true;
compresscmd = "${pkgs.xz}/bin/xz";
};
};
};
nextcloud = {
enable = true;
package = pkgs.nextcloud33;
appstoreEnable = true;
hostName = "127.0.0.1";
https = false;
config = {
adminpassFile = config.age.secrets.nextcloudadmin.path;
adminuser = "greg";
dbhost = "/run/postgresql";
dbtype = "pgsql";
};
settings = {
default_phone_region = "US";
overwriteprotocol = "http";
trusted_domains = [ "next.thehellings.com" ];
trusted_proxies = [
"localhost"
"127.0.0.1"
];
};
};
# Move to :8080 so that we can run haproxy as the primary HTTP service
nginx = {
virtualHosts."${config.services.nextcloud.hostName}".listen = [
{
addr = "127.0.0.1";
port = nextcloudPort;
}
];
# Route nginx access logs through syslog/journald (rather than only to
# /var/log/nginx/access.log, which the read-only monitoring account
# can't read) so `journalctl -t nginx_access` gives visibility into
# Nextcloud request traffic during bandwidth investigations.
#
# NOTE: nginx's syslog "tag" only allows alphanumeric characters and
# underscores (no hyphens) - an earlier version of this used
# tag=nginx-access, which fails nginx's config test with:
# nginx: [emerg] syslog "tag" only allows alphanumeric characters
# and underscore in .../nginx.conf:114
# That broke nginx.service (and, transitively, Nextcloud/next.thehellings.com,
# which is proxied through nginx on 127.0.0.1:8080) until nginx hit its
# systemd restart limit and gave up (start-limit-hit).
appendHttpConfig = ''
access_log syslog:server=unix:/dev/log,tag=nginx_access combined;
'';
};
openssh.settings.PasswordAuthentication = false;
postgresql = {
enable = true;
package = pkgs.postgresql_15;
checkConfig = true;
ensureDatabases = [ "nextcloud" ];
#initialScript = pkgs.writeText "create-matrix-db.sql" ''
# CREATE ROLE "matrix-synapse" WITH LOGIN;
# CREATE DATABASE "synapse" WITH OWNER "matrix-synapse" TEMPLATE template0 LC_COLLATE = "C" LC_CTYPE = "C";
# GRANT ALL PRIVILEGES ON DATABASE "synapse" TO "matrix-synapse";
#''; # These are done manually in order to set the LC_COLLATE values properly
ensureUsers = [
{
name = "nextcloud";
ensureDBOwnership = true;
}
];
settings = {
log_connections = true;
log_statement = "all";
logging_collector = true;
log_filename = "postgresql.log";
};
identMap = ''
root root postgres
'';
};
postgresqlBackup = {
enable = true;
databases = [ "nextcloud" ];
};
};
systemd.services = {
haproxy = {
after = [
"nextcloud.service"
"network-online.target"
];
wants = [
"nextcloud.service"
"network-online.target"
];
};
};
users.users.haproxy.extraGroups = [ config.security.acme.certs."thehellings.com".group ];
# Actually serve the content from here
virtualisation.oci-containers = {
backend = "podman";
containers."homepage" = {
image = "src.thehellings.com/greg/homepage:latest";
ports = [ "${homepage}:80" ];
};
};
virtualisation.podman = {
enable = true;
dockerCompat = true;
dockerSocket.enable = true;
};
}