From 763e21e0536855e138a5725d685189d96b51c4b8 Mon Sep 17 00:00:00 2001 From: Vladimir Dubrovin <3proxy@3proxy.ru> Date: Sat, 1 Aug 2026 10:22:03 +0300 Subject: [PATCH] Fixed: maxchild dropped to 100 on config reload if not explicitly set, system-specific tuning hints added to doc --- doc/html/highload.html | 267 +++++++++++++++++++++++++++++++++++--- man/3proxy.cfg.5 | 3 +- scripts/3proxy.service.in | 13 +- scripts/init.d/3proxy.in | 10 ++ src/common.c | 2 +- src/conf.c | 2 +- src/proxy.h | 1 + src/proxymain.c | 26 ++++ 8 files changed, 299 insertions(+), 25 deletions(-) diff --git a/doc/html/highload.html b/doc/html/highload.html index dea64a5..8ada547 100644 --- a/doc/html/highload.html +++ b/doc/html/highload.html @@ -5,8 +5,8 @@
maxconn 1000 proxy -p3129 @@ -19,6 +19,10 @@ simultaneous connections to 3proxy.Avoid setting 'maxconn' to an arbitrarily high value; it should be carefully chosen to protect the system and proxy from resource exhaustion. Setting maxconn above available resources can lead to denial of service conditions. +
'maxconn' is not reduced automatically to fit the open file limit. If the limit is +too low 3proxy only prints a warning at startup +("current open file ulimits are too low") and then fails to accept connections once +the limit is reached, so check for this warning after changing 'maxconn'.
Understanding Resource Requirements
Each running service requires:
-DefaultLimitDATA=infinity -DefaultLimitSTACK=infinity -DefaultLimitCORE=infinity -DefaultLimitRSS=infinity -DefaultLimitNOFILE=102400 -DefaultLimitAS=infinity -DefaultLimitNPROC=10240 -DefaultLimitMEMLOCK=infinity +LimitNOFILE=1048576 +LimitNPROC=infinity +TasksMax=infinity ++To change them on an installed system use an override instead of editing the unit: +
+systemctl edit 3proxy +systemctl daemon-reload && systemctl restart 3proxy +systemctl show 3proxy -p LimitNOFILE -p LimitNPROC -p TasksMax ++TasksMax is the one that is easy to miss. It is the cgroup limit on the number +of threads, and if it is not set the unit inherits DefaultTasksMax, which is 15% of +kernel.threads-max (about 9000 on a typical host). Since 3proxy uses one thread per +connection, that caps concurrent connections at that number regardless of LimitNPROC +and maxconn, and the only symptom is "pthread_create()" errors in the log. +
On systemd older than 227, which has no TasksMax, and for limits that must apply to +several services, the same values can be set globally as DefaultLimitNOFILE / +DefaultLimitNPROC in /etc/systemd/system.conf, but prefer the per-unit settings. + +
With SysV init the limits are not applied by limits.conf either, because +start-stop-daemon does not open a PAM session, so the daemon simply inherits the limits +of init. The shipped init script raises them itself before starting 3proxy: +
+ulimit -n 65536 +ulimit -u 32768 ++adjust these values in the script to match 'maxconn'. + +
On FreeBSD rc.subr applies limits(1) with the login class of the service (the +"daemon" class by default), so the limits can be set either in /etc/login.conf for that +class, or per service in rc.conf: +
+3proxy_limits="-n 65536"-in user.conf / system.conf
File descriptors. 3proxy needs 2 descriptors per connection (4 for FTP), plus +one per service, plus temporary ones for name resolution and RADIUS. +
+fs.nr_open = 1048576 # (1048576) upper bound for any process' RLIMIT_NOFILE ++ulimit -n (RLIMIT_NOFILE) is the limit that actually applies and is commonly +left at 1024; it must be raised for the 3proxy process itself, see "Setting ulimits" +above. fs.file-max is effectively unlimited on 64-bit kernels and rarely needs +changing. + +
Threads. Because of the "one connection - one thread" model these limits are +reached earlier with 3proxy than with event-driven servers. Each thread also consumes +one or two mappings, so vm.max_map_count matters too. +
+kernel.threads-max = 200000 # (~60000 on a 16G host, scales with RAM) +kernel.pid_max = 4194304 # (4194304) +vm.max_map_count = 1048576 # (1048576) ++RLIMIT_NPROC (ulimit -u) limits threads per user and must be raised as well. +Check the actual thread count with grep Threads /proc/PID/status. + +
Listen queue. 3proxy uses a listen backlog of 1+(maxconn/8) unless the +'backlog' command is given, so a large 'maxconn' does not automatically give a large +queue, and the kernel caps it at somaxconn: +
+net.core.somaxconn = 4096 # (4096) +net.ipv4.tcp_max_syn_backlog = 4096 # (512) raise for bursty connection rates +net.ipv4.tcp_syncookies = 1 # (1) keep enabled ++ +
Ephemeral ports and TIME_WAIT. See "Extending the Ephemeral Port Range" above +for the multi-IP case. The range gives about 28000 outgoing connections per +destination address by default: +
+net.ipv4.ip_local_port_range = 10240 65535 # (32768 60999) +net.ipv4.tcp_tw_reuse = 2 # (2) reuse TIME_WAIT for outgoing connections +net.ipv4.tcp_fin_timeout = 30 # (60) ++Do not enable tcp_tw_recycle; it was removed in kernel 4.12 and breaks NAT clients. + +
Socket buffers. Autotuning is usually right. Buffer memory is per connection, +so raising the maximums with tens of thousands of connections costs a lot of RAM: +
+net.core.rmem_max = 4194304 # (212992) +net.core.wmem_max = 4194304 # (212992) +net.ipv4.tcp_rmem = 4096 131072 6291456 # (same) min default max +net.ipv4.tcp_wmem = 4096 16384 4194304 # (same) ++Raise these only for high bandwidth-delay product links, and prefer raising the third +(max) value and leaving the default alone. + +
Conntrack. Only relevant if netfilter/nftables tracks the proxy's traffic. If +it does, the table is exhausted long before 3proxy's own limits, with +"nf_conntrack: table full, dropping packet" in dmesg: +
+net.netfilter.nf_conntrack_max = 1048576 +net.netfilter.nf_conntrack_buckets = 262144 +net.netfilter.nf_conntrack_tcp_timeout_established = 3600 # (432000, i.e. 5 days) +net.netfilter.nf_conntrack_tcp_timeout_time_wait = 30 # (120) ++nf_conntrack_max defaults to nf_conntrack_buckets, which itself is derived from the +amount of RAM, so it is often much lower than expected on small machines. Each +connection takes two entries (one per direction). The default established timeout of +5 days matters more than the table size with high connection churn: entries for +connections that are long gone keep occupying the table. +
If no rules need conntrack, not loading it at all is faster: the modules are loaded +on demand by the first rule that needs them ("-m state", "-m conntrack", any NAT +rule), so a ruleset without such rules keeps the proxy traffic untracked. If conntrack +is needed for other traffic but not for the proxy's, exempt the proxy's traffic +explicitly in the raw table: +
+iptables -t raw -A PREROUTING -p tcp --dport 3128 -j CT --notrack +iptables -t raw -A OUTPUT -p tcp -m owner --uid-owner proxy -j CT --notrack ++ +
Conntrack helpers (ALGs). The helper modules - nf_conntrack_ftp, +nf_conntrack_sip, nf_conntrack_h323, nf_conntrack_pptp, nf_conntrack_irc, +nf_conntrack_tftp - inspect the payload of every matching packet and create additional +"expectation" entries, so they cost both CPU and table space, and they have a long +history of security issues. Unload and blacklist the ones you do not actually need: +
+lsmod | grep nf_conntrack +modprobe -r nf_conntrack_sip nf_conntrack_h323 nf_conntrack_ftp nf_conntrack_pptp +echo "blacklist nf_conntrack_sip" >> /etc/modprobe.d/no-alg.conf ++On current kernels a helper only acts when it is attached explicitly +("-j CT --helper ftp"), so simply not attaching it is enough; automatic helper +assignment was deprecated and later removed. Older kernels, and most router firmware, +still enable them by default. + +
Checking the result. ss -s for socket state totals, +ss -lnt for listen queue overflow, nstat -az TcpExtListenOverflows +TcpExtListenDrops for accept queue drops, and +cat /proc/PID/limits for the limits actually applied to the running process. + +
Dynamic (ephemeral) port range. Since Windows Vista / Server 2008 the default +range is 49152-65535, i.e. only 16384 outgoing connections per local address, which is +reached quickly by a busy proxy. Show and change it with: +
+netsh int ipv4 show dynamicport tcp +netsh int ipv4 set dynamicport tcp start=10000 num=55535 ++The minimum start port is 1025, the minimum size of the range is 255, and the end of +the range cannot exceed 65535. The range is set separately for TCP and UDP, and for +IPv4 and IPv6. On pre-Vista systems the equivalent is the MaxUserPort registry value +in HKLM\SYSTEM\CurrentControlSet\Services\Tcpip\Parameters. + +
Listening socket. 3proxy sets SO_REUSEADDR on the listening socket by +default on Unix, but not on Windows: there it is not needed to rebind the port, and it +only allows another local process to bind the same address and port, with undefined +behaviour as to which of them receives the connections. If the machine is shared or +untrusted, harden the listening socket instead: +
+proxy -olSO_EXCLUSIVEADDRUSE ++Note that a socket with SO_EXCLUSIVEADDRUSE may not be immediately rebindable after a +restart if accepted connections are still active, so test restarts before using it. + +
Port reuse. 3proxy always binds the outgoing socket before connecting, so +Windows does not apply its automatic ephemeral port reuse (which it does only for +connections with an implicit bind). Setting the option explicitly on the +proxy-to-server socket therefore helps against port exhaustion: +
+proxy -osSO_REUSE_UNICASTPORT ++SO_REUSE_UNICASTPORT requires Windows 10 / Server 2019 or later. On older systems +(Windows 7 / Server 2008 and later) use SO_PORT_SCALABILITY instead; where both are +available Microsoft recommends SO_REUSE_UNICASTPORT. Note that SO_REUSEADDR has +different, weaker semantics on Windows than on Unix and allows another socket to bind +the same address and port, so do not use it on the listening socket as a substitute. + +
TIME_WAIT. Closed connections hold their port for the TcpTimedWaitDelay +period, set in +HKLM\SYSTEM\CurrentControlSet\Services\Tcpip\Parameters (DWORD, seconds). The +effective default differs between Windows versions (2 to 4 minutes); check the current +behaviour before changing it, and lower it only together with an extended port range. +Count the connections in that state with: +
+netstat -ano -p tcp | find /c "TIME_WAIT" ++ +
Threads and address space. Windows has no ulimit equivalent, and the handle +count is not normally the limit. On 32-bit builds the 2 GB of user address space is: +each connection thread reserves its stack there, so a few thousand connections can +exhaust the address space while physical memory is still free. Use a 64-bit build for high load, and +see "Setting Stack Size" above. + +
Filter drivers. Antivirus, endpoint protection and other LSP/WFP filter +drivers inspect every connection and are frequently the actual bottleneck on Windows, +costing far more than any tuning above can recover. Exclude the 3proxy process and its +ports, or test with the protection temporarily disabled to see the difference before +tuning anything else. +
-socks -olSO_REUSEPORT -p3128 -e 1.1.1.1 -socks -olSO_REUSEPORT -p3128 -e 2.2.2.2 -socks -olSO_REUSEPORT -p3128 -e 3.3.3.3 -socks -olSO_REUSEPORT -p3128 -e 4.4.4.4 +socks -olSO_REUSEPORT -p3128 -e1.1.1.1 +socks -olSO_REUSEPORT -p3128 -e2.2.2.2 +socks -olSO_REUSEPORT -p3128 -e3.3.3.3 +socks -olSO_REUSEPORT -p3128 -e4.4.4.4For web browsing, the last two examples are not recommended because the same client can get a different external address for different requests; you should choose the external @@ -172,6 +364,39 @@ randomly fail due to IP+port pair collisions if the remote or local system doesn't support this trick. +
On a Linux based router the same knobs apply and have to be raised there as well: +nf_conntrack_max / nf_conntrack_buckets and the conntrack timeouts (see "Linux Tuning +Hints" above), plus the port range used for translation, which is ip_local_port_range +for MASQUERADE, or the explicit range if SNAT is configured with --to-ports. Note that +the range is per translated address: with a single public IP, all clients share it. + +
Entry level and SOHO routers are the usual bottleneck here. They typically have a +small fixed NAT/conntrack table (a few thousand entries), aggressive or non-adjustable +timeouts, and no way to change either. Symptoms are seen on the proxy but caused by the +router: connections that fail or hang at random under load while the proxy is far from +its own limits, no error in the 3proxy log except a failed outgoing connect, and +recovery after a pause or a router reboot. Before tuning 3proxy further, check the +router's session/NAT table counters. For high load either give the proxy a public +address without NAT in the path, or use a router where the table size and timeouts are +configurable. + +
On the router, also turn off the application layer gateways that are not actually +used - they usually appear in the web interface as "SIP ALG", "FTP ALG", "H.323 ALG", +"PPTP passthrough", "IPsec/VPN passthrough". They are commonly enabled by default, they +parse the payload of matching connections, and they consume additional session table +entries for the connections they expect. If nothing behind the proxy uses FTP, VoIP or +those VPN protocols, disabling them frees table space and CPU on exactly the device +that is the bottleneck. +
For 32-bit systems, address space can be a bottleneck you should consider. If -you're short on address space, you can try using a negative stack size. +you're short on address space, you can try using a negative stack size. The result is +never lowered below the system minimum (PTHREAD_STACK_MIN), so a large negative value +can not disable the thread stack. The base value the 'stacksize' is added to is 48K +(64K on FreeBSD/NetBSD/OpenBSD/DragonFly, where libc uses more stack, e.g. in +vfprintf() called by syslog()).
logdump 1 1is useful to see how grace delays work; choose a delay value to avoid filling the read -pipe/buffer (typically 64K) but keep the request sizes close to the chosen average +buffer (typically 64K) but keep the request sizes close to the chosen average on large file uploads/downloads. diff --git a/man/3proxy.cfg.5 b/man/3proxy.cfg.5 index e21d31a..a0bbb10 100644 --- a/man/3proxy.cfg.5 +++ b/man/3proxy.cfg.5 @@ -178,7 +178,8 @@ connect to given remote HOST:port instead of listening local connection on -p or .br .B -oc\fIOPTIONS\fB, -os\fIOPTIONS\fB, -ol\fIOPTIONS\fB, -or\fIOPTIONS\fB, -oR\fIOPTIONS\fR options for proxy-to-client (\fB-oc\fR), proxy-to-server (\fB-os\fR), proxy listening (\fB-ol\fR), connect back client (\fB-or\fR), connect back listening (\fB-oR\fR) sockets. -Options like TCP_CORK, TCP_NODELAY, TCP_DEFER_ACCEPT, TCP_QUICKACK, TCP_TIMESTAMPS, USE_TCP_FASTOPEN, SO_REUSEADDR, SO_REUSEPORT, SO_PORT_SCALABILITY, SO_REUSE_UNICASTPORT, SO_KEEPALIVE, SO_DONTROUTE may be supported depending on OS. +Options like TCP_CORK, TCP_NODELAY, TCP_DEFER_ACCEPT, TCP_QUICKACK, TCP_TIMESTAMPS, TCP_FASTOPEN, SO_REUSEADDR, SO_REUSEPORT, SO_EXCLUSIVEADDRUSE, SO_PORT_SCALABILITY, SO_REUSE_UNICASTPORT, SO_KEEPALIVE, SO_DONTROUTE may be supported depending on OS. +SO_REUSEADDR and SO_REUSEPORT are set on the listening socket by default on Unix. On Windows SO_REUSEADDR is not set: it is not required to rebind a listening port and it only lets another local process bind the same address and port. Use SO_EXCLUSIVEADDRUSE (Windows) on the listening socket (\fB-ol\fR) to prevent that. .br .B -H (for all services) Expect HAProxy PROXY protocol v1 header on incoming connection. diff --git a/scripts/3proxy.service.in b/scripts/3proxy.service.in index 3c84a15..c5870bd 100644 --- a/scripts/3proxy.service.in +++ b/scripts/3proxy.service.in @@ -1,6 +1,6 @@ [Unit] Description=3proxy tiny proxy server -Documentation=man:3proxy(1) +Documentation=man:3proxy(8) man:3proxy.cfg(5) After=network.target [Service] @@ -13,8 +13,15 @@ ExecReload=/bin/kill -SIGUSR1 $MAINPID KillMode=process Restart=on-failure RestartSec=60s -LimitNOFILE=65536 -LimitNPROC=32768 +# 3proxy uses one thread and two descriptors per connection (four for ftppr), +# so it reaches these limits much earlier than event driven servers. They are +# ceilings only: the actual number of connections is governed by 'maxconn' in +# the configuration file. TasksMax must be set explicitly, systemd's +# DefaultTasksMax (15% of kernel.threads-max, e.g. ~9000) otherwise caps the +# number of threads, and thus connections, regardless of LimitNPROC. +LimitNOFILE=1048576 +LimitNPROC=infinity +TasksMax=infinity RuntimeDirectory=3proxy RuntimeDirectoryMode=0755 diff --git a/scripts/init.d/3proxy.in b/scripts/init.d/3proxy.in index efaa027..0936b04 100644 --- a/scripts/init.d/3proxy.in +++ b/scripts/init.d/3proxy.in @@ -24,9 +24,19 @@ if [ -f /etc/init.d/functions ]; then . /etc/init.d/functions fi +# SysV init does not apply limits.conf consistently: start-stop-daemon does not +# open a PAM session, so pam_limits is not involved and the daemon inherits the +# limits of init. 3proxy needs two descriptors per connection (four for ftppr), +# so raise them here. Adjust to match 'maxconn' in the configuration file. +set_limits() { + ulimit -n 65536 2>/dev/null || ulimit -n 4096 2>/dev/null || true + ulimit -u 32768 2>/dev/null || true +} + case "$1" in start) echo -n "Starting 3Proxy: " + set_limits if [ ! -d /var/run/3proxy ]; then mkdir -p /var/run/3proxy diff --git a/src/common.c b/src/common.c index 2c118bf..ad4333a 100644 --- a/src/common.c +++ b/src/common.c @@ -190,7 +190,7 @@ struct extparam conf = { .paused = 0, .archiverc = 0, .demon = 0, - .maxchild = 500, + .maxchild = DEFAULT_MAXCHILD, .backlog = 0, .needreload = 0, .timetoexit = 0, diff --git a/src/conf.c b/src/conf.c index 9f32070..3acb3a0 100644 --- a/src/conf.c +++ b/src/conf.c @@ -2025,7 +2025,7 @@ void freeconf(struct extparam *confp){ #endif *SAFAMILY(&confp->intsa) = AF_INET; *SAFAMILY(&confp->extsa) = AF_INET; - confp->maxchild = 100; + confp->maxchild = DEFAULT_MAXCHILD; confp->backlog = 0; resolvfunc = NULL; numservers = 0; diff --git a/src/proxy.h b/src/proxy.h index 17c3875..878c6d6 100644 --- a/src/proxy.h +++ b/src/proxy.h @@ -34,6 +34,7 @@ #define MAXUSERNAME 128 #define _PASSWORD_LEN 256 #define MAXNSERVERS 5 +#define DEFAULT_MAXCHILD 500 #define TCPBUFSIZE 65536 #define SRVBUFSIZE (param->srv->bufsize?param->srv->bufsize:((param->service == S_UDPPM)?UDPBUFSIZE:TCPBUFSIZE)) diff --git a/src/proxymain.c b/src/proxymain.c index 0cbda7f..cdd0680 100644 --- a/src/proxymain.c +++ b/src/proxymain.c @@ -149,6 +149,20 @@ void * threadfunc (void *p) { } #undef param +#ifdef _WIN32 +/* Present since Windows 7 (SO_PORT_SCALABILITY) and Windows 10 / Server 2019 + (SO_REUSE_UNICASTPORT), define them if the SDK is older so the options can + still be requested. setsockopt() just fails on a system which does not + support them and the failure is ignored. + */ +#ifndef SO_PORT_SCALABILITY +#define SO_PORT_SCALABILITY 0x3006 +#endif +#ifndef SO_REUSE_UNICASTPORT +#define SO_REUSE_UNICASTPORT 0x3007 +#endif +#endif + struct socketoptions sockopts[] = { #ifdef TCP_NODELAY {TCP_NODELAY, "TCP_NODELAY"}, @@ -171,6 +185,9 @@ struct socketoptions sockopts[] = { #ifdef SO_REUSEPORT {SO_REUSEPORT, "SO_REUSEPORT"}, #endif +#ifdef SO_EXCLUSIVEADDRUSE + {SO_EXCLUSIVEADDRUSE, "SO_EXCLUSIVEADDRUSE"}, +#endif #ifdef SO_PORT_SCALABILITY {SO_PORT_SCALABILITY, "SO_PORT_SCALABILITY"}, #endif @@ -794,8 +811,15 @@ int MODULEMAINFUNC (int argc, char** argv){ if(*SAFAMILY(&srv.intsa) != AF_UNIX) #endif { +/* SO_REUSEADDR is not set on Windows: it is not needed to rebind a listening + port there, and it only allows another local process to bind the same + address and port, with undefined behaviour as to which socket receives the + connections. Use -olSO_EXCLUSIVEADDRUSE to prevent that instead. + */ +#ifndef _WIN32 opt = 1; if(srv.so._setsockopt(srv.so.state, sock, SOL_SOCKET, SO_REUSEADDR, (char *)&opt, sizeof(int)))perror("setsockopt()"); +#endif #ifdef SO_REUSEPORT opt = 1; srv.so._setsockopt(srv.so.state, sock, SOL_SOCKET, SO_REUSEPORT, (char *)&opt, sizeof(int)); @@ -911,8 +935,10 @@ int MODULEMAINFUNC (int argc, char** argv){ freesrvstrings(&srv, cbc_string, cbl_string); return -6; } +#ifndef _WIN32 opt = 1; srv.so._setsockopt(srv.so.state, srv.cbsock, SOL_SOCKET, SO_REUSEADDR, (char *)&opt, sizeof(int)); +#endif #ifdef SO_REUSEPORT opt = 1; srv.so._setsockopt(srv.so.state, srv.cbsock, SOL_SOCKET, SO_REUSEPORT, (char *)&opt, sizeof(int));