Load Generator Sizing, JVM Tuning, OS Limits, and Network Capacity: Guided Hands-On Workflow
The guided workflow uses one deliberately inefficient HTTP behavior rather than a risky OS tweak: the baseline disables connection reuse. The target records every client source port, so socket churn is independently measurable. After profiling heap/GC, CPU, sockets, network and JTL, the only change is checking Use KeepAlive. This is reversible in the JMX and does not require Administrator/root privileges.
Learning objectives
- Build and run one bounded HTTP generator profile in CLI mode.
- Monitor JVM heap/GC and process CPU during the run.
- Inspect sockets/handles/FDs, dynamic-port range, DNS and NIC counters without changing them.
- Measure JTL size/bytes and achieved RPS.
- Identify socket/connection churn as the deliberately amplified resource pressure.
- Make one reversible plan change and rerun under identical conditions.
1. Safety envelope
127.0.0.1:8026. Each normal run =
10 threads ×100 loops = 1,000 requests, 100 ms pacing, 8 KiB
response, ≤20 seconds. Two normal variants plus optional smoke only.
Abort on >1,000 target requests/run, target errors, JVM
OOM/full-GC loop, sustained generator CPU saturation, unexpectedly
high socket/handle growth, disk free-space risk, non-loopback
connection, or missing profile evidence.
2. Create the target with connection evidence
Save fixtures/generator_fixture.py:
from http.server import BaseHTTPRequestHandler, ThreadingHTTPServer
from pathlib import Path
from urllib.parse import urlparse, parse_qs
import argparse
import json
import re
import threading
import time
FIXTURE_VERSION = "prompt26-generator-fixture-v1"
SAFE = re.compile(r"^[A-Za-z0-9_.-]{1,64}$")
lock = threading.Lock()
event_log = None
run_state = {}
def now_ms():
return int(time.time() * 1000)
def ensure_run(run_id):
with lock:
if run_id not in run_state:
run_state[run_id] = {
"requests": 0,
"errors": 0,
"client_ports": set(),
"bytes_sent": 0,
}
return run_state[run_id]
def snapshot(run_id=None):
with lock:
if run_id is not None:
s = run_state.get(run_id, {
"requests": 0,
"errors": 0,
"client_ports": set(),
"bytes_sent": 0,
})
return {
"run_id": run_id,
"requests": s["requests"],
"errors": s["errors"],
"unique_client_ports": len(s["client_ports"]),
"bytes_sent": s["bytes_sent"],
}
return {
rid: {
"requests": s["requests"],
"errors": s["errors"],
"unique_client_ports": len(s["client_ports"]),
"bytes_sent": s["bytes_sent"],
}
for rid, s in run_state.items()
}
def reset(run_id=None):
with lock:
if run_id is None:
run_state.clear()
else:
run_state.pop(run_id, None)
def write_event(event):
if event_log is None:
return
with lock:
with event_log.open("a", encoding="utf-8") as handle:
handle.write(json.dumps(event, sort_keys=True) + "\n")
class Handler(BaseHTTPRequestHandler):
protocol_version = "HTTP/1.1"
def send_json(self, status, payload):
raw = json.dumps(payload, sort_keys=True).encode("utf-8")
self.send_response(status)
self.send_header("Content-Type", "application/json")
self.send_header("Content-Length", str(len(raw)))
self.send_header("X-Fixture-Version", FIXTURE_VERSION)
self.end_headers()
self.wfile.write(raw)
return len(raw)
def do_POST(self):
parsed = urlparse(self.path)
if parsed.path != "/reset":
self.send_json(404, {"status": "not_found"})
return
q = parse_qs(parsed.query)
run_id = q.get("run_id", [None])[0]
if run_id is not None and not SAFE.fullmatch(run_id):
self.send_json(400, {"status": "invalid_run_id"})
return
reset(run_id)
self.send_json(200, {"status": "reset", "run_id": run_id, "epoch_ms": now_ms()})
def do_GET(self):
started = now_ms()
parsed = urlparse(self.path)
if parsed.path == "/health":
self.send_json(200, {
"status": "ok",
"fixture_version": FIXTURE_VERSION,
"epoch_ms": now_ms(),
})
return
if parsed.path == "/stats":
q = parse_qs(parsed.query)
run_id = q.get("run_id", [None])[0]
self.send_json(200, {
"fixture_version": FIXTURE_VERSION,
"epoch_ms": now_ms(),
"state": snapshot(run_id),
})
return
if parsed.path != "/work":
self.send_json(404, {"status": "not_found"})
return
q = parse_qs(parsed.query)
run_id = q.get("run_id", [""])[0]
thread_id = q.get("thread", [""])[0]
seq_raw = q.get("seq", [""])[0]
payload_raw = q.get("payload", ["8192"])[0]
if not SAFE.fullmatch(run_id) or not SAFE.fullmatch(thread_id):
self.send_json(400, {"status": "invalid_metadata"})
return
try:
seq = int(seq_raw)
payload_bytes = int(payload_raw)
except ValueError:
self.send_json(400, {"status": "invalid_number"})
return
if not 1 <= seq <= 100000 or not 0 <= payload_bytes <= 65536:
self.send_json(400, {"status": "out_of_bounds"})
return
time.sleep(0.010)
padding = "X" * payload_bytes
payload = {
"status": "ok",
"run_id": run_id,
"thread": thread_id,
"seq": seq,
"payload_bytes": payload_bytes,
"padding": padding,
}
raw_len = self.send_json(200, payload)
ended = now_ms()
with lock:
if run_id not in run_state:
run_state[run_id] = {
"requests": 0,
"errors": 0,
"client_ports": set(),
"bytes_sent": 0,
}
s = run_state[run_id]
s["requests"] += 1
s["client_ports"].add(int(self.client_address[1]))
s["bytes_sent"] += raw_len
write_event({
"ts_ms": ended,
"operation": "work",
"status": 200,
"run_id": run_id,
"thread": thread_id,
"seq": seq,
"client_ip": self.client_address[0],
"client_port": int(self.client_address[1]),
"connection_header": self.headers.get("Connection", ""),
"payload_bytes": payload_bytes,
"service_wall_ms": ended - started,
})
def log_message(self, format, *args):
return
def main():
parser = argparse.ArgumentParser()
parser.add_argument("--host", default="127.0.0.1")
parser.add_argument("--port", type=int, default=8026)
parser.add_argument("--log", default="results/server-events.jsonl")
args = parser.parse_args()
global event_log
event_log = Path(args.log).resolve()
event_log.parent.mkdir(parents=True, exist_ok=True)
event_log.write_text("", encoding="utf-8")
print(f"fixture_version={FIXTURE_VERSION}")
print(f"listen=http://{args.host}:{args.port}")
print(f"event_log={event_log}")
ThreadingHTTPServer((args.host, args.port), Handler).serve_forever()
if __name__ == "__main__":
main()
Start:
python .\fixtures\generator_fixture.py `
--host 127.0.0.1 `
--port 8026 `
--log .\results\server-events.jsonl
The HTTP/1.1 server records each work request's client source port. Many requests on one source port imply TCP reuse; one/few requests per source port imply churn. Target service time is intentionally about 10 ms and response size is bounded to ≤64 KiB.
3. Create the conservative properties
config/generator-local.properties:
# Prompt 26 conservative generator profile
target.host=127.0.0.1
target.port=8026
threads=10
loops=100
pacing.ms=100
payload.bytes=8192
connect.timeout.ms=500
response.timeout.ms=2000
# Lean result contract.
jmeter.save.saveservice.output_format=csv
jmeter.save.saveservice.print_field_names=true
jmeter.save.saveservice.response_data=false
jmeter.save.saveservice.response_data.on_error=false
jmeter.save.saveservice.samplerData=false
jmeter.save.saveservice.responseHeaders=false
jmeter.save.saveservice.requestHeaders=false
jmeter.save.saveservice.url=false
jmeter.save.saveservice.bytes=true
jmeter.save.saveservice.sent_bytes=true
jmeter.save.saveservice.thread_counts=true
jmeter.save.saveservice.assertion_results_failure_message=true
# Make HTTP implementation explicit.
jmeter.httpsampler=HttpClient4
httpclient4.retrycount=0
State changed: JMeter will use explicit HttpClient4 with no retry and a lean CSV result schema. No heap/OS/network setting changes yet.
4. Author two JMX files differing by one checkbox
plans/generator-close.jmx:
Test Plan
├── HTTP Request Defaults
│ host=${__P(target.host,127.0.0.1)}
│ port=${__P(target.port,8026)}
│ implementation=HttpClient4
│ connect/response timeouts from properties
└── Thread Group
threads=${__P(threads,1)}
loops=${__P(loops,1)}
├── Counter -> SEQ (per user)
└── HTTP Request — Work — CONNECTION CHURN BASELINE
GET /work
run_id=${__P(run.id,p26-local)}
thread=T${__threadNum}
seq=${SEQ}
payload=${__P(payload.bytes,8192)}
Use KeepAlive = unchecked
├── Constant Timer ${__P(pacing.ms,100)} ms
└── Response Assertion: response code = 200
Copy it to plans/generator-keepalive.jmx and make
exactly one behavioral change:
Test Plan
├── HTTP Request Defaults
│ host=${__P(target.host,127.0.0.1)}
│ port=${__P(target.port,8026)}
│ implementation=HttpClient4
└── Thread Group
threads=${__P(threads,1)}
loops=${__P(loops,1)}
├── Counter -> SEQ (per user)
└── HTTP Request — Work — NORMAL REUSE
same URL/properties/timer/assertion as baseline
Use KeepAlive = checked
Hash both plans and record the difference. Every workload/property/result setting must otherwise remain identical.
5. Smoke with 1×5 before profiling
& "$env:JMETER_HOME\bin\jmeter.bat" `
-n -t .\plans\generator-close.jmx `
-q .\config\generator-local.properties `
-Jthreads=1 -Jloops=5 -Jrun.id=p26-smoke `
-l .\results\p26-smoke\results.jtl `
-j .\results\p26-smoke\jmeter.log
Require five successes and five target events. Clear only the
p26-smoke target state if desired; do not erase the
event log before review.
6. Record hardware/JVM/OS inventory
Windows:
Get-CimInstance Win32_Processor |
Select-Object Name,NumberOfCores,NumberOfLogicalProcessors
Get-CimInstance Win32_ComputerSystem |
Select-Object TotalPhysicalMemory
Get-Volume | Select-Object DriveLetter,SizeRemaining,Size
Get-NetAdapter | Where-Object Status -eq "Up" |
Select-Object Name,LinkSpeed,MacAddress
& "$env:JMETER_HOME\bin\jmeter.bat" -v
java -version
Linux:
lscpu
free -h
df -h .
ip -br link
"$JMETER_HOME/bin/jmeter" -v
java -version
This inventory is descriptive, not a formula for thread limits.
7. Reset target and start the connection-churn baseline
curl.exe --fail --silent -X POST `
"http://127.0.0.1:8026/reset?run_id=p26-close"
New-Item -ItemType Directory -Force .\results\p26-close | Out-Null
& "$env:JMETER_HOME\bin\jmeter.bat" `
-n -t .\plans\generator-close.jmx `
-q .\config\generator-local.properties `
-Jrun.id=p26-close `
-l .\results\p26-close\results.jtl `
-j .\results\p26-close\jmeter.log
Run monitoring commands in another terminal during the ~10–15 second test.
8. Windows monitoring during the baseline
# 1) Find the JMeter JVM PID (inspect CommandLine before using it).
Get-CimInstance Win32_Process |
Where-Object { $_.Name -eq "java.exe" -and $_.CommandLine -like "*ApacheJMeter.jar*" } |
Select-Object ProcessId, CommandLine
# 2) JVM / process snapshots. Replace <PID>.
jcmd <PID> VM.version
jcmd <PID> VM.flags
jcmd <PID> GC.heap_info
jstat -gcutil <PID> 1000 12
Get-Process -Id <PID> |
Select-Object Id,CPU,WorkingSet64,PrivateMemorySize64,Handles,Threads
# 3) TCP/socket state owned by the JVM.
Get-NetTCPConnection -OwningProcess <PID> -ErrorAction SilentlyContinue |
Group-Object State | Select-Object Name,Count
# 4) Inspect the OS dynamic TCP port range; DO NOT change it in this lab.
netsh int ipv4 show dynamicport tcp
netsh int ipv6 show dynamicport tcp
# 5) DNS state / lookup path.
Resolve-DnsName localhost
Get-DnsClientCache | Select-Object -First 20
# 6) Network adapter counters.
Get-NetAdapterStatistics |
Select-Object Name,ReceivedBytes,SentBytes,ReceivedDiscardedPackets,OutboundDiscardedPackets
# 7) Disk/result artifact.
Get-Item .\results\<run>\results.jtl |
Select-Object FullName,Length,LastWriteTime
Get-Volume | Select-Object DriveLetter,SizeRemaining,Size
Capture at least two process snapshots plus one
socket-state/dynamic-port/NIC snapshot. Do not run
netsh ... set dynamicport, disable firewall/antivirus,
or change registry TCP settings.
9. Linux monitoring during the baseline
# Identify the JMeter JVM.
jps -lv
# JVM heap/GC (replace PID).
jcmd PID VM.version
jcmd PID VM.flags
jcmd PID GC.heap_info
jstat -gcutil PID 1000 12
# Process CPU/memory.
ps -p PID -o pid,pcpu,pmem,rss,vsz,nlwp,etime,cmd
pidstat -p PID 1 10 # if sysstat/pidstat is installed
# File descriptors and sockets.
ls /proc/PID/fd | wc -l
ss -tanp
ulimit -n
cat /proc/sys/fs/file-max
# Dynamic local port range (read-only).
sysctl net.ipv4.ip_local_port_range
# DNS state.
getent hosts localhost
resolvectl statistics # where systemd-resolved is present
# Network/disk state.
ip -s link
df -h .
du -h results/*/results.jtl
The POSIX file-descriptor limit and Linux ephemeral-port range are
inspection data only. Do not run ulimit -n changes or
sysctl -w in the mandatory lab.
10. Interpret heap/GC evidence
Record:
- effective heap/GC flags from
jcmd VM.flags; -
heap occupancy before/near end from
GC.heap_info; -
jstat -gcutilyoung/full GC counts and cumulative GC time; - process CPU seconds/working set/thread count.
Do not call GC.run or take a large heap dump during the
measurement run; those diagnostic actions can perturb the JVM.
11. Analyze JTL/target evidence
Save tools/profile_analyzer.py:
import csv
import json
import math
import sys
from pathlib import Path
def percentile(values, pct):
data = sorted(values)
if not data:
return 0
idx = max(0, min(len(data)-1, math.ceil(len(data) * pct / 100.0) - 1))
return data[idx]
def summarize_jtl(path):
rows = list(csv.DictReader(Path(path).open(newline="", encoding="utf-8")))
if not rows:
return {"samples": 0}
elapsed = [int(float(r["elapsed"])) for r in rows]
start = min(int(r["timeStamp"]) for r in rows)
end = max(int(r["timeStamp"]) + int(float(r["elapsed"])) for r in rows)
span_s = max((end - start)/1000.0, 0.001)
return {
"samples": len(rows),
"failures": sum(r["success"].lower() != "true" for r in rows),
"span_s": round(span_s, 3),
"achieved_rps": round(len(rows)/span_s, 3),
"avg_ms": round(sum(elapsed)/len(elapsed), 3),
"p95_nearest_rank_ms": percentile(elapsed, 95),
"received_bytes": sum(int(float(r.get("bytes") or 0)) for r in rows),
"sent_bytes": sum(int(float(r.get("sentBytes") or 0)) for r in rows),
"jtl_file_bytes": Path(path).stat().st_size,
}
if len(sys.argv) < 2:
raise SystemExit("usage: profile_analyzer.py <jtl> [jtl ...]")
for raw in sys.argv[1:]:
print(json.dumps({"path": raw, "jtl": summarize_jtl(raw)}, indent=2))
Save tools/analyze_target.py:
import json
import sys
from collections import Counter
from pathlib import Path
if len(sys.argv) != 3:
raise SystemExit("usage: analyze_target.py <server-events.jsonl> <run_id>")
path = Path(sys.argv[1])
run_id = sys.argv[2]
events = [
json.loads(line)
for line in path.read_text(encoding="utf-8").splitlines()
if line.strip()
]
work = [e for e in events if e.get("operation") == "work" and e.get("run_id") == run_id]
ports = [int(e["client_port"]) for e in work]
services = [int(e["service_wall_ms"]) for e in work]
print(json.dumps({
"run_id": run_id,
"events": len(work),
"statuses": dict(Counter(e.get("status") for e in work)),
"unique_client_ports": len(set(ports)),
"connection_headers": dict(Counter(e.get("connection_header") for e in work)),
"avg_service_wall_ms": round(sum(services)/len(services), 3) if services else 0,
"max_service_wall_ms": max(services) if services else 0,
"payload_bytes_values": sorted({e.get("payload_bytes") for e in work}),
}, indent=2))
python tools/profile_analyzer.py results/p26-close/results.jtl
python tools/analyze_target.py results/server-events.jsonl p26-close
Expected: 1,000 JTL rows, zero failures, stable ~10 ms target service time, and a high unique-client-port count because KeepAlive is disabled.
12. Identify the first observed generator pressure
At this deliberately bounded scale, the baseline is designed to make socket/local-port churn the dominant observable pressure, not to force actual OS exhaustion. Confirm that:
- unique client ports are a large fraction of the 1,000 requests;
- TIME_WAIT/connection state grows during/after the run;
- target service time remains stable;
- heap/GC and CPU are not the first hard failure;
- JTL/disk size remains small/lean.
If your machine instead shows CPU/GC/disk pressure first, record that reality and do not force the expected diagnosis.
13. Make one reversible plan change
Use generator-keepalive.jmx, where only
Use KeepAlive is checked. This changes HTTP
connection/session behavior; it does not change threads, loops,
pacing, target, payload, assertion or JTL schema.
14. Rerun identical workload
curl.exe --fail --silent -X POST `
"http://127.0.0.1:8026/reset?run_id=p26-keepalive"
New-Item -ItemType Directory -Force .\results\p26-keepalive | Out-Null
& "$env:JMETER_HOME\bin\jmeter.bat" `
-n -t .\plans\generator-keepalive.jmx `
-q .\config\generator-local.properties `
-Jrun.id=p26-keepalive `
-l .\results\p26-keepalive\results.jtl `
-j .\results\p26-keepalive\jmeter.log
Capture the same JVM/process/socket/NIC/JTL evidence at comparable times.
15. Compare before/after
Save tools/compare_profiles.py:
import csv
import json
import math
import sys
from pathlib import Path
def p95(values):
values = sorted(values)
if not values:
return 0
return values[max(0, min(len(values)-1, math.ceil(len(values)*0.95)-1))]
def jtl(path):
rows = list(csv.DictReader(Path(path).open(newline="", encoding="utf-8")))
start = min(int(r["timeStamp"]) for r in rows)
end = max(int(r["timeStamp"]) + int(float(r["elapsed"])) for r in rows)
elapsed = [int(float(r["elapsed"])) for r in rows]
return {
"samples": len(rows),
"failures": sum(r["success"].lower() != "true" for r in rows),
"rps": len(rows)/max((end-start)/1000.0, 0.001),
"avg_ms": sum(elapsed)/len(elapsed),
"p95_ms": p95(elapsed),
"jtl_bytes": Path(path).stat().st_size,
}
if len(sys.argv) != 3:
raise SystemExit("usage: compare_profiles.py close.jtl keepalive.jtl")
a = jtl(sys.argv[1])
b = jtl(sys.argv[2])
print(json.dumps({
"connection_close": a,
"keepalive": b,
"delta_keepalive_minus_close": {
"rps": round(b["rps"]-a["rps"], 3),
"avg_ms": round(b["avg_ms"]-a["avg_ms"], 3),
"p95_ms": round(b["p95_ms"]-a["p95_ms"], 3),
"jtl_bytes": b["jtl_bytes"]-a["jtl_bytes"],
}
}, indent=2))
python tools/compare_profiles.py results/p26-close/results.jtl results/p26-keepalive/results.jtl
python tools/analyze_target.py results/server-events.jsonl p26-close
python tools/analyze_target.py results/server-events.jsonl p26-keepalive
The strongest expected change is connection count/port churn, not necessarily a dramatic RPS gain at only 10 threads. Report exactly what improved: fewer TCP connections/TIME_WAIT, lower connect overhead, CPU change if measurable, achieved RPS/latency change if measurable, and unchanged target service time.
16. Measure JTL/disk cost explicitly
For each run record file bytes and
bytes per sample = JTL bytes / 1000. The response body
is 8 KiB but is not written into lean CSV. If you enable verbose
XML/body retention for a tiny separate debug run, cap it at
1×20 and immediately restore the lean profile; do not mix that
diagnostic run into the keepalive capacity comparison.
17. Inspect DNS without contaminating the benchmark
The benchmark uses an IP. Inspect localhost resolution
separately with Resolve-DnsName/Get-DnsClientCache or
getent/resolvectl. If production behavior requires
per-thread DNS rotation, model it with DNS Cache Manager in a
separate controlled experiment rather than flushing the host/JVM
resolver mid-run.
18. Challenge
Your keepalive run shows 20 unique client ports instead of ~1,000, but achieved RPS is unchanged and CPU differs by only 1%. Did “capacity double”?
No. Connection pressure/headroom improved, but this workload did not measure a higher maximum sustainable load. State the measured reduction in socket churn and retain a separate, safely stepped capacity experiment if a maximum is required.
Knowledge check
Why is keep-alive a good checkpoint change?
It is a small reversible test-plan change that directly changes connection reuse without requiring privileged OS tuning.
Why can unique client ports validate connection churn?
Each new TCP connection normally uses a local source port; reused persistent connections issue many requests on fewer ports.
What evidence prevents you from blaming the target?
Stable target service_wall_ms plus generator/socket/JVM evidence shows the changed pressure is on the injector connection path.
Why avoid a heap dump during the benchmark?
Heap dumps/full diagnostic actions can pause or materially perturb the JVM and invalidate timing/throughput evidence.
Why is unchanged RPS not proof the tuning was useless?
Resource headroom/port churn may improve even when the bounded workload was already below both configurations' throughput limits.
Official references and version notes
- Apache JMeter downloads — current stable release and Java requirement.
- Apache JMeter current changes — Java guidance for the 5.6.x line.
- JMeter Getting Started — load-generator sizing, CLI guidance, default heap/GC launcher settings and JVM environment variables.
- JMeter Best Practices — effective thread count, CLI execution, listener/result minimization, CSV output and generator resource reduction.
- Component Reference — HTTP Request — HttpClient4 default, keep-alive behavior and response-body MD5 option.
- Component Reference — DNS Cache Manager — JVM DNS cache behavior and HttpClient4-only per-thread DNS control.
- JMeter Properties Reference — HttpClient4 retry, connection TTL/validation and result-save properties.
-
JDK 17
jcmd— JVM process/heap inspection. -
JDK 17
jstat— GC/heap utilization statistics; documented as experimental/unsupported.
Version-sensitive statements were rechecked against current
primary documentation on 2026-09-05. The course baseline remains
Apache JMeter 5.6.3 with a Java 17 JDK. JMeter
5.6.3 requires Java 8+; the 5.6.x changes page recommends Java 17
or later. Current JMeter launcher scripts default to a 1 GiB heap
(-Xms1g -Xmx1g plus a 256 MiB metaspace cap) and G1GC
with
-XX:MaxGCPauseMillis=250/-XX:G1ReservePercent=20. Those defaults are a
starting point, not a universal sizing rule. JMeter's HTTP sampler
default is HttpClient4. Its retry count defaults to 0; its
documented connection TTL defaults to 60 seconds. The HTTP Request
Use KeepAlive option is effective with the Apache
HttpComponents implementation and is the only plan change used in
the checkpoint. JMeter's own best-practice guidance says effective
thread capacity depends on hardware, plan design and how fast the
target responds; CLI mode, minimal listeners, CSV and only
required fields reduce generator cost. The mandatory lab never
changes OS ephemeral-port ranges, file-descriptor limits,
firewall/antivirus state or global DNS/JDK security settings.
Keep the academy open
Support free, practical DevOps education.
Every lesson is designed to remain readable in a browser, downloadable from GitHub, and usable without a paid learning platform. Contributions help expand and maintain the curriculum.
0x716c4Ab160C4B66F31a28AE2448BfF68fc3a2ef0
Send only Ethereum/ERC-20 compatible assets to this
address.