Regular Expressions, CSS/JQuery, XPath, JSONPath, and Boundary Extraction: Guided Hands-On Workflow
The lab exposes the same business concept—a primary synthetic ID—in JSON, XML, HTML, and plain text. Each format uses the selector that best matches its structure, while the plain-text response is extracted twice to compare Boundary and Regex approaches.
Learning objectives
- Run a local multi-format fixture and inspect its synthetic responses.
- Extract primary IDs with JSONPath, XPath2, CSS Selector, Boundary, and Regex components.
- Compare first-match and all-match behavior using candidate lists.
- Use explicit sentinel defaults and fail before a downstream request when correlation is missing.
-
Verify each extracted value independently through a local
/verifyendpoint. - Measure small-scale extractor overhead with JTL, wall-clock, and generator observations.
1. Safety envelope
http://127.0.0.1:8000.
Normal functional workflow uses 1 thread × 2 loops. Multi-user proof
uses at most 2 threads × 2 loops. Overhead profiles use at most 1
thread × 20 loops. No profile exceeds 40 target requests. Abort on
target mismatch, unexpected external URL, unexpected errors, or
unsafe generator pressure.
2. Start the multi-format fixture
Save as fixtures/extraction_fixture.py:
from http.server import BaseHTTPRequestHandler, ThreadingHTTPServer
from urllib.parse import urlparse, parse_qs
from pathlib import Path
import argparse
import json
import threading
import time
lock = threading.Lock()
active = 0
max_active = 0
total = 0
by_path = {}
event_log = None
EXPECTED = {
"json": "JSON-PRIMARY-100",
"xml": "XML-PRIMARY-200",
"html": "HTML-PRIMARY-300",
"text": "TEXT-PRIMARY-400",
}
def record(event):
if event_log is None:
return
with lock:
with event_log.open("a", encoding="utf-8") as handle:
handle.write(json.dumps(event, sort_keys=True) + "\n")
def json_body(obj):
return json.dumps(obj, sort_keys=True).encode("utf-8")
class Handler(BaseHTTPRequestHandler):
protocol_version = "HTTP/1.1"
def _send(self, status, body, content_type):
if isinstance(body, str):
body = body.encode("utf-8")
self.send_response(status)
self.send_header("Content-Type", content_type)
self.send_header("Content-Length", str(len(body)))
self.end_headers()
self.wfile.write(body)
def _track_start(self, path):
global active, max_active, total
with lock:
total += 1
active += 1
max_active = max(max_active, active)
by_path[path] = by_path.get(path, 0) + 1
request_no = total
active_now = active
return request_no, active_now
def _track_finish(self, path, request_no, active_now, status, started_ms, extra=None):
global active
finished_ms = int(time.time() * 1000)
event = {
"path": path,
"request_no": request_no,
"status": status,
"started_ms": started_ms,
"finished_ms": finished_ms,
"service_wall_ms": finished_ms - started_ms,
"active_at_start": active_now,
}
if extra:
event.update(extra)
record(event)
with lock:
active -= 1
def do_GET(self):
path = urlparse(self.path).path
if path == "/health":
self._send(200, json_body({"status": "ok"}), "application/json")
return
if path == "/stats":
with lock:
snap = {
"active": active,
"max_active": max_active,
"total": total,
"by_path": dict(by_path),
}
self._send(200, json_body(snap), "application/json")
return
request_no, active_now = self._track_start(path)
started_ms = int(time.time() * 1000)
status = 200
extra = {}
try:
time.sleep(0.02)
if path == "/json":
payload = {
"primary": {"id": EXPECTED["json"]},
"candidates": [
{"id": "JSON-CANDIDATE-A"},
{"id": "JSON-CANDIDATE-B"},
],
"meta": {"format": "json"},
}
self._send(200, json_body(payload), "application/json")
elif path == "/xml":
payload = f"""<?xml version="1.0" encoding="UTF-8"?>
<response>
<primary><id>{EXPECTED["xml"]}</id></primary>
<candidates>
<id>XML-CANDIDATE-A</id>
<id>XML-CANDIDATE-B</id>
</candidates>
</response>"""
self._send(200, payload, "application/xml")
elif path == "/xml-v2":
# Same business meaning, intentionally changed structure.
payload = f"""<?xml version="1.0" encoding="UTF-8"?>
<response>
<data>
<primary><id>{EXPECTED["xml"]}</id></primary>
</data>
<candidates>
<id>XML-CANDIDATE-A</id>
<id>XML-CANDIDATE-B</id>
</candidates>
</response>"""
self._send(200, payload, "application/xml")
elif path == "/html":
payload = f"""<!doctype html>
<html>
<body>
<main id="result">
<div id="primary" data-id="{EXPECTED["html"]}">Primary record</div>
<ul id="candidates">
<li class="candidate" data-id="HTML-CANDIDATE-A">A</li>
<li class="candidate" data-id="HTML-CANDIDATE-B">B</li>
</ul>
</main>
</body>
</html>"""
self._send(200, payload, "text/html; charset=utf-8")
elif path == "/text":
payload = (
f"BEGIN;PRIMARY_ID={EXPECTED['text']};"
"CANDIDATE_ID=TEXT-CANDIDATE-A;"
"CANDIDATE_ID=TEXT-CANDIDATE-B;END"
)
self._send(200, payload, "text/plain; charset=utf-8")
elif path == "/verify":
q = parse_qs(urlparse(self.path).query)
source = q.get("source", [""])[-1]
supplied = q.get("id", [""])[-1]
expected = EXPECTED.get(source)
ok = expected is not None and supplied == expected
status = 200 if ok else 400
extra = {"source": source, "supplied": supplied, "expected": expected}
self._send(
status,
json_body({
"status": "accepted" if ok else "rejected",
"source": source,
"supplied_id": supplied,
"expected_id": expected,
}),
"application/json",
)
else:
status = 404
self._send(404, json_body({"error": "not_found", "path": path}), "application/json")
finally:
self._track_finish(path, request_no, active_now, status, started_ms, extra)
def log_message(self, format, *args):
return
if __name__ == "__main__":
parser = argparse.ArgumentParser()
parser.add_argument("--log", default="results/server-events.jsonl")
args = parser.parse_args()
event_log = Path(args.log).resolve()
event_log.parent.mkdir(parents=True, exist_ok=True)
event_log.write_text("", encoding="utf-8")
print("fixture=http://127.0.0.1:8000")
print(f"event_log={event_log}")
ThreadingHTTPServer(("127.0.0.1", 8000), Handler).serve_forever()
Start it:
python fixtures/extraction_fixture.py --log results/server-events.jsonl
curl --fail --silent http://127.0.0.1:8000/health
curl --fail --silent http://127.0.0.1:8000/json
curl --fail --silent http://127.0.0.1:8000/xml
curl --fail --silent http://127.0.0.1:8000/html
curl --fail --silent http://127.0.0.1:8000/text
3. Authoring tree
Thread Group — 1 user × 2 loops
├── JSON Source — GET /json
│ └── JSON Extractor
├── Verify JSON — GET /verify?source=json&id=${JSON_ID}
├── XML Source — GET /xml
│ └── XPath2 Extractor
├── Verify XML — GET /verify?source=xml&id=${XML_ID}
├── HTML Source — GET /html
│ └── CSS Selector Extractor
├── Verify HTML — GET /verify?source=html&id=${HTML_ID}
├── Text Source — GET /text
│ ├── Boundary Extractor
│ └── Regular Expression Extractor
├── Verify Text Boundary — GET /verify?source=text&id=${TEXT_BOUNDARY_ID}
└── Verify Text Regex — GET /verify?source=text&id=${TEXT_REGEX_ID}
Authoring only: View Results Tree + Debug Sampler
Every extractor is a child of the source sampler whose response it parses. Verification requests are separate samplers so downstream variable use is visible.
4. JSONPath extraction
Under JSON Source, add JSON Extractor:
| Setting | Value |
|---|---|
| Variable | JSON_ID |
| JSONPath | $.primary.id |
| Default | __NOT_FOUND__ |
| Match Number | 1 |
Expected variable: JSON-PRIMARY-100. Verify JSON should
return HTTP 200/accepted.
For all candidates, make a debug copy with expression
$.candidates[*].id and Match -1. Inspect
the numbered variables/count, then revert to the primary selector
before load.
5. XPath2 extraction
Under XML Source, add XPath2 Extractor:
| Setting | Value |
|---|---|
| Variable | XML_ID |
| XPath2 Query | //primary/id/text() |
| Return fragment | Off |
| Match Number | 1 |
| Default | __NOT_FOUND__ |
Expected: XML-PRIMARY-200. The relative structural
selector survives the later wrapper change used in the failure lab.
6. Tiny legacy XPath comparison
For one authoring-only sample, add the legacy XPath Extractor with the same query and compare the value. Then remove/disable it. Current JMeter guidance since 5.0 prefers XPath2, so the legacy component is taught for recognition/migration—not as the course default.
7. CSS Selector extraction
Under HTML Source, add CSS Selector Extractor:
| Setting | Value |
|---|---|
| Implementation | Leave blank → current default JSoup |
| Variable | HTML_ID |
| Selector | #primary[data-id] |
| Attribute | data-id |
| Match Number | 1 |
| Default | __NOT_FOUND__ |
Expected: HTML-PRIMARY-300. For candidates, use
selector #candidates .candidate[data-id], attribute
data-id, and negative/all-match mode in a debug copy.
8. Boundary extraction from text
Under Text Source, add Boundary Extractor:
| Setting | Value |
|---|---|
| Variable | TEXT_BOUNDARY_ID |
| Left boundary | PRIMARY_ID= |
| Right boundary | ; |
| Match | 1 |
| Default | __NOT_FOUND__ |
Expected: TEXT-PRIMARY-400.
9. Regex extraction from the same text
Add Regular Expression Extractor under Text Source:
| Setting | Value |
|---|---|
| Variable | TEXT_REGEX_ID |
| Field | Body |
| Regular expression | PRIMARY_ID=([A-Z0-9-]+); |
| Template | $1$ |
| Match | 1 |
| Default | __NOT_FOUND__ |
It should produce the same TEXT-PRIMARY-400. For this
stable delimiter case, Boundary is simpler; regex becomes valuable
when the value itself requires a pattern constraint.
10. Assert extraction succeeded before downstream use
During authoring, attach a Response Assertion after each
source/extractor path that evaluates the created JMeter variable and
rejects the sentinel __NOT_FOUND__. Post-Processors run
before Assertions, so the assertion can validate the just-created
variable.
Do not let /verify?id=__NOT_FOUND__ become the first
place you discover correlation failure in a large run.
11. Match-number experiment
Each response has two candidate IDs. In separate debug copies, compare:
- Match
1: deterministic first candidate. - Match
2: deterministic second candidate. -
Match
0: random candidate—use only when randomness is intentional. -
All-match / negative/
-1mode: numbered variables plus match count.
Record the exact variables produced by each component. Do not assume every extractor creates identical auxiliary-variable names.
12. Observe structure change before repairing it
The fixture's /xml-v2 adds a harmless
<data> wrapper around
<primary>. Compare:
Brittle absolute XPath: /response/primary/id/text()
Robust-for-this-contract XPath2: //primary/id/text()
Switch only the XML sampler from /xml to
/xml-v2. The absolute selector should hit the sentinel;
the relative selector still resolves the same business field.
13. Meaningful CLI run
Disable View Results Tree/Debug Sampler. Run the 1-thread × 2-loop extraction/verification plan:
mkdir -p results/functional
jmeter -n -t plans/extractors-load.jmx -l results/functional/results.jtl -j results/functional/jmeter.log -Jjmeter.save.saveservice.print_field_names=true -Jjmeter.save.saveservice.thread_counts=true
python tools/analyze_extraction.py results/functional/results.jtl
curl --fail --silent http://127.0.0.1:8000/stats
PowerShell uses jmeter.bat, backtick continuation, and
Invoke-WebRequest for stats.
14. Small-scale extractor overhead method
For each format, compare two otherwise-identical 1-thread × 20-loop plans:
- baseline source request with no extractor;
- source request with exactly one primary-value extractor.
Save unique JTL/jmeter.log, record whole-command wall
time and Task Manager/top CPU/memory, then compare:
import csv
import math
import sys
from pathlib import Path
def summarize(path):
rows = list(csv.DictReader(Path(path).open(encoding="utf-8")))
if not rows:
raise SystemExit(f"No rows in {path}")
starts = [int(r["timeStamp"]) - int(r["elapsed"]) for r in rows]
ends = [int(r["timeStamp"]) for r in rows]
span_s = max((max(ends)-min(starts))/1000.0, 0.001)
elapsed = sorted(int(r["elapsed"]) for r in rows)
def p(pct):
rank = max(1, math.ceil((pct/100)*len(elapsed)))
return elapsed[rank-1]
return {
"samples": len(rows),
"failures": sum(r["success"].lower() != "true" for r in rows),
"sample_rps": len(rows)/span_s,
"p50_ms": p(50),
"p95_ms": p(95),
}
if len(sys.argv) != 3:
raise SystemExit("usage: compare_runs.py BASELINE.jtl EXTRACTOR.jtl")
a = summarize(sys.argv[1])
b = summarize(sys.argv[2])
print(f"baseline={a}")
print(f"extractor={b}")
if a["sample_rps"]:
print(f"sample_rps_delta_pct={100*(b['sample_rps']-a['sample_rps'])/a['sample_rps']:.2f}")
print(
"Use external wall-clock and generator CPU/GC observations too. "
"Sampler elapsed does not directly include post-processor extraction cost."
)
Do not rank JSONPath versus XPath2 versus CSS by raw wall time unless payload size/shape and run conditions are controlled. The valid claim is whether each extractor adds material injector cost to its own baseline at this tiny scale.
15. Challenge
An HTML page contains a hidden form input
<input name="csrf" value="SYNTHETIC">. A teammate
proposes regex value="(.*?)". What is the better
default?
Use CSS Selector scoped to the actual element, for example
input[name=csrf] with attribute value. The
regex is broad and can accidentally capture another element's value.
Knowledge check
Why does JSON Extractor use $.primary.id instead of a regex for the JSON body?
The JSONPath follows the structured field relationship and is more robust to whitespace/member formatting changes.
Why is CSS Selector preferred for the HTML fixture?
It targets DOM elements/attributes directly, and current JMeter docs recommend CSS Selector Extractor for HTML.
What does the sentinel __NOT_FOUND__ provide?
A visible fail-closed/debug value when a selector finds no match, preventing silent reuse or confusing literal-variable behavior.
Why can all-match mode be dangerous in a hot path?
It can parse/store an unbounded number of matches and create extra variables/allocations when only one value is needed.
What evidence should accompany an extractor-overhead claim?
Same workload baseline versus extractor run, JTL and jmeter.log, whole-run wall time, generator CPU/memory/GC observations, and target timing/counts.
Official references and version notes
- Component Reference — current Regular Expression, CSS Selector, XPath2/XPath, JSON JMESPath, JSON, and Boundary Extractor semantics.
- Elements of a Test Plan — Post-Processor scope/execution and thread-local JMeter variables.
- Regular Expressions — JMeter regular-expression guidance and extractor examples.
- Best Practices — non-GUI load execution, lean listeners, generator validity, and scripting guidance.
- Apache JMeter downloads — current production release and Java requirement.
Version-sensitive behavior was rechecked against current Apache
JMeter documentation on 2026-09-05. The course baseline remains
Apache JMeter 5.6.3 with a Java 17 JDK for labs
and no third-party plugins; JMeter 5.6.3 requires Java 8+. The
current HTML component is named
CSS Selector Extractor (formerly CSS/JQuery
Extractor) and supports JSoup and Jodd-Lagarto implementations,
with JSoup the default when no implementation is selected. For
HTML, current JMeter documentation recommends CSS Selector
Extractor rather than XPath. Since JMeter 5.0, the documentation
recommends XPath2 Extractor over the legacy XPath
Extractor because of easier namespace handling, better
performance, and XPath 2.0 support.
JSON Extractor uses JSONPath syntax;
JSON JMESPath Extractor is also a current
built-in alternative. Regular Expression and Boundary Extractors
can process text/body/header fields and expose
match-number/default behavior. For several extractors, match
0 selects a random match, a positive number selects
the Nth match, and negative/-1 modes expose all
matches through numbered variables and a match-count variable.
Defaults are useful during debugging, but a silent permissive
default must not be allowed to masquerade as successful
correlation. All meaningful load runs preserve both raw JTL and
matching jmeter.log.
Keep the academy open
Support free, practical DevOps education.
Every lesson is designed to remain readable in a browser, downloadable from GitHub, and usable without a paid learning platform. Contributions help expand and maintain the curriculum.
0x716c4Ab160C4B66F31a28AE2448BfF68fc3a2ef0
Send only Ethereum/ERC-20 compatible assets to this
address.