|
#!/usr/bin/env bash |
|
|
|
set -euo pipefail |
|
|
|
script_name="$(basename "$0")" |
|
|
|
usage() { |
|
cat <<EOF |
|
Profile-driven HTTP stress test using curl and a weighted TSV request profile. |
|
|
|
Usage: |
|
bash ${script_name} --base-url URL --profile FILE [options] |
|
|
|
Examples: |
|
bash ${script_name} \ |
|
--base-url https://example.com \ |
|
--profile ./profile.tsv |
|
|
|
bash ${script_name} \ |
|
--base-url https://catalog.data.gov \ |
|
--profile tools/profiles/catalog-realistic.tsv |
|
|
|
bash ${script_name} \ |
|
--base-url https://catalog.data.gov \ |
|
--profile tools/profiles/catalog-realistic.tsv \ |
|
--requests 500 \ |
|
--concurrency 20 \ |
|
--user-agent 'load-test/1.0' \ |
|
--cache-buster cb |
|
|
|
bash ${script_name} \ |
|
--base-url https://catalog.data.gov \ |
|
--profile tools/profiles/catalog-realistic.tsv \ |
|
--requests 500 \ |
|
--concurrency 20 \ |
|
--cookie 'auth_tkt=1' |
|
|
|
Profile format: |
|
Tab-separated columns: |
|
weight<TAB>label<TAB>path |
|
|
|
Example: |
|
10 homepage / |
|
20 search-climate /?q=climate |
|
10 api-search /search?q=climate&per_page=20 |
|
|
|
Options: |
|
--base-url URL Base URL, such as https://catalog.data.gov |
|
--profile FILE Weighted endpoint profile file |
|
--requests COUNT Total requests to send. Default: 200 |
|
--concurrency COUNT Parallel requests. Default: 10 |
|
--timeout SECONDS curl timeout per request. Default: 30 |
|
--cookie VALUE Cookie header value to send with every request |
|
--header VALUE Extra header to send with every request. Repeatable. |
|
--user-agent VALUE User-Agent header. Default: datagov-catalog-load-test/1.0 |
|
--cache-buster NAME Append a unique query param with this name to every |
|
request, for example: cb=timestamp-random |
|
--output-dir DIR Directory for the request plan and results. |
|
Default: ./stress-test-results/<timestamp>-profile |
|
--help Show this message. |
|
|
|
Notes: |
|
- This script uses curl and mixes page views, searches, filters, and JSON APIs. |
|
- Review the profile before running against production. |
|
- If you need to bypass cache behavior, pass the same cookie/header values |
|
used by your production setup. |
|
EOF |
|
} |
|
|
|
require_cmd() { |
|
if ! command -v "$1" >/dev/null 2>&1; then |
|
echo "Missing required command: $1" >&2 |
|
exit 1 |
|
fi |
|
} |
|
|
|
base_url="" |
|
profile_file="" |
|
total_requests=200 |
|
concurrency=10 |
|
timeout_seconds=30 |
|
cookie_value="" |
|
user_agent="datagov-catalog-load-test/1.0" |
|
cache_buster_name="" |
|
declare -a extra_headers=() |
|
timestamp="$(date +%Y%m%d-%H%M%S)" |
|
output_dir="./stress-test-results/${timestamp}-profile" |
|
|
|
while [[ $# -gt 0 ]]; do |
|
case "$1" in |
|
--base-url) |
|
base_url="${2:-}" |
|
shift 2 |
|
;; |
|
--profile) |
|
profile_file="${2:-}" |
|
shift 2 |
|
;; |
|
--requests) |
|
total_requests="${2:-}" |
|
shift 2 |
|
;; |
|
--concurrency) |
|
concurrency="${2:-}" |
|
shift 2 |
|
;; |
|
--timeout) |
|
timeout_seconds="${2:-}" |
|
shift 2 |
|
;; |
|
--cookie) |
|
cookie_value="${2:-}" |
|
shift 2 |
|
;; |
|
--header) |
|
extra_headers+=("${2:-}") |
|
shift 2 |
|
;; |
|
--user-agent) |
|
user_agent="${2:-}" |
|
shift 2 |
|
;; |
|
--cache-buster) |
|
cache_buster_name="${2:-}" |
|
shift 2 |
|
;; |
|
--output-dir) |
|
output_dir="${2:-}" |
|
shift 2 |
|
;; |
|
--help|-h) |
|
usage |
|
exit 0 |
|
;; |
|
*) |
|
echo "Unknown argument: $1" >&2 |
|
usage >&2 |
|
exit 1 |
|
;; |
|
esac |
|
done |
|
|
|
if [[ -z "${base_url}" ]] || [[ -z "${profile_file}" ]]; then |
|
echo "--base-url and --profile are required." >&2 |
|
usage >&2 |
|
exit 1 |
|
fi |
|
|
|
if [[ ! -f "${profile_file}" ]]; then |
|
echo "Profile file not found: ${profile_file}" >&2 |
|
exit 1 |
|
fi |
|
|
|
if [[ ! "${total_requests}" =~ ^[0-9]+$ ]] || [[ ! "${concurrency}" =~ ^[0-9]+$ ]] || [[ ! "${timeout_seconds}" =~ ^[0-9]+$ ]]; then |
|
echo "--requests, --concurrency, and --timeout must be integers." >&2 |
|
exit 1 |
|
fi |
|
|
|
require_cmd curl |
|
require_cmd awk |
|
require_cmd sort |
|
require_cmd xargs |
|
require_cmd mkdir |
|
require_cmd date |
|
require_cmd mktemp |
|
require_cmd wc |
|
require_cmd sed |
|
|
|
mkdir -p "${output_dir}" |
|
|
|
plan_file="${output_dir}/request-plan.tsv" |
|
results_file="${output_dir}/results.tsv" |
|
summary_file="${output_dir}/summary.txt" |
|
|
|
declare -a weighted_labels=() |
|
declare -a weighted_paths=() |
|
|
|
while IFS=$'\t' read -r weight label path _; do |
|
if [[ -z "${weight}" ]] || [[ -z "${label}" ]] || [[ -z "${path}" ]]; then |
|
continue |
|
fi |
|
if [[ "${weight}" == \#* ]]; then |
|
continue |
|
fi |
|
if [[ ! "${weight}" =~ ^[0-9]+$ ]]; then |
|
echo "Invalid weight in profile: ${weight}" >&2 |
|
exit 1 |
|
fi |
|
|
|
for ((i = 0; i < weight; i++)); do |
|
weighted_labels+=("${label}") |
|
weighted_paths+=("${path}") |
|
done |
|
done < "${profile_file}" |
|
|
|
if [[ "${#weighted_paths[@]}" -eq 0 ]]; then |
|
echo "Profile did not contain any usable endpoints." >&2 |
|
exit 1 |
|
fi |
|
|
|
: > "${plan_file}" |
|
: > "${results_file}" |
|
|
|
for ((i = 1; i <= total_requests; i++)); do |
|
idx=$((RANDOM % ${#weighted_paths[@]})) |
|
request_path="${weighted_paths[$idx]}" |
|
if [[ -n "${cache_buster_name}" ]]; then |
|
separator="?" |
|
if [[ "${request_path}" == *\?* ]]; then |
|
separator="&" |
|
fi |
|
request_path="${request_path}${separator}${cache_buster_name}=${timestamp}-${i}-${RANDOM}" |
|
fi |
|
printf "%s\t%s%s\n" \ |
|
"${weighted_labels[$idx]}" \ |
|
"${base_url%/}" \ |
|
"${request_path}" >> "${plan_file}" |
|
done |
|
|
|
echo "Base URL: ${base_url}" |
|
echo "Profile: ${profile_file}" |
|
echo "Requests: ${total_requests}" |
|
echo "Concurrency: ${concurrency}" |
|
echo "Timeout: ${timeout_seconds}s" |
|
echo "User-Agent: ${user_agent}" |
|
if [[ -n "${cookie_value}" ]]; then |
|
echo "Cookie: [set]" |
|
fi |
|
if (( ${#extra_headers[@]} > 0 )); then |
|
echo "Extra headers: ${#extra_headers[@]}" |
|
fi |
|
echo "Output directory: ${output_dir}" |
|
echo |
|
|
|
headers_file="${output_dir}/curl-headers.args" |
|
: > "${headers_file}" |
|
printf 'user-agent = "%s"\n' "${user_agent}" >> "${headers_file}" |
|
if [[ -n "${cookie_value}" ]]; then |
|
printf 'header = "Cookie: %s"\n' "${cookie_value}" >> "${headers_file}" |
|
fi |
|
if (( ${#extra_headers[@]} > 0 )); then |
|
for header in "${extra_headers[@]}"; do |
|
printf 'header = "%s"\n' "${header}" >> "${headers_file}" |
|
done |
|
fi |
|
|
|
export RESULTS_FILE="${results_file}" |
|
export TIMEOUT_SECONDS="${timeout_seconds}" |
|
export HEADERS_FILE="${headers_file}" |
|
|
|
run_request() { |
|
local line="$1" |
|
local label="${line%%$'\t'*}" |
|
local url="${line#*$'\t'}" |
|
local curl_output |
|
local curl_exit=0 |
|
|
|
curl_output="$( |
|
curl -sS -o /dev/null \ |
|
--config "${HEADERS_FILE}" \ |
|
--max-time "${TIMEOUT_SECONDS}" \ |
|
-w '%{http_code}\t%{time_total}\t%{size_download}\t'"${label}"$'\t%{url_effective}' \ |
|
"${url}" 2>/dev/null |
|
)" || curl_exit=$? |
|
|
|
if [[ "${curl_exit}" -ne 0 ]]; then |
|
printf "000\t0\t0\t%s\t%s\n" "${label}" "${url}" >> "${RESULTS_FILE}" |
|
return 0 |
|
fi |
|
|
|
printf "%s\n" "${curl_output}" >> "${RESULTS_FILE}" |
|
} |
|
|
|
export -f run_request |
|
|
|
start_epoch="$(date +%s)" |
|
# Preserve each TSV line as a single argument so query strings containing '&' |
|
# are not re-parsed by the shell when requests run concurrently. |
|
awk '{printf "%s%c", $0, 0}' "${plan_file}" \ |
|
| xargs -0 -P "${concurrency}" -I{} bash -c 'run_request "$1"' _ "{}" |
|
end_epoch="$(date +%s)" |
|
|
|
elapsed_seconds=$((end_epoch - start_epoch)) |
|
if [[ "${elapsed_seconds}" -le 0 ]]; then |
|
elapsed_seconds=1 |
|
fi |
|
|
|
total_completed="$(wc -l < "${results_file}" | tr -d '[:space:]')" |
|
success_count="$(awk '$1 ~ /^2[0-9][0-9]$/ {count++} END {print count + 0}' "${results_file}")" |
|
error_count="$(awk '$1 !~ /^2[0-9][0-9]$/ {count++} END {print count + 0}' "${results_file}")" |
|
mean_seconds="$(awk '{sum += $2} END {if (NR > 0) printf "%.3f", sum / NR; else print "0.000"}' "${results_file}")" |
|
req_per_sec="$(awk -v total="${total_completed}" -v elapsed="${elapsed_seconds}" 'BEGIN {printf "%.2f", total / elapsed}')" |
|
|
|
tmp_times="$(mktemp)" |
|
awk '{print $2}' "${results_file}" | sort -n > "${tmp_times}" |
|
percentile_value() { |
|
local percentile="$1" |
|
awk -v p="${percentile}" ' |
|
{ vals[NR] = $1 } |
|
END { |
|
if (NR == 0) { |
|
print "0.000" |
|
exit |
|
} |
|
idx = int((p / 100) * NR) |
|
if (idx < 1) idx = 1 |
|
if (idx > NR) idx = NR |
|
printf "%.3f", vals[idx] |
|
} |
|
' "${tmp_times}" |
|
} |
|
|
|
p50_seconds="$(percentile_value 50)" |
|
p95_seconds="$(percentile_value 95)" |
|
p99_seconds="$(percentile_value 99)" |
|
max_seconds="$(awk 'END {if (NR > 0) printf "%.3f", $1; else print "0.000"}' "${tmp_times}")" |
|
rm -f "${tmp_times}" |
|
|
|
{ |
|
echo "Run configuration" |
|
echo "Base URL: ${base_url}" |
|
echo "Profile: ${profile_file}" |
|
echo "Requests: ${total_requests}" |
|
echo "Concurrency: ${concurrency}" |
|
echo "Timeout (s): ${timeout_seconds}" |
|
echo "User-Agent: ${user_agent}" |
|
if [[ -n "${cookie_value}" ]]; then |
|
echo "Cookie: [set]" |
|
else |
|
echo "Cookie: [not set]" |
|
fi |
|
echo "Cache buster: ${cache_buster_name:-[not set]}" |
|
if (( ${#extra_headers[@]} > 0 )); then |
|
echo "Extra headers:" |
|
for header in "${extra_headers[@]}"; do |
|
echo " - ${header}" |
|
done |
|
else |
|
echo "Extra headers: [none]" |
|
fi |
|
echo "Output directory: ${output_dir}" |
|
echo |
|
echo "Summary" |
|
echo "Completed requests: ${total_completed}" |
|
echo "Elapsed time (s): ${elapsed_seconds}" |
|
echo "Requests per second: ${req_per_sec}" |
|
echo "HTTP 2xx responses: ${success_count}" |
|
echo "Non-2xx or failed requests: ${error_count}" |
|
echo "Mean latency (s): ${mean_seconds}" |
|
echo "P50 latency (s): ${p50_seconds}" |
|
echo "P95 latency (s): ${p95_seconds}" |
|
echo "P99 latency (s): ${p99_seconds}" |
|
echo "Max latency (s): ${max_seconds}" |
|
} | tee "${summary_file}" |
|
|
|
echo |
|
echo "Top labels by volume" |
|
awk -F '\t' '{counts[$4]++} END {for (label in counts) print counts[label] "\t" label}' "${results_file}" \ |
|
| sort -rn \ |
|
| sed -n '1,12p' |
|
|
|
echo |
|
echo "Top URLs with non-2xx responses" |
|
awk -F '\t' '$1 !~ /^2[0-9][0-9]$/ {counts[$5]++} END {for (url in counts) print counts[url] "\t" url}' "${results_file}" \ |
|
| sort -rn \ |
|
| sed -n '1,12p' |
|
|
|
echo |
|
if [[ "${success_count}" -eq 0 ]]; then |
|
echo "No successful requests were recorded. Inspect ${results_file} for 000 responses or misrouting." >&2 |
|
exit 2 |
|
fi |
|
|
|
echo |
|
echo "Results saved to ${output_dir}" |