Copilot commented on code in PR #2245: URL: https://github.com/apache/nifi-minifi-cpp/pull/2245#discussion_r4004837930
########## benchmarks/repository_benchmark/run_benchmark.py: ########## @@ -0,0 +1,421 @@ +#!/usr/bin/env python3 +# Licensed to the Apache Software Foundation (ASF) under one or more +# contributor license agreements. See the NOTICE file distributed with +# this work for additional information regarding copyright ownership. +# The ASF licenses this file to You under the Apache License, Version 2.0 +# (the "License"); you may not use this file except in compliance with +# the License. You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. + +import argparse +import re +import docker +import humanfriendly +import json +import os +import random +import shutil +import string +import tempfile +import threading +import time +import jinja2 +from datetime import datetime, timezone +from enum import Enum + +SCRIPT_DIR = os.path.dirname(os.path.abspath(__file__)) +RESOURCES_DIR = os.path.join(SCRIPT_DIR, "resources") +GET_CONFIG_FILE_TEMPLATE = "get_config.json" +GENERATE_CONFIG_FILE_TEMPLATE = "generate_config.json" +RESULTS_DIR = os.path.join(SCRIPT_DIR, "results") + +MINIFI_HOME = "/opt/minifi/minifi-current" +FLOWFILE_REPO_DIR = f"{MINIFI_HOME}/flowfile_repository" +CONTENT_REPO_DIR = f"{MINIFI_HOME}/content_repository" +INPUT_DIR = "/tmp/input" + +FLOWFILE_REPOSITORY_CLASSES = { + "rocksdb": "FlowFileRepository", + "lmdb": "LmdbFlowFileRepository", + "volatile": "VolatileFlowFileRepository", +} + +CONTENT_REPOSITORY_CLASSES = { + "rocksdb": "DatabaseContentRepository", + "lmdb": "LmdbContentRepository", + "filesystem": "FileSystemRepository", + "volatile": "VolatileContentRepository", +} + +LOG_TIMESTAMP_RE = re.compile(r'\[(\d{4}-\d{2}-\d{2} \d{2}:\d{2}:\d{2}\.\d+)\]') + + +class InputGenerationType(str, Enum): + TIMED_GETFILE = "timed_getfile" + TIMED_GENERATEFLOWFILE = "timed_generateflowfile" + BURST = "burst" + + +def build_properties(flowfile_repository: str, content_repository: str) -> dict[str, str]: + properties = { + "nifi.flow.configuration.file": f"{MINIFI_HOME}/conf/config.yml", + "nifi.extension.path": "../extensions/*", + "nifi.administrative.yield.duration": "1 sec", + "nifi.bored.yield.duration": "100 millis", + "nifi.openssl.fips.support.enable": "false", + "nifi.provenance.repository.class.name": "NoOpRepository", + "nifi.flowfile.repository.directory.default": FLOWFILE_REPO_DIR, + "nifi.database.content.repository.directory.default": CONTENT_REPO_DIR, + "nifi.flowfile.repository.class.name": FLOWFILE_REPOSITORY_CLASSES[flowfile_repository], + "nifi.content.repository.class.name": CONTENT_REPOSITORY_CLASSES[content_repository], + } + return properties + + +def write_properties_file(properties: dict[str, str], path: str) -> None: + with open(path, "w") as properties_file: + for key, value in properties.items(): + properties_file.write(f"{key}={value}\n") + + +def repo_size(container, path: str) -> int: + exit_code, output = container.exec_run(["du", "-sk", path]) + if exit_code != 0: + return 0 + try: + return int(output.decode().split()[0]) * 1024 + except (ValueError, IndexError): + return 0 + + +def read_container_stats(container) -> tuple[int, int, int, int]: + stats = container.stats(stream=False, one_shot=True) + memory_stats = stats.get("memory_stats", {}) + usage = memory_stats.get("usage") + if usage is None: + mem = 0 + else: + inactive_file = memory_stats.get("stats", {}).get("inactive_file", 0) + mem = max(usage - inactive_file, 0) + + cpu_stats = stats.get("cpu_stats", {}) + cpu_total = cpu_stats.get("cpu_usage", {}).get("total_usage", 0) + system_cpu = cpu_stats.get("system_cpu_usage", 0) + num_cpus = cpu_stats.get("online_cpus") or len(cpu_stats.get("cpu_usage", {}).get("percpu_usage") or [1]) + return mem, cpu_total, system_cpu, num_cpus + + +def generate_single_input(input_dir: str, file_size: int, index: int) -> None: + data = os.urandom(file_size) + # Write to a temp name then rename so GetFile never reads a partial file. + tmp_path = os.path.join(input_dir, f".{index}.tmp") + final_path = os.path.join(input_dir, f"input_{index}.bin") + with open(tmp_path, "wb") as input_file: + input_file.write(data) + os.rename(tmp_path, final_path) Review Comment: Generating each input with one `os.urandom(file_size)` allocation makes host memory usage scale with the requested file size. A documented `1G` input therefore allocates roughly 1 GiB before every write and can OOM the benchmark host; write random data in bounded chunks instead. ########## benchmarks/repository_benchmark/generate_report.py: ########## @@ -0,0 +1,345 @@ +#!/usr/bin/env python3 +# Licensed to the Apache Software Foundation (ASF) under one or more +# contributor license agreements. See the NOTICE file distributed with +# this work for additional information regarding copyright ownership. +# The ASF licenses this file to You under the Apache License, Version 2.0 +# (the "License"); you may not use this file except in compliance with +# the License. You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. + +import argparse +import json +import os +import statistics +from collections import Counter + +CHART_JS_CDN = "https://cdn.jsdelivr.net/npm/[email protected]/dist/chart.umd.min.js" + +MIB = 1024 * 1024 + +HTML_TEMPLATE = """<!DOCTYPE html> +<html lang="en"> +<head> +<meta charset="utf-8"> +<title>MiNiFi C++ Repository Benchmark Report</title> +<script src="{chart_js_cdn}"></script> +<style> + body {{ font-family: sans-serif; margin: 2rem; color: #222; }} + h1 {{ font-size: 1.5rem; }} + .chart-container {{ max-width: 900px; margin-bottom: 3rem; }} + table {{ border-collapse: collapse; margin-bottom: 2rem; font-size: 0.9rem; }} + th, td {{ border: 1px solid #ccc; padding: 4px 8px; text-align: left; }} + th {{ background: #f0f0f0; }} + .range {{ color: #888; font-size: 0.85em; }} +</style> +</head> +<body> +<h1>MiNiFi C++ Repository Benchmark Report</h1> +<h2>Summary</h2> +{summary_table} +<h2>Runs</h2> +{config_table} +<div class="chart-container"><canvas id="flowfileChart"></canvas></div> +<div class="chart-container"><canvas id="contentChart"></canvas></div> +<div class="chart-container"><canvas id="memoryChart"></canvas></div> +<div class="chart-container"><canvas id="cpuChart"></canvas></div> +<div class="chart-container"><canvas id="throughputChart"></canvas></div> +<script> +// One entry per repository combo. Samples of the combo's repeated runs are +// aggregated per sample index into a mean line with a min/max band. +const COMBOS = {runs_json}; + +function megabytes(bytes) {{ return bytes / (1024 * 1024); }} + +function comboColor(index, alpha) {{ + const hue = (index * 137.508) % 360; + return `hsla(${{hue}}, 65%, 45%, ${{alpha}})`; +}} + +function makeChart(canvasId, title, seriesKey, transform, yAxisLabel) {{ + const datasets = []; + COMBOS.forEach((combo, i) => {{ + const points = combo.series[seriesKey]; + const line = comboColor(i, 1); + const band = comboColor(i, 0.15); + // Lower bound (min), drawn invisibly; the next dataset fills down to it. + datasets.push({{ + label: combo.label + ' (min)', showInLegend: false, + data: points.map(p => ({{ x: p.x, y: transform(p.min) }})), + showLine: true, pointRadius: 0, borderWidth: 0, fill: false, + }}); + // Upper bound (max); fill to the previous dataset (min) shades the band. + datasets.push({{ + label: combo.label + ' (max)', showInLegend: false, + data: points.map(p => ({{ x: p.x, y: transform(p.max) }})), + showLine: true, pointRadius: 0, borderWidth: 0, backgroundColor: band, fill: '-1', + }}); + // Mean line on top. + datasets.push({{ + label: combo.label, + data: points.map(p => ({{ x: p.x, y: transform(p.mean) }})), + showLine: true, pointRadius: 0, borderColor: line, borderWidth: 2, fill: false, tension: 0.1, + }}); + }}); + new Chart(document.getElementById(canvasId), {{ + type: 'scatter', + data: {{ datasets }}, + options: {{ + plugins: {{ + title: {{ display: true, text: title }}, + legend: {{ labels: {{ filter: (item, data) => data.datasets[item.datasetIndex].showInLegend !== false }} }}, + }}, + scales: {{ + x: {{ title: {{ display: true, text: 'Elapsed time (s)' }} }}, + y: {{ title: {{ display: true, text: yAxisLabel }}, beginAtZero: true }}, + }}, + }}, + }}); +}} + +const identity = v => v; + +// Throughput is a single value per run, shown as the mean across runs per combo. +function makeBarChart(canvasId, title, valueKey, yAxisLabel) {{ + new Chart(document.getElementById(canvasId), {{ + type: 'bar', + data: {{ + labels: COMBOS.map(combo => combo.label), + datasets: [{{ + label: title, + data: COMBOS.map(combo => combo[valueKey]), + backgroundColor: COMBOS.map((combo, i) => comboColor(i, 0.7)), + }}], + }}, + options: {{ + plugins: {{ title: {{ display: true, text: title }}, legend: {{ display: false }} }}, + scales: {{ + y: {{ title: {{ display: true, text: yAxisLabel }}, beginAtZero: true }}, + }}, + }}, + }}); +}} + +makeChart('flowfileChart', 'FlowFile repository size', 'flowfile_repo_bytes', megabytes, 'Megabytes (MiB)'); +makeChart('contentChart', 'Content repository size', 'content_repo_bytes', megabytes, 'Megabytes (MiB)'); +makeChart('memoryChart', 'Process memory usage', 'memory_bytes', megabytes, 'Megabytes (MiB)'); +makeChart('cpuChart', 'Process CPU usage', 'cpu_percent', identity, 'CPU usage (%, 100 = 1 core)'); +makeBarChart('throughputChart', 'Throughput (mean)', 'throughput', 'Flow files / sec'); +</script> +</body> +</html> +""" + + +def build_config_table(runs: list[dict]) -> str: + columns = [ + ("Label", lambda r: r["label"]), + ("FlowFile repo", lambda r: r["config"].get("flowfile_repository", "")), + ("Content repo", lambda r: r["config"].get("content_repository", "")), + ("File size (B)", lambda r: r["config"].get("input_file_size_bytes", "")), + ("Input interval (s)", lambda r: r["config"].get("input_interval_s", "")), + ("Metrics interval (s)", lambda r: r["config"].get("metrics_interval_s", "")), + ("Duration (s)", lambda r: r["config"].get("duration_s", "")), + ("Samples", lambda r: len(r["samples"])), + ("Input generation type", lambda r: r["config"].get("input_file_generation_type", "")), + ("Input file count", lambda r: r["config"].get("input_file_count", "")), + ] + header = "".join(f"<th>{name}</th>" for name, _ in columns) + rows = "" + for run in runs: + cells = "".join(f"<td>{getter(run)}</td>" for _, getter in columns) + rows += f"<tr>{cells}</tr>" + return f"<table><thead><tr>{header}</tr></thead><tbody>{rows}</tbody></table>" + + +def load_run(path: str) -> dict: + with open(path) as result_file: + data = json.load(result_file) + config = data.get("config", {}) + combo = "{}/{}".format( + config.get("flowfile_repository", "?"), + config.get("content_repository", "?"), + ) + return { + "combo": combo, + "path": path, + "config": config, + "samples": data.get("samples", []), + "throughput": data.get("throughput", 0), + "flow_files_processed": data.get("flow_files_processed"), + } + + +def assign_labels(runs: list[dict]) -> None: + # Use the repo combination as the label, only disambiguating with the file + # name when the same combination appears more than once. + combo_counts = Counter(run["combo"] for run in runs) + for run in runs: + if combo_counts[run["combo"]] > 1: + run["label"] = f"{run['combo']} ({os.path.basename(run['path'])})" + else: + run["label"] = run["combo"] + + +def compute_summary(run: dict) -> dict: + samples = run["samples"] + memory = [s.get("memory_bytes", 0) for s in samples] + cpu = [s.get("cpu_percent", 0) for s in samples] + flowfile_sizes = [s.get("flowfile_repo_bytes", 0) for s in samples] + content_sizes = [s.get("content_repo_bytes", 0) for s in samples] + + def peak(values: list) -> float: + return max(values) if values else 0 + + def mean(values: list) -> float: + return statistics.fmean(values) if values else 0 + + def p95(values: list) -> float: + if not values: + return 0 + ordered = sorted(values) + index = min(len(ordered) - 1, int(round(0.95 * (len(ordered) - 1)))) + return ordered[index] + + processed = run.get("flow_files_processed") + final_content = content_sizes[-1] if content_sizes else 0 + + return { + "peak_memory_mib": peak(memory) / MIB, + "mean_memory_mib": mean(memory) / MIB, + "mean_cpu": mean(cpu), + "p95_cpu": p95(cpu), + "final_flowfile_mib": (flowfile_sizes[-1] if flowfile_sizes else 0) / MIB, + "final_content_mib": final_content / MIB, + "peak_content_mib": peak(content_sizes) / MIB, + "throughput": run["throughput"], + "flow_files_processed": processed if processed is not None else "n/a", + } + + +# Scalar summary metrics aggregated across a combo's repeated runs. +# (label, compute_summary key, decimal places) +AGGREGATE_METRICS = [ + ("Throughput (ff/s)", "throughput", 2), + ("Files processed", "flow_files_processed", 0), + ("Peak mem (MiB)", "peak_memory_mib", 1), + ("Mean mem (MiB)", "mean_memory_mib", 1), + ("Mean CPU (%)", "mean_cpu", 1), + ("p95 CPU (%)", "p95_cpu", 1), + ("Final FF repo (MiB)", "final_flowfile_mib", 1), + ("Final content repo (MiB)", "final_content_mib", 1), + ("Peak content repo (MiB)", "peak_content_mib", 1), +] + +# Time-series metrics collapsed into a per-combo mean/min/max band. +SERIES_KEYS = ["flowfile_repo_bytes", "content_repo_bytes", "memory_bytes", "cpu_percent"] + + +def group_runs(runs: list[dict]) -> dict[str, list[dict]]: + groups: dict[str, list[dict]] = {} + for run in runs: + groups.setdefault(run["combo"], []).append(run) + return groups Review Comment: Grouping only by repository pair merges runs with different images and workloads when `generate_report.py` is given arbitrary result files. Their throughput and time series are then averaged as though they were repetitions, producing misleading comparisons; include the benchmark configuration in the group identity or reject mismatched configurations within a combo. ########## benchmarks/repository_benchmark/run_benchmark.py: ########## @@ -0,0 +1,421 @@ +#!/usr/bin/env python3 +# Licensed to the Apache Software Foundation (ASF) under one or more +# contributor license agreements. See the NOTICE file distributed with +# this work for additional information regarding copyright ownership. +# The ASF licenses this file to You under the Apache License, Version 2.0 +# (the "License"); you may not use this file except in compliance with +# the License. You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. + +import argparse +import re +import docker +import humanfriendly +import json +import os +import random +import shutil +import string +import tempfile +import threading +import time +import jinja2 +from datetime import datetime, timezone +from enum import Enum + +SCRIPT_DIR = os.path.dirname(os.path.abspath(__file__)) +RESOURCES_DIR = os.path.join(SCRIPT_DIR, "resources") +GET_CONFIG_FILE_TEMPLATE = "get_config.json" +GENERATE_CONFIG_FILE_TEMPLATE = "generate_config.json" +RESULTS_DIR = os.path.join(SCRIPT_DIR, "results") + +MINIFI_HOME = "/opt/minifi/minifi-current" +FLOWFILE_REPO_DIR = f"{MINIFI_HOME}/flowfile_repository" +CONTENT_REPO_DIR = f"{MINIFI_HOME}/content_repository" +INPUT_DIR = "/tmp/input" + +FLOWFILE_REPOSITORY_CLASSES = { + "rocksdb": "FlowFileRepository", + "lmdb": "LmdbFlowFileRepository", + "volatile": "VolatileFlowFileRepository", +} + +CONTENT_REPOSITORY_CLASSES = { + "rocksdb": "DatabaseContentRepository", + "lmdb": "LmdbContentRepository", + "filesystem": "FileSystemRepository", + "volatile": "VolatileContentRepository", +} + +LOG_TIMESTAMP_RE = re.compile(r'\[(\d{4}-\d{2}-\d{2} \d{2}:\d{2}:\d{2}\.\d+)\]') + + +class InputGenerationType(str, Enum): + TIMED_GETFILE = "timed_getfile" + TIMED_GENERATEFLOWFILE = "timed_generateflowfile" + BURST = "burst" + + +def build_properties(flowfile_repository: str, content_repository: str) -> dict[str, str]: + properties = { + "nifi.flow.configuration.file": f"{MINIFI_HOME}/conf/config.yml", + "nifi.extension.path": "../extensions/*", + "nifi.administrative.yield.duration": "1 sec", + "nifi.bored.yield.duration": "100 millis", + "nifi.openssl.fips.support.enable": "false", + "nifi.provenance.repository.class.name": "NoOpRepository", + "nifi.flowfile.repository.directory.default": FLOWFILE_REPO_DIR, + "nifi.database.content.repository.directory.default": CONTENT_REPO_DIR, + "nifi.flowfile.repository.class.name": FLOWFILE_REPOSITORY_CLASSES[flowfile_repository], + "nifi.content.repository.class.name": CONTENT_REPOSITORY_CLASSES[content_repository], + } + return properties + + +def write_properties_file(properties: dict[str, str], path: str) -> None: + with open(path, "w") as properties_file: + for key, value in properties.items(): + properties_file.write(f"{key}={value}\n") + + +def repo_size(container, path: str) -> int: + exit_code, output = container.exec_run(["du", "-sk", path]) + if exit_code != 0: + return 0 + try: + return int(output.decode().split()[0]) * 1024 + except (ValueError, IndexError): + return 0 + + +def read_container_stats(container) -> tuple[int, int, int, int]: + stats = container.stats(stream=False, one_shot=True) + memory_stats = stats.get("memory_stats", {}) + usage = memory_stats.get("usage") + if usage is None: + mem = 0 + else: + inactive_file = memory_stats.get("stats", {}).get("inactive_file", 0) + mem = max(usage - inactive_file, 0) + + cpu_stats = stats.get("cpu_stats", {}) + cpu_total = cpu_stats.get("cpu_usage", {}).get("total_usage", 0) + system_cpu = cpu_stats.get("system_cpu_usage", 0) + num_cpus = cpu_stats.get("online_cpus") or len(cpu_stats.get("cpu_usage", {}).get("percpu_usage") or [1]) + return mem, cpu_total, system_cpu, num_cpus + + +def generate_single_input(input_dir: str, file_size: int, index: int) -> None: + data = os.urandom(file_size) + # Write to a temp name then rename so GetFile never reads a partial file. + tmp_path = os.path.join(input_dir, f".{index}.tmp") + final_path = os.path.join(input_dir, f"input_{index}.bin") + with open(tmp_path, "wb") as input_file: + input_file.write(data) + os.rename(tmp_path, final_path) + + +def generate_input(input_dir: str, input_count: int, file_size: int) -> None: + for i in range(1, input_count + 1): + generate_single_input(input_dir, file_size, i) + + +def input_generator_loop(stop_event: threading.Event, input_dir: str, interval: float, file_size: int) -> None: + counter = 0 + while not stop_event.is_set(): + counter += 1 + generate_single_input(input_dir, file_size, counter) + stop_event.wait(interval) + + +def metrics_collector_loop(stop_event: threading.Event, container, samples: list, interval: float, start: float) -> None: + next_sample = time.monotonic() + prev_cpu_total = None + prev_system_cpu = None + while not stop_event.is_set(): + memory_bytes, cpu_total, system_cpu, num_cpus = read_container_stats(container) + if prev_cpu_total is not None and system_cpu > prev_system_cpu: + cpu_percent = (cpu_total - prev_cpu_total) / (system_cpu - prev_system_cpu) * num_cpus * 100.0 + else: + cpu_percent = 0.0 + prev_cpu_total, prev_system_cpu = cpu_total, system_cpu + sample = { + "elapsed_s": round(time.monotonic() - start, 3), + "flowfile_repo_bytes": repo_size(container, FLOWFILE_REPO_DIR), + "content_repo_bytes": repo_size(container, CONTENT_REPO_DIR), + "memory_bytes": memory_bytes, + "cpu_percent": round(cpu_percent, 2), + } + samples.append(sample) + next_sample += interval + stop_event.wait(max(0.0, next_sample - time.monotonic())) + + +def wait_for_minifi_to_start(container, timeout: float = 30.0) -> None: + deadline = time.monotonic() + timeout + since = datetime.fromisoformat(container.attrs["Created"]) Review Comment: The `Container` returned by `containers.run(..., detach=True)` is populated from Docker's create response and does not contain inspected fields such as `Created` until it is reloaded, so this access can raise `KeyError` before startup polling begins. Reading all logs from this newly created container also avoids parsing Docker's RFC3339-nanosecond timestamp, which is incompatible with the documented Python 3.10 runtime. This issue also appears on line 222 of the same file. ########## benchmarks/repository_benchmark/run_batch.py: ########## @@ -0,0 +1,119 @@ +#!/usr/bin/env python3 +# Licensed to the Apache Software Foundation (ASF) under one or more +# contributor license agreements. See the NOTICE file distributed with +# this work for additional information regarding copyright ownership. +# The ASF licenses this file to You under the Apache License, Version 2.0 +# (the "License"); you may not use this file except in compliance with +# the License. You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. + +import argparse +import os +import humanfriendly +import run_benchmark +from argparse import Namespace +from datetime import datetime, timezone +from run_benchmark import CONTENT_REPOSITORY_CLASSES, FLOWFILE_REPOSITORY_CLASSES, InputGenerationType + + +def parse_combo(value: str) -> tuple[str, str]: + parts = value.split(":") + if len(parts) != 2: + raise argparse.ArgumentTypeError(f"Combo must be 'flowfile:content', got '{value}'.") + flowfile, content = parts + if flowfile not in FLOWFILE_REPOSITORY_CLASSES: + raise argparse.ArgumentTypeError( + f"Unknown flowfile repository '{flowfile}', choose from {sorted(FLOWFILE_REPOSITORY_CLASSES)}.") + if content not in CONTENT_REPOSITORY_CLASSES: + raise argparse.ArgumentTypeError( + f"Unknown content repository '{content}', choose from {sorted(CONTENT_REPOSITORY_CLASSES)}.") + return flowfile, content + + +def run_combo(args: argparse.Namespace, flowfile: str, content: str, output_dir: str, rep: int) -> str: + timestamp = datetime.now(timezone.utc).strftime("%Y%m%dT%H%M%SZ") + name = f"{timestamp}_{flowfile}_{content}_{args.input_file_generation_type}_run{rep:03d}.json" + output_path = os.path.join(output_dir, name) + combo_args = Namespace( + image=args.image, + flowfile_repository=flowfile, + content_repository=content, + input_file_generation_type=args.input_file_generation_type, + input_file_count=args.input_file_count, + duration=args.duration, + input_interval=args.input_interval, + input_file_size=args.input_file_size, + attribute_count=args.attribute_count, + attribute_size=args.attribute_size, + metrics_interval=args.metrics_interval, + output=output_path, + ) + return run_benchmark.run(combo_args) + + +def main() -> None: + parser = argparse.ArgumentParser( + description="Run the MiNiFi C++ repository benchmark across several repository combinations sequentially and (optionally) generate a single comparison report. " + "Runs are sequential by design so containers do not compete for CPU/IO.") + parser.add_argument("--image", required=True, help="Docker image to use for the benchmark.") + parser.add_argument("--combo", required=True, action="append", type=parse_combo, dest="combos", + metavar="FLOWFILE:CONTENT", + help="Repository combination to benchmark, e.g. --combo lmdb:lmdb. Repeatable.") + parser.add_argument("--input-file-generation-type", default=InputGenerationType.TIMED_GETFILE.value, + choices=sorted([e.value for e in InputGenerationType]), + help="Input file generation type shared by all combos (see run_benchmark.py).") + parser.add_argument("--input-file-count", type=int, default=100, + help="Number of input files for burst input generation type (default: 100).") + parser.add_argument("--duration", type=int, default=120, + help="Total benchmark session length in seconds (default: 120).") + parser.add_argument("--input-interval", type=float, default=1.0, + help="Seconds between input file generation cycles (default: 1).") + parser.add_argument("--input-file-size", type=humanfriendly.parse_size, default=humanfriendly.parse_size("1M"), + help="Size of each generated input file, e.g. 512K, 1M, 1G (default: 1M).") + parser.add_argument("--attribute-count", type=int, default=0, + help="Number of extra attributes to set on each flow file via UpdateAttribute (default: 0).") + parser.add_argument("--attribute-size", type=humanfriendly.parse_size, default=0, + help="Size of each extra attribute value, e.g. 64, 1K (default: 0).") + parser.add_argument("--metrics-interval", type=float, default=5.0, + help="Seconds between metric samples (default: 5).") + parser.add_argument("--repeat", type=int, default=1, + help="Number of times to run each combo; results are aggregated per combo in the report (default: 1).") + parser.add_argument("--output-dir", default=run_benchmark.RESULTS_DIR, + help="Directory for result JSON files (default: results/).") + parser.add_argument("--report", default=None, + help="If set, generate an HTML report at this path from all successful runs.") + args = parser.parse_args() + + os.makedirs(args.output_dir, exist_ok=True) + + result_paths: list[str] = [] + failures: list[tuple[str, str, str]] = [] + total_runs = len(args.combos) * args.repeat + for index, (flowfile, content) in enumerate(args.combos, start=1): + for rep in range(1, args.repeat + 1): + print(f"\n=== [combo {index}/{len(args.combos)} rep {rep}/{args.repeat}] Benchmarking {flowfile}/{content} ===") + try: + result_paths.append(run_combo(args, flowfile, content, args.output_dir, rep)) + except Exception as error: + print(f"Error: combo {flowfile}/{content} (rep {rep}) failed with: {error}") + failures.append((flowfile, content, str(error))) + + print("\n=== Batch summary ===") + print(f"Succeeded: {len(result_paths)}/{total_runs}") + for flowfile, content, error in failures: + print(f" FAILED {flowfile}/{content}: {error}") + + if args.report and result_paths: + import generate_report + generate_report.write_report(result_paths, args.report) Review Comment: Caught run failures are only printed, so the batch process exits with status 0 even when one or every benchmark failed. This makes automation treat an incomplete batch as successful; preserve the continue-on-error behavior but return a nonzero status after generating any report. ########## benchmarks/repository_benchmark/run_benchmark.py: ########## @@ -0,0 +1,421 @@ +#!/usr/bin/env python3 +# Licensed to the Apache Software Foundation (ASF) under one or more +# contributor license agreements. See the NOTICE file distributed with +# this work for additional information regarding copyright ownership. +# The ASF licenses this file to You under the Apache License, Version 2.0 +# (the "License"); you may not use this file except in compliance with +# the License. You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. + +import argparse +import re +import docker +import humanfriendly +import json +import os +import random +import shutil +import string +import tempfile +import threading +import time +import jinja2 +from datetime import datetime, timezone +from enum import Enum + +SCRIPT_DIR = os.path.dirname(os.path.abspath(__file__)) +RESOURCES_DIR = os.path.join(SCRIPT_DIR, "resources") +GET_CONFIG_FILE_TEMPLATE = "get_config.json" +GENERATE_CONFIG_FILE_TEMPLATE = "generate_config.json" +RESULTS_DIR = os.path.join(SCRIPT_DIR, "results") + +MINIFI_HOME = "/opt/minifi/minifi-current" +FLOWFILE_REPO_DIR = f"{MINIFI_HOME}/flowfile_repository" +CONTENT_REPO_DIR = f"{MINIFI_HOME}/content_repository" +INPUT_DIR = "/tmp/input" + +FLOWFILE_REPOSITORY_CLASSES = { + "rocksdb": "FlowFileRepository", + "lmdb": "LmdbFlowFileRepository", + "volatile": "VolatileFlowFileRepository", +} + +CONTENT_REPOSITORY_CLASSES = { + "rocksdb": "DatabaseContentRepository", + "lmdb": "LmdbContentRepository", + "filesystem": "FileSystemRepository", + "volatile": "VolatileContentRepository", +} + +LOG_TIMESTAMP_RE = re.compile(r'\[(\d{4}-\d{2}-\d{2} \d{2}:\d{2}:\d{2}\.\d+)\]') + + +class InputGenerationType(str, Enum): + TIMED_GETFILE = "timed_getfile" + TIMED_GENERATEFLOWFILE = "timed_generateflowfile" + BURST = "burst" + + +def build_properties(flowfile_repository: str, content_repository: str) -> dict[str, str]: + properties = { + "nifi.flow.configuration.file": f"{MINIFI_HOME}/conf/config.yml", + "nifi.extension.path": "../extensions/*", + "nifi.administrative.yield.duration": "1 sec", + "nifi.bored.yield.duration": "100 millis", + "nifi.openssl.fips.support.enable": "false", + "nifi.provenance.repository.class.name": "NoOpRepository", + "nifi.flowfile.repository.directory.default": FLOWFILE_REPO_DIR, + "nifi.database.content.repository.directory.default": CONTENT_REPO_DIR, + "nifi.flowfile.repository.class.name": FLOWFILE_REPOSITORY_CLASSES[flowfile_repository], + "nifi.content.repository.class.name": CONTENT_REPOSITORY_CLASSES[content_repository], + } + return properties + + +def write_properties_file(properties: dict[str, str], path: str) -> None: + with open(path, "w") as properties_file: + for key, value in properties.items(): + properties_file.write(f"{key}={value}\n") + + +def repo_size(container, path: str) -> int: + exit_code, output = container.exec_run(["du", "-sk", path]) + if exit_code != 0: + return 0 + try: + return int(output.decode().split()[0]) * 1024 + except (ValueError, IndexError): + return 0 + + +def read_container_stats(container) -> tuple[int, int, int, int]: + stats = container.stats(stream=False, one_shot=True) + memory_stats = stats.get("memory_stats", {}) + usage = memory_stats.get("usage") + if usage is None: + mem = 0 + else: + inactive_file = memory_stats.get("stats", {}).get("inactive_file", 0) + mem = max(usage - inactive_file, 0) + + cpu_stats = stats.get("cpu_stats", {}) + cpu_total = cpu_stats.get("cpu_usage", {}).get("total_usage", 0) + system_cpu = cpu_stats.get("system_cpu_usage", 0) + num_cpus = cpu_stats.get("online_cpus") or len(cpu_stats.get("cpu_usage", {}).get("percpu_usage") or [1]) + return mem, cpu_total, system_cpu, num_cpus + + +def generate_single_input(input_dir: str, file_size: int, index: int) -> None: + data = os.urandom(file_size) + # Write to a temp name then rename so GetFile never reads a partial file. + tmp_path = os.path.join(input_dir, f".{index}.tmp") + final_path = os.path.join(input_dir, f"input_{index}.bin") + with open(tmp_path, "wb") as input_file: + input_file.write(data) + os.rename(tmp_path, final_path) + + +def generate_input(input_dir: str, input_count: int, file_size: int) -> None: + for i in range(1, input_count + 1): + generate_single_input(input_dir, file_size, i) + + +def input_generator_loop(stop_event: threading.Event, input_dir: str, interval: float, file_size: int) -> None: + counter = 0 + while not stop_event.is_set(): + counter += 1 + generate_single_input(input_dir, file_size, counter) + stop_event.wait(interval) + + +def metrics_collector_loop(stop_event: threading.Event, container, samples: list, interval: float, start: float) -> None: + next_sample = time.monotonic() + prev_cpu_total = None + prev_system_cpu = None + while not stop_event.is_set(): + memory_bytes, cpu_total, system_cpu, num_cpus = read_container_stats(container) + if prev_cpu_total is not None and system_cpu > prev_system_cpu: + cpu_percent = (cpu_total - prev_cpu_total) / (system_cpu - prev_system_cpu) * num_cpus * 100.0 + else: + cpu_percent = 0.0 + prev_cpu_total, prev_system_cpu = cpu_total, system_cpu + sample = { + "elapsed_s": round(time.monotonic() - start, 3), + "flowfile_repo_bytes": repo_size(container, FLOWFILE_REPO_DIR), + "content_repo_bytes": repo_size(container, CONTENT_REPO_DIR), + "memory_bytes": memory_bytes, + "cpu_percent": round(cpu_percent, 2), + } + samples.append(sample) + next_sample += interval + stop_event.wait(max(0.0, next_sample - time.monotonic())) + + +def wait_for_minifi_to_start(container, timeout: float = 30.0) -> None: + deadline = time.monotonic() + timeout + since = datetime.fromisoformat(container.attrs["Created"]) + while time.monotonic() < deadline: + container.reload() + if container.status != "running": + time.sleep(0.5) + continue + now = datetime.now(timezone.utc) + logs = container.logs(since=since).decode(errors="replace") + since = now + if "MiNiFi started" in logs: + return + time.sleep(0.5) + raise RuntimeError(f"Container did not reach running state (status: {container.status})") + + +def build_extra_attributes(count: int, size: int) -> dict[str, str]: + alphabet = string.ascii_letters + string.digits + return {f"bench_attr_{index}": "".join(random.choices(alphabet, k=size)) for index in range(count)} + + +def write_config_yml(args: argparse.Namespace, work_dir: str) -> None: + jinja_env = jinja2.Environment(loader=jinja2.FileSystemLoader(RESOURCES_DIR)) + extra_attributes = build_extra_attributes(args.attribute_count, args.attribute_size) + if args.input_file_generation_type != InputGenerationType.TIMED_GENERATEFLOWFILE: + flow_config_template = jinja_env.get_template(GET_CONFIG_FILE_TEMPLATE) + flow_config = flow_config_template.render(get_file_interval=args.input_interval, + extra_attributes=extra_attributes) + with open(os.path.join(work_dir, "config.yml"), "w") as config_file: + config_file.write(flow_config) + else: + flow_config_template = jinja_env.get_template(GENERATE_CONFIG_FILE_TEMPLATE) + flow_config = flow_config_template.render(generate_file_interval=args.input_interval, + generate_file_size=args.input_file_size, + extra_attributes=extra_attributes) + with open(os.path.join(work_dir, "config.yml"), "w") as config_file: + config_file.write(flow_config) + + +def create_minifi_container(args: argparse.Namespace, work_dir: str, input_dir: str) -> docker.models.containers.Container: + properties_path = os.path.join(work_dir, "minifi.properties") + properties = build_properties(args.flowfile_repository, args.content_repository) + write_properties_file(properties, properties_path) + + client = docker.from_env() + + container = client.containers.run( + args.image, + detach=True, + volumes={ + properties_path: {"bind": f"{MINIFI_HOME}/conf/minifi.properties", "mode": "ro"}, + os.path.join(work_dir, "config.yml"): {"bind": f"{MINIFI_HOME}/conf/config.yml", "mode": "ro"}, + input_dir: {"bind": INPUT_DIR, "mode": "rw"}, + }, + ) + return container + + +def wait_for_flow_files_to_be_processed(container, expected_count: int, timeout: float = 300.0) -> None: + deadline = time.monotonic() + timeout + since = datetime.fromisoformat(container.attrs["Created"]) + while True: + now = datetime.now(timezone.utc) + logs = container.logs(since=since).decode(errors="replace") + since = now + if f"key:flow_file_count value:{expected_count - 1}" in logs: + break + container.reload() + if container.status != "running": + raise RuntimeError(f"Container exited before processing {expected_count} flow files (status: {container.status})") + if time.monotonic() > deadline: + raise RuntimeError(f"Timed out after {timeout}s waiting for {expected_count} flow files to be processed") + time.sleep(0.1) + + # Wait a bit more to see how the repositories behave after all flow files have been processed. + time.sleep(2) Review Comment: This fixed two-second delay is shorter than `run_batch.py`'s default five-second metrics interval. A fast burst can therefore finish before the collector's second sample, leaving the reported final and peak repository sizes equal to the startup sample; explicitly collect a final sample after processing or synchronize this delay with the metrics interval. ########## benchmarks/repository_benchmark/run_benchmark.py: ########## @@ -0,0 +1,421 @@ +#!/usr/bin/env python3 +# Licensed to the Apache Software Foundation (ASF) under one or more +# contributor license agreements. See the NOTICE file distributed with +# this work for additional information regarding copyright ownership. +# The ASF licenses this file to You under the Apache License, Version 2.0 +# (the "License"); you may not use this file except in compliance with +# the License. You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. + +import argparse +import re +import docker +import humanfriendly +import json +import os +import random +import shutil +import string +import tempfile +import threading +import time +import jinja2 +from datetime import datetime, timezone +from enum import Enum + +SCRIPT_DIR = os.path.dirname(os.path.abspath(__file__)) +RESOURCES_DIR = os.path.join(SCRIPT_DIR, "resources") +GET_CONFIG_FILE_TEMPLATE = "get_config.json" +GENERATE_CONFIG_FILE_TEMPLATE = "generate_config.json" +RESULTS_DIR = os.path.join(SCRIPT_DIR, "results") + +MINIFI_HOME = "/opt/minifi/minifi-current" +FLOWFILE_REPO_DIR = f"{MINIFI_HOME}/flowfile_repository" +CONTENT_REPO_DIR = f"{MINIFI_HOME}/content_repository" +INPUT_DIR = "/tmp/input" + +FLOWFILE_REPOSITORY_CLASSES = { + "rocksdb": "FlowFileRepository", + "lmdb": "LmdbFlowFileRepository", + "volatile": "VolatileFlowFileRepository", +} + +CONTENT_REPOSITORY_CLASSES = { + "rocksdb": "DatabaseContentRepository", + "lmdb": "LmdbContentRepository", + "filesystem": "FileSystemRepository", + "volatile": "VolatileContentRepository", +} + +LOG_TIMESTAMP_RE = re.compile(r'\[(\d{4}-\d{2}-\d{2} \d{2}:\d{2}:\d{2}\.\d+)\]') + + +class InputGenerationType(str, Enum): + TIMED_GETFILE = "timed_getfile" + TIMED_GENERATEFLOWFILE = "timed_generateflowfile" + BURST = "burst" + + +def build_properties(flowfile_repository: str, content_repository: str) -> dict[str, str]: + properties = { + "nifi.flow.configuration.file": f"{MINIFI_HOME}/conf/config.yml", + "nifi.extension.path": "../extensions/*", + "nifi.administrative.yield.duration": "1 sec", + "nifi.bored.yield.duration": "100 millis", + "nifi.openssl.fips.support.enable": "false", + "nifi.provenance.repository.class.name": "NoOpRepository", + "nifi.flowfile.repository.directory.default": FLOWFILE_REPO_DIR, + "nifi.database.content.repository.directory.default": CONTENT_REPO_DIR, + "nifi.flowfile.repository.class.name": FLOWFILE_REPOSITORY_CLASSES[flowfile_repository], + "nifi.content.repository.class.name": CONTENT_REPOSITORY_CLASSES[content_repository], + } + return properties + + +def write_properties_file(properties: dict[str, str], path: str) -> None: + with open(path, "w") as properties_file: + for key, value in properties.items(): + properties_file.write(f"{key}={value}\n") + + +def repo_size(container, path: str) -> int: + exit_code, output = container.exec_run(["du", "-sk", path]) + if exit_code != 0: + return 0 + try: + return int(output.decode().split()[0]) * 1024 + except (ValueError, IndexError): + return 0 + + +def read_container_stats(container) -> tuple[int, int, int, int]: + stats = container.stats(stream=False, one_shot=True) + memory_stats = stats.get("memory_stats", {}) + usage = memory_stats.get("usage") + if usage is None: + mem = 0 + else: + inactive_file = memory_stats.get("stats", {}).get("inactive_file", 0) + mem = max(usage - inactive_file, 0) + + cpu_stats = stats.get("cpu_stats", {}) + cpu_total = cpu_stats.get("cpu_usage", {}).get("total_usage", 0) + system_cpu = cpu_stats.get("system_cpu_usage", 0) + num_cpus = cpu_stats.get("online_cpus") or len(cpu_stats.get("cpu_usage", {}).get("percpu_usage") or [1]) + return mem, cpu_total, system_cpu, num_cpus + + +def generate_single_input(input_dir: str, file_size: int, index: int) -> None: + data = os.urandom(file_size) + # Write to a temp name then rename so GetFile never reads a partial file. + tmp_path = os.path.join(input_dir, f".{index}.tmp") + final_path = os.path.join(input_dir, f"input_{index}.bin") + with open(tmp_path, "wb") as input_file: + input_file.write(data) + os.rename(tmp_path, final_path) + + +def generate_input(input_dir: str, input_count: int, file_size: int) -> None: + for i in range(1, input_count + 1): + generate_single_input(input_dir, file_size, i) + + +def input_generator_loop(stop_event: threading.Event, input_dir: str, interval: float, file_size: int) -> None: + counter = 0 + while not stop_event.is_set(): + counter += 1 + generate_single_input(input_dir, file_size, counter) + stop_event.wait(interval) + + +def metrics_collector_loop(stop_event: threading.Event, container, samples: list, interval: float, start: float) -> None: + next_sample = time.monotonic() + prev_cpu_total = None + prev_system_cpu = None + while not stop_event.is_set(): + memory_bytes, cpu_total, system_cpu, num_cpus = read_container_stats(container) + if prev_cpu_total is not None and system_cpu > prev_system_cpu: + cpu_percent = (cpu_total - prev_cpu_total) / (system_cpu - prev_system_cpu) * num_cpus * 100.0 + else: + cpu_percent = 0.0 + prev_cpu_total, prev_system_cpu = cpu_total, system_cpu + sample = { + "elapsed_s": round(time.monotonic() - start, 3), + "flowfile_repo_bytes": repo_size(container, FLOWFILE_REPO_DIR), + "content_repo_bytes": repo_size(container, CONTENT_REPO_DIR), + "memory_bytes": memory_bytes, + "cpu_percent": round(cpu_percent, 2), + } + samples.append(sample) + next_sample += interval + stop_event.wait(max(0.0, next_sample - time.monotonic())) + + +def wait_for_minifi_to_start(container, timeout: float = 30.0) -> None: + deadline = time.monotonic() + timeout + since = datetime.fromisoformat(container.attrs["Created"]) + while time.monotonic() < deadline: + container.reload() + if container.status != "running": + time.sleep(0.5) + continue + now = datetime.now(timezone.utc) + logs = container.logs(since=since).decode(errors="replace") + since = now + if "MiNiFi started" in logs: + return + time.sleep(0.5) + raise RuntimeError(f"Container did not reach running state (status: {container.status})") + + +def build_extra_attributes(count: int, size: int) -> dict[str, str]: + alphabet = string.ascii_letters + string.digits + return {f"bench_attr_{index}": "".join(random.choices(alphabet, k=size)) for index in range(count)} + + +def write_config_yml(args: argparse.Namespace, work_dir: str) -> None: + jinja_env = jinja2.Environment(loader=jinja2.FileSystemLoader(RESOURCES_DIR)) + extra_attributes = build_extra_attributes(args.attribute_count, args.attribute_size) + if args.input_file_generation_type != InputGenerationType.TIMED_GENERATEFLOWFILE: + flow_config_template = jinja_env.get_template(GET_CONFIG_FILE_TEMPLATE) + flow_config = flow_config_template.render(get_file_interval=args.input_interval, + extra_attributes=extra_attributes) + with open(os.path.join(work_dir, "config.yml"), "w") as config_file: + config_file.write(flow_config) + else: + flow_config_template = jinja_env.get_template(GENERATE_CONFIG_FILE_TEMPLATE) + flow_config = flow_config_template.render(generate_file_interval=args.input_interval, + generate_file_size=args.input_file_size, + extra_attributes=extra_attributes) + with open(os.path.join(work_dir, "config.yml"), "w") as config_file: + config_file.write(flow_config) + + +def create_minifi_container(args: argparse.Namespace, work_dir: str, input_dir: str) -> docker.models.containers.Container: + properties_path = os.path.join(work_dir, "minifi.properties") + properties = build_properties(args.flowfile_repository, args.content_repository) + write_properties_file(properties, properties_path) + + client = docker.from_env() + + container = client.containers.run( + args.image, + detach=True, + volumes={ + properties_path: {"bind": f"{MINIFI_HOME}/conf/minifi.properties", "mode": "ro"}, + os.path.join(work_dir, "config.yml"): {"bind": f"{MINIFI_HOME}/conf/config.yml", "mode": "ro"}, + input_dir: {"bind": INPUT_DIR, "mode": "rw"}, + }, + ) + return container + + +def wait_for_flow_files_to_be_processed(container, expected_count: int, timeout: float = 300.0) -> None: + deadline = time.monotonic() + timeout + since = datetime.fromisoformat(container.attrs["Created"]) + while True: + now = datetime.now(timezone.utc) + logs = container.logs(since=since).decode(errors="replace") + since = now + if f"key:flow_file_count value:{expected_count - 1}" in logs: + break + container.reload() + if container.status != "running": + raise RuntimeError(f"Container exited before processing {expected_count} flow files (status: {container.status})") + if time.monotonic() > deadline: + raise RuntimeError(f"Timed out after {timeout}s waiting for {expected_count} flow files to be processed") + time.sleep(0.1) + + # Wait a bit more to see how the repositories behave after all flow files have been processed. + time.sleep(2) + + +def run_threads(samples: list[dict], args: argparse.Namespace) -> tuple[float, int]: + stop_event = threading.Event() + container = None + work_dir = tempfile.mkdtemp(prefix="repo_benchmark_") + input_dir = os.path.join(work_dir, "input") + os.makedirs(input_dir) + write_config_yml(args, work_dir) + try: + if args.input_file_generation_type == InputGenerationType.TIMED_GETFILE: + start = time.monotonic() + container = create_minifi_container(args, work_dir, input_dir) + wait_for_minifi_to_start(container) + threads = [ + threading.Thread( + target=input_generator_loop, + args=(stop_event, input_dir, args.input_interval, args.input_file_size), + daemon=True, + ), + threading.Thread( + target=metrics_collector_loop, + args=(stop_event, container, samples, args.metrics_interval, start), + daemon=True, + ), + ] + elif args.input_file_generation_type == InputGenerationType.BURST: + generate_input(input_dir, args.input_file_count, args.input_file_size) + start = time.monotonic() + container = create_minifi_container(args, work_dir, input_dir) + threads = [ + threading.Thread( + target=metrics_collector_loop, + args=(stop_event, container, samples, args.metrics_interval, start), + daemon=True, + ), + ] + else: + start = time.monotonic() + container = create_minifi_container(args, work_dir, input_dir) + wait_for_minifi_to_start(container) + threads = [ + threading.Thread( + target=metrics_collector_loop, + args=(stop_event, container, samples, args.metrics_interval, start), + daemon=True, + ), + ] + + for thread in threads: + thread.start() + + if args.input_file_generation_type != InputGenerationType.BURST: + time.sleep(args.duration) + stop_event.set() + else: + print(f"Waiting for {args.input_file_count} flow files to be processed...") + wait_for_flow_files_to_be_processed(container, args.input_file_count) + stop_event.set() + + for thread in threads: + thread.join(timeout=10) Review Comment: `Thread.join()` does not propagate worker exceptions, and the timeout result is not checked. If input generation or Docker metric collection fails (or remains blocked), the main thread still calculates throughput and writes an apparently successful but incomplete benchmark result; transport worker exceptions back to the caller and fail when a thread is still alive after the timeout. -- This is an automated message from the Apache Git Service. To respond to the message, please log on to GitHub and use the URL above to go to the specific comment. To unsubscribe, e-mail: [email protected] For queries about this service, please contact Infrastructure at: [email protected]
