lordgamez commented on code in PR #2245:
URL: https://github.com/apache/nifi-minifi-cpp/pull/2245#discussion_r4006683919


##########
benchmarks/repository_benchmark/run_benchmark.py:
##########
@@ -0,0 +1,421 @@
+#!/usr/bin/env python3
+# Licensed to the Apache Software Foundation (ASF) under one or more
+# contributor license agreements.  See the NOTICE file distributed with
+# this work for additional information regarding copyright ownership.
+# The ASF licenses this file to You under the Apache License, Version 2.0
+# (the "License"); you may not use this file except in compliance with
+# the License.  You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+import argparse
+import re
+import docker
+import humanfriendly
+import json
+import os
+import random
+import shutil
+import string
+import tempfile
+import threading
+import time
+import jinja2
+from datetime import datetime, timezone
+from enum import Enum
+
+SCRIPT_DIR = os.path.dirname(os.path.abspath(__file__))
+RESOURCES_DIR = os.path.join(SCRIPT_DIR, "resources")
+GET_CONFIG_FILE_TEMPLATE = "get_config.json"
+GENERATE_CONFIG_FILE_TEMPLATE = "generate_config.json"
+RESULTS_DIR = os.path.join(SCRIPT_DIR, "results")
+
+MINIFI_HOME = "/opt/minifi/minifi-current"
+FLOWFILE_REPO_DIR = f"{MINIFI_HOME}/flowfile_repository"
+CONTENT_REPO_DIR = f"{MINIFI_HOME}/content_repository"
+INPUT_DIR = "/tmp/input"
+
+FLOWFILE_REPOSITORY_CLASSES = {
+    "rocksdb": "FlowFileRepository",
+    "lmdb": "LmdbFlowFileRepository",
+    "volatile": "VolatileFlowFileRepository",
+}
+
+CONTENT_REPOSITORY_CLASSES = {
+    "rocksdb": "DatabaseContentRepository",
+    "lmdb": "LmdbContentRepository",
+    "filesystem": "FileSystemRepository",
+    "volatile": "VolatileContentRepository",
+}
+
+LOG_TIMESTAMP_RE = re.compile(r'\[(\d{4}-\d{2}-\d{2} 
\d{2}:\d{2}:\d{2}\.\d+)\]')
+
+
+class InputGenerationType(str, Enum):
+    TIMED_GETFILE = "timed_getfile"
+    TIMED_GENERATEFLOWFILE = "timed_generateflowfile"
+    BURST = "burst"
+
+
+def build_properties(flowfile_repository: str, content_repository: str) -> 
dict[str, str]:
+    properties = {
+        "nifi.flow.configuration.file": f"{MINIFI_HOME}/conf/config.yml",
+        "nifi.extension.path": "../extensions/*",
+        "nifi.administrative.yield.duration": "1 sec",
+        "nifi.bored.yield.duration": "100 millis",
+        "nifi.openssl.fips.support.enable": "false",
+        "nifi.provenance.repository.class.name": "NoOpRepository",
+        "nifi.flowfile.repository.directory.default": FLOWFILE_REPO_DIR,
+        "nifi.database.content.repository.directory.default": CONTENT_REPO_DIR,
+        "nifi.flowfile.repository.class.name": 
FLOWFILE_REPOSITORY_CLASSES[flowfile_repository],
+        "nifi.content.repository.class.name": 
CONTENT_REPOSITORY_CLASSES[content_repository],
+    }
+    return properties
+
+
+def write_properties_file(properties: dict[str, str], path: str) -> None:
+    with open(path, "w") as properties_file:
+        for key, value in properties.items():
+            properties_file.write(f"{key}={value}\n")
+
+
+def repo_size(container, path: str) -> int:
+    exit_code, output = container.exec_run(["du", "-sk", path])
+    if exit_code != 0:
+        return 0
+    try:
+        return int(output.decode().split()[0]) * 1024
+    except (ValueError, IndexError):
+        return 0
+
+
+def read_container_stats(container) -> tuple[int, int, int, int]:
+    stats = container.stats(stream=False, one_shot=True)
+    memory_stats = stats.get("memory_stats", {})
+    usage = memory_stats.get("usage")
+    if usage is None:
+        mem = 0
+    else:
+        inactive_file = memory_stats.get("stats", {}).get("inactive_file", 0)
+        mem = max(usage - inactive_file, 0)
+
+    cpu_stats = stats.get("cpu_stats", {})
+    cpu_total = cpu_stats.get("cpu_usage", {}).get("total_usage", 0)
+    system_cpu = cpu_stats.get("system_cpu_usage", 0)
+    num_cpus = cpu_stats.get("online_cpus") or len(cpu_stats.get("cpu_usage", 
{}).get("percpu_usage") or [1])
+    return mem, cpu_total, system_cpu, num_cpus
+
+
+def generate_single_input(input_dir: str, file_size: int, index: int) -> None:
+    data = os.urandom(file_size)
+    # Write to a temp name then rename so GetFile never reads a partial file.
+    tmp_path = os.path.join(input_dir, f".{index}.tmp")
+    final_path = os.path.join(input_dir, f"input_{index}.bin")
+    with open(tmp_path, "wb") as input_file:
+        input_file.write(data)
+    os.rename(tmp_path, final_path)
+
+
+def generate_input(input_dir: str, input_count: int, file_size: int) -> None:
+    for i in range(1, input_count + 1):
+        generate_single_input(input_dir, file_size, i)
+
+
+def input_generator_loop(stop_event: threading.Event, input_dir: str, 
interval: float, file_size: int) -> None:
+    counter = 0
+    while not stop_event.is_set():
+        counter += 1
+        generate_single_input(input_dir, file_size, counter)
+        stop_event.wait(interval)
+
+
+def metrics_collector_loop(stop_event: threading.Event, container, samples: 
list, interval: float, start: float) -> None:
+    next_sample = time.monotonic()
+    prev_cpu_total = None
+    prev_system_cpu = None
+    while not stop_event.is_set():
+        memory_bytes, cpu_total, system_cpu, num_cpus = 
read_container_stats(container)
+        if prev_cpu_total is not None and system_cpu > prev_system_cpu:
+            cpu_percent = (cpu_total - prev_cpu_total) / (system_cpu - 
prev_system_cpu) * num_cpus * 100.0
+        else:
+            cpu_percent = 0.0
+        prev_cpu_total, prev_system_cpu = cpu_total, system_cpu
+        sample = {
+            "elapsed_s": round(time.monotonic() - start, 3),
+            "flowfile_repo_bytes": repo_size(container, FLOWFILE_REPO_DIR),
+            "content_repo_bytes": repo_size(container, CONTENT_REPO_DIR),
+            "memory_bytes": memory_bytes,
+            "cpu_percent": round(cpu_percent, 2),
+        }
+        samples.append(sample)
+        next_sample += interval
+        stop_event.wait(max(0.0, next_sample - time.monotonic()))
+
+
+def wait_for_minifi_to_start(container, timeout: float = 30.0) -> None:
+    deadline = time.monotonic() + timeout
+    since = datetime.fromisoformat(container.attrs["Created"])

Review Comment:
   Fixed timestamp parsing in 
https://github.com/apache/nifi-minifi-cpp/pull/2245/commits/a6d1495efb23ba3081ddf4beb6970d3e30fe9c83.



##########
benchmarks/repository_benchmark/run_batch.py:
##########
@@ -0,0 +1,119 @@
+#!/usr/bin/env python3
+# Licensed to the Apache Software Foundation (ASF) under one or more
+# contributor license agreements.  See the NOTICE file distributed with
+# this work for additional information regarding copyright ownership.
+# The ASF licenses this file to You under the Apache License, Version 2.0
+# (the "License"); you may not use this file except in compliance with
+# the License.  You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+import argparse
+import os
+import humanfriendly
+import run_benchmark
+from argparse import Namespace
+from datetime import datetime, timezone
+from run_benchmark import CONTENT_REPOSITORY_CLASSES, 
FLOWFILE_REPOSITORY_CLASSES, InputGenerationType
+
+
+def parse_combo(value: str) -> tuple[str, str]:
+    parts = value.split(":")
+    if len(parts) != 2:
+        raise argparse.ArgumentTypeError(f"Combo must be 'flowfile:content', 
got '{value}'.")
+    flowfile, content = parts
+    if flowfile not in FLOWFILE_REPOSITORY_CLASSES:
+        raise argparse.ArgumentTypeError(
+            f"Unknown flowfile repository '{flowfile}', choose from 
{sorted(FLOWFILE_REPOSITORY_CLASSES)}.")
+    if content not in CONTENT_REPOSITORY_CLASSES:
+        raise argparse.ArgumentTypeError(
+            f"Unknown content repository '{content}', choose from 
{sorted(CONTENT_REPOSITORY_CLASSES)}.")
+    return flowfile, content
+
+
+def run_combo(args: argparse.Namespace, flowfile: str, content: str, 
output_dir: str, rep: int) -> str:
+    timestamp = datetime.now(timezone.utc).strftime("%Y%m%dT%H%M%SZ")
+    name = 
f"{timestamp}_{flowfile}_{content}_{args.input_file_generation_type}_run{rep:03d}.json"
+    output_path = os.path.join(output_dir, name)
+    combo_args = Namespace(
+        image=args.image,
+        flowfile_repository=flowfile,
+        content_repository=content,
+        input_file_generation_type=args.input_file_generation_type,
+        input_file_count=args.input_file_count,
+        duration=args.duration,
+        input_interval=args.input_interval,
+        input_file_size=args.input_file_size,
+        attribute_count=args.attribute_count,
+        attribute_size=args.attribute_size,
+        metrics_interval=args.metrics_interval,
+        output=output_path,
+    )
+    return run_benchmark.run(combo_args)
+
+
+def main() -> None:
+    parser = argparse.ArgumentParser(
+        description="Run the MiNiFi C++ repository benchmark across several 
repository combinations sequentially and (optionally) generate a single 
comparison report. "
+                    "Runs are sequential by design so containers do not 
compete for CPU/IO.")
+    parser.add_argument("--image", required=True, help="Docker image to use 
for the benchmark.")
+    parser.add_argument("--combo", required=True, action="append", 
type=parse_combo, dest="combos",
+                        metavar="FLOWFILE:CONTENT",
+                        help="Repository combination to benchmark, e.g. 
--combo lmdb:lmdb. Repeatable.")
+    parser.add_argument("--input-file-generation-type", 
default=InputGenerationType.TIMED_GETFILE.value,
+                        choices=sorted([e.value for e in InputGenerationType]),
+                        help="Input file generation type shared by all combos 
(see run_benchmark.py).")
+    parser.add_argument("--input-file-count", type=int, default=100,
+                        help="Number of input files for burst input generation 
type (default: 100).")
+    parser.add_argument("--duration", type=int, default=120,
+                        help="Total benchmark session length in seconds 
(default: 120).")
+    parser.add_argument("--input-interval", type=float, default=1.0,
+                        help="Seconds between input file generation cycles 
(default: 1).")
+    parser.add_argument("--input-file-size", type=humanfriendly.parse_size, 
default=humanfriendly.parse_size("1M"),
+                        help="Size of each generated input file, e.g. 512K, 
1M, 1G (default: 1M).")
+    parser.add_argument("--attribute-count", type=int, default=0,
+                        help="Number of extra attributes to set on each flow 
file via UpdateAttribute (default: 0).")
+    parser.add_argument("--attribute-size", type=humanfriendly.parse_size, 
default=0,
+                        help="Size of each extra attribute value, e.g. 64, 1K 
(default: 0).")
+    parser.add_argument("--metrics-interval", type=float, default=5.0,
+                        help="Seconds between metric samples (default: 5).")
+    parser.add_argument("--repeat", type=int, default=1,
+                        help="Number of times to run each combo; results are 
aggregated per combo in the report (default: 1).")
+    parser.add_argument("--output-dir", default=run_benchmark.RESULTS_DIR,
+                        help="Directory for result JSON files (default: 
results/).")
+    parser.add_argument("--report", default=None,
+                        help="If set, generate an HTML report at this path 
from all successful runs.")
+    args = parser.parse_args()
+
+    os.makedirs(args.output_dir, exist_ok=True)
+
+    result_paths: list[str] = []
+    failures: list[tuple[str, str, str]] = []
+    total_runs = len(args.combos) * args.repeat
+    for index, (flowfile, content) in enumerate(args.combos, start=1):
+        for rep in range(1, args.repeat + 1):
+            print(f"\n=== [combo {index}/{len(args.combos)} rep 
{rep}/{args.repeat}] Benchmarking {flowfile}/{content} ===")
+            try:
+                result_paths.append(run_combo(args, flowfile, content, 
args.output_dir, rep))
+            except Exception as error:
+                print(f"Error: combo {flowfile}/{content} (rep {rep}) failed 
with: {error}")
+                failures.append((flowfile, content, str(error)))
+
+    print("\n=== Batch summary ===")
+    print(f"Succeeded: {len(result_paths)}/{total_runs}")
+    for flowfile, content, error in failures:
+        print(f"  FAILED {flowfile}/{content}: {error}")
+
+    if args.report and result_paths:
+        import generate_report
+        generate_report.write_report(result_paths, args.report)

Review Comment:
   Fixed in 
https://github.com/apache/nifi-minifi-cpp/pull/2245/commits/a6d1495efb23ba3081ddf4beb6970d3e30fe9c83



-- 
This is an automated message from the Apache Git Service.
To respond to the message, please log on to GitHub and use the
URL above to go to the specific comment.

To unsubscribe, e-mail: [email protected]

For queries about this service, please contact Infrastructure at:
[email protected]

Reply via email to