bryancall commented on code in PR #13670:
URL: https://github.com/apache/trafficserver/pull/13670#discussion_r3997756961


##########
tools/benchmark/benchmark_Regex.cc:
##########
@@ -0,0 +1,728 @@
+/** @file
+
+  Benchmarks for the tsutil Regex wrapper: time per operation and the number 
and size
+  of heap allocations each operation makes.
+
+  The allocation half matters as much as the timing half. PCRE2 routes every 
allocation
+  it makes for a compile or a match through the callbacks the wrapper 
installs, and those
+  call the system allocator, so counting calls to malloc across a region 
counts exactly
+  what the wrapper caused. Under the just-in-time engine a match should reach 
the system
+  allocator zero times, because the match data comes out of the caller's own 
buffer. The
+  interpreter is the exception: it allocates a backtracking frames vector 
through the same
+  allocator, so a match that runs interpreted does show up in the count.
+
+  This file builds two targets. benchmark_Regex times operations and contains 
no
+  interposer at all, so a timed case measures the regex path and nothing else; 
that
+  matters because a wrapper call costs about 1.5 ns, which is a fifth of an 
operation as
+  short as copying a compiled pattern. benchmark_Regex_alloc is built with
+  BENCHMARK_REGEX_ALLOC, carries the interposer, and runs only the counting 
cases, where
+  the overhead is irrelevant because nothing is being timed.
+
+  Interposing malloc is only wired up on Linux, where defining these symbols 
in the
+  executable is enough. Elsewhere the counters stay at zero and the report 
says so, so a
+  run on another platform still gives timings without quietly reporting zero 
allocations
+  as a result.
+
+  @section license License
+
+  Licensed to the Apache Software Foundation (ASF) under one
+  or more contributor license agreements.  See the NOTICE file
+  distributed with this work for additional information
+  regarding copyright ownership.  The ASF licenses this file
+  to you under the Apache License, Version 2.0 (the
+  "License"); you may not use this file except in compliance
+  with the License.  You may obtain a copy of the License at
+
+      http://www.apache.org/licenses/LICENSE-2.0
+
+  Unless required by applicable law or agreed to in writing, software
+  distributed under the License is distributed on an "AS IS" BASIS,
+  WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+  See the License for the specific language governing permissions and
+  limitations under the License.
+ */
+
+#include <cstddef>
+#include <cstdint>
+#include <cstdio>
+#include <cstdlib>
+#include <cstring>
+#include <string>
+#include <string_view>
+#include <vector>
+
+#define CATCH_CONFIG_ENABLE_BENCHMARKING
+#include <catch2/catch_test_macros.hpp>
+#include <catch2/benchmark/catch_benchmark.hpp>
+
+#include "tsutil/Regex.h"
+
+#define PCRE2_CODE_UNIT_WIDTH 8
+#include <pcre2.h>
+
+// ---------------------------------------------------------------------------
+// Allocation counting
+// ---------------------------------------------------------------------------
+
+namespace
+{
+#if defined(BENCHMARK_REGEX_ALLOC)
+struct AllocStats {
+  unsigned long calls = 0;
+  unsigned long bytes = 0;
+};
+
+// Counting is per thread so a benchmark that spawns threads does not race the 
counters.
+// These benchmarks are single threaded; the qualifier is here so the numbers 
stay honest
+// if one is added later.
+thread_local AllocStats alloc_stats;
+thread_local bool       alloc_counting = false;
+
+class CountAllocations
+{
+public:
+  CountAllocations()
+  {
+    alloc_stats    = AllocStats{};
+    alloc_counting = true;
+  }
+  ~CountAllocations() { alloc_counting = false; }
+
+  AllocStats
+  stats() const
+  {
+    return alloc_stats;
+  }
+};
+
+#if defined(__linux__)
+constexpr bool ALLOC_COUNTING_AVAILABLE = true;
+#else
+constexpr bool ALLOC_COUNTING_AVAILABLE = false;
+#endif
+#endif // BENCHMARK_REGEX_ALLOC
+
+} // namespace
+
+#if defined(__linux__) && defined(BENCHMARK_REGEX_ALLOC)
+#include <dlfcn.h>
+
+// Interpose the system allocator. Defining these in the executable takes 
precedence over
+// libc for every caller in the process, which is what makes the count cover 
PCRE2's own
+// allocations as well as the wrapper's.
+namespace
+{
+using malloc_fn  = void *(*)(size_t);
+using free_fn    = void (*)(void *);
+using calloc_fn  = void *(*)(size_t, size_t);
+using realloc_fn = void *(*)(void *, size_t);
+
+malloc_fn  real_malloc  = nullptr;
+free_fn    real_free    = nullptr;
+calloc_fn  real_calloc  = nullptr;
+realloc_fn real_realloc = nullptr;
+
+// dlsym() itself can allocate while the real pointers are still being 
resolved. Hand
+// those few allocations out of a static buffer rather than recursing.
+//
+// Each block is preceded by a header holding its size, so a realloc of one 
can copy the
+// old contents rather than silently returning uninitialised storage. The 
header is one
+// max_align_t wide so the pointer handed back keeps the alignment malloc 
promises.
+constexpr size_t BOOTSTRAP_HEADER = alignof(std::max_align_t);
+static_assert(BOOTSTRAP_HEADER >= sizeof(size_t), "the bootstrap header must 
hold a size");
+
+alignas(std::max_align_t) char bootstrap_buffer[16384];
+size_t bootstrap_used = 0;
+bool   resolving      = false;
+
+// Compare addresses rather than pointers. Relational comparison of pointers 
into unrelated
+// objects has no defined ordering in C++, and a wrong answer here is not 
harmless: a false
+// positive makes free() leak the block and makes realloc() read a header that 
is not there.
+bool
+from_bootstrap(void *p)
+{
+  auto const address = reinterpret_cast<uintptr_t>(p);
+  auto const begin   = reinterpret_cast<uintptr_t>(bootstrap_buffer);
+  return address >= begin && address < begin + sizeof(bootstrap_buffer);
+}
+
+void *
+bootstrap_alloc(size_t size)
+{
+  // Check the request against what is left before rounding it up. Rounding 
first would let
+  // a huge size wrap to a small payload, pass the capacity test, and hand 
back storage far
+  // smaller than asked for. These wrappers stand in for malloc for every 
library in the
+  // process while dlsym resolves, so that block would corrupt somebody else's 
startup.
+  size_t const remaining = sizeof(bootstrap_buffer) - bootstrap_used;
+  if (remaining <= BOOTSTRAP_HEADER || size > remaining - BOOTSTRAP_HEADER) {
+    return nullptr;
+  }
+
+  size_t const payload = (size + alignof(std::max_align_t) - 1) & 
~(alignof(std::max_align_t) - 1);
+  if (payload > remaining - BOOTSTRAP_HEADER) {
+    return nullptr;
+  }
+  char *block = bootstrap_buffer + bootstrap_used;
+  memcpy(block, &size, sizeof(size));
+  bootstrap_used += BOOTSTRAP_HEADER + payload;
+  return block + BOOTSTRAP_HEADER;
+}
+
+size_t
+bootstrap_size(void *p)
+{
+  size_t size = 0;
+  memcpy(&size, static_cast<char *>(p) - BOOTSTRAP_HEADER, sizeof(size));
+  return size;
+}
+
+// Resolve all four into locals and publish them together, with real_malloc 
last. dlsym()
+// may allocate or free while these lookups are in progress, which re-enters 
the wrappers
+// below; they test their own pointer and fall back to the bootstrap path 
while it is still
+// null, so no wrapper can reach a half-resolved table.
+void
+resolve_real_allocators()
+{
+  if (real_malloc != nullptr || resolving) {
+    return;
+  }
+  resolving = true;
+
+  auto *m = reinterpret_cast<malloc_fn>(dlsym(RTLD_NEXT, "malloc"));
+  auto *f = reinterpret_cast<free_fn>(dlsym(RTLD_NEXT, "free"));
+  auto *c = reinterpret_cast<calloc_fn>(dlsym(RTLD_NEXT, "calloc"));
+  auto *r = reinterpret_cast<realloc_fn>(dlsym(RTLD_NEXT, "realloc"));
+
+  real_free    = f;
+  real_calloc  = c;
+  real_realloc = r;
+  real_malloc  = m; // published last: this is the pointer the early return 
above tests

Review Comment:
   Done in 2669b2b0a8, though the failure path described does not exist in the 
current code.
   
   free() has re-checked its own pointer since f70992eb6e: if real_free is 
still null after resolve_real_allocators() returns, it leaks the block rather 
than calling through null. All four wrappers do the same for their own pointer, 
so a partial table was never dereferenced.
   
   The change is still worth making and I have made it: the four pointers now 
publish together or not at all. One invariant is easier to keep true than four, 
and a dlsym failure means every wrapper should stay on its safe path rather 
than half of them proceeding.
   
   _🤖 Addressed by [Claude Code](https://claude.com/claude-code)_



-- 
This is an automated message from the Apache Git Service.
To respond to the message, please log on to GitHub and use the
URL above to go to the specific comment.

To unsubscribe, e-mail: [email protected]

For queries about this service, please contact Infrastructure at:
[email protected]

Reply via email to