Skip to content
Draft
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
171 changes: 171 additions & 0 deletions benchmarks/test_named_tuple_factory_benchmark.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,171 @@
# Copyright ScyllaDB, Inc.
#
# Licensed under the Apache License, Version 2.0 (the "License");
# you may not use this file except in compliance with the License.
# You may obtain a copy of the License at
#
# http://www.apache.org/licenses/LICENSE-2.0
#
# Unless required by applicable law or agreed to in writing, software
# distributed under the License is distributed on an "AS IS" BASIS,
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
# See the License for the specific language governing permissions and
# limitations under the License.

"""
Benchmarks for named_tuple_factory with and without namedtuple class caching.

Run with: pytest benchmarks/test_named_tuple_factory_benchmark.py -v
Comment thread
mykaul marked this conversation as resolved.

Correctness tests for the same cache (cache-hit, cache-key, and eviction
behavior) live in tests/unit/test_row_factories.py instead of here, because
the project's wheel test commands only run `pytest tests/unit` -- code left
only under benchmarks/ is never exercised in CI. Only pure timing/benchmark
code belongs in this file.
"""

import re
from collections import namedtuple

import pytest

from cassandra.query import named_tuple_factory, _named_tuple_cache
from cassandra.util import _sanitize_identifiers


# ---------------------------------------------------------------------------
# Reference: original uncached implementation (copied from master)
# ---------------------------------------------------------------------------

NON_ALPHA_REGEX = re.compile("[^a-zA-Z0-9]")
START_BADCHAR_REGEX = re.compile("^[^a-zA-Z0-9]*")
END_BADCHAR_REGEX = re.compile("[^a-zA-Z0-9_]*$")

_clean_name_cache_old = {}


def _clean_column_name_old(name):
try:
return _clean_name_cache_old[name]
except KeyError:
clean = NON_ALPHA_REGEX.sub(
"_", START_BADCHAR_REGEX.sub("", END_BADCHAR_REGEX.sub("", name))
)
_clean_name_cache_old[name] = clean
return clean


def named_tuple_factory_uncached(colnames, rows):
"""Original implementation without caching (for benchmark comparison)."""
clean_column_names = map(_clean_column_name_old, colnames)
try:
Row = namedtuple("Row", clean_column_names)
except SyntaxError:
raise
except Exception:
clean_column_names = list(map(_clean_column_name_old, colnames))
Row = namedtuple("Row", _sanitize_identifiers(clean_column_names))
return [Row(*row) for row in rows]


# ---------------------------------------------------------------------------
# Test data generators
# ---------------------------------------------------------------------------


def make_colnames(n):
return tuple(f"col_{i}" for i in range(n))


def make_rows(ncols, nrows):
return [tuple(range(ncols)) for _ in range(nrows)]


# ---------------------------------------------------------------------------
# Benchmarks
# ---------------------------------------------------------------------------


class TestNamedTupleFactoryBenchmark:
"""Benchmark cached vs uncached named_tuple_factory."""

# --- 5 columns, 100 rows ---

@pytest.mark.benchmark(group="ntf_5cols_100rows")
def test_uncached_5cols_100rows(self, benchmark):
colnames = make_colnames(5)
rows = make_rows(5, 100)
benchmark(named_tuple_factory_uncached, colnames, rows)

@pytest.mark.benchmark(group="ntf_5cols_100rows")
def test_cached_5cols_100rows(self, benchmark):
colnames = make_colnames(5)
rows = make_rows(5, 100)
_named_tuple_cache.clear()
# Warm the cache with one call
named_tuple_factory(colnames, rows)
benchmark(named_tuple_factory, colnames, rows)

# --- 10 columns, 100 rows ---

@pytest.mark.benchmark(group="ntf_10cols_100rows")
def test_uncached_10cols_100rows(self, benchmark):
colnames = make_colnames(10)
rows = make_rows(10, 100)
benchmark(named_tuple_factory_uncached, colnames, rows)

@pytest.mark.benchmark(group="ntf_10cols_100rows")
def test_cached_10cols_100rows(self, benchmark):
colnames = make_colnames(10)
rows = make_rows(10, 100)
_named_tuple_cache.clear()
named_tuple_factory(colnames, rows)
benchmark(named_tuple_factory, colnames, rows)

# --- 20 columns, 100 rows ---

@pytest.mark.benchmark(group="ntf_20cols_100rows")
def test_uncached_20cols_100rows(self, benchmark):
colnames = make_colnames(20)
rows = make_rows(20, 100)
benchmark(named_tuple_factory_uncached, colnames, rows)

@pytest.mark.benchmark(group="ntf_20cols_100rows")
def test_cached_20cols_100rows(self, benchmark):
colnames = make_colnames(20)
rows = make_rows(20, 100)
_named_tuple_cache.clear()
named_tuple_factory(colnames, rows)
benchmark(named_tuple_factory, colnames, rows)

# --- 5 columns, 1000 rows ---

@pytest.mark.benchmark(group="ntf_5cols_1000rows")
def test_uncached_5cols_1000rows(self, benchmark):
colnames = make_colnames(5)
rows = make_rows(5, 1000)
benchmark(named_tuple_factory_uncached, colnames, rows)

@pytest.mark.benchmark(group="ntf_5cols_1000rows")
def test_cached_5cols_1000rows(self, benchmark):
colnames = make_colnames(5)
rows = make_rows(5, 1000)
_named_tuple_cache.clear()
named_tuple_factory(colnames, rows)
benchmark(named_tuple_factory, colnames, rows)

# --- 10 columns, 1 row (measures class creation overhead most clearly) ---

@pytest.mark.benchmark(group="ntf_10cols_1row")
def test_uncached_10cols_1row(self, benchmark):
colnames = make_colnames(10)
rows = make_rows(10, 1)
benchmark(named_tuple_factory_uncached, colnames, rows)

@pytest.mark.benchmark(group="ntf_10cols_1row")
def test_cached_10cols_1row(self, benchmark):
colnames = make_colnames(10)
rows = make_rows(10, 1)
_named_tuple_cache.clear()
named_tuple_factory(colnames, rows)
benchmark(named_tuple_factory, colnames, rows)
102 changes: 77 additions & 25 deletions cassandra/query.py
Original file line number Diff line number Diff line change
Expand Up @@ -22,6 +22,7 @@
from datetime import datetime, timedelta, timezone
import re
import struct
import threading
import time
import warnings

Expand Down Expand Up @@ -117,6 +118,34 @@ def pseudo_namedtuple_factory(colnames, rows):
for od in ordered_dict_factory(colnames, rows)]


# Cache namedtuple Row classes to avoid repeated exec() calls in namedtuple()
# for the same column schema. Keyed on the exact, ordered tuple of raw column
# names, so two schemas only share a cached class if their column names match
# exactly (same names, same case, same order); cleaning/sanitizing is a pure
# function of that key, so the derived Row class is always correct for it.
#
# For typical usage (a bounded set of prepared statements) this cache is
# naturally bounded by the number of distinct queries. Applications that
# build many ad hoc queries against highly variable/generated schemas could
# otherwise grow this without bound, so it is capped and evicted FIFO
# (oldest entry first, relying on dict insertion order) once full.
_named_tuple_cache = {}
_NAMED_TUPLE_CACHE_MAX_SIZE = 10000

# Guards the check-evict-insert sequence on the cache-miss path in
# named_tuple_factory() below. This driver supports free-threaded Python, so
# without synchronization, one thread iterating _named_tuple_cache to pick an
# eviction victim (`next(iter(...))`) can race with another thread mutating
# the same dict, raising `RuntimeError: dictionary changed size during
# iteration`; separately, two threads that both observe the cache under its
# size bound before either inserts can together push it past that bound. The
# cache-HIT path (the plain `_named_tuple_cache[key]` lookup above) does NOT
# take this lock -- concurrent reads of a dict are safe, and this cache is on
# a hot path where the whole point is to avoid paying synchronization cost on
# every call.
_named_tuple_cache_lock = threading.Lock()


def named_tuple_factory(colnames, rows):
"""
Returns each row as a `namedtuple <https://docs.python.org/2/library/collections.html#collections.namedtuple>`_.
Expand Down Expand Up @@ -146,32 +175,55 @@ def named_tuple_factory(colnames, rows):
.. versionchanged:: 2.0.0
moved from ``cassandra.decoder`` to ``cassandra.query``
"""
clean_column_names = map(_clean_column_name, colnames)
key = tuple(colnames)
try:
Row = namedtuple('Row', clean_column_names)
except SyntaxError:
warnings.warn(
"Failed creating namedtuple for a result because there were too "
"many columns. This is due to a Python limitation that affects "
"namedtuple in Python 3.0-3.6 (see issue18896). The row will be "
"created with {substitute_factory_name}, which lacks some namedtuple "
"features and is slower. To avoid slower performance accessing "
"values on row objects, Upgrade to Python 3.7, or use a different "
"row factory. (column names: {colnames})".format(
substitute_factory_name=pseudo_namedtuple_factory.__name__,
colnames=colnames
)
)
return pseudo_namedtuple_factory(colnames, rows)
except Exception:
clean_column_names = list(map(_clean_column_name, colnames)) # create list because py3 map object will be consumed by first attempt
log.warning("Failed creating named tuple for results with column names %s (cleaned: %s) "
"(see Python 'namedtuple' documentation for details on name rules). "
"Results will be returned with positional names. "
"Avoid this by choosing different names, using SELECT \"<col name>\" AS aliases, "
"or specifying a different row_factory on your Session" %
(colnames, clean_column_names))
Row = namedtuple('Row', _sanitize_identifiers(clean_column_names))
Row = _named_tuple_cache[key]
except KeyError:
Comment thread
mykaul marked this conversation as resolved.
# Miss path: synchronize the whole check-evict-insert sequence.
# Re-check the cache once we hold the lock in case another thread
# already populated `key` while we were waiting for it (the
# double-checked-locking pattern), so we don't do redundant work or
# clobber a class other callers may already hold a reference to.
with _named_tuple_cache_lock:
try:
Row = _named_tuple_cache[key]
except KeyError:
clean_column_names = map(_clean_column_name, colnames)
try:
Row = namedtuple('Row', clean_column_names)
except SyntaxError:
warnings.warn(
"Failed creating namedtuple for a result because there were too "
"many columns. This is due to a Python limitation that affects "
"namedtuple in Python 3.0-3.6 (see issue18896). The row will be "
"created with {substitute_factory_name}, which lacks some namedtuple "
"features and is slower. To avoid slower performance accessing "
"values on row objects, Upgrade to Python 3.7, or use a different "
"row factory. (column names: {colnames})".format(
substitute_factory_name=pseudo_namedtuple_factory.__name__,
colnames=colnames
)
)
return pseudo_namedtuple_factory(colnames, rows)
except Exception:
clean_column_names = list(map(_clean_column_name, colnames)) # create list because py3 map object will be consumed by first attempt
log.warning("Failed creating named tuple for results with column names %s (cleaned: %s) "
"(see Python 'namedtuple' documentation for details on name rules). "
"Results will be returned with positional names. "
"Avoid this by choosing different names, using SELECT \"<col name>\" AS aliases, "
"or specifying a different row_factory on your Session" %
(colnames, clean_column_names))
Row = namedtuple('Row', _sanitize_identifiers(clean_column_names))
if len(_named_tuple_cache) >= _NAMED_TUPLE_CACHE_MAX_SIZE:
# Evict the oldest entry (dicts preserve insertion order) to
# keep memory bounded when many distinct column-name schemas
# are seen. Safe under the lock: no other thread can be
# iterating or mutating _named_tuple_cache concurrently.
try:
_named_tuple_cache.pop(next(iter(_named_tuple_cache)))
except (StopIteration, KeyError):
pass
_named_tuple_cache[key] = Row

return [Row(*row) for row in rows]

Expand Down
1 change: 1 addition & 0 deletions pyproject.toml
Original file line number Diff line number Diff line change
Expand Up @@ -43,6 +43,7 @@ auth-kerberos = [
[dependency-groups]
dev = [
"pytest~=8.0",
"pytest-benchmark",
"PyYAML",
"pure-sasl",
"twisted[tls]",
Expand Down
Loading
Loading