Script 'mail_helper' called by obssrc Hello community, here is the log from the commit of package python-parsel for openSUSE:Factory checked in at 2026-09-29 17:53:24 ++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++ Comparing /work/SRC/openSUSE:Factory/python-parsel (Old) and /work/SRC/openSUSE:Factory/.python-parsel.new.383539 (New) ++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
Package is "python-parsel" Tue Sep 29 17:53:24 2026 rev:17 rq:1381373 version:1.12.1 Changes: -------- --- /work/SRC/openSUSE:Factory/python-parsel/python-parsel.changes 2026-09-28 10:44:18.090545123 +0200 +++ /work/SRC/openSUSE:Factory/.python-parsel.new.383539/python-parsel.changes 2026-09-29 17:55:37.190802811 +0200 @@ -1,0 +2,12 @@ +Tue Sep 29 08:26:14 UTC 2026 - Martin Pluskal <[email protected]> + +- Update to 1.12.1: + * Fixed unbounded memory growth on every xpath() and css() call + caused by the EXSLT namespaces being re-registered on a cached + evaluator. Needs lxml 5.4.0 or later and libxml2 2.13 or later + to reproduce; selection and extraction results are unchanged. +- Dropped the unneeded python-packaging dependency, upstream + stopped declaring it in 1.12.0 and the code never imports it. +- Spec cleanup: sorted the test BuildRequires alphabetically. + +------------------------------------------------------------------- Old: ---- parsel-1.12.0.tar.gz New: ---- parsel-1.12.1.tar.gz ++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++ Other differences: ------------------ ++++++ python-parsel.spec ++++++ --- /var/tmp/diff_new_pack.9yfJMs/_old 2026-09-29 17:55:37.997836467 +0200 +++ /var/tmp/diff_new_pack.9yfJMs/_new 2026-09-29 17:55:37.999836550 +0200 @@ -18,7 +18,7 @@ %{?sle15_python_module_pythons} Name: python-parsel -Version: 1.12.0 +Version: 1.12.1 Release: 0 Summary: Library to extract data from HTML and XML using XPath and CSS selectors License: BSD-3-Clause @@ -32,15 +32,14 @@ Requires: python-cssselect >= 1.2.0 Requires: python-jmespath >= 1.0.0 Requires: python-lxml >= 5.1 -Requires: python-packaging >= 23 Requires: python-w3lib >= 1.19.0 BuildArch: noarch # SECTION test requirements -BuildRequires: %{python_module pytest} BuildRequires: %{python_module cssselect >= 1.2.0} BuildRequires: %{python_module jmespath >= 1.0.0} BuildRequires: %{python_module lxml >= 5.1} BuildRequires: %{python_module psutil} +BuildRequires: %{python_module pytest} BuildRequires: %{python_module sybil} BuildRequires: %{python_module w3lib >= 1.19.0} # /SECTION ++++++ parsel-1.12.0.tar.gz -> parsel-1.12.1.tar.gz ++++++ diff -urN '--exclude=CVS' '--exclude=.cvsignore' '--exclude=.svn' '--exclude=.svnignore' old/parsel-1.12.0/NEWS new/parsel-1.12.1/NEWS --- old/parsel-1.12.0/NEWS 2020-02-02 01:00:00.000000000 +0100 +++ new/parsel-1.12.1/NEWS 2020-02-02 01:00:00.000000000 +0100 @@ -3,6 +3,12 @@ History ------- +1.12.1 (2026-09-28) +~~~~~~~~~~~~~~~~~~~ + +* Fixed memory usage growing with every ``xpath()`` and ``css()`` call when + using lxml 5.4.0 or later. + 1.12.0 (2026-09-25) ~~~~~~~~~~~~~~~~~~~ diff -urN '--exclude=CVS' '--exclude=.cvsignore' '--exclude=.svn' '--exclude=.svnignore' old/parsel-1.12.0/PKG-INFO new/parsel-1.12.1/PKG-INFO --- old/parsel-1.12.0/PKG-INFO 2020-02-02 01:00:00.000000000 +0100 +++ new/parsel-1.12.1/PKG-INFO 2020-02-02 01:00:00.000000000 +0100 @@ -1,6 +1,6 @@ Metadata-Version: 2.5 Name: parsel -Version: 1.12.0 +Version: 1.12.1 Summary: Parsel is a library to extract data from HTML and XML using XPath and CSS selectors Project-URL: Homepage, https://github.com/scrapy/parsel Project-URL: Documentation, https://parsel.readthedocs.io/en/latest/ diff -urN '--exclude=CVS' '--exclude=.cvsignore' '--exclude=.svn' '--exclude=.svnignore' old/parsel-1.12.0/parsel/__init__.py new/parsel-1.12.1/parsel/__init__.py --- old/parsel-1.12.0/parsel/__init__.py 2020-02-02 01:00:00.000000000 +0100 +++ new/parsel-1.12.1/parsel/__init__.py 2020-02-02 01:00:00.000000000 +0100 @@ -5,7 +5,7 @@ __author__ = "Scrapy project" __email__ = "[email protected]" -__version__ = "1.12.0" +__version__ = "1.12.1" __all__ = [ "Selector", "SelectorList", diff -urN '--exclude=CVS' '--exclude=.cvsignore' '--exclude=.svn' '--exclude=.svnignore' old/parsel-1.12.0/parsel/selector.py new/parsel-1.12.1/parsel/selector.py --- old/parsel-1.12.0/parsel/selector.py 2020-02-02 01:00:00.000000000 +0100 +++ new/parsel-1.12.1/parsel/selector.py 2020-02-02 01:00:00.000000000 +0100 @@ -390,6 +390,15 @@ return root, type_ +# lxml registers the functions of these namespaces on every call of an +# evaluator that maps them, and with libxml2 2.13+ every registration after the +# first adds an entry to the evaluator's error log, which cannot be cleared. +_EXSLT_NAMESPACES = frozenset( + f"http://exslt.org/{name}" + for name in ("dates-and-times", "math", "sets", "strings") +) + + @lru_cache(maxsize=2048) def _compile_xpath( query: str, namespaces: tuple[tuple[str, str], ...], smart_strings: bool @@ -397,6 +406,21 @@ return etree.XPath(query, namespaces=dict(namespaces), smart_strings=smart_strings) +def _get_xpath_evaluator( + query: str, namespaces: dict[str, str], smart_strings: bool +) -> etree.XPath: + namespaces = { + prefix: uri + for prefix, uri in namespaces.items() + if uri not in _EXSLT_NAMESPACES + or not isinstance(query, str) + or f"{prefix}:" in query + } + if _EXSLT_NAMESPACES.intersection(namespaces.values()): + return etree.XPath(query, namespaces=namespaces, smart_strings=smart_strings) + return _compile_xpath(query, tuple(sorted(namespaces.items())), smart_strings) + + def _get_root_type(root: Any, *, input_type: str | None) -> str: if isinstance(root, etree._Element): if input_type in {"json", "text"}: @@ -650,9 +674,7 @@ if namespaces is not None: nsp.update(namespaces) try: - xpathev = _compile_xpath( - query, tuple(sorted(nsp.items())), self._lxml_smart_strings - ) + xpathev = _get_xpath_evaluator(query, nsp, self._lxml_smart_strings) result = xpathev(root, **kwargs) except etree.XPathError as exc: raise ValueError(f"XPath error: {exc} in {query}") diff -urN '--exclude=CVS' '--exclude=.cvsignore' '--exclude=.svn' '--exclude=.svnignore' old/parsel-1.12.0/pyproject.toml new/parsel-1.12.1/pyproject.toml --- old/parsel-1.12.0/pyproject.toml 2020-02-02 01:00:00.000000000 +0100 +++ new/parsel-1.12.1/pyproject.toml 2020-02-02 01:00:00.000000000 +0100 @@ -61,7 +61,7 @@ ] [tool.bumpversion] -current_version = "1.12.0" +current_version = "1.12.1" commit = true tag = true tag_name = "v{new_version}" diff -urN '--exclude=CVS' '--exclude=.cvsignore' '--exclude=.svn' '--exclude=.svnignore' old/parsel-1.12.0/tests/test_selector.py new/parsel-1.12.1/tests/test_selector.py --- old/parsel-1.12.0/tests/test_selector.py 2020-02-02 01:00:00.000000000 +0100 +++ new/parsel-1.12.1/tests/test_selector.py 2020-02-02 01:00:00.000000000 +0100 @@ -1435,3 +1435,28 @@ class TestExsltBytes(TestExslt): sscls = SelectorBytesInput # type: ignore[assignment] + + +def test_repeated_queries_do_not_leak_memory() -> None: + tracemalloc = pytest.importorskip("tracemalloc") + text = "<html><body>" + '<p class="a">a</p>' * 20 + "</body></html>" + + def run(iterations: int) -> None: + for _ in range(iterations): + sel = Selector(text=text, namespaces={"s": "http://exslt.org/strings"}) + for p in sel.xpath("//p"): + p.xpath("string(.)").get() + p.xpath('s:padding(3, "x")').get() + p.xpath("has-class('a')").get() + p.css(".a::text").get() + sel.xpath("set:distinct(//p)").getall() + + run(10) + tracemalloc.start() + try: + before = tracemalloc.get_traced_memory()[0] + run(100) + after = tracemalloc.get_traced_memory()[0] + finally: + tracemalloc.stop() + assert after - before < 50_000
