gabotorresruiz commented on code in PR #44656:
URL: https://github.com/apache/superset/pull/44656#discussion_r4170670093


##########
superset/mcp_service/server.py:
##########
@@ -301,30 +303,71 @@ def _strip_titles(obj: Any, in_properties_map: bool = 
False) -> Any:
     return obj
 
 
+_PARAGRAPH_BREAK = re.compile(r"\n\s*\n")
+# A list item, an IMPORTANT marker, or a first line ending in a colon 
(heading).
+_STRUCTURED_PARAGRAPH = re.compile(
+    r"^\s*(?:[-*+]|\d+[.)])\s|^\W*IMPORTANT\b|\A[^\n]*:[ \t]*$", re.MULTILINE
+)
+
+
+def _complete_sentences(text: str, max_length: int) -> str:
+    """Return the longest prefix of *text* made of complete sentences."""
+    if max_length <= 0:
+        return ""
+    # Look one character past the budget so a boundary exactly at the limit 
counts.
+    boundaries = re.finditer(r"[.!?](?=\s|$)", text[: max_length + 1])
+    ends = [match.end() for match in boundaries if match.end() <= max_length]
+    return text[: ends[-1]].strip() if ends else ""
+
+
 def _truncate_description(text: str, max_length: int) -> str:
-    """Truncate a tool description for search results.
-
-    Cuts at the last sentence boundary before *max_length*, or at
-    *max_length* with an ellipsis if no sentence boundary is found.
-
-    Dedents first: Python 3.13 has the compiler strip a docstring's common
-    leading whitespace at compile time (``__doc__`` comes out already
-    cleaned), while 3.11/3.12 store it raw and leave that to the caller. A
-    multi-line tool docstring's raw, un-dedented form is longer per line, so
-    the same character budget lands at a different point in the text
-    depending on which Python compiled it. Cleaning here first makes the cut
-    point (and this function's callers' byte budgets) consistent regardless
-    of interpreter version.
+    """Keep whole paragraphs, then whole sentences of the next one, within 
budget.
+
+    Clean docstring indentation before applying the budget so the cut point
+    is consistent across Python versions that store docstrings differently.
     """
+    if max_length <= 0:
+        return ""
     text = inspect.cleandoc(text) if text else text
     if not text or len(text) <= max_length:
         return text
-    # Try to cut at the last sentence boundary
-    truncated = text[:max_length]
-    last_period = truncated.rfind(". ")
-    if last_period > max_length // 2:
-        return truncated[: last_period + 1]
-    return truncated.rstrip() + "..."
+    kept, rest = "", text
+    for match in _PARAGRAPH_BREAK.finditer(text):
+        if match.start() > max_length:
+            break
+        kept, rest = text[: match.start()].strip(), text[match.end() :]
+    following = _PARAGRAPH_BREAK.split(rest, maxsplit=1)[0]
+    # Do not leave a heading, IMPORTANT block or list workflow partly 
advertised.
+    if kept and _STRUCTURED_PARAGRAPH.search(following):

Review Comment:
   Just a small NIT, and it is the same family as my earlier paragraph note 
rather than anything new. The guard asks whether the *next* paragraph is 
structured, but not whether the paragraph it keeps ends in a heading itself.
   
   `update_dashboard` now serves `...for incremental metadata and styling 
edits. An LLM can:` and stops there, because `An LLM can:` closes the kept 
paragraph and the bullet list that answers it is the following one. I saw it on 
the wire on this branch: 159 characters with 141 of the 300 unused, where base 
served the first two bullets. Would trimming a trailing `:`-terminated sentence 
off `kept` be worth it, so it ends on `...for incremental metadata and styling 
edits.`? Entirely fine to leave the rule alone if you would rather not poke at 
it again.



-- 
This is an automated message from the Apache Git Service.
To respond to the message, please log on to GitHub and use the
URL above to go to the specific comment.

To unsubscribe, e-mail: [email protected]

For queries about this service, please contact Infrastructure at:
[email protected]


---------------------------------------------------------------------
To unsubscribe, e-mail: [email protected]
For additional commands, e-mail: [email protected]

Reply via email to