aminghadersohi commented on code in PR #44656:
URL: https://github.com/apache/superset/pull/44656#discussion_r4170787616
##########
superset/mcp_service/server.py:
##########
@@ -301,30 +303,71 @@ def _strip_titles(obj: Any, in_properties_map: bool =
False) -> Any:
return obj
+_PARAGRAPH_BREAK = re.compile(r"\n\s*\n")
+# A list item, an IMPORTANT marker, or a first line ending in a colon
(heading).
+_STRUCTURED_PARAGRAPH = re.compile(
+ r"^\s*(?:[-*+]|\d+[.)])\s|^\W*IMPORTANT\b|\A[^\n]*:[ \t]*$", re.MULTILINE
+)
+
+
+def _complete_sentences(text: str, max_length: int) -> str:
+ """Return the longest prefix of *text* made of complete sentences."""
+ if max_length <= 0:
+ return ""
+ # Look one character past the budget so a boundary exactly at the limit
counts.
+ boundaries = re.finditer(r"[.!?](?=\s|$)", text[: max_length + 1])
+ ends = [match.end() for match in boundaries if match.end() <= max_length]
+ return text[: ends[-1]].strip() if ends else ""
+
+
def _truncate_description(text: str, max_length: int) -> str:
- """Truncate a tool description for search results.
-
- Cuts at the last sentence boundary before *max_length*, or at
- *max_length* with an ellipsis if no sentence boundary is found.
-
- Dedents first: Python 3.13 has the compiler strip a docstring's common
- leading whitespace at compile time (``__doc__`` comes out already
- cleaned), while 3.11/3.12 store it raw and leave that to the caller. A
- multi-line tool docstring's raw, un-dedented form is longer per line, so
- the same character budget lands at a different point in the text
- depending on which Python compiled it. Cleaning here first makes the cut
- point (and this function's callers' byte budgets) consistent regardless
- of interpreter version.
+ """Keep whole paragraphs, then whole sentences of the next one, within
budget.
+
+ Clean docstring indentation before applying the budget so the cut point
+ is consistent across Python versions that store docstrings differently.
"""
+ if max_length <= 0:
+ return ""
text = inspect.cleandoc(text) if text else text
if not text or len(text) <= max_length:
return text
- # Try to cut at the last sentence boundary
- truncated = text[:max_length]
- last_period = truncated.rfind(". ")
- if last_period > max_length // 2:
- return truncated[: last_period + 1]
- return truncated.rstrip() + "..."
+ kept, rest = "", text
+ for match in _PARAGRAPH_BREAK.finditer(text):
+ if match.start() > max_length:
+ break
+ kept, rest = text[: match.start()].strip(), text[match.end() :]
+ following = _PARAGRAPH_BREAK.split(rest, maxsplit=1)[0]
+ # Do not leave a heading, IMPORTANT block or list workflow partly
advertised.
+ if kept and _STRUCTURED_PARAGRAPH.search(following):
Review Comment:
Done in e2db70f291: a kept paragraph whose last sentence ends in a colon now
drops that sentence, so `update_dashboard` ends on the complete sentence
instead of `An LLM can:`. If it is the only sentence it is left alone. Unit
tests cover both cases.
--
This is an automated message from the Apache Git Service.
To respond to the message, please log on to GitHub and use the
URL above to go to the specific comment.
To unsubscribe, e-mail: [email protected]
For queries about this service, please contact Infrastructure at:
[email protected]
---------------------------------------------------------------------
To unsubscribe, e-mail: [email protected]
For additional commands, e-mail: [email protected]