gabotorresruiz commented on code in PR #44656:
URL: https://github.com/apache/superset/pull/44656#discussion_r4170670093
##########
superset/mcp_service/server.py:
##########
@@ -301,30 +303,71 @@ def _strip_titles(obj: Any, in_properties_map: bool =
False) -> Any:
return obj
+_PARAGRAPH_BREAK = re.compile(r"\n\s*\n")
+# A list item, an IMPORTANT marker, or a first line ending in a colon
(heading).
+_STRUCTURED_PARAGRAPH = re.compile(
+ r"^\s*(?:[-*+]|\d+[.)])\s|^\W*IMPORTANT\b|\A[^\n]*:[ \t]*$", re.MULTILINE
+)
+
+
+def _complete_sentences(text: str, max_length: int) -> str:
+ """Return the longest prefix of *text* made of complete sentences."""
+ if max_length <= 0:
+ return ""
+ # Look one character past the budget so a boundary exactly at the limit
counts.
+ boundaries = re.finditer(r"[.!?](?=\s|$)", text[: max_length + 1])
+ ends = [match.end() for match in boundaries if match.end() <= max_length]
+ return text[: ends[-1]].strip() if ends else ""
+
+
def _truncate_description(text: str, max_length: int) -> str:
- """Truncate a tool description for search results.
-
- Cuts at the last sentence boundary before *max_length*, or at
- *max_length* with an ellipsis if no sentence boundary is found.
-
- Dedents first: Python 3.13 has the compiler strip a docstring's common
- leading whitespace at compile time (``__doc__`` comes out already
- cleaned), while 3.11/3.12 store it raw and leave that to the caller. A
- multi-line tool docstring's raw, un-dedented form is longer per line, so
- the same character budget lands at a different point in the text
- depending on which Python compiled it. Cleaning here first makes the cut
- point (and this function's callers' byte budgets) consistent regardless
- of interpreter version.
+ """Keep whole paragraphs, then whole sentences of the next one, within
budget.
+
+ Clean docstring indentation before applying the budget so the cut point
+ is consistent across Python versions that store docstrings differently.
"""
+ if max_length <= 0:
+ return ""
text = inspect.cleandoc(text) if text else text
if not text or len(text) <= max_length:
return text
- # Try to cut at the last sentence boundary
- truncated = text[:max_length]
- last_period = truncated.rfind(". ")
- if last_period > max_length // 2:
- return truncated[: last_period + 1]
- return truncated.rstrip() + "..."
+ kept, rest = "", text
+ for match in _PARAGRAPH_BREAK.finditer(text):
+ if match.start() > max_length:
+ break
+ kept, rest = text[: match.start()].strip(), text[match.end() :]
+ following = _PARAGRAPH_BREAK.split(rest, maxsplit=1)[0]
+ # Do not leave a heading, IMPORTANT block or list workflow partly
advertised.
+ if kept and _STRUCTURED_PARAGRAPH.search(following):
Review Comment:
Just a small NIT, and it is the same family as my earlier paragraph note
rather than anything new. The guard asks whether the *next* paragraph is
structured, but not whether the paragraph it keeps ends in a heading itself.
`update_dashboard` now serves `...for incremental metadata and styling
edits. An LLM can:` and stops there, because `An LLM can:` closes the kept
paragraph and the bullet list that answers it is the following one. I saw it on
the wire on this branch: 159 characters with 141 of the 300 unused, where base
served the first two bullets. Would trimming a trailing `:`-terminated sentence
off `kept` be worth it, so it ends on `...for incremental metadata and styling
edits.`? Entirely fine to leave the rule alone if you would rather not poke at
it again.
--
This is an automated message from the Apache Git Service.
To respond to the message, please log on to GitHub and use the
URL above to go to the specific comment.
To unsubscribe, e-mail: [email protected]
For queries about this service, please contact Infrastructure at:
[email protected]
---------------------------------------------------------------------
To unsubscribe, e-mail: [email protected]
For additional commands, e-mail: [email protected]