Copilot commented on code in PR #4020:
URL: 
https://github.com/apache/incubator-kie-tools/pull/4020#discussion_r4069311007


##########
packages/xml-parser-ts/src/index.ts:
##########
@@ -74,10 +74,196 @@ export const domParser = {
     }
     // console.timeEnd("parsing dom took (DOMParser) parsererror");
 
+    registerAttributesSourceOrder(domdoc, xml.toString());
+
     return domdoc;
   },
 };
 
+/**
+ * The DOM standard doesn't guarantee that `Element.attributes` follows the 
order of the attributes on the XML source.
+ * `DOMParser` implementations do differ: Chrome and WebKit move `xmlns` 
declarations to the front, Chrome 153+
+ * additionally sorts them alphabetically, while jsdom keeps the source order. 
Since the JSON produced by `parse` (and
+ * thus the XML produced by `build`) follows the order of 
`Element.attributes`, the serialized XML would depend on where
+ * it was parsed, breaking round-trip fidelity. To avoid that, the order of 
the attributes of each element is recovered
+ * from the XML text, the only place where it is actually defined, and kept 
here keyed by the Document parsed from it.
+ */
+const attributesSourceOrderByDocument = new WeakMap<Document, Map<Element, 
string[]>>();
+
+function isXmlWhitespace(c: string) {
+  return c === " " || c === "\n" || c === "\t" || c === "\r";
+}
+
+/**
+ * Scans the XML text and returns, for each start tag in document order, the 
names of its attributes in the order they
+ * appear in the source. The returned array has one entry per element 
(possibly empty), so it can be aligned with the
+ * elements of the parsed Document. Comments, CDATA sections, processing 
instructions and the DOCTYPE are skipped.
+ * Attribute values may contain `>`.
+ */
+export function scanAttributeNamesInSourceOrder(xml: string): string[][] {
+  const result: string[][] = [];
+  const len = xml.length;
+  let i = 0;
+  while (i < len) {
+    const lt = xml.indexOf("<", i);
+    if (lt < 0) {
+      break;
+    }
+
+    if (xml.startsWith("<!--", lt)) {
+      const end = xml.indexOf("-->", lt + 4);
+      i = end < 0 ? len : end + 3;
+      continue;
+    }
+    if (xml.startsWith("<![CDATA[", lt)) {
+      const end = xml.indexOf("]]>", lt + 9);
+      i = end < 0 ? len : end + 3;
+      continue;
+    }
+    if (xml.startsWith("<?", lt)) {
+      const end = xml.indexOf("?>", lt + 2);
+      i = end < 0 ? len : end + 2;
+      continue;
+    }
+    if (xml.startsWith("<!", lt)) {
+      // DOCTYPE, possibly with an internal subset (`[ ... ]`) that contains 
`>`.
+      let j = lt + 2;
+      let depth = 0;
+      while (j < len) {
+        const c = xml[j];
+        if (c === "[") {
+          depth++;
+        } else if (c === "]") {
+          depth--;
+        } else if (c === ">" && depth <= 0) {
+          break;
+        }
+        j++;
+      }
+      i = j + 1;
+      continue;
+    }
+    if (xml.startsWith("</", lt)) {
+      const end = xml.indexOf(">", lt + 2);
+      i = end < 0 ? len : end + 1;
+      continue;
+    }
+
+    // Start tag (or empty-element tag). Skip the element name.
+    let j = lt + 1;
+    while (j < len && !isXmlWhitespace(xml[j]) && xml[j] !== ">" && xml[j] !== 
"/") {
+      j++;
+    }
+
+    const attributeNames: string[] = [];
+    while (j < len) {
+      while (j < len && isXmlWhitespace(xml[j])) {
+        j++;
+      }
+      if (j >= len) {
+        break;
+      }
+      const c = xml[j];
+      if (c === ">") {
+        j++;
+        break;
+      }
+      if (c === "/") {
+        j++;
+        continue;
+      }
+
+      const nameStart = j;
+      while (j < len && !isXmlWhitespace(xml[j]) && xml[j] !== "=" && xml[j] 
!== ">" && xml[j] !== "/") {
+        j++;
+      }
+      const attrName = xml.slice(nameStart, j);
+
+      while (j < len && isXmlWhitespace(xml[j])) {
+        j++;
+      }
+      if (xml[j] === "=") {
+        j++;
+        while (j < len && isXmlWhitespace(xml[j])) {
+          j++;
+        }
+        const quote = xml[j];
+        if (quote === '"' || quote === "'") {
+          const end = xml.indexOf(quote, j + 1);
+          j = end < 0 ? len : end + 1;
+        } else {
+          // Not well-formed (unquoted value). Consume it anyway to keep going.
+          while (j < len && !isXmlWhitespace(xml[j]) && xml[j] !== ">") {
+            j++;
+          }
+        }
+      }
+
+      if (attrName) {
+        attributeNames.push(attrName);
+      }
+    }
+
+    result.push(attributeNames);
+    i = j;
+  }
+  return result;
+}
+
+/**
+ * Associates `domdoc` with the order of the attributes found on `xml`, the 
text it was parsed from. Only elements with
+ * more than one attribute are recorded, since order is irrelevant otherwise. 
If the scanned start tags can't be
+ * aligned with the Document's elements (e.g., the XML is not well-formed), 
nothing is recorded and the order given by
+ * the DOM is used.
+ */
+export function registerAttributesSourceOrder(domdoc: Document, xml: string) {
+  const scanned = scanAttributeNamesInSourceOrder(xml);

Review Comment:
   A failed re-registration returns without clearing any source-order map 
already associated with this document. Consequently, calling this exported 
function with text that cannot be aligned can keep applying stale ordering 
instead of the documented DOM-order fallback. Clear the existing entry before 
attempting the new registration.



##########
packages/xml-parser-ts/src/index.ts:
##########
@@ -74,10 +74,196 @@ export const domParser = {
     }
     // console.timeEnd("parsing dom took (DOMParser) parsererror");
 
+    registerAttributesSourceOrder(domdoc, xml.toString());
+
     return domdoc;
   },
 };
 
+/**
+ * The DOM standard doesn't guarantee that `Element.attributes` follows the 
order of the attributes on the XML source.
+ * `DOMParser` implementations do differ: Chrome and WebKit move `xmlns` 
declarations to the front, Chrome 153+
+ * additionally sorts them alphabetically, while jsdom keeps the source order. 
Since the JSON produced by `parse` (and
+ * thus the XML produced by `build`) follows the order of 
`Element.attributes`, the serialized XML would depend on where
+ * it was parsed, breaking round-trip fidelity. To avoid that, the order of 
the attributes of each element is recovered
+ * from the XML text, the only place where it is actually defined, and kept 
here keyed by the Document parsed from it.
+ */
+const attributesSourceOrderByDocument = new WeakMap<Document, Map<Element, 
string[]>>();
+
+function isXmlWhitespace(c: string) {
+  return c === " " || c === "\n" || c === "\t" || c === "\r";
+}
+
+/**
+ * Scans the XML text and returns, for each start tag in document order, the 
names of its attributes in the order they
+ * appear in the source. The returned array has one entry per element 
(possibly empty), so it can be aligned with the
+ * elements of the parsed Document. Comments, CDATA sections, processing 
instructions and the DOCTYPE are skipped.
+ * Attribute values may contain `>`.
+ */
+export function scanAttributeNamesInSourceOrder(xml: string): string[][] {
+  const result: string[][] = [];
+  const len = xml.length;
+  let i = 0;
+  while (i < len) {
+    const lt = xml.indexOf("<", i);
+    if (lt < 0) {
+      break;
+    }
+
+    if (xml.startsWith("<!--", lt)) {
+      const end = xml.indexOf("-->", lt + 4);
+      i = end < 0 ? len : end + 3;
+      continue;
+    }
+    if (xml.startsWith("<![CDATA[", lt)) {
+      const end = xml.indexOf("]]>", lt + 9);
+      i = end < 0 ? len : end + 3;
+      continue;
+    }
+    if (xml.startsWith("<?", lt)) {
+      const end = xml.indexOf("?>", lt + 2);
+      i = end < 0 ? len : end + 2;
+      continue;
+    }
+    if (xml.startsWith("<!", lt)) {
+      // DOCTYPE, possibly with an internal subset (`[ ... ]`) that contains 
`>`.
+      let j = lt + 2;
+      let depth = 0;
+      while (j < len) {
+        const c = xml[j];
+        if (c === "[") {
+          depth++;
+        } else if (c === "]") {
+          depth--;
+        } else if (c === ">" && depth <= 0) {
+          break;
+        }
+        j++;
+      }

Review Comment:
   The DOCTYPE scanner counts `[` and `]` inside quoted literals, comments, and 
processing instructions as subset delimiters. For valid XML such as `<!DOCTYPE 
root [<!ENTITY open "[">]><root .../>`, `depth` never returns to zero, so 
scanning consumes the root tag and source-order registration falls back to the 
browser's reordered attributes. These lexical regions need to be skipped while 
balancing the internal subset.



-- 
This is an automated message from the Apache Git Service.
To respond to the message, please log on to GitHub and use the
URL above to go to the specific comment.

To unsubscribe, e-mail: [email protected]

For queries about this service, please contact Infrastructure at:
[email protected]


---------------------------------------------------------------------
To unsubscribe, e-mail: [email protected]
For additional commands, e-mail: [email protected]

Reply via email to