Hi
My previous email reported an Ivy defect and proposed correction.
The defect relates to parsing the POM XML when installing a module
from a Maven repository. The Ivy PomReader class defines a class
AddDTDFilterInputStream which inserts a DOCTYPE declaration into
the POM XML before executing the parsing operation. The defect is that
this would result in two DOCTYPE declarations if the POM already
had a DOCTYPE declaration. The fix removes any existing DOCTYPE
declaration before inserting the required declaration.
The fix assumes that if the POM contains an XML declaration
("<?xml version...?>") that this is not on the same line as
the root element declaration.
However, the POM in Maven Central for the module
{ "org": "xml-apis", "name": "xml-apis", "rev": "1.3.04" }
begins like this:
<?xml version="1.0" encoding="UTF-8"?><project>
<parent>
<artifactId>apache</artifactId>
This is transformed to:
<?xml version="1.0" encoding="UTF-8"?><project>
<!DOCTYPE project SYSTEM "m2-entities.ent">
<parent>
<artifactId>apache</artifactId>
which is clearly wrong and produces the exception
org.xml.sax.SAXException: Scanner State 24 not Recognized
I have rewritten the fix to parse the POM XML prolog with
a recursive-descent parser and separate lexical analyser.
The parser uses grammar production rules taken directly from
the W3C "Extensible Markup Language (XML) 1.0 (Fifth Edition)" document
[https://www.w3.org/TR/REC-xml/]. This is considerably more
general than the original fix but still makes
some simplifying assumptions. A completely general solution would
require executing a full-blown XML parser on the POM XML prolog.
The rewritten 'AddDTDFilterInputStream' class, contained in
'PomReader.java' (package org.apache.ivy.plugins.parser.m2),
is as follows:
private static final class AddDTDFilterInputStream extends
FilterInputStream {
/**
* Represents a token produced by lexical analysis of POM XML.
*/
public enum Token {
// Token Name Scanned text content of token.
LT, // <
START_XMLDECL, // <?xml
START_PI, // <?PITarget (PI=processing instruction)
QUERY_GT, // ?>
START_COMMENT, // <!--
END_COMMENT, // -->
START_DOCTYPE, // <!DOCTYPE
GT, // >
LEFT_BRACKET, // [
RIGHT_BRACKET, // ]
S, // (sp|tab|cr|nl)+
OTHER, // single character distinct from above
tokens
EOF // end-of-file
}
/**
* Performs lexical analysis of POM XML, producing tokens.
*/
private static final class Scanner{
/**
* Reader of document being scanned.
*/
private LineNumberReader reader;
/**
* Holds contents of last line read from {@link #reader}
* or null if end of stream reached.
*/
private String line;
/**
* Holds index in {@link #line} of first character not yet
incorporated
* in a token if the end of the input file has not been reached.
* Otherwise is undefined.
*/
private int index;
/**
* Token most-recently advanced-to.
*/
private Token token;
/**
* Holds text of {@link #token}.
*/
private StringBuilder tokenText = new StringBuilder();
/**
* Holds text of tokens saved.
*/
private StringBuilder saved = new StringBuilder();
/**
* Construct.
*
* Postconditions:
* First token from reader has been advanced-to.
*/
public Scanner(LineNumberReader reader) throws IOException {
this.reader = reader;
// Advance to first character in input file.
// Fake at end of line to force read from reader.
line = "x";
index = 0;
advanceChar();
tokenText.setLength(0);
// Incorporate input file characters in first token.
advance();
}
/**
* Add saved token text content to specified builder.
*/
public void addSavedTo(StringBuilder builder) {
builder.append(saved);
saved.setLength(0);
}
/**
* Advance to next token in input document if not at EOF.
*
* Postconditions:
* Value of token_text() on entry to this method has been
saved.
* token() and token_text() represent next token after token
on entry.
*/
public void advance() throws IOException {
saved.append(tokenText);
tokenText.setLength(0);
if (nextChar() == -1) {
token = Token.EOF;
} else if (nextChar() == '<') {
advanceChar();
if (nextChar() == '?') {
int startPos = tokenText.length() + 1;
do {
advanceChar();
} while (isNameChar(nextChar()));
if (tokenText.substring(startPos,
tokenText.length())
.equals("xml")) {
token = Token.START_XMLDECL;
} else {
token = Token.START_PI;
}
} else if (nextChar() == '!') {
advanceChar();
if (nextChar() == '-') {
advanceChar();
if (nextChar() == '-') {
token = Token.START_COMMENT;
advanceChar();
} else {
token = Token.OTHER;
}
} else if (nextChar() == 'D') {
int startPos = tokenText.length();
do {
advanceChar();
} while (isNameChar(nextChar()));
if (tokenText.substring(startPos,
tokenText.length())
.equals("DOCTYPE")) {
token = Token.START_DOCTYPE;
} else {
token = Token.OTHER;
}
} else {
token = Token.OTHER;
}
} else {
token = Token.LT;
}
} else if (nextChar() == '?') {
advanceChar();
if (nextChar() == '>') {
token = token.QUERY_GT;
advanceChar();
} else {
token = token.OTHER;
}
} else if (nextChar() == '-') {
advanceChar();
if (nextChar() == '-') {
advanceChar();
if (nextChar() == '>') {
token = token.END_COMMENT;
advanceChar();
} else {
token = token.OTHER;
}
} else {
token = token.OTHER;
}
} else if (nextChar() == '>') {
token = token.GT;
advanceChar();
} else if (nextChar() == '[') {
token = token.LEFT_BRACKET;
advanceChar();
} else if (nextChar() == ']') {
token = token.RIGHT_BRACKET;
advanceChar();
} else if ((nextChar() == ' ') ||
(nextChar() == '\t') ||
(nextChar() == '\r') ||
(nextChar() == '\n')) {
do {
advanceChar();
} while ((nextChar() == ' ') ||
(nextChar() == '\t') ||
(nextChar() == '\r') ||
(nextChar() == '\n'));
token = token.S;
} else {
token = token.OTHER;
advanceChar();
}
}
/**
* Advance to first token after specified token or EOF.
*/
public void advanceToAfter(Token finish) throws IOException {
do {
advance();
} while (!token().equals(finish) &&
!token().equals(Token.EOF));
advance();
}
/**
* Advance until at first token of specified list or EOF.
*/
public void advanceToFirstOf(Token... tokens) throws
IOException {
do {
advance();
} while (!contains(tokens, token()) &&
!token().equals(Token.EOF));
}
/**
* Discard tokens saved since last add or discard of saved token
text.
*/
public void discardSaved() {
saved.setLength(0);
}
/**
* Current token.
*/
public Token token() {
return token;
}
/**
* Text of current token.
*/
public String tokenText() {
return tokenText.toString();
}
/**
* Unused text of last line read.
*/
public String unusedLineText() {
return (index != -1)
? line.substring(index, line.length())
: "";
}
/**
* Advance to next character in input document
* if not already at EOF.
*/
private void advanceChar() throws IOException {
if (line != null) {
tokenText.append((char)nextChar());
if (line.length() <= ++index) {
int n = reader.getLineNumber();
line = reader.readLine();
if (n < reader.getLineNumber()) {
line += "\n";
}
index = 0;
}
}
}
/**
* Does the array contain the specified item?
*/
private static <T> boolean contains(final T[] a_array, final T
a_item) {
for (T array_item: a_array) {
if (a_item.equals(array_item)) {
return true;
}
}
return false;
}
/**
* Is specified character a NAME character?
*/
private boolean isNameChar(int c) {
return (('A' <= c) && (c <= 'Z')) ||
(('a' <= c) && (c <= 'z')) ||
(('0' <= c) && (c <= '9')) ||
(0 <= "_:.-".indexOf(c));
}
/**
* Next character in input document that has not
* been included in a token
* or -1 if EOF has been reached.
*/
private int nextChar() {
return (line == null) ? -1 : line.charAt(index);
}
}
private static final int MARK = 10000;
/**
* DOCTYPE to be inserted.
*/
private static final String DOCTYPE = "<!DOCTYPE project SYSTEM
\"m2-entities.ent\">";
private int count;
/**
* Will contain replacement prefix of document.
*/
private byte[] prefix;
private StringBuilder prefixBuilder = new StringBuilder();
/**
* Will hold lexical analyser of document.
*/
private Scanner scanner;
// Process "Misc*" (see grammar definition below).
//
private void misc_star() throws IOException {
for (;;) {
// Misc is Comment or PI or S.
if (scanner.token().equals(Token.START_COMMENT)) {
// Process "Comment".
scanner.advanceToAfter(Token.END_COMMENT);
} else if (scanner.token().equals(Token.START_PI)) {
// Process "PI".
scanner.advanceToAfter(Token.QUERY_GT);
} else if (scanner.token().equals(Token.S)) {
// Process "S".
scanner.advance();
} else {
// Not at start of "Misc".
break;
}
}
}
private AddDTDFilterInputStream(InputStream in) throws IOException {
super(new BufferedInputStream(in));
this.in.mark(MARK);
// TODO: we should really find a better solution for this...
// maybe we could use a FilterReader instead of a
FilterInputStream?
int byte1 = this.in.read();
int byte2 = this.in.read();
int byte3 = this.in.read();
if (byte1 == 239 && byte2 == 187 && byte3 == 191) {
// skip the UTF-8 BOM
this.in.mark(MARK);
} else {
this.in.reset();
}
// Read prefix of document up to and including any DOCTYPE
declaration.
// Construct replacement prefix by inserting POM DOCTYPE and
removing
// existing DOCTYPE, if present.
// Prefix of document is parsed according to following grammar.
// Document prefix is considered to be up to the start of the
root element.
//
// These production rules are taken directly from
// the W3C "Extensible Markup Language (XML) 1.0 (Fifth
Edition)" document
// [https://www.w3.org/TR/REC-xml/].
//
// The "Remarks" are added by me to indicate roughly how to
treat the rule using
// the tokens defined below. This is a simplification which
should be good enough
// given the relatively limited variation in POM document
content before the
// root element.
//
// document ::= prolog element Misc*
//
// prolog ::= XMLDecl? Misc* (doctypedecl Misc*)?
//
// XMLDecl ::= '<?xml' VersionInfo EncodingDecl?
SDDecl? S? '?>'
// Remark: This can be treated as:
<?xml .* ?>
//
// Misc ::= Comment | PI | S
// Remark: "S" is space. See Tokens
below.
//
// Comment ::= '<!--' ((Char - '-') | ('-' (Char -
'-')))* '-->'
// Remark: This can be treated as:
<!-- .* -->
//
// PI ::= '<?' PITarget (S (Char* - (Char* '?>'
Char*)))? '?>'
// Remark: This can be treated as:
<? .* ?>
//
// PITarget ::= Name - (('X' | 'x') ('M' | 'm') ('L' |
'l'))
// Remark: "PI" is Processing
Instruction.
//
// doctypedecl ::= '<!DOCTYPE' S Name (S ExternalID)? S?
('[' intSubset ']' S?)? '>'
// Remark: This can be treated as:
// <!DOCTYPE .* [ .* ] >
// or
// <!DOCTYPE .* >
//
// Tokens (produced by Scanner)
//
// LT <
// START_XMLDECL <?xml
// START_PI <?PITarget
// QUERY_GT ?>
// START_COMMENT <!--
// END_COMMENT -->
// START_DOCTYPE <!DOCTYPE
// GT >
// LEFT_BRACKET [
// RIGHT_BRACKET ]
// NAME [A-Za-z_:][A-Za-z0-9_:.-]*
// S (sp|tab|cr|nl)+
// OTHER ?
//
// The parser is essentially a recursive-descent parser,
although the grammar
// is simple enough that there is not much recursing.
LineNumberReader reader =
new LineNumberReader(
new InputStreamReader(this.in,
StandardCharsets.UTF_8),
100);
scanner = new Scanner(reader);
// At start of "prolog".
// Starts with possible "XMLDecl".
if (scanner.token().equals(Token.START_XMLDECL)) {
// Process "XMLDecl".
scanner.advanceToAfter(Token.QUERY_GT);
}
// At start of "Misc*".
misc_star();
// Add all tokens read (and saved) up to this point to the
prefix
// being constructed and clear saved tokens.
scanner.addSavedTo(prefixBuilder);
// At start of possible "doctypedecl Misc*".
if (scanner.token().equals(Token.START_DOCTYPE)) {
// Process "doctypedecl". May contain "[...]".
scanner.advanceToFirstOf(Token.LEFT_BRACKET, Token.GT);
if (scanner.token().equals(Token.LEFT_BRACKET)) {
// Process content of "[...]" and then up to ">".
scanner.advanceToAfter(Token.RIGHT_BRACKET);
scanner.advanceToAfter(Token.GT);
}
else {
// Advance to after '>'.
scanner.advance();
}
// Do not add the "doctypedecl" just read to the prefix
// being constructed (by clearing saved tokens).
scanner.discardSaved();
// Add the required "doctypedecl" to the prefix
// being constructed in place of the discarded
"doctypedecl".
prefixBuilder.append(DOCTYPE);
// Process "Misc*".
misc_star();
// Add the "Misc*" tokens to the prefix being constructed.
scanner.addSavedTo(prefixBuilder);
} else {
// There is no "doctypedecl" in the document being
processed.
// Add the required "doctypedecl" at this point to the
prefix
// being constructed.
prefixBuilder.append(DOCTYPE).append('\n');
}
// Have now constructed tokens for the prolog and the first
token
// of the (root) element. The prefix being constructed contains
// everything needed for the prolog. However the input skipped
// below is everything up to the end of the current line buffer,
// so we need to add to the constructed prefix the text of the
// first token of the root element and all text after this in
// the line buffer (so that the constructed prefix replaces
// exactly all the original text up to the end of the line
buffer).
prefixBuilder.append(scanner.tokenText())
.append(scanner.unusedLineText());
// The prefix being constructed is now complete and no more
// preprocessing of the document is required.
prefix = prefixBuilder.toString().getBytes();
// Reset input position to just after prefix that was read.
int lines_skipped = 0;
this.in.reset();
do {
int c = this.in.read();
if (c == -1) {
break;
}
if (c == '\n') {
++lines_skipped;
}
} while (lines_skipped < reader.getLineNumber());
}
@Override
public int read() throws IOException {
if (count < prefix.length) {
return prefix[count++];
}
return super.read();
}
@Override
public int read(byte[] b, int off, int len) throws IOException {
if (b == null) {
throw new NullPointerException();
} else if (off < 0 || off > b.length || len < 0 || (off + len)
> b.length
|| (off + len) < 0) {
throw new IndexOutOfBoundsException();
} else if (len == 0) {
return 0;
}
int nbrBytesCopied = 0;
if (count < prefix.length) {
int nbrBytesFromPrefix = Math.min(prefix.length - count,
len);
System.arraycopy(prefix, count, b, off, nbrBytesFromPrefix);
nbrBytesCopied = nbrBytesFromPrefix;
}
if (nbrBytesCopied < len) {
nbrBytesCopied += in.read(b, off + nbrBytesCopied, len -
nbrBytesCopied);
}
count += nbrBytesCopied;
return nbrBytesCopied;
}
}
Regards
Colin Chambers