diff --git a/src/nu/validator/htmlparser/common/TokenHandler.java b/src/nu/validator/htmlparser/common/TokenHandler.java
index 8c05b5b1..ba4f79a7 100755
--- a/src/nu/validator/htmlparser/common/TokenHandler.java
+++ b/src/nu/validator/htmlparser/common/TokenHandler.java
@@ -119,6 +119,22 @@ public void startTag(ElementName eltName, HtmlAttributes attributes,
*/
public void comment(@NoLength char[] buf, int start, int length) throws SAXException;
+ /**
+ * Receive a processing instruction token.
+ *
+ * @param target
+ * the target of the processing instruction
+ * @param buf
+ * a buffer holding the data
+ * @param start the offset into the buffer
+ * @param length
+ * the number of code units to read
+ * @throws SAXException
+ * if something went wrong
+ */
+ public void processingInstruction(String target, @NoLength char[] buf,
+ int start, int length) throws SAXException;
+
/**
* Receive character tokens. This method has the same semantics as the SAX
* method of the same name.
diff --git a/src/nu/validator/htmlparser/dom/DOMTreeBuilder.java b/src/nu/validator/htmlparser/dom/DOMTreeBuilder.java
index aa8cccba..3b5eb43c 100644
--- a/src/nu/validator/htmlparser/dom/DOMTreeBuilder.java
+++ b/src/nu/validator/htmlparser/dom/DOMTreeBuilder.java
@@ -149,6 +149,36 @@ protected DOMTreeBuilder(DOMImplementation implementation) {
}
}
+ /**
+ *
+ * @see nu.validator.htmlparser.impl.CoalescingTreeBuilder#appendProcessingInstruction(java.lang.Object,
+ * java.lang.String, java.lang.String)
+ */
+ @Override protected void appendProcessingInstruction(Element parent,
+ String target, String data) throws SAXException {
+ try {
+ parent.appendChild(
+ document.createProcessingInstruction(target, data));
+ } catch (DOMException e) {
+ fatal(e);
+ }
+ }
+
+ /**
+ *
+ * @see nu.validator.htmlparser.impl.CoalescingTreeBuilder#appendProcessingInstructionToDocument(java.lang.String,
+ * java.lang.String)
+ */
+ @Override protected void appendProcessingInstructionToDocument(
+ String target, String data) throws SAXException {
+ try {
+ document.appendChild(
+ document.createProcessingInstruction(target, data));
+ } catch (DOMException e) {
+ fatal(e);
+ }
+ }
+
/**
*
* @see nu.validator.htmlparser.impl.TreeBuilder#createElement(String, String, nu.validator.htmlparser.impl.HtmlAttributes, Object)
diff --git a/src/nu/validator/htmlparser/impl/CoalescingTreeBuilder.java b/src/nu/validator/htmlparser/impl/CoalescingTreeBuilder.java
index 951c8498..3892900d 100644
--- a/src/nu/validator/htmlparser/impl/CoalescingTreeBuilder.java
+++ b/src/nu/validator/htmlparser/impl/CoalescingTreeBuilder.java
@@ -71,6 +71,32 @@ protected final void accumulateCharacters(@NoLength char[] buf, int start,
protected abstract void appendCommentToDocument(String comment) throws SAXException;
+ /**
+ * @see nu.validator.htmlparser.impl.TreeBuilder#appendProcessingInstruction(java.lang.Object, java.lang.String, char[], int, int)
+ */
+ @Override final protected void appendProcessingInstruction(T parent,
+ String target, char[] buf, int start, int length)
+ throws SAXException {
+ appendProcessingInstruction(parent, target,
+ new String(buf, start, length));
+ }
+
+ protected abstract void appendProcessingInstruction(T parent,
+ String target, String data) throws SAXException;
+
+ /**
+ * @see nu.validator.htmlparser.impl.TreeBuilder#appendProcessingInstructionToDocument(java.lang.String, char[], int, int)
+ */
+ @Override protected final void appendProcessingInstructionToDocument(
+ String target, char[] buf, int start, int length)
+ throws SAXException {
+ appendProcessingInstructionToDocument(target,
+ new String(buf, start, length));
+ }
+
+ protected abstract void appendProcessingInstructionToDocument(
+ String target, String data) throws SAXException;
+
/**
* @see nu.validator.htmlparser.impl.TreeBuilder#insertFosterParentedCharacters(char[], int, int, java.lang.Object, java.lang.Object)
*/
diff --git a/src/nu/validator/htmlparser/impl/ErrorReportingTokenizer.java b/src/nu/validator/htmlparser/impl/ErrorReportingTokenizer.java
index 6c8e7617..c58b9394 100644
--- a/src/nu/validator/htmlparser/impl/ErrorReportingTokenizer.java
+++ b/src/nu/validator/htmlparser/impl/ErrorReportingTokenizer.java
@@ -492,6 +492,25 @@ private boolean isAstralPrivateUse(int c) {
err("Saw \u201C\u201D. Probable cause: Attempt to use an XML processing instruction in HTML. (XML processing instructions are not supported in HTML.)");
}
+ @Override protected void errEofInProcessingInstruction() throws SAXException {
+ err("End of file inside processing instruction.");
+ }
+
+ @Override protected void errInvalidFirstCharacterOfProcessingInstructionTarget()
+ throws SAXException {
+ err("Invalid first character of processing instruction target. Treating the processing instruction as a comment.");
+ }
+
+ @Override protected void errInvalidProcessingInstructionTarget()
+ throws SAXException {
+ err("Invalid character in processing instruction target. Treating the processing instruction as a comment.");
+ }
+
+ @Override protected void errDisallowedProcessingInstructionTarget()
+ throws SAXException {
+ err("Processing instruction targets \u201Cxml\u201D and \u201Cxml-stylesheet\u201D are not allowed. Treating the processing instruction as a comment.");
+ }
+
@Override protected void errUnescapedAmpersandInterpretedAsCharacterReference()
throws SAXException {
if (errorHandler == null) {
diff --git a/src/nu/validator/htmlparser/impl/Tokenizer.java b/src/nu/validator/htmlparser/impl/Tokenizer.java
index c414ec66..2acc2203 100755
--- a/src/nu/validator/htmlparser/impl/Tokenizer.java
+++ b/src/nu/validator/htmlparser/impl/Tokenizer.java
@@ -231,6 +231,16 @@ public class Tokenizer implements Locator, Locator2 {
public static final int COMMENT_LESSTHAN_BANG_DASH_DASH = 79;
+ public static final int PROCESSING_INSTRUCTION_OPEN = 80;
+
+ public static final int PROCESSING_INSTRUCTION_TARGET = 81;
+
+ public static final int AFTER_PROCESSING_INSTRUCTION_TARGET = 82;
+
+ public static final int PROCESSING_INSTRUCTION_DATA = 83;
+
+ public static final int PROCESSING_INSTRUCTION_QUESTIONABLE = 84;
+
/**
* Magic value for UTF-16 operations.
*/
@@ -242,6 +252,9 @@ public class Tokenizer implements Locator, Locator2 {
*/
private static final @NoLength char[] LT_GT = { '<', '>' };
+ private static final @NoLength char[] XML_STYLESHEET = { 'x', 'm', 'l',
+ '-', 's', 't', 'y', 'l', 'e', 's', 'h', 'e', 'e', 't' };
+
/**
* UTF-16 code unit array containing less than and solidus for emitting
* those characters on certain parse errors.
@@ -482,6 +495,17 @@ public class Tokenizer implements Locator, Locator2 {
*/
private String systemIdentifier;
+ /**
+ * The target of the current processing instruction token.
+ */
+ private String piTarget;
+
+ /**
+ * Whether <? starts a processing instruction instead of
+ * a bogus comment.
+ */
+ private boolean processingInstructionsEnabled;
+
/**
* The attribute holder.
*/
@@ -570,6 +594,8 @@ public Tokenizer(TokenHandler tokenHandler, boolean newAttributesEachTime) {
this.doctypeName = null;
this.publicIdentifier = null;
this.systemIdentifier = null;
+ this.piTarget = null;
+ this.processingInstructionsEnabled = false;
this.attributes = null;
this.shouldSuspend = false;
this.keepBuffer = false;
@@ -629,6 +655,8 @@ public Tokenizer(TokenHandler tokenHandler
this.doctypeName = null;
this.publicIdentifier = null;
this.systemIdentifier = null;
+ this.piTarget = null;
+ this.processingInstructionsEnabled = false;
// [NOCPP[
this.attributes = null;
// ]NOCPP]
@@ -647,6 +675,11 @@ public void setInterner(Interner interner) {
this.interner = interner;
}
+ public void setProcessingInstructionsEnabled(
+ boolean processingInstructionsEnabled) {
+ this.processingInstructionsEnabled = processingInstructionsEnabled;
+ }
+
public void initLocation(String newPublicId, String newSystemId) {
this.systemId = newSystemId;
this.publicId = newPublicId;
@@ -1149,6 +1182,51 @@ private void emitComment(int provisionalHyphens, int pos)
suspendIfRequestedAfterCurrentNonTextToken();
}
+ /**
+ * Checks whether the processing instruction target accumulated after the
+ * initial U+003F in the buffer is an ASCII case-insensitive match for
+ * "xml" or "xml-stylesheet".
+ */
+ private boolean isDisallowedPiTarget() {
+ int targetLength = strBufLen - 1;
+ if (targetLength != 3 && targetLength != 14) {
+ return false;
+ }
+ for (int i = 0; i < targetLength; i++) {
+ char folded = strBuf[i + 1];
+ if (folded >= 'A' && folded <= 'Z') {
+ folded += 0x20;
+ }
+ if (folded != XML_STYLESHEET[i]) {
+ return false;
+ }
+ }
+ return true;
+ }
+
+ /**
+ * Takes the target out of the buffer, leaving the buffer ready for
+ * accumulating the processing instruction data. The initial U+003F stays
+ * in the buffer only while the bogus comment fallback is still possible,
+ * so it is excluded here.
+ */
+ private void finishPiTarget() {
+ piTarget = Portability.newStringFromBuffer(strBuf, 1, strBufLen - 1
+ // CPPONLY: , tokenHandler, null
+ );
+ clearStrBufAfterUse();
+ }
+
+ private void emitProcessingInstruction(int pos) throws SAXException {
+ // CPPONLY: RememberGt(pos);
+ tokenHandler.processingInstruction(piTarget, strBuf, 0, strBufLen);
+ clearStrBufAfterUse();
+ Portability.releaseString(piTarget);
+ piTarget = null;
+ cstart = pos + 1;
+ suspendIfRequestedAfterCurrentNonTextToken();
+ }
+
/**
* Flushes coalesced character tokens.
*
@@ -1760,6 +1838,22 @@ private void ensureBufferSpace(int inputLength) throws SAXException {
// CPPONLY: pos);
// CPPONLY: continue stateloop;
// CPPONLY: }
+ if (processingInstructionsEnabled) {
+ /*
+ * U+003F QUESTION MARK (?) Set the
+ * temporary buffer to the empty string.
+ * Switch to the processing instruction
+ * open state.
+ *
+ * The U+003F is kept at the start of the
+ * buffer so that converting the buffer to
+ * a comment needs no further work.
+ */
+ clearStrBufBeforeUse();
+ appendStrBuf(c);
+ state = transition(state, Tokenizer.PROCESSING_INSTRUCTION_OPEN, reconsume, pos);
+ continue stateloop;
+ }
/*
* U+003F QUESTION MARK (?) Parse error.
*/
@@ -6299,6 +6393,278 @@ private void ensureBufferSpace(int inputLength) throws SAXException {
}
}
// no fallthrough, reordering opportunity
+ case PROCESSING_INSTRUCTION_OPEN:
+ if (++pos == endPos) {
+ break stateloop;
+ }
+ c = checkChar(buf, pos);
+ /*
+ * Consume the next input character:
+ */
+ if ((c >= 'a' && c <= 'z') || (c >= 'A' && c <= 'Z')
+ || c == '_') {
+ /*
+ * ASCII alpha, U+005F LOW LINE (_) Reconsume in the
+ * processing instruction target state.
+ */
+ appendStrBuf(c);
+ state = transition(state, Tokenizer.PROCESSING_INSTRUCTION_TARGET, reconsume, pos);
+ // Reconsuming the appended character is equivalent
+ // to falling through having consumed it.
+ } else {
+ /*
+ * Anything else This is an
+ * invalid-first-character-of-processing-instruction-target
+ * parse error. Convert the temporary buffer to a
+ * comment. Reconsume in the bogus comment state.
+ */
+ errInvalidFirstCharacterOfProcessingInstructionTarget();
+ reconsume = true;
+ state = transition(state, Tokenizer.BOGUS_COMMENT, reconsume, pos);
+ continue stateloop;
+ }
+ // CPPONLY: MOZ_FALLTHROUGH;
+ case PROCESSING_INSTRUCTION_TARGET:
+ pitargetloop: for (;;) {
+ if (++pos == endPos) {
+ break stateloop;
+ }
+ c = checkChar(buf, pos);
+ /*
+ * Consume the next input character:
+ */
+ if ((c >= 'a' && c <= 'z') || (c >= 'A' && c <= 'Z')
+ || (c >= '0' && c <= '9') || c == '-'
+ || c == '_') {
+ /*
+ * ASCII alphanumeric, U+002D HYPHEN-MINUS (-),
+ * U+005F LOW LINE (_) Append the current input
+ * character to the temporary buffer.
+ */
+ appendStrBuf(c);
+ continue;
+ }
+ switch (c) {
+ case '\r':
+ case '\n':
+ case '\u000C':
+ case ' ':
+ case '\t':
+ case '?':
+ case '>':
+ /*
+ * Let target be the concatenation of the code
+ * points in the temporary buffer.
+ */
+ if (isDisallowedPiTarget()) {
+ /*
+ * If target is an ASCII case-insensitive
+ * match for "xml" or "xml-stylesheet",
+ * this is a
+ * disallowed-processing-instruction-target
+ * parse error. Convert the temporary
+ * buffer to a comment. Reconsume in the
+ * bogus comment state.
+ */
+ errDisallowedProcessingInstructionTarget();
+ reconsume = true;
+ state = transition(state, Tokenizer.BOGUS_COMMENT, reconsume, pos);
+ continue stateloop;
+ }
+ /*
+ * Otherwise, create a processing instruction
+ * token whose target is target and data is
+ * the empty string. Reconsume in the after
+ * processing instruction target state. The
+ * reconsumed character is handled here
+ * directly.
+ */
+ finishPiTarget();
+ if (c == '>') {
+ emitProcessingInstruction(pos);
+ state = transition(state, Tokenizer.DATA, reconsume, pos);
+ if (shouldSuspend) {
+ break stateloop;
+ }
+ continue stateloop;
+ }
+ if (c == '?') {
+ state = transition(state, Tokenizer.PROCESSING_INSTRUCTION_QUESTIONABLE, reconsume, pos);
+ continue stateloop;
+ }
+ if (c == '\r') {
+ silentCarriageReturn();
+ state = transition(state, Tokenizer.AFTER_PROCESSING_INSTRUCTION_TARGET, reconsume, pos);
+ break stateloop;
+ }
+ if (c == '\n') {
+ silentLineFeed();
+ }
+ state = transition(state, Tokenizer.AFTER_PROCESSING_INSTRUCTION_TARGET, reconsume, pos);
+ break pitargetloop;
+ default:
+ /*
+ * Anything else This is an
+ * invalid-processing-instruction-target parse
+ * error. Convert the temporary buffer to a
+ * comment. Reconsume in the bogus comment
+ * state.
+ */
+ errInvalidProcessingInstructionTarget();
+ if (c == '\u0000') {
+ c = '\uFFFD';
+ }
+ reconsume = true;
+ state = transition(state, Tokenizer.BOGUS_COMMENT, reconsume, pos);
+ continue stateloop;
+ }
+ }
+ // CPPONLY: MOZ_FALLTHROUGH;
+ case AFTER_PROCESSING_INSTRUCTION_TARGET:
+ afterpitargetloop: for (;;) {
+ if (++pos == endPos) {
+ break stateloop;
+ }
+ c = checkChar(buf, pos);
+ /*
+ * Consume the next input character:
+ */
+ switch (c) {
+ case '\r':
+ silentCarriageReturn();
+ break stateloop;
+ case '\n':
+ silentLineFeed();
+ // CPPONLY: MOZ_FALLTHROUGH;
+ case ' ':
+ case '\t':
+ case '\u000C':
+ /*
+ * U+0009 CHARACTER TABULATION (tab), U+000A
+ * LINE FEED (LF), U+000C FORM FEED (FF),
+ * U+0020 SPACE Ignore the character.
+ */
+ continue;
+ default:
+ /*
+ * Anything else Reconsume in the processing
+ * instruction data state.
+ */
+ reconsume = true;
+ state = transition(state, Tokenizer.PROCESSING_INSTRUCTION_DATA, reconsume, pos);
+ break afterpitargetloop;
+ }
+ }
+ // CPPONLY: MOZ_FALLTHROUGH;
+ case PROCESSING_INSTRUCTION_DATA:
+ pidataloop: for (;;) {
+ if (reconsume) {
+ reconsume = false;
+ } else {
+ if (++pos == endPos) {
+ break stateloop;
+ }
+ c = checkChar(buf, pos);
+ }
+ /*
+ * Consume the next input character:
+ */
+ switch (c) {
+ case '?':
+ /*
+ * U+003F QUESTION MARK (?) Switch to the
+ * processing instruction questionable state.
+ */
+ state = transition(state, Tokenizer.PROCESSING_INSTRUCTION_QUESTIONABLE, reconsume, pos);
+ break pidataloop;
+ case '>':
+ /*
+ * U+003E GREATER-THAN SIGN (>) Switch to the
+ * data state. Emit the current processing
+ * instruction token.
+ */
+ emitProcessingInstruction(pos);
+ state = transition(state, Tokenizer.DATA, reconsume, pos);
+ if (shouldSuspend) {
+ break stateloop;
+ }
+ continue stateloop;
+ case '\r':
+ appendStrBufCarriageReturn();
+ break stateloop;
+ case '\n':
+ appendStrBufLineFeed();
+ continue;
+ case '\u0000':
+ c = '\uFFFD';
+ // CPPONLY: MOZ_FALLTHROUGH;
+ default:
+ /*
+ * Anything else Append the current input
+ * character to the current processing
+ * instruction token's data.
+ */
+ appendStrBuf(c);
+ continue;
+ }
+ }
+ // CPPONLY: MOZ_FALLTHROUGH;
+ case PROCESSING_INSTRUCTION_QUESTIONABLE:
+ for (;;) {
+ if (++pos == endPos) {
+ break stateloop;
+ }
+ c = checkChar(buf, pos);
+ /*
+ * Consume the next input character:
+ */
+ switch (c) {
+ case '>':
+ /*
+ * U+003E GREATER-THAN SIGN (>) Switch to the
+ * data state. Emit the current processing
+ * instruction token.
+ */
+ emitProcessingInstruction(pos);
+ state = transition(state, Tokenizer.DATA, reconsume, pos);
+ if (shouldSuspend) {
+ break stateloop;
+ }
+ continue stateloop;
+ case '?':
+ /*
+ * Anything else Append U+003F (?) to the
+ * data. Reconsuming the question mark in the
+ * data state leads back here.
+ */
+ appendStrBuf('?');
+ continue;
+ case '\u0000':
+ c = '\uFFFD';
+ // CPPONLY: MOZ_FALLTHROUGH;
+ default:
+ /*
+ * Anything else Append U+003F (?) to the
+ * current processing instruction token's
+ * data. Reconsume in the processing
+ * instruction data state.
+ */
+ appendStrBuf('?');
+ if (c == '\r') {
+ appendStrBufCarriageReturn();
+ state = transition(state, Tokenizer.PROCESSING_INSTRUCTION_DATA, reconsume, pos);
+ break stateloop;
+ }
+ if (c == '\n') {
+ appendStrBufLineFeed();
+ } else {
+ appendStrBuf(c);
+ }
+ state = transition(state, Tokenizer.PROCESSING_INSTRUCTION_DATA, reconsume, pos);
+ continue stateloop;
+ }
+ }
+ // no fallthrough, reordering opportunity
case PROCESSING_INSTRUCTION:
processinginstructionloop: for (;;) {
if (++pos == endPos) {
@@ -6665,6 +7031,21 @@ public void eof() throws SAXException {
* Reconsume the EOF character in the data state.
*/
break eofloop;
+ case PROCESSING_INSTRUCTION_OPEN:
+ case PROCESSING_INSTRUCTION_TARGET:
+ case AFTER_PROCESSING_INSTRUCTION_TARGET:
+ case PROCESSING_INSTRUCTION_DATA:
+ case PROCESSING_INSTRUCTION_QUESTIONABLE:
+ /*
+ * EOF This is an eof-in-processing-instruction parse
+ * error. Emit an end-of-file token.
+ */
+ errEofInProcessingInstruction();
+ if (piTarget != null) {
+ Portability.releaseString(piTarget);
+ piTarget = null;
+ }
+ break eofloop;
case BOGUS_COMMENT:
emitComment(0, 0);
break eofloop;
@@ -7211,6 +7592,11 @@ private void emitDoctypeToken(int pos) throws SAXException {
case CDATA_RSQB_RSQB:
case PROCESSING_INSTRUCTION:
case PROCESSING_INSTRUCTION_QUESTION_MARK:
+ case PROCESSING_INSTRUCTION_OPEN:
+ case PROCESSING_INSTRUCTION_TARGET:
+ case AFTER_PROCESSING_INSTRUCTION_TARGET:
+ case PROCESSING_INSTRUCTION_DATA:
+ case PROCESSING_INSTRUCTION_QUESTIONABLE:
break;
case CONSUME_CHARACTER_REFERENCE:
case CONSUME_NCR:
@@ -7295,6 +7681,10 @@ public void end() throws SAXException {
Portability.releaseString(publicIdentifier);
publicIdentifier = null;
}
+ if (piTarget != null) {
+ Portability.releaseString(piTarget);
+ piTarget = null;
+ }
tagName = null;
nonInternedTagName.setNameForNonInterned(null
// CPPONLY: , false
@@ -7359,6 +7749,10 @@ public int getCol() {
public void resetToDataState() {
clearStrBufAfterUse();
+ if (piTarget != null) {
+ Portability.releaseString(piTarget);
+ piTarget = null;
+ }
charRefBufLen = 0;
stateSave = Tokenizer.DATA;
// line = 1; XXX line numbers
@@ -7435,6 +7829,14 @@ public void loadState(Tokenizer other) throws SAXException {
publicIdentifier = Portability.newStringFromString(other.publicIdentifier);
}
+ Portability.releaseString(piTarget);
+ if (other.piTarget == null) {
+ piTarget = null;
+ } else {
+ piTarget = Portability.newStringFromString(other.piTarget);
+ }
+ processingInstructionsEnabled = other.processingInstructionsEnabled;
+
containsHyphen = other.containsHyphen;
if (other.tagName == null) {
tagName = null;
@@ -7558,6 +7960,21 @@ protected void errLtGt() throws SAXException {
protected void errProcessingInstruction() throws SAXException {
}
+ protected void errEofInProcessingInstruction() throws SAXException {
+ }
+
+ protected void errInvalidFirstCharacterOfProcessingInstructionTarget()
+ throws SAXException {
+ }
+
+ protected void errInvalidProcessingInstructionTarget()
+ throws SAXException {
+ }
+
+ protected void errDisallowedProcessingInstructionTarget()
+ throws SAXException {
+ }
+
protected void errUnescapedAmpersandInterpretedAsCharacterReference()
throws SAXException {
}
diff --git a/src/nu/validator/htmlparser/impl/TreeBuilder.java b/src/nu/validator/htmlparser/impl/TreeBuilder.java
index 464a9d40..0cc5f688 100644
--- a/src/nu/validator/htmlparser/impl/TreeBuilder.java
+++ b/src/nu/validator/htmlparser/impl/TreeBuilder.java
@@ -851,6 +851,45 @@ public final void comment(@NoLength char[] buf, int start, int length)
return;
}
+ public final void processingInstruction(String target, @NoLength char[] buf,
+ int start, int length) throws SAXException {
+ needToDropLF = false;
+ if (!isInForeign()) {
+ switch (mode) {
+ case INITIAL:
+ case BEFORE_HTML:
+ case AFTER_AFTER_BODY:
+ case AFTER_AFTER_FRAMESET:
+ /*
+ * A processing instruction token Append a
+ * ProcessingInstruction node to the Document object.
+ */
+ appendProcessingInstructionToDocument(target, buf, start,
+ length);
+ return;
+ case AFTER_BODY:
+ /*
+ * A processing instruction token Append a
+ * ProcessingInstruction node to the first element in the
+ * stack of open elements (the html element).
+ */
+ flushCharacters();
+ appendProcessingInstruction(stack[0].node, target, buf,
+ start, length);
+ return;
+ default:
+ break;
+ }
+ }
+ /*
+ * A processing instruction token Append a ProcessingInstruction node
+ * to the current node.
+ */
+ flushCharacters();
+ appendProcessingInstruction(stack[currentPtr].node, target, buf, start,
+ length);
+ }
+
/**
* @see nu.validator.htmlparser.common.TokenHandler#characters(char[], int,
* int)
@@ -5780,6 +5819,14 @@ protected abstract void appendComment(T parent, @NoLength char[] buf,
protected abstract void appendCommentToDocument(@NoLength char[] buf,
int start, int length) throws SAXException;
+ protected abstract void appendProcessingInstruction(T parent,
+ String target, @NoLength char[] buf, int start, int length)
+ throws SAXException;
+
+ protected abstract void appendProcessingInstructionToDocument(
+ String target, @NoLength char[] buf, int start, int length)
+ throws SAXException;
+
protected abstract void addAttributesToElement(T element,
HtmlAttributes attributes) throws SAXException;
diff --git a/src/nu/validator/htmlparser/sax/SAXStreamer.java b/src/nu/validator/htmlparser/sax/SAXStreamer.java
index 025a3c40..4e0628a4 100644
--- a/src/nu/validator/htmlparser/sax/SAXStreamer.java
+++ b/src/nu/validator/htmlparser/sax/SAXStreamer.java
@@ -77,6 +77,21 @@ protected void appendCommentToDocument(char[] buf, int start, int length)
}
}
+ @Override
+ protected void appendProcessingInstruction(Attributes parent,
+ String target, char[] buf, int start, int length)
+ throws SAXException {
+ contentHandler.processingInstruction(target,
+ new String(buf, start, length));
+ }
+
+ @Override
+ protected void appendProcessingInstructionToDocument(String target,
+ char[] buf, int start, int length) throws SAXException {
+ contentHandler.processingInstruction(target,
+ new String(buf, start, length));
+ }
+
@Override
protected Attributes createElement(String ns, String name, HtmlAttributes attributes, Attributes intendedParent) throws SAXException {
return attributes;
diff --git a/src/nu/validator/htmlparser/sax/SAXTreeBuilder.java b/src/nu/validator/htmlparser/sax/SAXTreeBuilder.java
index 06378e2b..7b1faadb 100644
--- a/src/nu/validator/htmlparser/sax/SAXTreeBuilder.java
+++ b/src/nu/validator/htmlparser/sax/SAXTreeBuilder.java
@@ -35,6 +35,7 @@
import nu.validator.saxtree.Element;
import nu.validator.saxtree.Node;
import nu.validator.saxtree.NodeType;
+import nu.validator.saxtree.ProcessingInstruction;
import nu.validator.saxtree.ParentNode;
class SAXTreeBuilder extends TreeBuilder {
@@ -59,6 +60,20 @@ protected void appendCommentToDocument(char[] buf, int start, int length) {
document.appendChild(new Comment(tokenizer, buf, start, length));
}
+ @Override
+ protected void appendProcessingInstruction(Element parent, String target,
+ char[] buf, int start, int length) {
+ parent.appendChild(new ProcessingInstruction(tokenizer, target,
+ new String(buf, start, length)));
+ }
+
+ @Override
+ protected void appendProcessingInstructionToDocument(String target,
+ char[] buf, int start, int length) {
+ document.appendChild(new ProcessingInstruction(tokenizer, target,
+ new String(buf, start, length)));
+ }
+
@Override
protected void appendCharacters(Element parent, char[] buf, int start, int length) {
parent.appendChild(new Characters(tokenizer, buf, start, length));
diff --git a/src/nu/validator/htmlparser/xom/SimpleNodeFactory.java b/src/nu/validator/htmlparser/xom/SimpleNodeFactory.java
index 147b5d93..7824d4d1 100644
--- a/src/nu/validator/htmlparser/xom/SimpleNodeFactory.java
+++ b/src/nu/validator/htmlparser/xom/SimpleNodeFactory.java
@@ -26,6 +26,7 @@
import nu.xom.Comment;
import nu.xom.Document;
import nu.xom.Element;
+import nu.xom.ProcessingInstruction;
import nu.xom.Text;
import nu.xom.Attribute.Type;
@@ -67,6 +68,17 @@ public Comment makeComment(String string) {
return new Comment(string);
}
+ /**
+ * return new ProcessingInstruction(target, data);
+ * @param target
+ * @param data
+ * @return
+ */
+ public ProcessingInstruction makeProcessingInstruction(String target,
+ String data) {
+ return new ProcessingInstruction(target, data);
+ }
+
/**
* return new Element(name, namespace);
* @param name
diff --git a/src/nu/validator/htmlparser/xom/XOMTreeBuilder.java b/src/nu/validator/htmlparser/xom/XOMTreeBuilder.java
index 54564326..f8a4f203 100644
--- a/src/nu/validator/htmlparser/xom/XOMTreeBuilder.java
+++ b/src/nu/validator/htmlparser/xom/XOMTreeBuilder.java
@@ -129,6 +129,35 @@ protected void appendCommentToDocument(String comment)
}
}
+ @Override
+ protected void appendProcessingInstruction(Element parent, String target,
+ String data) throws SAXException {
+ try {
+ parent.appendChild(
+ nodeFactory.makeProcessingInstruction(target, data));
+ } catch (XMLException e) {
+ fatal(e);
+ }
+ }
+
+ @Override
+ protected void appendProcessingInstructionToDocument(String target,
+ String data) throws SAXException {
+ try {
+ Element root = document.getRootElement();
+ if ("http://www.xom.nu/fakeRoot".equals(root.getNamespaceURI())) {
+ document.insertChild(
+ nodeFactory.makeProcessingInstruction(target, data),
+ document.indexOf(root));
+ } else {
+ document.appendChild(
+ nodeFactory.makeProcessingInstruction(target, data));
+ }
+ } catch (XMLException e) {
+ fatal(e);
+ }
+ }
+
@Override
protected Element createElement(String ns, String name,
HtmlAttributes attributes, Element intendedParent) throws SAXException {