mirror of
https://gitflic.ru/project/openide/openide.git
synced 2026-09-27 10:03:11 +07:00
IJPL-163136 introduce XML lexer
GitOrigin-RevId: d98b50c28df6cc86ce964d6432e5518f498e5e8b
This commit is contained in:
committed by
intellij-monorepo-bot
parent
506a4b43cf
commit
77b888bda1
@@ -23,6 +23,24 @@
|
||||
- getTokenStart():I
|
||||
- getTokenType():com.intellij.platform.syntax.SyntaxElementType
|
||||
- start(java.lang.CharSequence,I,I,I):V
|
||||
*f:com.intellij.platform.syntax.util.lexer.FilterLexer
|
||||
- com.intellij.platform.syntax.util.lexer.DelegateLexer
|
||||
- <init>(com.intellij.platform.syntax.lexer.Lexer,com.intellij.platform.syntax.util.lexer.FilterLexer$Filter):V
|
||||
- <init>(com.intellij.platform.syntax.lexer.Lexer,com.intellij.platform.syntax.util.lexer.FilterLexer$Filter,Z[]):V
|
||||
- b:<init>(com.intellij.platform.syntax.lexer.Lexer,com.intellij.platform.syntax.util.lexer.FilterLexer$Filter,Z[],I,kotlin.jvm.internal.DefaultConstructorMarker):V
|
||||
- advance():V
|
||||
- getCurrentPosition():com.intellij.platform.syntax.lexer.LexerPosition
|
||||
- f:getOriginal():com.intellij.platform.syntax.lexer.Lexer
|
||||
- f:getPrevTokenEnd():I
|
||||
- f:locateToken():V
|
||||
- restore(com.intellij.platform.syntax.lexer.LexerPosition):V
|
||||
- start(java.lang.CharSequence,I,I,I):V
|
||||
*:com.intellij.platform.syntax.util.lexer.FilterLexer$Filter
|
||||
- a:reject(com.intellij.platform.syntax.SyntaxElementType):Z
|
||||
*f:com.intellij.platform.syntax.util.lexer.FilterLexer$SetFilter
|
||||
- com.intellij.platform.syntax.util.lexer.FilterLexer$Filter
|
||||
- <init>(com.intellij.platform.syntax.SyntaxElementTypeSet):V
|
||||
- reject(com.intellij.platform.syntax.SyntaxElementType):Z
|
||||
*c:com.intellij.platform.syntax.util.lexer.FlexAdapter
|
||||
- com.intellij.platform.syntax.util.lexer.LexerBase
|
||||
- <init>(com.intellij.platform.syntax.util.lexer.FlexLexer):V
|
||||
|
||||
+66
@@ -0,0 +1,66 @@
|
||||
// Copyright 2000-2023 JetBrains s.r.o. and contributors. Use of this source code is governed by the Apache 2.0 license.
|
||||
package com.intellij.platform.syntax.util.lexer
|
||||
|
||||
import com.intellij.platform.syntax.SyntaxElementType
|
||||
import com.intellij.platform.syntax.SyntaxElementTypeSet
|
||||
import com.intellij.platform.syntax.lexer.Lexer
|
||||
import com.intellij.platform.syntax.lexer.LexerPosition
|
||||
import org.jetbrains.annotations.ApiStatus
|
||||
import kotlin.jvm.JvmOverloads
|
||||
|
||||
@ApiStatus.Experimental
|
||||
class FilterLexer @JvmOverloads constructor(
|
||||
original: Lexer,
|
||||
private val filter: Filter?,
|
||||
private val stateFilter: BooleanArray? = null
|
||||
) : DelegateLexer(original) {
|
||||
var prevTokenEnd: Int = 0
|
||||
private set
|
||||
|
||||
interface Filter {
|
||||
fun reject(type: SyntaxElementType): Boolean
|
||||
}
|
||||
|
||||
class SetFilter(private val set: SyntaxElementTypeSet) : Filter {
|
||||
override fun reject(type: SyntaxElementType): Boolean = type in set
|
||||
}
|
||||
|
||||
val original: Lexer
|
||||
get() = delegate
|
||||
|
||||
override fun start(buffer: CharSequence, startOffset: Int, endOffset: Int, initialState: Int) {
|
||||
super.start(buffer, startOffset, endOffset, initialState)
|
||||
prevTokenEnd = -1
|
||||
locateToken()
|
||||
}
|
||||
|
||||
|
||||
override fun advance() {
|
||||
prevTokenEnd = delegate.getTokenEnd()
|
||||
super.advance()
|
||||
locateToken()
|
||||
}
|
||||
|
||||
override fun getCurrentPosition(): LexerPosition {
|
||||
return delegate.getCurrentPosition()
|
||||
}
|
||||
|
||||
override fun restore(position: LexerPosition) {
|
||||
delegate.restore(position)
|
||||
this.prevTokenEnd = -1
|
||||
}
|
||||
|
||||
fun locateToken() {
|
||||
while (true) {
|
||||
val delegate = delegate
|
||||
val tokenType = delegate.getTokenType() ?: break
|
||||
if (filter == null || !filter.reject(tokenType)) {
|
||||
if (stateFilter == null || !stateFilter[delegate.getState()]) {
|
||||
break
|
||||
}
|
||||
}
|
||||
delegate.advance()
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
@@ -18,11 +18,8 @@ abstract class LexerTestCase : UsefulTestCase() {
|
||||
|
||||
protected abstract fun createLexer(): Lexer
|
||||
|
||||
fun doTest(text: String, expected: String) {
|
||||
doTest(text, expected, createLexer())
|
||||
}
|
||||
|
||||
private fun doTest(text: String, expected: String? = null, lexer: Lexer = createLexer()) {
|
||||
@JvmOverloads
|
||||
protected fun doTest(text: String, expected: String? = null, lexer: Lexer = createLexer()) {
|
||||
val result = printTokens(lexer, text, 0)
|
||||
|
||||
if (expected != null) {
|
||||
@@ -37,7 +34,7 @@ abstract class LexerTestCase : UsefulTestCase() {
|
||||
return printTokens(text, start, lexer)
|
||||
}
|
||||
|
||||
protected fun getPathToTestDataFile(extension: String): String {
|
||||
protected open fun getPathToTestDataFile(extension: String): String {
|
||||
return IdeaTestExecutionPolicy.getHomePathWithPolicy() + "/" + this.dirPath + "/" + getTestName(true) + extension
|
||||
}
|
||||
|
||||
|
||||
@@ -61,6 +61,8 @@ jvm_library(
|
||||
"//tools/intellij.tools.ide.metrics.benchmark:ide-metrics-benchmark_test_lib",
|
||||
"//platform/polySymbols:polySymbols-testFramework",
|
||||
"//libraries/xerces",
|
||||
"//platform/syntax/syntax-scripts:scripts",
|
||||
"//xml/xml-syntax:syntax",
|
||||
],
|
||||
runtime_deps = [
|
||||
":tests_test_resources",
|
||||
|
||||
@@ -1,14 +1,14 @@
|
||||
// Copyright 2000-2023 JetBrains s.r.o. and contributors. Use of this source code is governed by the Apache 2.0 license.
|
||||
package com.intellij.xml;
|
||||
|
||||
import com.intellij.lang.xml.XMLParserDefinition;
|
||||
import com.intellij.lexer.FilterLexer;
|
||||
import com.intellij.lexer.Lexer;
|
||||
import com.intellij.lexer.XmlLexer;
|
||||
import com.intellij.testFramework.LexerTestCase;
|
||||
import com.intellij.platform.syntax.lexer.Lexer;
|
||||
import com.intellij.platform.syntax.util.lexer.FilterLexer;
|
||||
import com.intellij.testFramework.ParsingTestCase;
|
||||
import com.intellij.testFramework.PlatformTestUtil;
|
||||
import com.intellij.testFramework.syntax.LexerTestCase;
|
||||
import com.intellij.tools.ide.metrics.benchmark.Benchmark;
|
||||
import com.intellij.xml.syntax.XmlSyntaxDefinition;
|
||||
import com.intellij.xml.syntax.lexer.XmlLexer;
|
||||
import org.jetbrains.annotations.NotNull;
|
||||
|
||||
import java.io.File;
|
||||
@@ -60,8 +60,10 @@ public class XmlLexerTest extends LexerTestCase {
|
||||
PlatformTestUtil.getCommunityPath().replace(File.separatorChar, '/') + "/xml/tests/testData/psi/xml",
|
||||
fileName);
|
||||
final XmlLexer lexer = new XmlLexer();
|
||||
final FilterLexer filterLexer = new FilterLexer(new XmlLexer(),
|
||||
new FilterLexer.SetFilter(new XMLParserDefinition().getWhitespaceTokens()));
|
||||
final FilterLexer filterLexer = new com.intellij.platform.syntax.util.lexer.FilterLexer(
|
||||
new XmlLexer(),
|
||||
new FilterLexer.SetFilter(XmlSyntaxDefinition.INSTANCE.getWHITESPACES())
|
||||
);
|
||||
|
||||
Benchmark.newBenchmark("XML Lexer Performance on " + fileName, () -> {
|
||||
for (int i = 0; i < 10; i++) {
|
||||
|
||||
+2
-2
@@ -2,7 +2,7 @@
|
||||
package com.intellij.psi.impl.cache.impl.idCache;
|
||||
|
||||
import com.intellij.lexer.Lexer;
|
||||
import com.intellij.lexer.XmlLexer;
|
||||
import com.intellij.lexer.XmlLexerKt;
|
||||
import com.intellij.psi.impl.cache.impl.OccurrenceConsumer;
|
||||
import com.intellij.psi.impl.cache.impl.id.LexerBasedIdIndexer;
|
||||
import org.jetbrains.annotations.NotNull;
|
||||
@@ -14,6 +14,6 @@ public class XmlIdIndexer extends LexerBasedIdIndexer {
|
||||
}
|
||||
|
||||
static XmlFilterLexer createIndexingLexer(OccurrenceConsumer consumer) {
|
||||
return new XmlFilterLexer(new XmlLexer(), consumer);
|
||||
return new XmlFilterLexer(XmlLexerKt.createXmlLexer(), consumer);
|
||||
}
|
||||
}
|
||||
|
||||
Vendored
+2
-2
@@ -2,7 +2,7 @@
|
||||
package com.intellij.psi.impl.cache.impl.idCache;
|
||||
|
||||
import com.intellij.lexer.Lexer;
|
||||
import com.intellij.lexer.XmlLexer;
|
||||
import com.intellij.lexer.XmlLexerKt;
|
||||
import com.intellij.psi.PsiFile;
|
||||
import com.intellij.psi.impl.search.IndexPatternBuilder;
|
||||
import com.intellij.psi.tree.IElementType;
|
||||
@@ -18,7 +18,7 @@ public class XmlIndexPatternBuilder implements IndexPatternBuilder {
|
||||
@Override
|
||||
public @Nullable Lexer getIndexingLexer(@NotNull PsiFile file) {
|
||||
if (file instanceof XmlFile) {
|
||||
return new XmlLexer();
|
||||
return XmlLexerKt.createXmlLexer();
|
||||
}
|
||||
return null;
|
||||
}
|
||||
|
||||
+2
-2
@@ -8,7 +8,7 @@ import com.intellij.codeInspection.ProblemsHolder;
|
||||
import com.intellij.codeInspection.XmlSuppressableInspectionTool;
|
||||
import com.intellij.ide.highlighter.XmlLikeFileType;
|
||||
import com.intellij.lexer.Lexer;
|
||||
import com.intellij.lexer.XmlLexer;
|
||||
import com.intellij.lexer.XmlLexerKt;
|
||||
import com.intellij.openapi.fileTypes.FileType;
|
||||
import com.intellij.openapi.util.TextRange;
|
||||
import com.intellij.psi.PsiElement;
|
||||
@@ -46,7 +46,7 @@ public class CheckValidXmlInScriptBodyInspectionBase extends XmlSuppressableInsp
|
||||
|
||||
if (fileType instanceof XmlLikeFileType) {
|
||||
synchronized(CheckValidXmlInScriptBodyInspectionBase.class) {
|
||||
if (myXmlLexer == null) myXmlLexer = new XmlLexer();
|
||||
if (myXmlLexer == null) myXmlLexer = XmlLexerKt.createXmlLexer();
|
||||
final XmlTagValue tagValue = tag.getValue();
|
||||
final String tagBodyText = tagValue.getText();
|
||||
|
||||
|
||||
@@ -21,6 +21,7 @@ jvm_library(
|
||||
"//platform/syntax/syntax-api:syntax",
|
||||
"//platform/syntax/syntax-psi:psi",
|
||||
"//xml/xml-syntax:syntax",
|
||||
"//platform/syntax/syntax-util:util",
|
||||
],
|
||||
runtime_deps = [":parser_resources"]
|
||||
)
|
||||
|
||||
@@ -17,5 +17,6 @@
|
||||
<orderEntry type="module" module-name="intellij.platform.syntax" />
|
||||
<orderEntry type="module" module-name="intellij.platform.syntax.psi" />
|
||||
<orderEntry type="module" module-name="intellij.xml.syntax" />
|
||||
<orderEntry type="module" module-name="intellij.platform.syntax.util" />
|
||||
</component>
|
||||
</module>
|
||||
@@ -31,11 +31,11 @@ public class XHtmlLexer extends HtmlLexer {
|
||||
}
|
||||
|
||||
public XHtmlLexer() {
|
||||
this(new XmlLexer(true));
|
||||
this(XmlLexerKt.createXmlLexer(true));
|
||||
}
|
||||
|
||||
public XHtmlLexer(boolean highlightMode) {
|
||||
this(new XmlLexer(true), highlightMode);
|
||||
this(XmlLexerKt.createXmlLexer(true), highlightMode);
|
||||
}
|
||||
|
||||
@Override
|
||||
|
||||
@@ -5,7 +5,7 @@ import com.intellij.psi.tree.IElementType
|
||||
import com.intellij.psi.xml.XmlTokenType
|
||||
|
||||
class XmlHighlightingLexer :
|
||||
DelegateLexer(XmlLexer()) {
|
||||
DelegateLexer(createXmlLexer()) {
|
||||
|
||||
override fun getTokenType(): IElementType? {
|
||||
var tokenType = delegate.tokenType
|
||||
|
||||
@@ -1,23 +1,32 @@
|
||||
// Copyright 2000-2023 JetBrains s.r.o. and contributors. Use of this source code is governed by the Apache 2.0 license.
|
||||
package com.intellij.lexer
|
||||
|
||||
import com.intellij.platform.syntax.psi.asTokenSet
|
||||
import com.intellij.platform.syntax.psi.lexer.LexerAdapter
|
||||
import com.intellij.psi.tree.TokenSet
|
||||
import com.intellij.psi.xml.XmlTokenType
|
||||
import com.intellij.psi.xml.xmlElementTypeConverter
|
||||
import com.intellij.xml.syntax.lexer.XmlLexer
|
||||
import org.jetbrains.annotations.ApiStatus
|
||||
|
||||
class XmlLexer(baseLexer: Lexer) :
|
||||
MergingLexerAdapter(baseLexer, TOKENS_TO_MERGE) {
|
||||
/**
|
||||
* Obsolete: Use [com.intellij.xml.syntax.lexer.XmlLexer] instead. This class is kept for binary compatibility and won't be developed further.
|
||||
*
|
||||
* If you still need an instance of [com.intellij.lexer.Lexer], use [createXmlLexer] function which returns a new XmlLexer instance wrapped into an adapter.
|
||||
*/
|
||||
@ApiStatus.Obsolete
|
||||
class XmlLexer(baseLexer: Lexer) : MergingLexerAdapter(baseLexer, TOKENS_TO_MERGE) {
|
||||
|
||||
@Suppress("unused") // used by external plugins
|
||||
@JvmOverloads
|
||||
constructor(conditionalCommentsSupport: Boolean = false) :
|
||||
this(_XmlLexer(__XmlLexer(null), conditionalCommentsSupport))
|
||||
|
||||
private companion object {
|
||||
private val TOKENS_TO_MERGE = TokenSet.create(
|
||||
XmlTokenType.XML_DATA_CHARACTERS,
|
||||
XmlTokenType.XML_TAG_CHARACTERS,
|
||||
XmlTokenType.XML_ATTRIBUTE_VALUE_TOKEN,
|
||||
XmlTokenType.XML_PI_TARGET,
|
||||
XmlTokenType.XML_COMMENT_CHARACTERS,
|
||||
)
|
||||
}
|
||||
constructor(conditionalCommentsSupport: Boolean = false) : this(_XmlLexer(__XmlLexer(null), conditionalCommentsSupport))
|
||||
}
|
||||
|
||||
@JvmOverloads
|
||||
fun createXmlLexer(conditionalCommentsSupport: Boolean = false): Lexer =
|
||||
LexerAdapter(
|
||||
XmlLexer(conditionalCommentsSupport),
|
||||
xmlElementTypeConverter
|
||||
)
|
||||
|
||||
private val TOKENS_TO_MERGE: TokenSet =
|
||||
com.intellij.xml.syntax.lexer.TOKENS_TO_MERGE.asTokenSet(xmlElementTypeConverter)
|
||||
|
||||
@@ -9,6 +9,8 @@ import com.intellij.lang.ParserDefinition.SpaceRequirements
|
||||
import com.intellij.lang.PsiParser
|
||||
import com.intellij.lexer.Lexer
|
||||
import com.intellij.lexer.XmlLexer
|
||||
import com.intellij.lexer.createXmlLexer
|
||||
import com.intellij.openapi.components.service
|
||||
import com.intellij.openapi.project.Project
|
||||
import com.intellij.psi.FileViewProvider
|
||||
import com.intellij.psi.PsiElement
|
||||
@@ -26,7 +28,7 @@ open class XMLParserDefinition :
|
||||
ParserDefinition {
|
||||
|
||||
override fun createLexer(project: Project?): Lexer =
|
||||
XmlLexer()
|
||||
createXmlLexer()
|
||||
|
||||
override fun getFileNodeType(): IFileElementType =
|
||||
XmlElementType.XML_FILE
|
||||
|
||||
@@ -15,6 +15,10 @@ jvm_library(
|
||||
deps = [
|
||||
"@lib//:kotlin-stdlib",
|
||||
"//platform/syntax/syntax-api:syntax",
|
||||
"//platform/util/multiplatform",
|
||||
"@lib//:jetbrains-annotations",
|
||||
"//platform/syntax/syntax-i18n:i18n",
|
||||
"//platform/syntax/syntax-util:util",
|
||||
],
|
||||
runtime_deps = [":syntax_resources"]
|
||||
)
|
||||
|
||||
File diff suppressed because it is too large
Load Diff
@@ -11,5 +11,9 @@
|
||||
<orderEntry type="sourceFolder" forTests="false" />
|
||||
<orderEntry type="library" name="kotlin-stdlib" level="project" />
|
||||
<orderEntry type="module" module-name="intellij.platform.syntax" />
|
||||
<orderEntry type="module" module-name="intellij.platform.util.multiplatform" />
|
||||
<orderEntry type="library" name="jetbrains-annotations" level="project" />
|
||||
<orderEntry type="module" module-name="intellij.platform.syntax.i18n" />
|
||||
<orderEntry type="module" module-name="intellij.platform.syntax.util" />
|
||||
</component>
|
||||
</module>
|
||||
@@ -0,0 +1,72 @@
|
||||
// Copyright 2000-2025 JetBrains s.r.o. and contributors. Use of this source code is governed by the Apache 2.0 license.
|
||||
package com.intellij.xml.syntax.lexer
|
||||
|
||||
import com.intellij.platform.syntax.SyntaxElementTypeSet
|
||||
import com.intellij.platform.syntax.lexer.Lexer
|
||||
import com.intellij.platform.syntax.syntaxElementTypeSetOf
|
||||
import com.intellij.platform.syntax.util.lexer.FlexAdapter
|
||||
import com.intellij.platform.syntax.util.lexer.MergingLexerAdapter
|
||||
import com.intellij.xml.syntax.XmlSyntaxTokenType
|
||||
import org.jetbrains.annotations.ApiStatus
|
||||
|
||||
class XmlLexer(baseLexer: Lexer) : MergingLexerAdapter(baseLexer, TOKENS_TO_MERGE) {
|
||||
|
||||
@JvmOverloads
|
||||
constructor(conditionalCommentsSupport: Boolean = false) :
|
||||
this(_XmlLexer(__XmlLexer(), conditionalCommentsSupport))
|
||||
|
||||
}
|
||||
|
||||
@ApiStatus.Internal
|
||||
val TOKENS_TO_MERGE: SyntaxElementTypeSet = syntaxElementTypeSetOf(
|
||||
XmlSyntaxTokenType.XML_DATA_CHARACTERS,
|
||||
XmlSyntaxTokenType.XML_TAG_CHARACTERS,
|
||||
XmlSyntaxTokenType.XML_ATTRIBUTE_VALUE_TOKEN,
|
||||
XmlSyntaxTokenType.XML_PI_TARGET,
|
||||
XmlSyntaxTokenType.XML_COMMENT_CHARACTERS,
|
||||
)
|
||||
|
||||
class _XmlLexer @JvmOverloads constructor(
|
||||
flexLexer: __XmlLexer,
|
||||
conditionalCommentsSupport: Boolean = false,
|
||||
) : FlexAdapter(flexLexer) {
|
||||
|
||||
private var myState: Int = __XmlLexer.YYINITIAL
|
||||
|
||||
init {
|
||||
flexLexer.setConditionalCommentsSupport(conditionalCommentsSupport)
|
||||
}
|
||||
|
||||
override fun getState(): Int = myState
|
||||
|
||||
private fun packState() {
|
||||
val flex = flex as __XmlLexer
|
||||
this.myState = ((flex.yyprevstate() and STATE_MASK) shl STATE_SHIFT) or (flex.yystate() and STATE_MASK)
|
||||
}
|
||||
|
||||
private fun handleState(initialState: Int) {
|
||||
val flex = flex as __XmlLexer
|
||||
flex.yybegin(initialState and STATE_MASK)
|
||||
flex.pushState((initialState shr STATE_SHIFT) and STATE_MASK)
|
||||
packState()
|
||||
}
|
||||
|
||||
override fun start(buffer: CharSequence, startOffset: Int, endOffset: Int, initialState: Int) {
|
||||
super.start(buffer, startOffset, endOffset, initialState)
|
||||
handleState(initialState)
|
||||
}
|
||||
|
||||
override fun advance() {
|
||||
super.advance()
|
||||
packState()
|
||||
}
|
||||
|
||||
companion object {
|
||||
private const val STATE_SHIFT = 5
|
||||
private val STATE_MASK = (1 shl STATE_SHIFT) - 1
|
||||
|
||||
init {
|
||||
assert((STATE_MASK shl 1) <= HtmlLexerConstants.BASE_STATE_MASK)
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,175 @@
|
||||
/* It's an automatically generated code. Do not modify it. */
|
||||
package com.intellij.xml.syntax.lexer
|
||||
|
||||
import com.intellij.platform.syntax.SyntaxElementType
|
||||
import com.intellij.platform.syntax.util.lexer.FlexLexer
|
||||
import com.intellij.xml.syntax.XmlSyntaxTokenType
|
||||
import com.intellij.xml.syntax.XmlSyntaxElementType
|
||||
%%
|
||||
|
||||
%{
|
||||
private var elTokenType = XmlSyntaxTokenType.XML_DATA_CHARACTERS
|
||||
private var elTokenType2 = XmlSyntaxTokenType.XML_ATTRIBUTE_VALUE_TOKEN
|
||||
private var javaEmbeddedTokenType = XmlSyntaxTokenType.XML_ATTRIBUTE_VALUE_TOKEN
|
||||
private var myConditionalCommentsSupport: Boolean = false
|
||||
|
||||
fun setConditionalCommentsSupport(b: Boolean) {
|
||||
myConditionalCommentsSupport = b
|
||||
}
|
||||
|
||||
public fun setElTypes(_elTokenType: SyntaxElementType, _elTokenType2: SyntaxElementType) {
|
||||
elTokenType = _elTokenType;
|
||||
elTokenType2 = _elTokenType2;
|
||||
}
|
||||
|
||||
public fun setJavaEmbeddedType(_tokenType: SyntaxElementType) {
|
||||
javaEmbeddedTokenType = _tokenType;
|
||||
}
|
||||
|
||||
private var myPrevState = YYINITIAL
|
||||
|
||||
fun yyprevstate() = myPrevState
|
||||
|
||||
private fun popState(): Int {
|
||||
val prev = myPrevState
|
||||
myPrevState = YYINITIAL
|
||||
return prev
|
||||
}
|
||||
|
||||
fun pushState(state: Int){
|
||||
myPrevState = state
|
||||
}
|
||||
%}
|
||||
|
||||
%unicode
|
||||
%class __XmlLexer
|
||||
%public
|
||||
%implements FlexLexer
|
||||
%function advance
|
||||
%type SyntaxElementType
|
||||
|
||||
%state TAG
|
||||
%state PROCESSING_INSTRUCTION
|
||||
%state PI_ANY
|
||||
%state END_TAG
|
||||
%xstate COMMENT
|
||||
%state ATTR_LIST
|
||||
%state ATTR
|
||||
%state ATTR_VALUE_START
|
||||
%state ATTR_VALUE_DQ
|
||||
%state ATTR_VALUE_SQ
|
||||
%state DTD_MARKUP
|
||||
%state DOCTYPE
|
||||
%xstate CDATA
|
||||
%state C_COMMENT_START
|
||||
/* this state should be last, number of states should be less than 16 */
|
||||
%state C_COMMENT_END
|
||||
|
||||
ALPHA=[:letter:]
|
||||
DIGIT=[0-9]
|
||||
WS=[\ \n\r\t\f\u2028\u2029\u0085]
|
||||
S={WS}+
|
||||
|
||||
EL_EMBEDMENT_START="${" | "#{"
|
||||
NAME=({ALPHA}|"_"|":")({ALPHA}|{DIGIT}|"_"|"."|"-")*(":"({ALPHA}|"_")?({ALPHA}|{DIGIT}|"_"|"."|"-")*)?
|
||||
|
||||
END_COMMENT="-->"
|
||||
CONDITIONAL_COMMENT_CONDITION=({ALPHA})({ALPHA}|{S}|{DIGIT}|"."|"("|")"|"|"|"!"|"&")*
|
||||
|
||||
%%
|
||||
"<![CDATA[" {yybegin(CDATA); return XmlSyntaxTokenType.XML_CDATA_START; }
|
||||
<CDATA>{
|
||||
"]]>" {yybegin(YYINITIAL); return XmlSyntaxTokenType.XML_CDATA_END; }
|
||||
[^] {return XmlSyntaxTokenType.XML_DATA_CHARACTERS; }
|
||||
}
|
||||
|
||||
"<!--" { yybegin(COMMENT); return XmlSyntaxTokenType.XML_COMMENT_START; }
|
||||
<COMMENT> "[" { if (myConditionalCommentsSupport) {
|
||||
yybegin(C_COMMENT_START);
|
||||
return XmlSyntaxTokenType.XML_CONDITIONAL_COMMENT_START;
|
||||
} else return XmlSyntaxTokenType.XML_COMMENT_CHARACTERS; }
|
||||
<COMMENT> "<![" { if (myConditionalCommentsSupport) {
|
||||
yybegin(C_COMMENT_END);
|
||||
return XmlSyntaxTokenType.XML_CONDITIONAL_COMMENT_END_START;
|
||||
} else return XmlSyntaxTokenType.XML_COMMENT_CHARACTERS; }
|
||||
<COMMENT> {END_COMMENT} { yybegin(YYINITIAL); return XmlSyntaxTokenType.XML_COMMENT_END; }
|
||||
<COMMENT> [^\-]|(-[^\-]) { return XmlSyntaxTokenType.XML_COMMENT_CHARACTERS; }
|
||||
<COMMENT> [^] { return XmlSyntaxTokenType.XML_BAD_CHARACTER; }
|
||||
|
||||
<C_COMMENT_START,C_COMMENT_END> {CONDITIONAL_COMMENT_CONDITION} { return XmlSyntaxTokenType.XML_COMMENT_CHARACTERS; }
|
||||
<C_COMMENT_START> [^] { yybegin(COMMENT); return XmlSyntaxTokenType.XML_COMMENT_CHARACTERS; }
|
||||
<C_COMMENT_START> "]>" { yybegin(COMMENT); return XmlSyntaxTokenType.XML_CONDITIONAL_COMMENT_START_END; }
|
||||
<C_COMMENT_START,C_COMMENT_END> {END_COMMENT} { yybegin(YYINITIAL); return XmlSyntaxTokenType.XML_COMMENT_END; }
|
||||
<C_COMMENT_END> "]" { yybegin(COMMENT); return XmlSyntaxTokenType.XML_CONDITIONAL_COMMENT_END; }
|
||||
<C_COMMENT_END> [^] { yybegin(COMMENT); return XmlSyntaxTokenType.XML_COMMENT_CHARACTERS; }
|
||||
|
||||
"<" |
|
||||
">" |
|
||||
"'" |
|
||||
""" |
|
||||
" " |
|
||||
"&" |
|
||||
"&#"{DIGIT}+";" |
|
||||
"&#x"({DIGIT}|[a-fA-F])+";" { return XmlSyntaxTokenType.XML_CHAR_ENTITY_REF; }
|
||||
"&"{NAME}";" { return XmlSyntaxTokenType.XML_ENTITY_REF_TOKEN; }
|
||||
|
||||
<YYINITIAL> "<!DOCTYPE" { yybegin(DOCTYPE); return XmlSyntaxTokenType.XML_DOCTYPE_START; }
|
||||
<DOCTYPE> "SYSTEM" { return XmlSyntaxTokenType.XML_DOCTYPE_SYSTEM; }
|
||||
<DOCTYPE> "PUBLIC" { return XmlSyntaxTokenType.XML_DOCTYPE_PUBLIC; }
|
||||
<DOCTYPE> {NAME} { return XmlSyntaxTokenType.XML_NAME; }
|
||||
<DOCTYPE> "\"" [^\"]* "\""? { return XmlSyntaxTokenType.XML_ATTRIBUTE_VALUE_TOKEN;}
|
||||
<DOCTYPE> "'" [^']* "'"? { return XmlSyntaxTokenType.XML_ATTRIBUTE_VALUE_TOKEN;}
|
||||
<DOCTYPE> "[" (([^\]\"]*)|(\"[^\"]*\"))* "]"? { return XmlSyntaxElementType.XML_MARKUP_DECL;}
|
||||
<DOCTYPE> ">" { yybegin(YYINITIAL); return XmlSyntaxTokenType.XML_DOCTYPE_END; }
|
||||
|
||||
<YYINITIAL> "<?" { yybegin(PROCESSING_INSTRUCTION); return XmlSyntaxTokenType.XML_PI_START; }
|
||||
<PROCESSING_INSTRUCTION> "xml" { yybegin(ATTR_LIST); pushState(PROCESSING_INSTRUCTION); return XmlSyntaxTokenType.XML_NAME; }
|
||||
<PROCESSING_INSTRUCTION> {NAME} { yybegin(PI_ANY); return XmlSyntaxTokenType.XML_NAME; }
|
||||
<PI_ANY, PROCESSING_INSTRUCTION> "?>" { yybegin(YYINITIAL); return XmlSyntaxTokenType.XML_PI_END; }
|
||||
<PI_ANY> {S} { return XmlSyntaxTokenType.XML_WHITE_SPACE; }
|
||||
<PI_ANY> [^] { return XmlSyntaxTokenType.XML_TAG_CHARACTERS; }
|
||||
|
||||
<YYINITIAL> {EL_EMBEDMENT_START} [^<\}]* "}"? {
|
||||
return elTokenType;
|
||||
}
|
||||
|
||||
<YYINITIAL> "<" { yybegin(TAG); return XmlSyntaxTokenType.XML_START_TAG_START; }
|
||||
<TAG> {NAME} { yybegin(ATTR_LIST); pushState(TAG); return XmlSyntaxTokenType.XML_NAME; }
|
||||
<TAG> "/>" { yybegin(YYINITIAL); return XmlSyntaxTokenType.XML_EMPTY_ELEMENT_END; }
|
||||
<TAG> ">" { yybegin(YYINITIAL); return XmlSyntaxTokenType.XML_TAG_END; }
|
||||
|
||||
<YYINITIAL> "</" { yybegin(END_TAG); return XmlSyntaxTokenType.XML_END_TAG_START; }
|
||||
<END_TAG> {NAME} { return XmlSyntaxTokenType.XML_NAME; }
|
||||
<END_TAG> ">" { yybegin(YYINITIAL); return XmlSyntaxTokenType.XML_TAG_END; }
|
||||
|
||||
<ATTR_LIST> {NAME} {yybegin(ATTR); return XmlSyntaxTokenType.XML_NAME;}
|
||||
<ATTR> "=" { return XmlSyntaxTokenType.XML_EQ;}
|
||||
<ATTR> "'" { yybegin(ATTR_VALUE_SQ); return XmlSyntaxTokenType.XML_ATTRIBUTE_VALUE_START_DELIMITER;}
|
||||
<ATTR> "\"" { yybegin(ATTR_VALUE_DQ); return XmlSyntaxTokenType.XML_ATTRIBUTE_VALUE_START_DELIMITER;}
|
||||
<ATTR> [^\ \n\r\t\f] {yybegin(ATTR_LIST); yypushback(yylength()); }
|
||||
|
||||
<ATTR_VALUE_DQ>{
|
||||
"\"" { yybegin(ATTR_LIST); return XmlSyntaxTokenType.XML_ATTRIBUTE_VALUE_END_DELIMITER;}
|
||||
"&" { return XmlSyntaxTokenType.XML_BAD_CHARACTER; }
|
||||
{EL_EMBEDMENT_START} [^\}\"]* "}"? { return elTokenType2; }
|
||||
"%=" [^%\"]* "%" { return javaEmbeddedTokenType; }
|
||||
[^] { return XmlSyntaxTokenType.XML_ATTRIBUTE_VALUE_TOKEN;}
|
||||
}
|
||||
|
||||
<ATTR_VALUE_SQ>{
|
||||
"&" { return XmlSyntaxTokenType.XML_BAD_CHARACTER; }
|
||||
"'" { yybegin(ATTR_LIST); return XmlSyntaxTokenType.XML_ATTRIBUTE_VALUE_END_DELIMITER;}
|
||||
{EL_EMBEDMENT_START} [^\}\']* "}"? { return elTokenType2; }
|
||||
"%=" [^%\']* "%" { return javaEmbeddedTokenType; }
|
||||
[^] { return XmlSyntaxTokenType.XML_ATTRIBUTE_VALUE_TOKEN;}
|
||||
}
|
||||
|
||||
<YYINITIAL> {S} { return XmlSyntaxTokenType.XML_REAL_WHITE_SPACE; }
|
||||
<ATTR_LIST,ATTR,TAG,END_TAG,DOCTYPE> {S} { return XmlSyntaxTokenType.XML_WHITE_SPACE; }
|
||||
<YYINITIAL> ([^<&\$# \n\r\t\f]|(\\\$)|(\\#))* { return XmlSyntaxTokenType.XML_DATA_CHARACTERS; }
|
||||
<YYINITIAL> [^<&\ \n\r\t\f]|(\\\$)|(\\#) { return XmlSyntaxTokenType.XML_DATA_CHARACTERS; }
|
||||
|
||||
[^] { if(yystate() == YYINITIAL){
|
||||
return XmlSyntaxTokenType.XML_BAD_CHARACTER;
|
||||
}
|
||||
else yybegin(popState()); yypushback(yylength());}
|
||||
Reference in New Issue
Block a user