diff --git a/.github/workflows/conventionalpr.yml b/.github/workflows/conventionalpr.yml
index 2ef51f7..758bee7 100644
--- a/.github/workflows/conventionalpr.yml
+++ b/.github/workflows/conventionalpr.yml
@@ -2,7 +2,7 @@ name: Conventional Commit Parser
on:
pull_request:
branches: [main, master, develop]
- types: [opened, reopened, edited, review_requested, synchronize]
+ types: [opened, reopened, edited, synchronize]
jobs:
conventional_commit:
@@ -11,6 +11,6 @@ jobs:
permissions:
pull-requests: read
steps:
- - uses: webiny/action-conventional-commits@v1.3.0
- with:
+ - uses: amannn/action-semantic-pull-request@v6.1.1
+ env:
GITHUB_TOKEN: ${{ secrets.GITHUB_TOKEN }}
diff --git a/.github/workflows/java.yml b/.github/workflows/java.yml
index ec2a7d5..14b70ad 100644
--- a/.github/workflows/java.yml
+++ b/.github/workflows/java.yml
@@ -1,10 +1,9 @@
name: Build and Deploy Java Parser
on:
- push:
- branches: [develop, main]
- tags: ['v*']
-
+ release:
+ types: [published]
+
jobs:
build:
runs-on: ubuntu-latest
@@ -22,7 +21,6 @@ jobs:
server-username: MAVEN_USERNAME
server-password: MAVEN_PASSWORD
gpg-private-key: ${{ secrets.GPG_PRIVATE_KEY }}
- gpg-passphrase: MAVEN_GPG_PASSPHRASE
- name: Configure GPG for batch mode
run: |
@@ -41,27 +39,12 @@ jobs:
- name: Install ANTLR4
run: make dev
- - name: Set SNAPSHOT version (develop branch)
- if: github.ref == 'refs/heads/develop'
- run: |
- cd java
- BASE_VERSION=$(mvn help:evaluate -Dexpression=project.version -q -DforceStdout | sed 's/-SNAPSHOT//')
- mvn versions:set -DnewVersion="${BASE_VERSION}-SNAPSHOT" -DgenerateBackupPoms=false
-
- - name: Set release version (main branch)
- if: github.ref == 'refs/heads/main'
+ - name: Set release version
run: |
cd java
BASE_VERSION=$(mvn help:evaluate -Dexpression=project.version -q -DforceStdout | sed 's/-SNAPSHOT//')
mvn versions:set -DnewVersion="${BASE_VERSION}" -DgenerateBackupPoms=false
- - name: Set release version (tag)
- if: startsWith(github.ref, 'refs/tags/v')
- run: |
- cd java
- TAG_VERSION=${GITHUB_REF_NAME#v}
- mvn versions:set -DnewVersion="${TAG_VERSION}" -DgenerateBackupPoms=false
-
- name: Generate and Compile Java Code
run: make java_parser
@@ -72,17 +55,8 @@ jobs:
key: ${{ runner.os }}-m2-${{ hashFiles('**/pom.xml') }}
restore-keys: ${{ runner.os }}-m2
- - name: Deploy SNAPSHOT to Sonatype Snapshots
- if: github.ref == 'refs/heads/develop'
- run: cd java && mvn clean deploy -DskipCentralPublishing=true
- env:
- MAVEN_USERNAME: ${{ secrets.SONATYPE_USERNAME }}
- MAVEN_PASSWORD: ${{ secrets.SONATYPE_PASSWORD }}
- MAVEN_GPG_PASSPHRASE: ${{ secrets.GPG_PASSPHRASE }}
-
- name: Deploy Release to Central Portal
- if: github.ref == 'refs/heads/main' || startsWith(github.ref, 'refs/tags/v')
- run: cd java && mvn clean deploy
+ run: cd java && mvn --settings ../.github/settings.xml clean deploy
env:
MAVEN_USERNAME: ${{ secrets.SONATYPE_USERNAME }}
MAVEN_PASSWORD: ${{ secrets.SONATYPE_PASSWORD }}
diff --git a/README.md b/README.md
index f73ef93..7e09897 100644
--- a/README.md
+++ b/README.md
@@ -34,7 +34,7 @@ To use UVL in your projects, you can either:
io.github.universal-variability-language
uvl-parser
- 0.3
+ 0.5.1
```
### Python Parser
diff --git a/java/pom.xml b/java/pom.xml
index 88ec875..a5df1eb 100644
--- a/java/pom.xml
+++ b/java/pom.xml
@@ -11,7 +11,7 @@
io.github.universal-variability-language
uvl-parser
- 0.5.0
+ 0.5.1
@@ -172,7 +172,7 @@
org.sonatype.central
central-publishing-maven-plugin
- 0.5.0
+ 0.11.0
true
central
diff --git a/js/src/UVLJavaScriptCustomLexer.js b/js/src/UVLJavaScriptCustomLexer.js
index 3f09df5..0f6f0a8 100644
--- a/js/src/UVLJavaScriptCustomLexer.js
+++ b/js/src/UVLJavaScriptCustomLexer.js
@@ -10,6 +10,7 @@ export default class UVLJavaScriptCustomLexer extends UVLJavaScriptLexer {
this.indents = [];
this.opened = 0;
this.lastToken = null;
+ this.startChecked = false;
}
emitToken(t) {
@@ -18,6 +19,11 @@ export default class UVLJavaScriptCustomLexer extends UVLJavaScriptLexer {
}
nextToken() {
+ if (!this.startChecked) {
+ this.startChecked = true;
+ this.handleLeadingIndentation();
+ }
+
if (this._input.LA(1) === antlr4.Token.EOF && this.indents.length !== 0) {
while (this.tokens.length > 0 && this.tokens[this.tokens.length - 1].type === antlr4.Token.EOF) {
this.tokens.pop();
@@ -70,8 +76,34 @@ export default class UVLJavaScriptCustomLexer extends UVLJavaScriptLexer {
this.skip();
}
- atStartOfInput() {
- return this._interp.column === 0 && this._interp.line === 1;
+ // Indentation before the first token is handled here, once, rather than with an
+ // {atStartOfInput()}? predicate in the NEWLINE rule: a predicate reachable from the lexer
+ // start state stops ANTLR from caching its DFA start state, which makes every token
+ // recompute the whole ATN closure. The spaces themselves are then skipped by SKIP_.
+ // Queues the same tokens handleNewline() emits for a leading line: an indented first
+ // line yields NEWLINE + INDENT, which the parser rejects.
+ handleLeadingIndentation() {
+ let length = 0;
+ while (this._input.LA(length + 1) === 32 || this._input.LA(length + 1) === 9) {
+ length += 1;
+ }
+ if (length === 0) {
+ return;
+ }
+ const next = String.fromCharCode(this._input.LA(length + 1));
+ if (next === '\r' || next === '\n' || next === '\f' || next === '#') {
+ return;
+ }
+ const spaces = this._input.getText(0, length - 1);
+ this.indents.push(UVLJavaScriptCustomLexer.getIndentationCount(spaces));
+ this.emitToken(this.leadingToken(UVLJavaScriptLexer.NEWLINE, length - 1, length - 1, length));
+ this.emitToken(this.leadingToken(UVLJavaScriptParser.INDENT, 0, length - 1, length));
+ }
+
+ leadingToken(type, start, stop, column) {
+ const token = new antlr4.CommonToken(this._tokenFactorySourcePair, type, antlr4.Token.DEFAULT_CHANNEL, start, stop);
+ token.column = column;
+ return token;
}
handleNewline() {
diff --git a/python/test_grammar.py b/python/test_grammar.py
index 1c5658e..a1dde91 100644
--- a/python/test_grammar.py
+++ b/python/test_grammar.py
@@ -46,3 +46,15 @@ def test_valid_model(uvl_file):
def test_faulty_model(uvl_file):
with pytest.raises(Exception):
get_tree(uvl_file)
+
+
+def test_lexer_caches_its_dfa_start_state():
+ # A semantic predicate reachable from the lexer start state stops ANTLR from caching the
+ # DFA start state (dfa.s0), so every token recomputes the full ATN closure: parsing gets
+ # several times slower. Keep lexer rules predicate-free at the start state.
+ from antlr4 import CommonTokenStream, FileStream
+ from uvl.UVLCustomLexer import UVLCustomLexer
+
+ lexer = UVLCustomLexer(FileStream(os.path.join(TEST_MODELS_DIR, "complex", "bike.uvl")))
+ CommonTokenStream(lexer).fill()
+ assert lexer._interp.decisionToDFA[lexer._mode].s0 is not None
diff --git a/python/uvl/UVLCustomLexer.py b/python/uvl/UVLCustomLexer.py
index 27a0d11..8e82869 100644
--- a/python/uvl/UVLCustomLexer.py
+++ b/python/uvl/UVLCustomLexer.py
@@ -13,12 +13,17 @@ def __init__(self, input_stream):
self.indents = []
self.opened = 0
self.lastToken = None
+ self.startChecked = False
def emitToken(self, t):
super().emitToken(t)
self.tokens.append(t)
def nextToken(self):
+ if not self.startChecked:
+ self.startChecked = True
+ self.handleLeadingIndentation()
+
# Check if the end-of-file is ahead and there are still some DEDENTS expected.
if self._input.LA(1) == Token.EOF and len(self.indents) != 0:
# Remove any trailing EOF tokens from our buffer.
@@ -67,9 +72,32 @@ def getIndentationCount(spaces):
def skipToken(self):
self.skip()
- def atStartOfInput(self):
- return self._interp.column == 0 and self._interp.line == 1
-
+ def handleLeadingIndentation(self):
+ # Indentation before the first token is handled here, once, rather than with a
+ # {atStartOfInput()}? predicate in the NEWLINE rule: a predicate reachable from the lexer
+ # start state stops ANTLR from caching its DFA start state, which makes every token
+ # recompute the whole ATN closure. The spaces themselves are then skipped by SKIP_.
+ # Queues the same tokens handleNewline() emits for a leading line: an indented first
+ # line yields NEWLINE + INDENT, which the parser rejects.
+ length = 0
+ while self._input.LA(length + 1) in (ord(' '), ord('\t')):
+ length += 1
+ if length == 0:
+ return
+ next_code = self._input.LA(length + 1)
+ next_char = chr(next_code) if next_code != -1 else ''
+ if next_char in '\r\n\f' or next_char == '/' or next_char == '#':
+ return
+ spaces = self._input.getText(0, length - 1)
+ self.indents.append(self.getIndentationCount(spaces))
+ self.emitToken(self.leading_token(self.NEWLINE, length - 1, length - 1, length))
+ self.emitToken(self.leading_token(UVLPythonParser.INDENT, 0, length - 1, length))
+
+ def leading_token(self, type, start, stop, column):
+ token = CommonToken(self._tokenFactorySourcePair, type, Token.DEFAULT_CHANNEL, start, stop)
+ token.column = column
+ return token
+
def handleNewline(self):
new_line = re.sub(r"[^\r\n\f]+", "", self._interp.getText(self._input)) #.replaceAll("[^\r\n\f]+", "")
spaces = re.sub(r"[\r\n\f]+", "", self._interp.getText(self._input)) #.replaceAll("[\r\n\f]+", "")
diff --git a/test_models/faulty/indented_first_line.uvl b/test_models/faulty/indented_first_line.uvl
new file mode 100644
index 0000000..6c7f2de
--- /dev/null
+++ b/test_models/faulty/indented_first_line.uvl
@@ -0,0 +1,4 @@
+ features
+ Root
+ optional
+ A
diff --git a/uvl/Java/UVLJavaLexer.g4 b/uvl/Java/UVLJavaLexer.g4
index 8965902..50163f4 100644
--- a/uvl/Java/UVLJavaLexer.g4
+++ b/uvl/Java/UVLJavaLexer.g4
@@ -15,6 +15,8 @@ package uvl;
private int opened = 0;
// The most recently produced token.
private Token lastToken = null;
+ // Whether the indentation before the first token has been handled.
+ private boolean startChecked = false;
@Override
public void emit(Token t) {
@@ -24,6 +26,10 @@ package uvl;
@Override
public Token nextToken() {
+ if (!startChecked) {
+ startChecked = true;
+ handleLeadingIndentation();
+ }
// Check if the end-of-file is ahead and there are still some DEDENTS expected.
if (_input.LA(1) == EOF && !this.indents.isEmpty()) {
// Remove any trailing EOF tokens from our buffer.
@@ -52,6 +58,37 @@ package uvl;
return tokens.isEmpty() ? next : tokens.poll();
}
+ // Indentation before the first token is handled here, once, rather than with an
+ // {atStartOfInput()}? predicate in the NEWLINE rule: a predicate reachable from the lexer
+ // start state stops ANTLR from caching its DFA start state, which makes every token
+ // recompute the whole ATN closure. The spaces themselves are then skipped by SKIP_.
+ // Queues the same tokens the NEWLINE rule emits for a leading line: an indented first
+ // line yields NEWLINE + INDENT, which the parser rejects.
+ private void handleLeadingIndentation() {
+ int length = 0;
+ while (_input.LA(length + 1) == ' ' || _input.LA(length + 1) == '\t') {
+ length++;
+ }
+ if (length == 0) {
+ return;
+ }
+ int next = _input.LA(length + 1);
+ int nextNext = _input.LA(length + 2);
+ if (next == '\r' || next == '\n' || (next == '/' && nextNext == '/')) {
+ return;
+ }
+ String spaces = _input.getText(org.antlr.v4.runtime.misc.Interval.of(0, length - 1));
+ indents.push(getIndentationCount(spaces));
+ emit(leadingToken(UVLJavaParser.NEWLINE, length - 1, length - 1, length));
+ emit(leadingToken(UVLJavaParser.INDENT, 0, length - 1, length));
+ }
+
+ private CommonToken leadingToken(int type, int start, int stop, int column) {
+ CommonToken token = new CommonToken(this._tokenFactorySourcePair, type, DEFAULT_TOKEN_CHANNEL, start, stop);
+ token.setCharPositionInLine(column);
+ return token;
+ }
+
private Token createDedent() {
CommonToken dedent = commonToken(UVLJavaParser.DEDENT, "");
dedent.setLine(this.lastToken.getLine());
@@ -86,10 +123,6 @@ package uvl;
}
return count;
}
-
- boolean atStartOfInput() {
- return super.getCharPositionInLine() == 0 && super.getLine() == 1;
- }
}
// Indentation-sensitive tokens
@@ -102,11 +135,11 @@ CLOSE_BRACE: '}' {this.opened -= 1;};
OPEN_COMMENT: '/*' {this.opened += 1;};
CLOSE_COMMENT: '*/' {this.opened -= 1;};
-// Indentation-sensitive NEWLINE rule
-NEWLINE: (
- {atStartOfInput()}? SPACES
- | ( '\r'? '\n' | '\r') SPACES?
- ) {
+// Indentation-sensitive NEWLINE rule.
+// No semantic predicate here: a predicate reachable from the lexer start state stops ANTLR
+// from caching its DFA start state, so every token would recompute the full ATN closure.
+// Indentation before the first token is handled once, in handleLeadingIndentation().
+NEWLINE: ( '\r'? '\n' | '\r') SPACES? {
String newLine = getText().replaceAll("[^\r\n]+", "");
String spaces = getText().replaceAll("[\r\n]+", "");
int next = _input.LA(1);
diff --git a/uvl/JavaScript/UVLJavaScriptLexer.g4 b/uvl/JavaScript/UVLJavaScriptLexer.g4
index a3c5d1d..12b6599 100644
--- a/uvl/JavaScript/UVLJavaScriptLexer.g4
+++ b/uvl/JavaScript/UVLJavaScriptLexer.g4
@@ -22,7 +22,7 @@ CLOSE_BRACE: '}' {this.opened -= 1;};
OPEN_COMMENT: '/*' {this.opened += 1;};
CLOSE_COMMENT: '*/' {this.opened -= 1;};
-NEWLINE: (
- {this.atStartOfInput()}? SPACES
- | ( '\r'? '\n' | '\r') SPACES?
- ) {this.handleNewline();};
\ No newline at end of file
+// No semantic predicate here: a predicate reachable from the lexer start state stops ANTLR
+// from caching its DFA start state, so every token would recompute the full ATN closure.
+// Indentation before the first token is handled once, in UVLJavaScriptCustomLexer.handleLeadingIndentation().
+NEWLINE: ( '\r'? '\n' | '\r') SPACES? {this.handleNewline();};
\ No newline at end of file
diff --git a/uvl/Python/UVLPythonLexer.g4 b/uvl/Python/UVLPythonLexer.g4
index 6ac23b9..e8cd80b 100644
--- a/uvl/Python/UVLPythonLexer.g4
+++ b/uvl/Python/UVLPythonLexer.g4
@@ -10,8 +10,8 @@ CLOSE_BRACE: '}' {self.opened -= 1;};
OPEN_COMMENT: '/*' {self.opened += 1;};
CLOSE_COMMENT: '*/' {self.opened -= 1;};
-//This is here because the way python manage tabs and new lines
-NEWLINE: (
- {self.atStartOfInput()}? SPACES
- | ( '\r'? '\n' | '\r') SPACES?
- ) {self.handleNewline();};
\ No newline at end of file
+//This is here because the way python manage tabs and new lines.
+// No semantic predicate here: a predicate reachable from the lexer start state stops ANTLR
+// from caching its DFA start state, so every token would recompute the full ATN closure.
+// Indentation before the first token is handled once, in UVLCustomLexer.handleLeadingIndentation().
+NEWLINE: ( '\r'? '\n' | '\r') SPACES? {self.handleNewline();};
\ No newline at end of file