From 49cbf2158e323135520bbbf9958cef744b9b21e1 Mon Sep 17 00:00:00 2001 From: Chico Sundermann Date: Thu, 2 Jul 2026 12:44:50 +0200 Subject: [PATCH 1/7] chore: Update Java maven release and conventional commits version (#74) * Updates Java maven deploy to react to releases and only to releases Replace webiny/action-conventional-commits (checks all PR commit messages) with amannn/action-semantic-pull-request (checks PR title), which validates the Conventional Commits format on the squash-merge entry point instead of each individual commit. * chore: follow yaml styling conventions Co-authored-by: Copilot Autofix powered by AI <175728472+Copilot@users.noreply.github.com> * Chore: updates version to 0.5.1 --------- Co-authored-by: copilot-swe-agent[bot] <198982749+Copilot@users.noreply.github.com> Co-authored-by: Copilot Autofix powered by AI <175728472+Copilot@users.noreply.github.com> --- .github/workflows/conventionalpr.yml | 6 ++--- .github/workflows/java.yml | 33 ++++------------------------ README.md | 2 +- java/pom.xml | 2 +- 4 files changed, 9 insertions(+), 34 deletions(-) diff --git a/.github/workflows/conventionalpr.yml b/.github/workflows/conventionalpr.yml index 2ef51f7..758bee7 100644 --- a/.github/workflows/conventionalpr.yml +++ b/.github/workflows/conventionalpr.yml @@ -2,7 +2,7 @@ name: Conventional Commit Parser on: pull_request: branches: [main, master, develop] - types: [opened, reopened, edited, review_requested, synchronize] + types: [opened, reopened, edited, synchronize] jobs: conventional_commit: @@ -11,6 +11,6 @@ jobs: permissions: pull-requests: read steps: - - uses: webiny/action-conventional-commits@v1.3.0 - with: + - uses: amannn/action-semantic-pull-request@v6.1.1 + env: GITHUB_TOKEN: ${{ secrets.GITHUB_TOKEN }} diff --git a/.github/workflows/java.yml b/.github/workflows/java.yml index ec2a7d5..b3e2506 100644 --- a/.github/workflows/java.yml +++ b/.github/workflows/java.yml @@ -1,10 +1,9 @@ name: Build and Deploy Java Parser on: - push: - branches: [develop, main] - tags: ['v*'] - + release: + types: [published] + jobs: build: runs-on: ubuntu-latest @@ -41,27 +40,12 @@ jobs: - name: Install ANTLR4 run: make dev - - name: Set SNAPSHOT version (develop branch) - if: github.ref == 'refs/heads/develop' - run: | - cd java - BASE_VERSION=$(mvn help:evaluate -Dexpression=project.version -q -DforceStdout | sed 's/-SNAPSHOT//') - mvn versions:set -DnewVersion="${BASE_VERSION}-SNAPSHOT" -DgenerateBackupPoms=false - - - name: Set release version (main branch) - if: github.ref == 'refs/heads/main' + - name: Set release version run: | cd java BASE_VERSION=$(mvn help:evaluate -Dexpression=project.version -q -DforceStdout | sed 's/-SNAPSHOT//') mvn versions:set -DnewVersion="${BASE_VERSION}" -DgenerateBackupPoms=false - - name: Set release version (tag) - if: startsWith(github.ref, 'refs/tags/v') - run: | - cd java - TAG_VERSION=${GITHUB_REF_NAME#v} - mvn versions:set -DnewVersion="${TAG_VERSION}" -DgenerateBackupPoms=false - - name: Generate and Compile Java Code run: make java_parser @@ -72,16 +56,7 @@ jobs: key: ${{ runner.os }}-m2-${{ hashFiles('**/pom.xml') }} restore-keys: ${{ runner.os }}-m2 - - name: Deploy SNAPSHOT to Sonatype Snapshots - if: github.ref == 'refs/heads/develop' - run: cd java && mvn clean deploy -DskipCentralPublishing=true - env: - MAVEN_USERNAME: ${{ secrets.SONATYPE_USERNAME }} - MAVEN_PASSWORD: ${{ secrets.SONATYPE_PASSWORD }} - MAVEN_GPG_PASSPHRASE: ${{ secrets.GPG_PASSPHRASE }} - - name: Deploy Release to Central Portal - if: github.ref == 'refs/heads/main' || startsWith(github.ref, 'refs/tags/v') run: cd java && mvn clean deploy env: MAVEN_USERNAME: ${{ secrets.SONATYPE_USERNAME }} diff --git a/README.md b/README.md index f73ef93..7e09897 100644 --- a/README.md +++ b/README.md @@ -34,7 +34,7 @@ To use UVL in your projects, you can either: io.github.universal-variability-language uvl-parser - 0.3 + 0.5.1 ``` ### Python Parser diff --git a/java/pom.xml b/java/pom.xml index 88ec875..984d1b3 100644 --- a/java/pom.xml +++ b/java/pom.xml @@ -11,7 +11,7 @@ io.github.universal-variability-language uvl-parser - 0.5.0 + 0.5.1 From ec3e630dd19e49936ac065585ec5f1309008390d Mon Sep 17 00:00:00 2001 From: Chico Sundermann Date: Thu, 2 Jul 2026 12:58:50 +0200 Subject: [PATCH 2/7] Fix: removes redundant entry from yaml that may cause issues --- .github/workflows/java.yml | 1 - 1 file changed, 1 deletion(-) diff --git a/.github/workflows/java.yml b/.github/workflows/java.yml index b3e2506..e9fb050 100644 --- a/.github/workflows/java.yml +++ b/.github/workflows/java.yml @@ -21,7 +21,6 @@ jobs: server-username: MAVEN_USERNAME server-password: MAVEN_PASSWORD gpg-private-key: ${{ secrets.GPG_PRIVATE_KEY }} - gpg-passphrase: MAVEN_GPG_PASSPHRASE - name: Configure GPG for batch mode run: | From 118c307771dd0b6c11dc5da89a24441558379c3f Mon Sep 17 00:00:00 2001 From: Chico Sundermann Date: Thu, 2 Jul 2026 13:09:13 +0200 Subject: [PATCH 3/7] Fix: further cleanup of Java workflow --- .github/workflows/java.yml | 2 +- java/pom.xml | 1 - 2 files changed, 1 insertion(+), 2 deletions(-) diff --git a/.github/workflows/java.yml b/.github/workflows/java.yml index e9fb050..028b0dd 100644 --- a/.github/workflows/java.yml +++ b/.github/workflows/java.yml @@ -60,4 +60,4 @@ jobs: env: MAVEN_USERNAME: ${{ secrets.SONATYPE_USERNAME }} MAVEN_PASSWORD: ${{ secrets.SONATYPE_PASSWORD }} - MAVEN_GPG_PASSPHRASE: ${{ secrets.GPG_PASSPHRASE }} + GPG_PASSPHRASE: ${{ secrets.GPG_PASSPHRASE }} diff --git a/java/pom.xml b/java/pom.xml index 984d1b3..6605fbc 100644 --- a/java/pom.xml +++ b/java/pom.xml @@ -156,7 +156,6 @@ --pinentry-mode loopback - ${env.MAVEN_GPG_PASSPHRASE} From 263827a5086e6fe1d87753f34ddfab104bd541c0 Mon Sep 17 00:00:00 2001 From: Chico Sundermann Date: Thu, 2 Jul 2026 13:30:20 +0200 Subject: [PATCH 4/7] Chore: version bump maven publishing plugin --- java/pom.xml | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/java/pom.xml b/java/pom.xml index 6605fbc..ef14571 100644 --- a/java/pom.xml +++ b/java/pom.xml @@ -171,7 +171,7 @@ org.sonatype.central central-publishing-maven-plugin - 0.5.0 + 0.11.0 true central From 888bb8f9fb88cf91f1bf8d6821f814e152facbf1 Mon Sep 17 00:00:00 2001 From: Chico Sundermann Date: Thu, 2 Jul 2026 13:38:11 +0200 Subject: [PATCH 5/7] Fix: Adds reference to settings.xml --- .github/workflows/java.yml | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/.github/workflows/java.yml b/.github/workflows/java.yml index 028b0dd..76a9a5d 100644 --- a/.github/workflows/java.yml +++ b/.github/workflows/java.yml @@ -56,7 +56,7 @@ jobs: restore-keys: ${{ runner.os }}-m2 - name: Deploy Release to Central Portal - run: cd java && mvn clean deploy + run: cd java && mvn --settings ../.github/settings.xml clean deploy env: MAVEN_USERNAME: ${{ secrets.SONATYPE_USERNAME }} MAVEN_PASSWORD: ${{ secrets.SONATYPE_PASSWORD }} From d84c9a9e677cdbe726164452fa0b7829b84691c3 Mon Sep 17 00:00:00 2001 From: Chico Sundermann Date: Thu, 2 Jul 2026 13:56:30 +0200 Subject: [PATCH 6/7] Chore: Replaces gpg with maven_gpg variables --- .github/workflows/java.yml | 2 +- java/pom.xml | 1 + 2 files changed, 2 insertions(+), 1 deletion(-) diff --git a/.github/workflows/java.yml b/.github/workflows/java.yml index 76a9a5d..14b70ad 100644 --- a/.github/workflows/java.yml +++ b/.github/workflows/java.yml @@ -60,4 +60,4 @@ jobs: env: MAVEN_USERNAME: ${{ secrets.SONATYPE_USERNAME }} MAVEN_PASSWORD: ${{ secrets.SONATYPE_PASSWORD }} - GPG_PASSPHRASE: ${{ secrets.GPG_PASSPHRASE }} + MAVEN_GPG_PASSPHRASE: ${{ secrets.GPG_PASSPHRASE }} diff --git a/java/pom.xml b/java/pom.xml index ef14571..a5df1eb 100644 --- a/java/pom.xml +++ b/java/pom.xml @@ -156,6 +156,7 @@ --pinentry-mode loopback + ${env.MAVEN_GPG_PASSPHRASE} From 9626e58f999475ecc373368a5061b59c6ad3ecb7 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Jos=C3=A9=20A=2E=20Galindo?= Date: Wed, 30 Sep 2026 14:57:00 +0300 Subject: [PATCH 7/7] perf: remove the semantic predicate from the NEWLINE lexer rule The NEWLINE rule of the Python, Java and JavaScript lexers started with the predicate {atStartOfInput()}?. ANTLR never caches a lexer's DFA start state when a semantic predicate is reachable from it (LexerATNSimulator.matchATN suppresses the s0 edge), so every token recomputed the full ATN closure: 177,973 closure computations for 224,698 tokens on a 1.5 MB model. Indentation before the first token is now handled once, in the custom lexers' handleLeadingIndentation(), which queues the same NEWLINE + INDENT tokens the predicated alternative emitted. The token streams are unchanged, so an indented first line is still rejected by the parser. Adds a faulty test model for the indented first line (picked up by the Python, JavaScript and Java test suites) and a Python test that the lexer's DFA start state is cached. Co-Authored-By: Claude Opus 5.5 (1M context) --- js/src/UVLJavaScriptCustomLexer.js | 36 ++++++++++++++- python/test_grammar.py | 12 +++++ python/uvl/UVLCustomLexer.py | 34 +++++++++++++-- test_models/faulty/indented_first_line.uvl | 4 ++ uvl/Java/UVLJavaLexer.g4 | 51 ++++++++++++++++++---- uvl/JavaScript/UVLJavaScriptLexer.g4 | 8 ++-- uvl/Python/UVLPythonLexer.g4 | 10 ++--- 7 files changed, 132 insertions(+), 23 deletions(-) create mode 100644 test_models/faulty/indented_first_line.uvl diff --git a/js/src/UVLJavaScriptCustomLexer.js b/js/src/UVLJavaScriptCustomLexer.js index 3f09df5..0f6f0a8 100644 --- a/js/src/UVLJavaScriptCustomLexer.js +++ b/js/src/UVLJavaScriptCustomLexer.js @@ -10,6 +10,7 @@ export default class UVLJavaScriptCustomLexer extends UVLJavaScriptLexer { this.indents = []; this.opened = 0; this.lastToken = null; + this.startChecked = false; } emitToken(t) { @@ -18,6 +19,11 @@ export default class UVLJavaScriptCustomLexer extends UVLJavaScriptLexer { } nextToken() { + if (!this.startChecked) { + this.startChecked = true; + this.handleLeadingIndentation(); + } + if (this._input.LA(1) === antlr4.Token.EOF && this.indents.length !== 0) { while (this.tokens.length > 0 && this.tokens[this.tokens.length - 1].type === antlr4.Token.EOF) { this.tokens.pop(); @@ -70,8 +76,34 @@ export default class UVLJavaScriptCustomLexer extends UVLJavaScriptLexer { this.skip(); } - atStartOfInput() { - return this._interp.column === 0 && this._interp.line === 1; + // Indentation before the first token is handled here, once, rather than with an + // {atStartOfInput()}? predicate in the NEWLINE rule: a predicate reachable from the lexer + // start state stops ANTLR from caching its DFA start state, which makes every token + // recompute the whole ATN closure. The spaces themselves are then skipped by SKIP_. + // Queues the same tokens handleNewline() emits for a leading line: an indented first + // line yields NEWLINE + INDENT, which the parser rejects. + handleLeadingIndentation() { + let length = 0; + while (this._input.LA(length + 1) === 32 || this._input.LA(length + 1) === 9) { + length += 1; + } + if (length === 0) { + return; + } + const next = String.fromCharCode(this._input.LA(length + 1)); + if (next === '\r' || next === '\n' || next === '\f' || next === '#') { + return; + } + const spaces = this._input.getText(0, length - 1); + this.indents.push(UVLJavaScriptCustomLexer.getIndentationCount(spaces)); + this.emitToken(this.leadingToken(UVLJavaScriptLexer.NEWLINE, length - 1, length - 1, length)); + this.emitToken(this.leadingToken(UVLJavaScriptParser.INDENT, 0, length - 1, length)); + } + + leadingToken(type, start, stop, column) { + const token = new antlr4.CommonToken(this._tokenFactorySourcePair, type, antlr4.Token.DEFAULT_CHANNEL, start, stop); + token.column = column; + return token; } handleNewline() { diff --git a/python/test_grammar.py b/python/test_grammar.py index 1c5658e..a1dde91 100644 --- a/python/test_grammar.py +++ b/python/test_grammar.py @@ -46,3 +46,15 @@ def test_valid_model(uvl_file): def test_faulty_model(uvl_file): with pytest.raises(Exception): get_tree(uvl_file) + + +def test_lexer_caches_its_dfa_start_state(): + # A semantic predicate reachable from the lexer start state stops ANTLR from caching the + # DFA start state (dfa.s0), so every token recomputes the full ATN closure: parsing gets + # several times slower. Keep lexer rules predicate-free at the start state. + from antlr4 import CommonTokenStream, FileStream + from uvl.UVLCustomLexer import UVLCustomLexer + + lexer = UVLCustomLexer(FileStream(os.path.join(TEST_MODELS_DIR, "complex", "bike.uvl"))) + CommonTokenStream(lexer).fill() + assert lexer._interp.decisionToDFA[lexer._mode].s0 is not None diff --git a/python/uvl/UVLCustomLexer.py b/python/uvl/UVLCustomLexer.py index 27a0d11..8e82869 100644 --- a/python/uvl/UVLCustomLexer.py +++ b/python/uvl/UVLCustomLexer.py @@ -13,12 +13,17 @@ def __init__(self, input_stream): self.indents = [] self.opened = 0 self.lastToken = None + self.startChecked = False def emitToken(self, t): super().emitToken(t) self.tokens.append(t) def nextToken(self): + if not self.startChecked: + self.startChecked = True + self.handleLeadingIndentation() + # Check if the end-of-file is ahead and there are still some DEDENTS expected. if self._input.LA(1) == Token.EOF and len(self.indents) != 0: # Remove any trailing EOF tokens from our buffer. @@ -67,9 +72,32 @@ def getIndentationCount(spaces): def skipToken(self): self.skip() - def atStartOfInput(self): - return self._interp.column == 0 and self._interp.line == 1 - + def handleLeadingIndentation(self): + # Indentation before the first token is handled here, once, rather than with a + # {atStartOfInput()}? predicate in the NEWLINE rule: a predicate reachable from the lexer + # start state stops ANTLR from caching its DFA start state, which makes every token + # recompute the whole ATN closure. The spaces themselves are then skipped by SKIP_. + # Queues the same tokens handleNewline() emits for a leading line: an indented first + # line yields NEWLINE + INDENT, which the parser rejects. + length = 0 + while self._input.LA(length + 1) in (ord(' '), ord('\t')): + length += 1 + if length == 0: + return + next_code = self._input.LA(length + 1) + next_char = chr(next_code) if next_code != -1 else '' + if next_char in '\r\n\f' or next_char == '/' or next_char == '#': + return + spaces = self._input.getText(0, length - 1) + self.indents.append(self.getIndentationCount(spaces)) + self.emitToken(self.leading_token(self.NEWLINE, length - 1, length - 1, length)) + self.emitToken(self.leading_token(UVLPythonParser.INDENT, 0, length - 1, length)) + + def leading_token(self, type, start, stop, column): + token = CommonToken(self._tokenFactorySourcePair, type, Token.DEFAULT_CHANNEL, start, stop) + token.column = column + return token + def handleNewline(self): new_line = re.sub(r"[^\r\n\f]+", "", self._interp.getText(self._input)) #.replaceAll("[^\r\n\f]+", "") spaces = re.sub(r"[\r\n\f]+", "", self._interp.getText(self._input)) #.replaceAll("[\r\n\f]+", "") diff --git a/test_models/faulty/indented_first_line.uvl b/test_models/faulty/indented_first_line.uvl new file mode 100644 index 0000000..6c7f2de --- /dev/null +++ b/test_models/faulty/indented_first_line.uvl @@ -0,0 +1,4 @@ + features + Root + optional + A diff --git a/uvl/Java/UVLJavaLexer.g4 b/uvl/Java/UVLJavaLexer.g4 index 8965902..50163f4 100644 --- a/uvl/Java/UVLJavaLexer.g4 +++ b/uvl/Java/UVLJavaLexer.g4 @@ -15,6 +15,8 @@ package uvl; private int opened = 0; // The most recently produced token. private Token lastToken = null; + // Whether the indentation before the first token has been handled. + private boolean startChecked = false; @Override public void emit(Token t) { @@ -24,6 +26,10 @@ package uvl; @Override public Token nextToken() { + if (!startChecked) { + startChecked = true; + handleLeadingIndentation(); + } // Check if the end-of-file is ahead and there are still some DEDENTS expected. if (_input.LA(1) == EOF && !this.indents.isEmpty()) { // Remove any trailing EOF tokens from our buffer. @@ -52,6 +58,37 @@ package uvl; return tokens.isEmpty() ? next : tokens.poll(); } + // Indentation before the first token is handled here, once, rather than with an + // {atStartOfInput()}? predicate in the NEWLINE rule: a predicate reachable from the lexer + // start state stops ANTLR from caching its DFA start state, which makes every token + // recompute the whole ATN closure. The spaces themselves are then skipped by SKIP_. + // Queues the same tokens the NEWLINE rule emits for a leading line: an indented first + // line yields NEWLINE + INDENT, which the parser rejects. + private void handleLeadingIndentation() { + int length = 0; + while (_input.LA(length + 1) == ' ' || _input.LA(length + 1) == '\t') { + length++; + } + if (length == 0) { + return; + } + int next = _input.LA(length + 1); + int nextNext = _input.LA(length + 2); + if (next == '\r' || next == '\n' || (next == '/' && nextNext == '/')) { + return; + } + String spaces = _input.getText(org.antlr.v4.runtime.misc.Interval.of(0, length - 1)); + indents.push(getIndentationCount(spaces)); + emit(leadingToken(UVLJavaParser.NEWLINE, length - 1, length - 1, length)); + emit(leadingToken(UVLJavaParser.INDENT, 0, length - 1, length)); + } + + private CommonToken leadingToken(int type, int start, int stop, int column) { + CommonToken token = new CommonToken(this._tokenFactorySourcePair, type, DEFAULT_TOKEN_CHANNEL, start, stop); + token.setCharPositionInLine(column); + return token; + } + private Token createDedent() { CommonToken dedent = commonToken(UVLJavaParser.DEDENT, ""); dedent.setLine(this.lastToken.getLine()); @@ -86,10 +123,6 @@ package uvl; } return count; } - - boolean atStartOfInput() { - return super.getCharPositionInLine() == 0 && super.getLine() == 1; - } } // Indentation-sensitive tokens @@ -102,11 +135,11 @@ CLOSE_BRACE: '}' {this.opened -= 1;}; OPEN_COMMENT: '/*' {this.opened += 1;}; CLOSE_COMMENT: '*/' {this.opened -= 1;}; -// Indentation-sensitive NEWLINE rule -NEWLINE: ( - {atStartOfInput()}? SPACES - | ( '\r'? '\n' | '\r') SPACES? - ) { +// Indentation-sensitive NEWLINE rule. +// No semantic predicate here: a predicate reachable from the lexer start state stops ANTLR +// from caching its DFA start state, so every token would recompute the full ATN closure. +// Indentation before the first token is handled once, in handleLeadingIndentation(). +NEWLINE: ( '\r'? '\n' | '\r') SPACES? { String newLine = getText().replaceAll("[^\r\n]+", ""); String spaces = getText().replaceAll("[\r\n]+", ""); int next = _input.LA(1); diff --git a/uvl/JavaScript/UVLJavaScriptLexer.g4 b/uvl/JavaScript/UVLJavaScriptLexer.g4 index a3c5d1d..12b6599 100644 --- a/uvl/JavaScript/UVLJavaScriptLexer.g4 +++ b/uvl/JavaScript/UVLJavaScriptLexer.g4 @@ -22,7 +22,7 @@ CLOSE_BRACE: '}' {this.opened -= 1;}; OPEN_COMMENT: '/*' {this.opened += 1;}; CLOSE_COMMENT: '*/' {this.opened -= 1;}; -NEWLINE: ( - {this.atStartOfInput()}? SPACES - | ( '\r'? '\n' | '\r') SPACES? - ) {this.handleNewline();}; \ No newline at end of file +// No semantic predicate here: a predicate reachable from the lexer start state stops ANTLR +// from caching its DFA start state, so every token would recompute the full ATN closure. +// Indentation before the first token is handled once, in UVLJavaScriptCustomLexer.handleLeadingIndentation(). +NEWLINE: ( '\r'? '\n' | '\r') SPACES? {this.handleNewline();}; \ No newline at end of file diff --git a/uvl/Python/UVLPythonLexer.g4 b/uvl/Python/UVLPythonLexer.g4 index 6ac23b9..e8cd80b 100644 --- a/uvl/Python/UVLPythonLexer.g4 +++ b/uvl/Python/UVLPythonLexer.g4 @@ -10,8 +10,8 @@ CLOSE_BRACE: '}' {self.opened -= 1;}; OPEN_COMMENT: '/*' {self.opened += 1;}; CLOSE_COMMENT: '*/' {self.opened -= 1;}; -//This is here because the way python manage tabs and new lines -NEWLINE: ( - {self.atStartOfInput()}? SPACES - | ( '\r'? '\n' | '\r') SPACES? - ) {self.handleNewline();}; \ No newline at end of file +//This is here because the way python manage tabs and new lines. +// No semantic predicate here: a predicate reachable from the lexer start state stops ANTLR +// from caching its DFA start state, so every token would recompute the full ATN closure. +// Indentation before the first token is handled once, in UVLCustomLexer.handleLeadingIndentation(). +NEWLINE: ( '\r'? '\n' | '\r') SPACES? {self.handleNewline();}; \ No newline at end of file