Skip to content
Draft
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
6 changes: 3 additions & 3 deletions .github/workflows/conventionalpr.yml
Original file line number Diff line number Diff line change
Expand Up @@ -2,7 +2,7 @@ name: Conventional Commit Parser
on:
pull_request:
branches: [main, master, develop]
types: [opened, reopened, edited, review_requested, synchronize]
types: [opened, reopened, edited, synchronize]

jobs:
conventional_commit:
Expand All @@ -11,6 +11,6 @@ jobs:
permissions:
pull-requests: read
steps:
- uses: webiny/action-conventional-commits@v1.3.0
with:
- uses: amannn/action-semantic-pull-request@v6.1.1
env:
GITHUB_TOKEN: ${{ secrets.GITHUB_TOKEN }}
36 changes: 5 additions & 31 deletions .github/workflows/java.yml
Original file line number Diff line number Diff line change
@@ -1,10 +1,9 @@
name: Build and Deploy Java Parser

on:
push:
branches: [develop, main]
tags: ['v*']

release:
types: [published]

jobs:
build:
runs-on: ubuntu-latest
Expand All @@ -22,7 +21,6 @@ jobs:
server-username: MAVEN_USERNAME
server-password: MAVEN_PASSWORD
gpg-private-key: ${{ secrets.GPG_PRIVATE_KEY }}
gpg-passphrase: MAVEN_GPG_PASSPHRASE

- name: Configure GPG for batch mode
run: |
Expand All @@ -41,27 +39,12 @@ jobs:
- name: Install ANTLR4
run: make dev

- name: Set SNAPSHOT version (develop branch)
if: github.ref == 'refs/heads/develop'
run: |
cd java
BASE_VERSION=$(mvn help:evaluate -Dexpression=project.version -q -DforceStdout | sed 's/-SNAPSHOT//')
mvn versions:set -DnewVersion="${BASE_VERSION}-SNAPSHOT" -DgenerateBackupPoms=false

- name: Set release version (main branch)
if: github.ref == 'refs/heads/main'
- name: Set release version
run: |
cd java
BASE_VERSION=$(mvn help:evaluate -Dexpression=project.version -q -DforceStdout | sed 's/-SNAPSHOT//')
mvn versions:set -DnewVersion="${BASE_VERSION}" -DgenerateBackupPoms=false

- name: Set release version (tag)
if: startsWith(github.ref, 'refs/tags/v')
run: |
cd java
TAG_VERSION=${GITHUB_REF_NAME#v}
mvn versions:set -DnewVersion="${TAG_VERSION}" -DgenerateBackupPoms=false

- name: Generate and Compile Java Code
run: make java_parser

Expand All @@ -72,17 +55,8 @@ jobs:
key: ${{ runner.os }}-m2-${{ hashFiles('**/pom.xml') }}
restore-keys: ${{ runner.os }}-m2

- name: Deploy SNAPSHOT to Sonatype Snapshots
if: github.ref == 'refs/heads/develop'
run: cd java && mvn clean deploy -DskipCentralPublishing=true
env:
MAVEN_USERNAME: ${{ secrets.SONATYPE_USERNAME }}
MAVEN_PASSWORD: ${{ secrets.SONATYPE_PASSWORD }}
MAVEN_GPG_PASSPHRASE: ${{ secrets.GPG_PASSPHRASE }}

- name: Deploy Release to Central Portal
if: github.ref == 'refs/heads/main' || startsWith(github.ref, 'refs/tags/v')
run: cd java && mvn clean deploy
run: cd java && mvn --settings ../.github/settings.xml clean deploy
env:
MAVEN_USERNAME: ${{ secrets.SONATYPE_USERNAME }}
MAVEN_PASSWORD: ${{ secrets.SONATYPE_PASSWORD }}
Expand Down
2 changes: 1 addition & 1 deletion README.md
Original file line number Diff line number Diff line change
Expand Up @@ -34,7 +34,7 @@ To use UVL in your projects, you can either:
<dependency>
<groupId>io.github.universal-variability-language</groupId>
<artifactId>uvl-parser</artifactId>
<version>0.3</version>
<version>0.5.1</version>
</dependency>
```
### Python Parser
Expand Down
4 changes: 2 additions & 2 deletions java/pom.xml
Original file line number Diff line number Diff line change
Expand Up @@ -11,7 +11,7 @@

<groupId>io.github.universal-variability-language</groupId>
<artifactId>uvl-parser</artifactId>
<version>0.5.0</version>
<version>0.5.1</version>

<licenses>
<license>
Expand Down Expand Up @@ -172,7 +172,7 @@
<plugin>
<groupId>org.sonatype.central</groupId>
<artifactId>central-publishing-maven-plugin</artifactId>
<version>0.5.0</version>
<version>0.11.0</version>
<extensions>true</extensions>
<configuration>
<publishingServerId>central</publishingServerId>
Expand Down
36 changes: 34 additions & 2 deletions js/src/UVLJavaScriptCustomLexer.js
Original file line number Diff line number Diff line change
Expand Up @@ -10,6 +10,7 @@ export default class UVLJavaScriptCustomLexer extends UVLJavaScriptLexer {
this.indents = [];
this.opened = 0;
this.lastToken = null;
this.startChecked = false;
}

emitToken(t) {
Expand All @@ -18,6 +19,11 @@ export default class UVLJavaScriptCustomLexer extends UVLJavaScriptLexer {
}

nextToken() {
if (!this.startChecked) {
this.startChecked = true;
this.handleLeadingIndentation();
}

if (this._input.LA(1) === antlr4.Token.EOF && this.indents.length !== 0) {
while (this.tokens.length > 0 && this.tokens[this.tokens.length - 1].type === antlr4.Token.EOF) {
this.tokens.pop();
Expand Down Expand Up @@ -70,8 +76,34 @@ export default class UVLJavaScriptCustomLexer extends UVLJavaScriptLexer {
this.skip();
}

atStartOfInput() {
return this._interp.column === 0 && this._interp.line === 1;
// Indentation before the first token is handled here, once, rather than with an
// {atStartOfInput()}? predicate in the NEWLINE rule: a predicate reachable from the lexer
// start state stops ANTLR from caching its DFA start state, which makes every token
// recompute the whole ATN closure. The spaces themselves are then skipped by SKIP_.
// Queues the same tokens handleNewline() emits for a leading line: an indented first
// line yields NEWLINE + INDENT, which the parser rejects.
handleLeadingIndentation() {
let length = 0;
while (this._input.LA(length + 1) === 32 || this._input.LA(length + 1) === 9) {
length += 1;
}
if (length === 0) {
return;
}
const next = String.fromCharCode(this._input.LA(length + 1));
if (next === '\r' || next === '\n' || next === '\f' || next === '#') {
return;
}
const spaces = this._input.getText(0, length - 1);
this.indents.push(UVLJavaScriptCustomLexer.getIndentationCount(spaces));
this.emitToken(this.leadingToken(UVLJavaScriptLexer.NEWLINE, length - 1, length - 1, length));
this.emitToken(this.leadingToken(UVLJavaScriptParser.INDENT, 0, length - 1, length));
}

leadingToken(type, start, stop, column) {
const token = new antlr4.CommonToken(this._tokenFactorySourcePair, type, antlr4.Token.DEFAULT_CHANNEL, start, stop);
token.column = column;
return token;
}

handleNewline() {
Expand Down
12 changes: 12 additions & 0 deletions python/test_grammar.py
Original file line number Diff line number Diff line change
Expand Up @@ -46,3 +46,15 @@ def test_valid_model(uvl_file):
def test_faulty_model(uvl_file):
with pytest.raises(Exception):
get_tree(uvl_file)


def test_lexer_caches_its_dfa_start_state():
# A semantic predicate reachable from the lexer start state stops ANTLR from caching the
# DFA start state (dfa.s0), so every token recomputes the full ATN closure: parsing gets
# several times slower. Keep lexer rules predicate-free at the start state.
from antlr4 import CommonTokenStream, FileStream
from uvl.UVLCustomLexer import UVLCustomLexer

lexer = UVLCustomLexer(FileStream(os.path.join(TEST_MODELS_DIR, "complex", "bike.uvl")))
CommonTokenStream(lexer).fill()
assert lexer._interp.decisionToDFA[lexer._mode].s0 is not None
34 changes: 31 additions & 3 deletions python/uvl/UVLCustomLexer.py
Original file line number Diff line number Diff line change
Expand Up @@ -13,12 +13,17 @@ def __init__(self, input_stream):
self.indents = []
self.opened = 0
self.lastToken = None
self.startChecked = False

def emitToken(self, t):
super().emitToken(t)
self.tokens.append(t)

def nextToken(self):
if not self.startChecked:
self.startChecked = True
self.handleLeadingIndentation()

# Check if the end-of-file is ahead and there are still some DEDENTS expected.
if self._input.LA(1) == Token.EOF and len(self.indents) != 0:
# Remove any trailing EOF tokens from our buffer.
Expand Down Expand Up @@ -67,9 +72,32 @@ def getIndentationCount(spaces):
def skipToken(self):
self.skip()

def atStartOfInput(self):
return self._interp.column == 0 and self._interp.line == 1

def handleLeadingIndentation(self):
# Indentation before the first token is handled here, once, rather than with a
# {atStartOfInput()}? predicate in the NEWLINE rule: a predicate reachable from the lexer
# start state stops ANTLR from caching its DFA start state, which makes every token
# recompute the whole ATN closure. The spaces themselves are then skipped by SKIP_.
# Queues the same tokens handleNewline() emits for a leading line: an indented first
# line yields NEWLINE + INDENT, which the parser rejects.
length = 0
while self._input.LA(length + 1) in (ord(' '), ord('\t')):
length += 1
if length == 0:
return
next_code = self._input.LA(length + 1)
next_char = chr(next_code) if next_code != -1 else ''
if next_char in '\r\n\f' or next_char == '/' or next_char == '#':
return
spaces = self._input.getText(0, length - 1)
self.indents.append(self.getIndentationCount(spaces))
self.emitToken(self.leading_token(self.NEWLINE, length - 1, length - 1, length))
self.emitToken(self.leading_token(UVLPythonParser.INDENT, 0, length - 1, length))

def leading_token(self, type, start, stop, column):
token = CommonToken(self._tokenFactorySourcePair, type, Token.DEFAULT_CHANNEL, start, stop)
token.column = column
return token

def handleNewline(self):
new_line = re.sub(r"[^\r\n\f]+", "", self._interp.getText(self._input)) #.replaceAll("[^\r\n\f]+", "")
spaces = re.sub(r"[\r\n\f]+", "", self._interp.getText(self._input)) #.replaceAll("[\r\n\f]+", "")
Expand Down
4 changes: 4 additions & 0 deletions test_models/faulty/indented_first_line.uvl
Original file line number Diff line number Diff line change
@@ -0,0 +1,4 @@
features
Root
optional
A
51 changes: 42 additions & 9 deletions uvl/Java/UVLJavaLexer.g4
Original file line number Diff line number Diff line change
Expand Up @@ -15,6 +15,8 @@ package uvl;
private int opened = 0;
// The most recently produced token.
private Token lastToken = null;
// Whether the indentation before the first token has been handled.
private boolean startChecked = false;

@Override
public void emit(Token t) {
Expand All @@ -24,6 +26,10 @@ package uvl;

@Override
public Token nextToken() {
if (!startChecked) {
startChecked = true;
handleLeadingIndentation();
}
// Check if the end-of-file is ahead and there are still some DEDENTS expected.
if (_input.LA(1) == EOF && !this.indents.isEmpty()) {
// Remove any trailing EOF tokens from our buffer.
Expand Down Expand Up @@ -52,6 +58,37 @@ package uvl;
return tokens.isEmpty() ? next : tokens.poll();
}

// Indentation before the first token is handled here, once, rather than with an
// {atStartOfInput()}? predicate in the NEWLINE rule: a predicate reachable from the lexer
// start state stops ANTLR from caching its DFA start state, which makes every token
// recompute the whole ATN closure. The spaces themselves are then skipped by SKIP_.
// Queues the same tokens the NEWLINE rule emits for a leading line: an indented first
// line yields NEWLINE + INDENT, which the parser rejects.
private void handleLeadingIndentation() {
int length = 0;
while (_input.LA(length + 1) == ' ' || _input.LA(length + 1) == '\t') {
length++;
}
if (length == 0) {
return;
}
int next = _input.LA(length + 1);
int nextNext = _input.LA(length + 2);
if (next == '\r' || next == '\n' || (next == '/' && nextNext == '/')) {
return;
}
String spaces = _input.getText(org.antlr.v4.runtime.misc.Interval.of(0, length - 1));
indents.push(getIndentationCount(spaces));
emit(leadingToken(UVLJavaParser.NEWLINE, length - 1, length - 1, length));
emit(leadingToken(UVLJavaParser.INDENT, 0, length - 1, length));
}

private CommonToken leadingToken(int type, int start, int stop, int column) {
CommonToken token = new CommonToken(this._tokenFactorySourcePair, type, DEFAULT_TOKEN_CHANNEL, start, stop);
token.setCharPositionInLine(column);
return token;
}

private Token createDedent() {
CommonToken dedent = commonToken(UVLJavaParser.DEDENT, "");
dedent.setLine(this.lastToken.getLine());
Expand Down Expand Up @@ -86,10 +123,6 @@ package uvl;
}
return count;
}

boolean atStartOfInput() {
return super.getCharPositionInLine() == 0 && super.getLine() == 1;
}
}

// Indentation-sensitive tokens
Expand All @@ -102,11 +135,11 @@ CLOSE_BRACE: '}' {this.opened -= 1;};
OPEN_COMMENT: '/*' {this.opened += 1;};
CLOSE_COMMENT: '*/' {this.opened -= 1;};

// Indentation-sensitive NEWLINE rule
NEWLINE: (
{atStartOfInput()}? SPACES
| ( '\r'? '\n' | '\r') SPACES?
) {
// Indentation-sensitive NEWLINE rule.
// No semantic predicate here: a predicate reachable from the lexer start state stops ANTLR
// from caching its DFA start state, so every token would recompute the full ATN closure.
// Indentation before the first token is handled once, in handleLeadingIndentation().
NEWLINE: ( '\r'? '\n' | '\r') SPACES? {
String newLine = getText().replaceAll("[^\r\n]+", "");
String spaces = getText().replaceAll("[\r\n]+", "");
int next = _input.LA(1);
Expand Down
8 changes: 4 additions & 4 deletions uvl/JavaScript/UVLJavaScriptLexer.g4
Original file line number Diff line number Diff line change
Expand Up @@ -22,7 +22,7 @@ CLOSE_BRACE: '}' {this.opened -= 1;};
OPEN_COMMENT: '/*' {this.opened += 1;};
CLOSE_COMMENT: '*/' {this.opened -= 1;};

NEWLINE: (
{this.atStartOfInput()}? SPACES
| ( '\r'? '\n' | '\r') SPACES?
) {this.handleNewline();};
// No semantic predicate here: a predicate reachable from the lexer start state stops ANTLR
// from caching its DFA start state, so every token would recompute the full ATN closure.
// Indentation before the first token is handled once, in UVLJavaScriptCustomLexer.handleLeadingIndentation().
NEWLINE: ( '\r'? '\n' | '\r') SPACES? {this.handleNewline();};
10 changes: 5 additions & 5 deletions uvl/Python/UVLPythonLexer.g4
Original file line number Diff line number Diff line change
Expand Up @@ -10,8 +10,8 @@ CLOSE_BRACE: '}' {self.opened -= 1;};
OPEN_COMMENT: '/*' {self.opened += 1;};
CLOSE_COMMENT: '*/' {self.opened -= 1;};

//This is here because the way python manage tabs and new lines
NEWLINE: (
{self.atStartOfInput()}? SPACES
| ( '\r'? '\n' | '\r') SPACES?
) {self.handleNewline();};
//This is here because the way python manage tabs and new lines.
// No semantic predicate here: a predicate reachable from the lexer start state stops ANTLR
// from caching its DFA start state, so every token would recompute the full ATN closure.
// Indentation before the first token is handled once, in UVLCustomLexer.handleLeadingIndentation().
NEWLINE: ( '\r'? '\n' | '\r') SPACES? {self.handleNewline();};
Loading