Skip to content

Commit 0a75c47

Browse files
committed
[GR-80056] TRegex: fix bugs found by fuzzing.
PullRequest: graal/25556
2 parents 8f93fa2 + 258973d commit 0a75c47

7 files changed

Lines changed: 38 additions & 18 deletions

File tree

‎regex/src/com.oracle.truffle.regex.test/src/com/oracle/truffle/regex/tregex/test/generated/OracleDBGeneratedTests.java‎

Lines changed: 3 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -1727,6 +1727,9 @@ public class OracleDBGeneratedTests {
17271727
testCase("w[v-w]\\W", "", UTF_8, match("w\ud839\udfefwww\ud839\udfefw\ud839\udfefw", 5, 6, 12)),
17281728
testCase("(a{1100,1100})\\1", "i", UTF_8, match("a".repeat(2400), 0, 0, 2200, 0, 1100)),
17291729
testCase("[a]\\S{213,213}bcdz", "", UTF_8, noMatch("a".repeat(215) + ("bcxd" + "a".repeat(213)).repeat(3), 0)),
1730+
testCase("z()|(d)[[=\ud97a\udcb9=]\ud9ba\udcb9]", "m", UTF_16BE, match("zz", 0, 0, 1, 1, 1, -1, -1)),
1731+
testCase("(|\udbda\udcf5)[[.\uda83\udd45.][.\uda43\udd45.]]", "", UTF_16BE, match("\udbda\udcf5\uda83\udd45", 0, 0, 4, 0, 2)),
1732+
testCase("[[=\u01c4=]]", "", UTF_8, match("\u01c6", 0, 0, 2)),
17301733

17311734
/* GENERATED CODE END - KEEP THIS MARKER FOR AUTOMATIC UPDATES */
17321735
// Checkstyle: resume

‎regex/src/com.oracle.truffle.regex/src/com/oracle/truffle/regex/tregex/parser/MultiCharacterCaseFolding.java‎

Lines changed: 10 additions & 11 deletions
Original file line numberDiff line numberDiff line change
@@ -258,18 +258,17 @@ public static void caseClosure(CaseFoldData.CaseFoldAlgorithm algorithm, CodePoi
258258
CodePointSet allowedCodePoints, boolean transitiveEquivalence) {
259259
tmp.clear();
260260
caseFoldCharClass(algorithm, charClass, (from, to) -> {
261-
if (transitiveEquivalence || hasNoCaseFolding(algorithm, to[0])) {
262-
if (to.length == 1) {
263-
// Add the case-folded version to the character class...
264-
if (filter.test(from, to[0])) {
265-
tmp.addCodePoint(to[0]);
266-
}
261+
if (to.length == 1 && (transitiveEquivalence || hasNoCaseFolding(algorithm, to[0]))) {
262+
// Add the case-folded version to the character class.
263+
if (filter.test(from, to[0])) {
264+
tmp.addCodePoint(to[0]);
267265
}
268-
// ... and also any characters which case-fold to the same.
269-
for (int unfolding : CaseUnfoldingTrie.findSingleCharUnfoldings(algorithm, to)) {
270-
if (unfolding != from && filter.test(from, unfolding)) {
271-
tmp.addCodePoint(unfolding);
272-
}
266+
}
267+
// Add any characters which case-fold to the same, even if the common case-folded value
268+
// itself has another case-folding.
269+
for (int unfolding : CaseUnfoldingTrie.findSingleCharUnfoldings(algorithm, to)) {
270+
if (unfolding != from && filter.test(from, unfolding)) {
271+
tmp.addCodePoint(unfolding);
273272
}
274273
}
275274
});

‎regex/src/com.oracle.truffle.regex/src/com/oracle/truffle/regex/tregex/parser/RegexASTPostProcessor.java‎

Lines changed: 1 addition & 2 deletions
Original file line numberDiff line numberDiff line change
@@ -127,8 +127,7 @@ private void checkInnerLiteral() {
127127

128128
private boolean isLiteralChar(Term t) {
129129
return t.isCharacterClass() &&
130-
(t.asCharacterClass().getCharSet().matchesSingleChar() || t.asCharacterClass().getCharSet().matches2CharsWith1BitDifference()) &&
131-
ast.getEncoding().isFixedCodePointWidth(t.asCharacterClass().getCharSet()) &&
130+
ast.getEncoding().canBeMatchedWithMask(t.asCharacterClass().getCharSet()) &&
132131
!(ast.getEncoding().isUTF16() && t.asCharacterClass().getCharSet().intersects(Constants.SURROGATES));
133132
}
134133

‎regex/src/com.oracle.truffle.regex/src/com/oracle/truffle/regex/tregex/parser/ast/CalcASTPropsVisitor.java‎

Lines changed: 1 addition & 1 deletion
Original file line numberDiff line numberDiff line change
@@ -579,7 +579,7 @@ static void setFlagsSubtreeRootNode(RegexASTSubtreeRootNode subtreeRootNode, int
579579
protected void visit(CharacterClass characterClass) {
580580
if (isForward()) {
581581
if (!characterClass.getCharSet().matchesSingleChar()) {
582-
if (!characterClass.getCharSet().matches2CharsWith1BitDifference()) {
582+
if (!ast.getEncoding().canBeMatchedWithMask(characterClass.getCharSet())) {
583583
ast.getProperties().unsetCharClassesCanBeMatchedWithMask();
584584
}
585585
if (!ast.getEncoding().isFixedCodePointWidth(characterClass.getCharSet())) {

‎regex/src/com.oracle.truffle.regex/src/com/oracle/truffle/regex/tregex/parser/ast/CharacterClass.java‎

Lines changed: 1 addition & 1 deletion
Original file line numberDiff line numberDiff line change
@@ -183,7 +183,7 @@ public void extractSingleChar(AbstractStringBuffer literal, AbstractStringBuffer
183183
mask.append(0);
184184
}
185185
} else {
186-
assert charSet.matches2CharsWith1BitDifference();
186+
assert mask.getEncoding().canBeMatchedWithMask(charSet);
187187
int c1 = charSet.getMin();
188188
int c2 = charSet.getMax();
189189
literal.appendOR(c1, c2);

‎regex/src/com.oracle.truffle.regex/src/com/oracle/truffle/regex/tregex/parser/ast/RegexAST.java‎

Lines changed: 2 additions & 3 deletions
Original file line numberDiff line numberDiff line change
@@ -707,10 +707,9 @@ public InnerLiteral extractInnerLiteral() {
707707
boolean hasMask = false;
708708
for (int i = literalStart; i < literalEnd; i++) {
709709
CharacterClass cc = root.getFirstAlternative().getTerms().get(i).asCharacterClass();
710-
assert cc.getCharSet().matchesSingleChar() || cc.getCharSet().matches2CharsWith1BitDifference();
711-
assert getEncoding().isFixedCodePointWidth(cc.getCharSet());
710+
assert getEncoding().canBeMatchedWithMask(cc.getCharSet());
712711
cc.extractSingleChar(literal, mask);
713-
hasMask |= cc.getCharSet().matches2CharsWith1BitDifference();
712+
hasMask |= !cc.getCharSet().matchesSingleChar();
714713
}
715714
int maxPrefixSize = root.getFirstAlternative().get(literalStart).getMaxPath() - 1;
716715
for (int i = 0; i < literalStart; i++) {

‎regex/src/com.oracle.truffle.regex/src/com/oracle/truffle/regex/tregex/string/Encoding.java‎

Lines changed: 20 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -202,6 +202,26 @@ public boolean isFixedCodePointWidth(CodePointSet set) {
202202
}
203203
}
204204

205+
/**
206+
* Returns {@code true} if the given code point set can be represented as a literal and a bit
207+
* mask without matching any additional code points.
208+
*/
209+
public boolean canBeMatchedWithMask(CodePointSet set) {
210+
if (set.matchesSingleChar()) {
211+
return true;
212+
}
213+
if (!set.matches2CharsWith1BitDifference() || !isFixedCodePointWidth(set)) {
214+
return false;
215+
}
216+
if (isUTF16() && set.getMin() > Character.MAX_VALUE) {
217+
int c1 = set.getMin();
218+
int c2 = set.getMax();
219+
return Integer.bitCount(Character.highSurrogate(c1) ^ Character.highSurrogate(c2)) +
220+
Integer.bitCount(Character.lowSurrogate(c1) ^ Character.lowSurrogate(c2)) == 1;
221+
}
222+
return true;
223+
}
224+
205225
public boolean isUnicode() {
206226
return switch (this) {
207227
case LATIN_1, ASCII, BYTES -> false;

0 commit comments

Comments
 (0)