Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
Original file line number Diff line number Diff line change
Expand Up @@ -7,6 +7,7 @@
import java.util.ArrayList;
import java.util.HashMap;
import java.util.HashSet;
import java.util.LinkedHashSet;
import java.util.List;
import java.util.Map;
import java.util.Set;
Expand All @@ -27,21 +28,45 @@ public class MultiCharFoldMapper {
// Format: fold string → characters
private static final Map<String, List<Integer>> REVERSE_FOLDS = new HashMap<>();
private static final Set<Integer> FOLD_COMPONENTS = new HashSet<>();
private static final Map<Integer, List<Integer>> SIMPLE_FOLD_CLASSES = new HashMap<>();

static {
// ICU4J is already the runtime's Unicode source of truth. Derive all
// full folds instead of maintaining a partial hand-written table.
// CHANGES_WHEN_CASEMAPPED is intentionally used as a superset:
// CHANGES_WHEN_CASEFOLDED excludes lowercase characters such as
// U+01F0 and U+0390 even though their full fold has multiple code
// points.
UnicodeSet foldCandidates = new UnicodeSet()
.applyIntPropertyValue(UProperty.CHANGES_WHEN_CASEFOLDED, 1);
.applyIntPropertyValue(UProperty.CHANGES_WHEN_CASEMAPPED, 1);
Map<Integer, LinkedHashSet<Integer>> simpleFoldClasses = new HashMap<>();
for (String original : foldCandidates) {
if (original.codePointCount(0, original.length()) != 1) continue;
int codePoint = original.codePointAt(0);
String fold = UCharacter.foldCase(original, true);
if (fold.codePointCount(0, fold.length()) > 1) {
MULTI_CHAR_FOLDS.put(codePoint, fold);
} else {
int foldedCodePoint = fold.codePointAt(0);
LinkedHashSet<Integer> foldClass = simpleFoldClasses.computeIfAbsent(
foldedCodePoint, ignored -> new LinkedHashSet<>());
foldClass.add(foldedCodePoint);
foldClass.add(codePoint);
}
}

for (LinkedHashSet<Integer> foldClass : simpleFoldClasses.values()) {
if (foldClass.size() < 2) continue;
List<Integer> variants = List.copyOf(foldClass);
String representative = new String(Character.toChars(variants.get(0)));
Pattern javaFoldPattern = Pattern.compile(Pattern.quote(representative),
Pattern.CASE_INSENSITIVE | Pattern.UNICODE_CASE);
boolean javaSupportsFoldClass = variants.stream().allMatch(codePoint ->
javaFoldPattern.matcher(new String(Character.toChars(codePoint))).matches());
if (javaSupportsFoldClass) continue;
for (int codePoint : variants) SIMPLE_FOLD_CLASSES.put(codePoint, variants);
}

// Build reverse map (lowercase versions only for simpler matching)
for (Map.Entry<Integer, String> entry : MULTI_CHAR_FOLDS.entrySet()) {
String fold = entry.getValue();
Expand Down Expand Up @@ -87,10 +112,16 @@ public static String expandToAlternation(int codePoint) {
StringBuilder sb = new StringBuilder("(?:");
String original = new String(Character.toChars(codePoint));
sb.append(Pattern.quote(original));
sb.append("|");

// Include sibling code points with the same full fold. For example,
// both U+00DF and U+1E9E fold to "ss".
for (int reverseFold : REVERSE_FOLDS.getOrDefault(fold, List.of())) {
if (reverseFold == codePoint) continue;
sb.append('|').append(Pattern.quote(new String(Character.toChars(reverseFold))));
}

// Add the basic fold
sb.append(Pattern.quote(fold));
sb.append('|').append(Pattern.quote(fold));

// Add case variations of the fold (if it's ASCII)
if (fold.chars().allMatch(c -> c >= 'a' && c <= 'z')) {
Expand Down Expand Up @@ -136,6 +167,32 @@ public static List<Integer> getReverseFolds(String str) {
return folds == null ? List.of() : folds;
}

/** Whether ICU knows a non-trivial single-code-point fold for this literal. */
public static boolean hasSimpleFold(int codePoint) {
return SIMPLE_FOLD_CLASSES.containsKey(codePoint);
}

/**
* Expand a single-code-point fold class into a quoted alternation. This
* supplements Java Pattern for characters newer than the JDK's Unicode
* tables while remaining harmless for fold classes Java already knows.
*/
public static String expandSimpleFoldToAlternation(int codePoint) {
List<Integer> variants = SIMPLE_FOLD_CLASSES.get(codePoint);
if (variants == null) return null;
StringBuilder result = new StringBuilder("(?:");
for (int i = 0; i < variants.size(); i++) {
if (i > 0) result.append('|');
result.append(Pattern.quote(new String(Character.toChars(variants.get(i)))));
}
return result.append(')').toString();
}

/** Get all members of a simple fold class, or an empty list. */
public static List<Integer> getSimpleFoldVariants(int codePoint) {
return SIMPLE_FOLD_CLASSES.getOrDefault(codePoint, List.of());
}

/** Whether a literal code point can participate in a reverse multi-char fold. */
public static boolean isFoldComponent(int codePoint) {
if (FOLD_COMPONENTS.contains(codePoint)) {
Expand Down
19 changes: 15 additions & 4 deletions src/main/java/org/perlonjava/runtime/regex/RegexPreprocessor.java
Original file line number Diff line number Diff line change
Expand Up @@ -995,8 +995,11 @@ private static String expandMultiCharFolds(String pattern, RegexFlags regexFlags
}

if (!foundReverseFold) {
String specialExpansion = regexFlags.isAsciiStrict()
? null : expandSpecialSingleCharFold(codePoint);
String specialExpansion = regexFlags.isAsciiStrict() ? null
: MultiCharFoldMapper.expandSimpleFoldToAlternation(codePoint);
if (specialExpansion == null && !regexFlags.isAsciiStrict()) {
specialExpansion = expandSpecialSingleCharFold(codePoint);
}
if (specialExpansion != null) {
result.append(specialExpansion);
} else {
Expand All @@ -1016,6 +1019,7 @@ private static String expandMultiCharFoldClass(String charClass, RegexFlags rege
if (regexFlags.isAsciiStrict() || charClass.length() < 3 || charClass.charAt(1) == '^') return null;

LinkedHashSet<String> folds = new LinkedHashSet<>();
LinkedHashSet<Integer> simpleVariants = new LinkedHashSet<>();
boolean escaped = false;
for (int i = 1; i < charClass.length() - 1; ) {
int codePoint = charClass.codePointAt(i);
Expand All @@ -1032,11 +1036,17 @@ private static String expandMultiCharFoldClass(String charClass, RegexFlags rege
if (codePoint == '-' || codePoint == '[' || codePoint == ']') return null;
String fold = MultiCharFoldMapper.getMultiCharFold(codePoint);
if (fold != null) folds.add(Pattern.quote(fold));
simpleVariants.addAll(MultiCharFoldMapper.getSimpleFoldVariants(codePoint));
i += Character.charCount(codePoint);
}
if (folds.isEmpty()) return null;
if (folds.isEmpty() && simpleVariants.isEmpty()) return null;

StringBuilder expandedClass = new StringBuilder(charClass.substring(0, charClass.length() - 1));
simpleVariants.forEach(expandedClass::appendCodePoint);
expandedClass.append(']');
if (folds.isEmpty()) return expandedClass.toString();

StringBuilder expansion = new StringBuilder("(?:").append(charClass);
StringBuilder expansion = new StringBuilder("(?:").append(expandedClass);
for (String fold : folds) expansion.append('|').append(fold);
return expansion.append(')').toString();
}
Expand All @@ -1058,6 +1068,7 @@ private static String materializeFoldableHexEscapes(String pattern) {
int codePoint = Integer.parseInt(pattern.substring(i + 3, close), 16);
if (Character.isValidCodePoint(codePoint)
&& (MultiCharFoldMapper.hasMultiCharFold(codePoint)
|| MultiCharFoldMapper.hasSimpleFold(codePoint)
|| MultiCharFoldMapper.isFoldComponent(codePoint))) {
result.appendCodePoint(codePoint);
i = close + 1;
Expand Down
Original file line number Diff line number Diff line change
@@ -0,0 +1,30 @@
package org.perlonjava.runtime.regex;

import org.junit.jupiter.api.Tag;
import org.junit.jupiter.api.Test;

import java.util.List;

import static org.junit.jupiter.api.Assertions.assertEquals;
import static org.junit.jupiter.api.Assertions.assertTrue;

@Tag("unit")
class MultiCharFoldMapperTest {
@Test
void includesLowercaseCharactersWithFullFolds() {
assertEquals("j\u030C", MultiCharFoldMapper.getMultiCharFold(0x01F0));
assertEquals("\u03B9\u0308\u0301", MultiCharFoldMapper.getMultiCharFold(0x0390));
}

@Test
void includesEveryCodePointSharingAFullFold() {
assertTrue(MultiCharFoldMapper.getReverseFolds("ss").containsAll(List.of(0x00DF, 0x1E9E)));
assertTrue(MultiCharFoldMapper.expandToAlternation(0x00DF).contains("\u1E9E"));
}

@Test
void supplementsSimpleFoldsMissingFromJavaPattern() {
assertTrue(MultiCharFoldMapper.getSimpleFoldVariants(0xA7CE).contains(0xA7CF));
assertTrue(MultiCharFoldMapper.getSimpleFoldVariants(0x16EA0).contains(0x16EBB));
}
}
20 changes: 20 additions & 0 deletions src/test/resources/unit/regex_icu_full_casefold.t
Original file line number Diff line number Diff line change
@@ -0,0 +1,20 @@
#!/usr/bin/env perl
use strict;
use warnings;
use utf8;
use Test::More tests => 6;

ok("\x{01F0}" =~ /^\x{006A}\x{030C}$/iu,
'lowercase j with caron expands to its full fold');
ok("\x{006A}\x{030C}" =~ /^\x{01F0}$/iu,
'j plus combining caron reverse-folds to one code point');

ok("\x{0390}" =~ /^\x{03B9}\x{0308}\x{0301}$/iu,
'Greek lowercase character expands to its three-code-point fold');
ok("\x{03B9}\x{0308}\x{0301}" =~ /^\x{0390}$/iu,
'Greek three-code-point sequence reverse-folds');

ok("\x{1E9E}" =~ /^\x{00DF}$/iu,
'capital sharp s matches its sibling full-fold source');
ok("\x{00DF}" =~ /^\x{1E9E}$/iu,
'sibling full-fold source matching is symmetric');
Loading