diff --git a/js/src/irregexp/RegExpCharRanges.cpp b/js/src/irregexp/RegExpCharRanges.cpp new file mode 100644 index 0000000000..3bd3f99de4 --- /dev/null +++ b/js/src/irregexp/RegExpCharRanges.cpp @@ -0,0 +1,623 @@ +/* -*- Mode: C++; tab-width: 8; indent-tabs-mode: nil; c-basic-offset: 4 -*- */ + +// Copyright 2012 the V8 project authors. All rights reserved. +// Redistribution and use in source and binary forms, with or without +// modification, are permitted provided that the following conditions are +// met: +// +// * Redistributions of source code must retain the above copyright +// notice, this list of conditions and the following disclaimer. +// * Redistributions in binary form must reproduce the above +// copyright notice, this list of conditions and the following +// disclaimer in the documentation and/or other materials provided +// with the distribution. +// * Neither the name of Google Inc. nor the names of its +// contributors may be used to endorse or promote products derived +// from this software without specific prior written permission. +// +// THIS SOFTWARE IS PROVIDED BY THE COPYRIGHT HOLDERS AND CONTRIBUTORS +// "AS IS" AND ANY EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT +// LIMITED TO, THE IMPLIED WARRANTIES OF MERCHANTABILITY AND FITNESS FOR +// A PARTICULAR PURPOSE ARE DISCLAIMED. IN NO EVENT SHALL THE COPYRIGHT +// OWNER OR CONTRIBUTORS BE LIABLE FOR ANY DIRECT, INDIRECT, INCIDENTAL, +// SPECIAL, EXEMPLARY, OR CONSEQUENTIAL DAMAGES (INCLUDING, BUT NOT +// LIMITED TO, PROCUREMENT OF SUBSTITUTE GOODS OR SERVICES; LOSS OF USE, +// DATA, OR PROFITS; OR BUSINESS INTERRUPTION) HOWEVER CAUSED AND ON ANY +// THEORY OF LIABILITY, WHETHER IN CONTRACT, STRICT LIABILITY, OR TORT +// (INCLUDING NEGLIGENCE OR OTHERWISE) ARISING IN ANY WAY OUT OF THE USE +// OF THIS SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE. + +#include "irregexp/RegExpCharRanges.h" + +// Generated table +#include "irregexp/RegExpCharacters-inl.h" + +using namespace js::irregexp; + +using mozilla::ArrayLength; + +void +CharacterRange::AddCaseEquivalents(bool is_ascii, bool unicode, CharacterRangeVector* ranges) +{ + char16_t bottom = from(); + char16_t top = to(); + + if (is_ascii && !RangeContainsLatin1Equivalents(*this, unicode)) { + if (bottom > kMaxOneByteCharCode) + return; + if (top > kMaxOneByteCharCode) + top = kMaxOneByteCharCode; + } + + for (char16_t c = bottom;; c++) { + char16_t chars[kEcma262UnCanonicalizeMaxWidth]; + size_t length = GetCaseIndependentLetters(c, is_ascii, unicode, chars); + + for (size_t i = 0; i < length; i++) { + char16_t other = chars[i]; + if (other == c) + continue; + + // Try to combine with an existing range. + bool found = false; + for (size_t i = 0; i < ranges->length(); i++) { + CharacterRange& range = (*ranges)[i]; + if (range.Contains(other)) { + found = true; + break; + } else if (other == range.from() - 1) { + range.set_from(other); + found = true; + break; + } else if (other == range.to() + 1) { + range.set_to(other); + found = true; + break; + } + } + + if (!found) + ranges->append(CharacterRange::Singleton(other)); + } + + if (c == top) + break; + } +} + +/* static */ +void +CharacterRange::AddClass(const int* elmv, int elmc, CharacterRangeVector* ranges) +{ + elmc--; + MOZ_ASSERT(elmv[elmc] == 0x10000); + for (int i = 0; i < elmc; i += 2) { + MOZ_ASSERT(elmv[i] < elmv[i + 1]); + ranges->append(CharacterRange(elmv[i], elmv[i + 1] - 1)); + } +} + +/* static */ void +CharacterRange::AddClassNegated(const int* elmv, int elmc, CharacterRangeVector* ranges) +{ + elmc--; + MOZ_ASSERT(elmv[elmc] == 0x10000); + MOZ_ASSERT(elmv[0] != 0x0000); + MOZ_ASSERT(elmv[elmc-1] != kMaxUtf16CodeUnit); + char16_t last = 0x0000; + for (int i = 0; i < elmc; i += 2) { + MOZ_ASSERT(last <= elmv[i] - 1); + MOZ_ASSERT(elmv[i] < elmv[i + 1]); + ranges->append(CharacterRange(last, elmv[i] - 1)); + last = elmv[i + 1]; + } + ranges->append(CharacterRange(last, kMaxUtf16CodeUnit)); +} + +/* static */ void +CharacterRange::AddClassEscape(LifoAlloc* alloc, char16_t type, + CharacterRangeVector* ranges) +{ + switch (type) { + case 's': + AddClass(kSpaceRanges, kSpaceRangeCount, ranges); + break; + case 'S': + AddClassNegated(kSpaceRanges, kSpaceRangeCount, ranges); + break; + case 'w': + AddClass(kWordRanges, kWordRangeCount, ranges); + break; + case 'W': + AddClassNegated(kWordRanges, kWordRangeCount, ranges); + break; + case 'd': + AddClass(kDigitRanges, kDigitRangeCount, ranges); + break; + case 'D': + AddClassNegated(kDigitRanges, kDigitRangeCount, ranges); + break; + case '.': + AddClassNegated(kLineTerminatorRanges, kLineTerminatorRangeCount, ranges); + break; + // This is not a character range as defined by the spec but a + // convenient shorthand for a character class that matches any + // character. + case '*': + ranges->append(CharacterRange::Everything()); + break; + // This is the set of characters matched by the $ and ^ symbols + // in multiline mode. + case 'n': + AddClass(kLineTerminatorRanges, kLineTerminatorRangeCount, ranges); + break; + default: + MOZ_CRASH("Bad character class escape"); + } +} + +// Add class escape, excluding surrogate pair range. +/* static */ void +CharacterRange::AddClassEscapeUnicode(LifoAlloc* alloc, char16_t type, + CharacterRangeVector* ranges, bool ignore_case) +{ + switch (type) { + case 's': + case 'd': + return AddClassEscape(alloc, type, ranges); + break; + case 'S': + AddClassNegated(kSpaceAndSurrogateRanges, kSpaceAndSurrogateRangeCount, ranges); + break; + case 'w': + if (ignore_case) + AddClass(kIgnoreCaseWordRanges, kIgnoreCaseWordRangeCount, ranges); + else + AddClassEscape(alloc, type, ranges); + break; + case 'W': + if (ignore_case) { + AddClass(kNegatedIgnoreCaseWordAndSurrogateRanges, + kNegatedIgnoreCaseWordAndSurrogateRangeCount, ranges); + } else { + AddClassNegated(kWordAndSurrogateRanges, kWordAndSurrogateRangeCount, ranges); + } + break; + case 'D': + AddClassNegated(kDigitAndSurrogateRanges, kDigitAndSurrogateRangeCount, ranges); + break; + default: + MOZ_CRASH("Bad type!"); + } +} + +/* static */ void +CharacterRange::AddCharOrEscape(LifoAlloc* alloc, CharacterRangeVector* ranges, + char16_t char_class, widechar c) +{ + if (char_class != kNoCharClass) + AddClassEscape(alloc, char_class, ranges); + else + ranges->append(CharacterRange::Singleton(c)); +} + +/* static */ void +CharacterRange::AddCharOrEscapeUnicode(LifoAlloc* alloc, + CharacterRangeVector* ranges, + CharacterRangeVector* lead_ranges, + CharacterRangeVector* trail_ranges, + WideCharRangeVector* wide_ranges, + char16_t char_class, + widechar c, + bool ignore_case) +{ + if (char_class != kNoCharClass) { + AddClassEscapeUnicode(alloc, char_class, ranges, ignore_case); + switch (char_class) { + case 'S': + case 'W': + case 'D': + lead_ranges->append(CharacterRange::LeadSurrogate()); + trail_ranges->append(CharacterRange::TrailSurrogate()); + wide_ranges->append(WideCharRange::NonBMP()); + break; + case '.': + MOZ_CRASH("Bad char_class!"); + } + return; + } + + if (unicode::IsLeadSurrogate(c)) + lead_ranges->append(CharacterRange::Singleton(c)); + else if (unicode::IsTrailSurrogate(c)) + trail_ranges->append(CharacterRange::Singleton(c)); + else if (c >= unicode::NonBMPMin) + wide_ranges->append(WideCharRange::Singleton(c)); + else + ranges->append(CharacterRange::Singleton(c)); +} + +/* static */ void +CharacterRange::AddCharUnicode(LifoAlloc* alloc, + CharacterRangeVector* ranges, + CharacterRangeVector* lead_ranges, + CharacterRangeVector* trail_ranges, + WideCharRangeVector* wide_ranges, + widechar c) +{ + if (unicode::IsLeadSurrogate(c)) + lead_ranges->append(CharacterRange::Singleton(c)); + else if (unicode::IsTrailSurrogate(c)) + trail_ranges->append(CharacterRange::Singleton(c)); + else if (c >= unicode::NonBMPMin) + wide_ranges->append(WideCharRange::Singleton(c)); + else + ranges->append(CharacterRange::Singleton(c)); +} + +/* static */ void +CharacterRange::AddUnicodeRange(LifoAlloc* alloc, + CharacterRangeVector* ranges, + CharacterRangeVector* lead_ranges, + CharacterRangeVector* trail_ranges, + WideCharRangeVector* wide_ranges, + widechar first, + widechar next) +{ + MOZ_ASSERT(first <= next); + if (first < unicode::LeadSurrogateMin) { + if (next < unicode::LeadSurrogateMin) { + ranges->append(CharacterRange::Range(first, next)); + return; + } + ranges->append(CharacterRange::Range(first, unicode::LeadSurrogateMin - 1)); + first = unicode::LeadSurrogateMin; + } + if (first <= unicode::LeadSurrogateMax) { + if (next <= unicode::LeadSurrogateMax) { + lead_ranges->append(CharacterRange::Range(first, next)); + return; + } + lead_ranges->append(CharacterRange::Range(first, unicode::LeadSurrogateMax)); + first = unicode::LeadSurrogateMax + 1; + } + MOZ_ASSERT(unicode::LeadSurrogateMax + 1 == unicode::TrailSurrogateMin); + if (first <= unicode::TrailSurrogateMax) { + if (next <= unicode::TrailSurrogateMax) { + trail_ranges->append(CharacterRange::Range(first, next)); + return; + } + trail_ranges->append(CharacterRange::Range(first, unicode::TrailSurrogateMax)); + first = unicode::TrailSurrogateMax + 1; + } + if (first <= unicode::UTF16Max) { + if (next <= unicode::UTF16Max) { + ranges->append(CharacterRange::Range(first, next)); + return; + } + ranges->append(CharacterRange::Range(first, unicode::UTF16Max)); + first = unicode::NonBMPMin; + } + MOZ_ASSERT(unicode::UTF16Max + 1 == unicode::NonBMPMin); + wide_ranges->append(WideCharRange::Range(first, next)); +} + +/* static */ bool +CharacterRange::RangesContainLatin1Equivalents(const CharacterRangeVector& ranges, bool unicode) +{ + for (size_t i = 0; i < ranges.length(); i++) { + // TODO(dcarney): this could be a lot more efficient. + if (RangeContainsLatin1Equivalents(ranges[i], unicode)) + return true; + } + return false; +} + +/* static */ bool +CharacterRange::CompareRanges(const CharacterRangeVector& ranges, const int* special_class, size_t length) +{ + length--; // Remove final 0x10000. + MOZ_ASSERT(special_class[length] == 0x10000); + if (ranges.length() * 2 != length) + return false; + for (size_t i = 0; i < length; i += 2) { + CharacterRange range = ranges[i >> 1]; + if (range.from() != special_class[i] || range.to() != special_class[i + 1] - 1) + return false; + } + return true; +} + +/* static */ bool +CharacterRange::CompareInverseRanges(const CharacterRangeVector& ranges, const int* special_class, size_t length) +{ + length--; // Remove final 0x10000. + MOZ_ASSERT(special_class[length] == 0x10000); + MOZ_ASSERT(ranges.length() != 0); + MOZ_ASSERT(length != 0); + MOZ_ASSERT(special_class[0] != 0); + if (ranges.length() != (length >> 1) + 1) + return false; + CharacterRange range = ranges[0]; + if (range.from() != 0) + return false; + for (size_t i = 0; i < length; i += 2) { + if (special_class[i] != (range.to() + 1)) + return false; + range = ranges[(i >> 1) + 1]; + if (special_class[i+1] != range.from()) + return false; + } + if (range.to() != 0xffff) + return false; + return true; +} + +template +/* static */ void +CharacterRange::NegateUnicodeRanges(LifoAlloc* alloc, InfallibleVector** ranges, + RangeType full_range) +{ + typedef InfallibleVector RangeVector; + RangeVector* tmp_ranges = alloc->newInfallible(*alloc); + tmp_ranges->append(full_range); + RangeVector* result_ranges = alloc->newInfallible(*alloc); + + // Perform the following calculation: + // result_ranges = tmp_ranges - ranges + // with the following steps: + // result_ranges = tmp_ranges - ranges[0] + // SWAP(result_ranges, tmp_ranges) + // result_ranges = tmp_ranges - ranges[1] + // SWAP(result_ranges, tmp_ranges) + // ... + // result_ranges = tmp_ranges - ranges[N-1] + // SWAP(result_ranges, tmp_ranges) + // The last SWAP is just for simplicity of the loop. + for (size_t i = 0; i < (*ranges)->length(); i++) { + result_ranges->clear(); + + const RangeType& range = (**ranges)[i]; + for (size_t j = 0; j < tmp_ranges->length(); j++) { + const RangeType& tmpRange = (*tmp_ranges)[j]; + auto from1 = tmpRange.from(); + auto to1 = tmpRange.to(); + auto from2 = range.from(); + auto to2 = range.to(); + + if (from1 < from2) { + if (to1 < from2) { + result_ranges->append(tmpRange); + } else if (to1 <= to2) { + result_ranges->append(RangeType::Range(from1, from2 - 1)); + } else { + result_ranges->append(RangeType::Range(from1, from2 - 1)); + result_ranges->append(RangeType::Range(to2 + 1, to1)); + } + } else if (from1 <= to2) { + if (to1 > to2) + result_ranges->append(RangeType::Range(to2 + 1, to1)); + } else { + result_ranges->append(tmpRange); + } + } + + auto tmp = tmp_ranges; + tmp_ranges = result_ranges; + result_ranges = tmp; + } + + // After the loop, result is pointed at by tmp_ranges, instead of + // result_ranges. + *ranges = tmp_ranges; +} + +// Explicit specialization for NegateUnicodeRanges +template void CharacterRange::NegateUnicodeRanges(LifoAlloc* alloc, InfallibleVector** ranges, CharacterRange full_range); +template void CharacterRange::NegateUnicodeRanges(LifoAlloc* alloc, InfallibleVector** ranges, WideCharRange full_range); + +/* static */ bool +CharacterRange::IsCanonical(const CharacterRangeVector& ranges) +{ + int n = ranges.length(); + if (n <= 1) + return true; + + int max = ranges[0].to(); + for (int i = 1; i < n; i++) { + CharacterRange next_range = ranges[i]; + if (next_range.from() <= max + 1) + return false; + max = next_range.to(); + } + return true; +} + +/* static */ void +CharacterRange::Canonicalize(CharacterRangeVector& character_ranges) +{ + if (character_ranges.length() <= 1) return; + // Check whether ranges are already canonical (increasing, non-overlapping, + // non-adjacent). + int n = character_ranges.length(); + int max = character_ranges[0].to(); + int i = 1; + while (i < n) { + CharacterRange current = character_ranges[i]; + if (current.from() <= max + 1) { + break; + } + max = current.to(); + i++; + } + // Canonical until the i'th range. If that's all of them, we are done. + if (i == n) return; + + // The ranges at index i and forward are not canonicalized. Make them so by + // doing the equivalent of insertion sort (inserting each into the previous + // list, in order). + // Notice that inserting a range can reduce the number of ranges in the + // result due to combining of adjacent and overlapping ranges. + int read = i; // Range to insert. + size_t num_canonical = i; // Length of canonicalized part of list. + do { + num_canonical = InsertRangeInCanonicalList(character_ranges, + num_canonical, + character_ranges[read]); + read++; + } while (read < n); + + while (character_ranges.length() > num_canonical) + character_ranges.popBack(); + + MOZ_ASSERT(IsCanonical(character_ranges)); +} + +/* static */ int +CharacterRange::InsertRangeInCanonicalList(CharacterRangeVector& list, + int count, + CharacterRange insert) +{ + // Inserts a range into list[0..count[, which must be sorted + // by from value and non-overlapping and non-adjacent, using at most + // list[0..count] for the result. Returns the number of resulting + // canonicalized ranges. Inserting a range may collapse existing ranges into + // fewer ranges, so the return value can be anything in the range 1..count+1. + char16_t from = insert.from(); + char16_t to = insert.to(); + int start_pos = 0; + int end_pos = count; + for (int i = count - 1; i >= 0; i--) { + CharacterRange current = list[i]; + if (current.from() > to + 1) { + end_pos = i; + } else if (current.to() + 1 < from) { + start_pos = i + 1; + break; + } + } + + // Inserted range overlaps, or is adjacent to, ranges at positions + // [start_pos..end_pos[. Ranges before start_pos or at or after end_pos are + // not affected by the insertion. + // If start_pos == end_pos, the range must be inserted before start_pos. + // if start_pos < end_pos, the entire range from start_pos to end_pos + // must be merged with the insert range. + + if (start_pos == end_pos) { + // Insert between existing ranges at position start_pos. + if (start_pos < count) { + list.moveReplace(start_pos, start_pos + 1, count - start_pos); + } + list[start_pos] = insert; + return count + 1; + } + if (start_pos + 1 == end_pos) { + // Replace single existing range at position start_pos. + CharacterRange to_replace = list[start_pos]; + int new_from = Min(to_replace.from(), from); + int new_to = Max(to_replace.to(), to); + list[start_pos] = CharacterRange(new_from, new_to); + return count; + } + // Replace a number of existing ranges from start_pos to end_pos - 1. + // Move the remaining ranges down. + + int new_from = Min(list[start_pos].from(), from); + int new_to = Max(list[end_pos - 1].to(), to); + if (end_pos < count) { + list.moveReplace(end_pos, start_pos + 1, count - end_pos); + } + list[start_pos] = CharacterRange(new_from, new_to); + return count - (end_pos - start_pos) + 1; +} + +int +irregexp::GetCaseIndependentLetters(char16_t character, + bool ascii_subject, + bool unicode, + const char16_t* choices, + size_t choices_length, + char16_t* letters) +{ + size_t count = 0; + for (size_t i = 0; i < choices_length; i++) { + char16_t c = choices[i]; + + // Skip characters that can't appear in one byte strings. + if (!unicode && ascii_subject && c > kMaxOneByteCharCode) + continue; + + // Watch for duplicates. + bool found = false; + for (size_t j = 0; j < count; j++) { + if (letters[j] == c) { + found = true; + break; + } + } + if (found) + continue; + + letters[count++] = c; + } + + return count; +} + +int +irregexp::GetCaseIndependentLetters(char16_t character, + bool ascii_subject, + bool unicode, + char16_t* letters) +{ + if (unicode) { + const char16_t choices[] = { + character, + unicode::FoldCase(character), + unicode::ReverseFoldCase1(character), + unicode::ReverseFoldCase2(character), + unicode::ReverseFoldCase3(character), + }; + return GetCaseIndependentLetters(character, ascii_subject, unicode, + choices, ArrayLength(choices), letters); + } + + char16_t upper = unicode::ToUpperCase(character); + unicode::CodepointsWithSameUpperCase others(character); + char16_t other1 = others.other1(); + char16_t other2 = others.other2(); + char16_t other3 = others.other3(); + + // ES 2017 draft 996af87b7072b3c3dd2b1def856c66f456102215 21.2.4.2 + // step 3.g. + // The standard requires that non-ASCII characters cannot have ASCII + // character codes in their equivalence class, even though this + // situation occurs multiple times in the Unicode tables. + static const unsigned kMaxAsciiCharCode = 127; + if (upper <= kMaxAsciiCharCode) { + if (character > kMaxAsciiCharCode) { + // If Canonicalize(character) == character, all other characters + // should be ignored. + return GetCaseIndependentLetters(character, ascii_subject, unicode, + &character, 1, letters); + } + + if (other1 > kMaxAsciiCharCode) + other1 = character; + if (other2 > kMaxAsciiCharCode) + other2 = character; + if (other3 > kMaxAsciiCharCode) + other3 = character; + } + + const char16_t choices[] = { + character, + upper, + other1, + other2, + other3 + }; + return GetCaseIndependentLetters(character, ascii_subject, unicode, + choices, ArrayLength(choices), letters); +} diff --git a/js/src/irregexp/RegExpCharRanges.h b/js/src/irregexp/RegExpCharRanges.h new file mode 100644 index 0000000000..80736d961e --- /dev/null +++ b/js/src/irregexp/RegExpCharRanges.h @@ -0,0 +1,220 @@ +/* -*- Mode: C++; tab-width: 8; indent-tabs-mode: nil; c-basic-offset: 4 -*- */ + +// Copyright 2012 the V8 project authors. All rights reserved. +// Redistribution and use in source and binary forms, with or without +// modification, are permitted provided that the following conditions are +// met: +// +// * Redistributions of source code must retain the above copyright +// notice, this list of conditions and the following disclaimer. +// * Redistributions in binary form must reproduce the above +// copyright notice, this list of conditions and the following +// disclaimer in the documentation and/or other materials provided +// with the distribution. +// * Neither the name of Google Inc. nor the names of its +// contributors may be used to endorse or promote products derived +// from this software without specific prior written permission. +// +// THIS SOFTWARE IS PROVIDED BY THE COPYRIGHT HOLDERS AND CONTRIBUTORS +// "AS IS" AND ANY EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT +// LIMITED TO, THE IMPLIED WARRANTIES OF MERCHANTABILITY AND FITNESS FOR +// A PARTICULAR PURPOSE ARE DISCLAIMED. IN NO EVENT SHALL THE COPYRIGHT +// OWNER OR CONTRIBUTORS BE LIABLE FOR ANY DIRECT, INDIRECT, INCIDENTAL, +// SPECIAL, EXEMPLARY, OR CONSEQUENTIAL DAMAGES (INCLUDING, BUT NOT +// LIMITED TO, PROCUREMENT OF SUBSTITUTE GOODS OR SERVICES; LOSS OF USE, +// DATA, OR PROFITS; OR BUSINESS INTERRUPTION) HOWEVER CAUSED AND ON ANY +// THEORY OF LIABILITY, WHETHER IN CONTRACT, STRICT LIABILITY, OR TORT +// (INCLUDING NEGLIGENCE OR OTHERWISE) ARISING IN ANY WAY OUT OF THE USE +// OF THIS SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE. + +#ifndef V8_JSREGEXPCHARRANGES_H_ +#define V8_JSREGEXPCHARRANGES_H_ + +#include "irregexp/RegExpCharacters.h" +#include "irregexp/InfallibleVector.h" + +namespace js { + +namespace irregexp { + +// Characters parsed by RegExpParser can be either char16_t or kEndMarker. +typedef uint32_t widechar; + +static const int kMaxOneByteCharCode = 0xff; +static const int kMaxUtf16CodeUnit = 0xffff; +static const size_t kEcma262UnCanonicalizeMaxWidth = 4; +static const char16_t kNoCharClass = 0; + +static inline char16_t +MaximumCharacter(bool ascii) +{ + return ascii ? kMaxOneByteCharCode : kMaxUtf16CodeUnit; +} + + +// Returns the number of characters in the equivalence class, omitting those +// that cannot occur in the source string if it is a one byte string. +int +GetCaseIndependentLetters(char16_t character, + bool ascii_subject, + bool unicode, + const char16_t* choices, + size_t choices_length, + char16_t* letters); + +int +GetCaseIndependentLetters(char16_t character, + bool ascii_subject, + bool unicode, + char16_t* letters); + +class CharacterRange; +class WideCharRange; +typedef InfallibleVector CharacterRangeVector; +typedef InfallibleVector WideCharRangeVector; + +// Represents code units in the range from from_ to to_, both ends are +// inclusive. +class CharacterRange +{ + public: + // static methods for dealing with CharacterRangeVectors + + static void AddClass(const int* elmv, int elmc, CharacterRangeVector* ranges); + static void AddClassNegated(const int* elmv, int elmc, CharacterRangeVector* ranges); + static void AddClassEscape(LifoAlloc* alloc, char16_t type, CharacterRangeVector* ranges); + static void AddClassEscapeUnicode(LifoAlloc* alloc, char16_t type, + CharacterRangeVector* ranges, bool ignoreCase); + + // Adds a character or pre-defined character class to character ranges. + // If char_class is not kNoCharClass, it's interpreted as a class + // escape (i.e., 's' means whitespace, from '\s'). + static void AddCharOrEscape(LifoAlloc* alloc, CharacterRangeVector* ranges, + char16_t char_class, widechar c); + static void AddCharOrEscapeUnicode(LifoAlloc* alloc, + CharacterRangeVector* ranges, + CharacterRangeVector* lead_ranges, + CharacterRangeVector* trail_ranges, + WideCharRangeVector* wide_ranges, + char16_t char_class, + widechar c, + bool ignore_case); + // Simplified version of AddUnicodeRange for single characters + static void AddCharUnicode(LifoAlloc* alloc, + CharacterRangeVector* ranges, + CharacterRangeVector* lead_ranges, + CharacterRangeVector* trail_ranges, + WideCharRangeVector* wide_ranges, + widechar c); + static void AddUnicodeRange(LifoAlloc* alloc, + CharacterRangeVector* ranges, + CharacterRangeVector* lead_ranges, + CharacterRangeVector* trail_ranges, + WideCharRangeVector* wide_ranges, + widechar first, + widechar next); + + static bool RangesContainLatin1Equivalents(const CharacterRangeVector& ranges, bool unicode); + static bool CompareRanges(const CharacterRangeVector& ranges, const int* special_class, size_t length); + static bool CompareInverseRanges(const CharacterRangeVector& ranges, const int* special_class, size_t length); + + // Negate a vector of ranges by subtracting its ranges from a range + // encompassing the full range of possible values. + template + static void NegateUnicodeRanges(LifoAlloc* alloc, InfallibleVector** ranges, + RangeType full_range); + + // static methods for dealing with canonical CharacterRangeVectors + + // Whether a range list is in canonical form: Ranges ordered by from value, + // and ranges non-overlapping and non-adjacent. + static bool IsCanonical(const CharacterRangeVector& ranges); + + // Convert range list to canonical form. The characters covered by the ranges + // will still be the same, but no character is in more than one range, and + // adjacent ranges are merged. The resulting list may be shorter than the + // original, but cannot be longer. + static void Canonicalize(CharacterRangeVector& ranges); + + static int InsertRangeInCanonicalList(CharacterRangeVector& list, int count, CharacterRange insert); + + // Negate the contents of a character range in canonical form. + static void Negate(const LifoAlloc* alloc, + CharacterRangeVector src, + CharacterRangeVector* dst); + public: + CharacterRange() + : from_(0), to_(0) + {} + + CharacterRange(char16_t from, char16_t to) + : from_(from), to_(to) + {} + + static inline CharacterRange Singleton(char16_t value) { + return CharacterRange(value, value); + } + static inline CharacterRange Range(char16_t from, char16_t to) { + MOZ_ASSERT(from <= to); + return CharacterRange(from, to); + } + static inline CharacterRange Everything() { + return CharacterRange(0, kMaxUtf16CodeUnit); + } + static inline CharacterRange LeadSurrogate() { + return CharacterRange(unicode::LeadSurrogateMin, unicode::LeadSurrogateMax); + } + static inline CharacterRange TrailSurrogate() { + return CharacterRange(unicode::TrailSurrogateMin, unicode::TrailSurrogateMax); + } + bool Contains(char16_t i) { return from_ <= i && i <= to_; } + char16_t from() const { return from_; } + void set_from(char16_t value) { from_ = value; } + char16_t to() const { return to_; } + void set_to(char16_t value) { to_ = value; } + bool is_valid() { return from_ <= to_; } + bool IsEverything(char16_t max) { return from_ == 0 && to_ >= max; } + bool IsSingleton() { return (from_ == to_); } + + void AddCaseEquivalents(bool is_ascii, bool unicode, CharacterRangeVector* ranges); + private: + char16_t from_; + char16_t to_; +}; + + +class WideCharRange +{ + public: + WideCharRange() + : from_(0), to_(0) + {} + + WideCharRange(widechar from, widechar to) + : from_(from), to_(to) + {} + + static inline WideCharRange Singleton(widechar value) { + return WideCharRange(value, value); + } + static inline WideCharRange Range(widechar from, widechar to) { + MOZ_ASSERT(from <= to); + return WideCharRange(from, to); + } + static inline WideCharRange NonBMP() { + return WideCharRange(unicode::NonBMPMin, unicode::NonBMPMax); + } + + bool Contains(widechar i) const { return from_ <= i && i <= to_; } + widechar from() const { return from_; } + widechar to() const { return to_; } + + private: + widechar from_; + widechar to_; +}; + + +} } // namespace js::irregexp + +#endif // V8_JSREGEXPCHARRANGES_H_ diff --git a/js/src/irregexp/RegExpEngine.cpp b/js/src/irregexp/RegExpEngine.cpp index bf95b59dda..f3db7c1847 100644 --- a/js/src/irregexp/RegExpEngine.cpp +++ b/js/src/irregexp/RegExpEngine.cpp @@ -30,18 +30,14 @@ #include "irregexp/RegExpEngine.h" #include "irregexp/NativeRegExpMacroAssembler.h" -#include "irregexp/RegExpCharacters.h" +#include "irregexp/RegExpCharacters.h" #include "irregexp/RegExpMacroAssembler.h" #include "jit/ExecutableAllocator.h" #include "jit/JitCommon.h" -// Generated table -#include "irregexp/RegExpCharacters-inl.h" - using namespace js; using namespace js::irregexp; -using mozilla::ArrayLength; using mozilla::DebugOnly; using mozilla::Maybe; @@ -64,317 +60,6 @@ RegExpNode::RegExpNode(LifoAlloc* alloc) bm_info_[0] = bm_info_[1] = nullptr; } -static const int kMaxOneByteCharCode = 0xff; -static const int kMaxUtf16CodeUnit = 0xffff; - -static char16_t -MaximumCharacter(bool ascii) -{ - return ascii ? kMaxOneByteCharCode : kMaxUtf16CodeUnit; -} - -static void -AddClass(const int* elmv, int elmc, - CharacterRangeVector* ranges) -{ - elmc--; - MOZ_ASSERT(elmv[elmc] == 0x10000); - for (int i = 0; i < elmc; i += 2) { - MOZ_ASSERT(elmv[i] < elmv[i + 1]); - ranges->append(CharacterRange(elmv[i], elmv[i + 1] - 1)); - } -} - -static void -AddClassNegated(const int* elmv, - int elmc, - CharacterRangeVector* ranges) -{ - elmc--; - MOZ_ASSERT(elmv[elmc] == 0x10000); - MOZ_ASSERT(elmv[0] != 0x0000); - MOZ_ASSERT(elmv[elmc-1] != kMaxUtf16CodeUnit); - char16_t last = 0x0000; - for (int i = 0; i < elmc; i += 2) { - MOZ_ASSERT(last <= elmv[i] - 1); - MOZ_ASSERT(elmv[i] < elmv[i + 1]); - ranges->append(CharacterRange(last, elmv[i] - 1)); - last = elmv[i + 1]; - } - ranges->append(CharacterRange(last, kMaxUtf16CodeUnit)); -} - -void -CharacterRange::AddClassEscape(LifoAlloc* alloc, char16_t type, - CharacterRangeVector* ranges) -{ - switch (type) { - case 's': - AddClass(kSpaceRanges, kSpaceRangeCount, ranges); - break; - case 'S': - AddClassNegated(kSpaceRanges, kSpaceRangeCount, ranges); - break; - case 'w': - AddClass(kWordRanges, kWordRangeCount, ranges); - break; - case 'W': - AddClassNegated(kWordRanges, kWordRangeCount, ranges); - break; - case 'd': - AddClass(kDigitRanges, kDigitRangeCount, ranges); - break; - case 'D': - AddClassNegated(kDigitRanges, kDigitRangeCount, ranges); - break; - case '.': - AddClassNegated(kLineTerminatorRanges, kLineTerminatorRangeCount, ranges); - break; - // This is not a character range as defined by the spec but a - // convenient shorthand for a character class that matches any - // character. - case '*': - ranges->append(CharacterRange::Everything()); - break; - // This is the set of characters matched by the $ and ^ symbols - // in multiline mode. - case 'n': - AddClass(kLineTerminatorRanges, kLineTerminatorRangeCount, ranges); - break; - default: - MOZ_CRASH("Bad character class escape"); - } -} - -// Add class escape, excluding surrogate pair range. -void -CharacterRange::AddClassEscapeUnicode(LifoAlloc* alloc, char16_t type, - CharacterRangeVector* ranges, bool ignore_case) -{ - switch (type) { - case 's': - case 'd': - return AddClassEscape(alloc, type, ranges); - break; - case 'S': - AddClassNegated(kSpaceAndSurrogateRanges, kSpaceAndSurrogateRangeCount, ranges); - break; - case 'w': - if (ignore_case) - AddClass(kIgnoreCaseWordRanges, kIgnoreCaseWordRangeCount, ranges); - else - AddClassEscape(alloc, type, ranges); - break; - case 'W': - if (ignore_case) { - AddClass(kNegatedIgnoreCaseWordAndSurrogateRanges, - kNegatedIgnoreCaseWordAndSurrogateRangeCount, ranges); - } else { - AddClassNegated(kWordAndSurrogateRanges, kWordAndSurrogateRangeCount, ranges); - } - break; - case 'D': - AddClassNegated(kDigitAndSurrogateRanges, kDigitAndSurrogateRangeCount, ranges); - break; - default: - MOZ_CRASH("Bad type!"); - } -} - -static bool -RangesContainLatin1Equivalents(const CharacterRangeVector& ranges, bool unicode) -{ - for (size_t i = 0; i < ranges.length(); i++) { - // TODO(dcarney): this could be a lot more efficient. - if (RangeContainsLatin1Equivalents(ranges[i], unicode)) - return true; - } - return false; -} - -static const size_t kEcma262UnCanonicalizeMaxWidth = 4; - -// Returns the number of characters in the equivalence class, omitting those -// that cannot occur in the source string if it is a one byte string. -static int -GetCaseIndependentLetters(char16_t character, - bool ascii_subject, - bool unicode, - const char16_t* choices, - size_t choices_length, - char16_t* letters) -{ - size_t count = 0; - for (size_t i = 0; i < choices_length; i++) { - char16_t c = choices[i]; - - // Skip characters that can't appear in one byte strings. - if (!unicode && ascii_subject && c > kMaxOneByteCharCode) - continue; - - // Watch for duplicates. - bool found = false; - for (size_t j = 0; j < count; j++) { - if (letters[j] == c) { - found = true; - break; - } - } - if (found) - continue; - - letters[count++] = c; - } - - return count; -} - -static int -GetCaseIndependentLetters(char16_t character, - bool ascii_subject, - bool unicode, - char16_t* letters) -{ - if (unicode) { - const char16_t choices[] = { - character, - unicode::FoldCase(character), - unicode::ReverseFoldCase1(character), - unicode::ReverseFoldCase2(character), - unicode::ReverseFoldCase3(character), - }; - return GetCaseIndependentLetters(character, ascii_subject, unicode, - choices, ArrayLength(choices), letters); - } - - char16_t upper = unicode::ToUpperCase(character); - unicode::CodepointsWithSameUpperCase others(character); - char16_t other1 = others.other1(); - char16_t other2 = others.other2(); - char16_t other3 = others.other3(); - - // ES 2017 draft 996af87b7072b3c3dd2b1def856c66f456102215 21.2.4.2 - // step 3.g. - // The standard requires that non-ASCII characters cannot have ASCII - // character codes in their equivalence class, even though this - // situation occurs multiple times in the Unicode tables. - static const unsigned kMaxAsciiCharCode = 127; - if (upper <= kMaxAsciiCharCode) { - if (character > kMaxAsciiCharCode) { - // If Canonicalize(character) == character, all other characters - // should be ignored. - return GetCaseIndependentLetters(character, ascii_subject, unicode, - &character, 1, letters); - } - - if (other1 > kMaxAsciiCharCode) - other1 = character; - if (other2 > kMaxAsciiCharCode) - other2 = character; - if (other3 > kMaxAsciiCharCode) - other3 = character; - } - - const char16_t choices[] = { - character, - upper, - other1, - other2, - other3 - }; - return GetCaseIndependentLetters(character, ascii_subject, unicode, - choices, ArrayLength(choices), letters); -} - -void -CharacterRange::AddCaseEquivalents(bool is_ascii, bool unicode, CharacterRangeVector* ranges) -{ - char16_t bottom = from(); - char16_t top = to(); - - if (is_ascii && !RangeContainsLatin1Equivalents(*this, unicode)) { - if (bottom > kMaxOneByteCharCode) - return; - if (top > kMaxOneByteCharCode) - top = kMaxOneByteCharCode; - } - - for (char16_t c = bottom;; c++) { - char16_t chars[kEcma262UnCanonicalizeMaxWidth]; - size_t length = GetCaseIndependentLetters(c, is_ascii, unicode, chars); - - for (size_t i = 0; i < length; i++) { - char16_t other = chars[i]; - if (other == c) - continue; - - // Try to combine with an existing range. - bool found = false; - for (size_t i = 0; i < ranges->length(); i++) { - CharacterRange& range = (*ranges)[i]; - if (range.Contains(other)) { - found = true; - break; - } else if (other == range.from() - 1) { - range.set_from(other); - found = true; - break; - } else if (other == range.to() + 1) { - range.set_to(other); - found = true; - break; - } - } - - if (!found) - ranges->append(CharacterRange::Singleton(other)); - } - - if (c == top) - break; - } -} - -static bool -CompareInverseRanges(const CharacterRangeVector& ranges, const int* special_class, size_t length) -{ - length--; // Remove final 0x10000. - MOZ_ASSERT(special_class[length] == 0x10000); - MOZ_ASSERT(ranges.length() != 0); - MOZ_ASSERT(length != 0); - MOZ_ASSERT(special_class[0] != 0); - if (ranges.length() != (length >> 1) + 1) - return false; - CharacterRange range = ranges[0]; - if (range.from() != 0) - return false; - for (size_t i = 0; i < length; i += 2) { - if (special_class[i] != (range.to() + 1)) - return false; - range = ranges[(i >> 1) + 1]; - if (special_class[i+1] != range.from()) - return false; - } - if (range.to() != 0xffff) - return false; - return true; -} - -static bool -CompareRanges(const CharacterRangeVector& ranges, const int* special_class, size_t length) -{ - length--; // Remove final 0x10000. - MOZ_ASSERT(special_class[length] == 0x10000); - if (ranges.length() * 2 != length) - return false; - for (size_t i = 0; i < length; i += 2) { - CharacterRange range = ranges[i >> 1]; - if (range.from() != special_class[i] || range.to() != special_class[i + 1] - 1) - return false; - } - return true; -} - bool RegExpCharacterClass::is_standard(LifoAlloc* alloc) { @@ -384,168 +69,37 @@ RegExpCharacterClass::is_standard(LifoAlloc* alloc) return false; if (set_.is_standard()) return true; - if (CompareRanges(set_.ranges(alloc), kSpaceRanges, kSpaceRangeCount)) { + if (CharacterRange::CompareRanges(set_.ranges(alloc), kSpaceRanges, kSpaceRangeCount)) { set_.set_standard_set_type('s'); return true; } - if (CompareInverseRanges(set_.ranges(alloc), kSpaceRanges, kSpaceRangeCount)) { + if (CharacterRange::CompareInverseRanges(set_.ranges(alloc), kSpaceRanges, kSpaceRangeCount)) { set_.set_standard_set_type('S'); return true; } - if (CompareInverseRanges(set_.ranges(alloc), + if (CharacterRange::CompareInverseRanges(set_.ranges(alloc), kLineTerminatorRanges, kLineTerminatorRangeCount)) { set_.set_standard_set_type('.'); return true; } - if (CompareRanges(set_.ranges(alloc), + if (CharacterRange::CompareRanges(set_.ranges(alloc), kLineTerminatorRanges, kLineTerminatorRangeCount)) { set_.set_standard_set_type('n'); return true; } - if (CompareRanges(set_.ranges(alloc), kWordRanges, kWordRangeCount)) { + if (CharacterRange::CompareRanges(set_.ranges(alloc), kWordRanges, kWordRangeCount)) { set_.set_standard_set_type('w'); return true; } - if (CompareInverseRanges(set_.ranges(alloc), kWordRanges, kWordRangeCount)) { + if (CharacterRange::CompareInverseRanges(set_.ranges(alloc), kWordRanges, kWordRangeCount)) { set_.set_standard_set_type('W'); return true; } return false; } -bool -CharacterRange::IsCanonical(const CharacterRangeVector& ranges) -{ - int n = ranges.length(); - if (n <= 1) - return true; - - int max = ranges[0].to(); - for (int i = 1; i < n; i++) { - CharacterRange next_range = ranges[i]; - if (next_range.from() <= max + 1) - return false; - max = next_range.to(); - } - return true; -} - -// Move a number of elements in a zonelist to another position -// in the same list. Handles overlapping source and target areas. -static -void MoveRanges(CharacterRangeVector& list, int from, int to, int count) -{ - // Ranges are potentially overlapping. - if (from < to) { - for (int i = count - 1; i >= 0; i--) - list[to + i] = list[from + i]; - } else { - for (int i = 0; i < count; i++) - list[to + i] = list[from + i]; - } -} - -static int -InsertRangeInCanonicalList(CharacterRangeVector& list, - int count, - CharacterRange insert) -{ - // Inserts a range into list[0..count[, which must be sorted - // by from value and non-overlapping and non-adjacent, using at most - // list[0..count] for the result. Returns the number of resulting - // canonicalized ranges. Inserting a range may collapse existing ranges into - // fewer ranges, so the return value can be anything in the range 1..count+1. - char16_t from = insert.from(); - char16_t to = insert.to(); - int start_pos = 0; - int end_pos = count; - for (int i = count - 1; i >= 0; i--) { - CharacterRange current = list[i]; - if (current.from() > to + 1) { - end_pos = i; - } else if (current.to() + 1 < from) { - start_pos = i + 1; - break; - } - } - - // Inserted range overlaps, or is adjacent to, ranges at positions - // [start_pos..end_pos[. Ranges before start_pos or at or after end_pos are - // not affected by the insertion. - // If start_pos == end_pos, the range must be inserted before start_pos. - // if start_pos < end_pos, the entire range from start_pos to end_pos - // must be merged with the insert range. - - if (start_pos == end_pos) { - // Insert between existing ranges at position start_pos. - if (start_pos < count) { - MoveRanges(list, start_pos, start_pos + 1, count - start_pos); - } - list[start_pos] = insert; - return count + 1; - } - if (start_pos + 1 == end_pos) { - // Replace single existing range at position start_pos. - CharacterRange to_replace = list[start_pos]; - int new_from = Min(to_replace.from(), from); - int new_to = Max(to_replace.to(), to); - list[start_pos] = CharacterRange(new_from, new_to); - return count; - } - // Replace a number of existing ranges from start_pos to end_pos - 1. - // Move the remaining ranges down. - - int new_from = Min(list[start_pos].from(), from); - int new_to = Max(list[end_pos - 1].to(), to); - if (end_pos < count) { - MoveRanges(list, end_pos, start_pos + 1, count - end_pos); - } - list[start_pos] = CharacterRange(new_from, new_to); - return count - (end_pos - start_pos) + 1; -} - -void -CharacterRange::Canonicalize(CharacterRangeVector& character_ranges) -{ - if (character_ranges.length() <= 1) return; - // Check whether ranges are already canonical (increasing, non-overlapping, - // non-adjacent). - int n = character_ranges.length(); - int max = character_ranges[0].to(); - int i = 1; - while (i < n) { - CharacterRange current = character_ranges[i]; - if (current.from() <= max + 1) { - break; - } - max = current.to(); - i++; - } - // Canonical until the i'th range. If that's all of them, we are done. - if (i == n) return; - - // The ranges at index i and forward are not canonicalized. Make them so by - // doing the equivalent of insertion sort (inserting each into the previous - // list, in order). - // Notice that inserting a range can reduce the number of ranges in the - // result due to combining of adjacent and overlapping ranges. - int read = i; // Range to insert. - size_t num_canonical = i; // Length of canonicalized part of list. - do { - num_canonical = InsertRangeInCanonicalList(character_ranges, - num_canonical, - character_ranges[read]); - read++; - } while (read < n); - - while (character_ranges.length() > num_canonical) - character_ranges.popBack(); - - MOZ_ASSERT(CharacterRange::IsCanonical(character_ranges)); -} - // ------------------------------------------------------------------- // SeqRegExpNode @@ -790,7 +344,7 @@ TextNode::FilterASCII(int depth, bool ignore_case, bool unicode) ranges[0].to() >= kMaxOneByteCharCode) { // This will be handled in a later filter. - if (ignore_case && RangesContainLatin1Equivalents(ranges, unicode)) + if (ignore_case && CharacterRange::RangesContainLatin1Equivalents(ranges, unicode)) continue; return set_replacement(nullptr); } @@ -799,7 +353,7 @@ TextNode::FilterASCII(int depth, bool ignore_case, bool unicode) ranges[0].from() > kMaxOneByteCharCode) { // This will be handled in a later filter. - if (ignore_case && RangesContainLatin1Equivalents(ranges, unicode)) + if (ignore_case && CharacterRange::RangesContainLatin1Equivalents(ranges, unicode)) continue; return set_replacement(nullptr); } diff --git a/js/src/irregexp/RegExpEngine.h b/js/src/irregexp/RegExpEngine.h index 1b011458cd..22e9d944a4 100644 --- a/js/src/irregexp/RegExpEngine.h +++ b/js/src/irregexp/RegExpEngine.h @@ -34,6 +34,9 @@ #include "ds/SplayTree.h" #include "jit/Label.h" + +#include "irregexp/InfallibleVector.h" +#include "irregexp/RegExpCharRanges.h" #include "vm/RegExpObject.h" namespace js { @@ -142,75 +145,6 @@ InterpretCode(JSContext* cx, const uint8_t* byteCode, const CharT* chars, size_t FOR_EACH_REG_EXP_TREE_TYPE(FORWARD_DECLARE) #undef FORWARD_DECLARE -class CharacterRange; -typedef InfallibleVector CharacterRangeVector; - -// Represents code units in the range from from_ to to_, both ends are -// inclusive. -class CharacterRange -{ - public: - CharacterRange() - : from_(0), to_(0) - {} - - CharacterRange(char16_t from, char16_t to) - : from_(from), to_(to) - {} - - static void AddClassEscape(LifoAlloc* alloc, char16_t type, CharacterRangeVector* ranges); - static void AddClassEscapeUnicode(LifoAlloc* alloc, char16_t type, - CharacterRangeVector* ranges, bool ignoreCase); - - static inline CharacterRange Singleton(char16_t value) { - return CharacterRange(value, value); - } - static inline CharacterRange Range(char16_t from, char16_t to) { - MOZ_ASSERT(from <= to); - return CharacterRange(from, to); - } - static inline CharacterRange Everything() { - return CharacterRange(0, 0xFFFF); - } - bool Contains(char16_t i) { return from_ <= i && i <= to_; } - char16_t from() const { return from_; } - void set_from(char16_t value) { from_ = value; } - char16_t to() const { return to_; } - void set_to(char16_t value) { to_ = value; } - bool is_valid() { return from_ <= to_; } - bool IsEverything(char16_t max) { return from_ == 0 && to_ >= max; } - bool IsSingleton() { return (from_ == to_); } - void AddCaseEquivalents(bool is_ascii, bool unicode, CharacterRangeVector* ranges); - - static void Split(const LifoAlloc* alloc, - CharacterRangeVector base, - const Vector& overlay, - CharacterRangeVector* included, - CharacterRangeVector* excluded); - - // Whether a range list is in canonical form: Ranges ordered by from value, - // and ranges non-overlapping and non-adjacent. - static bool IsCanonical(const CharacterRangeVector& ranges); - - // Convert range list to canonical form. The characters covered by the ranges - // will still be the same, but no character is in more than one range, and - // adjacent ranges are merged. The resulting list may be shorter than the - // original, but cannot be longer. - static void Canonicalize(CharacterRangeVector& ranges); - - // Negate the contents of a character range in canonical form. - static void Negate(const LifoAlloc* alloc, - CharacterRangeVector src, - CharacterRangeVector* dst); - - static const int kStartMarker = (1 << 24); - static const int kPayloadMask = (1 << 24) - 1; - - private: - char16_t from_; - char16_t to_; -}; - // A set of unsigned integers that behaves especially well on small // integers (< 32). class OutSet diff --git a/js/src/irregexp/RegExpParser.cpp b/js/src/irregexp/RegExpParser.cpp index c46b8cf8ec..014951e320 100644 --- a/js/src/irregexp/RegExpParser.cpp +++ b/js/src/irregexp/RegExpParser.cpp @@ -670,215 +670,6 @@ RegExpParser::ParseClassCharacterEscape(widechar* code) return true; } -class WideCharRange -{ - public: - WideCharRange() - : from_(0), to_(0) - {} - - WideCharRange(widechar from, widechar to) - : from_(from), to_(to) - {} - - static inline WideCharRange Singleton(widechar value) { - return WideCharRange(value, value); - } - static inline WideCharRange Range(widechar from, widechar to) { - MOZ_ASSERT(from <= to); - return WideCharRange(from, to); - } - - bool Contains(widechar i) const { return from_ <= i && i <= to_; } - widechar from() const { return from_; } - widechar to() const { return to_; } - - private: - widechar from_; - widechar to_; -}; - -typedef InfallibleVector WideCharRangeVector; - -static inline CharacterRange -LeadSurrogateRange() -{ - return CharacterRange::Range(unicode::LeadSurrogateMin, unicode::LeadSurrogateMax); -} - -static inline CharacterRange -TrailSurrogateRange() -{ - return CharacterRange::Range(unicode::TrailSurrogateMin, unicode::TrailSurrogateMax); -} - -static inline WideCharRange -NonBMPRange() -{ - return WideCharRange::Range(unicode::NonBMPMin, unicode::NonBMPMax); -} - -static const char16_t kNoCharClass = 0; - -// Adds a character or pre-defined character class to character ranges. -// If char_class is not kInvalidClass, it's interpreted as a class -// escape (i.e., 's' means whitespace, from '\s'). -static inline void -AddCharOrEscape(LifoAlloc* alloc, - CharacterRangeVector* ranges, - char16_t char_class, - widechar c) -{ - if (char_class != kNoCharClass) - CharacterRange::AddClassEscape(alloc, char_class, ranges); - else - ranges->append(CharacterRange::Singleton(c)); -} - -static inline void -AddCharOrEscapeUnicode(LifoAlloc* alloc, - CharacterRangeVector* ranges, - CharacterRangeVector* lead_ranges, - CharacterRangeVector* trail_ranges, - WideCharRangeVector* wide_ranges, - char16_t char_class, - widechar c, - bool ignore_case) -{ - if (char_class != kNoCharClass) { - CharacterRange::AddClassEscapeUnicode(alloc, char_class, ranges, ignore_case); - switch (char_class) { - case 'S': - case 'W': - case 'D': - lead_ranges->append(LeadSurrogateRange()); - trail_ranges->append(TrailSurrogateRange()); - wide_ranges->append(NonBMPRange()); - break; - case '.': - MOZ_CRASH("Bad char_class!"); - } - return; - } - - if (unicode::IsLeadSurrogate(c)) - lead_ranges->append(CharacterRange::Singleton(c)); - else if (unicode::IsTrailSurrogate(c)) - trail_ranges->append(CharacterRange::Singleton(c)); - else if (c >= unicode::NonBMPMin) - wide_ranges->append(WideCharRange::Singleton(c)); - else - ranges->append(CharacterRange::Singleton(c)); -} - -static inline void -AddUnicodeRange(LifoAlloc* alloc, - CharacterRangeVector* ranges, - CharacterRangeVector* lead_ranges, - CharacterRangeVector* trail_ranges, - WideCharRangeVector* wide_ranges, - widechar first, - widechar next) -{ - MOZ_ASSERT(first <= next); - if (first < unicode::LeadSurrogateMin) { - if (next < unicode::LeadSurrogateMin) { - ranges->append(CharacterRange::Range(first, next)); - return; - } - ranges->append(CharacterRange::Range(first, unicode::LeadSurrogateMin - 1)); - first = unicode::LeadSurrogateMin; - } - if (first <= unicode::LeadSurrogateMax) { - if (next <= unicode::LeadSurrogateMax) { - lead_ranges->append(CharacterRange::Range(first, next)); - return; - } - lead_ranges->append(CharacterRange::Range(first, unicode::LeadSurrogateMax)); - first = unicode::LeadSurrogateMax + 1; - } - MOZ_ASSERT(unicode::LeadSurrogateMax + 1 == unicode::TrailSurrogateMin); - if (first <= unicode::TrailSurrogateMax) { - if (next <= unicode::TrailSurrogateMax) { - trail_ranges->append(CharacterRange::Range(first, next)); - return; - } - trail_ranges->append(CharacterRange::Range(first, unicode::TrailSurrogateMax)); - first = unicode::TrailSurrogateMax + 1; - } - if (first <= unicode::UTF16Max) { - if (next <= unicode::UTF16Max) { - ranges->append(CharacterRange::Range(first, next)); - return; - } - ranges->append(CharacterRange::Range(first, unicode::UTF16Max)); - first = unicode::NonBMPMin; - } - MOZ_ASSERT(unicode::UTF16Max + 1 == unicode::NonBMPMin); - wide_ranges->append(WideCharRange::Range(first, next)); -} - -// Negate a vector of ranges by subtracting its ranges from a range -// encompassing the full range of possible values. -template -static inline void -NegateUnicodeRanges(LifoAlloc* alloc, InfallibleVector** ranges, - RangeType full_range) -{ - typedef InfallibleVector RangeVector; - RangeVector* tmp_ranges = alloc->newInfallible(*alloc); - tmp_ranges->append(full_range); - RangeVector* result_ranges = alloc->newInfallible(*alloc); - - // Perform the following calculation: - // result_ranges = tmp_ranges - ranges - // with the following steps: - // result_ranges = tmp_ranges - ranges[0] - // SWAP(result_ranges, tmp_ranges) - // result_ranges = tmp_ranges - ranges[1] - // SWAP(result_ranges, tmp_ranges) - // ... - // result_ranges = tmp_ranges - ranges[N-1] - // SWAP(result_ranges, tmp_ranges) - // The last SWAP is just for simplicity of the loop. - for (size_t i = 0; i < (*ranges)->length(); i++) { - result_ranges->clear(); - - const RangeType& range = (**ranges)[i]; - for (size_t j = 0; j < tmp_ranges->length(); j++) { - const RangeType& tmpRange = (*tmp_ranges)[j]; - auto from1 = tmpRange.from(); - auto to1 = tmpRange.to(); - auto from2 = range.from(); - auto to2 = range.to(); - - if (from1 < from2) { - if (to1 < from2) { - result_ranges->append(tmpRange); - } else if (to1 <= to2) { - result_ranges->append(RangeType::Range(from1, from2 - 1)); - } else { - result_ranges->append(RangeType::Range(from1, from2 - 1)); - result_ranges->append(RangeType::Range(to2 + 1, to1)); - } - } else if (from1 <= to2) { - if (to1 > to2) - result_ranges->append(RangeType::Range(to2 + 1, to1)); - } else { - result_ranges->append(tmpRange); - } - } - - auto tmp = tmp_ranges; - tmp_ranges = result_ranges; - result_ranges = tmp; - } - - // After the loop, result is pointed at by tmp_ranges, instead of - // result_ranges. - *ranges = tmp_ranges; -} - static bool WideCharRangesContain(WideCharRangeVector* wide_ranges, widechar c) { @@ -948,9 +739,9 @@ UnicodeRangesAtom(LifoAlloc* alloc, } if (is_negated) { - NegateUnicodeRanges(alloc, &lead_ranges, LeadSurrogateRange()); - NegateUnicodeRanges(alloc, &trail_ranges, TrailSurrogateRange()); - NegateUnicodeRanges(alloc, &wide_ranges, NonBMPRange()); + CharacterRange::NegateUnicodeRanges(alloc, &lead_ranges, CharacterRange::LeadSurrogate()); + CharacterRange::NegateUnicodeRanges(alloc, &trail_ranges, CharacterRange::TrailSurrogate()); + CharacterRange::NegateUnicodeRanges(alloc, &wide_ranges, WideCharRange::NonBMP()); } RegExpBuilder* builder = alloc->newInfallible(alloc); @@ -958,8 +749,8 @@ UnicodeRangesAtom(LifoAlloc* alloc, bool added = false; if (is_negated) { - ranges->append(LeadSurrogateRange()); - ranges->append(TrailSurrogateRange()); + ranges->append(CharacterRange::LeadSurrogate()); + ranges->append(CharacterRange::TrailSurrogate()); } if (ranges->length() > 0) { builder->AddAtom(alloc->newInfallible(ranges, is_negated)); @@ -1077,9 +868,9 @@ RegExpParser::ParseCharacterClass() } while (has_more() && current() != ']') { - char16_t char_class = kNoCharClass; - widechar first = 0; - if (!ParseClassAtom(&char_class, &first)) + char16_t char_class_1 = kNoCharClass; + widechar char_1 = 0; + if (!ParseClassEscape(&char_class_1, &char_1, ranges, lead_ranges, trail_ranges, wide_ranges)) return nullptr; if (current() == '-') { Advance(); @@ -1088,41 +879,49 @@ RegExpParser::ParseCharacterClass() // following code report an error. break; } else if (current() == ']') { - if (unicode_) { - AddCharOrEscapeUnicode(alloc, ranges, lead_ranges, trail_ranges, wide_ranges, - char_class, first, ignore_case_); - } else { - AddCharOrEscape(alloc, ranges, char_class, first); + // if the last item was not a class, add it verbatim. + if (char_class_1 == kNoCharClass) { + if (unicode_) { + CharacterRange::AddCharUnicode(alloc, ranges, lead_ranges, trail_ranges, wide_ranges, char_1); + } else { + ranges->append(CharacterRange::Singleton(char_1)); + } } + // Hyphen at the end of a class. Treat the '-' verbatim. ranges->append(CharacterRange::Singleton('-')); break; } char16_t char_class_2 = kNoCharClass; - widechar next = 0; - if (!ParseClassAtom(&char_class_2, &next)) + widechar char_2 = 0; + if (!ParseClassEscape(&char_class_2, &char_2, ranges, lead_ranges, trail_ranges, wide_ranges)) return nullptr; - if (char_class != kNoCharClass || char_class_2 != kNoCharClass) { + if (char_class_1 != kNoCharClass || char_class_2 != kNoCharClass) { if (unicode_) return ReportError(JSMSG_RANGE_WITH_CLASS_ESCAPE); - // Either end is an escaped character class. Treat the '-' verbatim. - AddCharOrEscape(alloc, ranges, char_class, first); + // Either end is an escaped character class. Treat the '-' verbatim and add the + // character that isn't a class + if (char_class_1 == kNoCharClass) + ranges->append(CharacterRange::Singleton(char_1)); ranges->append(CharacterRange::Singleton('-')); - AddCharOrEscape(alloc, ranges, char_class_2, next); + if (char_class_1 == kNoCharClass) + ranges->append(CharacterRange::Singleton(char_2)); continue; } - if (first > next) + if (char_1 > char_2) return ReportError(JSMSG_BAD_CLASS_RANGE); if (unicode_) - AddUnicodeRange(alloc, ranges, lead_ranges, trail_ranges,wide_ranges, first, next); + CharacterRange::AddUnicodeRange(alloc, ranges, lead_ranges, trail_ranges, wide_ranges, char_1, char_2); else - ranges->append(CharacterRange::Range(first, next)); + ranges->append(CharacterRange::Range(char_1, char_2)); } else { - if (unicode_) { - AddCharOrEscapeUnicode(alloc, ranges, lead_ranges, trail_ranges, wide_ranges, - char_class, first, ignore_case_); - } else { - AddCharOrEscape(alloc, ranges, char_class, first); + // if the last item was not a class, add it verbatim. + if (char_class_1 == kNoCharClass) { + if (unicode_) { + CharacterRange::AddCharUnicode(alloc, ranges, lead_ranges, trail_ranges, wide_ranges, char_1); + } else { + ranges->append(CharacterRange::Singleton(char_1)); + } } } } @@ -1135,22 +934,26 @@ RegExpParser::ParseCharacterClass() is_negated = !is_negated; } return alloc->newInfallible(ranges, is_negated); - } + } else { + if (!is_negated && ranges->length() == 0 && lead_ranges->length() == 0 && + trail_ranges->length() == 0 && wide_ranges->length() == 0) + { + ranges->append(CharacterRange::Everything()); + return alloc->newInfallible(ranges, true); + } - if (!is_negated && ranges->length() == 0 && lead_ranges->length() == 0 && - trail_ranges->length() == 0 && wide_ranges->length() == 0) - { - ranges->append(CharacterRange::Everything()); - return alloc->newInfallible(ranges, true); + return UnicodeRangesAtom(alloc, ranges, lead_ranges, trail_ranges, wide_ranges, is_negated, + ignore_case_); } - - return UnicodeRangesAtom(alloc, ranges, lead_ranges, trail_ranges, wide_ranges, is_negated, - ignore_case_); } template bool -RegExpParser::ParseClassAtom(char16_t* char_class, widechar* value) +RegExpParser::ParseClassEscape(char16_t* char_class, widechar *value, + CharacterRangeVector* ranges, + CharacterRangeVector* lead_ranges, + CharacterRangeVector* trail_ranges, + WideCharRangeVector* wide_ranges) { MOZ_ASSERT(*char_class == kNoCharClass); widechar first = current(); @@ -1159,6 +962,13 @@ RegExpParser::ParseClassAtom(char16_t* char_class, widechar* value) case 'w': case 'W': case 'd': case 'D': case 's': case 'S': { *char_class = Next(); Advance(2); + // add character range to ranges immediately + if (unicode_) { + CharacterRange::AddCharOrEscapeUnicode(alloc, ranges, lead_ranges, trail_ranges, wide_ranges, + *char_class, 0, ignore_case_); + } else { + CharacterRange::AddCharOrEscape(alloc, ranges, *char_class, 0); + } return true; } case kEndMarker: @@ -1720,8 +1530,8 @@ UnicodeCharacterClassEscapeAtom(LifoAlloc* alloc, char16_t char_class, bool igno CharacterRangeVector* lead_ranges = alloc->newInfallible(*alloc); CharacterRangeVector* trail_ranges = alloc->newInfallible(*alloc); WideCharRangeVector* wide_ranges = alloc->newInfallible(*alloc); - AddCharOrEscapeUnicode(alloc, ranges, lead_ranges, trail_ranges, wide_ranges, char_class, 0, - ignore_case); + CharacterRange::AddCharOrEscapeUnicode(alloc, ranges, lead_ranges, trail_ranges, wide_ranges, + char_class, 0, ignore_case); return UnicodeRangesAtom(alloc, ranges, lead_ranges, trail_ranges, wide_ranges, false, false); } @@ -1951,25 +1761,23 @@ RegExpParser::ParseDisjunction() // CharacterClassEscape :: one of // d D s S w W case 'D': case 'S': case 'W': - if (unicode_) { - Advance(); - builder->AddAtom(UnicodeCharacterClassEscapeAtom(alloc, current(), - ignore_case_)); - Advance(); - break; - } - MOZ_FALLTHROUGH; case 'd': case 's': case 'w': { widechar c = Next(); + bool negated = c <= 'Z'; Advance(2); - CharacterRangeVector* ranges = - alloc->newInfallible(*alloc); - if (unicode_) - CharacterRange::AddClassEscapeUnicode(alloc, c, ranges, ignore_case_); - else - CharacterRange::AddClassEscape(alloc, c, ranges); - RegExpTree* atom = alloc->newInfallible(ranges, false); - builder->AddAtom(atom); + if (unicode_ && negated) { + // must generate negative lookarounds for lone surrogates, done by AddCharOrEscapeUnicode + builder->AddAtom(UnicodeCharacterClassEscapeAtom(alloc, c, ignore_case_)); + } else { + // only match positive ranges + CharacterRangeVector* ranges = alloc->newInfallible(*alloc); + if (unicode_) + CharacterRange::AddClassEscapeUnicode(alloc, c, ranges, ignore_case_); + else + CharacterRange::AddClassEscape(alloc, c, ranges); + RegExpTree* atom = alloc->newInfallible(ranges, false); + builder->AddAtom(atom); + } break; } case '1': case '2': case '3': case '4': case '5': case '6': diff --git a/js/src/irregexp/RegExpParser.h b/js/src/irregexp/RegExpParser.h index 48236530ab..01d05caf78 100644 --- a/js/src/irregexp/RegExpParser.h +++ b/js/src/irregexp/RegExpParser.h @@ -133,9 +133,6 @@ class BufferedVector }; -// Characters parsed by RegExpParser can be either char16_t or kEndMarker. -typedef uint32_t widechar; - // Accumulates RegExp atoms and assertions into lists of terms and alternatives. class RegExpBuilder { @@ -215,7 +212,14 @@ class RegExpParser // can be reparsed. bool ParseBackReferenceIndex(int* index_out); - bool ParseClassAtom(char16_t* char_class, widechar *value); + // Parse a thing inside a character class. Either add escaped class to the range and return + // the matched range as |char_class|, or return a single character as |value| + // Unicode ranges can be null if not in Unicode mode + bool ParseClassEscape(char16_t* char_class, widechar *value, + CharacterRangeVector* ranges, + CharacterRangeVector* lead_ranges, + CharacterRangeVector* trail_ranges, + WideCharRangeVector* wide_ranges); RegExpTree* ReportError(unsigned errorNumber, const char* param = nullptr); void Advance(); void Advance(int dist) { diff --git a/js/src/moz.build b/js/src/moz.build index 642dd7d911..5ac4fcd669 100644 --- a/js/src/moz.build +++ b/js/src/moz.build @@ -153,6 +153,7 @@ UNIFIED_SOURCES += [ 'irregexp/NativeRegExpMacroAssembler.cpp', 'irregexp/RegExpAST.cpp', 'irregexp/RegExpCharacters.cpp', + 'irregexp/RegExpCharRanges.cpp', 'irregexp/RegExpEngine.cpp', 'irregexp/RegExpInterpreter.cpp', 'irregexp/RegExpMacroAssembler.cpp',