Files
sdk/tests/corelib_2/regexp/unicode-property-special_test.dart
T
Stevie Strickland 4028fec3b5 Reland "[vm] Finish adding support for ECMAScript 2018 features."
This work pulls in v8 support for these features with
appropriate changes for Dart and closes
https://github.com/dart-lang/sdk/issues/34935.

This adds support for the following features:

* Interpreting patterns as Unicode patterns instead of
  BMP patterns
* the dotAll flag (`/s`) for changing the behavior
  of '.' to also match line terminators
* Escapes for character classes described by Unicode
  property groups (e.g., \p{Greek} to match all Greek
  characters, or \P{Greek} for all non-Greek characters).

The following TC39 proposals describe some of the added features:

* https://github.com/tc39/proposal-regexp-dotall-flag
* https://github.com/tc39/proposal-regexp-unicode-property-escapes

These additional changes are included:

* Extends named capture group names to include the full
  range of identifier characters supported by ECMAScript,
  not just ASCII.
* Changing the RegExp interface to return RegExpMatch
  objects, not Match objects, so that downcasting is
  not necessary to use named capture groups from Dart

**Note**: The changes to the RegExp interface are a
breaking change for implementers of the RegExp interface.
Current users of the RegExp interface (i.e., code using Dart
RegExp objects) will not be affected.

Change-Id: Ie62e6082a0e2fedc1680ef2576ce0c6db80fc19a
Reviewed-on: https://dart-review.googlesource.com/c/sdk/+/100641
Reviewed-by: Martin Kustermann <kustermann@google.com>
Commit-Queue: Stevie Strickland <sstrickl@google.com>
2019-04-29 09:11:48 +00:00

111 lines
5.2 KiB
Dart

// Copyright (c) 2019, the Dart project authors. All rights reserved.
// Copyright 2016 the V8 project authors. All rights reserved.
// Redistribution and use in source and binary forms, with or without
// modification, are permitted provided that the following conditions are
// met:
//
// * Redistributions of source code must retain the above copyright
// notice, this list of conditions and the following disclaimer.
// * Redistributions in binary form must reproduce the above
// copyright notice, this list of conditions and the following
// disclaimer in the documentation and/or other materials provided
// with the distribution.
// * Neither the name of Google Inc. nor the names of its
// contributors may be used to endorse or promote products derived
// from this software without specific prior written permission.
//
// THIS SOFTWARE IS PROVIDED BY THE COPYRIGHT HOLDERS AND CONTRIBUTORS
// "AS IS" AND ANY EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT
// LIMITED TO, THE IMPLIED WARRANTIES OF MERCHANTABILITY AND FITNESS FOR
// A PARTICULAR PURPOSE ARE DISCLAIMED. IN NO EVENT SHALL THE COPYRIGHT
// OWNER OR CONTRIBUTORS BE LIABLE FOR ANY DIRECT, INDIRECT, INCIDENTAL,
// SPECIAL, EXEMPLARY, OR CONSEQUENTIAL DAMAGES (INCLUDING, BUT NOT
// LIMITED TO, PROCUREMENT OF SUBSTITUTE GOODS OR SERVICES; LOSS OF USE,
// DATA, OR PROFITS; OR BUSINESS INTERRUPTION) HOWEVER CAUSED AND ON ANY
// THEORY OF LIABILITY, WHETHER IN CONTRACT, STRICT LIABILITY, OR TORT
// (INCLUDING NEGLIGENCE OR OTHERWISE) ARISING IN ANY WAY OUT OF THE USE
// OF THIS SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.
import 'package:expect/expect.dart';
import 'v8_regexp_utils.dart';
void main() {
void t(RegExp re, String s) {
assertTrue(re.hasMatch(s));
}
void f(RegExp re, String s) {
assertFalse(re.hasMatch(s));
}
t(RegExp(r"\p{ASCII}+", unicode: true), "abc123");
f(RegExp(r"\p{ASCII}+", unicode: true), "ⓐⓑⓒ①②③");
f(RegExp(r"\p{ASCII}+", unicode: true), "🄰🄱🄲①②③");
f(RegExp(r"\P{ASCII}+", unicode: true), "abcd123");
t(RegExp(r"\P{ASCII}+", unicode: true), "ⓐⓑⓒ①②③");
t(RegExp(r"\P{ASCII}+", unicode: true), "🄰🄱🄲①②③");
f(RegExp(r"[^\p{ASCII}]+", unicode: true), "abc123");
f(RegExp(r"[\p{ASCII}]+", unicode: true), "ⓐⓑⓒ①②③");
f(RegExp(r"[\p{ASCII}]+", unicode: true), "🄰🄱🄲①②③");
t(RegExp(r"[^\P{ASCII}]+", unicode: true), "abcd123");
t(RegExp(r"[\P{ASCII}]+", unicode: true), "ⓐⓑⓒ①②③");
f(RegExp(r"[^\P{ASCII}]+", unicode: true), "🄰🄱🄲①②③");
t(RegExp(r"\p{Any}+", unicode: true), "🄰🄱🄲①②③");
shouldBe(
RegExp(r"\p{Any}", unicode: true).firstMatch("\ud800\ud801"), ["\ud800"]);
shouldBe(
RegExp(r"\p{Any}", unicode: true).firstMatch("\udc00\udc01"), ["\udc00"]);
shouldBe(RegExp(r"\p{Any}", unicode: true).firstMatch("\ud800\udc01"),
["\ud800\udc01"]);
shouldBe(RegExp(r"\p{Any}", unicode: true).firstMatch("\udc01"), ["\udc01"]);
f(RegExp(r"\P{Any}+", unicode: true), "123");
f(RegExp(r"[\P{Any}]+", unicode: true), "123");
t(RegExp(r"[\P{Any}\d]+", unicode: true), "123");
t(RegExp(r"[^\P{Any}]+", unicode: true), "123");
t(RegExp(r"\p{Assigned}+", unicode: true), "123");
t(RegExp(r"\p{Assigned}+", unicode: true), "🄰🄱🄲");
f(RegExp(r"\p{Assigned}+", unicode: true), "\ufdd0");
f(RegExp(r"\p{Assigned}+", unicode: true), "\u{fffff}");
f(RegExp(r"\P{Assigned}+", unicode: true), "123");
f(RegExp(r"\P{Assigned}+", unicode: true), "🄰🄱🄲");
t(RegExp(r"\P{Assigned}+", unicode: true), "\ufdd0");
t(RegExp(r"\P{Assigned}+", unicode: true), "\u{fffff}");
f(RegExp(r"\P{Assigned}", unicode: true), "");
t(RegExp(r"[^\P{Assigned}]+", unicode: true), "123");
f(RegExp(r"[\P{Assigned}]+", unicode: true), "🄰🄱🄲");
f(RegExp(r"[^\P{Assigned}]+", unicode: true), "\ufdd0");
t(RegExp(r"[\P{Assigned}]+", unicode: true), "\u{fffff}");
f(RegExp(r"[\P{Assigned}]", unicode: true), "");
f(RegExp(r"[^\u1234\p{ASCII}]+", unicode: true), "\u1234");
t(RegExp(r"[x\P{ASCII}]+", unicode: true), "x");
t(RegExp(r"[\u1234\p{ASCII}]+", unicode: true), "\u1234");
// Contributory binary properties are not supported.
assertThrows(() => RegExp("\\p{Other_Alphabetic}", unicode: true));
assertThrows(() => RegExp("\\P{OAlpha}", unicode: true));
assertThrows(
() => RegExp("\\p{Other_Default_Ignorable_Code_Point}", unicode: true));
assertThrows(() => RegExp("\\P{ODI}", unicode: true));
assertThrows(() => RegExp("\\p{Other_Grapheme_Extend}", unicode: true));
assertThrows(() => RegExp("\\P{OGr_Ext}", unicode: true));
assertThrows(() => RegExp("\\p{Other_ID_Continue}", unicode: true));
assertThrows(() => RegExp("\\P{OIDC}", unicode: true));
assertThrows(() => RegExp("\\p{Other_ID_Start}", unicode: true));
assertThrows(() => RegExp("\\P{OIDS}", unicode: true));
assertThrows(() => RegExp("\\p{Other_Lowercase}", unicode: true));
assertThrows(() => RegExp("\\P{OLower}", unicode: true));
assertThrows(() => RegExp("\\p{Other_Math}", unicode: true));
assertThrows(() => RegExp("\\P{OMath}", unicode: true));
assertThrows(() => RegExp("\\p{Other_Uppercase}", unicode: true));
assertThrows(() => RegExp("\\P{OUpper}", unicode: true));
}