Files
cpython/Lib/re/_constants.py
T
794b42ff8a gh-95555: Support Unicode property escapes \p{...} in regular expressions (GH-151969)
Add support for \p{property} and \P{property} escapes in Unicode (str)
regular expressions, for the properties the engine can resolve without
the unicodedata database.  They are matched as CATEGORY opcodes or as
fixed sets of character ranges.

Supported in this change: many General_Category values (the groups L, N,
Z, C and the values Lu, Lt, Lm, Nd, Nl, No, Zs, Zl, Zp, Cc, Cf, Cs, Co
and Cn); the binary properties Alphabetic, Lowercase, Uppercase, Numeric,
Printable, XID_Start, XID_Continue, Cased and Case_Ignorable; the POSIX
compatibility classes; the code-point classes ASCII, Any, Assigned,
Noncharacter_Code_Point, Join_Control, Pattern_Syntax and
Pattern_White_Space; and Regional_Indicator, ASCII_Hex_Digit and
Hex_Digit.

Property and value names use loose matching (UAX #44 UAX44-LM3), so a
property may be spelled \p{Lu}, \p{gc=Lu} or \p{name=yes}.

Co-Authored-By: Claude Opus 4.8 <[email protected]>
2026-06-26 07:33:33 +03:00

287 lines
8.3 KiB
Python

#
# Secret Labs' Regular Expression Engine
#
# various symbols used by the regular expression engine.
# run this script to update the _sre include files!
#
# Copyright (c) 1998-2001 by Secret Labs AB. All rights reserved.
#
# See the __init__.py file for information on usage and redistribution.
#
"""Internal support module for sre"""
# update when constants are added or removed
MAGIC = 20260622
from _sre import MAXREPEAT, MAXGROUPS # noqa: F401
# SRE standard exception (access as sre.error)
# should this really be here?
class PatternError(Exception):
"""Exception raised for invalid regular expressions.
Attributes:
msg: The unformatted error message
pattern: The regular expression pattern
pos: The index in the pattern where compilation failed (may be None)
lineno: The line corresponding to pos (may be None)
colno: The column corresponding to pos (may be None)
"""
__module__ = 're'
def __init__(self, msg, pattern=None, pos=None):
self.msg = msg
self.pattern = pattern
self.pos = pos
if pattern is not None and pos is not None:
msg = '%s at position %d' % (msg, pos)
if isinstance(pattern, str):
newline = '\n'
else:
newline = b'\n'
self.lineno = pattern.count(newline, 0, pos) + 1
self.colno = pos - pattern.rfind(newline, 0, pos)
if newline in pattern:
msg = '%s (line %d, column %d)' % (msg, self.lineno, self.colno)
else:
self.lineno = self.colno = None
super().__init__(msg)
# Backward compatibility after renaming in 3.13
error = PatternError
class _NamedIntConstant(int):
def __new__(cls, value, name):
self = super(_NamedIntConstant, cls).__new__(cls, value)
self.name = name
return self
def __repr__(self):
return self.name
__reduce__ = None
MAXREPEAT = _NamedIntConstant(MAXREPEAT, 'MAXREPEAT')
def _makecodes(*names):
items = [_NamedIntConstant(i, name) for i, name in enumerate(names)]
globals().update({item.name: item for item in items})
return items
# operators
OPCODES = _makecodes(
# failure=0 success=1 (just because it looks better that way :-)
'FAILURE', 'SUCCESS',
'ANY', 'ANY_ALL',
'ASSERT', 'ASSERT_NOT',
'AT',
'BRANCH',
'CATEGORY',
'CHARSET', 'BIGCHARSET',
'GROUPREF', 'GROUPREF_EXISTS',
'IN',
'INFO',
'JUMP',
'LITERAL',
'MARK',
'MAX_UNTIL',
'MIN_UNTIL',
'NOT_LITERAL',
'NEGATE',
'RANGE',
'REPEAT',
'REPEAT_ONE',
'SUBPATTERN',
'MIN_REPEAT_ONE',
'ATOMIC_GROUP',
'POSSESSIVE_REPEAT',
'POSSESSIVE_REPEAT_ONE',
'GROUPREF_IGNORE',
'IN_IGNORE',
'LITERAL_IGNORE',
'NOT_LITERAL_IGNORE',
'GROUPREF_LOC_IGNORE',
'IN_LOC_IGNORE',
'LITERAL_LOC_IGNORE',
'NOT_LITERAL_LOC_IGNORE',
'GROUPREF_UNI_IGNORE',
'IN_UNI_IGNORE',
'LITERAL_UNI_IGNORE',
'NOT_LITERAL_UNI_IGNORE',
'RANGE_UNI_IGNORE',
# The following opcodes are only occurred in the parser output,
# but not in the compiled code.
'MIN_REPEAT', 'MAX_REPEAT',
)
del OPCODES[-2:] # remove MIN_REPEAT and MAX_REPEAT
# positions
ATCODES = _makecodes(
'AT_BEGINNING', 'AT_BEGINNING_LINE', 'AT_BEGINNING_STRING',
'AT_BOUNDARY', 'AT_NON_BOUNDARY',
'AT_END', 'AT_END_LINE', 'AT_END_STRING',
'AT_LOC_BOUNDARY', 'AT_LOC_NON_BOUNDARY',
'AT_UNI_BOUNDARY', 'AT_UNI_NON_BOUNDARY',
)
# categories
CHCODES = _makecodes(
'CATEGORY_DIGIT', 'CATEGORY_NOT_DIGIT',
'CATEGORY_SPACE', 'CATEGORY_NOT_SPACE',
'CATEGORY_WORD', 'CATEGORY_NOT_WORD',
'CATEGORY_LINEBREAK', 'CATEGORY_NOT_LINEBREAK',
'CATEGORY_LOC_WORD', 'CATEGORY_LOC_NOT_WORD',
'CATEGORY_UNI_DIGIT', 'CATEGORY_UNI_NOT_DIGIT',
'CATEGORY_UNI_SPACE', 'CATEGORY_UNI_NOT_SPACE',
'CATEGORY_UNI_WORD', 'CATEGORY_UNI_NOT_WORD',
'CATEGORY_UNI_LINEBREAK', 'CATEGORY_UNI_NOT_LINEBREAK',
# Unicode property categories. These are not affected by the ASCII,
# LOCALE or UNICODE flags.
'CATEGORY_ALPHA', 'CATEGORY_NOT_ALPHA',
'CATEGORY_LOWER', 'CATEGORY_NOT_LOWER',
'CATEGORY_UPPER', 'CATEGORY_NOT_UPPER',
'CATEGORY_NUMERIC', 'CATEGORY_NOT_NUMERIC',
'CATEGORY_PRINTABLE', 'CATEGORY_NOT_PRINTABLE',
'CATEGORY_ALNUM', 'CATEGORY_NOT_ALNUM',
'CATEGORY_XID_START', 'CATEGORY_NOT_XID_START',
'CATEGORY_XID_CONTINUE', 'CATEGORY_NOT_XID_CONTINUE',
'CATEGORY_TITLE', 'CATEGORY_NOT_TITLE',
'CATEGORY_CASED', 'CATEGORY_NOT_CASED',
'CATEGORY_CASE_IGNORABLE', 'CATEGORY_NOT_CASE_IGNORABLE',
# Compound categories: Lu = uppercase letter, N = number.
'CATEGORY_LU', 'CATEGORY_NOT_LU',
'CATEGORY_N', 'CATEGORY_NOT_N',
'CATEGORY_LM', 'CATEGORY_NOT_LM',
'CATEGORY_NL', 'CATEGORY_NOT_NL',
'CATEGORY_NO', 'CATEGORY_NOT_NO',
'CATEGORY_CF', 'CATEGORY_NOT_CF',
'CATEGORY_Z', 'CATEGORY_NOT_Z',
'CATEGORY_ZS', 'CATEGORY_NOT_ZS',
'CATEGORY_C', 'CATEGORY_NOT_C',
'CATEGORY_CN', 'CATEGORY_NOT_CN',
'CATEGORY_ASSIGNED', 'CATEGORY_NOT_ASSIGNED',
'CATEGORY_BLANK', 'CATEGORY_NOT_BLANK',
'CATEGORY_GRAPH', 'CATEGORY_NOT_GRAPH',
'CATEGORY_PRINT', 'CATEGORY_NOT_PRINT',
)
# replacement operations for "ignore case" mode
OP_IGNORE = {
LITERAL: LITERAL_IGNORE,
NOT_LITERAL: NOT_LITERAL_IGNORE,
}
OP_LOCALE_IGNORE = {
LITERAL: LITERAL_LOC_IGNORE,
NOT_LITERAL: NOT_LITERAL_LOC_IGNORE,
}
OP_UNICODE_IGNORE = {
LITERAL: LITERAL_UNI_IGNORE,
NOT_LITERAL: NOT_LITERAL_UNI_IGNORE,
}
AT_MULTILINE = {
AT_BEGINNING: AT_BEGINNING_LINE,
AT_END: AT_END_LINE
}
AT_LOCALE = {
AT_BOUNDARY: AT_LOC_BOUNDARY,
AT_NON_BOUNDARY: AT_LOC_NON_BOUNDARY
}
AT_UNICODE = {
AT_BOUNDARY: AT_UNI_BOUNDARY,
AT_NON_BOUNDARY: AT_UNI_NON_BOUNDARY
}
CH_LOCALE = {
CATEGORY_DIGIT: CATEGORY_DIGIT,
CATEGORY_NOT_DIGIT: CATEGORY_NOT_DIGIT,
CATEGORY_SPACE: CATEGORY_SPACE,
CATEGORY_NOT_SPACE: CATEGORY_NOT_SPACE,
CATEGORY_WORD: CATEGORY_LOC_WORD,
CATEGORY_NOT_WORD: CATEGORY_LOC_NOT_WORD,
CATEGORY_LINEBREAK: CATEGORY_LINEBREAK,
CATEGORY_NOT_LINEBREAK: CATEGORY_NOT_LINEBREAK
}
CH_UNICODE = {
CATEGORY_DIGIT: CATEGORY_UNI_DIGIT,
CATEGORY_NOT_DIGIT: CATEGORY_UNI_NOT_DIGIT,
CATEGORY_SPACE: CATEGORY_UNI_SPACE,
CATEGORY_NOT_SPACE: CATEGORY_UNI_NOT_SPACE,
CATEGORY_WORD: CATEGORY_UNI_WORD,
CATEGORY_NOT_WORD: CATEGORY_UNI_NOT_WORD,
CATEGORY_LINEBREAK: CATEGORY_UNI_LINEBREAK,
CATEGORY_NOT_LINEBREAK: CATEGORY_UNI_NOT_LINEBREAK
}
# The Unicode property categories are the same regardless of the flags.
CH_PROPERTY = (
CATEGORY_ALPHA, CATEGORY_NOT_ALPHA,
CATEGORY_LOWER, CATEGORY_NOT_LOWER,
CATEGORY_UPPER, CATEGORY_NOT_UPPER,
CATEGORY_NUMERIC, CATEGORY_NOT_NUMERIC,
CATEGORY_PRINTABLE, CATEGORY_NOT_PRINTABLE,
CATEGORY_ALNUM, CATEGORY_NOT_ALNUM,
CATEGORY_XID_START, CATEGORY_NOT_XID_START,
CATEGORY_XID_CONTINUE, CATEGORY_NOT_XID_CONTINUE,
CATEGORY_TITLE, CATEGORY_NOT_TITLE,
CATEGORY_CASED, CATEGORY_NOT_CASED,
CATEGORY_CASE_IGNORABLE, CATEGORY_NOT_CASE_IGNORABLE,
CATEGORY_LU, CATEGORY_NOT_LU,
CATEGORY_N, CATEGORY_NOT_N,
CATEGORY_LM, CATEGORY_NOT_LM,
CATEGORY_NL, CATEGORY_NOT_NL,
CATEGORY_NO, CATEGORY_NOT_NO,
CATEGORY_CF, CATEGORY_NOT_CF,
CATEGORY_Z, CATEGORY_NOT_Z,
CATEGORY_ZS, CATEGORY_NOT_ZS,
CATEGORY_C, CATEGORY_NOT_C,
CATEGORY_CN, CATEGORY_NOT_CN,
CATEGORY_ASSIGNED, CATEGORY_NOT_ASSIGNED,
CATEGORY_BLANK, CATEGORY_NOT_BLANK,
CATEGORY_GRAPH, CATEGORY_NOT_GRAPH,
CATEGORY_PRINT, CATEGORY_NOT_PRINT,
)
for _cat in CH_PROPERTY:
CH_LOCALE[_cat] = _cat
CH_UNICODE[_cat] = _cat
del _cat
CH_NEGATE = dict(zip(CHCODES[::2] + CHCODES[1::2], CHCODES[1::2] + CHCODES[::2]))
# flags
SRE_FLAG_IGNORECASE = 2 # case insensitive
SRE_FLAG_LOCALE = 4 # honour system locale
SRE_FLAG_MULTILINE = 8 # treat target as multiline string
SRE_FLAG_DOTALL = 16 # treat target as a single string
SRE_FLAG_UNICODE = 32 # use unicode "locale"
SRE_FLAG_VERBOSE = 64 # ignore whitespace and comments
SRE_FLAG_DEBUG = 128 # debugging
SRE_FLAG_ASCII = 256 # use ascii "locale"
# flags for INFO primitive
SRE_INFO_PREFIX = 1 # has prefix
SRE_INFO_LITERAL = 2 # entire pattern is literal (given by prefix)
SRE_INFO_CHARSET = 4 # pattern starts with character from given set