mirror of
https://github.com/python/cpython.git
synced 2026-08-02 23:25:41 -04:00
Add support for \p{property} and \P{property} escapes in Unicode (str)
regular expressions, for the properties the engine can resolve without
the unicodedata database. They are matched as CATEGORY opcodes or as
fixed sets of character ranges.
Supported in this change: many General_Category values (the groups L, N,
Z, C and the values Lu, Lt, Lm, Nd, Nl, No, Zs, Zl, Zp, Cc, Cf, Cs, Co
and Cn); the binary properties Alphabetic, Lowercase, Uppercase, Numeric,
Printable, XID_Start, XID_Continue, Cased and Case_Ignorable; the POSIX
compatibility classes; the code-point classes ASCII, Any, Assigned,
Noncharacter_Code_Point, Join_Control, Pattern_Syntax and
Pattern_White_Space; and Regional_Indicator, ASCII_Hex_Digit and
Hex_Digit.
Property and value names use loose matching (UAX #44 UAX44-LM3), so a
property may be spelled \p{Lu}, \p{gc=Lu} or \p{name=yes}.
Co-Authored-By: Claude Opus 4.8 <[email protected]>
287 lines
8.3 KiB
Python
287 lines
8.3 KiB
Python
#
|
|
# Secret Labs' Regular Expression Engine
|
|
#
|
|
# various symbols used by the regular expression engine.
|
|
# run this script to update the _sre include files!
|
|
#
|
|
# Copyright (c) 1998-2001 by Secret Labs AB. All rights reserved.
|
|
#
|
|
# See the __init__.py file for information on usage and redistribution.
|
|
#
|
|
|
|
"""Internal support module for sre"""
|
|
|
|
# update when constants are added or removed
|
|
|
|
MAGIC = 20260622
|
|
|
|
from _sre import MAXREPEAT, MAXGROUPS # noqa: F401
|
|
|
|
# SRE standard exception (access as sre.error)
|
|
# should this really be here?
|
|
|
|
class PatternError(Exception):
|
|
"""Exception raised for invalid regular expressions.
|
|
|
|
Attributes:
|
|
|
|
msg: The unformatted error message
|
|
pattern: The regular expression pattern
|
|
pos: The index in the pattern where compilation failed (may be None)
|
|
lineno: The line corresponding to pos (may be None)
|
|
colno: The column corresponding to pos (may be None)
|
|
"""
|
|
|
|
__module__ = 're'
|
|
|
|
def __init__(self, msg, pattern=None, pos=None):
|
|
self.msg = msg
|
|
self.pattern = pattern
|
|
self.pos = pos
|
|
if pattern is not None and pos is not None:
|
|
msg = '%s at position %d' % (msg, pos)
|
|
if isinstance(pattern, str):
|
|
newline = '\n'
|
|
else:
|
|
newline = b'\n'
|
|
self.lineno = pattern.count(newline, 0, pos) + 1
|
|
self.colno = pos - pattern.rfind(newline, 0, pos)
|
|
if newline in pattern:
|
|
msg = '%s (line %d, column %d)' % (msg, self.lineno, self.colno)
|
|
else:
|
|
self.lineno = self.colno = None
|
|
super().__init__(msg)
|
|
|
|
|
|
# Backward compatibility after renaming in 3.13
|
|
error = PatternError
|
|
|
|
class _NamedIntConstant(int):
|
|
def __new__(cls, value, name):
|
|
self = super(_NamedIntConstant, cls).__new__(cls, value)
|
|
self.name = name
|
|
return self
|
|
|
|
def __repr__(self):
|
|
return self.name
|
|
|
|
__reduce__ = None
|
|
|
|
MAXREPEAT = _NamedIntConstant(MAXREPEAT, 'MAXREPEAT')
|
|
|
|
def _makecodes(*names):
|
|
items = [_NamedIntConstant(i, name) for i, name in enumerate(names)]
|
|
globals().update({item.name: item for item in items})
|
|
return items
|
|
|
|
# operators
|
|
OPCODES = _makecodes(
|
|
# failure=0 success=1 (just because it looks better that way :-)
|
|
'FAILURE', 'SUCCESS',
|
|
|
|
'ANY', 'ANY_ALL',
|
|
'ASSERT', 'ASSERT_NOT',
|
|
'AT',
|
|
'BRANCH',
|
|
'CATEGORY',
|
|
'CHARSET', 'BIGCHARSET',
|
|
'GROUPREF', 'GROUPREF_EXISTS',
|
|
'IN',
|
|
'INFO',
|
|
'JUMP',
|
|
'LITERAL',
|
|
'MARK',
|
|
'MAX_UNTIL',
|
|
'MIN_UNTIL',
|
|
'NOT_LITERAL',
|
|
'NEGATE',
|
|
'RANGE',
|
|
'REPEAT',
|
|
'REPEAT_ONE',
|
|
'SUBPATTERN',
|
|
'MIN_REPEAT_ONE',
|
|
'ATOMIC_GROUP',
|
|
'POSSESSIVE_REPEAT',
|
|
'POSSESSIVE_REPEAT_ONE',
|
|
|
|
'GROUPREF_IGNORE',
|
|
'IN_IGNORE',
|
|
'LITERAL_IGNORE',
|
|
'NOT_LITERAL_IGNORE',
|
|
|
|
'GROUPREF_LOC_IGNORE',
|
|
'IN_LOC_IGNORE',
|
|
'LITERAL_LOC_IGNORE',
|
|
'NOT_LITERAL_LOC_IGNORE',
|
|
|
|
'GROUPREF_UNI_IGNORE',
|
|
'IN_UNI_IGNORE',
|
|
'LITERAL_UNI_IGNORE',
|
|
'NOT_LITERAL_UNI_IGNORE',
|
|
'RANGE_UNI_IGNORE',
|
|
|
|
# The following opcodes are only occurred in the parser output,
|
|
# but not in the compiled code.
|
|
'MIN_REPEAT', 'MAX_REPEAT',
|
|
)
|
|
del OPCODES[-2:] # remove MIN_REPEAT and MAX_REPEAT
|
|
|
|
# positions
|
|
ATCODES = _makecodes(
|
|
'AT_BEGINNING', 'AT_BEGINNING_LINE', 'AT_BEGINNING_STRING',
|
|
'AT_BOUNDARY', 'AT_NON_BOUNDARY',
|
|
'AT_END', 'AT_END_LINE', 'AT_END_STRING',
|
|
|
|
'AT_LOC_BOUNDARY', 'AT_LOC_NON_BOUNDARY',
|
|
|
|
'AT_UNI_BOUNDARY', 'AT_UNI_NON_BOUNDARY',
|
|
)
|
|
|
|
# categories
|
|
CHCODES = _makecodes(
|
|
'CATEGORY_DIGIT', 'CATEGORY_NOT_DIGIT',
|
|
'CATEGORY_SPACE', 'CATEGORY_NOT_SPACE',
|
|
'CATEGORY_WORD', 'CATEGORY_NOT_WORD',
|
|
'CATEGORY_LINEBREAK', 'CATEGORY_NOT_LINEBREAK',
|
|
|
|
'CATEGORY_LOC_WORD', 'CATEGORY_LOC_NOT_WORD',
|
|
|
|
'CATEGORY_UNI_DIGIT', 'CATEGORY_UNI_NOT_DIGIT',
|
|
'CATEGORY_UNI_SPACE', 'CATEGORY_UNI_NOT_SPACE',
|
|
'CATEGORY_UNI_WORD', 'CATEGORY_UNI_NOT_WORD',
|
|
'CATEGORY_UNI_LINEBREAK', 'CATEGORY_UNI_NOT_LINEBREAK',
|
|
|
|
# Unicode property categories. These are not affected by the ASCII,
|
|
# LOCALE or UNICODE flags.
|
|
'CATEGORY_ALPHA', 'CATEGORY_NOT_ALPHA',
|
|
'CATEGORY_LOWER', 'CATEGORY_NOT_LOWER',
|
|
'CATEGORY_UPPER', 'CATEGORY_NOT_UPPER',
|
|
'CATEGORY_NUMERIC', 'CATEGORY_NOT_NUMERIC',
|
|
'CATEGORY_PRINTABLE', 'CATEGORY_NOT_PRINTABLE',
|
|
'CATEGORY_ALNUM', 'CATEGORY_NOT_ALNUM',
|
|
'CATEGORY_XID_START', 'CATEGORY_NOT_XID_START',
|
|
'CATEGORY_XID_CONTINUE', 'CATEGORY_NOT_XID_CONTINUE',
|
|
'CATEGORY_TITLE', 'CATEGORY_NOT_TITLE',
|
|
'CATEGORY_CASED', 'CATEGORY_NOT_CASED',
|
|
'CATEGORY_CASE_IGNORABLE', 'CATEGORY_NOT_CASE_IGNORABLE',
|
|
# Compound categories: Lu = uppercase letter, N = number.
|
|
'CATEGORY_LU', 'CATEGORY_NOT_LU',
|
|
'CATEGORY_N', 'CATEGORY_NOT_N',
|
|
'CATEGORY_LM', 'CATEGORY_NOT_LM',
|
|
'CATEGORY_NL', 'CATEGORY_NOT_NL',
|
|
'CATEGORY_NO', 'CATEGORY_NOT_NO',
|
|
'CATEGORY_CF', 'CATEGORY_NOT_CF',
|
|
'CATEGORY_Z', 'CATEGORY_NOT_Z',
|
|
'CATEGORY_ZS', 'CATEGORY_NOT_ZS',
|
|
'CATEGORY_C', 'CATEGORY_NOT_C',
|
|
'CATEGORY_CN', 'CATEGORY_NOT_CN',
|
|
'CATEGORY_ASSIGNED', 'CATEGORY_NOT_ASSIGNED',
|
|
'CATEGORY_BLANK', 'CATEGORY_NOT_BLANK',
|
|
'CATEGORY_GRAPH', 'CATEGORY_NOT_GRAPH',
|
|
'CATEGORY_PRINT', 'CATEGORY_NOT_PRINT',
|
|
)
|
|
|
|
|
|
# replacement operations for "ignore case" mode
|
|
OP_IGNORE = {
|
|
LITERAL: LITERAL_IGNORE,
|
|
NOT_LITERAL: NOT_LITERAL_IGNORE,
|
|
}
|
|
|
|
OP_LOCALE_IGNORE = {
|
|
LITERAL: LITERAL_LOC_IGNORE,
|
|
NOT_LITERAL: NOT_LITERAL_LOC_IGNORE,
|
|
}
|
|
|
|
OP_UNICODE_IGNORE = {
|
|
LITERAL: LITERAL_UNI_IGNORE,
|
|
NOT_LITERAL: NOT_LITERAL_UNI_IGNORE,
|
|
}
|
|
|
|
AT_MULTILINE = {
|
|
AT_BEGINNING: AT_BEGINNING_LINE,
|
|
AT_END: AT_END_LINE
|
|
}
|
|
|
|
AT_LOCALE = {
|
|
AT_BOUNDARY: AT_LOC_BOUNDARY,
|
|
AT_NON_BOUNDARY: AT_LOC_NON_BOUNDARY
|
|
}
|
|
|
|
AT_UNICODE = {
|
|
AT_BOUNDARY: AT_UNI_BOUNDARY,
|
|
AT_NON_BOUNDARY: AT_UNI_NON_BOUNDARY
|
|
}
|
|
|
|
CH_LOCALE = {
|
|
CATEGORY_DIGIT: CATEGORY_DIGIT,
|
|
CATEGORY_NOT_DIGIT: CATEGORY_NOT_DIGIT,
|
|
CATEGORY_SPACE: CATEGORY_SPACE,
|
|
CATEGORY_NOT_SPACE: CATEGORY_NOT_SPACE,
|
|
CATEGORY_WORD: CATEGORY_LOC_WORD,
|
|
CATEGORY_NOT_WORD: CATEGORY_LOC_NOT_WORD,
|
|
CATEGORY_LINEBREAK: CATEGORY_LINEBREAK,
|
|
CATEGORY_NOT_LINEBREAK: CATEGORY_NOT_LINEBREAK
|
|
}
|
|
|
|
CH_UNICODE = {
|
|
CATEGORY_DIGIT: CATEGORY_UNI_DIGIT,
|
|
CATEGORY_NOT_DIGIT: CATEGORY_UNI_NOT_DIGIT,
|
|
CATEGORY_SPACE: CATEGORY_UNI_SPACE,
|
|
CATEGORY_NOT_SPACE: CATEGORY_UNI_NOT_SPACE,
|
|
CATEGORY_WORD: CATEGORY_UNI_WORD,
|
|
CATEGORY_NOT_WORD: CATEGORY_UNI_NOT_WORD,
|
|
CATEGORY_LINEBREAK: CATEGORY_UNI_LINEBREAK,
|
|
CATEGORY_NOT_LINEBREAK: CATEGORY_UNI_NOT_LINEBREAK
|
|
}
|
|
|
|
# The Unicode property categories are the same regardless of the flags.
|
|
CH_PROPERTY = (
|
|
CATEGORY_ALPHA, CATEGORY_NOT_ALPHA,
|
|
CATEGORY_LOWER, CATEGORY_NOT_LOWER,
|
|
CATEGORY_UPPER, CATEGORY_NOT_UPPER,
|
|
CATEGORY_NUMERIC, CATEGORY_NOT_NUMERIC,
|
|
CATEGORY_PRINTABLE, CATEGORY_NOT_PRINTABLE,
|
|
CATEGORY_ALNUM, CATEGORY_NOT_ALNUM,
|
|
CATEGORY_XID_START, CATEGORY_NOT_XID_START,
|
|
CATEGORY_XID_CONTINUE, CATEGORY_NOT_XID_CONTINUE,
|
|
CATEGORY_TITLE, CATEGORY_NOT_TITLE,
|
|
CATEGORY_CASED, CATEGORY_NOT_CASED,
|
|
CATEGORY_CASE_IGNORABLE, CATEGORY_NOT_CASE_IGNORABLE,
|
|
CATEGORY_LU, CATEGORY_NOT_LU,
|
|
CATEGORY_N, CATEGORY_NOT_N,
|
|
CATEGORY_LM, CATEGORY_NOT_LM,
|
|
CATEGORY_NL, CATEGORY_NOT_NL,
|
|
CATEGORY_NO, CATEGORY_NOT_NO,
|
|
CATEGORY_CF, CATEGORY_NOT_CF,
|
|
CATEGORY_Z, CATEGORY_NOT_Z,
|
|
CATEGORY_ZS, CATEGORY_NOT_ZS,
|
|
CATEGORY_C, CATEGORY_NOT_C,
|
|
CATEGORY_CN, CATEGORY_NOT_CN,
|
|
CATEGORY_ASSIGNED, CATEGORY_NOT_ASSIGNED,
|
|
CATEGORY_BLANK, CATEGORY_NOT_BLANK,
|
|
CATEGORY_GRAPH, CATEGORY_NOT_GRAPH,
|
|
CATEGORY_PRINT, CATEGORY_NOT_PRINT,
|
|
)
|
|
for _cat in CH_PROPERTY:
|
|
CH_LOCALE[_cat] = _cat
|
|
CH_UNICODE[_cat] = _cat
|
|
del _cat
|
|
|
|
CH_NEGATE = dict(zip(CHCODES[::2] + CHCODES[1::2], CHCODES[1::2] + CHCODES[::2]))
|
|
|
|
# flags
|
|
SRE_FLAG_IGNORECASE = 2 # case insensitive
|
|
SRE_FLAG_LOCALE = 4 # honour system locale
|
|
SRE_FLAG_MULTILINE = 8 # treat target as multiline string
|
|
SRE_FLAG_DOTALL = 16 # treat target as a single string
|
|
SRE_FLAG_UNICODE = 32 # use unicode "locale"
|
|
SRE_FLAG_VERBOSE = 64 # ignore whitespace and comments
|
|
SRE_FLAG_DEBUG = 128 # debugging
|
|
SRE_FLAG_ASCII = 256 # use ascii "locale"
|
|
|
|
# flags for INFO primitive
|
|
SRE_INFO_PREFIX = 1 # has prefix
|
|
SRE_INFO_LITERAL = 2 # entire pattern is literal (given by prefix)
|
|
SRE_INFO_CHARSET = 4 # pattern starts with character from given set
|