diff --git a/.github/workflows/build.yml b/.github/workflows/build.yml index f4ac6f0d..7de2f423 100644 --- a/.github/workflows/build.yml +++ b/.github/workflows/build.yml @@ -17,11 +17,8 @@ jobs: max-parallel: 4 matrix: platform: [ubuntu-latest, windows-latest] - tox-env: [py39, py310, py311, py312, py313, py314, nolxml, nohtml5lib] + tox-env: [py310, py311, py312, py313, py314, nolxml, nohtml5lib] include: - - tox-env: py39 - python-version: 3.9 - continue-on-error: false - tox-env: py310 python-version: '3.10' continue-on-error: false @@ -43,9 +40,12 @@ jobs: - tox-env: nohtml5lib python-version: '3.13' continue-on-error: false - exclude: - - platform: windows-latest - tox-env: py314 + - tox-env: nohtml5lib + python-version: '3.14' + continue-on-error: false + # exclude: + # - platform: windows-latest + # tox-env: py314 env: TOXENV: ${{ matrix.tox-env }} @@ -80,7 +80,7 @@ jobs: strategy: max-parallel: 4 matrix: - python-version: [3.13] + python-version: [3.14] env: TOXENV: lint @@ -104,7 +104,7 @@ jobs: strategy: max-parallel: 4 matrix: - python-version: [3.13] + python-version: [3.14] env: TOXENV: documents diff --git a/.github/workflows/deploy.yml b/.github/workflows/deploy.yml index 6a6d2d77..4b9b595c 100644 --- a/.github/workflows/deploy.yml +++ b/.github/workflows/deploy.yml @@ -13,7 +13,7 @@ jobs: - uses: actions/checkout@v4 - uses: actions/setup-python@v5 with: - python-version: 3.13 + python-version: 3.14 - name: Build run: | pip install --upgrade pip build diff --git a/.github/workflows/publish.yaml b/.github/workflows/publish.yaml index b38f4b13..2a9ebdb0 100644 --- a/.github/workflows/publish.yaml +++ b/.github/workflows/publish.yaml @@ -12,7 +12,7 @@ jobs: strategy: max-parallel: 4 matrix: - python-version: ['3.13'] + python-version: ['3.14'] runs-on: ubuntu-latest diff --git a/docs/src/markdown/about/changelog.md b/docs/src/markdown/about/changelog.md index eb56aafb..31d4613d 100644 --- a/docs/src/markdown/about/changelog.md +++ b/docs/src/markdown/about/changelog.md @@ -3,6 +3,21 @@ icon: lucide/scroll-text --- # Changelog +## 2.9 + +- **NEW**: Drop Python 3.9 support. +- **NEW**: Lazy compile selector patterns to improve initial import speed. +- **FIX**: Correct `:nth-child`/`:nth-of-type` (and `-last-` variants) for `An+B` values whose sequence steps onto + index 0 or onto the last child (e.g. `:nth-child(2n-2)`, `:nth-child(n-1)`, `:nth-child(n+5)`), which previously + matched the wrong elements or nothing at all (@gaoflow). +- **FIX**: More efficient CSS ID matching (@kaimandalic). +- **FIX**: Fix inefficient trimming of comments and white space (@kaimandalic). + +## 2.8.4 + +- **FIX**: Fix another inefficient attribute pattern (@mauriceng98). +- **FIX**: Limit total number of selectors processed in a pattern to prevent massive selector requests (@mauriceng98). + ## 2.8.3 - **FIX**: Fix inefficient attribute pattern. diff --git a/docs/src/markdown/api.md b/docs/src/markdown/api.md index c3a34255..e5c0399c 100644 --- a/docs/src/markdown/api.md +++ b/docs/src/markdown/api.md @@ -26,7 +26,7 @@ change. > [!tip] Getting Proper Namespaces > The `html5lib` parser provides proper namespaces for HTML5, but `lxml`'s HTML parser will not. If you need > namespace support for HTML5, consider using `html5lib`. - > + > > For XML, the `lxml-xml` parser (`xml` for short) will provide proper namespaces. It is generally suggested that > `lxml-xml` is used to parse XHTML documents to take advantage of namespaces. @@ -37,6 +37,12 @@ change. While Soup Sieve access is exposed through Beautiful Soup's API, Soup Sieve's API can always be imported and accessed directly for more controlled tag selection if needed. +> [!note] Selector Limits +> Starting in 2.8.4, number of selectors in a given pattern are arbitrarily capped at ~8192. This limitation was added +> to prevent cases where impractical, massive selectors. +> +> Some selectors are defined as a series of other pre-defined selectors which also contribute towards the count. + ## Flags ### `soupseive.DEBUG` diff --git a/docs/src/markdown/selectors/pseudo-classes.md b/docs/src/markdown/selectors/pseudo-classes.md index 9958bff4..e057a819 100644 --- a/docs/src/markdown/selectors/pseudo-classes.md +++ b/docs/src/markdown/selectors/pseudo-classes.md @@ -1707,7 +1707,7 @@ While the level 4 specifications state that [compound](#compound-selector) selec Selects elements that contain the provided text. Text can be found in either itself, or its descendants. Originally, there was a pseudo-class called `:contains()` that was originally included in a [CSS early draft][contains-draft], -but was dropped from the draft in the end. Soup Sieve implements it how it was originally proposed accept for two +but was dropped from the draft in the end. Soup Sieve implements it how it was originally proposed except for two differences: it is called `:-soup-contains()` instead of `:contains()`, and it can accept either a single value, or a comma separated list of values. An element needs only to match at least one of the items in the comma separated list to be considered matching. diff --git a/hatch_build.py b/hatch_build.py index 6849c3d9..300461a3 100644 --- a/hatch_build.py +++ b/hatch_build.py @@ -29,7 +29,6 @@ def update(self, metadata): 'License :: OSI Approved :: MIT License', 'Operating System :: OS Independent', 'Programming Language :: Python :: 3', - 'Programming Language :: Python :: 3.9', 'Programming Language :: Python :: 3.10', 'Programming Language :: Python :: 3.11', 'Programming Language :: Python :: 3.12', diff --git a/pyproject.toml b/pyproject.toml index b0267085..32f0b634 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -1,6 +1,6 @@ [build-system] requires = [ - "hatchling>=0.21.1", + "hatchling>=1.26", ] build-backend = "hatchling.build" @@ -9,7 +9,7 @@ name = "soupsieve" description = "A modern CSS selector implementation for Beautiful Soup." readme = "README.md" license = "MIT" -requires-python = ">=3.9" +requires-python = ">=3.10" authors = [ { name = "Isaac Muse", email = "Isaac.Muse@gmail.com" }, ] @@ -54,7 +54,7 @@ include = [ "/tests/**/*.py", "/.pyspelling.yml", "/.coveragerc", - "/mkdocs.yml" + "/zensical.yml" ] [tool.mypy] @@ -109,7 +109,7 @@ legacy_tox_ini = """ [tox] isolated_build = true envlist = - py{39,310,311,312,313,314}, + py{310,311,312,313,314}, lint, nolxml, nohtml5lib [testenv] @@ -128,7 +128,7 @@ passenv = * deps = -rrequirements/docs.txt commands = - {envbindir}/zensical build -f zensical.yml --clean --strict + {envbindir}/zensical build -f zensical.yml --clean pyspelling -j 8 [testenv:lint] diff --git a/soupsieve/__meta__.py b/soupsieve/__meta__.py index c472d5c6..b1c018dc 100644 --- a/soupsieve/__meta__.py +++ b/soupsieve/__meta__.py @@ -193,5 +193,5 @@ def parse_version(ver: str) -> Version: return Version(major, minor, micro, release, pre, post, dev) -__version_info__ = Version(2, 8, 3, "final") +__version_info__ = Version(2, 9, 0, "final") __version__ = __version_info__._get_canonical() diff --git a/soupsieve/css_match.py b/soupsieve/css_match.py index 15e3917c..5f4dc1e1 100644 --- a/soupsieve/css_match.py +++ b/soupsieve/css_match.py @@ -995,44 +995,18 @@ def match_nth(self, el: bs4.Tag, nth: tuple[ct.SelectorNth, ...]) -> bool: # We can only adjust bounds within a variable index if var: - # Abort if our nth index is out of bounds and only getting further out of bounds as we increment. - # Otherwise, increment to try to get in bounds. - adjust = None - while idx < 1 or idx > last_index: - if idx < 0: - diff_low = 0 - idx - if adjust is not None and adjust == 1: - break - adjust = -1 - count += count_incr - idx = last_idx = a * count + b if var else a - diff = 0 - idx - if diff >= diff_low: - break - else: - diff_high = idx - last_index - if adjust is not None and adjust == -1: - break - adjust = 1 - count += count_incr - idx = last_idx = a * count + b if var else a - diff = idx - last_index - if diff >= diff_high: - break - diff_high = diff - - # If a < 0, our count is working backwards, so floor the index by increasing the count. - # Find the count that yields the lowest, in bound value and use that. - # Lastly reverse count increment so that we'll increase our index. - lowest = count - if a < 0: - while idx >= 1: - lowest = count - count += count_incr - idx = last_idx = a * count + b if var else a + # Find the count `n` that yields the smallest in-bounds index + # (>= 1), then set the increment direction so that the index + # ascends from there as the evaluation loop below walks children. + if a > 0: + # Ascending sequence: smallest n with a * n + b >= 1. + count = 0 if b >= 1 else -(-(1 - b) // a) + elif a < 0: + # Descending sequence: largest n with a * n + b >= 1, then + # walk n back down so the index increases. + count = (b - 1) // -a if b >= 1 else 0 count_incr = -1 - count = lowest - idx = last_idx = a * count + b if var else a + idx = last_idx = a * count + b # Evaluate elements while our calculated nth index is still in range while 1 <= idx <= last_index + 1: diff --git a/soupsieve/css_parser.py b/soupsieve/css_parser.py index 3140ba82..663abff1 100644 --- a/soupsieve/css_parser.py +++ b/soupsieve/css_parser.py @@ -8,9 +8,17 @@ from .util import SelectorSyntaxError import warnings from typing import Match, Any, Iterator, cast +from dataclasses import dataclass +from collections import UserDict +import threading + +RE_LOCK = threading.Lock() +SEL_LOCK = threading.RLock() UNICODE_REPLACEMENT_CHAR = 0xFFFD +SELECTOR_LIMIT = 8192 + # Simple pseudo classes that take no parameters PSEUDO_SIMPLE = { ":any-link", @@ -104,7 +112,7 @@ NEWLINE = r'(?:\r\n|(?!\r\n)[\n\f\r])' WS = fr'(?:[ \t]|{NEWLINE})' # Comments -COMMENTS = r'(?:/\*[^*]*\*+(?:[^/*][^*]*\*+)*/)' +COMMENTS = r'(?:/\*(?:[^*]|\*(?!/))*\*/)' # Whitespace with comments included WSC = fr'(?:{WS}|{COMMENTS})' # CSS escapes @@ -112,13 +120,13 @@ CSS_STRING_ESCAPES = fr'(?:\\(?:[a-f0-9]{{1,6}}{WS}?|[^\r\n\f]|$|{NEWLINE}))' # CSS Identifier IDENTIFIER = fr''' -(?:(?:-?(?:[^\x00-\x2f\x30-\x40\x5B-\x5E\x60\x7B-\x9f]|{CSS_ESCAPES})+|--) +(?:(?:--|-?(?:[^\x00-\x2f\x30-\x40\x5B-\x5E\x60\x7B-\x9f]|{CSS_ESCAPES})) (?:[^\x00-\x2c\x2e\x2f\x3A-\x40\x5B-\x5E\x60\x7B-\x9f]|{CSS_ESCAPES})*) ''' # `nth` content NTH = fr'(?:[-+])?(?:[0-9]+n?|n)(?:(?<=n){WSC}*(?:[-+]){WSC}*(?:[0-9]+))?' # Value: quoted string or identifier -VALUE = fr'''(?:"(?:\\(?:.|{NEWLINE})|[^\\"\r\n\f]+)*?"|'(?:\\(?:.|{NEWLINE})|[^\\'\r\n\f]+)*?'|{IDENTIFIER})''' +VALUE = fr'''(?:"(?:\\(?:.|{NEWLINE})|[^\\"\r\n\f])*?"|'(?:\\(?:.|{NEWLINE})|[^\\'\r\n\f])*?'|{IDENTIFIER})''' # Attribute value comparison. `!=` is handled special as it is non-standard. ATTR = fr'(?:{WSC}*(?P[!~^|*$]?=){WSC}*(?P{VALUE})(?:{WSC}*(?P[is]))?)?{WSC}*' @@ -175,8 +183,9 @@ # Whitespace checks RE_WS = re.compile(WS) RE_WS_BEGIN = re.compile(fr'^{WSC}*') -RE_WS_END = re.compile(fr'{WSC}*$') +RE_WS_END = re.compile(fr'^(?:[ \t]|(?:\n\r|(?!\n\r)[\n\f\r])|{COMMENTS})*') RE_CUSTOM = re.compile(fr'^{PAT_PSEUDO_CLASS_CUSTOM}$', re.X) +RE_PSEUDO_CLASS_SPECIAL = re.compile(PAT_PSEUDO_CLASS_SPECIAL, re.I | re.X | re.U) # Constants # List split token @@ -307,7 +316,17 @@ def __init__(self, name: str, pattern: str) -> None: """Initialize.""" self.name = name - self.re_pattern = re.compile(pattern, re.I | re.X | re.U) + self.pattern = pattern + self._re_pattern: re.Pattern[str] | None = None + + @property + def re_pattern(self) -> re.Pattern[str]: + """Retrieve the compiled regular expression pattern.""" + + with RE_LOCK: + if self._re_pattern is None: + self._re_pattern = re.compile(self.pattern, re.I | re.X | re.U) + return self._re_pattern def get_name(self) -> str: """Get name.""" @@ -334,7 +353,6 @@ def __init__(self, patterns: tuple[tuple[str, tuple[str, ...], str, type[Selecto self.patterns[pseudo] = pattern self.matched_name = None # type: SelectorPattern | None - self.re_pseudo_name = re.compile(PAT_PSEUDO_CLASS_SPECIAL, re.I | re.X | re.U) def get_name(self) -> str: """Get name.""" @@ -345,7 +363,7 @@ def match(self, selector: str, index: int, flags: int) -> Match[str] | None: """Match the selector.""" pseudo = None - m = self.re_pseudo_name.match(selector, index) + m = RE_PSEUDO_CLASS_SPECIAL.match(selector, index) if m: name = util.lower(css_unescape(m.group('name'))) pattern = self.patterns.get(name) @@ -425,10 +443,187 @@ def __str__(self) -> str: # pragma: no cover __repr__ = __str__ +@dataclass +class CSSPattern: + """A CSS pattern that hasn't been processed by `CSSParser` yet.""" + + selector: str + flags: int + + +class PseudoSelectorMap(UserDict[str, CSSPattern | ct.SelectorList]): + """Pseudo selector map.""" + + def __setitem__(self, key: str, value: CSSPattern | ct.SelectorList) -> None: + """Set item.""" + + self.data[key] = value + + def __getitem__(self, key: str) -> ct.SelectorList: + """Get item.""" + + with SEL_LOCK: + value = self.data[key] + if isinstance(value, CSSPattern): + value = CSSParser(value.selector).process_selectors(flags=value.flags) + self.data[key] = value + + return value + + +# CSS pattern for `:link` and `:any-link` +CSS_LINK = CSSPattern('html|*:is(a, area)[href]', FLG_PSEUDO | FLG_HTML) +# CSS pattern for `:checked` +CSS_CHECKED = CSSPattern( + ''' + html|*:is(input[type=checkbox], input[type=radio])[checked], html|option[selected] + ''', + FLG_PSEUDO | FLG_HTML +) +# CSS pattern for `:default` (must compile CSS_CHECKED first) +CSS_DEFAULT = CSSPattern( + ''' + :checked, + + /* + This pattern must be at the end. + Special logic is applied to the last selector. + */ + html|form html|*:is(button, input)[type="submit"] + ''', + FLG_PSEUDO | FLG_HTML | FLG_DEFAULT +) +# CSS pattern for `:indeterminate` +CSS_INDETERMINATE = CSSPattern( + ''' + html|input[type="checkbox"][indeterminate], + html|input[type="radio"]:is(:not([name]), [name=""]):not([checked]), + html|progress:not([value]), + + /* + This pattern must be at the end. + Special logic is applied to the last selector. + */ + html|input[type="radio"][name]:not([name='']):not([checked]) + ''', + FLG_PSEUDO | FLG_HTML | FLG_INDETERMINATE +) +# CSS pattern for `:disabled` +CSS_DISABLED = CSSPattern( + ''' + html|*:is(input:not([type=hidden]), button, select, textarea, fieldset, optgroup, option, fieldset)[disabled], + html|optgroup[disabled] > html|option, + html|fieldset[disabled] > html|*:is(input:not([type=hidden]), button, select, textarea, fieldset), + html|fieldset[disabled] > + html|*:not(legend:nth-of-type(1)) html|*:is(input:not([type=hidden]), button, select, textarea, fieldset) + ''', + FLG_PSEUDO | FLG_HTML +) +# CSS pattern for `:enabled` +CSS_ENABLED = CSSPattern( + ''' + html|*:is(input:not([type=hidden]), button, select, textarea, fieldset, optgroup, option, fieldset):not(:disabled) + ''', + FLG_PSEUDO | FLG_HTML +) +# CSS pattern for `:required` +CSS_REQUIRED = CSSPattern('html|*:is(input, textarea, select)[required]', FLG_PSEUDO | FLG_HTML) +# CSS pattern for `:optional` +CSS_OPTIONAL = CSSPattern('html|*:is(input, textarea, select):not([required])', FLG_PSEUDO | FLG_HTML) +# CSS pattern for `:placeholder-shown` +CSS_PLACEHOLDER_SHOWN = CSSPattern( + ''' + html|input:is( + :not([type]), + [type=""], + [type=text], + [type=search], + [type=url], + [type=tel], + [type=email], + [type=password], + [type=number] + )[placeholder]:not([placeholder='']):is(:not([value]), [value=""]), + html|textarea[placeholder]:not([placeholder='']) + ''', + FLG_PSEUDO | FLG_HTML | FLG_PLACEHOLDER_SHOWN +) +# CSS pattern for `:read-write` (CSS_DISABLED must be compiled first) +CSS_READ_WRITE = CSSPattern( + ''' + html|*:is( + textarea, + input:is( + :not([type]), + [type=""], + [type=text], + [type=search], + [type=url], + [type=tel], + [type=email], + [type=number], + [type=password], + [type=date], + [type=datetime-local], + [type=month], + [type=time], + [type=week] + ) + ):not([readonly], :disabled), + html|*:is([contenteditable=""], [contenteditable="true" i]) + ''', + FLG_PSEUDO | FLG_HTML +) +# CSS pattern for `:read-only` +CSS_READ_ONLY = CSSPattern('html|*:not(:read-write)', FLG_PSEUDO | FLG_HTML) +# CSS pattern for `:in-range` +CSS_IN_RANGE = CSSPattern( + ''' + html|input:is( + [type="date"], + [type="month"], + [type="week"], + [type="time"], + [type="datetime-local"], + [type="number"], + [type="range"] + ):is( + [min], + [max] + ) + ''', + FLG_PSEUDO | FLG_HTML | FLG_IN_RANGE +) +# CSS pattern for `:out-of-range` +CSS_OUT_OF_RANGE = CSSPattern( + ''' + html|input:is( + [type="date"], + [type="month"], + [type="week"], + [type="time"], + [type="datetime-local"], + [type="number"], + [type="range"] + ):is( + [min], + [max] + ) + ''', + FLG_PSEUDO | FLG_HTML | FLG_OUT_OF_RANGE +) +# CSS pattern for :open +CSS_OPEN = CSSPattern('html|*:is(details, dialog)[open]', FLG_PSEUDO | FLG_HTML) +# CSS pattern for :muted +CSS_MUTED = CSSPattern('html|*:is(video, audio)[muted]', FLG_PSEUDO | FLG_HTML) +# CSS pattern default for `:nth-child` "of S" feature +CSS_NTH_OF_S_DEFAULT = CSSPattern("*|*", FLG_PSEUDO) + + class CSSParser: """Parse CSS selectors.""" - css_tokens = ( + CSS_TOKENS = ( SelectorPattern("pseudo_close", PAT_PSEUDO_CLOSE), SpecialPseudoPattern( ( @@ -456,6 +651,29 @@ class CSSParser: SelectorPattern("combine", PAT_COMBINE) ) + # Pseudos that expand to selectors + PSEUDO_SELECTORS = PseudoSelectorMap( + { + ':link': CSS_LINK, + ':any-link': CSS_LINK, + ':checked': CSS_CHECKED, + ':default': CSS_DEFAULT, + ':indeterminate': CSS_INDETERMINATE, + ':disabled': CSS_DISABLED, + ':enabled': CSS_ENABLED, + ':required': CSS_REQUIRED, + ':muted': CSS_MUTED, + ':open': CSS_OPEN, + ':optional': CSS_OPTIONAL, + ':read-only': CSS_READ_ONLY, + ':read-write': CSS_READ_WRITE, + ':in-range': CSS_IN_RANGE, + ':out-of-range': CSS_OUT_OF_RANGE, + ':placeholder-shown': CSS_PLACEHOLDER_SHOWN, + '': CSS_NTH_OF_S_DEFAULT + } + ) + def __init__( self, selector: str, @@ -468,6 +686,13 @@ def __init__( self.flags = flags self.debug = self.flags & util.DEBUG self.custom = {} if custom is None else custom + self.count = 0 + + def check_count(self) -> None: + """Check the current selector count.""" + + if self.count > SELECTOR_LIMIT: + raise ValueError(f'Selector exceeds pseudo-class nesting limit of {SELECTOR_LIMIT}') def parse_attribute_selector(self, sel: _Selector, m: Match[str], has_selector: bool) -> bool: """Create attribute selector from the returned regex match.""" @@ -572,6 +797,9 @@ def parse_pseudo_class_custom(self, sel: _Selector, m: Match[str], has_selector: ).process_selectors(flags=FLG_PSEUDO) self.custom[pseudo] = selector + self.count += selector.count + self.check_count() + sel.selectors.append(selector) has_selector = True return has_selector @@ -602,36 +830,11 @@ def parse_pseudo_class( sel.flags |= ct.SEL_SCOPE elif pseudo == ':empty': sel.flags |= ct.SEL_EMPTY - elif pseudo in (':link', ':any-link'): - sel.selectors.append(CSS_LINK) - elif pseudo == ':checked': - sel.selectors.append(CSS_CHECKED) - elif pseudo == ':default': - sel.selectors.append(CSS_DEFAULT) - elif pseudo == ':indeterminate': - sel.selectors.append(CSS_INDETERMINATE) - elif pseudo == ":disabled": - sel.selectors.append(CSS_DISABLED) - elif pseudo == ":enabled": - sel.selectors.append(CSS_ENABLED) - elif pseudo == ":required": - sel.selectors.append(CSS_REQUIRED) - elif pseudo == ":muted": - sel.selectors.append(CSS_MUTED) - elif pseudo == ":open": - sel.selectors.append(CSS_OPEN) - elif pseudo == ":optional": - sel.selectors.append(CSS_OPTIONAL) - elif pseudo == ":read-only": - sel.selectors.append(CSS_READ_ONLY) - elif pseudo == ":read-write": - sel.selectors.append(CSS_READ_WRITE) - elif pseudo == ":in-range": - sel.selectors.append(CSS_IN_RANGE) - elif pseudo == ":out-of-range": - sel.selectors.append(CSS_OUT_OF_RANGE) - elif pseudo == ":placeholder-shown": - sel.selectors.append(CSS_PLACEHOLDER_SHOWN) + elif pseudo in self.PSEUDO_SELECTORS: + pseudo_selector = self.PSEUDO_SELECTORS[pseudo] + self.count += pseudo_selector.count + self.check_count() + sel.selectors.append(pseudo_selector) elif pseudo == ':first-child': sel.nth.append(ct.SelectorNth(1, False, 0, False, False, ct.SelectorList())) elif pseudo == ':last-child': @@ -730,7 +933,9 @@ def parse_pseudo_nth( nth_sel = self.parse_selectors(iselector, m.end(0), FLG_PSEUDO | FLG_OPEN) else: # Use default `*|*` for `of S`. - nth_sel = CSS_NTH_OF_S_DEFAULT + nth_sel = self.PSEUDO_SELECTORS[''] + self.count += nth_sel.count + self.check_count() if pseudo_sel == ':nth-child': sel.nth.append(ct.SelectorNth(s1, var, s2, False, False, nth_sel)) elif pseudo_sel == ':nth-last-child': @@ -937,6 +1142,7 @@ def parse_selectors( closed = False relations = [] # type: list[_Selector] rel_type = ":" + WS_COMBINATOR + count = self.count # Setup various flags is_open = bool(flags & FLG_OPEN) @@ -984,6 +1190,10 @@ def parse_selectors( while True: key, m = next(iselector) + if key not in ('combine', 'pseudo_close'): + self.count += 1 + self.check_count() + # Handle parts if key == "at_rule": raise NotImplementedError(f"At-rules found at position {m.start(0)}") @@ -1103,7 +1313,7 @@ def parse_selectors( selectors[-1].flags = ct.SEL_PLACEHOLDER_SHOWN # Return selector list - return ct.SelectorList([s.freeze() for s in selectors], is_not, is_html) + return ct.SelectorList([s.freeze() for s in selectors], is_not, is_html, self.count - count) def selector_iter(self, pattern: str) -> Iterator[tuple[str, Match[str]]]: """Iterate selector tokens.""" @@ -1111,14 +1321,15 @@ def selector_iter(self, pattern: str) -> Iterator[tuple[str, Match[str]]]: # Ignore whitespace and comments at start and end of pattern m = RE_WS_BEGIN.search(pattern) index = m.end(0) if m else 0 - m = RE_WS_END.search(pattern) - end = (m.start(0) - 1) if m else (len(pattern) - 1) + m = RE_WS_END.search(pattern[::-1]) + offset = m.end(0) if m else 0 + end = len(pattern) - (1 + offset) if self.debug: # pragma: no cover print(f'## PARSING: {pattern!r}') while index <= end: m = None - for v in self.css_tokens: + for v in self.CSS_TOKENS: m = v.match(pattern, index, self.flags) if m: name = v.get_name() @@ -1150,169 +1361,3 @@ def process_selectors(self, index: int = 0, flags: int = 0) -> ct.SelectorList: """Process selectors.""" return self.parse_selectors(self.selector_iter(self.pattern), index, flags) - - -# Precompile CSS selector lists for pseudo-classes (additional logic may be required beyond the pattern) -# A few patterns are order dependent as they use patterns previous compiled. - -# CSS pattern for `:link` and `:any-link` -CSS_LINK = CSSParser( - 'html|*:is(a, area)[href]' -).process_selectors(flags=FLG_PSEUDO | FLG_HTML) -# CSS pattern for `:checked` -CSS_CHECKED = CSSParser( - ''' - html|*:is(input[type=checkbox], input[type=radio])[checked], html|option[selected] - ''' -).process_selectors(flags=FLG_PSEUDO | FLG_HTML) -# CSS pattern for `:default` (must compile CSS_CHECKED first) -CSS_DEFAULT = CSSParser( - ''' - :checked, - - /* - This pattern must be at the end. - Special logic is applied to the last selector. - */ - html|form html|*:is(button, input)[type="submit"] - ''' -).process_selectors(flags=FLG_PSEUDO | FLG_HTML | FLG_DEFAULT) -# CSS pattern for `:indeterminate` -CSS_INDETERMINATE = CSSParser( - ''' - html|input[type="checkbox"][indeterminate], - html|input[type="radio"]:is(:not([name]), [name=""]):not([checked]), - html|progress:not([value]), - - /* - This pattern must be at the end. - Special logic is applied to the last selector. - */ - html|input[type="radio"][name]:not([name='']):not([checked]) - ''' -).process_selectors(flags=FLG_PSEUDO | FLG_HTML | FLG_INDETERMINATE) -# CSS pattern for `:disabled` -CSS_DISABLED = CSSParser( - ''' - html|*:is(input:not([type=hidden]), button, select, textarea, fieldset, optgroup, option, fieldset)[disabled], - html|optgroup[disabled] > html|option, - html|fieldset[disabled] > html|*:is(input:not([type=hidden]), button, select, textarea, fieldset), - html|fieldset[disabled] > - html|*:not(legend:nth-of-type(1)) html|*:is(input:not([type=hidden]), button, select, textarea, fieldset) - ''' -).process_selectors(flags=FLG_PSEUDO | FLG_HTML) -# CSS pattern for `:enabled` -CSS_ENABLED = CSSParser( - ''' - html|*:is(input:not([type=hidden]), button, select, textarea, fieldset, optgroup, option, fieldset):not(:disabled) - ''' -).process_selectors(flags=FLG_PSEUDO | FLG_HTML) -# CSS pattern for `:required` -CSS_REQUIRED = CSSParser( - 'html|*:is(input, textarea, select)[required]' -).process_selectors(flags=FLG_PSEUDO | FLG_HTML) -# CSS pattern for `:optional` -CSS_OPTIONAL = CSSParser( - 'html|*:is(input, textarea, select):not([required])' -).process_selectors(flags=FLG_PSEUDO | FLG_HTML) -# CSS pattern for `:placeholder-shown` -CSS_PLACEHOLDER_SHOWN = CSSParser( - ''' - html|input:is( - :not([type]), - [type=""], - [type=text], - [type=search], - [type=url], - [type=tel], - [type=email], - [type=password], - [type=number] - )[placeholder]:not([placeholder='']):is(:not([value]), [value=""]), - html|textarea[placeholder]:not([placeholder='']) - ''' -).process_selectors(flags=FLG_PSEUDO | FLG_HTML | FLG_PLACEHOLDER_SHOWN) -# CSS pattern default for `:nth-child` "of S" feature -CSS_NTH_OF_S_DEFAULT = CSSParser( - '*|*' -).process_selectors(flags=FLG_PSEUDO) -# CSS pattern for `:read-write` (CSS_DISABLED must be compiled first) -CSS_READ_WRITE = CSSParser( - ''' - html|*:is( - textarea, - input:is( - :not([type]), - [type=""], - [type=text], - [type=search], - [type=url], - [type=tel], - [type=email], - [type=number], - [type=password], - [type=date], - [type=datetime-local], - [type=month], - [type=time], - [type=week] - ) - ):not([readonly], :disabled), - html|*:is([contenteditable=""], [contenteditable="true" i]) - ''' -).process_selectors(flags=FLG_PSEUDO | FLG_HTML) -# CSS pattern for `:read-only` -CSS_READ_ONLY = CSSParser( - ''' - html|*:not(:read-write) - ''' -).process_selectors(flags=FLG_PSEUDO | FLG_HTML) -# CSS pattern for `:in-range` -CSS_IN_RANGE = CSSParser( - ''' - html|input:is( - [type="date"], - [type="month"], - [type="week"], - [type="time"], - [type="datetime-local"], - [type="number"], - [type="range"] - ):is( - [min], - [max] - ) - ''' -).process_selectors(flags=FLG_PSEUDO | FLG_IN_RANGE | FLG_HTML) -# CSS pattern for `:out-of-range` -CSS_OUT_OF_RANGE = CSSParser( - ''' - html|input:is( - [type="date"], - [type="month"], - [type="week"], - [type="time"], - [type="datetime-local"], - [type="number"], - [type="range"] - ):is( - [min], - [max] - ) - ''' -).process_selectors(flags=FLG_PSEUDO | FLG_OUT_OF_RANGE | FLG_HTML) - -# CSS pattern for :open -CSS_OPEN = CSSParser( - ''' - html|*:is(details, dialog)[open] - ''' -).process_selectors(flags=FLG_PSEUDO | FLG_HTML) - - -# CSS pattern for :muted -CSS_MUTED = CSSParser( - ''' - html|*:is(video, audio)[muted] - ''' -).process_selectors(flags=FLG_PSEUDO | FLG_HTML) diff --git a/soupsieve/css_types.py b/soupsieve/css_types.py index b69fe8cf..f1af46ef 100644 --- a/soupsieve/css_types.py +++ b/soupsieve/css_types.py @@ -351,24 +351,27 @@ def __getitem__(self, index: int) -> str: # pragma: no cover class SelectorList(Immutable): """Selector list.""" - __slots__ = ("selectors", "is_not", "is_html", "_hash") + __slots__ = ("selectors", "is_not", "is_html", "count", "_hash") selectors: tuple[Selector | SelectorNull, ...] is_not: bool is_html: bool + count: int def __init__( self, selectors: Iterable[Selector | SelectorNull] | None = None, is_not: bool = False, - is_html: bool = False + is_html: bool = False, + count: int = 0, ) -> None: """Initialize.""" super().__init__( selectors=tuple(selectors) if selectors is not None else (), is_not=is_not, - is_html=is_html + is_html=is_html, + count=count ) def __iter__(self) -> Iterator[Selector | SelectorNull]: diff --git a/tests/test_api.py b/tests/test_api.py index 1ef84021..b60fc231 100644 --- a/tests/test_api.py +++ b/tests/test_api.py @@ -592,6 +592,38 @@ def test_invalid_type_input_filter(self): with self.assertRaises(TypeError): sv.filter('div', "not a tag", flags=flags) + def test_excessive_selectors(self): + """Test excessive selectors.""" + + # Build a 500 KB selector string: "a,a,a,...,a" (250,000 items) + count = 10000 + selector = ",".join("a" for _ in range(count)) + + # Compile the selector + with self.assertRaises(ValueError): + sv.compile(selector) + + def test_excessive_custom_selectors(self): + """Test excessive custom selectors.""" + + # Build a 500 KB selector string: "a,a,a,...,a" (250,000 items) + count = 10000 + selector = ",".join("a" for _ in range(count)) + + # Compile the selector + with self.assertRaises(ValueError): + sv.compile('div:--custom', custom={':--custom': selector}) + + def test_excessive_custom_and_normal_selectors(self): + """Test excessive custom and normal selectors.""" + + count = 5000 + selector = ",".join("a" for _ in range(count)) + + # Compile the selector + with self.assertRaises(ValueError): + sv.compile(f':is({selector}):--custom', custom={':--custom': selector}) + class TestSyntaxErrorReporting(util.TestCase): """Test reporting of syntax errors.""" diff --git a/tests/test_extra/test_attribute.py b/tests/test_extra/test_attribute.py index 08d96f24..814ced30 100644 --- a/tests/test_extra/test_attribute.py +++ b/tests/test_extra/test_attribute.py @@ -62,3 +62,31 @@ def test_bad_attribute(self): self.assertEqual(e.context, '[\\]!=D4XXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXX\n^') self.assertEqual(e.line, 1) self.assertEqual(e.col, 1) + + def test_bad_attribute_unclused(self): + """Test bad attribute fails for syntax error, not timeout error.""" + + import platform + + if platform.system() == 'Windows': + with self.assertRaises(sv.SelectorSyntaxError): + sv.compile('[a="' + ('x' * 300)) + else: + import signal + + def timeout_handler(signum, frame): + raise TimeoutError + + signal.signal(signal.SIGALRM, timeout_handler) + signal.alarm(3) + + passed = False + try: + with self.assertRaises(sv.SelectorSyntaxError): + sv.compile('[a="' + ('x' * 300)) + passed = True + except TimeoutError: + pass + finally: + signal.alarm(0) + self.assertTrue(passed) diff --git a/tests/test_level3/test_nth_child.py b/tests/test_level3/test_nth_child.py index f2d0c215..4b96ba4c 100644 --- a/tests/test_level3/test_nth_child.py +++ b/tests/test_level3/test_nth_child.py @@ -244,3 +244,30 @@ def test_nth_child_with_bad_parameters(self): """Test that pseudo class fails with bad parameters (basically it doesn't match).""" self.assert_raises(':nth-child(a)', SelectorSyntaxError) + + def test_nth_child_an_plus_b_boundaries(self): + """Test `An+B` forms whose sequence steps onto index 0 or the last child.""" + + markup = """ + + + + + + + + """ + + # `B < 1` sequences whose first valid index is > 1 or is all indices + self.assert_selector(markup, "i:nth-child(2n-2)", ['1', '3'], flags=util.HTML) + self.assert_selector( + markup, "i:nth-child(n-1)", ['0', '1', '2', '3', '4'], flags=util.HTML + ) + # `B` equal to the child count (index lands on the last child) + self.assert_selector(markup, "i:nth-child(n+5)", ['4'], flags=util.HTML) + self.assert_selector(markup, "i:nth-child(2n+5)", ['4'], flags=util.HTML) + # same boundaries, counted from the end + self.assert_selector( + markup, "i:nth-last-child(2n-2)", ['1', '3'], flags=util.HTML + ) + self.assert_selector(markup, "i:nth-last-child(n+5)", ['0'], flags=util.HTML) diff --git a/tests/test_performance.py b/tests/test_performance.py new file mode 100644 index 00000000..6bef8da0 --- /dev/null +++ b/tests/test_performance.py @@ -0,0 +1,61 @@ +"""Test performance cases.""" +import unittest +import sys +import signal +import time +import soupsieve as sv +from soupsieve.util import SelectorSyntaxError + + +class Timeout(Exception): + """Timeout exception.""" + + +@unittest.skipUnless(not sys.platform.startswith('win'), "Unix/Linux test") +class TestPerformance(unittest.TestCase): + """Test performance.""" + + def assert_performance(self, selector): + """Test performance of of specific cases.""" + + LIMIT = 5.0 + + signal.signal(signal.SIGALRM, lambda *_: (_ for _ in ()).throw(Timeout())) + signal.setitimer(signal.ITIMER_REAL, LIMIT) + dt = None + t = 0 + try: + t = time.perf_counter() + sv.compile(selector) + dt = time.perf_counter() - t + except SelectorSyntaxError: + dt = time.perf_counter() - t + except Timeout: + pass + finally: + signal.setitimer(signal.ITIMER_REAL, 0) + + hung = dt is None + success = not hung and dt < 1.0 + + if not success: + print('Selector:', selector) + print(f"{'> %.0f s (HANG)' % LIMIT if hung else '%.3f s' % dt}") + + self.assertTrue(success) + + def test_performance_caase(self): + """Test performance cases.""" + + for n in (1000, 2000, 4000, 8000): + self.assert_performance("[a=" + "a" * n) + + for n in (2000, 4000, 8000, 16000): + self.assert_performance("a" * n + "!") + + self.assert_performance("[a=" + "a" * 12000) + + for n in (2000, 4000, 8000, 16000): + self.assert_performance("a" + " " * n + "b") + + self.assert_performance("a" + " " * 20000 + "b")