Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
2 changes: 2 additions & 0 deletions CHANGES_1.in.rst
Original file line number Diff line number Diff line change
Expand Up @@ -33,6 +33,7 @@ Bug fixes:
- `Issue #146`_: pattern_to_regex raises on three lines git accepts (bare '!', '! ', lone '\').
- `Pull #147`_: Match directory entries with directory-only patterns. `PathSpec.match_entries()` and `.match_tree_entries()` will now match directory entries as directory paths (i.e., with a trailing `/`), instead of as file paths (no trailing `/`).
- `Pull #149`_: Handle directory-open errors through on_error during tree walks.
- `Pull #152`_: Anchor gitignore filename matches at the actual end of the path.
- `Pull #156`_: Avoid following symbolic link targets when checking file types in `iter_tree_files(follow_links=False)` / `iter_tree_entries(follow_links=False)`.
- `Pull #158`_: Ignore unmatched exclusions in `detailed_match_files()`.
- `Pull #159`_: Preserve PathSpec state when backend reconstruction fails.
Expand All @@ -56,6 +57,7 @@ Bug fixes:
.. _`Pull #147`: https://github.com/cpburnz/python-pathspec/pull/147
.. _`Pull #149`: https://github.com/cpburnz/python-pathspec/pull/149
.. _`Pull #150`: https://github.com/cpburnz/python-pathspec/pull/150
.. _`Pull #152`: https://github.com/cpburnz/python-pathspec/pull/152
.. _`Pull #156`: https://github.com/cpburnz/python-pathspec/pull/156
.. _`Pull #158`: https://github.com/cpburnz/python-pathspec/pull/158
.. _`Pull #159`: https://github.com/cpburnz/python-pathspec/pull/159
Expand Down
25 changes: 24 additions & 1 deletion pathspec/_backends/_utils.py
Original file line number Diff line number Diff line change
Expand Up @@ -5,17 +5,40 @@
contents and structure are likely to change.
"""

import re
from collections.abc import (
Iterable)
from typing import (
TypeVar)
TypeVar,
Union,
overload)

from pathspec.pattern import (
Pattern)

TPattern = TypeVar("TPattern", bound=Pattern)


@overload
def translate_end_anchor(regex: str) -> str: ...


@overload
def translate_end_anchor(regex: bytes) -> bytes: ...


def translate_end_anchor(regex: Union[str, bytes]) -> Union[str, bytes]:
"""
Translate Python's strict end anchor to the RE2 and Hyperscan spelling.

Escaped backslashes are consumed as pairs, keeping literal ``\\Z`` names intact.
"""
if isinstance(regex, bytes):
return re.sub(rb'\\.', lambda match: rb'\z' if match[0] == rb'\Z' else match[0], regex)
else:
return re.sub(r'\\.', lambda match: r'\z' if match[0] == r'\Z' else match[0], regex)


def enumerate_patterns(
patterns: Iterable[TPattern],
filter: bool,
Expand Down
8 changes: 6 additions & 2 deletions pathspec/_backends/hyperscan/gitignore.py
Original file line number Diff line number Diff line change
Expand Up @@ -29,6 +29,9 @@
from pathspec._typing import (
override) # Added in 3.12.

from .._utils import (
translate_end_anchor)

from ._base import (
HS_FLAGS,
HS_VERSION,
Expand Down Expand Up @@ -140,18 +143,19 @@ def _init_db(
# direct match and not an excluded ancestor.
base_regex = regex_str[:-len(_DIR_MARK_OPT)]
use_regexes.append((f'{base_regex}/(?s:.)', True))
use_regexes.append((f'{base_regex}/?$', False))
use_regexes.append((rf'{base_regex}/?\z', False))
else:
# Remove capture group.
base_regex = regex_str.replace(_DIR_MARK_CG, '/')
use_regexes.append((f'{base_regex}(?s:.)', True))
use_regexes.append((f'{base_regex}$', False))
use_regexes.append((rf'{base_regex}\z', False))

if not use_regexes:
# No special case for regex.
use_regexes.append((regex, False))

for regex, is_dir_pattern in use_regexes:
regex = translate_end_anchor(regex)
if isinstance(regex, bytes):
regex_bytes = regex
else:
Expand Down
5 changes: 3 additions & 2 deletions pathspec/_backends/hyperscan/pathspec.py
Original file line number Diff line number Diff line change
Expand Up @@ -26,7 +26,8 @@
override) # Added in 3.12.

from .._utils import (
enumerate_patterns)
enumerate_patterns,
translate_end_anchor)

from .base import (
hyperscan_error)
Expand Down Expand Up @@ -156,7 +157,7 @@ def _init_db(

# Encode regex.
assert isinstance(pattern, RegexPattern), pattern
regex = pattern.regex.pattern
regex = translate_end_anchor(pattern.regex.pattern)

if isinstance(regex, bytes):
regex_bytes = regex
Expand Down
3 changes: 3 additions & 0 deletions pathspec/_backends/re2/gitignore.py
Original file line number Diff line number Diff line change
Expand Up @@ -29,6 +29,8 @@
from ._base import (
Re2RegexDat,
Re2RegexDebug)
from .._utils import (
translate_end_anchor)
from .pathspec import (
Re2PsBackend)

Expand Down Expand Up @@ -114,6 +116,7 @@ def _init_set(
use_regexes.append((regex, False))

for regex, is_dir_pattern in use_regexes:
regex = translate_end_anchor(regex)
if debug:
regex_data.append(Re2RegexDebug(
include=pattern.include,
Expand Down
5 changes: 3 additions & 2 deletions pathspec/_backends/re2/pathspec.py
Original file line number Diff line number Diff line change
Expand Up @@ -25,7 +25,8 @@
override) # Added in 3.12.

from .._utils import (
enumerate_patterns)
enumerate_patterns,
translate_end_anchor)

from .base import (
re2_error)
Expand Down Expand Up @@ -132,7 +133,7 @@ def _init_set(

assert pattern.regex is not None, pattern
assert isinstance(pattern, RegexPattern), pattern
regex = pattern.regex.pattern
regex = translate_end_anchor(pattern.regex.pattern)

if debug:
regex_data.append(Re2RegexDebug(
Expand Down
4 changes: 2 additions & 2 deletions pathspec/patterns/gitignore/basic.py
Original file line number Diff line number Diff line change
Expand Up @@ -374,13 +374,13 @@ def __translate_segments(
# A pattern ending with an asterisk ('*') will match a file or
# directory (without matching descendant paths). E.g., "foo/*"
# matches "foo/test.json", "foo/bar/", but not "foo/bar/hello.c".
out_parts.append('/?$')
out_parts.append(r'/?\Z')

else:
# A pattern ending without a slash ('/') will match a file or a
# directory (with paths underneath it). E.g., "foo" matches "foo",
# "foo/bar", "foo/bar/baz", etc.
out_parts.append('(?:/|$)')
out_parts.append(r'(?:/|\Z)')

need_slash = True

Expand Down
2 changes: 1 addition & 1 deletion pathspec/patterns/gitignore/spec.py
Original file line number Diff line number Diff line change
Expand Up @@ -38,7 +38,7 @@
This regular expression matches the directory marker.
"""

_DIR_MARK_OPT = f'(?:{_DIR_MARK_CG}|$)'
_DIR_MARK_OPT = rf'(?:{_DIR_MARK_CG}|\Z)'
"""
This regular expression matches the optional directory marker and sub-path.
"""
Expand Down
4 changes: 2 additions & 2 deletions tests/test_03_gitignore_basic.py
Original file line number Diff line number Diff line change
Expand Up @@ -18,7 +18,7 @@
from pathspec.util import (
lookup_pattern)

_DIR_OPT = '(?:/|$)'
_DIR_OPT = r'(?:/|\Z)'
"""
Optional directory ending.
"""
Expand Down Expand Up @@ -909,7 +909,7 @@ def test_b_files(self):
"""
pattern = GitIgnoreBasicPattern('!libfoo/*')

self.assertEqual(pattern.regex.pattern, f'^libfoo/[^/]+/?$')
self.assertEqual(pattern.regex.pattern, rf'^libfoo/[^/]+/?\Z')
self.assertIs(pattern.include, False)
self.assertTrue(pattern.match_file('libfoo/__init__.py'))

Expand Down
67 changes: 67 additions & 0 deletions tests/test_07_gitignore_end_anchor.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,67 @@
"""Test that a terminal newline remains part of a filename."""

import unittest

from pathspec import GitIgnoreSpec, PathSpec
from pathspec.patterns.gitignore.basic import GitIgnoreBasicPattern
from pathspec.patterns.gitignore.spec import GitIgnoreSpecPattern

from .util import require_backend


class EndAnchorTest(unittest.TestCase):

def test_pattern_filename_end(self):
for factory in (GitIgnoreBasicPattern, GitIgnoreSpecPattern):
for source, filename in (
('foo', 'foo'),
('foo', 'x/foo'),
('foo', 'x\ny/foo'),
('/foo', 'foo'),
('**/foo', 'x/foo'),
('foo?', 'foo1'),
('foo[0-9]', 'foo1'),
(r'foo\\Z', r'foo\Z'),
):
for as_bytes in (False, True):
with self.subTest(factory=factory, source=source, as_bytes=as_bytes):
pattern = factory(source.encode() if as_bytes else source)
path = filename.encode() if as_bytes else filename
newline = b'\n' if as_bytes else '\n'
self.assertIsNotNone(pattern.match_file(path))
self.assertIsNone(pattern.match_file(path + newline))

def test_backend_filename_end(self):
for backend in ('simple', 're2', 'hyperscan'):
with self.subTest(backend=backend):
require_backend(backend)
for factory in (GitIgnoreBasicPattern, GitIgnoreSpecPattern):
for as_bytes in (False, True):
if as_bytes and backend == 'simple':
continue # The simple backend requires string patterns for string paths.
with self.subTest(factory=factory, as_bytes=as_bytes):
line = b'foo' if as_bytes else 'foo'
spec = PathSpec.from_lines(factory, [line], backend=backend)
self.assertTrue(spec.match_file('foo'))
self.assertFalse(spec.match_file('foo\n'))
spec = GitIgnoreSpec.from_lines([line], backend=backend)
self.assertTrue(spec.match_file('foo'))
self.assertFalse(spec.match_file('foo\n'))

def test_negation_keeps_newline_filename(self):
for backend in ('simple', 're2', 'hyperscan'):
with self.subTest(backend=backend):
require_backend(backend)
spec = GitIgnoreSpec.from_lines(['*', '!foo'], backend=backend)
self.assertFalse(spec.match_file('foo'))
self.assertTrue(spec.match_file('foo\n'))

def test_backend_literal_backslash(self):
for backend in ('simple', 're2', 'hyperscan'):
with self.subTest(backend=backend):
require_backend(backend)
for factory in (GitIgnoreBasicPattern, GitIgnoreSpecPattern):
with self.subTest(factory=factory):
spec = PathSpec.from_lines(factory, [r'foo\\Z'], backend=backend)
self.assertTrue(spec.match_file(r'foo\Z', separators=('/',)))
self.assertFalse(spec.match_file(r'foo\z', separators=('/',)))
Loading