Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
19 changes: 8 additions & 11 deletions docs/ignore.md
Original file line number Diff line number Diff line change
Expand Up @@ -49,10 +49,6 @@ Minecraft/Website
!!! note ""
_This section sourced and adapted from Git's[^1] `.gitignore` [documentation](https://git-scm.com/docs/gitignore)._

### Internal Processes

When scanning your library directories, the `.ts_ignore` file is read by either the [`wcmatch`](https://facelessuser.github.io/wcmatch/glob/) library or [`ripgrep`](https://github.com/BurntSushi/ripgrep) in glob mode depending if you have the later installed on your system and it's detected by TagStudio. Ripgrep is the preferred method for scanning directories due to its improved performance and identical pattern matching to `.gitignore`. This mixture of tools may lead to slight inconsistencies if not using `ripgrep`.

---

### Comments ( `#` )
Expand Down Expand Up @@ -142,10 +138,6 @@ A `!` prefix before a pattern negates the pattern, allowing any files matched ma
```
<!-- prettier-ignore-end -->

<!-- prettier-ignore -->
!!! bug "Directory Exclusion Negation"
TagStudio attempts to match the behavior of a `.gitignore` file 1:1, however if you don't have `ripgrep` installed on your system and TagStudio falls back to its internal pattern matcher, excluded directories can be overwritten by further negations, unlike `.gitignore` behavior. Be wary that this is **not officially supported**, and this behavior may be removed at any time.

---

### Wildcards
Expand Down Expand Up @@ -275,18 +267,22 @@ Character sets and ranges are specific and powerful forms of wildcards that use
```
=== "Ignore all files EXCEPT .jpg files"
```toml
# Ignore everything to start,
# reinclude subfolders,
# then reinclude .jpg files located anywhere
*
!*/
!*.jpg
```
=== "Ignore all .jpg files in specific folders"
```toml
./Photos/Worst Vacation/*.jpg
Photos/Worst Vacation/*.jpg
Music/Artwork Art/*.jpg
```

<!-- prettier-ignore -->
!!! tip "Ensuring Complete Extension Matches"
For some filetypes, it may be nessisary to specify different casing and alternative spellings in order to match with all possible variations of an extension in your library.
For some filetypes, it may be necessary to specify different casing and alternative spellings in order to match with all possible variations of an extension in your library.

```toml title="Ignore (Most) Possible JPEG File Extensions"
# The JPEG Cinematic Universe
Expand All @@ -305,7 +301,8 @@ Character sets and ranges are specific and powerful forms of wildcards that use
<!-- prettier-ignore -->
=== "Ignore all "Cache" folders"
```toml
# Matches any folder called "Cache" no matter where it is in your library.
# Matches any folder called "cache"/"Cache" no matter where it is in your library.
Cache/
cache/
```
=== "Ignore a "Downloads" folder"
Expand Down
24 changes: 24 additions & 0 deletions src/tagstudio/core/library/alchemy/migrations.py
Original file line number Diff line number Diff line change
Expand Up @@ -662,3 +662,27 @@ def run(cls, conn: Connection, library_dir: Path, fmt_log: LoggingMethod):
"suffix = :suffix WHERE id = :id",
updates,
)

logger.info(fmt_log("Repairing 'reinclude folders' pattern in the .ts_ignore file..."))
cls._repair_reinclude_folders_pattern(library_dir)

@classmethod
def _repair_reinclude_folders_pattern(cls, library_dir: Path):
"""Add "!*/" following "*" lines in the `.ts_ignore` if one isn't already present.

Under wcmatch's rules paths could be reincluded after a "*" without requiring a
"!*/" after it, unlike the desired `.gitignore`-type behavior.
"""
ts_ignore = library_dir / TS_FOLDER_NAME / IGNORE_NAME
if not ts_ignore.exists():
return

# If "!*/" already follows "*", do nothing and return
lines = ts_ignore.read_text(encoding="utf8").splitlines()
patterns = [line.rstrip() for line in lines]
if "*" not in patterns or "!*/" in patterns:
return

# Add "!*/" after the first "*" found, if any
lines.insert(patterns.index("*") + 1, "!*/")
ts_ignore.write_text("\n".join(lines) + "\n", encoding="utf8")
Original file line number Diff line number Diff line change
Expand Up @@ -36,10 +36,7 @@ def refresh_ignored_entries(self) -> Iterator[int]:

for i, entry in enumerate(self.lib.all_entries()):
yield i
if not Ignore.compiled_patterns:
# If the compiled_patterns has malfunctioned, don't consider that a false positive
yield i
elif Ignore.compiled_patterns.match(entry.path):
if Ignore.matcher.is_ignored(entry.path):
self.ignored_entries.append(entry)

def remove_ignored_entries(self) -> None:
Expand Down
151 changes: 74 additions & 77 deletions src/tagstudio/core/library/ignore.py
Original file line number Diff line number Diff line change
Expand Up @@ -3,6 +3,7 @@


from pathlib import Path
from typing import NamedTuple

import structlog
from wcmatch import glob
Expand All @@ -12,12 +13,10 @@

logger = structlog.get_logger()

PATH_GLOB_FLAGS: int = glob.GLOBSTARLONG | glob.DOTGLOB | glob.NEGATE
_RULE_FLAGS: int = glob.GLOBSTAR | glob.DOTGLOB


GLOBAL_IGNORE = [
# TagStudio -------------------
f"{TS_FOLDER_NAME}",
# Trash -----------------------
".Trash-*",
".Trash",
Expand All @@ -35,65 +34,52 @@
]


def ignore_to_glob(ignore_patterns: list[str]) -> list[str]:
"""Convert .gitignore-like patterns to Unix-like glob syntax.

Args:
ignore_patterns (list[str]): The .gitignore-like patterns to convert.
"""
glob_patterns: list[str] = list(ignore_patterns)
glob_patterns_remove: list[str] = []
additional_patterns: list[str] = []
root_patterns: list[str] = []

# Expand .gitignore patterns to mimic the same behavior with unix-like glob patterns.
for pattern in glob_patterns:
# Temporarily remove any exclusion character before processing
exclusion_char = ""
gp = pattern
if pattern.startswith("!"):
gp = pattern[1:]
exclusion_char = "!"

if not gp.startswith("**/") and not gp.startswith("*/") and not gp.startswith("/"):
# Create a version of a prefix-less pattern that starts with "**/"
gp = "**/" + gp
additional_patterns.append(exclusion_char + gp)

gp = gp.removeprefix("**/").removeprefix("*/")
additional_patterns.append(exclusion_char + gp)

elif gp.startswith("/"):
# Matches "/file" case for .gitignore behavior where it should only match
# a file or folder in the root directory and nowhere else.
glob_patterns_remove.append(pattern)
gp = gp.lstrip("/")
root_patterns.append(exclusion_char + gp)

remove_set = set(glob_patterns_remove)
glob_patterns = [p for p in glob_patterns if p not in remove_set]
# root_patterns must be merged in before the "/**" suffix pass below, otherwise a rooted
# directory pattern (e.g. "/Downloads/") never gets a "/**" variant and matches nothing.
glob_patterns = glob_patterns + additional_patterns + root_patterns

# Add "/**" suffix to suffix-less patterns to match implicit .gitignore behavior.
for pattern in list(glob_patterns):
if pattern.endswith("/**"):
continue

glob_patterns.append(pattern.removesuffix("/*").removesuffix("/") + "/**")

# Fix wcmatch interpreting "**" as "one or more" to be a .gitignore style "zero or more".
# Otherwise "**/foo" won't match a root "foo" and "a/**/b" won't match match "a/b".
for pattern in list(glob_patterns):
collapsed = pattern.removeprefix("**/").replace("/**/", "/")
if collapsed != pattern:
glob_patterns.append(collapsed)

glob_patterns = list(dict.fromkeys(glob_patterns)) # Ordered deduplication

logger.info("[Ignore]", glob_patterns=glob_patterns)
return glob_patterns
class _Rule(NamedTuple):
matcher: glob.WcMatcher
negated: bool
dir_only: bool
name_only: bool


def _parse_rule(pattern: str) -> _Rule | None:
negated = pattern.startswith("!")
pattern = pattern.removeprefix("!")
dir_only = pattern.endswith("/")
pattern = pattern.rstrip("/")
if not pattern:
return None
# A slashed pattern is relative to the root, otherwise it matches any name
name_only = "/" not in pattern
pattern = pattern.removeprefix("/")
return _Rule(glob.compile(pattern, flags=_RULE_FLAGS), negated, dir_only, name_only)


class IgnoreMatcher:
"""Matches paths relative to the library against .gitignore-style patterns."""

def __init__(self, patterns: list[str]) -> None:
self._rules = [rule for pattern in patterns if (rule := _parse_rule(pattern))]
self._folder_verdicts: dict[str, bool] = {}

def match(self, path: str, is_dir: bool) -> bool:
"""Whether `path` is ignored, without checking its parent folders."""
name = path.rpartition("/")[2]
for rule in reversed(self._rules):
target = name if rule.name_only else path
if (is_dir or not rule.dir_only) and rule.matcher.match(target):
return not rule.negated
return False

def is_ignored(self, path: Path | str) -> bool:
"""Whether the file at `path` is ignored, including by any ignored parent folder."""
parts = Path(path).as_posix().split("/")
for depth in range(1, len(parts)):
folder = "/".join(parts[:depth])
if folder not in self._folder_verdicts:
self._folder_verdicts[folder] = self.match(folder, is_dir=True)
if self._folder_verdicts[folder]:
return True
return self.match("/".join(parts), is_dir=False)


def migrate_ext_list(exts: list[str], is_exclude_list: bool) -> str:
Expand All @@ -108,7 +94,7 @@ def migrate_ext_list(exts: list[str], is_exclude_list: bool) -> str:
prefix = ""
if not is_exclude_list:
prefix = "!"
out += "*\n"
out += "*\n!*/\n"
out += "\n".join([f"{prefix}*.{x.lstrip('.')}\n" for x in exts])
return out

Expand All @@ -129,8 +115,8 @@ class Ignore(metaclass=Singleton):
"""Class for processing and managing glob-like file ignore file patterns."""

_last_loaded: tuple[Path, float] | None = None
_patterns: list[str] = []
compiled_patterns: glob.WcMatcher | None = None
_patterns: list[str] = [*GLOBAL_IGNORE, TS_FOLDER_NAME]
matcher: IgnoreMatcher = IgnoreMatcher(_patterns)

@staticmethod
def read_ignore_file(library_dir: Path) -> list[str]:
Expand Down Expand Up @@ -165,46 +151,57 @@ def write_ignore_file(library_dir: Path, lines: list[str]) -> None:
f.writelines(lines)

@staticmethod
def get_patterns(library_dir: Path, include_global: bool = True) -> list[str]:
def get_patterns(
library_dir: Path,
include_global: bool = True,
update_state: bool = True,
) -> list[str]:
"""Get the ignore patterns for the given library directory.

The library's .TagStudio folder always comes last so it doesn't get reincluded.

Args:
library_dir (Path): The path of the library to load patterns from.
include_global (bool): Flag for including the global ignore set.
In most scenarios, this should be True.
include_global (bool): Flag for including the global ignore list.
update_state (bool): Flag for also loading the patterns into the class's state.
Should be True outside of exceptions that may include tests, migrations, etc.
"""
patterns = GLOBAL_IGNORE if include_global else []
global_patterns = GLOBAL_IGNORE if include_global else []
ts_ignore_path = Path(library_dir / TS_FOLDER_NAME / IGNORE_NAME)

# Return computed patterns if the state of the Ignore singleton shouldn't be updated.
if not update_state:
return [*global_patterns, *Ignore._load_ignore_file(ts_ignore_path), TS_FOLDER_NAME]

# Return default internal patterns if no .ts_ignore exists.
if not ts_ignore_path.exists():
logger.info(
"[Ignore] No .ts_ignore file found",
path=ts_ignore_path,
)
Ignore._last_loaded = None
Ignore._patterns = patterns
Ignore._patterns = [*global_patterns, TS_FOLDER_NAME]
Ignore.matcher = IgnoreMatcher(Ignore._patterns)

return Ignore._patterns

# Process the .ts_ignore file if the previous result is non-existent or outdated.
loaded = (ts_ignore_path, ts_ignore_path.stat().st_mtime)
if not Ignore._last_loaded or (Ignore._last_loaded and Ignore._last_loaded != loaded):
if Ignore._last_loaded != loaded:
logger.info(
"[Ignore] Processing the .ts_ignore file...",
library=library_dir,
last_mtime=Ignore._last_loaded[1] if Ignore._last_loaded else None,
new_mtime=loaded[1],
)
Ignore._patterns = patterns + Ignore._load_ignore_file(ts_ignore_path)
Ignore.compiled_patterns = glob.compile(
patterns=ignore_to_glob(Ignore._patterns),
flags=PATH_GLOB_FLAGS,
)
user_patterns = Ignore._load_ignore_file(ts_ignore_path)
Ignore._patterns = [*global_patterns, *user_patterns, TS_FOLDER_NAME]
Ignore.matcher = IgnoreMatcher(Ignore._patterns)
else:
logger.info(
"[Ignore] No updates to the .ts_ignore detected",
library=library_dir,
last_mtime=Ignore._last_loaded[1],
last_mtime=loaded[1],
new_mtime=loaded[1],
)
Ignore._last_loaded = loaded
Expand Down
9 changes: 4 additions & 5 deletions src/tagstudio/core/library/scanners.py
Original file line number Diff line number Diff line change
Expand Up @@ -10,10 +10,9 @@
from pathlib import Path

import structlog
from wcmatch import glob

from tagstudio.core.constants import TS_FOLDER_NAME
from tagstudio.core.library.ignore import PATH_GLOB_FLAGS, ignore_to_glob
from tagstudio.core.library.ignore import IgnoreMatcher
from tagstudio.core.utils.ripgrep_status import RipgrepStatus
from tagstudio.core.utils.silent_subprocess import silent_popen # pyright: ignore

Expand Down Expand Up @@ -99,7 +98,7 @@ def _scan_with_ripgrep(scan_dir: Path, ignore_patterns: list[str]) -> Iterator[P
def _scan_with_internal_scanner(scan_dir: Path, ignore_patterns: list[str]) -> Iterator[Path]:
"""Scan for files with the internal scanner."""
logger.info("[Scanners] Using internal scanner for scanning", path=scan_dir)
matcher = glob.compile(patterns=ignore_to_glob(ignore_patterns), flags=PATH_GLOB_FLAGS)
matcher = IgnoreMatcher(ignore_patterns)

def walk(dir_path: Path, ancestors: frozenset[str]) -> Iterator[Path]:
try:
Expand All @@ -110,9 +109,9 @@ def walk(dir_path: Path, ancestors: frozenset[str]) -> Iterator[Path]:

for item in dir_items:
rel = Path(item.path).relative_to(scan_dir)
if matcher.match(rel.as_posix()):
continue
try:
if matcher.match(rel.as_posix(), is_dir=item.is_dir()):
continue
item_stat = item.stat(follow_symlinks=True)
except OSError:
continue
Expand Down
13 changes: 9 additions & 4 deletions src/tagstudio/core/library/sync.py
Original file line number Diff line number Diff line change
Expand Up @@ -105,16 +105,21 @@ def sync_dir(
sleep(0)
start_time_loop = time()

# NOTE: Ignored files are skipped during the scan.
if not self.cancelled:
for entry in self.library.get_entries([cache[key] for key in unvisited]):
if self.cancelled:
break
if not (library_dir / entry.path).is_file():
self.unlinked_entries.append(entry)

if self.cancelled:
yield count, len(self.new_paths)
logger.info("[Sync] Directory scan cancelled", path=library_dir, files_scanned=count)
return

unlinked_ids = {cache[key] for key in unvisited}
if self.library.duplicate_path_entry_ids:
unlinked_ids.update(self.library.duplicate_path_entry_ids)
if unlinked_ids:
self.unlinked_entries = self.library.get_entries(list(unlinked_ids))
self.unlinked_entries += self.library.get_entries(self.library.duplicate_path_entry_ids)

if self.unlinked_entries:
yield -1, -1 # Signals the UI that repair work is starting
Expand Down
8 changes: 2 additions & 6 deletions src/tagstudio/previews/file_renderer.py
Original file line number Diff line number Diff line change
Expand Up @@ -656,12 +656,8 @@ def fetch_cached_image(file_name: Path):

# Check if the file is supposed to be ignored and render an overlay if needed
try:
if (
image
and Ignore.compiled_patterns
and Ignore.compiled_patterns.match(
filepath.relative_to(unwrap(self.lib.library_dir))
)
if image and Ignore.matcher.is_ignored(
filepath.relative_to(unwrap(self.lib.library_dir))
):
image = render_ignored((scaled_size, scaled_size), image)
except TypeError:
Expand Down
Loading
Loading