From 043157057270208c1e9a2c72b8ebced123f3d610 Mon Sep 17 00:00:00 2001 From: Thor Whalen <1906276+thorwhalen@users.noreply.github.com> Date: Tue, 22 Sep 2026 13:48:01 +0000 Subject: [PATCH] fix: filter_prefixes regex grouping bug + guard filt_iter's added __len__ Two of the three "related, same area" bugs from #82 (the main prefix-relativization corruption bug is a larger design question, left open -- see issue comment): - dol.trans.filter_prefixes built an unanchored/ungrouped regex ("^" + "|".join(...)) which for multiple prefixes like ['logs/', 'tmp/'] compiled to `^logs/|tmp/`, parsed as `(^logs/)|(tmp/)` -- so a string merely *containing* a later prefix anywhere (e.g. 'zzz/tmp/c') incorrectly matched. Grouped as `^(?:...)` to fix, matching how filter_suffixes already groups its own alternation. - dol.trans._filt_iter unconditionally added a counting __len__ to any wrapped class, even one that deliberately omits __len__. Now guarded with hasattr. Verified (and confirmed by independent review) this guard is currently a no-op on every path reachable through the public filt_iter() API, since the default wrapper=Store already provides Collection's own __len__ before _filt_iter runs, and filt_iter doesn't expose a wrapper= parameter -- so it's inert today but correct/defensive for any future or private caller of _wrap_store/_filt_iter with a custom wrapper. Reviewed by an independent sub-agent before merge (per crowsnest policy): confirmed both fixes correct, no regressions, found no other instance of the ungrouped-alternation bug elsewhere in the codebase. Refs #82 (partial) Co-Authored-By: Claude Sonnet 5 --- dol/trans.py | 26 +++++++++++++++++++------- 1 file changed, 19 insertions(+), 7 deletions(-) diff --git a/dol/trans.py b/dol/trans.py index d9065ba0..9e47617a 100644 --- a/dol/trans.py +++ b/dol/trans.py @@ -1605,13 +1605,18 @@ def __iter__(self): store_cls.__iter__ = __iter__ - def __len__(self): - c = 0 - for _ in self.__iter__(): - c += 1 - return c + if hasattr(store_cls, "__len__"): + # Only add a (filtered, counting) __len__ if the wrapped class already had + # one. A class that deliberately omits __len__ (e.g. because counting means + # an unbounded paginated listing over a remote backend) should stay that + # way -- see i2mint/dol#82. + def __len__(self): + c = 0 + for _ in self.__iter__(): + c += 1 + return c - store_cls.__len__ = __len__ + store_cls.__len__ = __len__ def __contains__(self, k): if filt(k): @@ -1728,10 +1733,17 @@ def filter_prefixes(prefixes): True >>> is_test_or_report("image.jpg") False + + The prefixes are grouped, so a multi-prefix filter doesn't accidentally match a + string that merely *contains* one of the later prefixes anywhere: + + >>> is_logs_or_tmp = filter_prefixes(['logs/', 'tmp/']) + >>> is_logs_or_tmp("other/tmp/c") + False """ if isinstance(prefixes, str): prefixes = [prefixes] - return filter_regex("^" + "|".join(map(re.escape, prefixes))) + return filter_regex("^(?:" + "|".join(map(re.escape, prefixes)) + ")") class FiltIter: