From ecabc828eaf707cc8239480b18ccbdccbaa23f83 Mon Sep 17 00:00:00 2001 From: rmsitz Date: Mon, 24 Aug 2026 18:06:57 -0400 Subject: [PATCH] Cache the category tree walk (fixes ~1s File Management page load) _build_category_index() was the one tree-scanning function never wrapped in the app's existing 5-minute cache - it ran fresh on every File Management page load and Sort Unsorted scan (not a manual "this may take a moment" action like Duplicate Courses or Storage Usage), so the cost was invisible until it was already slow. Measured ~1.0-1.1s consistently against the live NAS-mounted library at 161 courses, confirmed via profiling that a single call makes dozens of iterdir() round-trips - each one a network hop over SMB. Now shares the same cache_get_or_compute pattern and invalidate_cache() call sites as get_all_course_dirs(), so no new invalidation logic was needed. Co-Authored-By: Claude Sonnet 5 --- README.md | 8 +++- VERSION | 2 +- offlineu_core.py | 103 ++++++++++++++++++++++++++--------------------- 3 files changed, 65 insertions(+), 48 deletions(-) diff --git a/README.md b/README.md index 06f101f..53662fb 100644 --- a/README.md +++ b/README.md @@ -137,8 +137,12 @@ silently stay blank instead of erroring. ones that recur across many categories (a prolific creator's name, etc.) so they can't outvote a genuinely specific word just by sharing more of them. - - *Refresh Library*: manually bypasses the 5-minute filesystem-scan cache, - for when files were added/removed directly on disk. + - *Refresh Library*: manually bypasses the 5-minute filesystem-scan cache + (this now includes the category tree Sort Unsorted/Manage Library's + picker builds - previously rebuilt on every page load, a full, + uncached walk of the whole library that got noticeably slow over a + NAS-mounted (SMB) library as the course count grew), for when files + were added/removed directly on disk. - *Bulk Rename*: find & replace across every course/folder name in the library at once, with a per-match preview and the ability to drop individual matches before applying. Three match modes: plain text diff --git a/VERSION b/VERSION index 74e0a84..1a0c6a8 100644 --- a/VERSION +++ b/VERSION @@ -1 +1 @@ -2026-08-24 18:40 UTC — search, bulk actions, undo, storage usage +2026-08-24 21:41 UTC — cache category tree walk (fixes ~1s picker lag) diff --git a/offlineu_core.py b/offlineu_core.py index 3667e53..959fb6c 100644 --- a/offlineu_core.py +++ b/offlineu_core.py @@ -1445,54 +1445,67 @@ def _build_category_index() -> List[Dict[str, Any]]: courses and their non-lecture contents never show up as if they were categories to file things under. The Unsorted folder itself is excluded - it's the source, never a valid destination. + + This walk visits every directory in the library and, over a + NAS-mounted (SMB) library, each of those is a network round-trip - + the same reasoning behind get_all_course_dirs()'s cache, so this + result is cached the same way (same 5-minute TTL, same + invalidate_cache() call sites already busting it - no separate + invalidation needed): otherwise this ran fresh on every File + Management page load and Sort Unsorted scan, neither of which is a + manual "this may take a moment" action like Duplicate Courses or + Storage Usage, so the cost was invisible until it was already slow. """ - library_root = Path(get_library_root()) - try: - unsorted_root = (library_root / UNSORTED_FOLDER_NAME).resolve() - except OSError: - unsorted_root = None - categories: List[Dict[str, Any]] = [] - - def walk(directory: Path, path_parts: List[str]): + def compute() -> List[Dict[str, Any]]: + library_root = Path(get_library_root()) try: - entries = sorted( - (p for p in directory.iterdir() if p.is_dir() and not p.name.startswith('.')), - key=lambda p: p.name.lower() - ) - except (PermissionError, OSError): - return - for entry in entries: - try: - if unsorted_root is not None and entry.resolve() == unsorted_root: - continue - except OSError: - pass - if (_looks_like_course(entry) or _is_unrecognized_leaf_item(entry) - or _is_media_free_subtree(entry) or _is_course_wrapper(entry)): - continue - new_parts = path_parts + [entry.name] - path_tokens: Set[str] = set() - for part in new_parts: - path_tokens |= _tokenize(part) - bonus_tokens: Set[str] = set() - try: - for child in entry.iterdir(): - if child.is_dir() and not child.name.startswith('.') and _looks_like_course(child): - bonus_tokens |= _tokenize(child.name) - except (PermissionError, OSError): - pass - bonus_tokens -= path_tokens - categories.append({ - 'path': str(entry), - 'relative': '/'.join(new_parts), - 'depth': len(new_parts), - 'path_tokens': path_tokens, - 'bonus_tokens': bonus_tokens, - }) - walk(entry, new_parts) + unsorted_root = (library_root / UNSORTED_FOLDER_NAME).resolve() + except OSError: + unsorted_root = None + categories: List[Dict[str, Any]] = [] - walk(library_root, []) - return categories + def walk(directory: Path, path_parts: List[str]): + try: + entries = sorted( + (p for p in directory.iterdir() if p.is_dir() and not p.name.startswith('.')), + key=lambda p: p.name.lower() + ) + except (PermissionError, OSError): + return + for entry in entries: + try: + if unsorted_root is not None and entry.resolve() == unsorted_root: + continue + except OSError: + pass + if (_looks_like_course(entry) or _is_unrecognized_leaf_item(entry) + or _is_media_free_subtree(entry) or _is_course_wrapper(entry)): + continue + new_parts = path_parts + [entry.name] + path_tokens: Set[str] = set() + for part in new_parts: + path_tokens |= _tokenize(part) + bonus_tokens: Set[str] = set() + try: + for child in entry.iterdir(): + if child.is_dir() and not child.name.startswith('.') and _looks_like_course(child): + bonus_tokens |= _tokenize(child.name) + except (PermissionError, OSError): + pass + bonus_tokens -= path_tokens + categories.append({ + 'path': str(entry), + 'relative': '/'.join(new_parts), + 'depth': len(new_parts), + 'path_tokens': path_tokens, + 'bonus_tokens': bonus_tokens, + }) + walk(entry, new_parts) + + walk(library_root, []) + return categories + + return cache_get_or_compute('category_index', compute) def _bonus_token_frequency(categories: List[Dict[str, Any]]) -> Dict[str, int]: