From bbf05b3863552985e4b03eceb401766bedcb9d59 Mon Sep 17 00:00:00 2001 From: jim Date: Thu, 3 Sep 2026 14:34:57 -0500 Subject: [PATCH] fix(collect): exclude posts from IsActive=0 blogs in --collect 1 Blogs.IsActive is the crawler's work-selection flag (Rolodex removal sets it to 0) and is independent of Posts.IsActive/Notes.IsActive -- deactivating a blog never touches its posts' own IsActive column, so --collect 1 kept re-queuing posts for blogs that had been deactivated. Join Blogs into the PostsWithCount CTE's source filter and require COALESCE(BL.IsActive, 1) = 1. Since the hardcoded zomb-eh re-queue branch reads from PostsWithCount rather than Posts directly, it now inherits this filter automatically -- if zomb-eh is ever deactivated, its rows disappear from PostsWithCount and the union branch contributes nothing, with no special-case code needed. Co-Authored-By: Claude Sonnet 5 --- URLNotesGrabberCORE/DataAccess.cs | 21 +++++++++++++++------ 1 file changed, 15 insertions(+), 6 deletions(-) diff --git a/URLNotesGrabberCORE/DataAccess.cs b/URLNotesGrabberCORE/DataAccess.cs index aa24a40..cc31e61 100644 --- a/URLNotesGrabberCORE/DataAccess.cs +++ b/URLNotesGrabberCORE/DataAccess.cs @@ -888,16 +888,25 @@ namespace URLNotesGrabberCORE beforeDateFilter = $"WHERE (U.NotesGatheredDateTime < {unixTimestamp} OR U.NotesGatheredDateTime IS NULL)" + Environment.NewLine; } + // Blogs.IsActive is the crawler's work-selection flag (Rolodex removal sets it to 0) + // and is independent of Posts.IsActive/Notes.IsActive -- deactivating a blog does not + // touch its posts' own IsActive column. AndIsActive("Posts", ...) above therefore does + // not catch a deactivated blog; this join against the source rows is what does, so a + // blog taken IsActive = 0 in Blogs stops being re-queued by --collect 1 even if its + // posts were never individually marked inactive. Blogs.BlogName is that table's PRIMARY + // KEY, so the join rides an index rather than scanning it. + // // Same WHERE/AND juggling WhereIsActive does, extended to the optional blog predicate: // either clause may be absent, so the first one present has to open the WHERE. - string sourceClause = AndIsActive("Posts", "P", DBPath) + (filterByBlog ? " AND P.BlogName = @blogName" : string.Empty); + string sourceClause = AndIsActive("Posts", "P", DBPath) + " AND COALESCE(BL.IsActive, 1) = 1" + (filterByBlog ? " AND P.BlogName = @blogName" : string.Empty); string sourceFilter = sourceClause.Length == 0 ? string.Empty : " WHERE" + sourceClause.Substring(" AND".Length); // The zomb-eh branch re-queues that blog's *already collected* posts every 3 days. It needs - // no blog-filter handling of its own: it reads PostsWithCount, which the filter has already - // scoped, so it contributes its rows when the filter names zomb-eh and nothing otherwise. - // That keeps a filtered worklist a strict subset of the unfiltered one -- "--collect 1 X" - // returns exactly the rows "--collect 1" would have returned for X. + // no blog-filter or IsActive handling of its own: it reads PostsWithCount, which the source + // filter above -- Blogs.IsActive included -- has already scoped, so it contributes its rows + // only when zomb-eh itself is still IsActive = 1 there. That keeps a filtered worklist a + // strict subset of the unfiltered one -- "--collect 1 X" returns exactly the rows + // "--collect 1" would have returned for X. // // --force drops the age gate only. NotFound = 0 and the IsActive/blog scoping above still // apply: the flag is "re-collect early", not "collect rows every other path excludes". @@ -950,7 +959,7 @@ namespace URLNotesGrabberCORE " P.HasNotesGathered," + Environment.NewLine + " P.NotFound," + Environment.NewLine + " P.PostDate" + Environment.NewLine + - " FROM Posts P" + sourceFilter + Environment.NewLine + + " FROM Posts P LEFT JOIN Blogs BL ON BL.BlogName = P.BlogName" + sourceFilter + Environment.NewLine + ")," + Environment.NewLine + "Unioned AS" + Environment.NewLine + "(" + Environment.NewLine +