diff --git a/.github/copilot-instructions.md b/.github/copilot-instructions.md index b99baac..a98095c 100644 --- a/.github/copilot-instructions.md +++ b/.github/copilot-instructions.md @@ -31,7 +31,7 @@ dotnet run -- --test [blogname] [postID] # Test API for specific post - `--test [blogname] [postID]`: Test API note collection - `--posts`: Export post blogs to file - `--blogs`: Export blog list to file -- `--collect`: Collect notes for all posts in DB +- `--collect [0|1] [datetime] [blogname]`: Collect notes for posts in DB. Optional `blogname` restricts the run to one blog (exact match), e.g. `--collect 1 zomb-eh` - `--blogsR`: Export reply blogs to file - `--blogsO [start] [stop]`: Export blogs within range diff --git a/URLNotesGrabberCORE/DataAccess.cs b/URLNotesGrabberCORE/DataAccess.cs index dd7fd1e..5e95d7e 100644 --- a/URLNotesGrabberCORE/DataAccess.cs +++ b/URLNotesGrabberCORE/DataAccess.cs @@ -855,12 +855,16 @@ namespace URLNotesGrabberCORE /// /// /// blogName, postID, lastNoteTimestamp, notesGatheredTimestamp - public static List> GetPosts(bool withoutNotesOnly = false, DateTime? beforeDate = null, string? DBPath = null) + public static List> GetPosts(bool withoutNotesOnly = false, DateTime? beforeDate = null, string? blogName = null, string? DBPath = null) { DBPath ??= GetDefaultDbPath(); using SQLiteConnection connection = new SQLiteConnection("Data Source=" + DBPath); List> posts = new List>(); + // A blog filter matches BlogName exactly: the column is BINARY-collated and leads the + // Posts primary key, so "= @blogName" rides that index instead of scanning 1.18M rows. + bool filterByBlog = !string.IsNullOrWhiteSpace(blogName); + try { connection.Open(); @@ -876,31 +880,17 @@ namespace URLNotesGrabberCORE beforeDateFilter = $"WHERE (U.NotesGatheredDateTime < {unixTimestamp} OR U.NotesGatheredDateTime IS NULL)" + Environment.NewLine; } - sql = "WITH PostsWithCount AS" + Environment.NewLine + - "(" + Environment.NewLine + - " SELECT " + Environment.NewLine + - " P.BlogName," + Environment.NewLine + - " P.PostID," + Environment.NewLine + - " 1925013599 AS LatestNoteTimestamp," + Environment.NewLine + - " P.NotesGatheredDateTime," + Environment.NewLine + - " COUNT(P.PostID) OVER(PARTITION BY P.BlogName) AS CNT," + Environment.NewLine + - " P.HasNotesGathered," + Environment.NewLine + - " P.NotFound," + Environment.NewLine + - " P.PostDate" + Environment.NewLine + - " FROM Posts P" + WhereIsActive("Posts", "P", DBPath) + Environment.NewLine + - ")," + Environment.NewLine + - "Unioned AS" + Environment.NewLine + - "(" + Environment.NewLine + - " SELECT " + Environment.NewLine + - " BlogName," + Environment.NewLine + - " PostID," + Environment.NewLine + - " LatestNoteTimestamp," + Environment.NewLine + - " NotesGatheredDateTime," + Environment.NewLine + - " CNT," + Environment.NewLine + - " PostDate" + Environment.NewLine + - " FROM PostsWithCount" + Environment.NewLine + - " WHERE NotFound = 0" + Environment.NewLine + - " AND HasNotesGathered = 0" + Environment.NewLine + + // Same WHERE/AND juggling WhereIsActive does, extended to the optional blog predicate: + // either clause may be absent, so the first one present has to open the WHERE. + string sourceClause = AndIsActive("Posts", "P", DBPath) + (filterByBlog ? " AND P.BlogName = @blogName" : string.Empty); + string sourceFilter = sourceClause.Length == 0 ? string.Empty : " WHERE" + sourceClause.Substring(" AND".Length); + + // The zomb-eh branch re-queues that blog's *already collected* posts every 3 days. It needs + // no blog-filter handling of its own: it reads PostsWithCount, which the filter has already + // scoped, so it contributes its rows when the filter names zomb-eh and nothing otherwise. + // That keeps a filtered worklist a strict subset of the unfiltered one -- "--collect 1 X" + // returns exactly the rows "--collect 1" would have returned for X. + string refreshBranch = "" + Environment.NewLine + " UNION " + Environment.NewLine + "" + Environment.NewLine + @@ -914,7 +904,34 @@ namespace URLNotesGrabberCORE " FROM PostsWithCount" + Environment.NewLine + " WHERE BlogName = 'zomb-eh'" + Environment.NewLine + " AND NotFound = 0" + Environment.NewLine + - " AND NotesGatheredDateTime < unixepoch('now', 'localtime', '-3 days')" + Environment.NewLine + + " AND NotesGatheredDateTime < unixepoch('now', 'localtime', '-3 days')" + Environment.NewLine; + + sql = "WITH PostsWithCount AS" + Environment.NewLine + + "(" + Environment.NewLine + + " SELECT " + Environment.NewLine + + " P.BlogName," + Environment.NewLine + + " P.PostID," + Environment.NewLine + + " 1925013599 AS LatestNoteTimestamp," + Environment.NewLine + + " P.NotesGatheredDateTime," + Environment.NewLine + + " COUNT(P.PostID) OVER(PARTITION BY P.BlogName) AS CNT," + Environment.NewLine + + " P.HasNotesGathered," + Environment.NewLine + + " P.NotFound," + Environment.NewLine + + " P.PostDate" + Environment.NewLine + + " FROM Posts P" + sourceFilter + Environment.NewLine + + ")," + Environment.NewLine + + "Unioned AS" + Environment.NewLine + + "(" + Environment.NewLine + + " SELECT " + Environment.NewLine + + " BlogName," + Environment.NewLine + + " PostID," + Environment.NewLine + + " LatestNoteTimestamp," + Environment.NewLine + + " NotesGatheredDateTime," + Environment.NewLine + + " CNT," + Environment.NewLine + + " PostDate" + Environment.NewLine + + " FROM PostsWithCount" + Environment.NewLine + + " WHERE NotFound = 0" + Environment.NewLine + + " AND HasNotesGathered = 0" + Environment.NewLine + + refreshBranch + ")" + Environment.NewLine + "SELECT" + Environment.NewLine + " U.BlogName," + Environment.NewLine + @@ -944,6 +961,11 @@ namespace URLNotesGrabberCORE " ( select BlogName, count(PostID) as CNT from Posts" + WhereIsActive("Posts", "", DBPath) + " group by BlogName) CNT on CNT.blogName = Posts.BlogName " + "WHERE NotFound = 0 " + AndIsActive("Posts", "Posts", DBPath) + Environment.NewLine; + if (filterByBlog) + { + sql += " AND Posts.BlogName = @blogName " + Environment.NewLine; + } + if (beforeDate.HasValue) { long unixTimestamp = new DateTimeOffset(beforeDate.Value).ToUnixTimeSeconds(); @@ -963,6 +985,9 @@ namespace URLNotesGrabberCORE using (SQLiteCommand command = new SQLiteCommand(sql, connection)) { + if (filterByBlog) + command.Parameters.AddWithValue("@blogName", blogName); + using (SQLiteDataReader reader = command.ExecuteReader()) { while (reader.Read()) diff --git a/URLNotesGrabberCORE/Program.cs b/URLNotesGrabberCORE/Program.cs index dddfbcd..6552e5e 100644 --- a/URLNotesGrabberCORE/Program.cs +++ b/URLNotesGrabberCORE/Program.cs @@ -220,7 +220,7 @@ namespace URLNotesGrabberCORE if (args.Length < 2) { - Console.WriteLine("--Expected WITHOUTNOTESONLY (0, 1) [OPTIONAL: BEFOREDATE]--"); + Console.WriteLine("--Expected WITHOUTNOTESONLY (0, 1) [OPTIONAL: BEFOREDATE] [OPTIONAL: BLOGNAME]--"); exitCode = 2; break; } @@ -242,28 +242,63 @@ namespace URLNotesGrabberCORE Console.WriteLine("Without Notes Only: {0}\t{1}", withoutNotesOnly, args[1]); } - // Parse optional beforeDate parameter - if (args.Length >= 3 && !string.IsNullOrEmpty(args[2])) + // Trailing arguments are the optional cutoff date and the optional blog filter, in + // either order. A token that parses as a date is the cutoff; anything else is a blog + // name -- which is why an unparseable token is no longer an error here. "--blog=name" + // forces the blog reading for the rare name that would otherwise parse as a date. + string? collectBlogName = null; + bool badCollectArg = false; + + for (int i = 2; i < args.Length; i++) { - if (DateTime.TryParse(args[2], out DateTime parsedDate)) + string arg = args[i]; + if (string.IsNullOrWhiteSpace(arg)) + continue; + + if (arg.StartsWith("--blog=", StringComparison.OrdinalIgnoreCase)) + { + collectBlogName = arg.Substring("--blog=".Length); + if (string.IsNullOrWhiteSpace(collectBlogName)) + { + Console.WriteLine("ERROR: --blog= requires a blog name"); + badCollectArg = true; + break; + } + } + else if (!explicitDateSupplied && DateTime.TryParse(arg, out DateTime parsedDate)) { beforeDate = parsedDate; explicitDateSupplied = true; Console.WriteLine($"Filter: Collecting notes for posts with NotesGatheredDateTime < {beforeDate}"); } + else if (collectBlogName == null) + { + collectBlogName = arg; + } else { - Console.WriteLine($"ERROR: Invalid date format '{args[2]}'"); - exitCode = 2; + Console.WriteLine($"ERROR: Unexpected argument '{arg}'"); + badCollectArg = true; break; } } + if (badCollectArg) + { + exitCode = 2; + break; + } + + if (collectBlogName != null) + Console.WriteLine($"Filter: Collecting notes for posts by '{collectBlogName}' only"); + // Mode 0 (full re-check) with no explicit date is a *managed* run: freeze the cutoff and // persist it so an interrupted run resumes against the same cutoff and a completed run stops - // instead of restarting. Mode 1 and explicit-date runs keep their existing behavior. + // instead of restarting. Mode 1, explicit-date and blog-scoped runs keep their existing + // behavior -- a single blog covers a slice of the worklist, so letting it write the shared + // run state would mark the whole re-check complete after collecting one blog. bool managedCollectRun = false; - if (!withoutNotesOnly && !explicitDateSupplied) + if (!withoutNotesOnly && !explicitDateSupplied && collectBlogName == null) { DataAccess.EnsureCollectRunStateTableExists(); var runState = DataAccess.GetCollectRunState(); @@ -281,7 +316,7 @@ namespace URLNotesGrabberCORE managedCollectRun = true; } - exitCode = CollectNotes(settings.GetValue("PathOutput"), withoutNotesOnly, beforeDate, managedCollectRun).GetAwaiter().GetResult(); + exitCode = CollectNotes(settings.GetValue("PathOutput"), withoutNotesOnly, beforeDate, managedCollectRun, collectBlogName).GetAwaiter().GetResult(); break; case "--blogsR": //collect notes from all posts @@ -413,7 +448,7 @@ namespace URLNotesGrabberCORE Console.WriteLine("--blogs\t For each Blog in DB, write blogname to file"); - Console.WriteLine("--collect [0|1] [datetime]\t Collect Notes from API. 1=only posts without notes. 0=full re-check of all posts: a single resumable pass (interrupt & relaunch to resume; stops when complete, retrigger for a new pass). Optional datetime overrides the cutoff and runs as a one-off (bypasses resume tracking)."); + Console.WriteLine("--collect [0|1] [datetime] [blogname]\t Collect Notes from API. 1=only posts without notes. 0=full re-check of all posts: a single resumable pass (interrupt & relaunch to resume; stops when complete, retrigger for a new pass). Optional datetime overrides the cutoff and runs as a one-off (bypasses resume tracking). Optional blogname restricts the run to that blog (exact, case-sensitive match) and also runs as a one-off; e.g. \"--collect 1 zomb-eh\". datetime and blogname may be given in either order - use --blog=name if a blog name would otherwise parse as a date."); Console.WriteLine("--blogsR\t For each Note that is a REPLY, write blogname to file "); @@ -1294,9 +1329,17 @@ if (shouldInsert) // blipping on one post. Past this, skipping post-by-post would just hammer a closed door. const int MaxConsecutiveTransient = 10; - static async Task CollectNotes(string outPath, bool withoutNotesOnly = true, DateTime? beforeDate = null, bool managedRun = false) + static async Task CollectNotes(string outPath, bool withoutNotesOnly = true, DateTime? beforeDate = null, bool managedRun = false, string? blogName = null) { - List> posts = DataAccess.GetPosts(withoutNotesOnly, beforeDate); + List> posts = DataAccess.GetPosts(withoutNotesOnly, beforeDate, blogName); + + if (posts.Count == 0 && !string.IsNullOrWhiteSpace(blogName)) + { + // BlogName is matched exactly, so a typo or a case mismatch looks identical to "nothing + // left to collect". Say so rather than reporting a silent, instant success. + Console.WriteLine($"No posts to collect for blog '{blogName}'. Either it is fully collected, or the name does not match a stored blog (the match is case-sensitive)."); + return 0; + } // Posts attempted (with a definitive, non-throttle result) during *this* process. Guarantees a single // attempt pass: once every remaining post has been attempted, the loop stops instead of spinning on a @@ -1392,7 +1435,7 @@ if (shouldInsert) } // Re-fetch the updated list after processing the current post - posts = DataAccess.GetPosts(withoutNotesOnly, beforeDate); + posts = DataAccess.GetPosts(withoutNotesOnly, beforeDate, blogName); } }