diff --git a/.github/copilot-instructions.md b/.github/copilot-instructions.md
index b99baac..a98095c 100644
--- a/.github/copilot-instructions.md
+++ b/.github/copilot-instructions.md
@@ -31,7 +31,7 @@ dotnet run -- --test [blogname] [postID] # Test API for specific post
- `--test [blogname] [postID]`: Test API note collection
- `--posts`: Export post blogs to file
- `--blogs`: Export blog list to file
-- `--collect`: Collect notes for all posts in DB
+- `--collect [0|1] [datetime] [blogname]`: Collect notes for posts in DB. Optional `blogname` restricts the run to one blog (exact match), e.g. `--collect 1 zomb-eh`
- `--blogsR`: Export reply blogs to file
- `--blogsO [start] [stop]`: Export blogs within range
diff --git a/URLNotesGrabberCORE/DataAccess.cs b/URLNotesGrabberCORE/DataAccess.cs
index dd7fd1e..5e95d7e 100644
--- a/URLNotesGrabberCORE/DataAccess.cs
+++ b/URLNotesGrabberCORE/DataAccess.cs
@@ -855,12 +855,16 @@ namespace URLNotesGrabberCORE
///
///
/// blogName, postID, lastNoteTimestamp, notesGatheredTimestamp
- public static List> GetPosts(bool withoutNotesOnly = false, DateTime? beforeDate = null, string? DBPath = null)
+ public static List> GetPosts(bool withoutNotesOnly = false, DateTime? beforeDate = null, string? blogName = null, string? DBPath = null)
{
DBPath ??= GetDefaultDbPath();
using SQLiteConnection connection = new SQLiteConnection("Data Source=" + DBPath);
List> posts = new List>();
+ // A blog filter matches BlogName exactly: the column is BINARY-collated and leads the
+ // Posts primary key, so "= @blogName" rides that index instead of scanning 1.18M rows.
+ bool filterByBlog = !string.IsNullOrWhiteSpace(blogName);
+
try
{
connection.Open();
@@ -876,31 +880,17 @@ namespace URLNotesGrabberCORE
beforeDateFilter = $"WHERE (U.NotesGatheredDateTime < {unixTimestamp} OR U.NotesGatheredDateTime IS NULL)" + Environment.NewLine;
}
- sql = "WITH PostsWithCount AS" + Environment.NewLine +
- "(" + Environment.NewLine +
- " SELECT " + Environment.NewLine +
- " P.BlogName," + Environment.NewLine +
- " P.PostID," + Environment.NewLine +
- " 1925013599 AS LatestNoteTimestamp," + Environment.NewLine +
- " P.NotesGatheredDateTime," + Environment.NewLine +
- " COUNT(P.PostID) OVER(PARTITION BY P.BlogName) AS CNT," + Environment.NewLine +
- " P.HasNotesGathered," + Environment.NewLine +
- " P.NotFound," + Environment.NewLine +
- " P.PostDate" + Environment.NewLine +
- " FROM Posts P" + WhereIsActive("Posts", "P", DBPath) + Environment.NewLine +
- ")," + Environment.NewLine +
- "Unioned AS" + Environment.NewLine +
- "(" + Environment.NewLine +
- " SELECT " + Environment.NewLine +
- " BlogName," + Environment.NewLine +
- " PostID," + Environment.NewLine +
- " LatestNoteTimestamp," + Environment.NewLine +
- " NotesGatheredDateTime," + Environment.NewLine +
- " CNT," + Environment.NewLine +
- " PostDate" + Environment.NewLine +
- " FROM PostsWithCount" + Environment.NewLine +
- " WHERE NotFound = 0" + Environment.NewLine +
- " AND HasNotesGathered = 0" + Environment.NewLine +
+ // Same WHERE/AND juggling WhereIsActive does, extended to the optional blog predicate:
+ // either clause may be absent, so the first one present has to open the WHERE.
+ string sourceClause = AndIsActive("Posts", "P", DBPath) + (filterByBlog ? " AND P.BlogName = @blogName" : string.Empty);
+ string sourceFilter = sourceClause.Length == 0 ? string.Empty : " WHERE" + sourceClause.Substring(" AND".Length);
+
+ // The zomb-eh branch re-queues that blog's *already collected* posts every 3 days. It needs
+ // no blog-filter handling of its own: it reads PostsWithCount, which the filter has already
+ // scoped, so it contributes its rows when the filter names zomb-eh and nothing otherwise.
+ // That keeps a filtered worklist a strict subset of the unfiltered one -- "--collect 1 X"
+ // returns exactly the rows "--collect 1" would have returned for X.
+ string refreshBranch =
"" + Environment.NewLine +
" UNION " + Environment.NewLine +
"" + Environment.NewLine +
@@ -914,7 +904,34 @@ namespace URLNotesGrabberCORE
" FROM PostsWithCount" + Environment.NewLine +
" WHERE BlogName = 'zomb-eh'" + Environment.NewLine +
" AND NotFound = 0" + Environment.NewLine +
- " AND NotesGatheredDateTime < unixepoch('now', 'localtime', '-3 days')" + Environment.NewLine +
+ " AND NotesGatheredDateTime < unixepoch('now', 'localtime', '-3 days')" + Environment.NewLine;
+
+ sql = "WITH PostsWithCount AS" + Environment.NewLine +
+ "(" + Environment.NewLine +
+ " SELECT " + Environment.NewLine +
+ " P.BlogName," + Environment.NewLine +
+ " P.PostID," + Environment.NewLine +
+ " 1925013599 AS LatestNoteTimestamp," + Environment.NewLine +
+ " P.NotesGatheredDateTime," + Environment.NewLine +
+ " COUNT(P.PostID) OVER(PARTITION BY P.BlogName) AS CNT," + Environment.NewLine +
+ " P.HasNotesGathered," + Environment.NewLine +
+ " P.NotFound," + Environment.NewLine +
+ " P.PostDate" + Environment.NewLine +
+ " FROM Posts P" + sourceFilter + Environment.NewLine +
+ ")," + Environment.NewLine +
+ "Unioned AS" + Environment.NewLine +
+ "(" + Environment.NewLine +
+ " SELECT " + Environment.NewLine +
+ " BlogName," + Environment.NewLine +
+ " PostID," + Environment.NewLine +
+ " LatestNoteTimestamp," + Environment.NewLine +
+ " NotesGatheredDateTime," + Environment.NewLine +
+ " CNT," + Environment.NewLine +
+ " PostDate" + Environment.NewLine +
+ " FROM PostsWithCount" + Environment.NewLine +
+ " WHERE NotFound = 0" + Environment.NewLine +
+ " AND HasNotesGathered = 0" + Environment.NewLine +
+ refreshBranch +
")" + Environment.NewLine +
"SELECT" + Environment.NewLine +
" U.BlogName," + Environment.NewLine +
@@ -944,6 +961,11 @@ namespace URLNotesGrabberCORE
" ( select BlogName, count(PostID) as CNT from Posts" + WhereIsActive("Posts", "", DBPath) + " group by BlogName) CNT on CNT.blogName = Posts.BlogName " +
"WHERE NotFound = 0 " + AndIsActive("Posts", "Posts", DBPath) + Environment.NewLine;
+ if (filterByBlog)
+ {
+ sql += " AND Posts.BlogName = @blogName " + Environment.NewLine;
+ }
+
if (beforeDate.HasValue)
{
long unixTimestamp = new DateTimeOffset(beforeDate.Value).ToUnixTimeSeconds();
@@ -963,6 +985,9 @@ namespace URLNotesGrabberCORE
using (SQLiteCommand command = new SQLiteCommand(sql, connection))
{
+ if (filterByBlog)
+ command.Parameters.AddWithValue("@blogName", blogName);
+
using (SQLiteDataReader reader = command.ExecuteReader())
{
while (reader.Read())
diff --git a/URLNotesGrabberCORE/Program.cs b/URLNotesGrabberCORE/Program.cs
index dddfbcd..6552e5e 100644
--- a/URLNotesGrabberCORE/Program.cs
+++ b/URLNotesGrabberCORE/Program.cs
@@ -220,7 +220,7 @@ namespace URLNotesGrabberCORE
if (args.Length < 2)
{
- Console.WriteLine("--Expected WITHOUTNOTESONLY (0, 1) [OPTIONAL: BEFOREDATE]--");
+ Console.WriteLine("--Expected WITHOUTNOTESONLY (0, 1) [OPTIONAL: BEFOREDATE] [OPTIONAL: BLOGNAME]--");
exitCode = 2;
break;
}
@@ -242,28 +242,63 @@ namespace URLNotesGrabberCORE
Console.WriteLine("Without Notes Only: {0}\t{1}", withoutNotesOnly, args[1]);
}
- // Parse optional beforeDate parameter
- if (args.Length >= 3 && !string.IsNullOrEmpty(args[2]))
+ // Trailing arguments are the optional cutoff date and the optional blog filter, in
+ // either order. A token that parses as a date is the cutoff; anything else is a blog
+ // name -- which is why an unparseable token is no longer an error here. "--blog=name"
+ // forces the blog reading for the rare name that would otherwise parse as a date.
+ string? collectBlogName = null;
+ bool badCollectArg = false;
+
+ for (int i = 2; i < args.Length; i++)
{
- if (DateTime.TryParse(args[2], out DateTime parsedDate))
+ string arg = args[i];
+ if (string.IsNullOrWhiteSpace(arg))
+ continue;
+
+ if (arg.StartsWith("--blog=", StringComparison.OrdinalIgnoreCase))
+ {
+ collectBlogName = arg.Substring("--blog=".Length);
+ if (string.IsNullOrWhiteSpace(collectBlogName))
+ {
+ Console.WriteLine("ERROR: --blog= requires a blog name");
+ badCollectArg = true;
+ break;
+ }
+ }
+ else if (!explicitDateSupplied && DateTime.TryParse(arg, out DateTime parsedDate))
{
beforeDate = parsedDate;
explicitDateSupplied = true;
Console.WriteLine($"Filter: Collecting notes for posts with NotesGatheredDateTime < {beforeDate}");
}
+ else if (collectBlogName == null)
+ {
+ collectBlogName = arg;
+ }
else
{
- Console.WriteLine($"ERROR: Invalid date format '{args[2]}'");
- exitCode = 2;
+ Console.WriteLine($"ERROR: Unexpected argument '{arg}'");
+ badCollectArg = true;
break;
}
}
+ if (badCollectArg)
+ {
+ exitCode = 2;
+ break;
+ }
+
+ if (collectBlogName != null)
+ Console.WriteLine($"Filter: Collecting notes for posts by '{collectBlogName}' only");
+
// Mode 0 (full re-check) with no explicit date is a *managed* run: freeze the cutoff and
// persist it so an interrupted run resumes against the same cutoff and a completed run stops
- // instead of restarting. Mode 1 and explicit-date runs keep their existing behavior.
+ // instead of restarting. Mode 1, explicit-date and blog-scoped runs keep their existing
+ // behavior -- a single blog covers a slice of the worklist, so letting it write the shared
+ // run state would mark the whole re-check complete after collecting one blog.
bool managedCollectRun = false;
- if (!withoutNotesOnly && !explicitDateSupplied)
+ if (!withoutNotesOnly && !explicitDateSupplied && collectBlogName == null)
{
DataAccess.EnsureCollectRunStateTableExists();
var runState = DataAccess.GetCollectRunState();
@@ -281,7 +316,7 @@ namespace URLNotesGrabberCORE
managedCollectRun = true;
}
- exitCode = CollectNotes(settings.GetValue("PathOutput"), withoutNotesOnly, beforeDate, managedCollectRun).GetAwaiter().GetResult();
+ exitCode = CollectNotes(settings.GetValue("PathOutput"), withoutNotesOnly, beforeDate, managedCollectRun, collectBlogName).GetAwaiter().GetResult();
break;
case "--blogsR": //collect notes from all posts
@@ -413,7 +448,7 @@ namespace URLNotesGrabberCORE
Console.WriteLine("--blogs\t For each Blog in DB, write blogname to file");
- Console.WriteLine("--collect [0|1] [datetime]\t Collect Notes from API. 1=only posts without notes. 0=full re-check of all posts: a single resumable pass (interrupt & relaunch to resume; stops when complete, retrigger for a new pass). Optional datetime overrides the cutoff and runs as a one-off (bypasses resume tracking).");
+ Console.WriteLine("--collect [0|1] [datetime] [blogname]\t Collect Notes from API. 1=only posts without notes. 0=full re-check of all posts: a single resumable pass (interrupt & relaunch to resume; stops when complete, retrigger for a new pass). Optional datetime overrides the cutoff and runs as a one-off (bypasses resume tracking). Optional blogname restricts the run to that blog (exact, case-sensitive match) and also runs as a one-off; e.g. \"--collect 1 zomb-eh\". datetime and blogname may be given in either order - use --blog=name if a blog name would otherwise parse as a date.");
Console.WriteLine("--blogsR\t For each Note that is a REPLY, write blogname to file ");
@@ -1294,9 +1329,17 @@ if (shouldInsert)
// blipping on one post. Past this, skipping post-by-post would just hammer a closed door.
const int MaxConsecutiveTransient = 10;
- static async Task CollectNotes(string outPath, bool withoutNotesOnly = true, DateTime? beforeDate = null, bool managedRun = false)
+ static async Task CollectNotes(string outPath, bool withoutNotesOnly = true, DateTime? beforeDate = null, bool managedRun = false, string? blogName = null)
{
- List> posts = DataAccess.GetPosts(withoutNotesOnly, beforeDate);
+ List> posts = DataAccess.GetPosts(withoutNotesOnly, beforeDate, blogName);
+
+ if (posts.Count == 0 && !string.IsNullOrWhiteSpace(blogName))
+ {
+ // BlogName is matched exactly, so a typo or a case mismatch looks identical to "nothing
+ // left to collect". Say so rather than reporting a silent, instant success.
+ Console.WriteLine($"No posts to collect for blog '{blogName}'. Either it is fully collected, or the name does not match a stored blog (the match is case-sensitive).");
+ return 0;
+ }
// Posts attempted (with a definitive, non-throttle result) during *this* process. Guarantees a single
// attempt pass: once every remaining post has been attempted, the loop stops instead of spinning on a
@@ -1392,7 +1435,7 @@ if (shouldInsert)
}
// Re-fetch the updated list after processing the current post
- posts = DataAccess.GetPosts(withoutNotesOnly, beforeDate);
+ posts = DataAccess.GetPosts(withoutNotesOnly, beforeDate, blogName);
}
}