feat(collect): add an optional blog filter to --collect

"--collect 1 zomb-eh" now restricts the run to a single blog. The name is
bound as a SQLite parameter and matched exactly against Posts.BlogName,
which leads the primary key, so the predicate uses that index.

Trailing arguments are scanned rather than positionally fixed: a token that
parses as a date is the cutoff, anything else is the blog name, in either
order. --blog=name forces the blog reading for a name that would otherwise
parse as a date.

The filter is a pure filter -- the zomb-eh 3-day refresh branch reads from
the already-scoped PostsWithCount CTE, so a filtered worklist is a strict
subset of the unfiltered one. Verified against TL.db: --collect 1 returns
651 posts, of which --collect 1 zomb-eh returns exactly its 164 and
--collect 1 cs1d3blog exactly its 18.

Blog-scoped runs are excluded from managed-run state, so collecting one
blog cannot mark a full re-check complete.

Co-Authored-By: Claude Opus 5 <[email protected]>
This commit is contained in:
jim
2026-08-08 21:47:38 -05:00
co-authored by Claude Opus 5
parent b31d5842cc
commit 0d9641db08
3 changed files with 109 additions and 41 deletions
+56 -13
View File
@@ -220,7 +220,7 @@ namespace URLNotesGrabberCORE
if (args.Length < 2)
{
Console.WriteLine("--Expected WITHOUTNOTESONLY (0, 1) [OPTIONAL: BEFOREDATE]--");
Console.WriteLine("--Expected WITHOUTNOTESONLY (0, 1) [OPTIONAL: BEFOREDATE] [OPTIONAL: BLOGNAME]--");
exitCode = 2;
break;
}
@@ -242,28 +242,63 @@ namespace URLNotesGrabberCORE
Console.WriteLine("Without Notes Only: {0}\t{1}", withoutNotesOnly, args[1]);
}
// Parse optional beforeDate parameter
if (args.Length >= 3 && !string.IsNullOrEmpty(args[2]))
// Trailing arguments are the optional cutoff date and the optional blog filter, in
// either order. A token that parses as a date is the cutoff; anything else is a blog
// name -- which is why an unparseable token is no longer an error here. "--blog=name"
// forces the blog reading for the rare name that would otherwise parse as a date.
string? collectBlogName = null;
bool badCollectArg = false;
for (int i = 2; i < args.Length; i++)
{
if (DateTime.TryParse(args[2], out DateTime parsedDate))
string arg = args[i];
if (string.IsNullOrWhiteSpace(arg))
continue;
if (arg.StartsWith("--blog=", StringComparison.OrdinalIgnoreCase))
{
collectBlogName = arg.Substring("--blog=".Length);
if (string.IsNullOrWhiteSpace(collectBlogName))
{
Console.WriteLine("ERROR: --blog= requires a blog name");
badCollectArg = true;
break;
}
}
else if (!explicitDateSupplied && DateTime.TryParse(arg, out DateTime parsedDate))
{
beforeDate = parsedDate;
explicitDateSupplied = true;
Console.WriteLine($"Filter: Collecting notes for posts with NotesGatheredDateTime < {beforeDate}");
}
else if (collectBlogName == null)
{
collectBlogName = arg;
}
else
{
Console.WriteLine($"ERROR: Invalid date format '{args[2]}'");
exitCode = 2;
Console.WriteLine($"ERROR: Unexpected argument '{arg}'");
badCollectArg = true;
break;
}
}
if (badCollectArg)
{
exitCode = 2;
break;
}
if (collectBlogName != null)
Console.WriteLine($"Filter: Collecting notes for posts by '{collectBlogName}' only");
// Mode 0 (full re-check) with no explicit date is a *managed* run: freeze the cutoff and
// persist it so an interrupted run resumes against the same cutoff and a completed run stops
// instead of restarting. Mode 1 and explicit-date runs keep their existing behavior.
// instead of restarting. Mode 1, explicit-date and blog-scoped runs keep their existing
// behavior -- a single blog covers a slice of the worklist, so letting it write the shared
// run state would mark the whole re-check complete after collecting one blog.
bool managedCollectRun = false;
if (!withoutNotesOnly && !explicitDateSupplied)
if (!withoutNotesOnly && !explicitDateSupplied && collectBlogName == null)
{
DataAccess.EnsureCollectRunStateTableExists();
var runState = DataAccess.GetCollectRunState();
@@ -281,7 +316,7 @@ namespace URLNotesGrabberCORE
managedCollectRun = true;
}
exitCode = CollectNotes(settings.GetValue<string>("PathOutput"), withoutNotesOnly, beforeDate, managedCollectRun).GetAwaiter().GetResult();
exitCode = CollectNotes(settings.GetValue<string>("PathOutput"), withoutNotesOnly, beforeDate, managedCollectRun, collectBlogName).GetAwaiter().GetResult();
break;
case "--blogsR": //collect notes from all posts
@@ -413,7 +448,7 @@ namespace URLNotesGrabberCORE
Console.WriteLine("--blogs\t For each Blog in DB, write blogname to file");
Console.WriteLine("--collect [0|1] [datetime]\t Collect Notes from API. 1=only posts without notes. 0=full re-check of all posts: a single resumable pass (interrupt & relaunch to resume; stops when complete, retrigger for a new pass). Optional datetime overrides the cutoff and runs as a one-off (bypasses resume tracking).");
Console.WriteLine("--collect [0|1] [datetime] [blogname]\t Collect Notes from API. 1=only posts without notes. 0=full re-check of all posts: a single resumable pass (interrupt & relaunch to resume; stops when complete, retrigger for a new pass). Optional datetime overrides the cutoff and runs as a one-off (bypasses resume tracking). Optional blogname restricts the run to that blog (exact, case-sensitive match) and also runs as a one-off; e.g. \"--collect 1 zomb-eh\". datetime and blogname may be given in either order - use --blog=name if a blog name would otherwise parse as a date.");
Console.WriteLine("--blogsR\t For each Note that is a REPLY, write blogname to file ");
@@ -1294,9 +1329,17 @@ if (shouldInsert)
// blipping on one post. Past this, skipping post-by-post would just hammer a closed door.
const int MaxConsecutiveTransient = 10;
static async Task<int> CollectNotes(string outPath, bool withoutNotesOnly = true, DateTime? beforeDate = null, bool managedRun = false)
static async Task<int> CollectNotes(string outPath, bool withoutNotesOnly = true, DateTime? beforeDate = null, bool managedRun = false, string? blogName = null)
{
List<Tuple<string, long, long, long>> posts = DataAccess.GetPosts(withoutNotesOnly, beforeDate);
List<Tuple<string, long, long, long>> posts = DataAccess.GetPosts(withoutNotesOnly, beforeDate, blogName);
if (posts.Count == 0 && !string.IsNullOrWhiteSpace(blogName))
{
// BlogName is matched exactly, so a typo or a case mismatch looks identical to "nothing
// left to collect". Say so rather than reporting a silent, instant success.
Console.WriteLine($"No posts to collect for blog '{blogName}'. Either it is fully collected, or the name does not match a stored blog (the match is case-sensitive).");
return 0;
}
// Posts attempted (with a definitive, non-throttle result) during *this* process. Guarantees a single
// attempt pass: once every remaining post has been attempted, the loop stops instead of spinning on a
@@ -1392,7 +1435,7 @@ if (shouldInsert)
}
// Re-fetch the updated list after processing the current post
posts = DataAccess.GetPosts(withoutNotesOnly, beforeDate);
posts = DataAccess.GetPosts(withoutNotesOnly, beforeDate, blogName);
}
}