Merge branch 'claude/collect-1-query-sorting-e12f6c' into master

This commit is contained in:
jim
2026-08-08 21:47:45 -05:00
3 changed files with 109 additions and 41 deletions
+1 -1
View File
@@ -31,7 +31,7 @@ dotnet run -- --test [blogname] [postID] # Test API for specific post
- `--test [blogname] [postID]`: Test API note collection - `--test [blogname] [postID]`: Test API note collection
- `--posts`: Export post blogs to file - `--posts`: Export post blogs to file
- `--blogs`: Export blog list to file - `--blogs`: Export blog list to file
- `--collect`: Collect notes for all posts in DB - `--collect [0|1] [datetime] [blogname]`: Collect notes for posts in DB. Optional `blogname` restricts the run to one blog (exact match), e.g. `--collect 1 zomb-eh`
- `--blogsR`: Export reply blogs to file - `--blogsR`: Export reply blogs to file
- `--blogsO [start] [stop]`: Export blogs within range - `--blogsO [start] [stop]`: Export blogs within range
+52 -27
View File
@@ -855,12 +855,16 @@ namespace URLNotesGrabberCORE
/// <param name="withoutNotesOnly"></param> /// <param name="withoutNotesOnly"></param>
/// <param name="DBPath"></param> /// <param name="DBPath"></param>
/// <returns>blogName, postID, lastNoteTimestamp, notesGatheredTimestamp</returns> /// <returns>blogName, postID, lastNoteTimestamp, notesGatheredTimestamp</returns>
public static List<Tuple<string, long, long, long>> GetPosts(bool withoutNotesOnly = false, DateTime? beforeDate = null, string? DBPath = null) public static List<Tuple<string, long, long, long>> GetPosts(bool withoutNotesOnly = false, DateTime? beforeDate = null, string? blogName = null, string? DBPath = null)
{ {
DBPath ??= GetDefaultDbPath(); DBPath ??= GetDefaultDbPath();
using SQLiteConnection connection = new SQLiteConnection("Data Source=" + DBPath); using SQLiteConnection connection = new SQLiteConnection("Data Source=" + DBPath);
List<Tuple<string, long, long, long>> posts = new List<Tuple<string, long, long, long>>(); List<Tuple<string, long, long, long>> posts = new List<Tuple<string, long, long, long>>();
// A blog filter matches BlogName exactly: the column is BINARY-collated and leads the
// Posts primary key, so "= @blogName" rides that index instead of scanning 1.18M rows.
bool filterByBlog = !string.IsNullOrWhiteSpace(blogName);
try try
{ {
connection.Open(); connection.Open();
@@ -876,31 +880,17 @@ namespace URLNotesGrabberCORE
beforeDateFilter = $"WHERE (U.NotesGatheredDateTime < {unixTimestamp} OR U.NotesGatheredDateTime IS NULL)" + Environment.NewLine; beforeDateFilter = $"WHERE (U.NotesGatheredDateTime < {unixTimestamp} OR U.NotesGatheredDateTime IS NULL)" + Environment.NewLine;
} }
sql = "WITH PostsWithCount AS" + Environment.NewLine + // Same WHERE/AND juggling WhereIsActive does, extended to the optional blog predicate:
"(" + Environment.NewLine + // either clause may be absent, so the first one present has to open the WHERE.
" SELECT " + Environment.NewLine + string sourceClause = AndIsActive("Posts", "P", DBPath) + (filterByBlog ? " AND P.BlogName = @blogName" : string.Empty);
" P.BlogName," + Environment.NewLine + string sourceFilter = sourceClause.Length == 0 ? string.Empty : " WHERE" + sourceClause.Substring(" AND".Length);
" P.PostID," + Environment.NewLine +
" 1925013599 AS LatestNoteTimestamp," + Environment.NewLine + // The zomb-eh branch re-queues that blog's *already collected* posts every 3 days. It needs
" P.NotesGatheredDateTime," + Environment.NewLine + // no blog-filter handling of its own: it reads PostsWithCount, which the filter has already
" COUNT(P.PostID) OVER(PARTITION BY P.BlogName) AS CNT," + Environment.NewLine + // scoped, so it contributes its rows when the filter names zomb-eh and nothing otherwise.
" P.HasNotesGathered," + Environment.NewLine + // That keeps a filtered worklist a strict subset of the unfiltered one -- "--collect 1 X"
" P.NotFound," + Environment.NewLine + // returns exactly the rows "--collect 1" would have returned for X.
" P.PostDate" + Environment.NewLine + string refreshBranch =
" FROM Posts P" + WhereIsActive("Posts", "P", DBPath) + Environment.NewLine +
")," + Environment.NewLine +
"Unioned AS" + Environment.NewLine +
"(" + Environment.NewLine +
" SELECT " + Environment.NewLine +
" BlogName," + Environment.NewLine +
" PostID," + Environment.NewLine +
" LatestNoteTimestamp," + Environment.NewLine +
" NotesGatheredDateTime," + Environment.NewLine +
" CNT," + Environment.NewLine +
" PostDate" + Environment.NewLine +
" FROM PostsWithCount" + Environment.NewLine +
" WHERE NotFound = 0" + Environment.NewLine +
" AND HasNotesGathered = 0" + Environment.NewLine +
"" + Environment.NewLine + "" + Environment.NewLine +
" UNION " + Environment.NewLine + " UNION " + Environment.NewLine +
"" + Environment.NewLine + "" + Environment.NewLine +
@@ -914,7 +904,34 @@ namespace URLNotesGrabberCORE
" FROM PostsWithCount" + Environment.NewLine + " FROM PostsWithCount" + Environment.NewLine +
" WHERE BlogName = 'zomb-eh'" + Environment.NewLine + " WHERE BlogName = 'zomb-eh'" + Environment.NewLine +
" AND NotFound = 0" + Environment.NewLine + " AND NotFound = 0" + Environment.NewLine +
" AND NotesGatheredDateTime < unixepoch('now', 'localtime', '-3 days')" + Environment.NewLine + " AND NotesGatheredDateTime < unixepoch('now', 'localtime', '-3 days')" + Environment.NewLine;
sql = "WITH PostsWithCount AS" + Environment.NewLine +
"(" + Environment.NewLine +
" SELECT " + Environment.NewLine +
" P.BlogName," + Environment.NewLine +
" P.PostID," + Environment.NewLine +
" 1925013599 AS LatestNoteTimestamp," + Environment.NewLine +
" P.NotesGatheredDateTime," + Environment.NewLine +
" COUNT(P.PostID) OVER(PARTITION BY P.BlogName) AS CNT," + Environment.NewLine +
" P.HasNotesGathered," + Environment.NewLine +
" P.NotFound," + Environment.NewLine +
" P.PostDate" + Environment.NewLine +
" FROM Posts P" + sourceFilter + Environment.NewLine +
")," + Environment.NewLine +
"Unioned AS" + Environment.NewLine +
"(" + Environment.NewLine +
" SELECT " + Environment.NewLine +
" BlogName," + Environment.NewLine +
" PostID," + Environment.NewLine +
" LatestNoteTimestamp," + Environment.NewLine +
" NotesGatheredDateTime," + Environment.NewLine +
" CNT," + Environment.NewLine +
" PostDate" + Environment.NewLine +
" FROM PostsWithCount" + Environment.NewLine +
" WHERE NotFound = 0" + Environment.NewLine +
" AND HasNotesGathered = 0" + Environment.NewLine +
refreshBranch +
")" + Environment.NewLine + ")" + Environment.NewLine +
"SELECT" + Environment.NewLine + "SELECT" + Environment.NewLine +
" U.BlogName," + Environment.NewLine + " U.BlogName," + Environment.NewLine +
@@ -944,6 +961,11 @@ namespace URLNotesGrabberCORE
" ( select BlogName, count(PostID) as CNT from Posts" + WhereIsActive("Posts", "", DBPath) + " group by BlogName) CNT on CNT.blogName = Posts.BlogName " + " ( select BlogName, count(PostID) as CNT from Posts" + WhereIsActive("Posts", "", DBPath) + " group by BlogName) CNT on CNT.blogName = Posts.BlogName " +
"WHERE NotFound = 0 " + AndIsActive("Posts", "Posts", DBPath) + Environment.NewLine; "WHERE NotFound = 0 " + AndIsActive("Posts", "Posts", DBPath) + Environment.NewLine;
if (filterByBlog)
{
sql += " AND Posts.BlogName = @blogName " + Environment.NewLine;
}
if (beforeDate.HasValue) if (beforeDate.HasValue)
{ {
long unixTimestamp = new DateTimeOffset(beforeDate.Value).ToUnixTimeSeconds(); long unixTimestamp = new DateTimeOffset(beforeDate.Value).ToUnixTimeSeconds();
@@ -963,6 +985,9 @@ namespace URLNotesGrabberCORE
using (SQLiteCommand command = new SQLiteCommand(sql, connection)) using (SQLiteCommand command = new SQLiteCommand(sql, connection))
{ {
if (filterByBlog)
command.Parameters.AddWithValue("@blogName", blogName);
using (SQLiteDataReader reader = command.ExecuteReader()) using (SQLiteDataReader reader = command.ExecuteReader())
{ {
while (reader.Read()) while (reader.Read())
+56 -13
View File
@@ -220,7 +220,7 @@ namespace URLNotesGrabberCORE
if (args.Length < 2) if (args.Length < 2)
{ {
Console.WriteLine("--Expected WITHOUTNOTESONLY (0, 1) [OPTIONAL: BEFOREDATE]--"); Console.WriteLine("--Expected WITHOUTNOTESONLY (0, 1) [OPTIONAL: BEFOREDATE] [OPTIONAL: BLOGNAME]--");
exitCode = 2; exitCode = 2;
break; break;
} }
@@ -242,28 +242,63 @@ namespace URLNotesGrabberCORE
Console.WriteLine("Without Notes Only: {0}\t{1}", withoutNotesOnly, args[1]); Console.WriteLine("Without Notes Only: {0}\t{1}", withoutNotesOnly, args[1]);
} }
// Parse optional beforeDate parameter // Trailing arguments are the optional cutoff date and the optional blog filter, in
if (args.Length >= 3 && !string.IsNullOrEmpty(args[2])) // either order. A token that parses as a date is the cutoff; anything else is a blog
// name -- which is why an unparseable token is no longer an error here. "--blog=name"
// forces the blog reading for the rare name that would otherwise parse as a date.
string? collectBlogName = null;
bool badCollectArg = false;
for (int i = 2; i < args.Length; i++)
{ {
if (DateTime.TryParse(args[2], out DateTime parsedDate)) string arg = args[i];
if (string.IsNullOrWhiteSpace(arg))
continue;
if (arg.StartsWith("--blog=", StringComparison.OrdinalIgnoreCase))
{
collectBlogName = arg.Substring("--blog=".Length);
if (string.IsNullOrWhiteSpace(collectBlogName))
{
Console.WriteLine("ERROR: --blog= requires a blog name");
badCollectArg = true;
break;
}
}
else if (!explicitDateSupplied && DateTime.TryParse(arg, out DateTime parsedDate))
{ {
beforeDate = parsedDate; beforeDate = parsedDate;
explicitDateSupplied = true; explicitDateSupplied = true;
Console.WriteLine($"Filter: Collecting notes for posts with NotesGatheredDateTime < {beforeDate}"); Console.WriteLine($"Filter: Collecting notes for posts with NotesGatheredDateTime < {beforeDate}");
} }
else if (collectBlogName == null)
{
collectBlogName = arg;
}
else else
{ {
Console.WriteLine($"ERROR: Invalid date format '{args[2]}'"); Console.WriteLine($"ERROR: Unexpected argument '{arg}'");
exitCode = 2; badCollectArg = true;
break; break;
} }
} }
if (badCollectArg)
{
exitCode = 2;
break;
}
if (collectBlogName != null)
Console.WriteLine($"Filter: Collecting notes for posts by '{collectBlogName}' only");
// Mode 0 (full re-check) with no explicit date is a *managed* run: freeze the cutoff and // Mode 0 (full re-check) with no explicit date is a *managed* run: freeze the cutoff and
// persist it so an interrupted run resumes against the same cutoff and a completed run stops // persist it so an interrupted run resumes against the same cutoff and a completed run stops
// instead of restarting. Mode 1 and explicit-date runs keep their existing behavior. // instead of restarting. Mode 1, explicit-date and blog-scoped runs keep their existing
// behavior -- a single blog covers a slice of the worklist, so letting it write the shared
// run state would mark the whole re-check complete after collecting one blog.
bool managedCollectRun = false; bool managedCollectRun = false;
if (!withoutNotesOnly && !explicitDateSupplied) if (!withoutNotesOnly && !explicitDateSupplied && collectBlogName == null)
{ {
DataAccess.EnsureCollectRunStateTableExists(); DataAccess.EnsureCollectRunStateTableExists();
var runState = DataAccess.GetCollectRunState(); var runState = DataAccess.GetCollectRunState();
@@ -281,7 +316,7 @@ namespace URLNotesGrabberCORE
managedCollectRun = true; managedCollectRun = true;
} }
exitCode = CollectNotes(settings.GetValue<string>("PathOutput"), withoutNotesOnly, beforeDate, managedCollectRun).GetAwaiter().GetResult(); exitCode = CollectNotes(settings.GetValue<string>("PathOutput"), withoutNotesOnly, beforeDate, managedCollectRun, collectBlogName).GetAwaiter().GetResult();
break; break;
case "--blogsR": //collect notes from all posts case "--blogsR": //collect notes from all posts
@@ -413,7 +448,7 @@ namespace URLNotesGrabberCORE
Console.WriteLine("--blogs\t For each Blog in DB, write blogname to file"); Console.WriteLine("--blogs\t For each Blog in DB, write blogname to file");
Console.WriteLine("--collect [0|1] [datetime]\t Collect Notes from API. 1=only posts without notes. 0=full re-check of all posts: a single resumable pass (interrupt & relaunch to resume; stops when complete, retrigger for a new pass). Optional datetime overrides the cutoff and runs as a one-off (bypasses resume tracking)."); Console.WriteLine("--collect [0|1] [datetime] [blogname]\t Collect Notes from API. 1=only posts without notes. 0=full re-check of all posts: a single resumable pass (interrupt & relaunch to resume; stops when complete, retrigger for a new pass). Optional datetime overrides the cutoff and runs as a one-off (bypasses resume tracking). Optional blogname restricts the run to that blog (exact, case-sensitive match) and also runs as a one-off; e.g. \"--collect 1 zomb-eh\". datetime and blogname may be given in either order - use --blog=name if a blog name would otherwise parse as a date.");
Console.WriteLine("--blogsR\t For each Note that is a REPLY, write blogname to file "); Console.WriteLine("--blogsR\t For each Note that is a REPLY, write blogname to file ");
@@ -1294,9 +1329,17 @@ if (shouldInsert)
// blipping on one post. Past this, skipping post-by-post would just hammer a closed door. // blipping on one post. Past this, skipping post-by-post would just hammer a closed door.
const int MaxConsecutiveTransient = 10; const int MaxConsecutiveTransient = 10;
static async Task<int> CollectNotes(string outPath, bool withoutNotesOnly = true, DateTime? beforeDate = null, bool managedRun = false) static async Task<int> CollectNotes(string outPath, bool withoutNotesOnly = true, DateTime? beforeDate = null, bool managedRun = false, string? blogName = null)
{ {
List<Tuple<string, long, long, long>> posts = DataAccess.GetPosts(withoutNotesOnly, beforeDate); List<Tuple<string, long, long, long>> posts = DataAccess.GetPosts(withoutNotesOnly, beforeDate, blogName);
if (posts.Count == 0 && !string.IsNullOrWhiteSpace(blogName))
{
// BlogName is matched exactly, so a typo or a case mismatch looks identical to "nothing
// left to collect". Say so rather than reporting a silent, instant success.
Console.WriteLine($"No posts to collect for blog '{blogName}'. Either it is fully collected, or the name does not match a stored blog (the match is case-sensitive).");
return 0;
}
// Posts attempted (with a definitive, non-throttle result) during *this* process. Guarantees a single // Posts attempted (with a definitive, non-throttle result) during *this* process. Guarantees a single
// attempt pass: once every remaining post has been attempted, the loop stops instead of spinning on a // attempt pass: once every remaining post has been attempted, the loop stops instead of spinning on a
@@ -1392,7 +1435,7 @@ if (shouldInsert)
} }
// Re-fetch the updated list after processing the current post // Re-fetch the updated list after processing the current post
posts = DataAccess.GetPosts(withoutNotesOnly, beforeDate); posts = DataAccess.GetPosts(withoutNotesOnly, beforeDate, blogName);
} }
} }