Author SHA1 Message Date
jimandClaude Opus 5 f41957fd2f feat(collect): let --force ignore the re-collect cooldown
--collect 1 <blog> returned an empty worklist whenever the periodic
re-queue branch was still inside its 3-day window, with no way to ask
for the re-collect early. GetPosts now composes that age predicate
conditionally, and the existing global --force flag - already "ignore
cooldown" for --likes - drives it.

Only the age gate drops: NotFound = 0, the IsActive filter and the blog
scoping still apply. The flag is a no-op for mode 0, which re-collects
every post regardless, and says so rather than pretending to act.

Co-Authored-By: Claude Opus 5 <[email protected]>
2026-08-23 04:34:45 -05:00
jimandClaude Sonnet 5 a28c5cc9ec fix(dbbrowser): rewrite saved queries for the Notes integer schema
Notes.RootBlogName/NoteBlogName/Type were replaced by RootBlogId/NoteBlogId/
TypeId (resolved via BlogNames and NoteTypes) back on 2026-08-07. The saved
DB Browser for SQLite queries in RERUN.sqbpro and TL.sqbpro still referenced
the old text columns and failed against the migrated TL.db.

Rewrote the 5 affected queries per TL.db.md's porting guide: Notes<->Blogs
joins go through Blogs.BlogId in one hop, Notes<->Posts joins route through
BlogNames (Posts has no BlogId), and type filters resolve through NoteTypes.
Verified read-only against the live TL.db -- all 5 execute without error.

Co-Authored-By: Claude Sonnet 5 <[email protected]>
2026-08-20 09:07:08 -05:00
jim d6a96f7885 Merge branch 'claude/posttype-noargs-backfill' into master 2026-08-20 08:15:54 -05:00
jimandClaude Opus 5 864b468d96 fix(posttype): run the migration and backfill in no-args mode
The backfill hung off --ingest, --output, --correct, --importposts and
--updatepaths, because those were where EnsureTTFileHelperColumnsExist was
already being called. But the no-argument traversal is the mode that
actually gets run day to day, and it called none of them -- so the command
used most often was the one command that never repaired an untyped row.

TraverseDirectory already types the posts it inserts, from the filename it
is reading. This closes the other half: the rows already sitting untyped
now get fixed by an ordinary run, with no separate maintenance command.

Called before BeginImportSession so it uses its own connection rather than
contending with the import session's, and before the traversal so existing
rows are typed first and newly inserted ones arrive already typed.

Co-Authored-By: Claude Opus 5 <[email protected]>
2026-08-20 08:15:54 -05:00
5 changed files with 169 additions and 133 deletions
+1 -1
View File
@@ -31,7 +31,7 @@ dotnet run -- --test [blogname] [postID] # Test API for specific post
- `--test [blogname] [postID]`: Test API note collection - `--test [blogname] [postID]`: Test API note collection
- `--posts`: Export post blogs to file - `--posts`: Export post blogs to file
- `--blogs`: Export blog list to file - `--blogs`: Export blog list to file
- `--collect [0|1] [datetime] [blogname]`: Collect notes for posts in DB. Optional `blogname` restricts the run to one blog (exact match), e.g. `--collect 1 zomb-eh` - `--collect [0|1] [datetime] [blogname]`: Collect notes for posts in DB. Optional `blogname` restricts the run to one blog (exact match), e.g. `--collect 1 zomb-eh`. Add `--force` to ignore the periodic re-collect cooldown so already-collected posts are re-queued immediately (mode 1 only)
- `--blogsR`: Export reply blogs to file - `--blogsR`: Export reply blogs to file
- `--blogsO [start] [stop]`: Export blogs within range - `--blogsO [start] [stop]`: Export blogs within range
+25 -16
View File
@@ -9,8 +9,8 @@ WHERE (BlogName, PostID) IN (
SELECT 1 SELECT 1
FROM Notes n FROM Notes n
WHERE n.PostID = p.PostID WHERE n.PostID = p.PostID
AND n.RootBlogName = p.BlogName AND n.RootBlogId = (SELECT BlogId FROM BlogNames WHERE BlogName = p.BlogName)
--AND n.Type NOT IN ('reblog', 'reply') --AND n.TypeId NOT IN (SELECT TypeId FROM NoteTypes WHERE Type IN ('reblog', 'reply'))
) )
ORDER BY P.PostDate ASC ORDER BY P.PostDate ASC
--LIMIT 500 --LIMIT 500
@@ -29,10 +29,14 @@ blogname in
'nudenymph', 'nudenymph',
'caylachief' 'caylachief'
)</sql><sql name="New Notes">select P.slug, N.replyText, n.RootBlogName, n.PostID, NoteBlogName || '.tumblr.com' as NoteBlogName, DatetimeCrawled, TimeStamp, type, n.RootBlogName || '.tumblr.com/post/' || n.postid, datetime(timestamp, 'unixepoch') )</sql><sql name="New Notes">select P.slug, N.replyText, rbn.BlogName as RootBlogName, n.PostID, nbn.BlogName || '.tumblr.com' as NoteBlogName, DatetimeCrawled, TimeStamp, nt.Type, rbn.BlogName || '.tumblr.com/post/' || n.postid, datetime(timestamp, 'unixepoch')
from Notes N inner join Posts P on p.PostID = n.PostID from Notes N
inner join Posts P on p.PostID = n.PostID
inner join BlogNames rbn on rbn.BlogId = n.RootBlogId
inner join BlogNames nbn on nbn.BlogId = n.NoteBlogId
inner join NoteTypes nt on nt.TypeId = n.TypeId
where where
DatetimeCrawled &gt; '2026-08-07 11:47:22' and type like 'r%' DatetimeCrawled &gt; '2026-08-07 11:47:22' and nt.Type like 'r%'
and P.IsActive = 1 and P.IsActive = 1
order by n.DatetimeCrawled</sql><sql name="Pull Blogs">SELECT distinct order by n.DatetimeCrawled</sql><sql name="Pull Blogs">SELECT distinct
'''' || blogname || ''',', '''' || blogname || ''',',
@@ -41,31 +45,36 @@ order by n.DatetimeCrawled</sql><sql name="Pull Blogs">SELECT distinct
FROM FROM
Blogs Blogs
inner JOIN inner JOIN
Notes on notes.noteBlogName = blogs.BlogName Notes on notes.noteBlogId = blogs.BlogId
inner JOIN
NoteTypes on NoteTypes.TypeId = Notes.TypeId
WHERE WHERE
HasBeenOutput = 0 and type = 'reblog' HasBeenOutput = 0 and NoteTypes.Type = 'reblog'
order by order by
Notes.Type desc, NoteTypes.Type desc,
DateAdded desc DateAdded desc
LIMIT 100;</sql><sql name="SQL 7">WITH ReplyCounts AS ( LIMIT 100;</sql><sql name="SQL 7">WITH ReplyCounts AS (
SELECT SELECT
NoteBlogName, NoteBlogId,
COUNT(DISTINCT replyText) AS DistinctReplyCount COUNT(DISTINCT replyText) AS DistinctReplyCount
FROM Notes FROM Notes
where replyText &lt;&gt; '.' where replyText &lt;&gt; '.'
GROUP BY NoteBlogName GROUP BY NoteBlogId
) )
SELECT SELECT
n.RootBlogName || '.tumblr.com/post/' || n.PostID AS PostURL, postid, rbn.BlogName || '.tumblr.com/post/' || n.PostID AS PostURL, postid,
n.NoteBlogName, nbn.BlogName AS NoteBlogName,
n.replyText, n.replyText,
c.DistinctReplyCount c.DistinctReplyCount
FROM Notes n FROM Notes n
JOIN ReplyCounts c ON n.NoteBlogName = c.NoteBlogName JOIN ReplyCounts c ON n.NoteBlogId = c.NoteBlogId
where replyText &lt;&gt; '.' and type &lt;&gt; 'reply' JOIN BlogNames rbn ON rbn.BlogId = n.RootBlogId
--AND N.NoteBlogName NOT IN ( 'roadblocker21', 'thesaddemon666', 'edwardabbeyhoffman', 'tattedsoldier20', 'zomb-eh', 'animalistic13', 'indken', 'maccloud1592', JOIN BlogNames nbn ON nbn.BlogId = n.NoteBlogId
JOIN NoteTypes t ON t.TypeId = n.TypeId
where replyText &lt;&gt; '.' and t.Type &lt;&gt; 'reply'
--AND nbn.BlogName NOT IN ( 'roadblocker21', 'thesaddemon666', 'edwardabbeyhoffman', 'tattedsoldier20', 'zomb-eh', 'animalistic13', 'indken', 'maccloud1592',
--'moss-wizard', 'supertrucker12682', 'exploringthrupics', 'padeyepete' ) --'moss-wizard', 'supertrucker12682', 'exploringthrupics', 'padeyepete' )
order by c.DistinctReplyCount desc, n.NoteBlogName, n.DateModified desc, replyText, RootBlogName, PostID</sql><sql name="Collect">WITH PostsWithCount AS ( SELECT P.BlogName, P.PostID, 1925013599 AS LatestNoteTimestamp, P.NotesGatheredDateTime, COUNT(P.PostID) OVER(PARTITION BY P.BlogName) AS CNT, P.HasNotesGathered, P.NotFound, P.PostDate FROM Posts P WHERE COALESCE(P.IsActive, 1) = 1 ), Unioned AS ( SELECT BlogName, PostID, LatestNoteTimestamp, NotesGatheredDateTime, CNT, PostDate FROM PostsWithCount WHERE NotFound = 0 AND HasNotesGathered = 0 UNION SELECT BlogName, PostID, LatestNoteTimestamp, NotesGatheredDateTime, CNT, PostDate FROM PostsWithCount WHERE BlogName = 'zomb-eh' AND NotFound = 0 AND NotesGatheredDateTime &lt; unixepoch('now', 'localtime', '-3 days') ) SELECT U.BlogName, U.PostID, U.LatestNoteTimestamp, U.NotesGatheredDateTime, U.CNT FROM Unioned U WHERE (U.NotesGatheredDateTime &lt; 1786134037 OR U.NotesGatheredDateTime IS NULL) ORDER BY U.NotesGatheredDateTime, U.PostDate DESC, U.BlogName, U.PostID;</sql><sql name="Del Posts">delete from posts where postid in order by c.DistinctReplyCount desc, nbn.BlogName, n.DateModified desc, replyText, rbn.BlogName, PostID</sql><sql name="Collect">WITH PostsWithCount AS ( SELECT P.BlogName, P.PostID, 1925013599 AS LatestNoteTimestamp, P.NotesGatheredDateTime, COUNT(P.PostID) OVER(PARTITION BY P.BlogName) AS CNT, P.HasNotesGathered, P.NotFound, P.PostDate FROM Posts P WHERE COALESCE(P.IsActive, 1) = 1 ), Unioned AS ( SELECT BlogName, PostID, LatestNoteTimestamp, NotesGatheredDateTime, CNT, PostDate FROM PostsWithCount WHERE NotFound = 0 AND HasNotesGathered = 0 UNION SELECT BlogName, PostID, LatestNoteTimestamp, NotesGatheredDateTime, CNT, PostDate FROM PostsWithCount WHERE BlogName = 'zomb-eh' AND NotFound = 0 AND NotesGatheredDateTime &lt; unixepoch('now', 'localtime', '-3 days') ) SELECT U.BlogName, U.PostID, U.LatestNoteTimestamp, U.NotesGatheredDateTime, U.CNT FROM Unioned U WHERE (U.NotesGatheredDateTime &lt; 1786134037 OR U.NotesGatheredDateTime IS NULL) ORDER BY U.NotesGatheredDateTime, U.PostDate DESC, U.BlogName, U.PostID;</sql><sql name="Del Posts">delete from posts where postid in
( (
'741662499571728384', '741662499571728384',
178892849664, 178892849664,
+12 -6
View File
@@ -9,16 +9,22 @@ WHERE
ORDER BY ORDER BY
postdate desc</sql><sql name="SQL 2*">SELECT postdate desc</sql><sql name="SQL 2*">SELECT
datetime(TimeStamp, 'unixepoch'), datetime(TimeStamp, 'unixepoch'),
RootBlogName || '.tumblr.com/post/' || N.postid, rbn.BlogName || '.tumblr.com/post/' || N.postid,
*, *,
NoteBlogName || '.tumblr.com' nbn.BlogName || '.tumblr.com'
FROM FROM
Notes N Notes N
inner JOIN inner JOIN
Posts P on P.PostID = N.PostID and P.BlogName = N.RootBlogName BlogNames rbn on rbn.BlogId = N.RootBlogId
WHERE RootBlogName NOT IN ('xlittle-ghost', 'glimmerin-darlin', 'vvenus-child')␍ inner JOIN
and type like 'r%'␍ BlogNames nbn on nbn.BlogId = N.NoteBlogId
and RootBlogName = 'zomb-eh'␍ inner JOIN
NoteTypes t on t.TypeId = N.TypeId
inner JOIN
Posts P on P.PostID = N.PostID and P.BlogName = rbn.BlogName
WHERE rbn.BlogName NOT IN ('xlittle-ghost', 'glimmerin-darlin', 'vvenus-child')
and t.Type like 'r%'
and rbn.BlogName = 'zomb-eh'
and P.HasImage = 1 and P.HasImage = 1
ORDER BY ORDER BY
TimeStamp desc</sql><current_tab id="1"/></tab_sql></sqlb_project> TimeStamp desc</sql><current_tab id="1"/></tab_sql></sqlb_project>
+10 -2
View File
@@ -858,9 +858,10 @@ namespace URLNotesGrabberCORE
/// ///
/// </summary> /// </summary>
/// <param name="withoutNotesOnly"></param> /// <param name="withoutNotesOnly"></param>
/// <param name="ignoreRefreshCooldown">Drops the age gate on the periodic re-queue branch (--force).</param>
/// <param name="DBPath"></param> /// <param name="DBPath"></param>
/// <returns>blogName, postID, lastNoteTimestamp, notesGatheredTimestamp</returns> /// <returns>blogName, postID, lastNoteTimestamp, notesGatheredTimestamp</returns>
public static List<Tuple<string, long, long, long>> GetPosts(bool withoutNotesOnly = false, DateTime? beforeDate = null, string? blogName = null, string? DBPath = null) public static List<Tuple<string, long, long, long>> GetPosts(bool withoutNotesOnly = false, DateTime? beforeDate = null, string? blogName = null, bool ignoreRefreshCooldown = false, string? DBPath = null)
{ {
DBPath ??= GetDefaultDbPath(); DBPath ??= GetDefaultDbPath();
using SQLiteConnection connection = new SQLiteConnection("Data Source=" + DBPath); using SQLiteConnection connection = new SQLiteConnection("Data Source=" + DBPath);
@@ -895,6 +896,13 @@ namespace URLNotesGrabberCORE
// scoped, so it contributes its rows when the filter names zomb-eh and nothing otherwise. // scoped, so it contributes its rows when the filter names zomb-eh and nothing otherwise.
// That keeps a filtered worklist a strict subset of the unfiltered one -- "--collect 1 X" // That keeps a filtered worklist a strict subset of the unfiltered one -- "--collect 1 X"
// returns exactly the rows "--collect 1" would have returned for X. // returns exactly the rows "--collect 1" would have returned for X.
//
// --force drops the age gate only. NotFound = 0 and the IsActive/blog scoping above still
// apply: the flag is "re-collect early", not "collect rows every other path excludes".
string refreshCooldownClause = ignoreRefreshCooldown
? string.Empty
: " AND NotesGatheredDateTime < unixepoch('now', 'localtime', '-3 days')" + Environment.NewLine;
string refreshBranch = string refreshBranch =
"" + Environment.NewLine + "" + Environment.NewLine +
" UNION " + Environment.NewLine + " UNION " + Environment.NewLine +
@@ -909,7 +917,7 @@ namespace URLNotesGrabberCORE
" FROM PostsWithCount" + Environment.NewLine + " FROM PostsWithCount" + Environment.NewLine +
" WHERE BlogName = 'zomb-eh'" + Environment.NewLine + " WHERE BlogName = 'zomb-eh'" + Environment.NewLine +
" AND NotFound = 0" + Environment.NewLine + " AND NotFound = 0" + Environment.NewLine +
" AND NotesGatheredDateTime < unixepoch('now', 'localtime', '-3 days')" + Environment.NewLine; refreshCooldownClause;
sql = "WITH PostsWithCount AS" + Environment.NewLine + sql = "WITH PostsWithCount AS" + Environment.NewLine +
"(" + Environment.NewLine + "(" + Environment.NewLine +
+19 -6
View File
@@ -148,6 +148,12 @@ namespace URLNotesGrabberCORE
if (args.Length == 0) //Traverse folder structure to add posts and thus blogs to DB if (args.Length == 0) //Traverse folder structure to add posts and thus blogs to DB
{ {
int postsAdded = 0; int postsAdded = 0;
// This is the mode that actually gets run day to day, so the schema migration and
// the PostType backfill have to happen here too. They used to hang off --ingest,
// --output and friends only, which meant the untyped rows this traversal creates
// could sit unrepaired indefinitely while the one command everyone runs skipped
// the fix entirely. Idempotent, so paying it on every run costs nothing.
DataAccess.EnsureTTFileHelperColumnsExist();
try try
{ {
DataAccess.EnableImportModePragmas(); DataAccess.EnableImportModePragmas();
@@ -316,7 +322,12 @@ namespace URLNotesGrabberCORE
managedCollectRun = true; managedCollectRun = true;
} }
exitCode = CollectNotes(settings.GetValue<string>("PathOutput"), withoutNotesOnly, beforeDate, managedCollectRun, collectBlogName).GetAwaiter().GetResult(); if (forceIgnoreCooldown)
Console.WriteLine(withoutNotesOnly
? "--force: ignoring the periodic re-collect cooldown; already-collected posts in scope are re-queued now"
: "--force: no effect in mode 0 - a full re-check already re-collects every post");
exitCode = CollectNotes(settings.GetValue<string>("PathOutput"), withoutNotesOnly, beforeDate, managedCollectRun, collectBlogName, forceIgnoreCooldown).GetAwaiter().GetResult();
break; break;
case "--blogsR": //collect notes from all posts case "--blogsR": //collect notes from all posts
@@ -448,7 +459,7 @@ namespace URLNotesGrabberCORE
Console.WriteLine("--blogs\t For each Blog in DB, write blogname to file"); Console.WriteLine("--blogs\t For each Blog in DB, write blogname to file");
Console.WriteLine("--collect [0|1] [datetime] [blogname]\t Collect Notes from API. 1=only posts without notes. 0=full re-check of all posts: a single resumable pass (interrupt & relaunch to resume; stops when complete, retrigger for a new pass). Optional datetime overrides the cutoff and runs as a one-off (bypasses resume tracking). Optional blogname restricts the run to that blog (exact, case-sensitive match) and also runs as a one-off; e.g. \"--collect 1 zomb-eh\". datetime and blogname may be given in either order - use --blog=name if a blog name would otherwise parse as a date."); Console.WriteLine("--collect [0|1] [datetime] [blogname]\t Collect Notes from API. 1=only posts without notes. 0=full re-check of all posts: a single resumable pass (interrupt & relaunch to resume; stops when complete, retrigger for a new pass). Optional datetime overrides the cutoff and runs as a one-off (bypasses resume tracking). Optional blogname restricts the run to that blog (exact, case-sensitive match) and also runs as a one-off; e.g. \"--collect 1 zomb-eh\". datetime and blogname may be given in either order - use --blog=name if a blog name would otherwise parse as a date. Add --force to ignore the periodic re-collect cooldown and re-queue already-collected posts immediately (mode 1 only).");
Console.WriteLine("--blogsR\t For each Note that is a REPLY, write blogname to file "); Console.WriteLine("--blogsR\t For each Note that is a REPLY, write blogname to file ");
@@ -460,7 +471,7 @@ namespace URLNotesGrabberCORE
Console.WriteLine("--likes\t Fetch likes: initial backfill for new blogs, incremental refresh for blogs past cooldown. Optional blog name forces single-blog run."); Console.WriteLine("--likes\t Fetch likes: initial backfill for new blogs, incremental refresh for blogs past cooldown. Optional blog name forces single-blog run.");
Console.WriteLine("--force\t (with --likes) Ignore cooldown and refresh every fully-backfilled blog"); Console.WriteLine("--force\t Ignore refresh cooldowns: with --likes, refresh every fully-backfilled blog; with --collect 1, re-queue already-collected posts without waiting out their cooldown");
Console.WriteLine("--urldump\t Scan all posts' text columns and extract suspected URLs to configured file"); Console.WriteLine("--urldump\t Scan all posts' text columns and extract suspected URLs to configured file");
@@ -1339,15 +1350,17 @@ if (shouldInsert)
// blipping on one post. Past this, skipping post-by-post would just hammer a closed door. // blipping on one post. Past this, skipping post-by-post would just hammer a closed door.
const int MaxConsecutiveTransient = 10; const int MaxConsecutiveTransient = 10;
static async Task<int> CollectNotes(string outPath, bool withoutNotesOnly = true, DateTime? beforeDate = null, bool managedRun = false, string? blogName = null) static async Task<int> CollectNotes(string outPath, bool withoutNotesOnly = true, DateTime? beforeDate = null, bool managedRun = false, string? blogName = null, bool ignoreRefreshCooldown = false)
{ {
List<Tuple<string, long, long, long>> posts = DataAccess.GetPosts(withoutNotesOnly, beforeDate, blogName); List<Tuple<string, long, long, long>> posts = DataAccess.GetPosts(withoutNotesOnly, beforeDate, blogName, ignoreRefreshCooldown);
if (posts.Count == 0 && !string.IsNullOrWhiteSpace(blogName)) if (posts.Count == 0 && !string.IsNullOrWhiteSpace(blogName))
{ {
// BlogName is matched exactly, so a typo or a case mismatch looks identical to "nothing // BlogName is matched exactly, so a typo or a case mismatch looks identical to "nothing
// left to collect". Say so rather than reporting a silent, instant success. // left to collect". Say so rather than reporting a silent, instant success.
Console.WriteLine($"No posts to collect for blog '{blogName}'. Either it is fully collected, or the name does not match a stored blog (the match is case-sensitive)."); Console.WriteLine($"No posts to collect for blog '{blogName}'. Either it is fully collected, or the name does not match a stored blog (the match is case-sensitive).");
if (withoutNotesOnly && !ignoreRefreshCooldown)
Console.WriteLine("Already-collected posts are re-queued only once their cooldown elapses; add --force to re-collect them now.");
return 0; return 0;
} }
@@ -1445,7 +1458,7 @@ if (shouldInsert)
} }
// Re-fetch the updated list after processing the current post // Re-fetch the updated list after processing the current post
posts = DataAccess.GetPosts(withoutNotesOnly, beforeDate, blogName); posts = DataAccess.GetPosts(withoutNotesOnly, beforeDate, blogName, ignoreRefreshCooldown);
} }
} }