select * from Blogs --update blogs set HasBeenOutput = 1 where HasBeenOutput = 0 AND blogname in ('udontn33dh1m', 'tyrantsxblood', 'sentry-34', 'deathcabforfrankie', 'abheith-sasta', 'kuwaiikittenghost', 'kansasmud', '03diesel', 'itzameallieee', 'fireball-temptations', 'mamaisamess', '906raised-and-dogobsessed', 'the-queerist-wolf', 'counting-corpsess', 'aqueenbby', 'maybememoriesx', 'queenofnevers', 'obsidian-psyche', 'lilmissellexo', 'alittlebunny95', 'rage--and--grace', 'savage-deniz', 'daddyspuddleprincess', 'littledefenstration', 'bearded-snorlax', 'thosesummerskiess', 'tubadtoph', 'lieutenant-dan-ice-cream', 'brittvnybitch', 'a-smol-gayologist', 'sum1random', 'samsternelly', 'littlemouseylauren', 'princessleiaorgasma', 'bloodstaineddkisses', 'letsfacerealitybabe', 'x--marks--thespot', 'space-and-suffering', 'rinarootski', 'thiccandtired', 'fvcking-scvmbag', 'fullblownwizard', 'bigjewface', 'unleash-the-krayken', 'bumpintheroad', 'liltexasjedii', 'nawtydude', 'queenpeachqueen', 'the-clansman', 'balmain-bxtch' )select P.slug, N.replyText, rbn.BlogName as RootBlogName, n.PostID, nbn.BlogName || '.tumblr.com' as NoteBlogName, DatetimeCrawled, TimeStamp, nt.Type, rbn.BlogName || '.tumblr.com/post/' || n.postid, datetime(timestamp, 'unixepoch') from Notes N inner join Posts P on p.PostID = n.PostID inner join Blogs rbn on rbn.BlogId = n.RootBlogId inner join Blogs nbn on nbn.BlogId = n.NoteBlogId inner join NoteTypes nt on nt.TypeId = n.TypeId where DatetimeCrawled > '2026-08-07 11:47:22' and nt.Type like 'r%' and P.IsActive = 1 order by n.DatetimeCrawledWITH ReplyCounts AS ( SELECT NoteBlogId, COUNT(DISTINCT replyText) AS DistinctReplyCount FROM Notes where replyText <> '.' GROUP BY NoteBlogId ) SELECT rbn.BlogName || '.tumblr.com/post/' || n.PostID AS PostURL, postid, nbn.BlogName AS NoteBlogName, n.replyText, c.DistinctReplyCount FROM Notes n JOIN ReplyCounts c ON n.NoteBlogId = c.NoteBlogId JOIN Blogs rbn ON rbn.BlogId = n.RootBlogId JOIN Blogs nbn ON nbn.BlogId = n.NoteBlogId JOIN NoteTypes t ON t.TypeId = n.TypeId where replyText <> '.' and t.Type <> 'reply' --AND nbn.BlogName NOT IN ( 'roadblocker21', 'thesaddemon666', 'edwardabbeyhoffman', 'tattedsoldier20', 'zomb-eh', 'animalistic13', 'indken', 'maccloud1592', --'moss-wizard', 'supertrucker12682', 'exploringthrupics', 'padeyepete' ) order by c.DistinctReplyCount desc, nbn.BlogName, n.DateModified desc, replyText, rbn.BlogName, PostIDWITH PostsWithCount AS ( SELECT P.BlogName, P.PostID, 1925013599 AS LatestNoteTimestamp, P.NotesGatheredDateTime, COUNT(P.PostID) OVER(PARTITION BY P.BlogName) AS CNT, P.HasNotesGathered, P.NotFound, P.PostDate FROM Posts P WHERE COALESCE(P.IsActive, 1) = 1 ), Unioned AS ( SELECT BlogName, PostID, LatestNoteTimestamp, NotesGatheredDateTime, CNT, PostDate FROM PostsWithCount WHERE NotFound = 0 AND HasNotesGathered = 0 UNION SELECT BlogName, PostID, LatestNoteTimestamp, NotesGatheredDateTime, CNT, PostDate FROM PostsWithCount WHERE BlogName = 'zomb-eh' AND NotFound = 0 AND NotesGatheredDateTime < unixepoch('now', 'localtime', '-3 days') ) SELECT U.BlogName, U.PostID, U.LatestNoteTimestamp, U.NotesGatheredDateTime, U.CNT FROM Unioned U WHERE (U.NotesGatheredDateTime < 1787237598 OR U.NotesGatheredDateTime IS NULL) ORDER BY U.NotesGatheredDateTime, U.PostDate DESC, U.BlogName, U.PostID;delete from posts where postid in ( '741662499571728384', 178892849664, 178264721139, 177012868749, 169950081964, 755440787056099328 )select * -- delete from notes where postid not in (select distinct postid from posts where IsActive = 1)-- ============================================================================ -- blogs-added-after-august-2026-with-reblog-or-reply.sql -- -- Purpose: Of all blogs in Blogs, find the ones that show up in Notes as the -- engager (NoteBlogId) on a 'reblog' or 'reply' note, sorted by -- DateAdded. (Originally scoped to "added after August 2026" -- -- that cutoff is now removed per request; QUERY 2 shows how to put -- a date floor back if needed.) -- -- Read-only. No INSERT/UPDATE/DELETE/DDL anywhere in this file. -- -- How to use (DB Browser for SQLite): -- 1. File > Open Database -> TL.db -- 2. Execute SQL tab. Each QUERY below is independent; Ctrl+Enter runs just -- the one your cursor is in. -- -- The join, once: -- "Added after August 2026" filters Blogs.DateAdded. "Has a reblog/reply -- note" means the blog is the engager, which is NoteBlogId -- not -- RootBlogId, which is the blog that *owns* the post being reacted to -- (see find-notes-on-inactive-posts.sql for that side). Per TL.db.md, the -- Blogs<->Notes join is a single integer hop and should not be routed -- through Blogs: -- Blogs.BlogId = Notes.NoteBlogId -- EXISTS is used rather than a JOIN so a blog with many qualifying notes -- still contributes one output row. -- -- Excluding notes on an inactive post: same shape as -- find-notes-on-inactive-posts.sql -- Notes only carries RootBlogId (an -- integer), so reaching Posts.IsActive needs the one text hop the rest of -- this file avoids: RootBlogId -> Blogs.BlogName = Posts.BlogName, -- matched on PostID. It's a LEFT JOIN, not an inner one: only 3,867 of -- 20,430 root/engager blogs have any stored Posts rows at all (TL.db.md), -- so most reblog/reply notes have no Posts row to check and must be kept, -- not dropped by an inner join. COALESCE(p.IsActive, 1) = 1 keeps a note -- unless its post is explicitly IsActive = 0 -- NULL (no Posts row, or a -- stored row with no flag written) means live, per the schema's own -- convention (TL.db.md, "Posts.IsActive and Notes.IsActive"). This is a -- big filter in practice: of the blogs that qualified before it, most -- have every one of their reblog/reply notes pointing at a since-removed -- post, not just some -- verified against the live data, not assumed. -- -- On DateAdded: this column is not written consistently -- most rows hold -- ISO 'yyyy-MM-dd HH:mm:ss', but 17k+ hold US 'M/d/yy' from a 2025 bulk -- import (see TL.db.md, "DateAdded is not written consistently"). As text, -- those two shapes do not sort or compare against each other correctly, so -- QUERY 0 normalises both to an ISO date before filtering. In the live data -- every US-format row predates August 2026 anyway (only '12/23/25' and -- '12/24/25' occur), so this makes no difference to the current answer -- -- it's here so the query stays correct if that ever changes. -- ============================================================================ -- ---------------------------------------------------------------------------- -- QUERY 0 / THE ANSWER -- one row per qualifying blog, sorted by DateAdded -- descending (normalised -- see the note above). No date cutoff, but now -- scoped to HasBeenOutput = 0 AND IsActive = 1. 4,739 rows in the live -- data. -- -- earliest_reblog_or_reply_utc is the MIN(TimeStamp) among this blog's -- reblog-or-reply notes (either type counts -- see the column name). -- Getting this meant switching QUERY 0 from EXISTS to an inner JOIN + -- GROUP BY: EXISTS can only tell you a qualifying row is present, not -- aggregate over which ones. No CASE is needed inside the MIN() because -- the WHERE below already restricts the joined rows to reblog/reply, so -- every row a blog brings into the aggregate is one this column should -- consider. A blog appears exactly once, same as before, and this column -- is never NULL for a row that's in the result at all (an earlier -- revision aggregated reblog only, which left it NULL for the 181 blogs -- that had replies but no reblogs). -- ---------------------------------------------------------------------------- WITH BlogsSplit AS ( SELECT b.BlogId, b.BlogName, b.DateAdded, CASE WHEN b.DateAdded LIKE '____-__-__%' THEN 1 ELSE 0 END AS IsIso, -- for the US 'M/d/yy' shape only: everything after the first '/' substr(b.DateAdded, instr(b.DateAdded, '/') + 1) AS RestAfterMonth FROM Blogs b WHERE b.BlogId IS NOT NULL and HasBeenOutput = 0 and IsActive = 1 -- a blog can only match Notes if it has one ), BlogsNorm AS ( SELECT BlogId, BlogName, DateAdded, CASE WHEN IsIso = 1 THEN date(DateAdded) ELSE date( '20' || substr(RestAfterMonth, instr(RestAfterMonth, '/') + 1) || '-' || substr('00' || substr(DateAdded, 1, instr(DateAdded, '/') - 1), -2) || '-' || substr('00' || substr(RestAfterMonth, 1, instr(RestAfterMonth, '/') - 1), -2) ) END AS DateAddedNorm FROM BlogsSplit ) SELECT bn.BlogId, bn.BlogName, bn.DateAdded, bn.DateAddedNorm, datetime(MIN(n.TimeStamp), 'unixepoch') AS earliest_reblog_or_reply_utc FROM BlogsNorm bn JOIN Notes n ON n.NoteBlogId = bn.BlogId JOIN NoteTypes t ON t.TypeId = n.TypeId JOIN Blogs root_bn ON root_bn.BlogId = n.RootBlogId LEFT JOIN Posts p ON p.BlogName = root_bn.BlogName AND p.PostID = n.PostID WHERE t.Type IN ('reblog')--, 'reply') AND COALESCE(p.IsActive, 1) = 1 -- exclude notes on a post explicitly marked removed GROUP BY bn.BlogId, bn.BlogName, bn.DateAdded, bn.DateAddedNorm ORDER BY bn.DateAddedNorm desc; -- ---------------------------------------------------------------------------- -- QUERY 1 -- same answer, with a per-blog breakdown of which type(s) fired -- and how many. Useful once QUERY 0 has rows; redundant while it's empty. -- ---------------------------------------------------------------------------- -- WITH BlogsSplit AS ( ... ), BlogsNorm AS ( ... ) -- reuse the CTEs above -- -- SELECT -- bn.BlogId, -- bn.BlogName, -- bn.DateAddedNorm, -- SUM(CASE WHEN t.Type = 'reblog' THEN 1 ELSE 0 END) AS reblog_count, -- SUM(CASE WHEN t.Type = 'reply' THEN 1 ELSE 0 END) AS reply_count -- FROM BlogsNorm bn -- JOIN Notes n ON n.NoteBlogId = bn.BlogId -- JOIN NoteTypes t ON t.TypeId = n.TypeId -- WHERE t.Type IN ('reblog', 'reply') -- GROUP BY bn.BlogId, bn.BlogName, bn.DateAddedNorm -- ORDER BY bn.DateAddedNorm; -- ---------------------------------------------------------------------------- -- QUERY 2 -- put a date floor back, if wanted later. -- Same as QUERY 0, with one extra line in the outer WHERE: -- AND bn.DateAddedNorm >= '2026-09-01' -- or whatever cutoff -- ---------------------------------------------------------------------------- -- ============================================================================ -- blogs-added-after-august-2026-with-reblog-or-reply.sql -- -- Purpose: Of all blogs in Blogs, find the ones that show up in Notes as the -- engager (NoteBlogId) on a 'reblog' or 'reply' note, sorted by -- DateAdded. (Originally scoped to "added after August 2026" -- -- that cutoff is now removed per request; QUERY 2 shows how to put -- a date floor back if needed.) -- -- Read-only. No INSERT/UPDATE/DELETE/DDL anywhere in this file. -- -- How to use (DB Browser for SQLite): -- 1. File > Open Database -> TL.db -- 2. Execute SQL tab. Each QUERY below is independent; Ctrl+Enter runs just -- the one your cursor is in. -- -- The join, once: -- "Added after August 2026" filters Blogs.DateAdded. "Has a reblog/reply -- note" means the blog is the engager, which is NoteBlogId -- not -- RootBlogId, which is the blog that *owns* the post being reacted to -- (see find-notes-on-inactive-posts.sql for that side). Per TL.db.md, the -- Blogs<->Notes join is a single integer hop and should not be routed -- through Blogs: -- Blogs.BlogId = Notes.NoteBlogId -- EXISTS is used rather than a JOIN so a blog with many qualifying notes -- still contributes one output row. -- -- Excluding notes on an inactive post: same shape as -- find-notes-on-inactive-posts.sql -- Notes only carries RootBlogId (an -- integer), so reaching Posts.IsActive needs the one text hop the rest of -- this file avoids: RootBlogId -> Blogs.BlogName = Posts.BlogName, -- matched on PostID. It's a LEFT JOIN, not an inner one: only 3,867 of -- 20,430 root/engager blogs have any stored Posts rows at all (TL.db.md), -- so most reblog/reply notes have no Posts row to check and must be kept, -- not dropped by an inner join. COALESCE(p.IsActive, 1) = 1 keeps a note -- unless its post is explicitly IsActive = 0 -- NULL (no Posts row, or a -- stored row with no flag written) means live, per the schema's own -- convention (TL.db.md, "Posts.IsActive and Notes.IsActive"). This is a -- big filter in practice: of the blogs that qualified before it, most -- have every one of their reblog/reply notes pointing at a since-removed -- post, not just some -- verified against the live data, not assumed. -- -- On DateAdded: this column is not written consistently -- most rows hold -- ISO 'yyyy-MM-dd HH:mm:ss', but 17k+ hold US 'M/d/yy' from a 2025 bulk -- import (see TL.db.md, "DateAdded is not written consistently"). As text, -- those two shapes do not sort or compare against each other correctly, so -- QUERY 0 normalises both to an ISO date before filtering. In the live data -- every US-format row predates August 2026 anyway (only '12/23/25' and -- '12/24/25' occur), so this makes no difference to the current answer -- -- it's here so the query stays correct if that ever changes. -- ============================================================================ -- ---------------------------------------------------------------------------- -- QUERY 0 / THE ANSWER -- one row per qualifying blog, sorted by -- earliest_reblog_or_reply_utc then DateAdded descending (normalised -- -- see the note above). No date cutoff, but scoped to HasBeenOutput = 0 -- AND IsActive = 1, and now excluding notes on a removed post (see the -- header note above). 1,804 rows in the live data as of this revision -- -- down from 4,396 just before this exclusion was added, because most of -- the blogs that dropped out had *every* reblog/reply note pointing at a -- now-inactive post, not just some (the number moves between runs -- regardless -- crawling and output flip HasBeenOutput/IsActive on live -- rows). -- -- earliest_reblog_or_reply_utc is the earliest TimeStamp among this -- blog's reblog-or-reply notes (either type counts -- see the column -- name); earliest_reblog_or_reply_postid and _root_blogid identify that -- specific note's post: PostID + RootBlogId together, not PostID alone -- -- see TL.db.md ("345 post IDs exist under more than one blog"), same -- caution as in find-notes-on-inactive-posts.sql. Resolve RootBlogId to a -- name via Blogs (or Blogs, tolerating a miss) if you need it. -- -- Getting "which note" rather than just "when" doesn't fit a plain -- MIN()/GROUP BY -- an aggregate can tell you the earliest value but not -- which row it came from. EarliestNote instead ranks each blog's -- reblog/reply notes with ROW_NUMBER() OVER (PARTITION BY NoteBlogId -- ORDER BY TimeStamp), and QUERY 0 takes rn = 1. The ORDER BY carries a -- PostID tiebreak because (NoteBlogId, TimeStamp) is not unique in this -- data -- ties exist (e.g. NoteBlogId 12 has 7 notes at the same -- TimeStamp) -- so without a tiebreak the "earliest" pick would be -- arbitrary among ties rather than deterministic. -- -- EarliestNote also excludes notes on an inactive post before ranking -- (see the header note above), so "earliest" means earliest surviving -- note, not earliest overall -- a blog whose true-earliest note pointed -- at a since-removed post now surfaces its next-earliest live one -- instead. Applying the exclusion here, not as a filter on QUERY 0's -- final rows, matters: filtering after ROW_NUMBER would have picked the -- removed-post note as rn = 1 and then dropped the whole row instead of -- promoting the next candidate. -- ---------------------------------------------------------------------------- WITH BlogsSplit AS ( SELECT b.BlogId, b.BlogName, b.DateAdded, CASE WHEN b.DateAdded LIKE '____-__-__%' THEN 1 ELSE 0 END AS IsIso, -- for the US 'M/d/yy' shape only: everything after the first '/' substr(b.DateAdded, instr(b.DateAdded, '/') + 1) AS RestAfterMonth FROM Blogs b WHERE b.BlogId IS NOT NULL and HasBeenOutput = 0 and IsActive = 1 -- a blog can only match Notes if it has one ), BlogsNorm AS ( SELECT BlogId, BlogName, DateAdded, CASE WHEN IsIso = 1 THEN date(DateAdded) ELSE date( '20' || substr(RestAfterMonth, instr(RestAfterMonth, '/') + 1) || '-' || substr('00' || substr(DateAdded, 1, instr(DateAdded, '/') - 1), -2) || '-' || substr('00' || substr(RestAfterMonth, 1, instr(RestAfterMonth, '/') - 1), -2) ) END AS DateAddedNorm FROM BlogsSplit ), EarliestNote AS ( SELECT n.NoteBlogId, n.RootBlogId, n.PostID, n.TimeStamp, ROW_NUMBER() OVER ( PARTITION BY n.NoteBlogId ORDER BY n.TimeStamp ASC, n.PostID ASC ) AS rn FROM Notes n JOIN NoteTypes t ON t.TypeId = n.TypeId JOIN Blogs root_bn ON root_bn.BlogId = n.RootBlogId LEFT JOIN Posts p ON p.BlogName = root_bn.BlogName AND p.PostID = n.PostID WHERE t.Type IN ('reblog')--, 'reply') AND n.NoteBlogId IN (SELECT BlogId FROM BlogsNorm) -- scope the window to blogs we care about AND COALESCE(p.IsActive, 1) = 1 -- exclude notes on a post explicitly marked removed ) SELECT bn.BlogId, bn.BlogName || '.tumblr.com', '''' || bn.blogname || ''',', bn.DateAdded, bn.DateAddedNorm, datetime(en.TimeStamp, 'unixepoch') AS earliest_reblog_or_reply_utc, en.PostID AS earliest_reblog_or_reply_postid, en.RootBlogId AS earliest_reblog_or_reply_root_blogid FROM BlogsNorm bn JOIN EarliestNote en ON en.NoteBlogId = bn.BlogId AND en.rn = 1 ORDER BY earliest_reblog_or_reply_utc, bn.DateAddedNorm desc limit 50; -- ---------------------------------------------------------------------------- -- QUERY 1 -- same answer, with a per-blog breakdown of which type(s) fired -- and how many. Useful once QUERY 0 has rows; redundant while it's empty. -- ---------------------------------------------------------------------------- -- WITH BlogsSplit AS ( ... ), BlogsNorm AS ( ... ) -- reuse the CTEs above -- -- SELECT -- bn.BlogId, -- bn.BlogName, -- bn.DateAddedNorm, -- SUM(CASE WHEN t.Type = 'reblog' THEN 1 ELSE 0 END) AS reblog_count, -- SUM(CASE WHEN t.Type = 'reply' THEN 1 ELSE 0 END) AS reply_count -- FROM BlogsNorm bn -- JOIN Notes n ON n.NoteBlogId = bn.BlogId -- JOIN NoteTypes t ON t.TypeId = n.TypeId -- WHERE t.Type IN ('reblog', 'reply') -- GROUP BY bn.BlogId, bn.BlogName, bn.DateAddedNorm -- ORDER BY bn.DateAddedNorm; -- ---------------------------------------------------------------------------- -- QUERY 2 -- put a date floor back, if wanted later. -- Same as QUERY 0, with one extra line in the outer WHERE: -- AND bn.DateAddedNorm >= '2026-09-01' -- or whatever cutoff -- ----------------------------------------------------------------------------