select *
from Blogs
--update blogs set HasBeenOutput = 1
where HasBeenOutput = 0
AND
blogname in
('udontn33dh1m',
'tyrantsxblood',
'sentry-34',
'deathcabforfrankie',
'abheith-sasta',
'kuwaiikittenghost',
'kansasmud',
'03diesel',
'itzameallieee',
'fireball-temptations',
'mamaisamess',
'906raised-and-dogobsessed',
'the-queerist-wolf',
'counting-corpsess',
'aqueenbby',
'maybememoriesx',
'queenofnevers',
'obsidian-psyche',
'lilmissellexo',
'alittlebunny95',
'rage--and--grace',
'savage-deniz',
'daddyspuddleprincess',
'littledefenstration',
'bearded-snorlax',
'thosesummerskiess',
'tubadtoph',
'lieutenant-dan-ice-cream',
'brittvnybitch',
'a-smol-gayologist',
'sum1random',
'samsternelly',
'littlemouseylauren',
'princessleiaorgasma',
'bloodstaineddkisses',
'letsfacerealitybabe',
'x--marks--thespot',
'space-and-suffering',
'rinarootski',
'thiccandtired',
'fvcking-scvmbag',
'fullblownwizard',
'bigjewface',
'unleash-the-krayken',
'bumpintheroad',
'liltexasjedii',
'nawtydude',
'queenpeachqueen',
'the-clansman',
'balmain-bxtch'
)select P.slug, N.replyText, rbn.BlogName as RootBlogName, n.PostID, nbn.BlogName || '.tumblr.com' as NoteBlogName, DatetimeCrawled, TimeStamp, nt.Type, rbn.BlogName || '.tumblr.com/post/' || n.postid, datetime(timestamp, 'unixepoch')
from Notes N
inner join Posts P on p.PostID = n.PostID
inner join Blogs rbn on rbn.BlogId = n.RootBlogId
inner join Blogs nbn on nbn.BlogId = n.NoteBlogId
inner join NoteTypes nt on nt.TypeId = n.TypeId
where
DatetimeCrawled > '2026-08-07 11:47:22' and nt.Type like 'r%'
and P.IsActive = 1
order by n.DatetimeCrawledWITH ReplyCounts AS (
SELECT
NoteBlogId,
COUNT(DISTINCT replyText) AS DistinctReplyCount
FROM Notes
where replyText <> '.'
GROUP BY NoteBlogId
)
SELECT
rbn.BlogName || '.tumblr.com/post/' || n.PostID AS PostURL, postid,
nbn.BlogName AS NoteBlogName,
n.replyText,
c.DistinctReplyCount
FROM Notes n
JOIN ReplyCounts c ON n.NoteBlogId = c.NoteBlogId
JOIN Blogs rbn ON rbn.BlogId = n.RootBlogId
JOIN Blogs nbn ON nbn.BlogId = n.NoteBlogId
JOIN NoteTypes t ON t.TypeId = n.TypeId
where replyText <> '.' and t.Type <> 'reply'
--AND nbn.BlogName NOT IN ( 'roadblocker21', 'thesaddemon666', 'edwardabbeyhoffman', 'tattedsoldier20', 'zomb-eh', 'animalistic13', 'indken', 'maccloud1592',
--'moss-wizard', 'supertrucker12682', 'exploringthrupics', 'padeyepete' )
order by c.DistinctReplyCount desc, nbn.BlogName, n.DateModified desc, replyText, rbn.BlogName, PostIDWITH PostsWithCount AS ( SELECT P.BlogName, P.PostID, 1925013599 AS LatestNoteTimestamp, P.NotesGatheredDateTime, COUNT(P.PostID) OVER(PARTITION BY P.BlogName) AS CNT, P.HasNotesGathered, P.NotFound, P.PostDate FROM Posts P WHERE COALESCE(P.IsActive, 1) = 1 ), Unioned AS ( SELECT BlogName, PostID, LatestNoteTimestamp, NotesGatheredDateTime, CNT, PostDate FROM PostsWithCount WHERE NotFound = 0 AND HasNotesGathered = 0 UNION SELECT BlogName, PostID, LatestNoteTimestamp, NotesGatheredDateTime, CNT, PostDate FROM PostsWithCount WHERE BlogName = 'zomb-eh' AND NotFound = 0 AND NotesGatheredDateTime < unixepoch('now', 'localtime', '-3 days') ) SELECT U.BlogName, U.PostID, U.LatestNoteTimestamp, U.NotesGatheredDateTime, U.CNT FROM Unioned U WHERE (U.NotesGatheredDateTime < 1787237598 OR U.NotesGatheredDateTime IS NULL) ORDER BY U.NotesGatheredDateTime, U.PostDate DESC, U.BlogName, U.PostID;delete from posts where postid in
(
'741662499571728384',
178892849664,
178264721139,
177012868749,
169950081964,
755440787056099328
)select *
-- delete
from notes
where postid not in (select distinct postid from posts where IsActive = 1)-- ============================================================================
-- blogs-added-after-august-2026-with-reblog-or-reply.sql
--
-- Purpose: Of all blogs in Blogs, find the ones that show up in Notes as the
-- engager (NoteBlogId) on a 'reblog' or 'reply' note, sorted by
-- DateAdded. (Originally scoped to "added after August 2026" --
-- that cutoff is now removed per request; QUERY 2 shows how to put
-- a date floor back if needed.)
--
-- Read-only. No INSERT/UPDATE/DELETE/DDL anywhere in this file.
--
-- How to use (DB Browser for SQLite):
-- 1. File > Open Database -> TL.db
-- 2. Execute SQL tab. Each QUERY below is independent; Ctrl+Enter runs just
-- the one your cursor is in.
--
-- The join, once:
-- "Added after August 2026" filters Blogs.DateAdded. "Has a reblog/reply
-- note" means the blog is the engager, which is NoteBlogId -- not
-- RootBlogId, which is the blog that *owns* the post being reacted to
-- (see find-notes-on-inactive-posts.sql for that side). Per TL.db.md, the
-- Blogs<->Notes join is a single integer hop and should not be routed
-- through Blogs:
-- Blogs.BlogId = Notes.NoteBlogId
-- EXISTS is used rather than a JOIN so a blog with many qualifying notes
-- still contributes one output row.
--
-- Excluding notes on an inactive post: same shape as
-- find-notes-on-inactive-posts.sql -- Notes only carries RootBlogId (an
-- integer), so reaching Posts.IsActive needs the one text hop the rest of
-- this file avoids: RootBlogId -> Blogs.BlogName = Posts.BlogName,
-- matched on PostID. It's a LEFT JOIN, not an inner one: only 3,867 of
-- 20,430 root/engager blogs have any stored Posts rows at all (TL.db.md),
-- so most reblog/reply notes have no Posts row to check and must be kept,
-- not dropped by an inner join. COALESCE(p.IsActive, 1) = 1 keeps a note
-- unless its post is explicitly IsActive = 0 -- NULL (no Posts row, or a
-- stored row with no flag written) means live, per the schema's own
-- convention (TL.db.md, "Posts.IsActive and Notes.IsActive"). This is a
-- big filter in practice: of the blogs that qualified before it, most
-- have every one of their reblog/reply notes pointing at a since-removed
-- post, not just some -- verified against the live data, not assumed.
--
-- On DateAdded: this column is not written consistently -- most rows hold
-- ISO 'yyyy-MM-dd HH:mm:ss', but 17k+ hold US 'M/d/yy' from a 2025 bulk
-- import (see TL.db.md, "DateAdded is not written consistently"). As text,
-- those two shapes do not sort or compare against each other correctly, so
-- QUERY 0 normalises both to an ISO date before filtering. In the live data
-- every US-format row predates August 2026 anyway (only '12/23/25' and
-- '12/24/25' occur), so this makes no difference to the current answer --
-- it's here so the query stays correct if that ever changes.
-- ============================================================================
-- ----------------------------------------------------------------------------
-- QUERY 0 / THE ANSWER -- one row per qualifying blog, sorted by DateAdded
-- descending (normalised -- see the note above). No date cutoff, but now
-- scoped to HasBeenOutput = 0 AND IsActive = 1. 4,739 rows in the live
-- data.
--
-- earliest_reblog_or_reply_utc is the MIN(TimeStamp) among this blog's
-- reblog-or-reply notes (either type counts -- see the column name).
-- Getting this meant switching QUERY 0 from EXISTS to an inner JOIN +
-- GROUP BY: EXISTS can only tell you a qualifying row is present, not
-- aggregate over which ones. No CASE is needed inside the MIN() because
-- the WHERE below already restricts the joined rows to reblog/reply, so
-- every row a blog brings into the aggregate is one this column should
-- consider. A blog appears exactly once, same as before, and this column
-- is never NULL for a row that's in the result at all (an earlier
-- revision aggregated reblog only, which left it NULL for the 181 blogs
-- that had replies but no reblogs).
-- ----------------------------------------------------------------------------
WITH BlogsSplit AS (
SELECT
b.BlogId,
b.BlogName,
b.DateAdded,
CASE WHEN b.DateAdded LIKE '____-__-__%' THEN 1 ELSE 0 END AS IsIso,
-- for the US 'M/d/yy' shape only: everything after the first '/'
substr(b.DateAdded, instr(b.DateAdded, '/') + 1) AS RestAfterMonth
FROM Blogs b
WHERE b.BlogId IS NOT NULL and HasBeenOutput = 0 and IsActive = 1 -- a blog can only match Notes if it has one
),
BlogsNorm AS (
SELECT
BlogId,
BlogName,
DateAdded,
CASE
WHEN IsIso = 1 THEN date(DateAdded)
ELSE date(
'20' || substr(RestAfterMonth, instr(RestAfterMonth, '/') + 1) || '-' ||
substr('00' || substr(DateAdded, 1, instr(DateAdded, '/') - 1), -2) || '-' ||
substr('00' || substr(RestAfterMonth, 1, instr(RestAfterMonth, '/') - 1), -2)
)
END AS DateAddedNorm
FROM BlogsSplit
)
SELECT
bn.BlogId,
bn.BlogName,
bn.DateAdded,
bn.DateAddedNorm,
datetime(MIN(n.TimeStamp), 'unixepoch') AS earliest_reblog_or_reply_utc
FROM BlogsNorm bn
JOIN Notes n ON n.NoteBlogId = bn.BlogId
JOIN NoteTypes t ON t.TypeId = n.TypeId
JOIN Blogs root_bn ON root_bn.BlogId = n.RootBlogId
LEFT JOIN Posts p ON p.BlogName = root_bn.BlogName
AND p.PostID = n.PostID
WHERE t.Type IN ('reblog')--, 'reply')
AND COALESCE(p.IsActive, 1) = 1 -- exclude notes on a post explicitly marked removed
GROUP BY bn.BlogId, bn.BlogName, bn.DateAdded, bn.DateAddedNorm
ORDER BY bn.DateAddedNorm desc;
-- ----------------------------------------------------------------------------
-- QUERY 1 -- same answer, with a per-blog breakdown of which type(s) fired
-- and how many. Useful once QUERY 0 has rows; redundant while it's empty.
-- ----------------------------------------------------------------------------
-- WITH BlogsSplit AS ( ... ), BlogsNorm AS ( ... ) -- reuse the CTEs above
--
-- SELECT
-- bn.BlogId,
-- bn.BlogName,
-- bn.DateAddedNorm,
-- SUM(CASE WHEN t.Type = 'reblog' THEN 1 ELSE 0 END) AS reblog_count,
-- SUM(CASE WHEN t.Type = 'reply' THEN 1 ELSE 0 END) AS reply_count
-- FROM BlogsNorm bn
-- JOIN Notes n ON n.NoteBlogId = bn.BlogId
-- JOIN NoteTypes t ON t.TypeId = n.TypeId
-- WHERE t.Type IN ('reblog', 'reply')
-- GROUP BY bn.BlogId, bn.BlogName, bn.DateAddedNorm
-- ORDER BY bn.DateAddedNorm;
-- ----------------------------------------------------------------------------
-- QUERY 2 -- put a date floor back, if wanted later.
-- Same as QUERY 0, with one extra line in the outer WHERE:
-- AND bn.DateAddedNorm >= '2026-09-01' -- or whatever cutoff
-- ----------------------------------------------------------------------------
-- ============================================================================
-- blogs-added-after-august-2026-with-reblog-or-reply.sql
--
-- Purpose: Of all blogs in Blogs, find the ones that show up in Notes as the
-- engager (NoteBlogId) on a 'reblog' or 'reply' note, sorted by
-- DateAdded. (Originally scoped to "added after August 2026" --
-- that cutoff is now removed per request; QUERY 2 shows how to put
-- a date floor back if needed.)
--
-- Read-only. No INSERT/UPDATE/DELETE/DDL anywhere in this file.
--
-- How to use (DB Browser for SQLite):
-- 1. File > Open Database -> TL.db
-- 2. Execute SQL tab. Each QUERY below is independent; Ctrl+Enter runs just
-- the one your cursor is in.
--
-- The join, once:
-- "Added after August 2026" filters Blogs.DateAdded. "Has a reblog/reply
-- note" means the blog is the engager, which is NoteBlogId -- not
-- RootBlogId, which is the blog that *owns* the post being reacted to
-- (see find-notes-on-inactive-posts.sql for that side). Per TL.db.md, the
-- Blogs<->Notes join is a single integer hop and should not be routed
-- through Blogs:
-- Blogs.BlogId = Notes.NoteBlogId
-- EXISTS is used rather than a JOIN so a blog with many qualifying notes
-- still contributes one output row.
--
-- Excluding notes on an inactive post: same shape as
-- find-notes-on-inactive-posts.sql -- Notes only carries RootBlogId (an
-- integer), so reaching Posts.IsActive needs the one text hop the rest of
-- this file avoids: RootBlogId -> Blogs.BlogName = Posts.BlogName,
-- matched on PostID. It's a LEFT JOIN, not an inner one: only 3,867 of
-- 20,430 root/engager blogs have any stored Posts rows at all (TL.db.md),
-- so most reblog/reply notes have no Posts row to check and must be kept,
-- not dropped by an inner join. COALESCE(p.IsActive, 1) = 1 keeps a note
-- unless its post is explicitly IsActive = 0 -- NULL (no Posts row, or a
-- stored row with no flag written) means live, per the schema's own
-- convention (TL.db.md, "Posts.IsActive and Notes.IsActive"). This is a
-- big filter in practice: of the blogs that qualified before it, most
-- have every one of their reblog/reply notes pointing at a since-removed
-- post, not just some -- verified against the live data, not assumed.
--
-- On DateAdded: this column is not written consistently -- most rows hold
-- ISO 'yyyy-MM-dd HH:mm:ss', but 17k+ hold US 'M/d/yy' from a 2025 bulk
-- import (see TL.db.md, "DateAdded is not written consistently"). As text,
-- those two shapes do not sort or compare against each other correctly, so
-- QUERY 0 normalises both to an ISO date before filtering. In the live data
-- every US-format row predates August 2026 anyway (only '12/23/25' and
-- '12/24/25' occur), so this makes no difference to the current answer --
-- it's here so the query stays correct if that ever changes.
-- ============================================================================
-- ----------------------------------------------------------------------------
-- QUERY 0 / THE ANSWER -- one row per qualifying blog, sorted by
-- earliest_reblog_or_reply_utc then DateAdded descending (normalised --
-- see the note above). No date cutoff, but scoped to HasBeenOutput = 0
-- AND IsActive = 1, and now excluding notes on a removed post (see the
-- header note above). 1,804 rows in the live data as of this revision --
-- down from 4,396 just before this exclusion was added, because most of
-- the blogs that dropped out had *every* reblog/reply note pointing at a
-- now-inactive post, not just some (the number moves between runs
-- regardless -- crawling and output flip HasBeenOutput/IsActive on live
-- rows).
--
-- earliest_reblog_or_reply_utc is the earliest TimeStamp among this
-- blog's reblog-or-reply notes (either type counts -- see the column
-- name); earliest_reblog_or_reply_postid and _root_blogid identify that
-- specific note's post: PostID + RootBlogId together, not PostID alone --
-- see TL.db.md ("345 post IDs exist under more than one blog"), same
-- caution as in find-notes-on-inactive-posts.sql. Resolve RootBlogId to a
-- name via Blogs (or Blogs, tolerating a miss) if you need it.
--
-- Getting "which note" rather than just "when" doesn't fit a plain
-- MIN()/GROUP BY -- an aggregate can tell you the earliest value but not
-- which row it came from. EarliestNote instead ranks each blog's
-- reblog/reply notes with ROW_NUMBER() OVER (PARTITION BY NoteBlogId
-- ORDER BY TimeStamp), and QUERY 0 takes rn = 1. The ORDER BY carries a
-- PostID tiebreak because (NoteBlogId, TimeStamp) is not unique in this
-- data -- ties exist (e.g. NoteBlogId 12 has 7 notes at the same
-- TimeStamp) -- so without a tiebreak the "earliest" pick would be
-- arbitrary among ties rather than deterministic.
--
-- EarliestNote also excludes notes on an inactive post before ranking
-- (see the header note above), so "earliest" means earliest surviving
-- note, not earliest overall -- a blog whose true-earliest note pointed
-- at a since-removed post now surfaces its next-earliest live one
-- instead. Applying the exclusion here, not as a filter on QUERY 0's
-- final rows, matters: filtering after ROW_NUMBER would have picked the
-- removed-post note as rn = 1 and then dropped the whole row instead of
-- promoting the next candidate.
-- ----------------------------------------------------------------------------
WITH BlogsSplit AS (
SELECT
b.BlogId,
b.BlogName,
b.DateAdded,
CASE WHEN b.DateAdded LIKE '____-__-__%' THEN 1 ELSE 0 END AS IsIso,
-- for the US 'M/d/yy' shape only: everything after the first '/'
substr(b.DateAdded, instr(b.DateAdded, '/') + 1) AS RestAfterMonth
FROM Blogs b
WHERE b.BlogId IS NOT NULL and HasBeenOutput = 0 and IsActive = 1 -- a blog can only match Notes if it has one
),
BlogsNorm AS (
SELECT
BlogId,
BlogName,
DateAdded,
CASE
WHEN IsIso = 1 THEN date(DateAdded)
ELSE date(
'20' || substr(RestAfterMonth, instr(RestAfterMonth, '/') + 1) || '-' ||
substr('00' || substr(DateAdded, 1, instr(DateAdded, '/') - 1), -2) || '-' ||
substr('00' || substr(RestAfterMonth, 1, instr(RestAfterMonth, '/') - 1), -2)
)
END AS DateAddedNorm
FROM BlogsSplit
),
EarliestNote AS (
SELECT
n.NoteBlogId,
n.RootBlogId,
n.PostID,
n.TimeStamp,
ROW_NUMBER() OVER (
PARTITION BY n.NoteBlogId
ORDER BY n.TimeStamp ASC, n.PostID ASC
) AS rn
FROM Notes n
JOIN NoteTypes t ON t.TypeId = n.TypeId
JOIN Blogs root_bn ON root_bn.BlogId = n.RootBlogId
LEFT JOIN Posts p ON p.BlogName = root_bn.BlogName
AND p.PostID = n.PostID
WHERE t.Type IN ('reblog')--, 'reply')
AND n.NoteBlogId IN (SELECT BlogId FROM BlogsNorm) -- scope the window to blogs we care about
AND COALESCE(p.IsActive, 1) = 1 -- exclude notes on a post explicitly marked removed
)
SELECT
bn.BlogId,
bn.BlogName || '.tumblr.com',
'''' || bn.blogname || ''',',
bn.DateAdded,
bn.DateAddedNorm,
datetime(en.TimeStamp, 'unixepoch') AS earliest_reblog_or_reply_utc,
en.PostID AS earliest_reblog_or_reply_postid,
en.RootBlogId AS earliest_reblog_or_reply_root_blogid
FROM BlogsNorm bn
JOIN EarliestNote en ON en.NoteBlogId = bn.BlogId AND en.rn = 1
ORDER BY earliest_reblog_or_reply_utc, bn.DateAddedNorm desc
limit 50;
-- ----------------------------------------------------------------------------
-- QUERY 1 -- same answer, with a per-blog breakdown of which type(s) fired
-- and how many. Useful once QUERY 0 has rows; redundant while it's empty.
-- ----------------------------------------------------------------------------
-- WITH BlogsSplit AS ( ... ), BlogsNorm AS ( ... ) -- reuse the CTEs above
--
-- SELECT
-- bn.BlogId,
-- bn.BlogName,
-- bn.DateAddedNorm,
-- SUM(CASE WHEN t.Type = 'reblog' THEN 1 ELSE 0 END) AS reblog_count,
-- SUM(CASE WHEN t.Type = 'reply' THEN 1 ELSE 0 END) AS reply_count
-- FROM BlogsNorm bn
-- JOIN Notes n ON n.NoteBlogId = bn.BlogId
-- JOIN NoteTypes t ON t.TypeId = n.TypeId
-- WHERE t.Type IN ('reblog', 'reply')
-- GROUP BY bn.BlogId, bn.BlogName, bn.DateAddedNorm
-- ORDER BY bn.DateAddedNorm;
-- ----------------------------------------------------------------------------
-- QUERY 2 -- put a date floor back, if wanted later.
-- Same as QUERY 0, with one extra line in the outer WHERE:
-- AND bn.DateAddedNorm >= '2026-09-01' -- or whatever cutoff
-- ----------------------------------------------------------------------------