Blogs.BlogId was a one-time copy of BlogNames and nothing kept it current: 12,238 blogs first seen after 2026-08-07 had a BlogNames ID but a NULL Blogs.BlogId, so GetBlogs' join on BlogId silently skipped them and their 23,148 notes. - retire-blognames.sql: stub Blogs rows for the 17 unregistered note participants, backfill IDs (none renumbered), make ix_Blogs_BlogId UNIQUE, drop BlogNames, and add triggers that stop a Blogs row with a BlogId from being deleted, renamed or renumbered - AddNote registers both blogs via RegisterBlog (Blogs row + MAX+1 ID) and every query resolves names through Blogs instead of BlogNames - verify-db-schema.sql reports a DB that still has BlogNames (1e) - Update TL.db.md, AGENTS.md and the DB Browser saved queries Co-Authored-By: Claude Opus 5.5 <[email protected]>
416 lines
24 KiB
XML
416 lines
24 KiB
XML
<?xml version="1.0" encoding="UTF-8"?><sqlb_project><db path="C:/Users/jim/Nextcloud/C#/URLNotesGrabberCORE/URLNotesGrabberCORE/TL.db" readonly="0" foreign_keys="1" case_sensitive_like="0" temp_store="0" wal_autocheckpoint="1000" synchronous="2"/><attached/><window><main_tabs open="structure browser pragmas query" current="3"/></window><tab_structure><column_width id="0" width="300"/><column_width id="1" width="0"/><column_width id="2" width="100"/><column_width id="3" width="4486"/><column_width id="4" width="0"/><expanded_item id="0" parent="1"/><expanded_item id="1" parent="1"/><expanded_item id="2" parent="1"/><expanded_item id="3" parent="1"/></tab_structure><tab_browse><table title="Notes" custom_title="0" dock_id="4" table="4,5:mainNotes"/><dock_state state="000000ff00000000fd0000000100000002000005470000029efc0100000006fb000000160064006f0063006b00420072006f00770073006500310100000000000004a10000000000000000fb000000160064006f0063006b00420072006f00770073006500320100000000000004a10000000000000000fb000000160064006f0063006b00420072006f00770073006500330100000000000004a10000000000000000fb000000160064006f0063006b00420072006f00770073006500350100000000000005f40000000000000000fb000000160064006f0063006b00420072006f00770073006500340100000000000005470000011100fffffffb000000160064006f0063006b00420072006f00770073006500340100000000000005f40000000000000000000005470000000000000004000000040000000800000008fc00000000"/><default_encoding codec=""/><browse_table_settings><table schema="main" name="ApiKeyPoolMeta" show_row_id="0" encoding="" plot_x_axis="" unlock_view_pk="_rowid_" freeze_columns="0"><sort/><column_widths><column index="1" value="29"/><column index="2" value="64"/></column_widths><filter_values/><conditional_formats/><row_id_formats/><display_formats/><hidden_columns/><plot_y_axes/><global_filter/></table><table schema="main" name="Blogs" show_row_id="0" encoding="" plot_x_axis="" unlock_view_pk="_rowid_" freeze_columns="0"><sort><column index="7" mode="1"/></sort><column_widths><column index="1" value="257"/><column index="2" value="108"/><column index="3" value="63"/><column index="4" value="156"/><column index="5" value="60"/><column index="6" value="81"/><column index="7" value="85"/><column index="8" value="156"/><column index="9" value="156"/><column index="10" value="151"/><column index="11" value="125"/><column index="12" value="129"/></column_widths><filter_values><column index="4" value="1"/><column index="7" value=">2026-05-27 17:22:36"/></filter_values><conditional_formats/><row_id_formats/><display_formats/><hidden_columns/><plot_y_axes/><global_filter/></table><table schema="main" name="Notes" show_row_id="0" encoding="" plot_x_axis="" unlock_view_pk="_rowid_" freeze_columns="0"><sort><column index="4" mode="1"/></sort><column_widths><column index="1" value="81"/><column index="2" value="148"/><column index="3" value="83"/><column index="4" value="85"/><column index="5" value="56"/><column index="6" value="300"/><column index="7" value="156"/><column index="8" value="156"/><column index="9" value="156"/><column index="10" value="63"/></column_widths><filter_values><column index="2" value="4370"/></filter_values><conditional_formats/><row_id_formats/><display_formats/><hidden_columns/><plot_y_axes/><global_filter/></table><table schema="main" name="Posts" show_row_id="0" encoding="" plot_x_axis="" unlock_view_pk="_rowid_" freeze_columns="0"><sort><column index="14" mode="1"/></sort><column_widths><column index="1" value="241"/><column index="2" value="148"/><column index="3" value="126"/><column index="4" value="300"/><column index="5" value="75"/><column index="6" value="187"/><column index="7" value="159"/><column index="8" value="75"/><column index="9" value="0"/><column index="10" value="0"/><column index="11" value="0"/><column index="12" value="249"/><column index="13" value="300"/><column index="14" value="53"/><column index="15" value="300"/><column index="16" value="300"/><column index="17" value="41"/><column index="18" value="75"/><column index="19" value="96"/><column index="20" value="300"/><column index="21" value="96"/><column index="22" value="300"/><column index="23" value="300"/><column index="24" value="300"/><column index="25" value="60"/><column index="26" value="218"/><column index="27" value="300"/><column index="28" value="156"/><column index="29" value="156"/><column index="30" value="69"/><column index="31" value="63"/></column_widths><filter_values><column index="1" value="137735301451"/></filter_values><conditional_formats/><row_id_formats/><display_formats/><hidden_columns><column index="9" value="1"/><column index="10" value="1"/><column index="11" value="1"/></hidden_columns><plot_y_axes/><global_filter/></table></browse_table_settings></tab_browse><tab_sql><sql name="Mark Blogs">select *
|
|
from Blogs
|
|
--update blogs set HasBeenOutput = 1
|
|
where HasBeenOutput = 0
|
|
AND
|
|
blogname in
|
|
('udontn33dh1m',
|
|
'tyrantsxblood',
|
|
'sentry-34',
|
|
'deathcabforfrankie',
|
|
'abheith-sasta',
|
|
'kuwaiikittenghost',
|
|
'kansasmud',
|
|
'03diesel',
|
|
'itzameallieee',
|
|
'fireball-temptations',
|
|
'mamaisamess',
|
|
'906raised-and-dogobsessed',
|
|
'the-queerist-wolf',
|
|
'counting-corpsess',
|
|
'aqueenbby',
|
|
'maybememoriesx',
|
|
'queenofnevers',
|
|
'obsidian-psyche',
|
|
'lilmissellexo',
|
|
'alittlebunny95',
|
|
'rage--and--grace',
|
|
'savage-deniz',
|
|
'daddyspuddleprincess',
|
|
'littledefenstration',
|
|
'bearded-snorlax',
|
|
'thosesummerskiess',
|
|
'tubadtoph',
|
|
'lieutenant-dan-ice-cream',
|
|
'brittvnybitch',
|
|
'a-smol-gayologist',
|
|
'sum1random',
|
|
'samsternelly',
|
|
'littlemouseylauren',
|
|
'princessleiaorgasma',
|
|
'bloodstaineddkisses',
|
|
'letsfacerealitybabe',
|
|
'x--marks--thespot',
|
|
'space-and-suffering',
|
|
'rinarootski',
|
|
'thiccandtired',
|
|
'fvcking-scvmbag',
|
|
'fullblownwizard',
|
|
'bigjewface',
|
|
'unleash-the-krayken',
|
|
'bumpintheroad',
|
|
'liltexasjedii',
|
|
'nawtydude',
|
|
'queenpeachqueen',
|
|
'the-clansman',
|
|
'balmain-bxtch'
|
|
)</sql><sql name="New Notes">select P.slug, N.replyText, rbn.BlogName as RootBlogName, n.PostID, nbn.BlogName || '.tumblr.com' as NoteBlogName, DatetimeCrawled, TimeStamp, nt.Type, rbn.BlogName || '.tumblr.com/post/' || n.postid, datetime(timestamp, 'unixepoch')
|
|
from Notes N
|
|
inner join Posts P on p.PostID = n.PostID
|
|
inner join Blogs rbn on rbn.BlogId = n.RootBlogId
|
|
inner join Blogs nbn on nbn.BlogId = n.NoteBlogId
|
|
inner join NoteTypes nt on nt.TypeId = n.TypeId
|
|
where
|
|
DatetimeCrawled > '2026-08-07 11:47:22' and nt.Type like 'r%'
|
|
and P.IsActive = 1
|
|
order by n.DatetimeCrawled</sql><sql name="SQL 7">WITH ReplyCounts AS (
|
|
SELECT
|
|
NoteBlogId,
|
|
COUNT(DISTINCT replyText) AS DistinctReplyCount
|
|
FROM Notes
|
|
where replyText <> '.'
|
|
GROUP BY NoteBlogId
|
|
)
|
|
SELECT
|
|
rbn.BlogName || '.tumblr.com/post/' || n.PostID AS PostURL, postid,
|
|
nbn.BlogName AS NoteBlogName,
|
|
n.replyText,
|
|
c.DistinctReplyCount
|
|
FROM Notes n
|
|
JOIN ReplyCounts c ON n.NoteBlogId = c.NoteBlogId
|
|
JOIN Blogs rbn ON rbn.BlogId = n.RootBlogId
|
|
JOIN Blogs nbn ON nbn.BlogId = n.NoteBlogId
|
|
JOIN NoteTypes t ON t.TypeId = n.TypeId
|
|
where replyText <> '.' and t.Type <> 'reply'
|
|
--AND nbn.BlogName NOT IN ( 'roadblocker21', 'thesaddemon666', 'edwardabbeyhoffman', 'tattedsoldier20', 'zomb-eh', 'animalistic13', 'indken', 'maccloud1592',
|
|
--'moss-wizard', 'supertrucker12682', 'exploringthrupics', 'padeyepete' )
|
|
order by c.DistinctReplyCount desc, nbn.BlogName, n.DateModified desc, replyText, rbn.BlogName, PostID</sql><sql name="Collect">WITH PostsWithCount AS ( SELECT P.BlogName, P.PostID, 1925013599 AS LatestNoteTimestamp, P.NotesGatheredDateTime, COUNT(P.PostID) OVER(PARTITION BY P.BlogName) AS CNT, P.HasNotesGathered, P.NotFound, P.PostDate FROM Posts P WHERE COALESCE(P.IsActive, 1) = 1 ), Unioned AS ( SELECT BlogName, PostID, LatestNoteTimestamp, NotesGatheredDateTime, CNT, PostDate FROM PostsWithCount WHERE NotFound = 0 AND HasNotesGathered = 0 UNION SELECT BlogName, PostID, LatestNoteTimestamp, NotesGatheredDateTime, CNT, PostDate FROM PostsWithCount WHERE BlogName = 'zomb-eh' AND NotFound = 0 AND NotesGatheredDateTime < unixepoch('now', 'localtime', '-3 days') ) SELECT U.BlogName, U.PostID, U.LatestNoteTimestamp, U.NotesGatheredDateTime, U.CNT FROM Unioned U WHERE (U.NotesGatheredDateTime < 1787237598 OR U.NotesGatheredDateTime IS NULL) ORDER BY U.NotesGatheredDateTime, U.PostDate DESC, U.BlogName, U.PostID;</sql><sql name="Del Posts">delete from posts where postid in
|
|
(
|
|
'741662499571728384',
|
|
178892849664,
|
|
178264721139,
|
|
177012868749,
|
|
169950081964,
|
|
755440787056099328
|
|
)</sql><sql name="notes NO post">select *
|
|
-- delete
|
|
from notes
|
|
where postid not in (select distinct postid from posts where IsActive = 1)</sql><sql name="Pull Blogs">-- ============================================================================
|
|
-- blogs-added-after-august-2026-with-reblog-or-reply.sql
|
|
--
|
|
-- Purpose: Of all blogs in Blogs, find the ones that show up in Notes as the
|
|
-- engager (NoteBlogId) on a 'reblog' or 'reply' note, sorted by
|
|
-- DateAdded. (Originally scoped to "added after August 2026" --
|
|
-- that cutoff is now removed per request; QUERY 2 shows how to put
|
|
-- a date floor back if needed.)
|
|
--
|
|
-- Read-only. No INSERT/UPDATE/DELETE/DDL anywhere in this file.
|
|
--
|
|
-- How to use (DB Browser for SQLite):
|
|
-- 1. File > Open Database -> TL.db
|
|
-- 2. Execute SQL tab. Each QUERY below is independent; Ctrl+Enter runs just
|
|
-- the one your cursor is in.
|
|
--
|
|
-- The join, once:
|
|
-- "Added after August 2026" filters Blogs.DateAdded. "Has a reblog/reply
|
|
-- note" means the blog is the engager, which is NoteBlogId -- not
|
|
-- RootBlogId, which is the blog that *owns* the post being reacted to
|
|
-- (see find-notes-on-inactive-posts.sql for that side). Per TL.db.md, the
|
|
-- Blogs<->Notes join is a single integer hop and should not be routed
|
|
-- through Blogs:
|
|
-- Blogs.BlogId = Notes.NoteBlogId
|
|
-- EXISTS is used rather than a JOIN so a blog with many qualifying notes
|
|
-- still contributes one output row.
|
|
--
|
|
-- Excluding notes on an inactive post: same shape as
|
|
-- find-notes-on-inactive-posts.sql -- Notes only carries RootBlogId (an
|
|
-- integer), so reaching Posts.IsActive needs the one text hop the rest of
|
|
-- this file avoids: RootBlogId -> Blogs.BlogName = Posts.BlogName,
|
|
-- matched on PostID. It's a LEFT JOIN, not an inner one: only 3,867 of
|
|
-- 20,430 root/engager blogs have any stored Posts rows at all (TL.db.md),
|
|
-- so most reblog/reply notes have no Posts row to check and must be kept,
|
|
-- not dropped by an inner join. COALESCE(p.IsActive, 1) = 1 keeps a note
|
|
-- unless its post is explicitly IsActive = 0 -- NULL (no Posts row, or a
|
|
-- stored row with no flag written) means live, per the schema's own
|
|
-- convention (TL.db.md, "Posts.IsActive and Notes.IsActive"). This is a
|
|
-- big filter in practice: of the blogs that qualified before it, most
|
|
-- have every one of their reblog/reply notes pointing at a since-removed
|
|
-- post, not just some -- verified against the live data, not assumed.
|
|
--
|
|
-- On DateAdded: this column is not written consistently -- most rows hold
|
|
-- ISO 'yyyy-MM-dd HH:mm:ss', but 17k+ hold US 'M/d/yy' from a 2025 bulk
|
|
-- import (see TL.db.md, "DateAdded is not written consistently"). As text,
|
|
-- those two shapes do not sort or compare against each other correctly, so
|
|
-- QUERY 0 normalises both to an ISO date before filtering. In the live data
|
|
-- every US-format row predates August 2026 anyway (only '12/23/25' and
|
|
-- '12/24/25' occur), so this makes no difference to the current answer --
|
|
-- it's here so the query stays correct if that ever changes.
|
|
-- ============================================================================
|
|
|
|
|
|
-- ----------------------------------------------------------------------------
|
|
-- QUERY 0 / THE ANSWER -- one row per qualifying blog, sorted by DateAdded
|
|
-- descending (normalised -- see the note above). No date cutoff, but now
|
|
-- scoped to HasBeenOutput = 0 AND IsActive = 1. 4,739 rows in the live
|
|
-- data.
|
|
--
|
|
-- earliest_reblog_or_reply_utc is the MIN(TimeStamp) among this blog's
|
|
-- reblog-or-reply notes (either type counts -- see the column name).
|
|
-- Getting this meant switching QUERY 0 from EXISTS to an inner JOIN +
|
|
-- GROUP BY: EXISTS can only tell you a qualifying row is present, not
|
|
-- aggregate over which ones. No CASE is needed inside the MIN() because
|
|
-- the WHERE below already restricts the joined rows to reblog/reply, so
|
|
-- every row a blog brings into the aggregate is one this column should
|
|
-- consider. A blog appears exactly once, same as before, and this column
|
|
-- is never NULL for a row that's in the result at all (an earlier
|
|
-- revision aggregated reblog only, which left it NULL for the 181 blogs
|
|
-- that had replies but no reblogs).
|
|
-- ----------------------------------------------------------------------------
|
|
WITH BlogsSplit AS (
|
|
SELECT
|
|
b.BlogId,
|
|
b.BlogName,
|
|
b.DateAdded,
|
|
CASE WHEN b.DateAdded LIKE '____-__-__%' THEN 1 ELSE 0 END AS IsIso,
|
|
-- for the US 'M/d/yy' shape only: everything after the first '/'
|
|
substr(b.DateAdded, instr(b.DateAdded, '/') + 1) AS RestAfterMonth
|
|
FROM Blogs b
|
|
WHERE b.BlogId IS NOT NULL and HasBeenOutput = 0 and IsActive = 1 -- a blog can only match Notes if it has one
|
|
),
|
|
BlogsNorm AS (
|
|
SELECT
|
|
BlogId,
|
|
BlogName,
|
|
DateAdded,
|
|
CASE
|
|
WHEN IsIso = 1 THEN date(DateAdded)
|
|
ELSE date(
|
|
'20' || substr(RestAfterMonth, instr(RestAfterMonth, '/') + 1) || '-' ||
|
|
substr('00' || substr(DateAdded, 1, instr(DateAdded, '/') - 1), -2) || '-' ||
|
|
substr('00' || substr(RestAfterMonth, 1, instr(RestAfterMonth, '/') - 1), -2)
|
|
)
|
|
END AS DateAddedNorm
|
|
FROM BlogsSplit
|
|
)
|
|
SELECT
|
|
bn.BlogId,
|
|
bn.BlogName,
|
|
bn.DateAdded,
|
|
bn.DateAddedNorm,
|
|
datetime(MIN(n.TimeStamp), 'unixepoch') AS earliest_reblog_or_reply_utc
|
|
FROM BlogsNorm bn
|
|
JOIN Notes n ON n.NoteBlogId = bn.BlogId
|
|
JOIN NoteTypes t ON t.TypeId = n.TypeId
|
|
JOIN Blogs root_bn ON root_bn.BlogId = n.RootBlogId
|
|
LEFT JOIN Posts p ON p.BlogName = root_bn.BlogName
|
|
AND p.PostID = n.PostID
|
|
WHERE t.Type IN ('reblog')--, 'reply')
|
|
AND COALESCE(p.IsActive, 1) = 1 -- exclude notes on a post explicitly marked removed
|
|
GROUP BY bn.BlogId, bn.BlogName, bn.DateAdded, bn.DateAddedNorm
|
|
ORDER BY bn.DateAddedNorm desc;
|
|
|
|
|
|
-- ----------------------------------------------------------------------------
|
|
-- QUERY 1 -- same answer, with a per-blog breakdown of which type(s) fired
|
|
-- and how many. Useful once QUERY 0 has rows; redundant while it's empty.
|
|
-- ----------------------------------------------------------------------------
|
|
-- WITH BlogsSplit AS ( ... ), BlogsNorm AS ( ... ) -- reuse the CTEs above
|
|
--
|
|
-- SELECT
|
|
-- bn.BlogId,
|
|
-- bn.BlogName,
|
|
-- bn.DateAddedNorm,
|
|
-- SUM(CASE WHEN t.Type = 'reblog' THEN 1 ELSE 0 END) AS reblog_count,
|
|
-- SUM(CASE WHEN t.Type = 'reply' THEN 1 ELSE 0 END) AS reply_count
|
|
-- FROM BlogsNorm bn
|
|
-- JOIN Notes n ON n.NoteBlogId = bn.BlogId
|
|
-- JOIN NoteTypes t ON t.TypeId = n.TypeId
|
|
-- WHERE t.Type IN ('reblog', 'reply')
|
|
-- GROUP BY bn.BlogId, bn.BlogName, bn.DateAddedNorm
|
|
-- ORDER BY bn.DateAddedNorm;
|
|
|
|
|
|
-- ----------------------------------------------------------------------------
|
|
-- QUERY 2 -- put a date floor back, if wanted later.
|
|
-- Same as QUERY 0, with one extra line in the outer WHERE:
|
|
-- AND bn.DateAddedNorm >= '2026-09-01' -- or whatever cutoff
|
|
-- ----------------------------------------------------------------------------
|
|
-- ============================================================================
|
|
-- blogs-added-after-august-2026-with-reblog-or-reply.sql
|
|
--
|
|
-- Purpose: Of all blogs in Blogs, find the ones that show up in Notes as the
|
|
-- engager (NoteBlogId) on a 'reblog' or 'reply' note, sorted by
|
|
-- DateAdded. (Originally scoped to "added after August 2026" --
|
|
-- that cutoff is now removed per request; QUERY 2 shows how to put
|
|
-- a date floor back if needed.)
|
|
--
|
|
-- Read-only. No INSERT/UPDATE/DELETE/DDL anywhere in this file.
|
|
--
|
|
-- How to use (DB Browser for SQLite):
|
|
-- 1. File > Open Database -> TL.db
|
|
-- 2. Execute SQL tab. Each QUERY below is independent; Ctrl+Enter runs just
|
|
-- the one your cursor is in.
|
|
--
|
|
-- The join, once:
|
|
-- "Added after August 2026" filters Blogs.DateAdded. "Has a reblog/reply
|
|
-- note" means the blog is the engager, which is NoteBlogId -- not
|
|
-- RootBlogId, which is the blog that *owns* the post being reacted to
|
|
-- (see find-notes-on-inactive-posts.sql for that side). Per TL.db.md, the
|
|
-- Blogs<->Notes join is a single integer hop and should not be routed
|
|
-- through Blogs:
|
|
-- Blogs.BlogId = Notes.NoteBlogId
|
|
-- EXISTS is used rather than a JOIN so a blog with many qualifying notes
|
|
-- still contributes one output row.
|
|
--
|
|
-- Excluding notes on an inactive post: same shape as
|
|
-- find-notes-on-inactive-posts.sql -- Notes only carries RootBlogId (an
|
|
-- integer), so reaching Posts.IsActive needs the one text hop the rest of
|
|
-- this file avoids: RootBlogId -> Blogs.BlogName = Posts.BlogName,
|
|
-- matched on PostID. It's a LEFT JOIN, not an inner one: only 3,867 of
|
|
-- 20,430 root/engager blogs have any stored Posts rows at all (TL.db.md),
|
|
-- so most reblog/reply notes have no Posts row to check and must be kept,
|
|
-- not dropped by an inner join. COALESCE(p.IsActive, 1) = 1 keeps a note
|
|
-- unless its post is explicitly IsActive = 0 -- NULL (no Posts row, or a
|
|
-- stored row with no flag written) means live, per the schema's own
|
|
-- convention (TL.db.md, "Posts.IsActive and Notes.IsActive"). This is a
|
|
-- big filter in practice: of the blogs that qualified before it, most
|
|
-- have every one of their reblog/reply notes pointing at a since-removed
|
|
-- post, not just some -- verified against the live data, not assumed.
|
|
--
|
|
-- On DateAdded: this column is not written consistently -- most rows hold
|
|
-- ISO 'yyyy-MM-dd HH:mm:ss', but 17k+ hold US 'M/d/yy' from a 2025 bulk
|
|
-- import (see TL.db.md, "DateAdded is not written consistently"). As text,
|
|
-- those two shapes do not sort or compare against each other correctly, so
|
|
-- QUERY 0 normalises both to an ISO date before filtering. In the live data
|
|
-- every US-format row predates August 2026 anyway (only '12/23/25' and
|
|
-- '12/24/25' occur), so this makes no difference to the current answer --
|
|
-- it's here so the query stays correct if that ever changes.
|
|
-- ============================================================================
|
|
|
|
|
|
-- ----------------------------------------------------------------------------
|
|
-- QUERY 0 / THE ANSWER -- one row per qualifying blog, sorted by
|
|
-- earliest_reblog_or_reply_utc then DateAdded descending (normalised --
|
|
-- see the note above). No date cutoff, but scoped to HasBeenOutput = 0
|
|
-- AND IsActive = 1, and now excluding notes on a removed post (see the
|
|
-- header note above). 1,804 rows in the live data as of this revision --
|
|
-- down from 4,396 just before this exclusion was added, because most of
|
|
-- the blogs that dropped out had *every* reblog/reply note pointing at a
|
|
-- now-inactive post, not just some (the number moves between runs
|
|
-- regardless -- crawling and output flip HasBeenOutput/IsActive on live
|
|
-- rows).
|
|
--
|
|
-- earliest_reblog_or_reply_utc is the earliest TimeStamp among this
|
|
-- blog's reblog-or-reply notes (either type counts -- see the column
|
|
-- name); earliest_reblog_or_reply_postid and _root_blogid identify that
|
|
-- specific note's post: PostID + RootBlogId together, not PostID alone --
|
|
-- see TL.db.md ("345 post IDs exist under more than one blog"), same
|
|
-- caution as in find-notes-on-inactive-posts.sql. Resolve RootBlogId to a
|
|
-- name via Blogs (or Blogs, tolerating a miss) if you need it.
|
|
--
|
|
-- Getting "which note" rather than just "when" doesn't fit a plain
|
|
-- MIN()/GROUP BY -- an aggregate can tell you the earliest value but not
|
|
-- which row it came from. EarliestNote instead ranks each blog's
|
|
-- reblog/reply notes with ROW_NUMBER() OVER (PARTITION BY NoteBlogId
|
|
-- ORDER BY TimeStamp), and QUERY 0 takes rn = 1. The ORDER BY carries a
|
|
-- PostID tiebreak because (NoteBlogId, TimeStamp) is not unique in this
|
|
-- data -- ties exist (e.g. NoteBlogId 12 has 7 notes at the same
|
|
-- TimeStamp) -- so without a tiebreak the "earliest" pick would be
|
|
-- arbitrary among ties rather than deterministic.
|
|
--
|
|
-- EarliestNote also excludes notes on an inactive post before ranking
|
|
-- (see the header note above), so "earliest" means earliest surviving
|
|
-- note, not earliest overall -- a blog whose true-earliest note pointed
|
|
-- at a since-removed post now surfaces its next-earliest live one
|
|
-- instead. Applying the exclusion here, not as a filter on QUERY 0's
|
|
-- final rows, matters: filtering after ROW_NUMBER would have picked the
|
|
-- removed-post note as rn = 1 and then dropped the whole row instead of
|
|
-- promoting the next candidate.
|
|
-- ----------------------------------------------------------------------------
|
|
WITH BlogsSplit AS (
|
|
SELECT
|
|
b.BlogId,
|
|
b.BlogName,
|
|
b.DateAdded,
|
|
CASE WHEN b.DateAdded LIKE '____-__-__%' THEN 1 ELSE 0 END AS IsIso,
|
|
-- for the US 'M/d/yy' shape only: everything after the first '/'
|
|
substr(b.DateAdded, instr(b.DateAdded, '/') + 1) AS RestAfterMonth
|
|
FROM Blogs b
|
|
WHERE b.BlogId IS NOT NULL and HasBeenOutput = 0 and IsActive = 1 -- a blog can only match Notes if it has one
|
|
),
|
|
BlogsNorm AS (
|
|
SELECT
|
|
BlogId,
|
|
BlogName,
|
|
DateAdded,
|
|
CASE
|
|
WHEN IsIso = 1 THEN date(DateAdded)
|
|
ELSE date(
|
|
'20' || substr(RestAfterMonth, instr(RestAfterMonth, '/') + 1) || '-' ||
|
|
substr('00' || substr(DateAdded, 1, instr(DateAdded, '/') - 1), -2) || '-' ||
|
|
substr('00' || substr(RestAfterMonth, 1, instr(RestAfterMonth, '/') - 1), -2)
|
|
)
|
|
END AS DateAddedNorm
|
|
FROM BlogsSplit
|
|
),
|
|
EarliestNote AS (
|
|
SELECT
|
|
n.NoteBlogId,
|
|
n.RootBlogId,
|
|
n.PostID,
|
|
n.TimeStamp,
|
|
ROW_NUMBER() OVER (
|
|
PARTITION BY n.NoteBlogId
|
|
ORDER BY n.TimeStamp ASC, n.PostID ASC
|
|
) AS rn
|
|
FROM Notes n
|
|
JOIN NoteTypes t ON t.TypeId = n.TypeId
|
|
JOIN Blogs root_bn ON root_bn.BlogId = n.RootBlogId
|
|
LEFT JOIN Posts p ON p.BlogName = root_bn.BlogName
|
|
AND p.PostID = n.PostID
|
|
WHERE t.Type IN ('reblog')--, 'reply')
|
|
AND n.NoteBlogId IN (SELECT BlogId FROM BlogsNorm) -- scope the window to blogs we care about
|
|
AND COALESCE(p.IsActive, 1) = 1 -- exclude notes on a post explicitly marked removed
|
|
)
|
|
SELECT
|
|
bn.BlogId,
|
|
bn.BlogName || '.tumblr.com',
|
|
'''' || bn.blogname || ''',',
|
|
bn.DateAdded,
|
|
bn.DateAddedNorm,
|
|
datetime(en.TimeStamp, 'unixepoch') AS earliest_reblog_or_reply_utc,
|
|
en.PostID AS earliest_reblog_or_reply_postid,
|
|
en.RootBlogId AS earliest_reblog_or_reply_root_blogid
|
|
FROM BlogsNorm bn
|
|
JOIN EarliestNote en ON en.NoteBlogId = bn.BlogId AND en.rn = 1
|
|
ORDER BY earliest_reblog_or_reply_utc, bn.DateAddedNorm desc
|
|
limit 50;
|
|
|
|
|
|
-- ----------------------------------------------------------------------------
|
|
-- QUERY 1 -- same answer, with a per-blog breakdown of which type(s) fired
|
|
-- and how many. Useful once QUERY 0 has rows; redundant while it's empty.
|
|
-- ----------------------------------------------------------------------------
|
|
-- WITH BlogsSplit AS ( ... ), BlogsNorm AS ( ... ) -- reuse the CTEs above
|
|
--
|
|
-- SELECT
|
|
-- bn.BlogId,
|
|
-- bn.BlogName,
|
|
-- bn.DateAddedNorm,
|
|
-- SUM(CASE WHEN t.Type = 'reblog' THEN 1 ELSE 0 END) AS reblog_count,
|
|
-- SUM(CASE WHEN t.Type = 'reply' THEN 1 ELSE 0 END) AS reply_count
|
|
-- FROM BlogsNorm bn
|
|
-- JOIN Notes n ON n.NoteBlogId = bn.BlogId
|
|
-- JOIN NoteTypes t ON t.TypeId = n.TypeId
|
|
-- WHERE t.Type IN ('reblog', 'reply')
|
|
-- GROUP BY bn.BlogId, bn.BlogName, bn.DateAddedNorm
|
|
-- ORDER BY bn.DateAddedNorm;
|
|
|
|
|
|
-- ----------------------------------------------------------------------------
|
|
-- QUERY 2 -- put a date floor back, if wanted later.
|
|
-- Same as QUERY 0, with one extra line in the outer WHERE:
|
|
-- AND bn.DateAddedNorm >= '2026-09-01' -- or whatever cutoff
|
|
-- ----------------------------------------------------------------------------
|
|
</sql><current_tab id="0"/></tab_sql></sqlb_project>
|