fix(db): make Blogs.BlogId the only blog ID and drop BlogNames
Blogs.BlogId was a one-time copy of BlogNames and nothing kept it current: 12,238 blogs first seen after 2026-08-07 had a BlogNames ID but a NULL Blogs.BlogId, so GetBlogs' join on BlogId silently skipped them and their 23,148 notes. - retire-blognames.sql: stub Blogs rows for the 17 unregistered note participants, backfill IDs (none renumbered), make ix_Blogs_BlogId UNIQUE, drop BlogNames, and add triggers that stop a Blogs row with a BlogId from being deleted, renamed or renumbered - AddNote registers both blogs via RegisterBlog (Blogs row + MAX+1 ID) and every query resolves names through Blogs instead of BlogNames - verify-db-schema.sql reports a DB that still has BlogNames (1e) - Update TL.db.md, AGENTS.md and the DB Browser saved queries Co-Authored-By: Claude Opus 5.5 <[email protected]>
This commit is contained in:
@@ -216,18 +216,19 @@ namespace URLNotesGrabberCORE
|
||||
#region Notes integer schema
|
||||
|
||||
// Notes stopped storing names on 2026-08-07: RootBlogName/NoteBlogName/Type became
|
||||
// RootBlogId/NoteBlogId/TypeId, resolved through BlogNames and NoteTypes. There is no
|
||||
// RootBlogId/NoteBlogId/TypeId, resolved through Blogs.BlogId and NoteTypes. There is no
|
||||
// compatibility view -- a query naming an old column fails outright, so this is a hard
|
||||
// cut rather than an optional column like IsActive. See TL.db.md.
|
||||
//
|
||||
// Blogs.BlogId is the only ID authority as of 2026-09-28; the BlogNames table that used
|
||||
// to hold the IDs is gone. Every name in Notes has a Blogs row, created by RegisterBlog.
|
||||
//
|
||||
// Two shapes recur below and are spelled out inline rather than hidden behind a helper,
|
||||
// so that every statement reads as the SQL it actually runs:
|
||||
// (SELECT BlogId FROM BlogNames WHERE BlogName = @name) -- unique-index probe, 20k rows
|
||||
// (SELECT BlogId FROM Blogs WHERE BlogName = @name) -- primary-key probe
|
||||
// (SELECT TypeId FROM NoteTypes WHERE Type = 'reply') -- 5 rows, effectively free
|
||||
// Joining Notes to Blogs is the one case that must NOT route through BlogNames: Blogs
|
||||
// carries its own BlogId, so N.NoteBlogId = B.BlogId is a single integer hop. Joining
|
||||
// Notes to Posts is the opposite case -- Posts has only BlogName, so it has to go
|
||||
// through BlogNames.
|
||||
// Joining Notes to Blogs is N.NoteBlogId = B.BlogId, a single hop on the unique index.
|
||||
// Joining Notes to Posts goes through Blogs too -- Posts has only BlogName.
|
||||
|
||||
/// <summary>
|
||||
/// True when the exception is a duplicate-key collision on Notes. The message embeds the
|
||||
@@ -242,14 +243,30 @@ namespace URLNotesGrabberCORE
|
||||
}
|
||||
|
||||
/// <summary>
|
||||
/// Gives a blog name an ID if it does not have one. No read-back and no round trip -- a
|
||||
/// name that is already registered keeps the ID that 1.18M Notes rows point at.
|
||||
/// Ensures a blog has a Blogs row and a BlogId, so a note can point at it. No read-back
|
||||
/// and no round trip -- a blog that already has an ID keeps the one Notes rows point at.
|
||||
/// Unlike AddBlog this does not skip "deact" names: a note by a deactivated blog still
|
||||
/// needs an ID, and Blogs is the only place one can live.
|
||||
/// Assigning the ID is bookkeeping, not a content change, so DateModified is not touched.
|
||||
/// MAX(BlogId) + 1 cannot hand out a used ID because trg_Blogs_BlogId_NoDelete stops any
|
||||
/// row that holds one from being deleted.
|
||||
/// </summary>
|
||||
private static void RegisterBlogName(SQLiteConnection connection, SQLiteTransaction? transaction, string blogName)
|
||||
private static void RegisterBlog(SQLiteConnection connection, SQLiteTransaction? transaction, string blogName)
|
||||
{
|
||||
using SQLiteCommand command = new SQLiteCommand("INSERT OR IGNORE INTO BlogNames (BlogName) VALUES (@BlogName)", connection, transaction);
|
||||
command.Parameters.AddWithValue("@BlogName", blogName);
|
||||
command.ExecuteNonQuery();
|
||||
string now = DateTime.Now.ToString("yyyy-MM-dd HH:mm:ss");
|
||||
|
||||
using (SQLiteCommand command = new SQLiteCommand("INSERT OR IGNORE INTO Blogs (BlogName, DateAdded, DateModified, DateCreated) VALUES (@BlogName, @Now, @Now, @Now)", connection, transaction))
|
||||
{
|
||||
command.Parameters.AddWithValue("@BlogName", blogName);
|
||||
command.Parameters.AddWithValue("@Now", now);
|
||||
command.ExecuteNonQuery();
|
||||
}
|
||||
|
||||
using (SQLiteCommand command = new SQLiteCommand("UPDATE Blogs SET BlogId = (SELECT IFNULL(MAX(BlogId), 0) + 1 FROM Blogs) WHERE BlogName = @BlogName AND BlogId IS NULL", connection, transaction))
|
||||
{
|
||||
command.Parameters.AddWithValue("@BlogName", blogName);
|
||||
command.ExecuteNonQuery();
|
||||
}
|
||||
}
|
||||
|
||||
/// <summary>
|
||||
@@ -762,7 +779,6 @@ namespace URLNotesGrabberCORE
|
||||
{
|
||||
DBPath ??= GetDefaultDbPath();
|
||||
//try { AddPost(rootBlogName, postID, DBPath); } catch { }
|
||||
try { AddBlog(noteBlogName, false, DBPath); } catch { }
|
||||
|
||||
using SQLiteConnection connection2 = new SQLiteConnection("Data Source=" + DBPath);
|
||||
int rowsInserted = 0;
|
||||
@@ -771,23 +787,24 @@ namespace URLNotesGrabberCORE
|
||||
{
|
||||
connection2.Open();
|
||||
|
||||
// Notes stores integer IDs, so both participants and the type have to exist in
|
||||
// their lookup table before the note can point at them.
|
||||
// Notes stores integer IDs, so both participants need a Blogs row with a BlogId,
|
||||
// and the type a NoteTypes row, before the note can point at them. RegisterBlog
|
||||
// also does what the AddBlog call here used to: register the note's blog.
|
||||
//
|
||||
// All four statements run in one transaction so a crash cannot leave a name or a
|
||||
// All statements run in one transaction so a crash cannot leave a blog or a
|
||||
// type registered with no note. The transaction is committed before the console
|
||||
// output below, which sleeps -- a write lock must not be held across that.
|
||||
using (SQLiteTransaction transaction = connection2.BeginTransaction())
|
||||
{
|
||||
RegisterBlogName(connection2, transaction, rootBlogName);
|
||||
RegisterBlogName(connection2, transaction, noteBlogName);
|
||||
RegisterBlog(connection2, transaction, rootBlogName);
|
||||
RegisterBlog(connection2, transaction, noteBlogName);
|
||||
RegisterNoteType(connection2, transaction, type ?? string.Empty);
|
||||
|
||||
// INSERT OR IGNORE, and no IsActive in the column list: re-crawling a note
|
||||
// that was removed elsewhere leaves the existing row -- and its flag -- alone.
|
||||
string sql = "INSERT OR IGNORE INTO Notes (RootBlogId, NoteBlogId, PostID, TimeStamp, TypeId, DatetimeCrawled, DateModified, DateCreated) " +
|
||||
"SELECT (SELECT BlogId FROM BlogNames WHERE BlogName = @rootBlogName), " +
|
||||
" (SELECT BlogId FROM BlogNames WHERE BlogName = @noteBlogName), " +
|
||||
"SELECT (SELECT BlogId FROM Blogs WHERE BlogName = @rootBlogName), " +
|
||||
" (SELECT BlogId FROM Blogs WHERE BlogName = @noteBlogName), " +
|
||||
" @PostID, @TimeStamp, " +
|
||||
" (SELECT TypeId FROM NoteTypes WHERE Type = @Type), " +
|
||||
" @DatetimeCrawled, @DateModified, @DateCreated";
|
||||
@@ -1126,11 +1143,11 @@ namespace URLNotesGrabberCORE
|
||||
{
|
||||
connection.Open();
|
||||
|
||||
string sql = "SELECT DISTINCT BN.BlogName as blogName, N.PostID" +
|
||||
string sql = "SELECT DISTINCT RB.BlogName as blogName, N.PostID" +
|
||||
" FROM Notes N" +
|
||||
" INNER JOIN BlogNames BN ON BN.BlogId = N.RootBlogId" +
|
||||
" INNER JOIN Blogs RB ON RB.BlogId = N.RootBlogId" +
|
||||
" WHERE N.TypeId = (SELECT TypeId FROM NoteTypes WHERE Type = 'reply')" + AndIsActive("Notes", "N", DBPath) +
|
||||
" ORDER BY BN.BlogName, N.PostID";
|
||||
" ORDER BY RB.BlogName, N.PostID";
|
||||
|
||||
using (SQLiteCommand command = new SQLiteCommand(sql, connection))
|
||||
{
|
||||
@@ -1171,11 +1188,11 @@ namespace URLNotesGrabberCORE
|
||||
connection.Open();
|
||||
|
||||
// Grouped on the integer rather than the name: the group key is what gets sorted,
|
||||
// and BN.BlogName comes along for free off the join.
|
||||
string sql = @"SELECT BN.BlogName as blogName, N.PostID,
|
||||
// and RB.BlogName comes along for free off the join.
|
||||
string sql = @"SELECT RB.BlogName as blogName, N.PostID,
|
||||
MAX(N.TimeStamp) as LatestTimestamp
|
||||
FROM Notes N
|
||||
INNER JOIN BlogNames BN ON BN.BlogId = N.RootBlogId
|
||||
INNER JOIN Blogs RB ON RB.BlogId = N.RootBlogId
|
||||
WHERE N.TypeId = (SELECT TypeId FROM NoteTypes WHERE Type = 'reply')
|
||||
AND (N.replyText IS NULL OR N.replyText = '' OR N.replyText = '.')" + AndIsActive("Notes", "N", DBPath) + @"
|
||||
GROUP BY N.RootBlogId, N.PostID
|
||||
@@ -1220,13 +1237,13 @@ namespace URLNotesGrabberCORE
|
||||
{
|
||||
connection.Open();
|
||||
|
||||
// Posts carries only BlogName, so this is the one join to Notes that has to go
|
||||
// through BlogNames -- there is no Posts.BlogId to hop on. The name predicate is
|
||||
// pushed into the 20k-row lookup, which then feeds integers to the Notes key.
|
||||
// Posts carries only BlogName, so the join to Notes goes through Blogs -- there is
|
||||
// no Posts.BlogId to hop on. Each post's name is a primary-key probe on Blogs,
|
||||
// which then feeds an integer to the Notes key.
|
||||
string sql = @"SELECT DISTINCT P.BlogName, P.PostID, MAX(N.TimeStamp) as LatestTimestamp
|
||||
FROM Posts P
|
||||
INNER JOIN BlogNames RBN ON RBN.BlogName = P.BlogName
|
||||
INNER JOIN Notes N ON N.RootBlogId = RBN.BlogId AND N.PostID = P.PostID
|
||||
INNER JOIN Blogs RB ON RB.BlogName = P.BlogName
|
||||
INNER JOIN Notes N ON N.RootBlogId = RB.BlogId AND N.PostID = P.PostID
|
||||
WHERE P.NotFound = 0
|
||||
AND N.TypeId = (SELECT TypeId FROM NoteTypes WHERE Type = 'reply')
|
||||
AND (N.replyText IS NULL OR N.replyText = '' OR N.replyText = '.')" + AndIsActive("Posts", "P", DBPath) + AndIsActive("Notes", "N", DBPath) + @"
|
||||
@@ -1415,8 +1432,7 @@ namespace URLNotesGrabberCORE
|
||||
try
|
||||
{
|
||||
connection.Open();
|
||||
// Blogs is reached in one integer hop off Blogs.BlogId, not through BlogNames --
|
||||
// that would add a hop and end in the text comparison the migration removed.
|
||||
// Blogs is reached in one integer hop off Blogs.BlogId.
|
||||
// The negated form is only correct because Notes.TypeId is NOT NULL.
|
||||
string sql = "";
|
||||
if (reblogsOnly)
|
||||
@@ -1703,8 +1719,8 @@ namespace URLNotesGrabberCORE
|
||||
connection.Open();
|
||||
|
||||
string sql = "UPDATE Notes SET TimeStamp = @timestamp, DateModified = @dateModified " +
|
||||
"WHERE RootBlogId = (SELECT BlogId FROM BlogNames WHERE BlogName = @rootBlogName) " +
|
||||
"AND NoteBlogId = (SELECT BlogId FROM BlogNames WHERE BlogName = @noteBlogName) " +
|
||||
"WHERE RootBlogId = (SELECT BlogId FROM Blogs WHERE BlogName = @rootBlogName) " +
|
||||
"AND NoteBlogId = (SELECT BlogId FROM Blogs WHERE BlogName = @noteBlogName) " +
|
||||
"AND PostID = @postID AND IFNULL(TimeStamp, 0) <> @timestamp";
|
||||
using (SQLiteCommand command = new SQLiteCommand(sql, connection))
|
||||
{
|
||||
@@ -2018,7 +2034,7 @@ namespace URLNotesGrabberCORE
|
||||
// Only fan out to rows that match the SELECT criteria in GetRepliesWithFilledText (NULL/empty/legacy-'.'). Never overwrite '?' (confirmed-empty) or already-fetched text.
|
||||
// The ABS() term cannot use an index on TimeStamp, before or after the integer schema; the NoteBlogId probe is what keeps this off a full scan.
|
||||
string sql = "UPDATE Notes SET replyText = @replyText, DateModified = @dateModified " +
|
||||
"WHERE NoteBlogId = (SELECT BlogId FROM BlogNames WHERE BlogName = @noteBlogName) " +
|
||||
"WHERE NoteBlogId = (SELECT BlogId FROM Blogs WHERE BlogName = @noteBlogName) " +
|
||||
"AND ABS(TimeStamp - @TimeStamp) <= 5 " +
|
||||
"AND TypeId = (SELECT TypeId FROM NoteTypes WHERE Type = 'reply') " +
|
||||
"AND (replyText IS NULL OR replyText = '' OR replyText = '.') " +
|
||||
@@ -2064,7 +2080,7 @@ namespace URLNotesGrabberCORE
|
||||
connection.Open();
|
||||
|
||||
string sql = "UPDATE Notes SET replyText = @replyText, DateModified = @dateModified " +
|
||||
"WHERE RootBlogId = (SELECT BlogId FROM BlogNames WHERE BlogName = @rootBlogName) " +
|
||||
"WHERE RootBlogId = (SELECT BlogId FROM Blogs WHERE BlogName = @rootBlogName) " +
|
||||
"AND PostID = @PostID " +
|
||||
"AND TypeId = (SELECT TypeId FROM NoteTypes WHERE Type = 'reply') " +
|
||||
"AND IFNULL(replyText, '.') <> @replyText";
|
||||
|
||||
+112
-94
@@ -24,21 +24,43 @@ Everything below was read out of the live file, not inferred from code. Counts a
|
||||
> Applied by `../normalize-notes.sql`, which took the file from 207 MB to 148 MB. An
|
||||
> earlier change the same day (`../shrink-db.sql`) took it from 267 MB to 207 MB.
|
||||
|
||||
> ### ⚠ Breaking change, 2026-09-28: `BlogNames` is gone; `Blogs.BlogId` is the only ID authority
|
||||
>
|
||||
> The IDs in `Notes` used to live in a `BlogNames` table, with a copy in `Blogs.BlogId`.
|
||||
> Nothing kept the copy current, so by 2026-09-28 12,238 blogs first seen after the
|
||||
> migration had `Blogs.BlogId = NULL`. Every `Notes`-to-`Blogs` join on `BlogId` silently
|
||||
> skipped them and their 23,148 notes, which kept them out of `GetBlogs`.
|
||||
>
|
||||
> `../retire-blognames.sql` fixed this by giving every note participant a `Blogs` row,
|
||||
> backfilling the IDs (none renumbered), making `ix_Blogs_BlogId` unique, and **dropping
|
||||
> `BlogNames`**. There is no compatibility view: any query naming it fails with
|
||||
> `no such table: BlogNames`. Two triggers now protect the IDs.
|
||||
>
|
||||
> **Porting an app:** replace `BlogNames` with `Blogs` everywhere. The columns you used,
|
||||
> `BlogId` and `BlogName`, exist there with the same meaning. A name lookup
|
||||
> (`SELECT BlogId FROM Blogs WHERE BlogName = ?`) is a primary-key probe, and an ID
|
||||
> lookup or join (`JOIN Blogs b ON b.BlogId = n.NoteBlogId`) uses the unique
|
||||
> `ix_Blogs_BlogId`. Every ID in `Notes` resolves to exactly one `Blogs` row. `Blogs.BlogId`
|
||||
> is **no longer** a stale copy, so any code or docs that distrust it can drop that
|
||||
> caveat. Never write `BlogId` or `BlogName` on a row that has an ID, and never delete such
|
||||
> a row: the triggers reject all three. See [`Blogs`](#blogs).
|
||||
|
||||
---
|
||||
|
||||
## The three content tables
|
||||
|
||||
| Table | Rows | What it is |
|
||||
|---|--:|---|
|
||||
| `Blogs` | 188,620 | The crawl registry — one row per known blog, plus crawl-state flags |
|
||||
| `Blogs` | 198,560 | The crawl registry, one row per known blog plus crawl-state flags. Also the ID authority for blogs in `Notes` |
|
||||
| `Posts` | 22,468 | Stored post content. Only 3,867 blogs actually have any |
|
||||
| `Notes` | 1,182,333 | The engagement graph: `NoteBlogId` acted on `(RootBlogId, PostID)` |
|
||||
| `Notes` | 1,234,830 | The engagement graph: `NoteBlogId` acted on `(RootBlogId, PostID)` |
|
||||
|
||||
…supported by two lookup tables that exist only to keep `Notes` small:
|
||||
(`Blogs` and `Notes` counts as of 2026-09-28; the rest as of 2026-08-07.)
|
||||
|
||||
…supported by one lookup table that exists only to keep `Notes` small:
|
||||
|
||||
| Table | Rows | What it is |
|
||||
|---|--:|---|
|
||||
| `BlogNames` | 20,430 | `BlogId` ⇄ `BlogName`. The ID authority for everything in `Notes` |
|
||||
| `NoteTypes` | 5 | `TypeId` ⇄ `Type`. `like`, `reblog`, `reply`, `posted`, `post_attribution` |
|
||||
|
||||
The engagement graph is the interesting part. 20,311 distinct blogs appear as engagers —
|
||||
@@ -66,27 +88,47 @@ CREATE TABLE "Blogs" (
|
||||
PRIMARY KEY("BlogName")
|
||||
);
|
||||
|
||||
CREATE INDEX ix_Blogs_BlogId ON Blogs (BlogId);
|
||||
CREATE UNIQUE INDEX ix_Blogs_BlogId ON Blogs (BlogId);
|
||||
|
||||
CREATE TRIGGER trg_Blogs_BlogId_NoDelete -- no DELETE of a row that has a BlogId
|
||||
CREATE TRIGGER trg_Blogs_BlogId_Immutable -- no change to its BlogId or BlogName
|
||||
```
|
||||
|
||||
`BlogName` is the primary key, so it is the only indexed way in by name. There is no index
|
||||
on any flag or date — filtering or sorting on those scans all 188k rows, which is
|
||||
on any flag or date. Filtering or sorting on those scans the whole table, which is
|
||||
affordable here and is not on `Notes`.
|
||||
|
||||
**`BlogId` is new as of 2026-08-07 and is the join key to `Notes`.** It exists so that
|
||||
`Notes` can reach `Blogs` in a single integer hop rather than going through `BlogNames`
|
||||
and ending in a text comparison:
|
||||
**`BlogId` is the ID that `Notes.RootBlogId` and `Notes.NoteBlogId` store, and `Blogs` is
|
||||
the only place it lives** (since 2026-09-28; see the banner at the top). The join to
|
||||
`Notes` is one integer hop on the unique index:
|
||||
|
||||
```sql
|
||||
-- what you want
|
||||
FROM Blogs B JOIN Notes N ON N.NoteBlogId = B.BlogId
|
||||
|
||||
-- not this
|
||||
FROM Blogs B JOIN BlogNames BN ON BN.BlogName = B.BlogName
|
||||
JOIN Notes N ON N.NoteBlogId = BN.BlogId
|
||||
```
|
||||
|
||||
**`BlogId` is NULL on 168,202 of 188,620 rows** — every blog that has never appeared in a
|
||||
**Every blog that appears in `Notes` has a `Blogs` row with a `BlogId`.** `AddNote`
|
||||
guarantees it through `RegisterBlog`, which runs in the note's own transaction:
|
||||
|
||||
```sql
|
||||
INSERT OR IGNORE INTO Blogs (BlogName, DateAdded, DateModified, DateCreated)
|
||||
VALUES (@name, @now, @now, @now);
|
||||
UPDATE Blogs SET BlogId = (SELECT IFNULL(MAX(BlogId), 0) + 1 FROM Blogs)
|
||||
WHERE BlogName = @name AND BlogId IS NULL;
|
||||
```
|
||||
|
||||
- Unlike `AddBlog`, this does **not** skip names containing `deact`. A note by a
|
||||
deactivated blog still needs an ID, so such blogs now get registry rows too, with the
|
||||
usual defaults (`HasBeenOutput = 0`, `IsActive` left at its default).
|
||||
- Assigning a `BlogId` is bookkeeping, so it **does not move `DateModified`**.
|
||||
- `MAX(BlogId) + 1` is safe only because an ID can never be freed. The two triggers see
|
||||
to that: deleting a row that has a `BlogId`, or changing its `BlogId` or `BlogName`,
|
||||
aborts. Remove a blog with `IsActive = 0` instead. A blog renamed upstream gets a new
|
||||
row. Rows with no `BlogId` can still be deleted or renamed freely.
|
||||
- `INSERT OR REPLACE` on `Blogs` gets around the delete trigger (SQLite does not fire
|
||||
delete triggers for REPLACE unless `recursive_triggers` is on), and it would wipe the
|
||||
`BlogId`. It was already forbidden because it resets `IsActive`. Do not use it.
|
||||
|
||||
**`BlogId` is NULL on 165,887 of 198,560 rows**, every blog that has never appeared in a
|
||||
note. That is the large majority, and it is not an error: the registry is far bigger than
|
||||
the engagement graph. An inner join on `BlogId` therefore silently drops those blogs,
|
||||
which is usually what you want for engagement queries and is wrong for registry listings.
|
||||
@@ -187,8 +229,7 @@ CREATE INDEX ix_Notes_NoteBlogId ON Notes (NoteBlogId);
|
||||
|
||||
**Integer IDs since 2026-08-07 — this is the breaking change.** `RootBlogName`,
|
||||
`NoteBlogName` and `Type` are gone, replaced by `RootBlogId`, `NoteBlogId` and `TypeId`.
|
||||
Resolve them through [`BlogNames`](#blognames) and [`NoteTypes`](#notetypes), or join
|
||||
straight to `Blogs` on `BlogId`. The old names were text repeated on 1.18 million rows,
|
||||
Resolve blog IDs through `Blogs.BlogId` and types through [`NoteTypes`](#notetypes). The old names were text repeated on 1.18 million rows,
|
||||
in the table *and* in every index over it; the swap took the file from 207 MB to 148 MB.
|
||||
|
||||
The **primary key column order is deliberately unchanged**, so the leading-prefix access
|
||||
@@ -251,43 +292,21 @@ At 1.18M rows this is the table that dictates how the whole database has to be q
|
||||
- `replyText` is `'.'` on 1,167,464 rows — only `reply` notes carry real text. Those
|
||||
dots are inherited from the old column default; new rows get `NULL` instead.
|
||||
|
||||
**Resolve IDs by filtering the lookup, not by scanning `Notes`.** The lookup tables are
|
||||
tiny and uniquely indexed, so pushing a name predicate into them costs nothing and lets
|
||||
the `Notes` index do the work:
|
||||
**Resolve IDs by filtering `Blogs`, not by scanning `Notes`.** A name predicate on `Blogs`
|
||||
is a primary-key probe, so pushing it there costs nothing and lets the `Notes` index do
|
||||
the work:
|
||||
|
||||
```sql
|
||||
-- good: BlogNames resolves the name, then the index is searched
|
||||
-- good: Blogs resolves the name, then the index is searched
|
||||
SELECT * FROM Notes
|
||||
WHERE NoteBlogId = (SELECT BlogId FROM BlogNames WHERE BlogName = ?);
|
||||
WHERE NoteBlogId = (SELECT BlogId FROM Blogs WHERE BlogName = ?);
|
||||
|
||||
-- also good, same plan
|
||||
SELECT n.* FROM Notes n
|
||||
JOIN BlogNames b ON b.BlogId = n.NoteBlogId
|
||||
JOIN Blogs b ON b.BlogId = n.NoteBlogId
|
||||
WHERE b.BlogName = ?;
|
||||
```
|
||||
|
||||
### `BlogNames`
|
||||
|
||||
```sql
|
||||
CREATE TABLE BlogNames (
|
||||
BlogId INTEGER PRIMARY KEY,
|
||||
BlogName TEXT NOT NULL UNIQUE
|
||||
);
|
||||
```
|
||||
|
||||
20,430 rows — every name appearing in `Notes` as either participant, and nothing else.
|
||||
This is the **ID authority**: `Notes.RootBlogId` and `Notes.NoteBlogId` both point here,
|
||||
and `Blogs.BlogId` is a copy of the value for the blogs that have one.
|
||||
|
||||
**12 of these names have no `Blogs` row.** The registry has never been a superset of the
|
||||
engagement graph and still is not, so resolving an ID through `Blogs` rather than
|
||||
`BlogNames` will occasionally find nothing. Use `BlogNames` when you need the name itself
|
||||
and `Blogs` when you need registry columns.
|
||||
|
||||
IDs are assigned by SQLite and are **stable**: they are stored in over a million `Notes`
|
||||
rows. Never renumber them. A blog that is renamed upstream should get a new row, not an
|
||||
edit to an existing one, unless every `Notes` reference is migrated with it.
|
||||
|
||||
### `NoteTypes`
|
||||
|
||||
```sql
|
||||
@@ -317,6 +336,9 @@ code to this table's contents, so prefer the join in anything long-lived.
|
||||
|
||||
## Porting to the integer schema
|
||||
|
||||
> Written for the 2026-08-07 change, and updated for 2026-09-28: wherever this section
|
||||
> once said `BlogNames`, it now says `Blogs`. `BlogNames` no longer exists.
|
||||
|
||||
Everything here was checked against the live 148 MB file. There were 14 affected call
|
||||
sites in `DataAccess.cs` and 16 in `RolodexRepository.cs`. TumblThree needs no changes —
|
||||
its single statement touches `Blogs.IsActive` and `BlogName` only.
|
||||
@@ -333,8 +355,8 @@ the result.
|
||||
|
||||
| Was | Is now | Resolve via |
|
||||
|---|---|---|
|
||||
| `Notes.RootBlogName` | `Notes.RootBlogId` | `BlogNames.BlogId` → `.BlogName` |
|
||||
| `Notes.NoteBlogName` | `Notes.NoteBlogId` | `BlogNames.BlogId` → `.BlogName` |
|
||||
| `Notes.RootBlogName` | `Notes.RootBlogId` | `Blogs.BlogId` → `.BlogName` |
|
||||
| `Notes.NoteBlogName` | `Notes.NoteBlogId` | `Blogs.BlogId` → `.BlogName` |
|
||||
| `Notes.Type` | `Notes.TypeId` | `NoteTypes.TypeId` → `.Type` |
|
||||
| `ix_NoteBlogName01` | `ix_Notes_NoteBlogId` | — |
|
||||
|
||||
@@ -347,14 +369,14 @@ the result.
|
||||
-- was
|
||||
WHERE NoteBlogName = @Name
|
||||
|
||||
-- now, either form; both search ix_Notes_NoteBlogId after a unique-index lookup
|
||||
WHERE NoteBlogId = (SELECT BlogId FROM BlogNames WHERE BlogName = @Name)
|
||||
-- now, either form; both search ix_Notes_NoteBlogId after a primary-key lookup
|
||||
WHERE NoteBlogId = (SELECT BlogId FROM Blogs WHERE BlogName = @Name)
|
||||
-- or
|
||||
JOIN BlogNames b ON b.BlogId = n.NoteBlogId WHERE b.BlogName = @Name
|
||||
JOIN Blogs b ON b.BlogId = n.NoteBlogId WHERE b.BlogName = @Name
|
||||
```
|
||||
|
||||
Measured 73 ms against 63 ms for the old text form on the busiest blog — the extra hop is
|
||||
a unique-index probe on a 20k-row table and does not show.
|
||||
Measured at 73 ms against 63 ms for the old text form on the busiest blog, when the lookup
|
||||
was still `BlogNames`. The extra hop is one index probe and does not show.
|
||||
|
||||
### Joining `Notes` to `Blogs`
|
||||
|
||||
@@ -364,12 +386,17 @@ This is the join to get right; it is the most common shape in both applications.
|
||||
-- was
|
||||
FROM Blogs B INNER JOIN Notes N ON N.NoteBlogName = B.BlogName
|
||||
|
||||
-- now: one integer hop, using the new Blogs.BlogId
|
||||
-- now: one integer hop, using Blogs.BlogId
|
||||
FROM Blogs B INNER JOIN Notes N ON N.NoteBlogId = B.BlogId
|
||||
```
|
||||
|
||||
Do **not** route this through `BlogNames` — that adds a hop and ends in the text
|
||||
comparison the change was meant to remove.
|
||||
Joining `Notes` to `Posts` also goes through `Blogs`, since `Posts` has only a name:
|
||||
|
||||
```sql
|
||||
FROM Posts P
|
||||
JOIN Blogs RB ON RB.BlogName = P.BlogName
|
||||
JOIN Notes N ON N.RootBlogId = RB.BlogId AND N.PostID = P.PostID
|
||||
```
|
||||
|
||||
### Selecting a name back out
|
||||
|
||||
@@ -378,12 +405,12 @@ comparison the change was meant to remove.
|
||||
SELECT NoteBlogName AS blogName, COUNT(*) FROM Notes ... GROUP BY NoteBlogName
|
||||
|
||||
-- now
|
||||
SELECT bn.BlogName AS blogName, COUNT(*)
|
||||
FROM Notes n JOIN BlogNames bn ON bn.BlogId = n.NoteBlogId
|
||||
... GROUP BY bn.BlogName
|
||||
SELECT b.BlogName AS blogName, COUNT(*)
|
||||
FROM Notes n JOIN Blogs b ON b.BlogId = n.NoteBlogId
|
||||
... GROUP BY b.BlogName
|
||||
```
|
||||
|
||||
Group by `n.NoteBlogId` instead of `bn.BlogName` when you only need the name for display —
|
||||
Group by `n.NoteBlogId` instead of `b.BlogName` when you only need the name for display —
|
||||
grouping on the integer is cheaper and the name comes along for free.
|
||||
|
||||
### Filtering by type
|
||||
@@ -407,27 +434,25 @@ because `TypeId` is `NOT NULL`.
|
||||
|
||||
### Inserting a note
|
||||
|
||||
The crawler must ensure both names have IDs first. `INSERT OR IGNORE` on `BlogNames` is
|
||||
the whole of it — no read-back, no round trip, safe to run every time:
|
||||
The crawler must ensure both blogs have IDs first: run the `RegisterBlog` pair shown under
|
||||
[`Blogs`](#blogs) for each name. No read-back, no round trip, and safe to run every time.
|
||||
Then:
|
||||
|
||||
```sql
|
||||
INSERT OR IGNORE INTO BlogNames (BlogName) VALUES (@rootBlogName);
|
||||
INSERT OR IGNORE INTO BlogNames (BlogName) VALUES (@noteBlogName);
|
||||
|
||||
INSERT OR IGNORE INTO Notes
|
||||
(RootBlogId, PostID, NoteBlogId, TimeStamp, TypeId,
|
||||
DatetimeCrawled, DateModified, DateCreated)
|
||||
SELECT (SELECT BlogId FROM BlogNames WHERE BlogName = @rootBlogName),
|
||||
SELECT (SELECT BlogId FROM Blogs WHERE BlogName = @rootBlogName),
|
||||
@PostID,
|
||||
(SELECT BlogId FROM BlogNames WHERE BlogName = @noteBlogName),
|
||||
(SELECT BlogId FROM Blogs WHERE BlogName = @noteBlogName),
|
||||
@TimeStamp,
|
||||
(SELECT TypeId FROM NoteTypes WHERE Type = @Type),
|
||||
@DatetimeCrawled, @DateModified, @DateCreated;
|
||||
```
|
||||
|
||||
Verified: a genuinely new note inserts, and re-running the identical statement inserts 0.
|
||||
Run all three statements in one transaction so a crash cannot leave a name registered
|
||||
with no note.
|
||||
Run the registrations and the insert in one transaction so a crash cannot leave a blog
|
||||
registered with no note.
|
||||
|
||||
**The duplicate-key error message has changed.** `DataAccess.cs` compares against the
|
||||
literal string
|
||||
@@ -452,7 +477,7 @@ on `TimeStamp` either before or after:
|
||||
```sql
|
||||
-- now
|
||||
UPDATE Notes SET replyText = @replyText, DateModified = @dateModified
|
||||
WHERE NoteBlogId = (SELECT BlogId FROM BlogNames WHERE BlogName = @noteBlogName)
|
||||
WHERE NoteBlogId = (SELECT BlogId FROM Blogs WHERE BlogName = @noteBlogName)
|
||||
AND ABS(TimeStamp - @TimeStamp) <= 5
|
||||
AND TypeId = (SELECT TypeId FROM NoteTypes WHERE Type = 'reply')
|
||||
AND (replyText IS NULL OR replyText = '' OR replyText = '.')
|
||||
@@ -462,18 +487,15 @@ UPDATE Notes SET replyText = @replyText, DateModified = @dateModified
|
||||
Rolodex's soft-delete updates need no change beyond the `WHERE` clause — they set
|
||||
`IsActive`, which is untouched.
|
||||
|
||||
### Three traps
|
||||
### Two traps
|
||||
|
||||
**`Blogs.BlogId` is NULL on 168,202 of 188,620 rows.** Any inner join on it silently drops
|
||||
**`Blogs.BlogId` is NULL on 165,887 of 198,560 rows.** Any inner join on it silently drops
|
||||
every blog that has never appeared in a note. Correct for engagement queries; wrong for
|
||||
registry listings, which need a `LEFT JOIN` or no join at all.
|
||||
|
||||
**12 names in `BlogNames` have no `Blogs` row.** Resolving an ID to a name through `Blogs`
|
||||
will occasionally find nothing. Use `BlogNames` for names and `Blogs` for registry columns.
|
||||
|
||||
**IDs are stable and must stay so.** `BlogNames.BlogId` and `NoteTypes.TypeId` are stored
|
||||
in over a million `Notes` rows. Never renumber. A blog renamed upstream gets a new row,
|
||||
not an edited one, unless every `Notes` reference migrates with it.
|
||||
**IDs are stable and must stay so.** `Blogs.BlogId` and `NoteTypes.TypeId` are stored in
|
||||
over a million `Notes` rows. Never renumber. A blog renamed upstream gets a new row, not
|
||||
an edited one. The `Blogs` triggers reject both.
|
||||
|
||||
---
|
||||
|
||||
@@ -481,16 +503,12 @@ not an edited one, unless every `Notes` reference migrates with it.
|
||||
|
||||
There are no foreign keys, and the tables do not perfectly agree:
|
||||
|
||||
- 4 `Posts` rows name a blog with no `Blogs` row.
|
||||
- 12 of the 20,430 names in `BlogNames` have no `Blogs` row.
|
||||
|
||||
So a name appearing in `Notes` or `Posts` is not a guarantee that the registry knows about
|
||||
it. Joins from those tables back to `Blogs` should tolerate a miss.
|
||||
|
||||
The integer schema does not fix this and was not meant to. `BlogNames` is deliberately
|
||||
built from `Notes` rather than from `Blogs`, precisely so that the 12 unregistered
|
||||
engagers keep their IDs and their rows. Had it been built from the registry, those notes
|
||||
would have been dropped by the migration's inner joins.
|
||||
- 4 `Posts` rows name a blog with no `Blogs` row, so joins from `Posts` back to `Blogs`
|
||||
should tolerate a miss.
|
||||
- `Notes` is covered: every `RootBlogId` and `NoteBlogId` resolves to a `Blogs` row.
|
||||
`retire-blognames.sql` checked this before committing, and `RegisterBlog` keeps it true.
|
||||
Before 2026-09-28, 12 to 17 note participants had no registry row. They now have stub
|
||||
rows.
|
||||
|
||||
---
|
||||
|
||||
@@ -542,11 +560,10 @@ Crawler bookkeeping. Rolodex ignores all of these.
|
||||
`DataAccess.cs` joins on it to decide what to collect:
|
||||
|
||||
```sql
|
||||
-- shape only; the ported GetBlogs joins Blogs directly on BlogId and needs no BlogNames hop
|
||||
SELECT bn.BlogName, count(*)
|
||||
-- shape only
|
||||
SELECT b.BlogName, count(*)
|
||||
FROM Notes n
|
||||
JOIN Blogs b ON b.BlogId = n.NoteBlogId
|
||||
JOIN BlogNames bn ON bn.BlogId = n.NoteBlogId
|
||||
JOIN Blogs b ON b.BlogId = n.NoteBlogId
|
||||
WHERE b.IsActive = @isActive AND ...
|
||||
```
|
||||
|
||||
@@ -644,7 +661,7 @@ handled:
|
||||
SELECT 'Blogs', COUNT(*) FROM Blogs
|
||||
UNION ALL SELECT 'Posts', COUNT(*) FROM Posts
|
||||
UNION ALL SELECT 'Notes', COUNT(*) FROM Notes
|
||||
UNION ALL SELECT 'BlogNames', COUNT(*) FROM BlogNames;
|
||||
UNION ALL SELECT 'Blogs with a BlogId', COUNT(*) FROM Blogs WHERE BlogId IS NOT NULL;
|
||||
|
||||
-- note type mix (joins NoteTypes; Notes.Type no longer exists)
|
||||
SELECT t.Type, COUNT(*)
|
||||
@@ -668,8 +685,9 @@ SELECT COUNT(*) FROM (
|
||||
SELECT COUNT(*) FROM Posts p
|
||||
WHERE NOT EXISTS (SELECT 1 FROM Blogs b WHERE b.BlogName = p.BlogName);
|
||||
|
||||
SELECT COUNT(*) FROM BlogNames bn
|
||||
WHERE NOT EXISTS (SELECT 1 FROM Blogs b WHERE b.BlogName = bn.BlogName);
|
||||
-- note participants with no Blogs.BlogId (expect 0; anything else is the pre-2026-09-28 drift)
|
||||
SELECT COUNT(*) FROM (SELECT DISTINCT NoteBlogId AS Id FROM Notes) n
|
||||
WHERE NOT EXISTS (SELECT 1 FROM Blogs b WHERE b.BlogId = n.Id);
|
||||
|
||||
-- space by object, to see where the file actually goes
|
||||
SELECT name, SUM(pgsize)/1024/1024 AS mb
|
||||
|
||||
Reference in New Issue
Block a user