Files
URLNotesGrabberCore/URLNotesGrabberCORE/LegacyPostsDbImporter.cs
jimandClaude Opus 5 34da632e6a fix(posttype): stop Unknown.txt and type posts at their source
PostType becomes an output filename, so an unset or unvalidated value does
not stay a data problem -- it creates a file. OutputMode wrote untyped rows
to `PostType ?? "Unknown"`, IngestMode read that file back and derived the
literal type "Unknown" from its name, and the two would have regenerated
each other indefinitely.

Nothing was setting the type in the first place. AddPost -- the path every
notes/likes harvest goes through -- omitted PostType from its INSERT column
list entirely, so 1790 rows across 469 blogs had none. Ingest could never
repair them: it types a post only when it meets it inside a real export
file, and these posts appear in none.

Type at the source, from what each path actually knows:

- TraverseDirectory takes it from the filename it is already reading
  ("texts.txt" -> "texts"), the same rule ingest uses.
- CollectLikes has no file, so it reads the legacy-format `type` field that
  GrabLikes already requests with npf=false, mapped singular -> plural.
- AddPost/UpdatePost gained the plumbing to carry it. UpdatePost fills a
  missing type but never overwrites one, and its change-detection clause
  had to learn about PostType or the SET would be unreachable for a row
  whose content was already current.

PostTypes is the single source of truth: eight canonical names, and
anything else normalizes to null. Null is safe -- OutputMode skips those
rows -- while a stray value would have become a stray file. IngestMode and
TraverseDirectory now skip non-export .txt files outright, and the legacy
importer no longer passes a pre-column NULL straight back in.

For the rows already stored untyped, content is the only signal left, so
the migration infers from which columns they carry. Verified against a copy
of the live database: 1780 of 1790 typed, 10 left untyped for want of any
content at all, idempotent on a second pass. HasImage is deliberately not
consulted -- it is set on 12,420 of 19,828 known text posts.

Co-Authored-By: Claude Opus 5 <[email protected]>
2026-08-20 08:00:56 -05:00

162 lines
8.5 KiB
C#

using System.Data.SQLite;
namespace URLNotesGrabberCORE
{
// One-time migration: opens a legacy ThreeTxtFileHelper posts.db, copies its
// Blog + PostData rows into the merged TL.db via DataAccess.
// Conflict rule on (BlogName, PostId): ThreeTxtFileHelper wins on the 22 content
// columns + PostType + DateModified (handled inside UpsertPostFromTextFile).
// Engagement columns in TL.db (ByLikes, RootBlogName, RootURL, HasNotesGathered,
// NotFound, NotesGatheredDateTime) are preserved.
public static class LegacyPostsDbImporter
{
public static int Run(string legacyDbPath)
{
if (string.IsNullOrWhiteSpace(legacyDbPath))
{
Console.WriteLine("LegacyPostsDbImporter: path to legacy posts.db is required.");
return 1;
}
if (!File.Exists(legacyDbPath))
{
Console.WriteLine($"Legacy posts.db not found at: {legacyDbPath}");
return 1;
}
DataAccess.EnsureTTFileHelperColumnsExist();
Console.WriteLine($"Reading legacy posts.db: {legacyDbPath}");
int blogsCopied = 0;
int blogPathsWritten = 0;
int blogsWithoutPath = 0;
int postsUpserted = 0;
int errors = 0;
try
{
using var src = new SQLiteConnection("Data Source=" + legacyDbPath + ";Read Only=True;");
src.Open();
// 1) Copy Blogs (BlogName + TTFolderPath)
using (var cmd = new SQLiteCommand("SELECT BlogName, TTFolderPath FROM Blogs", src))
using (var reader = cmd.ExecuteReader())
{
while (reader.Read())
{
string blogName = reader.IsDBNull(0) ? string.Empty : reader.GetString(0);
string? ttFolderPath = reader.IsDBNull(1) ? null : reader.GetString(1);
if (string.IsNullOrWhiteSpace(blogName)) continue;
try
{
// A legacy row whose TTFolderPath was already NULL copies nothing.
// Counting it as "copied" is what hid the fact that this import has
// never populated a single path.
if (string.IsNullOrWhiteSpace(ttFolderPath))
blogsWithoutPath++;
else if (DataAccess.SetBlogTTFolderPath(blogName, ttFolderPath.Trim()))
blogPathsWritten++;
blogsCopied++;
}
catch (Exception ex)
{
errors++;
Console.WriteLine($" Blog copy failed for '{blogName}': {ex.Message}");
}
}
}
Console.WriteLine($" Blogs seen: {blogsCopied}, TTFolderPath written: {blogPathsWritten}, legacy rows with no path: {blogsWithoutPath}");
// 2) Copy Posts
try
{
DataAccess.EnableImportModePragmas();
DataAccess.BeginImportSession();
string sql = @"SELECT BlogName, PostId, ReblogUrl, Date, HasImage, PostUrl, Slug,
ReblogKey, ReblogName, Summary, Quote, Body, Tags, Link,
PhotoUrl, PhotoCaption, DownloadedFiles, AudioCaption,
Question, Answer, Title, PostType
FROM Posts";
using var cmd = new SQLiteCommand(sql, src);
using var reader = cmd.ExecuteReader();
while (reader.Read())
{
try
{
string blogName = reader.IsDBNull(0) ? string.Empty : reader.GetString(0);
string postId = reader.IsDBNull(1) ? string.Empty : reader.GetString(1);
if (string.IsNullOrWhiteSpace(blogName) || string.IsNullOrWhiteSpace(postId)) continue;
string? hasImageRaw = reader.IsDBNull(4) ? null : reader.GetValue(4)?.ToString();
bool hasImage = !string.IsNullOrWhiteSpace(hasImageRaw)
&& (hasImageRaw.Equals("true", StringComparison.OrdinalIgnoreCase)
|| hasImageRaw == "1"
|| hasImageRaw.Equals("yes", StringComparison.OrdinalIgnoreCase));
DataAccess.UpsertPostFromTextFile(
blogName: blogName,
postID: postId,
reblogURL: reader.IsDBNull(2) ? null : reader.GetString(2),
postDate: reader.IsDBNull(3) ? null : reader.GetString(3),
postURL: reader.IsDBNull(5) ? null : reader.GetString(5),
slug: reader.IsDBNull(6) ? null : reader.GetString(6),
reblogKey: reader.IsDBNull(7) ? null : reader.GetString(7),
reblogName: reader.IsDBNull(8) ? null : reader.GetString(8),
summary: reader.IsDBNull(9) ? null : reader.GetString(9),
quote: reader.IsDBNull(10) ? null : reader.GetString(10),
body: reader.IsDBNull(11) ? null : reader.GetString(11),
tags: reader.IsDBNull(12) ? null : reader.GetString(12),
link: reader.IsDBNull(13) ? null : reader.GetString(13),
photoURL: reader.IsDBNull(14) ? null : reader.GetString(14),
photoCaption: reader.IsDBNull(15) ? null : reader.GetString(15),
downloadedFiles: reader.IsDBNull(16) ? null : reader.GetString(16),
audioCaption: reader.IsDBNull(17) ? null : reader.GetString(17),
question: reader.IsDBNull(18) ? null : reader.GetString(18),
answer: reader.IsDBNull(19) ? null : reader.GetString(19),
title: reader.IsDBNull(20) ? null : reader.GetString(20),
// A legacy Posts.db predating the PostType column hands back NULL
// here, and on the INSERT branch that NULL is stored -- reseeding
// exactly the untyped rows the backfill exists to clear. Normalize
// so an unrecognized legacy value cannot become a filename either;
// the backfill types whatever comes through as null.
postType: PostTypes.Normalize(reader.IsDBNull(21) ? null : reader.GetString(21)),
hasImage: hasImage);
postsUpserted++;
if (postsUpserted % 500 == 0)
Console.WriteLine($" ... {postsUpserted} posts upserted");
}
catch (Exception ex)
{
errors++;
if (errors < 20)
Console.WriteLine($" Post upsert error: {ex.Message}");
}
}
}
finally
{
DataAccess.EndImportSession();
DataAccess.RestoreImportModePragmas();
}
Console.WriteLine($" Posts upserted: {postsUpserted}");
}
catch (Exception ex)
{
Console.WriteLine($"Fatal error reading legacy posts.db: {ex.Message}");
return 1;
}
Console.WriteLine($"\n========== Legacy import summary ==========");
Console.WriteLine($"Blogs seen: {blogsCopied}");
Console.WriteLine($"Paths written: {blogPathsWritten} (legacy rows with no path: {blogsWithoutPath})");
Console.WriteLine($"Posts upserted: {postsUpserted}");
Console.WriteLine($"Errors: {errors}");
return errors == 0 ? 0 : 2;
}
}
}