feat: merge ThreeTxtFileHelper into URLNotesGrabberCORE

Folds the standalone ThreeTxtFileHelper tool into URLNotesGrabberCORE so
text-file ingest/output/correct lives alongside the API scraper. Adds
new flags -ingest, -output, -correct (with -apply), -updatepaths, and a
one-time -importposts <posts.db> migration.

Schema: Blogs.TTFolderPath and Posts.PostType are added by an idempotent
migration. On (BlogName, PostID) collisions, content columns are
overwritten while engagement columns (ByLikes, RootBlogName, RootURL,
HasNotesGathered, NotFound, NotesGatheredDateTime, Likes*) are preserved.

Co-Authored-By: Claude Opus 4.7 <[email protected]>
This commit is contained in:
jim
2026-05-18 12:43:17 -05:00
co-authored by Claude Opus 4.7
parent 5781e121d2
commit 3aff849216
10 changed files with 1419 additions and 1 deletions
+199
View File
@@ -0,0 +1,199 @@
using System.Text.RegularExpressions;
using Microsoft.Extensions.Configuration;
namespace URLNotesGrabberCORE
{
// Port of ThreeTxtFileHelper RunIngestMode. Scans a root folder for .txt files,
// parses Tumblr-export fields (multi-line aware, prefix-driven), upserts each
// post into TL.db.Posts via DataAccess.UpsertPostFromTextFile.
public static class IngestMode
{
public static int Run(IConfiguration config, string[] args)
{
string? rootPath = args.Length > 0 ? args[0] : config["appSettings:PathTTRoot"];
if (string.IsNullOrWhiteSpace(rootPath))
rootPath = config["appSettings:PathInput"];
if (string.IsNullOrWhiteSpace(rootPath))
{
Console.WriteLine("Ingest: no root path provided. Set appSettings:PathTTRoot, appSettings:PathInput, or pass a path after -ingest.");
return 1;
}
if (!Directory.Exists(rootPath))
{
Console.WriteLine($"Directory not found: {rootPath}");
return 1;
}
DataAccess.EnsureTTFileHelperColumnsExist();
string prefixesPath = config["appSettings:PathPrefixes"] ?? "prefixes.txt";
if (!Path.IsPathRooted(prefixesPath))
prefixesPath = Path.Combine(AppContext.BaseDirectory, prefixesPath);
if (!File.Exists(prefixesPath))
{
Console.WriteLine($"Prefixes file not found: {prefixesPath}");
return 1;
}
var allowedPrefixes = new HashSet<string>(File.ReadLines(prefixesPath), StringComparer.OrdinalIgnoreCase);
Console.WriteLine($"Loaded {allowedPrefixes.Count} prefixes from {prefixesPath}");
Console.WriteLine($"========== Ingest Settings ==========");
Console.WriteLine($"Root path: {rootPath}");
Console.WriteLine($"=====================================");
var txtFiles = new List<string>();
try
{
var dirs = Directory.GetDirectories(rootPath, "*", SearchOption.AllDirectories);
Console.WriteLine($"Found {dirs.Length} directories under root.");
foreach (var dir in dirs)
{
try { txtFiles.AddRange(Directory.GetFiles(dir, "*.txt")); }
catch (Exception ex) { Console.WriteLine($" Skipping {dir}: {ex.Message}"); }
}
txtFiles.AddRange(Directory.GetFiles(rootPath, "*.txt"));
}
catch (Exception ex)
{
Console.WriteLine($"Error scanning root: {ex.Message}");
return 1;
}
Console.WriteLine($"Processing {txtFiles.Count} .txt file(s)...");
int filesProcessed = 0;
int postsTouched = 0;
try
{
DataAccess.EnableImportModePragmas();
DataAccess.BeginImportSession();
foreach (string file in txtFiles)
{
try
{
filesProcessed++;
if (filesProcessed % 50 == 0 || filesProcessed == 1)
Console.WriteLine($"[{filesProcessed}/{txtFiles.Count}] {Path.GetFileName(file)}");
string rawBlogName = Path.GetFileName(Path.GetDirectoryName(file) ?? "unknown");
string blogName = Regex.Replace(rawBlogName, @"_\d+$", "");
string postType = Path.GetFileNameWithoutExtension(file);
string currentPostId = "";
var currentPostData = new Dictionary<string, string>(StringComparer.OrdinalIgnoreCase);
void Flush()
{
if (!string.IsNullOrWhiteSpace(currentPostId) && currentPostData.Count > 0)
{
UpsertPostFromParsedData(blogName, currentPostId, postType, currentPostData);
postsTouched++;
}
}
var lines = File.ReadAllLines(file);
int lineIndex = 0;
while (lineIndex < lines.Length)
{
string line = lines[lineIndex];
string searchText = line.Length > 25 ? line.Substring(0, 25) : line;
int colonIndex = searchText.IndexOf(": ");
if (colonIndex > 0)
{
string prefix = line.Substring(0, colonIndex).Trim();
if (!string.IsNullOrWhiteSpace(prefix) && allowedPrefixes.Contains(prefix))
{
if (string.Equals(prefix, "Post ID", StringComparison.OrdinalIgnoreCase))
{
Flush();
currentPostId = line.Substring(colonIndex + 2).Trim();
currentPostData.Clear();
lineIndex++;
continue;
}
var valueLines = new List<string> { line.Substring(colonIndex + 2).Trim() };
int nextLineIndex = lineIndex + 1;
while (nextLineIndex < lines.Length)
{
string nextLine = lines[nextLineIndex];
string nextSearch = nextLine.Length > 25 ? nextLine.Substring(0, 25) : nextLine;
int nextColon = nextSearch.IndexOf(": ");
if (nextColon > 0)
{
string nextPrefix = nextLine.Substring(0, nextColon).Trim();
if (!string.IsNullOrWhiteSpace(nextPrefix) && allowedPrefixes.Contains(nextPrefix))
break;
}
valueLines.Add(nextLine);
nextLineIndex++;
}
currentPostData[prefix] = string.Join("\n", valueLines);
lineIndex = nextLineIndex;
continue;
}
}
lineIndex++;
}
Flush();
}
catch (Exception ex)
{
Console.WriteLine($" ERROR processing file {file}: {ex.Message}");
}
}
}
finally
{
DataAccess.EndImportSession();
DataAccess.RestoreImportModePragmas();
}
Console.WriteLine($"\nIngest complete. Files processed: {filesProcessed}. Posts touched: {postsTouched}.");
return 0;
}
private static void UpsertPostFromParsedData(string blogName, string postId, string postType, Dictionary<string, string> data)
{
string? G(string key) => data.TryGetValue(key, out var v) ? v : null;
string? hasImageStr = G("Has Image");
bool hasImage = !string.IsNullOrWhiteSpace(hasImageStr)
&& (hasImageStr.Equals("true", StringComparison.OrdinalIgnoreCase)
|| hasImageStr == "1"
|| hasImageStr.Equals("yes", StringComparison.OrdinalIgnoreCase));
DataAccess.UpsertPostFromTextFile(
blogName: blogName,
postID: postId,
reblogURL: G("reblog URL"),
postDate: G("Date"),
postURL: G("Post URL"),
slug: G("Slug"),
reblogKey: G("Reblog Key"),
reblogName: G("Reblog Name"),
summary: G("Summary"),
quote: G("Quote"),
body: G("Body"),
tags: G("Tags"),
link: G("Link"),
photoURL: G("Photo URL"),
photoCaption: G("Photo Caption"),
downloadedFiles: G("Downloaded Files"),
audioCaption: G("Audio Caption"),
question: G("Question"),
answer: G("Answer"),
title: G("Title"),
postType: postType,
hasImage: hasImage);
}
}
}