PostType becomes an output filename, so an unset or unvalidated value does
not stay a data problem -- it creates a file. OutputMode wrote untyped rows
to `PostType ?? "Unknown"`, IngestMode read that file back and derived the
literal type "Unknown" from its name, and the two would have regenerated
each other indefinitely.
Nothing was setting the type in the first place. AddPost -- the path every
notes/likes harvest goes through -- omitted PostType from its INSERT column
list entirely, so 1790 rows across 469 blogs had none. Ingest could never
repair them: it types a post only when it meets it inside a real export
file, and these posts appear in none.
Type at the source, from what each path actually knows:
- TraverseDirectory takes it from the filename it is already reading
("texts.txt" -> "texts"), the same rule ingest uses.
- CollectLikes has no file, so it reads the legacy-format `type` field that
GrabLikes already requests with npf=false, mapped singular -> plural.
- AddPost/UpdatePost gained the plumbing to carry it. UpdatePost fills a
missing type but never overwrites one, and its change-detection clause
had to learn about PostType or the SET would be unreachable for a row
whose content was already current.
PostTypes is the single source of truth: eight canonical names, and
anything else normalizes to null. Null is safe -- OutputMode skips those
rows -- while a stray value would have become a stray file. IngestMode and
TraverseDirectory now skip non-export .txt files outright, and the legacy
importer no longer passes a pre-column NULL straight back in.
For the rows already stored untyped, content is the only signal left, so
the migration infers from which columns they carry. Verified against a copy
of the live database: 1780 of 1790 typed, 10 left untyped for want of any
content at all, idempotent on a second pass. HasImage is deliberately not
consulted -- it is set on 12,420 of 19,828 known text posts.
Co-Authored-By: Claude Opus 5 <[email protected]>
238 lines
11 KiB
C#
238 lines
11 KiB
C#
using System.Text.RegularExpressions;
|
|
using Microsoft.Extensions.Configuration;
|
|
|
|
namespace URLNotesGrabberCORE
|
|
{
|
|
// Port of ThreeTxtFileHelper RunIngestMode. Scans a root folder for .txt files,
|
|
// parses Tumblr-export fields (multi-line aware, prefix-driven), upserts each
|
|
// post into TL.db.Posts via DataAccess.UpsertPostFromTextFile.
|
|
public static class IngestMode
|
|
{
|
|
public static int Run(IConfiguration config, string[] args)
|
|
{
|
|
string? targetBlog = args.Length > 0 ? args[0]?.Trim() : null;
|
|
if (string.IsNullOrWhiteSpace(targetBlog)) targetBlog = null;
|
|
|
|
string? rootPath = config["appSettings:PathTTRoot"];
|
|
if (string.IsNullOrWhiteSpace(rootPath))
|
|
rootPath = config["appSettings:PathInput"];
|
|
|
|
if (string.IsNullOrWhiteSpace(rootPath))
|
|
{
|
|
Console.WriteLine("Ingest: no root path configured. Set appSettings:PathTTRoot or appSettings:PathInput.");
|
|
return 1;
|
|
}
|
|
|
|
if (!Directory.Exists(rootPath))
|
|
{
|
|
Console.WriteLine($"Directory not found: {rootPath}");
|
|
return 1;
|
|
}
|
|
|
|
DataAccess.EnsureTTFileHelperColumnsExist();
|
|
|
|
string prefixesPath = config["appSettings:PathPrefixes"] ?? "prefixes.txt";
|
|
if (!Path.IsPathRooted(prefixesPath))
|
|
prefixesPath = Path.Combine(AppContext.BaseDirectory, prefixesPath);
|
|
|
|
if (!File.Exists(prefixesPath))
|
|
{
|
|
Console.WriteLine($"Prefixes file not found: {prefixesPath}");
|
|
return 1;
|
|
}
|
|
|
|
var allowedPrefixes = new HashSet<string>(File.ReadLines(prefixesPath), StringComparer.OrdinalIgnoreCase);
|
|
Console.WriteLine($"Loaded {allowedPrefixes.Count} prefixes from {prefixesPath}");
|
|
|
|
Console.WriteLine($"========== Ingest Settings ==========");
|
|
Console.WriteLine($"Root path: {rootPath}");
|
|
Console.WriteLine($"Blog filter: {(targetBlog == null ? "(all blogs)" : targetBlog)}");
|
|
Console.WriteLine($"=====================================");
|
|
|
|
var txtFiles = new List<string>();
|
|
try
|
|
{
|
|
var dirs = Directory.GetDirectories(rootPath, "*", SearchOption.AllDirectories);
|
|
Console.WriteLine($"Found {dirs.Length} directories under root.");
|
|
foreach (var dir in dirs)
|
|
{
|
|
try { txtFiles.AddRange(Directory.GetFiles(dir, "*.txt")); }
|
|
catch (Exception ex) { Console.WriteLine($" Skipping {dir}: {ex.Message}"); }
|
|
}
|
|
txtFiles.AddRange(Directory.GetFiles(rootPath, "*.txt"));
|
|
}
|
|
catch (Exception ex)
|
|
{
|
|
Console.WriteLine($"Error scanning root: {ex.Message}");
|
|
return 1;
|
|
}
|
|
|
|
Console.WriteLine($"Processing {txtFiles.Count} .txt file(s)...");
|
|
|
|
int filesProcessed = 0;
|
|
int filesSkipped = 0;
|
|
int postsTouched = 0;
|
|
string? lastBlogFolder = null;
|
|
|
|
try
|
|
{
|
|
DataAccess.EnableImportModePragmas();
|
|
DataAccess.BeginImportSession();
|
|
|
|
foreach (string file in txtFiles)
|
|
{
|
|
try
|
|
{
|
|
string rawBlogName = Path.GetFileName(Path.GetDirectoryName(file) ?? "unknown");
|
|
string blogName = Regex.Replace(rawBlogName, @"_\d+$", "");
|
|
|
|
// The filename becomes the row's PostType, and PostType later becomes an
|
|
// output filename -- so an unrecognized name here would mint a new type and
|
|
// a new file from any stray .txt that happens to sit in the tree. Only the
|
|
// eight real export files are ingestable.
|
|
//
|
|
// This is also what breaks the Unknown.txt cycle: OutputMode used to write
|
|
// untyped rows to Unknown.txt, and this scan would read it straight back
|
|
// and stamp those rows with the literal type "Unknown", making the file
|
|
// regenerate itself forever.
|
|
string? resolvedPostType = PostTypes.FromFileName(file);
|
|
if (resolvedPostType == null)
|
|
{
|
|
filesSkipped++;
|
|
continue;
|
|
}
|
|
// Non-nullable from here so the local Flush() below stays warning-clean:
|
|
// nullable flow analysis does not reach into local functions.
|
|
string postType = resolvedPostType;
|
|
|
|
if (targetBlog != null && !string.Equals(blogName, targetBlog, StringComparison.OrdinalIgnoreCase))
|
|
{
|
|
filesSkipped++;
|
|
continue;
|
|
}
|
|
|
|
filesProcessed++;
|
|
|
|
if (lastBlogFolder != rawBlogName)
|
|
{
|
|
Console.WriteLine($"[{filesProcessed}/{txtFiles.Count}] >> entering folder: {rawBlogName}");
|
|
lastBlogFolder = rawBlogName;
|
|
}
|
|
else if (filesProcessed % 50 == 0)
|
|
{
|
|
Console.WriteLine($"[{filesProcessed}/{txtFiles.Count}] {rawBlogName}/{Path.GetFileName(file)}");
|
|
}
|
|
|
|
string currentPostId = "";
|
|
var currentPostData = new Dictionary<string, string>(StringComparer.OrdinalIgnoreCase);
|
|
|
|
void Flush()
|
|
{
|
|
if (!string.IsNullOrWhiteSpace(currentPostId) && currentPostData.Count > 0)
|
|
{
|
|
UpsertPostFromParsedData(blogName, currentPostId, postType, currentPostData);
|
|
postsTouched++;
|
|
}
|
|
}
|
|
|
|
var lines = File.ReadAllLines(file);
|
|
int lineIndex = 0;
|
|
while (lineIndex < lines.Length)
|
|
{
|
|
string line = lines[lineIndex];
|
|
string searchText = line.Length > 25 ? line.Substring(0, 25) : line;
|
|
int colonIndex = searchText.IndexOf(": ");
|
|
|
|
if (colonIndex > 0)
|
|
{
|
|
string prefix = line.Substring(0, colonIndex).Trim();
|
|
if (!string.IsNullOrWhiteSpace(prefix) && allowedPrefixes.Contains(prefix))
|
|
{
|
|
if (string.Equals(prefix, "Post ID", StringComparison.OrdinalIgnoreCase))
|
|
{
|
|
Flush();
|
|
currentPostId = line.Substring(colonIndex + 2).Trim();
|
|
currentPostData.Clear();
|
|
lineIndex++;
|
|
continue;
|
|
}
|
|
|
|
var valueLines = new List<string> { line.Substring(colonIndex + 2).Trim() };
|
|
int nextLineIndex = lineIndex + 1;
|
|
while (nextLineIndex < lines.Length)
|
|
{
|
|
string nextLine = lines[nextLineIndex];
|
|
string nextSearch = nextLine.Length > 25 ? nextLine.Substring(0, 25) : nextLine;
|
|
int nextColon = nextSearch.IndexOf(": ");
|
|
if (nextColon > 0)
|
|
{
|
|
string nextPrefix = nextLine.Substring(0, nextColon).Trim();
|
|
if (!string.IsNullOrWhiteSpace(nextPrefix) && allowedPrefixes.Contains(nextPrefix))
|
|
break;
|
|
}
|
|
valueLines.Add(nextLine);
|
|
nextLineIndex++;
|
|
}
|
|
|
|
currentPostData[prefix] = string.Join("\n", valueLines);
|
|
lineIndex = nextLineIndex;
|
|
continue;
|
|
}
|
|
}
|
|
lineIndex++;
|
|
}
|
|
|
|
Flush();
|
|
}
|
|
catch (Exception ex)
|
|
{
|
|
Console.WriteLine($" ERROR processing file {file}: {ex.Message}");
|
|
}
|
|
}
|
|
}
|
|
finally
|
|
{
|
|
DataAccess.EndImportSession();
|
|
DataAccess.RestoreImportModePragmas();
|
|
}
|
|
|
|
Console.WriteLine($"\nIngest complete. Files processed: {filesProcessed}. Files skipped (blog filter): {filesSkipped}. Posts touched: {postsTouched}.");
|
|
return 0;
|
|
}
|
|
|
|
private static void UpsertPostFromParsedData(string blogName, string postId, string postType, Dictionary<string, string> data)
|
|
{
|
|
string? G(string key) => data.TryGetValue(key, out var v) ? v : null;
|
|
string? hasImageStr = G("Has Image");
|
|
bool hasImage = !string.IsNullOrWhiteSpace(hasImageStr)
|
|
&& (hasImageStr.Equals("true", StringComparison.OrdinalIgnoreCase)
|
|
|| hasImageStr == "1"
|
|
|| hasImageStr.Equals("yes", StringComparison.OrdinalIgnoreCase));
|
|
|
|
DataAccess.UpsertPostFromTextFile(
|
|
blogName: blogName,
|
|
postID: postId,
|
|
reblogURL: G("reblog URL"),
|
|
postDate: G("Date"),
|
|
postURL: G("Post URL"),
|
|
slug: G("Slug"),
|
|
reblogKey: G("Reblog Key"),
|
|
reblogName: G("Reblog Name"),
|
|
summary: G("Summary"),
|
|
quote: G("Quote"),
|
|
body: G("Body"),
|
|
tags: G("Tags"),
|
|
link: G("Link"),
|
|
photoURL: G("Photo URL"),
|
|
photoCaption: G("Photo Caption"),
|
|
downloadedFiles: G("Downloaded Files"),
|
|
audioCaption: G("Audio Caption"),
|
|
question: G("Question"),
|
|
answer: G("Answer"),
|
|
title: G("Title"),
|
|
postType: postType,
|
|
hasImage: hasImage);
|
|
}
|
|
}
|
|
}
|