Compare commits
2
Commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
e6a3efba5b | ||
|
|
34da632e6a |
@@ -590,12 +590,16 @@ namespace URLNotesGrabberCORE
|
||||
}
|
||||
}
|
||||
|
||||
public static void AddPost(string blogName, long postID, string reblogURL, string postDate, string postURL, string slug, string reblogKey, string reblogName, string summary, string quote, string body, string tags, string link, string photoURL, string photoCaption, string downloadedFiles, string audioCaption, string question, string answer, string title, bool hasImage, bool byLikes = false, string? DBPath = null, string? rootBlogName = null, string? rootURL = null)
|
||||
// postType: a canonical PostTypes name, or null when the caller has no trustworthy type.
|
||||
// Null is stored as NULL rather than guessed at -- OutputMode skips untyped rows, so a
|
||||
// null costs one export line, whereas a wrong value would create a wrongly named file.
|
||||
public static void AddPost(string blogName, long postID, string reblogURL, string postDate, string postURL, string slug, string reblogKey, string reblogName, string summary, string quote, string body, string tags, string link, string photoURL, string photoCaption, string downloadedFiles, string audioCaption, string question, string answer, string title, bool hasImage, bool byLikes = false, string? DBPath = null, string? rootBlogName = null, string? rootURL = null, string? postType = null)
|
||||
{
|
||||
DBPath ??= GetDefaultDbPath();
|
||||
postType = PostTypes.Normalize(postType);
|
||||
try { AddBlog(blogName, byLikes, DBPath); } catch { }
|
||||
try { UpdatePostSetDate(blogName, postID, postDate, DBPath); } catch { }
|
||||
try { UpdatePost(blogName, postID, reblogURL, postDate, postURL, slug, reblogKey, reblogName, summary, quote, body, tags, link, photoURL, photoCaption, downloadedFiles, audioCaption, question, answer, title, hasImage, byLikes, DBPath, rootBlogName, rootURL); } catch { }
|
||||
try { UpdatePost(blogName, postID, reblogURL, postDate, postURL, slug, reblogKey, reblogName, summary, quote, body, tags, link, photoURL, photoCaption, downloadedFiles, audioCaption, question, answer, title, hasImage, byLikes, DBPath, rootBlogName, rootURL, postType); } catch { }
|
||||
|
||||
SQLiteConnection connection;
|
||||
bool ownsConnection;
|
||||
@@ -637,9 +641,10 @@ namespace URLNotesGrabberCORE
|
||||
RootBlogName,
|
||||
RootURL,
|
||||
HasImage,
|
||||
ByLikes
|
||||
ByLikes,
|
||||
PostType
|
||||
) VALUES (" +
|
||||
Q(blogName) + ", " + postID + ", " + Q(reblogURL) + ", " + Q(postDate) + ", " + Q(postURL) + ", " + Q(slug) + ", " + Q(reblogKey) + ", " + Q(reblogName) + ", " + Q(summary) + ", " + Q(quote) + ", " + Q(body) + ", " + Q(tags) + ", " + Q(link) + ", " + Q(photoURL) + ", " + Q(photoCaption) + ", " + Q(downloadedFiles) + ", " + Q(audioCaption) + ", " + Q(question) + ", " + Q(answer) + ", " + Q(title) + ", " + Q(DateTime.Now.ToString("yyyy-MM-dd HH:mm:ss")) + ", " + Q(DateTime.Now.ToString("yyyy-MM-dd HH:mm:ss")) + ", " + Q(rootBlogName ?? ".") + ", " + Q(rootURL ?? ".") + ", " + (hasImage ? 1 : 0) + ", " + (byLikes ? 1 : 0) + ")";
|
||||
Q(blogName) + ", " + postID + ", " + Q(reblogURL) + ", " + Q(postDate) + ", " + Q(postURL) + ", " + Q(slug) + ", " + Q(reblogKey) + ", " + Q(reblogName) + ", " + Q(summary) + ", " + Q(quote) + ", " + Q(body) + ", " + Q(tags) + ", " + Q(link) + ", " + Q(photoURL) + ", " + Q(photoCaption) + ", " + Q(downloadedFiles) + ", " + Q(audioCaption) + ", " + Q(question) + ", " + Q(answer) + ", " + Q(title) + ", " + Q(DateTime.Now.ToString("yyyy-MM-dd HH:mm:ss")) + ", " + Q(DateTime.Now.ToString("yyyy-MM-dd HH:mm:ss")) + ", " + Q(rootBlogName ?? ".") + ", " + Q(rootURL ?? ".") + ", " + (hasImage ? 1 : 0) + ", " + (byLikes ? 1 : 0) + ", " + (postType == null ? "NULL" : Q(postType)) + ")";
|
||||
SQLiteCommand command = new SQLiteCommand(sql, connection);
|
||||
|
||||
int rowsInserted = 0;
|
||||
@@ -1679,9 +1684,10 @@ namespace URLNotesGrabberCORE
|
||||
return false;
|
||||
}
|
||||
|
||||
public static void UpdatePost(string blogName, long postID, string reblogURL, string postDate, string postURL, string slug, string reblogKey, string reblogName, string summary, string quote, string body, string tags, string link, string photoURL, string photoCaption, string downloadedFiles, string audioCaption, string question, string answer, string title, bool hasImage, bool byLikes = false, string? DBPath = null, string? rootBlogName = null, string? rootURL = null)
|
||||
public static void UpdatePost(string blogName, long postID, string reblogURL, string postDate, string postURL, string slug, string reblogKey, string reblogName, string summary, string quote, string body, string tags, string link, string photoURL, string photoCaption, string downloadedFiles, string audioCaption, string question, string answer, string title, bool hasImage, bool byLikes = false, string? DBPath = null, string? rootBlogName = null, string? rootURL = null, string? postType = null)
|
||||
{
|
||||
DBPath ??= GetDefaultDbPath();
|
||||
postType = PostTypes.Normalize(postType);
|
||||
|
||||
SQLiteConnection connection;
|
||||
bool ownsConnection;
|
||||
@@ -1729,6 +1735,10 @@ namespace URLNotesGrabberCORE
|
||||
sql += "RootBlogName = CASE WHEN @rootBlogName IS NULL OR @rootBlogName = '' OR @rootBlogName = '.' THEN RootBlogName ELSE @rootBlogName END, ";
|
||||
sql += "RootURL = CASE WHEN @rootURL IS NULL OR @rootURL = '' OR @rootURL = '.' THEN RootURL ELSE @rootURL END, ";
|
||||
sql += "hasImage = @hasImage, ";
|
||||
// Fill in a missing type, never overwrite one. A type derived by --ingest from a
|
||||
// real export filename is authoritative; this path's type is only as good as the
|
||||
// folder it was crawled from, so it must not win over an existing value.
|
||||
sql += "PostType = IFNULL(PostType, @postType), ";
|
||||
sql += "ByLikes = MAX(IFNULL(ByLikes, 0), @byLikes) ";
|
||||
sql += " WHERE BlogName = @BlogName AND PostID = @PostID AND (";
|
||||
sql += "(@postDate <> '.' AND IFNULL(postDate, '') <> @postDate) OR ";
|
||||
@@ -1752,7 +1762,10 @@ namespace URLNotesGrabberCORE
|
||||
sql += "IFNULL(hasImage, 0) <> @hasImage OR ";
|
||||
sql += "(@byLikes = 1 AND IFNULL(ByLikes, 0) = 0) OR ";
|
||||
sql += "((@rootBlogName IS NOT NULL AND @rootBlogName <> '' AND @rootBlogName <> '.') AND IFNULL(RootBlogName, '') <> @rootBlogName) OR ";
|
||||
sql += "((@rootURL IS NOT NULL AND @rootURL <> '' AND @rootURL <> '.') AND IFNULL(RootURL, '') <> @rootURL)";
|
||||
sql += "((@rootURL IS NOT NULL AND @rootURL <> '' AND @rootURL <> '.') AND IFNULL(RootURL, '') <> @rootURL) OR ";
|
||||
// Without this the SET above is unreachable for a row whose content is already
|
||||
// current: the UPDATE would not fire, and the type would stay NULL forever.
|
||||
sql += "(PostType IS NULL AND @postType IS NOT NULL)";
|
||||
sql += ")";
|
||||
|
||||
using (SQLiteCommand command = new SQLiteCommand(sql, connection))
|
||||
@@ -1780,6 +1793,7 @@ namespace URLNotesGrabberCORE
|
||||
command.Parameters.AddWithValue("@rootURL", string.IsNullOrWhiteSpace(rootURL) ? (object)DBNull.Value : rootURL);
|
||||
command.Parameters.AddWithValue("@hasImage", hasImage ? 1 : 0);
|
||||
command.Parameters.AddWithValue("@byLikes", byLikes ? 1 : 0);
|
||||
command.Parameters.AddWithValue("@postType", (object?)postType ?? DBNull.Value);
|
||||
command.Parameters.AddWithValue("@BlogName", blogName);
|
||||
command.Parameters.AddWithValue("@PostID", postID);
|
||||
|
||||
@@ -2089,6 +2103,8 @@ namespace URLNotesGrabberCORE
|
||||
cmd.ExecuteNonQuery();
|
||||
Console.WriteLine("[Migration] Added PostType column to Posts table");
|
||||
}
|
||||
|
||||
BackfillMissingPostTypes(connection);
|
||||
}
|
||||
catch (Exception ex)
|
||||
{
|
||||
@@ -2096,6 +2112,65 @@ namespace URLNotesGrabberCORE
|
||||
}
|
||||
}
|
||||
|
||||
/// <summary>
|
||||
/// Types rows that carry no PostType, inferring it from which content columns they hold.
|
||||
///
|
||||
/// These are posts harvested from notes and likes rather than read out of a TumblThree
|
||||
/// export, so no filename ever described them and --ingest can never reach them: it only
|
||||
/// types a post it meets inside a real .txt. Content is the only signal they have.
|
||||
///
|
||||
/// Runs on every migration pass and is idempotent -- it only touches PostType IS NULL,
|
||||
/// so a row typed once is never revisited. Rows whose columns give no signal at all stay
|
||||
/// NULL and are skipped by OutputMode.
|
||||
///
|
||||
/// Mirrors PostTypes.InferFromContent; the two must agree. Notably HasImage is not
|
||||
/// consulted, because most text posts carry it.
|
||||
/// </summary>
|
||||
private static void BackfillMissingPostTypes(SQLiteConnection connection)
|
||||
{
|
||||
const string set = @"
|
||||
UPDATE Posts SET PostType = CASE
|
||||
WHEN Has(Question) AND Has(Answer) THEN 'answers'
|
||||
WHEN Has(Quote) THEN 'quotes'
|
||||
WHEN Has(Link) THEN 'links'
|
||||
WHEN Has(AudioCaption) THEN 'audios'
|
||||
WHEN Has(Body) THEN 'texts'
|
||||
WHEN Has(PhotoURL) OR Has(PhotoCaption) THEN 'images'
|
||||
ELSE NULL END
|
||||
WHERE PostType IS NULL";
|
||||
|
||||
// SQLite has no user-defined predicate here, so expand the "field supplied" test
|
||||
// ("." is the not-supplied sentinel used throughout the export format) inline.
|
||||
string sql = System.Text.RegularExpressions.Regex.Replace(
|
||||
set, @"Has\((\w+)\)", "TRIM(IFNULL($1, '')) NOT IN ('', '.')");
|
||||
|
||||
try
|
||||
{
|
||||
long before;
|
||||
using (var count = new SQLiteCommand("SELECT COUNT(*) FROM Posts WHERE PostType IS NULL", connection))
|
||||
before = Convert.ToInt64(count.ExecuteScalar());
|
||||
|
||||
if (before == 0) return;
|
||||
|
||||
int changed;
|
||||
using (var cmd = new SQLiteCommand(sql, connection))
|
||||
changed = cmd.ExecuteNonQuery();
|
||||
|
||||
long after;
|
||||
using (var count = new SQLiteCommand("SELECT COUNT(*) FROM Posts WHERE PostType IS NULL", connection))
|
||||
after = Convert.ToInt64(count.ExecuteScalar());
|
||||
|
||||
if (changed > 0 || after != before)
|
||||
Console.WriteLine($"[Migration] Backfilled PostType for {before - after} post(s); {after} still untyped (no content signal).");
|
||||
}
|
||||
catch (Exception ex)
|
||||
{
|
||||
// A failed backfill must not stop the run: untyped rows are skipped on export,
|
||||
// which is inconvenient, not corrupting.
|
||||
Console.WriteLine($"[Migration] PostType backfill failed: {ex.Message}");
|
||||
}
|
||||
}
|
||||
|
||||
// INSERT-or-UPDATE for a post arriving from a Tumblr text-file export.
|
||||
// On collision, only content columns + PostType + DateModified are updated;
|
||||
// engagement columns (ByLikes, RootBlogName, RootURL, HasNotesGathered, NotFound,
|
||||
@@ -2126,6 +2201,10 @@ namespace URLNotesGrabberCORE
|
||||
string? DBPath = null)
|
||||
{
|
||||
DBPath ??= GetDefaultDbPath();
|
||||
// Central guarantee: whatever a caller believes, only a canonical type reaches the
|
||||
// column. PostType is used as an output filename, so this is the invariant that keeps
|
||||
// a stray value from becoming a stray file.
|
||||
postType = PostTypes.Normalize(postType);
|
||||
try { AddBlog(blogName, false, DBPath); } catch { }
|
||||
|
||||
SQLiteConnection connection;
|
||||
|
||||
@@ -85,7 +85,25 @@ namespace URLNotesGrabberCORE
|
||||
{
|
||||
string rawBlogName = Path.GetFileName(Path.GetDirectoryName(file) ?? "unknown");
|
||||
string blogName = Regex.Replace(rawBlogName, @"_\d+$", "");
|
||||
string postType = Path.GetFileNameWithoutExtension(file);
|
||||
|
||||
// The filename becomes the row's PostType, and PostType later becomes an
|
||||
// output filename -- so an unrecognized name here would mint a new type and
|
||||
// a new file from any stray .txt that happens to sit in the tree. Only the
|
||||
// eight real export files are ingestable.
|
||||
//
|
||||
// This is also what breaks the Unknown.txt cycle: OutputMode used to write
|
||||
// untyped rows to Unknown.txt, and this scan would read it straight back
|
||||
// and stamp those rows with the literal type "Unknown", making the file
|
||||
// regenerate itself forever.
|
||||
string? resolvedPostType = PostTypes.FromFileName(file);
|
||||
if (resolvedPostType == null)
|
||||
{
|
||||
filesSkipped++;
|
||||
continue;
|
||||
}
|
||||
// Non-nullable from here so the local Flush() below stays warning-clean:
|
||||
// nullable flow analysis does not reach into local functions.
|
||||
string postType = resolvedPostType;
|
||||
|
||||
if (targetBlog != null && !string.Equals(blogName, targetBlog, StringComparison.OrdinalIgnoreCase))
|
||||
{
|
||||
|
||||
@@ -117,7 +117,12 @@ namespace URLNotesGrabberCORE
|
||||
question: reader.IsDBNull(18) ? null : reader.GetString(18),
|
||||
answer: reader.IsDBNull(19) ? null : reader.GetString(19),
|
||||
title: reader.IsDBNull(20) ? null : reader.GetString(20),
|
||||
postType: reader.IsDBNull(21) ? null : reader.GetString(21),
|
||||
// A legacy Posts.db predating the PostType column hands back NULL
|
||||
// here, and on the INSERT branch that NULL is stored -- reseeding
|
||||
// exactly the untyped rows the backfill exists to clear. Normalize
|
||||
// so an unrecognized legacy value cannot become a filename either;
|
||||
// the backfill types whatever comes through as null.
|
||||
postType: PostTypes.Normalize(reader.IsDBNull(21) ? null : reader.GetString(21)),
|
||||
hasImage: hasImage);
|
||||
postsUpserted++;
|
||||
if (postsUpserted % 500 == 0)
|
||||
|
||||
@@ -65,10 +65,20 @@ namespace URLNotesGrabberCORE
|
||||
var posts = DataAccess.GetAllPostsForBlog(blogName);
|
||||
Console.WriteLine($" Found {posts.Count} post(s) for this blog.");
|
||||
|
||||
var grouped = posts.GroupBy(p => p.PostType ?? "Unknown");
|
||||
// A post's type becomes a filename, so only a recognized type may be written. The
|
||||
// old `PostType ?? "Unknown"` invented Unknown.txt for untyped rows, which --ingest
|
||||
// then read back as a type named "Unknown" -- the two regenerated each other.
|
||||
// Untyped rows are skipped instead: after the backfill these are only rows with no
|
||||
// content signal at all, so nothing meaningful is lost, and nothing is invented.
|
||||
var typed = posts.Where(p => PostTypes.Normalize(p.PostType) != null).ToList();
|
||||
int untyped = posts.Count - typed.Count;
|
||||
if (untyped > 0)
|
||||
Console.WriteLine($" Skipping {untyped} post(s) with no recognized PostType.");
|
||||
|
||||
var grouped = typed.GroupBy(p => PostTypes.Normalize(p.PostType)!);
|
||||
foreach (var typeGroup in grouped)
|
||||
{
|
||||
string postType = typeGroup.Key ?? "Unknown";
|
||||
string postType = typeGroup.Key;
|
||||
string outputFilePath = Path.Combine(folder, $"{postType}.txt");
|
||||
var ordered = typeGroup.OrderBy(p => p.Date).ToList();
|
||||
Console.WriteLine($" Writing {ordered.Count} post(s) to {postType}.txt");
|
||||
|
||||
@@ -0,0 +1,123 @@
|
||||
using System;
|
||||
using System.Collections.Generic;
|
||||
using System.IO;
|
||||
|
||||
namespace URLNotesGrabberCORE
|
||||
{
|
||||
/// <summary>
|
||||
/// The single source of truth for Posts.PostType values.
|
||||
///
|
||||
/// PostType exists so --output can write one .txt per type. Because the type becomes a
|
||||
/// *filename*, an unvalidated value is not a cosmetic problem: it creates a file. That is
|
||||
/// how "Unknown.txt" came about -- OutputMode used `PostType ?? "Unknown"` as a filename,
|
||||
/// --ingest then read that file straight back and derived the literal type "Unknown" from
|
||||
/// its name, and the pair would have kept regenerating each other indefinitely.
|
||||
///
|
||||
/// So every path that produces a type routes through <see cref="Normalize"/>, which admits
|
||||
/// only the eight known names and returns null for anything else. A null type is safe:
|
||||
/// OutputMode skips those rows rather than inventing a file for them.
|
||||
/// </summary>
|
||||
public static class PostTypes
|
||||
{
|
||||
// The canonical set. These are exactly the TumblThree .txt basenames, which is what
|
||||
// makes an ingested filename usable as a type without translation.
|
||||
public const string Texts = "texts";
|
||||
public const string Answers = "answers";
|
||||
public const string Quotes = "quotes";
|
||||
public const string Links = "links";
|
||||
public const string Conversations = "conversations";
|
||||
public const string Images = "images";
|
||||
public const string Videos = "videos";
|
||||
public const string Audios = "audios";
|
||||
|
||||
private static readonly HashSet<string> Known = new HashSet<string>(
|
||||
new[] { Texts, Answers, Quotes, Links, Conversations, Images, Videos, Audios },
|
||||
StringComparer.OrdinalIgnoreCase);
|
||||
|
||||
// Tumblr's legacy post format (npf=false) names types in the singular. The likes API is
|
||||
// the one source that reports a type directly rather than via a filename, so it is the
|
||||
// only place this mapping is needed.
|
||||
private static readonly Dictionary<string, string> ApiTypeMap = new Dictionary<string, string>(StringComparer.OrdinalIgnoreCase)
|
||||
{
|
||||
["text"] = Texts,
|
||||
["photo"] = Images,
|
||||
["quote"] = Quotes,
|
||||
["link"] = Links,
|
||||
["chat"] = Conversations,
|
||||
["answer"] = Answers,
|
||||
["audio"] = Audios,
|
||||
["video"] = Videos,
|
||||
};
|
||||
|
||||
/// <summary>
|
||||
/// Returns the canonical type name, or null if the value is not one of the eight.
|
||||
/// Returning null rather than passing the value through is the whole point: an
|
||||
/// unrecognized string must never reach a filename.
|
||||
/// </summary>
|
||||
public static string? Normalize(string? candidate)
|
||||
{
|
||||
if (string.IsNullOrWhiteSpace(candidate)) return null;
|
||||
string trimmed = candidate.Trim();
|
||||
return Known.TryGetValue(trimmed, out string? canonical) ? canonical : null;
|
||||
}
|
||||
|
||||
/// <summary>
|
||||
/// Type for a post read out of a TumblThree export file, taken from the filename
|
||||
/// ("texts.txt" -> "texts"). Anything else in the folder -- README.txt, a stray
|
||||
/// triage file, or a previously written Unknown.txt -- normalizes to null and is
|
||||
/// rejected by the caller.
|
||||
/// </summary>
|
||||
public static string? FromFileName(string? path)
|
||||
{
|
||||
if (string.IsNullOrWhiteSpace(path)) return null;
|
||||
return Normalize(Path.GetFileNameWithoutExtension(path));
|
||||
}
|
||||
|
||||
/// <summary>
|
||||
/// Type for a post from the likes API, whose legacy-format `type` field is singular.
|
||||
/// Null when the field is absent or unrecognized -- the access is dynamic, so a missing
|
||||
/// field yields null at runtime rather than failing to compile.
|
||||
/// </summary>
|
||||
public static string? FromApiType(string? apiType)
|
||||
{
|
||||
if (string.IsNullOrWhiteSpace(apiType)) return null;
|
||||
return ApiTypeMap.TryGetValue(apiType.Trim(), out string? mapped) ? mapped : null;
|
||||
}
|
||||
|
||||
/// <summary>
|
||||
/// Last-resort type inferred from which content columns a row actually carries. Used
|
||||
/// only to backfill rows written before any type was recorded; a filename or an API
|
||||
/// type is always preferred over this.
|
||||
///
|
||||
/// The order matters and is derived from the already-typed rows, where the column
|
||||
/// signatures are effectively disjoint: answers carry Question+Answer and no Body,
|
||||
/// images carry photo columns and no Body, texts carry Body and no photo columns.
|
||||
///
|
||||
/// HasImage is deliberately NOT consulted: it is set on 12,420 of 19,828 known text
|
||||
/// posts, so it says nothing about the post's type.
|
||||
///
|
||||
/// conversations cannot be separated from texts this way -- both carry only Body -- so
|
||||
/// a chat post with no other signal is labelled texts. A later --ingest that meets the
|
||||
/// post in a real conversations.txt corrects it.
|
||||
/// </summary>
|
||||
public static string? InferFromContent(string? question, string? answer, string? quote,
|
||||
string? link, string? audioCaption, string? body, string? photoUrl, string? photoCaption)
|
||||
{
|
||||
if (HasValue(question) && HasValue(answer)) return Answers;
|
||||
if (HasValue(quote)) return Quotes;
|
||||
if (HasValue(link)) return Links;
|
||||
if (HasValue(audioCaption)) return Audios;
|
||||
if (HasValue(body)) return Texts;
|
||||
if (HasValue(photoUrl) || HasValue(photoCaption)) return Images;
|
||||
return null;
|
||||
}
|
||||
|
||||
// "." is the codebase-wide "field not supplied" sentinel in export records, so it
|
||||
// counts as absent here just as it does in UpdatePost's CASE guards.
|
||||
private static bool HasValue(string? value)
|
||||
{
|
||||
if (string.IsNullOrWhiteSpace(value)) return false;
|
||||
return value.Trim() != ".";
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -1047,6 +1047,15 @@ namespace URLNotesGrabberCORE
|
||||
string reblogKey = post.reblog_key?.ToString() ?? ".";
|
||||
string link = ".";
|
||||
|
||||
// No file backs a liked post, so the filename trick used everywhere
|
||||
// else cannot apply here. GrabLikes requests npf=false, and in the
|
||||
// legacy format `type` is the discriminator that decides which content
|
||||
// fields a post carries -- singular there, mapped to our plural names.
|
||||
// liked_posts is List<dynamic>, so this is resolved at runtime and a
|
||||
// missing field yields null rather than a compile error; an absent or
|
||||
// unrecognized value leaves the type NULL instead of guessing.
|
||||
string? apiPostType = PostTypes.FromApiType(post.type?.ToString() as string);
|
||||
|
||||
// Only insert if any of the data contains strings from ContainsList
|
||||
bool shouldInsert = false;
|
||||
string matchedFieldName = string.Empty;
|
||||
@@ -1090,7 +1099,8 @@ if (shouldInsert)
|
||||
DataAccess.AddPost(authorBlog, postID, reblogURL, date, postURL, slug, reblogKey,
|
||||
reblogName, summary, quote, body, tags, link, photoURL,
|
||||
photoCaption, downloadedFiles, audioCaption, question, answer,
|
||||
title, hasImage, true, rootBlogName: rootBlogName, rootURL: rootURL);
|
||||
title, hasImage, true, rootBlogName: rootBlogName, rootURL: rootURL,
|
||||
postType: apiPostType);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -1508,7 +1518,15 @@ if (shouldInsert)
|
||||
string normalizedDirectoryName = NormalizeBlogFolderName(new DirectoryInfo(path).Name);
|
||||
bool isAtOrAfterStart = string.IsNullOrWhiteSpace(startFromBlogName) || string.Compare(normalizedDirectoryName, startFromBlogName, StringComparison.OrdinalIgnoreCase) >= 0;
|
||||
|
||||
// The filename is the post type ("texts.txt" -> "texts"), so only the eight
|
||||
// known export files are post sources. Everything else in a blog folder is
|
||||
// either not a post file at all (README.txt, url lists, triage scratch) or
|
||||
// is our own derived output -- Unknown.txt above all, which must never be
|
||||
// read back in as a source or it perpetuates itself.
|
||||
string? filePostType = PostTypes.FromFileName(file);
|
||||
|
||||
if (file.EndsWith(".txt", StringComparison.OrdinalIgnoreCase)
|
||||
&& filePostType != null
|
||||
&& (string.IsNullOrEmpty(blogName) || path.IndexOf(blogName, StringComparison.OrdinalIgnoreCase) >= 0)
|
||||
&& isAtOrAfterStart)
|
||||
{
|
||||
@@ -1541,7 +1559,7 @@ if (shouldInsert)
|
||||
DataAccess.AddPost(curDir, long.Parse(reblog.postID), reblog.reblogURL, reblog.date, reblog.postURL, reblog.slug, reblog.reblogKey,
|
||||
reblog.reblogName, reblog.summary, reblog.quote, reblog.body, reblog.tags, reblog.link, reblog.photoURL,
|
||||
reblog.photoCaption, reblog.downloadedFiles, reblog.audioCaption, reblog.question, reblog.answer,
|
||||
reblog.title, false, rootURL: reblog.rootURL);
|
||||
reblog.title, false, rootURL: reblog.rootURL, postType: filePostType);
|
||||
recordImportStopwatch.Stop();
|
||||
|
||||
postsAdded++;
|
||||
@@ -1689,7 +1707,7 @@ if (shouldInsert)
|
||||
DataAccess.AddPost(curDir, long.Parse(reblog.postID), reblog.reblogURL, reblog.date, reblog.postURL, reblog.slug, reblog.reblogKey,
|
||||
reblog.reblogName, reblog.summary, reblog.quote, reblog.body, reblog.tags, reblog.link, reblog.photoURL,
|
||||
reblog.photoCaption, reblog.downloadedFiles, reblog.audioCaption, reblog.question, reblog.answer,
|
||||
reblog.title, true, rootURL: reblog.rootURL);
|
||||
reblog.title, true, rootURL: reblog.rootURL, postType: filePostType);
|
||||
recordImportStopwatch.Stop();
|
||||
|
||||
postsAdded++;
|
||||
|
||||
Reference in New Issue
Block a user