diff --git a/config.toml.example b/config.toml.example index f46b371..ecbeb1b 100644 --- a/config.toml.example +++ b/config.toml.example @@ -82,10 +82,6 @@ min_positive_priority = 1 max_positive_priority = 720 bootstrap_priority = 10 bootstrap_ttl_hours = 24 -bootstrap_label_content_types = ["vocaloid", "maybe_vocaloid"] -bootstrap_label_origin = "rule" -bootstrap_label_writers = ["classification_apply", "classification_trigger"] -bootstrap_tid_v2_allowlist = [2022, 2061] processed_backfill_new_video_age_days = 7 collection_business_timezone = "Asia/Shanghai" @@ -110,11 +106,21 @@ max_recommendation_depth = 1 [processing.filtering] # Content filtering settings -type_id_whitelist = [] # Array of video type IDs to include (e.g., [28, 30, 130]) -copyright_whitelist = [] # Array of copyright types to include (1: Original, 2: Repost) content_blacklist = [] # Array of content blacklist keywords -content_whitelist = [] # Array of content whitelist keywords -pid_v2_whitelist = [] # Explicit pid_v2 values eligible for recommendation admission + +[whitelist.video] +type_ids = [] # Video type IDs to include (e.g., [28, 30, 130]) +copyright_types = [] # Copyright types to include (1: Original, 2: Repost) +content_keywords = [] # Keywords that bypass type and copyright checks + +[whitelist.recommendation] +pid_v2 = [] # Explicit pid_v2 values eligible for recommendation admission + +[whitelist.minute_bootstrap] +label_content_types = ["vocaloid", "maybe_vocaloid"] +label_origin = "rule" +label_writers = ["classification_apply", "classification_trigger"] +tid_v2 = [2022, 2061] [export] [export.mysql] diff --git a/docs/config/index.md b/docs/config/index.md index b107a3a..1d9e5e5 100644 --- a/docs/config/index.md +++ b/docs/config/index.md @@ -10,7 +10,15 @@ Start by copying the example file: cp config.toml.example config.toml ``` -Then fill only the sections you use. +Then fill only the sections you use. On startup, legacy whitelist TOML keys are +moved to `[whitelist.*]` automatically. Migration keeps a temporary backup +and replaces the file only after validating the migrated values. A conflicting +destination, quoted/dotted key, symlinked file needing migration, or +non-comment multiline string requires a manual edit. Invalid TOML produces a +warning and falls back to environment variables. Run migration when no other +process or editor is writing `config.toml`. If a process stops during migration +and leaves `config.toml.migration-lock`, remove that lock after confirming no +instance is migrating. ## Sections @@ -33,4 +41,3 @@ Each setting follows this order: This means a value in `config.toml` overrides the matching environment variable. Remove the TOML value or leave it empty when you want the environment variable to take effect. - diff --git a/docs/config/minute.md b/docs/config/minute.md index 8986f26..a7adda2 100644 --- a/docs/config/minute.md +++ b/docs/config/minute.md @@ -15,10 +15,6 @@ min_positive_priority = 1 max_positive_priority = 720 bootstrap_priority = 10 bootstrap_ttl_hours = 24 -bootstrap_label_content_types = ["vocaloid", "maybe_vocaloid"] -bootstrap_label_origin = "rule" -bootstrap_label_writers = ["classification_apply", "classification_trigger"] -bootstrap_tid_v2_allowlist = [2022, 2061] processed_backfill_new_video_age_days = 7 collection_business_timezone = "Asia/Shanghai" ``` @@ -121,10 +117,6 @@ rescheduling and increments | `max_positive_priority` | `MINUTE_MAX_POSITIVE_PRIORITY` | `720` | Maximum positive interval in minutes. | | `bootstrap_priority` | `MINUTE_BOOTSTRAP_PRIORITY` | `10` | Initial interval for newly tracked videos. | | `bootstrap_ttl_hours` | `MINUTE_BOOTSTRAP_TTL_HOURS` | `24` | Maximum bootstrap window. | -| `bootstrap_label_content_types` | `MINUTE_BOOTSTRAP_LABEL_CONTENT_TYPES` | `["vocaloid", "maybe_vocaloid"]` | Label content types eligible for bootstrap. | -| `bootstrap_label_origin` | `MINUTE_BOOTSTRAP_LABEL_ORIGIN` | `rule` | Required label origin for bootstrap. | -| `bootstrap_label_writers` | `MINUTE_BOOTSTRAP_LABEL_WRITERS` | `["classification_apply", "classification_trigger"]` | Label writers eligible for bootstrap. | -| `bootstrap_tid_v2_allowlist` | `MINUTE_BOOTSTRAP_TID_V2_ALLOWLIST` | `[2022, 2061]` | Fallback type IDs eligible for bootstrap. | | `processed_backfill_new_video_age_days` | `MINUTE_PROCESSED_BACKFILL_NEW_VIDEO_AGE_DAYS` | `7` | Age cutoff for processed-video bootstrap. | | `collection_business_timezone` | `MINUTE_COLLECTION_BUSINESS_TIMEZONE` | `Asia/Shanghai` | Business date timezone for daily refresh. | @@ -132,10 +124,40 @@ rescheduling and increments `target_delta_lower`..`target_delta_upper` range after parsing. `MINUTE_ENABLED` accepts `1`, `true`, `yes`, and `on` as true values. -Use TOML for array settings such as `bootstrap_label_content_types`, -`bootstrap_label_writers`, and `bootstrap_tid_v2_allowlist`. Unlike -`BILIBILI_COOKIE_FILES` and the processing filter lists, these minute array -settings are not parsed from comma-separated environment strings. +## Bootstrap eligibility + +Bootstrap eligibility is configured under `[whitelist.minute_bootstrap]`: + +```toml +[whitelist.minute_bootstrap] +label_content_types = ["vocaloid", "maybe_vocaloid"] +label_origin = "rule" +label_writers = ["classification_apply", "classification_trigger"] +tid_v2 = [2022, 2061] +``` + +| TOML key | Environment variable | Default | Effect | +| --- | --- | --- | --- | +| `label_content_types` | `MINUTE_BOOTSTRAP_LABEL_CONTENT_TYPES` | `["vocaloid", "maybe_vocaloid"]` | Eligible formal label types. | +| `label_origin` | `MINUTE_BOOTSTRAP_LABEL_ORIGIN` | `rule` | Required label origin. | +| `label_writers` | `MINUTE_BOOTSTRAP_LABEL_WRITERS` | `["classification_apply", "classification_trigger"]` | Eligible label writers. | +| `tid_v2` | `MINUTE_BOOTSTRAP_TID_V2_ALLOWLIST` | `[2022, 2061]` | Fallback values when no formal label exists. | + +On startup, the app automatically moves these former TOML keys in `config.toml` +to their current locations: + +| Former TOML key | Current TOML key | +| --- | --- | +| `minute.bootstrap_label_content_types` | `whitelist.minute_bootstrap.label_content_types` | +| `minute.bootstrap_label_origin` | `whitelist.minute_bootstrap.label_origin` | +| `minute.bootstrap_label_writers` | `whitelist.minute_bootstrap.label_writers` | +| `minute.bootstrap_tid_v2_allowlist` | `whitelist.minute_bootstrap.tid_v2` | + +Existing environment variable names remain valid. After changing +minute-bootstrap values, run `pnpm init-schema` against the database so stored +SQL function defaults use the new values, then restart the application. Without +schema initialization, calls that rely on the stored SQL defaults can continue +using the previously installed bootstrap values. ## Metrics diff --git a/docs/config/processing.md b/docs/config/processing.md index beabb51..a132f2b 100644 --- a/docs/config/processing.md +++ b/docs/config/processing.md @@ -1,6 +1,7 @@ # Processing Configuration -`[processing]` controls feature flags and filtering rules. +`[processing]` controls feature flags and the content blacklist. Video and +recommendation allowlists are configured under `[whitelist]`. ## Feature flags @@ -25,17 +26,57 @@ max_recommendation_depth = 1 ```toml [processing.filtering] -type_id_whitelist = [] -copyright_whitelist = [] content_blacklist = [] -content_whitelist = [] ``` | TOML key | Environment variable | Default | Meaning | | --- | --- | --- | --- | -| `type_id_whitelist` | `TYPE_ID_WHITE_LIST` | `[]` | Type IDs to include. | -| `copyright_whitelist` | `COPYRIGHT_WHITE_LIST` | `[]` | Copyright types to include. | | `content_blacklist` | `CONTENT_BLACK_LIST` | `[]` | Keywords to exclude. | -| `content_whitelist` | `CONTENT_WHITE_LIST` | `[]` | Keywords to include. | -For environment variables, list values are comma-separated. +For environment variables, list values are comma-separated. The blacklist is +applied after the video allowlists; matching an allowlist never bypasses it. + +## Video and recommendation allowlists + +Video and recommendation allowlists use these TOML keys. A TOML value takes +precedence over the corresponding environment variable; environment list values +are comma-separated. Empty TOML arrays explicitly disable a list. + +```toml +[whitelist.video] +type_ids = [] +copyright_types = [] +content_keywords = [] + +[whitelist.recommendation] +pid_v2 = [] +``` + +| TOML key | Environment variable | Default | Effect | +| --- | --- | --- | --- | +| `whitelist.video.type_ids` | `TYPE_ID_WHITE_LIST` | `[]` | Video types admitted by the type check. | +| `whitelist.video.copyright_types` | `COPYRIGHT_WHITE_LIST` | `[]` | Copyright types admitted by the copyright check. | +| `whitelist.video.content_keywords` | `CONTENT_WHITE_LIST` | `[]` | Keywords that bypass the type and copyright checks. Blank keywords are invalid. | +| `whitelist.recommendation.pid_v2` | `UPDATE_INFO_PID_V2_WHITELIST` | `[]` | Related videos admitted by recommendation collection and `--update-info`. | + +An empty video type or copyright list disables that check. A matching content +keyword bypasses those checks, while `processing.filtering.content_blacklist` +still excludes matching videos. Recommendation admission requires a listed +`pid_v2`; `--update-info --pid-v2-whitelist` overrides the configured list for +that run. + +On startup, the app automatically moves these former TOML keys in `config.toml` +to their current locations: + +| Former TOML key | Current TOML key | +| --- | --- | +| `processing.filtering.type_id_whitelist` | `whitelist.video.type_ids` | +| `processing.filtering.copyright_whitelist` | `whitelist.video.copyright_types` | +| `processing.filtering.content_whitelist` | `whitelist.video.content_keywords` | +| `processing.filtering.pid_v2_whitelist` | `whitelist.recommendation.pid_v2` | + +The migration preserves other configuration text and comments. It stops without +changing the file if a destination key already exists or the old key uses quoted +or dotted TOML syntax. Move those keys manually before restarting. Keep +`processing.filtering.content_blacklist` where it is. Existing environment +variable names remain valid. diff --git a/src/config/index.ts b/src/config/index.ts index b2dc2a6..201b173 100644 --- a/src/config/index.ts +++ b/src/config/index.ts @@ -1,7 +1,7 @@ -import { readFileSync } from "node:fs"; +import { existsSync } from "node:fs"; import { resolve } from "node:path"; -import { parse as parseToml } from "smol-toml"; import { z } from "zod"; +import { ConfigTomlParseError, loadConfigToml } from "./migrate-whitelist"; import { applicationSchema, bilibiliSchema, @@ -16,6 +16,7 @@ import { createRepairConfig, createServerConfig, createSubtitleConfig, + createWhitelistConfig, databaseSchema, exportSchema, metricsSchema, @@ -25,18 +26,21 @@ import { repairSchema, serverSchema, subtitleSchema, + whitelistSchema, } from "./schemas"; +const configPath = resolve(process.cwd(), "config.toml"); let tomlData: unknown = {}; -try { - const configPath = resolve(process.cwd(), "config.toml"); - const tomlString = readFileSync(configPath, "utf-8"); - tomlData = parseToml(tomlString); -} catch (error) { - console.warn( - "Warning: config.toml not found or invalid. Using environment variables as fallback.", - ); - console.warn("Actual error:", error); +if (existsSync(configPath)) { + try { + tomlData = loadConfigToml(configPath); + } catch (error) { + if (!(error instanceof ConfigTomlParseError)) throw error; + console.warn( + "Warning: config.toml not found or invalid. Using environment variables as fallback.", + ); + console.warn("Actual error:", error.cause); + } } // Helper function to get configuration value from TOML or environment variable @@ -90,6 +94,7 @@ const configSchema = z.object({ server: serverSchema, subtitle: subtitleSchema, notifications: notificationsSchema, + whitelist: whitelistSchema, }); export const config = configSchema.parse({ @@ -104,4 +109,5 @@ export const config = configSchema.parse({ server: createServerConfig(getConfigValue), subtitle: createSubtitleConfig(getConfigValue), notifications: createNotificationsConfig(getConfigValue), + whitelist: createWhitelistConfig(getConfigValue), }); diff --git a/src/config/migrate-whitelist.test.ts b/src/config/migrate-whitelist.test.ts new file mode 100644 index 0000000..906efd3 --- /dev/null +++ b/src/config/migrate-whitelist.test.ts @@ -0,0 +1,202 @@ +import assert from "node:assert/strict"; +import { spawnSync } from "node:child_process"; +import { + mkdtempSync, + readdirSync, + readFileSync, + rmSync, + statSync, + symlinkSync, + writeFileSync, +} from "node:fs"; +import { tmpdir } from "node:os"; +import { dirname, join } from "node:path"; +import test from "node:test"; +import { ConfigTomlParseError, loadConfigToml } from "./migrate-whitelist"; + +function configFile( + t: { after: (fn: () => void) => void }, + source: string, +): string { + const directory = mkdtempSync(join(tmpdir(), "whitelist-migration-")); + t.after(() => rmSync(directory, { recursive: true, force: true })); + const file = join(directory, "config.toml"); + writeFileSync(file, source, { mode: 0o600 }); + return file; +} + +test("migrates all legacy keys while preserving values, comments, and mode", (t) => { + const original = `# global comment +# body = ''' +[processing.filtering] +content_blacklist = ["spam"] # keep +content_whitelist = [ + "music", # inside array + "dance", +] # keyword comment +type_id_whitelist = [28] # type comment +copyright_whitelist = [1] +pid_v2_whitelist = [7] + +[minute] +bootstrap_label_content_types = ["vocaloid"] +bootstrap_label_origin = "rule" # origin comment +bootstrap_label_writers = ["classification_apply"] +bootstrap_tid_v2_allowlist = [2022] + +[whitelist.video] +# existing destination section +`; + const file = configFile(t, original); + const parsed = loadConfigToml(file); + const migrated = readFileSync(file, "utf-8"); + assert.deepEqual(parsed.whitelist, { + video: { + content_keywords: ["music", "dance"], + type_ids: [28], + copyright_types: [1], + }, + recommendation: { pid_v2: [7] }, + minute_bootstrap: { + label_content_types: ["vocaloid"], + label_origin: "rule", + label_writers: ["classification_apply"], + tid_v2: [2022], + }, + }); + assert.match(migrated, /content_blacklist = \["spam"\] # keep/); + assert.match(migrated, /"music", # inside array/); + assert.match(migrated, /type_ids = \[28\] # type comment/); + assert.match(migrated, /label_origin = "rule" # origin comment/); + assert.equal(statSync(file).mode & 0o777, 0o600); + assert.deepEqual(readdirSync(join(file, "..")), ["config.toml"]); + assert.deepEqual(loadConfigToml(file).whitelist, parsed.whitelist); + assert.equal(readFileSync(file, "utf-8"), migrated); +}); + +test("inline triple-quote comments do not block migration", (t) => { + const source = `[processing.filtering]\ncontent_whitelist = ["music"] # example: """text"""\n`; + const file = configFile(t, source); + loadConfigToml(file); + assert.match( + readFileSync(file, "utf-8"), + /content_keywords = \["music"\] # example: """text"""/, + ); +}); + +test("actual multiline strings still require manual migration", (t) => { + const source = `[processing.filtering]\ncontent_whitelist = ["music"]\ndescription = """first\nsecond"""\n`; + const file = configFile(t, source); + assert.throws(() => loadConfigToml(file), /multiline strings/); + assert.equal(readFileSync(file, "utf-8"), source); +}); + +test("attached comments follow their legacy key, but separated section comments stay", (t) => { + const source = `[processing.filtering]\n# section guidance\n\n# keywords for this video\n# keep this note with the key\ncontent_whitelist = ["music"]\ncontent_blacklist = ["spam"]\n`; + const file = configFile(t, source); + loadConfigToml(file); + const migrated = readFileSync(file, "utf-8"); + assert.match( + migrated, + /\[processing\.filtering\]\n# section guidance\n\ncontent_blacklist/, + ); + assert.match( + migrated, + /\[whitelist\.video\]\n# keywords for this video\n# keep this note with the key\ncontent_keywords = \["music"\]/, + ); +}); + +test("conflicts and unsupported old key forms leave config.toml unchanged", (t) => { + for (const source of [ + `[processing.filtering]\ncontent_whitelist = ["music"]\n[whitelist.video]\ncontent_keywords = ["dance"]\n`, + `[processing.filtering]\n"content_whitelist" = ["music"]\n`, + `processing.filtering.content_whitelist = ["music"]\n`, + `[processing.filtering]\ncontent_whitelist = [\n`, + ]) { + const file = configFile(t, source); + assert.throws(() => loadConfigToml(file), /config.toml|content_whitelist/); + assert.equal(readFileSync(file, "utf-8"), source); + assert.deepEqual(readdirSync(join(file, "..")), ["config.toml"]); + } +}); + +test("a symlinked config cannot be rewritten automatically", (t) => { + const target = configFile( + t, + `[processing.filtering]\ncontent_whitelist = ["music"]\n`, + ); + const link = join(target, "..", "linked.toml"); + symlinkSync(target, link); + const original = readFileSync(target, "utf-8"); + assert.throws(() => loadConfigToml(link), /symlinks cannot be migrated/); + assert.equal(readFileSync(target, "utf-8"), original); +}); + +test("a symlinked config without legacy keys remains readable", (t) => { + const target = configFile( + t, + `[whitelist.video]\ncontent_keywords = ["music"]\n`, + ); + const link = join(target, "..", "linked.toml"); + symlinkSync(target, link); + assert.deepEqual(loadConfigToml(link).whitelist, { + video: { content_keywords: ["music"] }, + }); + assert.equal(readFileSync(target, "utf-8"), readFileSync(link, "utf-8")); +}); + +test("migration retains numeric-string minute IDs", (t) => { + const file = configFile( + t, + `[minute]\nbootstrap_tid_v2_allowlist = ["2022"]\n`, + ); + const migrated = loadConfigToml(file); + assert.deepEqual(migrated.whitelist, { + minute_bootstrap: { tid_v2: ["2022"] }, + }); + assert.match(readFileSync(file, "utf-8"), /tid_v2 = \["2022"\]/); +}); + +test("invalid migrated whitelist leaves the original config untouched", (t) => { + const source = `[processing.filtering]\ncontent_whitelist = [" "]\n`; + const file = configFile(t, source); + assert.throws(() => loadConfigToml(file), /contentKeywords/); + assert.equal(readFileSync(file, "utf-8"), source); + assert.deepEqual(readdirSync(join(file, "..")), ["config.toml"]); +}); + +test("invalid source TOML warns and falls back to environment configuration", (t) => { + const source = "[processing.filtering\n"; + const file = configFile(t, source); + assert.throws(() => loadConfigToml(file), ConfigTomlParseError); + const root = process.cwd(); + const result = spawnSync( + join(root, "node_modules/.bin/tsx"), + [ + "-e", + `const { config } = require(${JSON.stringify(join(root, "src/config/index.ts"))}); console.log(config.bilibili.cookieFiles[0].path)`, + ], + { + cwd: dirname(file), + encoding: "utf-8", + env: { ...process.env, BILIBILI_COOKIE_FILE: "from-environment" }, + }, + ); + assert.equal(result.status, 0, result.stderr); + assert.match(result.stderr, /config.toml not found or invalid/); + assert.match(result.stdout, /from-environment/); + assert.equal(readFileSync(file, "utf-8"), source); +}); + +test("an existing migration lock leaves the source unchanged", (t) => { + const source = `[processing.filtering]\ncontent_whitelist = ["music"]\n`; + const file = configFile(t, source); + const lock = `${file}.migration-lock`; + writeFileSync(lock, "", { flag: "wx" }); + assert.throws(() => loadConfigToml(file), /EEXIST/); + assert.equal(readFileSync(file, "utf-8"), source); + assert.deepEqual(readdirSync(dirname(file)).sort(), [ + "config.toml", + "config.toml.migration-lock", + ]); +}); diff --git a/src/config/migrate-whitelist.ts b/src/config/migrate-whitelist.ts new file mode 100644 index 0000000..1b09e66 --- /dev/null +++ b/src/config/migrate-whitelist.ts @@ -0,0 +1,311 @@ +import { randomUUID } from "node:crypto"; +import { + chmodSync, + closeSync, + constants, + copyFileSync, + lstatSync, + openSync, + readFileSync, + renameSync, + rmSync, + writeFileSync, +} from "node:fs"; +import { basename, dirname, join } from "node:path"; +import { isDeepStrictEqual } from "node:util"; +import { parse as parseToml } from "smol-toml"; +import { + createWhitelistConfig, + legacyWhitelistPaths, +} from "./schemas/whitelist"; + +type Toml = Record; + +export class ConfigTomlParseError extends Error { + readonly cause: unknown; + + constructor(cause: unknown) { + super( + "config.toml is invalid TOML; using environment variables as fallback.", + ); + this.name = "ConfigTomlParseError"; + this.cause = cause; + } +} + +function atPath(data: unknown, path: string): unknown { + let value = data; + for (const part of path.split(".")) { + if (value === null || typeof value !== "object" || !(part in value)) { + return undefined; + } + value = (value as Toml)[part]; + } + return value; +} + +function parseConfig(source: string): Toml { + return parseToml(source); +} + +function hasMultilineString(line: string): boolean { + let quote: '"' | "'" | undefined; + let escaped = false; + for (let index = 0; index < line.length; index++) { + const char = line[index]; + if (quote) { + if (quote === '"' && !escaped && char === "\\") escaped = true; + else { + if (!escaped && char === quote) quote = undefined; + escaped = false; + } + } else if (char === "#") { + break; + } else if (char === '"' || char === "'") { + if (line.startsWith(char.repeat(3), index)) return true; + quote = char; + } + } + return false; +} + +function arrayEnd( + lines: string[], + start: number, + value: string, + path: string, +): number { + if (!value.trimStart().startsWith("[")) return start; + let depth = 0; + let quote: '"' | "'" | undefined; + let escaped = false; + for (let row = start; row < lines.length; row++) { + const part = row === start ? value : lines[row]; + for (const char of part) { + if (quote) { + if (quote === '"' && !escaped && char === "\\") escaped = true; + else { + if (!escaped && char === quote) quote = undefined; + escaped = false; + } + } else if (char === "#") break; + else if (char === '"' || char === "'") quote = char; + else if (char === "[") depth++; + else if (char === "]") { + depth--; + if (depth === 0) return row; + } + } + } + throw new Error( + `Cannot automatically migrate ${path}: array assignment is incomplete.`, + ); +} + +export function migrateWhitelistText(source: string): string { + const parsed = parseConfig(source); + const present = legacyWhitelistPaths.filter( + ([oldPath]) => atPath(parsed, oldPath) !== undefined, + ); + if (present.length === 0) return source; + if (source.split(/\r?\n/).some(hasMultilineString)) { + throw new Error( + "Cannot automatically migrate config.toml containing multiline strings; move legacy whitelist keys manually.", + ); + } + for (const [oldPath, newPath] of present) { + if (atPath(parsed, newPath) !== undefined) { + throw new Error( + `Cannot migrate ${oldPath}: ${newPath} already exists in config.toml.`, + ); + } + } + + const lines = source.split(/(?<=\n)/); + const newline = source.includes("\r\n") ? "\r\n" : "\n"; + const sections = new Map(); + const assignments = new Map< + string, + { start: number; end: number; leading: string; text: string } + >(); + let section = ""; + for (let row = 0; row < lines.length; row++) { + const currentLine = lines[row].replace(/\r?\n$/, ""); + const header = currentLine.match(/^\s*\[([A-Za-z0-9_.-]+)\]\s*(?:#.*)?$/); + if (header) { + const previous = sections.get(section); + if (previous) previous.end = row; + section = header[1]; + sections.set(section, { end: lines.length }); + continue; + } + if (/^\s*\[/.test(currentLine)) { + const previous = sections.get(section); + if (previous) previous.end = row; + section = ""; + continue; + } + const assignment = currentLine.match(/^(\s*)([A-Za-z0-9_-]+)(\s*=)(.*)$/); + if (!assignment) continue; + const path = section ? `${section}.${assignment[2]}` : assignment[2]; + if (!present.some(([oldPath]) => oldPath === path)) continue; + const end = arrayEnd(lines, row, assignment[4], path); + let start = row; + while (start > 0 && /^\s*#/.test(lines[start - 1])) start--; + assignments.set(path, { + start, + end, + leading: lines.slice(start, row).join(""), + text: lines.slice(row, end + 1).join(""), + }); + row = end; + } + + const insertions = new Map(); + const removed = new Set(); + const newSections = new Map(); + for (const [oldPath, newPath] of present) { + const match = assignments.get(oldPath); + if (!match) { + throw new Error( + `Cannot automatically migrate ${oldPath}: use bare table headers and keys, or move this key manually.`, + ); + } + const pieces = newPath.split("."); + const key = pieces.pop(); + const destination = pieces.join("."); + if (!key) throw new Error(`Invalid whitelist destination for ${oldPath}.`); + const renamed = match.text.replace( + /^(\s*)[A-Za-z0-9_-]+(\s*=)/, + `$1${key}$2`, + ); + const statement = `${match.leading}${renamed.endsWith("\n") ? renamed : `${renamed}${newline}`}`; + const target = sections.get(destination); + if (target) { + const pending = insertions.get(target.end) ?? []; + pending.push(statement); + insertions.set(target.end, pending); + } else { + const pending = newSections.get(destination) ?? []; + pending.push(statement); + newSections.set(destination, pending); + } + for (let row = match.start; row <= match.end; row++) removed.add(row); + } + + let result = ""; + for (let row = 0; row <= lines.length; row++) { + const pending = insertions.get(row); + if (pending) { + if (result.length > 0 && !result.endsWith("\n")) result += newline; + result += pending.join(""); + } + if (row < lines.length && !removed.has(row)) result += lines[row]; + } + for (const [sectionName, statements] of newSections) { + if (result.length > 0 && !result.endsWith("\n")) result += newline; + result += `${newline}[${sectionName}]${newline}${statements.join("")}`; + } + verifyMigration(parsed, parseConfig(result), present); + return result; +} + +function verifyMigration( + original: Toml, + migrated: Toml, + present: (typeof legacyWhitelistPaths)[number][], +): void { + for (const [oldPath, newPath] of present) { + if (atPath(migrated, oldPath) !== undefined) { + throw new Error(`Whitelist migration left ${oldPath} in config.toml.`); + } + if ( + !isDeepStrictEqual(atPath(migrated, newPath), atPath(original, oldPath)) + ) { + throw new Error(`Whitelist migration changed the value for ${newPath}.`); + } + } +} + +export function loadConfigToml(configPath: string): Toml { + const source = readFileSync(configPath, "utf-8"); + let original: Toml; + try { + original = parseConfig(source); + } catch (error) { + throw new ConfigTomlParseError(error); + } + const present = legacyWhitelistPaths.filter( + ([oldPath]) => atPath(original, oldPath) !== undefined, + ); + const migrated = migrateWhitelistText(source); + if (migrated === source) return original; + const stat = lstatSync(configPath); + if (!stat.isFile()) { + throw new Error( + "config.toml must be a regular file; symlinks cannot be migrated automatically.", + ); + } + const migratedData = parseConfig(migrated); + createWhitelistConfig((tomlPath, envKey, defaultValue) => { + const value = atPath(migratedData, tomlPath.join(".")); + if (value !== undefined && value !== "") return value; + const envValue = process.env[envKey]; + return envValue !== undefined && envValue !== "" ? envValue : defaultValue; + }); + + const directory = dirname(configPath); + const name = basename(configPath); + const lock = join(directory, `${name}.migration-lock`); + const backup = join(directory, `${name}.backup-${randomUUID()}`); + const temporary = join(directory, `${name}.tmp-${randomUUID()}`); + let backupMade = false; + let replaced = false; + const lockHandle = openSync(lock, "wx", 0o600); + try { + if (readFileSync(configPath, "utf-8") !== source) { + throw new Error( + "config.toml changed during whitelist migration; retry after stopping other editors.", + ); + } + copyFileSync(configPath, backup, constants.COPYFILE_EXCL); + backupMade = true; + if (readFileSync(backup, "utf-8") !== source) { + throw new Error( + "config.toml changed during whitelist migration; retry after stopping other editors.", + ); + } + const handle = openSync(temporary, "wx", stat.mode); + try { + writeFileSync(handle, migrated, "utf-8"); + } finally { + closeSync(handle); + } + chmodSync(temporary, stat.mode); + verifyMigration( + original, + parseConfig(readFileSync(temporary, "utf-8")), + present, + ); + if (readFileSync(configPath, "utf-8") !== source) { + throw new Error( + "config.toml changed during whitelist migration; retry after stopping other editors.", + ); + } + renameSync(temporary, configPath); + replaced = true; + const result = parseConfig(readFileSync(configPath, "utf-8")); + verifyMigration(original, result, present); + rmSync(backup); + backupMade = false; + return result; + } catch (error) { + if (replaced && backupMade) renameSync(backup, configPath); + throw error; + } finally { + rmSync(temporary, { force: true }); + if (!replaced) rmSync(backup, { force: true }); + closeSync(lockHandle); + rmSync(lock); + } +} diff --git a/src/config/schemas/index.ts b/src/config/schemas/index.ts index 15c0a34..217299d 100644 --- a/src/config/schemas/index.ts +++ b/src/config/schemas/index.ts @@ -12,3 +12,4 @@ export { createProcessingConfig, processingSchema } from "./processing"; export { createRepairConfig, repairSchema } from "./repair"; export { createServerConfig, serverSchema } from "./server"; export { createSubtitleConfig, subtitleSchema } from "./subtitle"; +export { createWhitelistConfig, whitelistSchema } from "./whitelist"; diff --git a/src/config/schemas/minute.ts b/src/config/schemas/minute.ts index d26711b..16321b5 100644 --- a/src/config/schemas/minute.ts +++ b/src/config/schemas/minute.ts @@ -16,16 +16,6 @@ export const minuteSchema = z.object({ maxPositivePriority: z.coerce.number().int().positive().default(720), bootstrapPriority: z.coerce.number().int().positive().default(10), bootstrapTtlHours: z.coerce.number().int().positive().max(24).default(24), - bootstrapLabelContentTypes: z - .array(z.string()) - .default(["vocaloid", "maybe_vocaloid"]), - bootstrapLabelOrigin: z.string().default("rule"), - bootstrapLabelWriters: z - .array(z.string()) - .default(["classification_apply", "classification_trigger"]), - bootstrapTidV2Allowlist: z - .array(z.coerce.number().int()) - .default([2022, 2061]), processedBackfillNewVideoAgeDays: z.coerce .number() .int() @@ -92,26 +82,6 @@ export function createMinuteConfig( "MINUTE_BOOTSTRAP_TTL_HOURS", 24, ), - bootstrapLabelContentTypes: getConfigValue( - ["minute", "bootstrap_label_content_types"], - "MINUTE_BOOTSTRAP_LABEL_CONTENT_TYPES", - ["vocaloid", "maybe_vocaloid"], - ), - bootstrapLabelOrigin: getConfigValue( - ["minute", "bootstrap_label_origin"], - "MINUTE_BOOTSTRAP_LABEL_ORIGIN", - "rule", - ), - bootstrapLabelWriters: getConfigValue( - ["minute", "bootstrap_label_writers"], - "MINUTE_BOOTSTRAP_LABEL_WRITERS", - ["classification_apply", "classification_trigger"], - ), - bootstrapTidV2Allowlist: getConfigValue( - ["minute", "bootstrap_tid_v2_allowlist"], - "MINUTE_BOOTSTRAP_TID_V2_ALLOWLIST", - [2022, 2061], - ), processedBackfillNewVideoAgeDays: getConfigValue( ["minute", "processed_backfill_new_video_age_days"], "MINUTE_PROCESSED_BACKFILL_NEW_VIDEO_AGE_DAYS", diff --git a/src/config/schemas/processing.test.ts b/src/config/schemas/processing.test.ts index 67b431f..9e34e44 100644 --- a/src/config/schemas/processing.test.ts +++ b/src/config/schemas/processing.test.ts @@ -1,25 +1,21 @@ import assert from "node:assert/strict"; import test from "node:test"; -import { createProcessingConfig, processingSchema } from "./processing.js"; +import { createProcessingConfig } from "./processing.js"; -test("pid_v2 whitelist prefers TOML values and validates every entry", () => { - const config = createProcessingConfig((path, envKey) => { - if (path.join(".") === "processing.filtering.pid_v2_whitelist") { - return [7, 11]; - } - return envKey === "UPDATE_INFO_PID_V2_WHITELIST" ? "13,17" : undefined; - }); - assert.deepEqual(config.filtering.pidV2Whitelist, [7, 11]); - const environmentConfig = createProcessingConfig((_path, envKey) => - envKey === "UPDATE_INFO_PID_V2_WHITELIST" ? "13,17" : undefined, +test("content blacklist keeps TOML precedence and parses environment entries", () => { + const fromToml = createProcessingConfig((path, envKey) => + path.join(".") === "processing.filtering.content_blacklist" + ? ["skip"] + : envKey === "CONTENT_BLACK_LIST" + ? "other, word" + : undefined, ); - assert.deepEqual(environmentConfig.filtering.pidV2Whitelist, [13, 17]); - assert.throws( - () => - processingSchema.parse({ - features: {}, - filtering: { pidV2Whitelist: [7, 1.5] }, - }), - /pidV2Whitelist/, + assert.deepEqual(fromToml.filtering.contentBlacklist, ["skip"]); + const fromEnvironment = createProcessingConfig((_path, envKey) => + envKey === "CONTENT_BLACK_LIST" ? "other, word" : undefined, ); + assert.deepEqual(fromEnvironment.filtering.contentBlacklist, [ + "other", + "word", + ]); }); diff --git a/src/config/schemas/processing.ts b/src/config/schemas/processing.ts index 5e1be63..870f3fb 100644 --- a/src/config/schemas/processing.ts +++ b/src/config/schemas/processing.ts @@ -12,11 +12,7 @@ export const processingSchema = z.object({ maxRelatedExpansionDepth: z.coerce.number().default(1), }), filtering: z.object({ - typeIdWhitelist: z.array(z.number()).default([]), contentBlacklist: z.array(z.string()).default([]), - contentWhitelist: z.array(z.string()).default([]), - copyrightWhitelist: z.array(z.number()).default([]), - pidV2Whitelist: z.array(z.number().int().positive()).default([]), }), }); @@ -39,9 +35,10 @@ export function createProcessingConfig( ["processing", "features", "max_recommendation_depth"], "MAX_RECOMMENDATION_DEPTH", ); - const pidV2Whitelist = getConfigValue( - ["processing", "filtering", "pid_v2_whitelist"], - "UPDATE_INFO_PID_V2_WHITELIST", + const contentBlacklist = getConfigValue( + ["processing", "filtering", "content_blacklist"], + "CONTENT_BLACK_LIST", + [], ); return { @@ -79,39 +76,10 @@ export function createProcessingConfig( 1, }, filtering: { - typeIdWhitelist: - getConfigValue( - ["processing", "filtering", "type_id_whitelist"], - "TYPE_ID_WHITE_LIST", - ) || - process.env.TYPE_ID_WHITE_LIST?.split(",").map(Number) || - [], contentBlacklist: - getConfigValue( - ["processing", "filtering", "content_blacklist"], - "CONTENT_BLACK_LIST", - ) || - process.env.CONTENT_BLACK_LIST?.split(",").map((s) => s.trim()) || - [], - contentWhitelist: - getConfigValue( - ["processing", "filtering", "content_whitelist"], - "CONTENT_WHITE_LIST", - ) || - process.env.CONTENT_WHITE_LIST?.split(",").map((s) => s.trim()) || - [], - copyrightWhitelist: - getConfigValue( - ["processing", "filtering", "copyright_whitelist"], - "COPYRIGHT_WHITE_LIST", - ) || - process.env.COPYRIGHT_WHITE_LIST?.split(",").map(Number) || - [], - pidV2Whitelist: Array.isArray(pidV2Whitelist) - ? pidV2Whitelist - : typeof pidV2Whitelist === "string" - ? pidV2Whitelist.split(",").map(Number) - : [], + typeof contentBlacklist === "string" + ? contentBlacklist.split(",").map((entry) => entry.trim()) + : contentBlacklist, }, }; } diff --git a/src/config/schemas/whitelist.test.ts b/src/config/schemas/whitelist.test.ts new file mode 100644 index 0000000..cefe4ef --- /dev/null +++ b/src/config/schemas/whitelist.test.ts @@ -0,0 +1,78 @@ +import assert from "node:assert/strict"; +import test from "node:test"; +import { createWhitelistConfig, whitelistSchema } from "./whitelist.js"; + +test("whitelist sections use TOML before environment values", () => { + const toml = new Map([ + ["whitelist.video.type_ids", [28]], + ["whitelist.video.copyright_types", [1]], + ["whitelist.video.content_keywords", ["music"]], + ["whitelist.recommendation.pid_v2", [7]], + ["whitelist.minute_bootstrap.label_content_types", ["vocaloid"]], + ["whitelist.minute_bootstrap.label_origin", "rule"], + ["whitelist.minute_bootstrap.label_writers", ["classification_apply"]], + ["whitelist.minute_bootstrap.tid_v2", [2022]], + ]); + const result = createWhitelistConfig((path) => toml.get(path.join("."))); + assert.deepEqual(result, { + video: { typeIds: [28], copyrightTypes: [1], contentKeywords: ["music"] }, + recommendation: { pidV2: [7] }, + minuteBootstrap: { + labelContentTypes: ["vocaloid"], + labelOrigin: "rule", + labelWriters: ["classification_apply"], + tidV2: [2022], + }, + }); + assert.deepEqual(whitelistSchema.parse(result), result); +}); + +test("whitelist environment lists parse and defaults retain minute admission", () => { + const env = new Map([ + ["TYPE_ID_WHITE_LIST", "28, 30"], + ["COPYRIGHT_WHITE_LIST", "1, 2"], + ["CONTENT_WHITE_LIST", "music, dance"], + ["UPDATE_INFO_PID_V2_WHITELIST", "7, 11"], + ["MINUTE_BOOTSTRAP_LABEL_WRITERS", "writer_one, writer_two"], + ["MINUTE_BOOTSTRAP_TID_V2_ALLOWLIST", "2022, 2061"], + ]); + const result = createWhitelistConfig((_path, key) => env.get(key)); + assert.deepEqual(result.video, { + typeIds: [28, 30], + copyrightTypes: [1, 2], + contentKeywords: ["music", "dance"], + }); + assert.deepEqual(result.recommendation.pidV2, [7, 11]); + assert.deepEqual(result.minuteBootstrap, { + labelContentTypes: ["vocaloid", "maybe_vocaloid"], + labelOrigin: "rule", + labelWriters: ["writer_one", "writer_two"], + tidV2: [2022, 2061], + }); + assert.throws( + () => + createWhitelistConfig((_path, key) => + key === "UPDATE_INFO_PID_V2_WHITELIST" ? "7,,11" : undefined, + ), + /pidV2/, + ); +}); + +test("content keywords reject blank entries from environment and TOML", () => { + assert.throws( + () => + createWhitelistConfig((_path, key) => + key === "CONTENT_WHITE_LIST" ? "music, " : undefined, + ), + /contentKeywords/, + ); + assert.throws( + () => + createWhitelistConfig((path) => + path.join(".") === "whitelist.video.content_keywords" + ? ["music", " "] + : undefined, + ), + /contentKeywords/, + ); +}); diff --git a/src/config/schemas/whitelist.ts b/src/config/schemas/whitelist.ts new file mode 100644 index 0000000..ce2a537 --- /dev/null +++ b/src/config/schemas/whitelist.ts @@ -0,0 +1,137 @@ +import { z } from "zod"; + +const stringList = z.array(z.string()); +const numberList = z.array(z.number()); + +export const legacyWhitelistPaths = [ + ["processing.filtering.type_id_whitelist", "whitelist.video.type_ids"], + [ + "processing.filtering.copyright_whitelist", + "whitelist.video.copyright_types", + ], + [ + "processing.filtering.content_whitelist", + "whitelist.video.content_keywords", + ], + ["processing.filtering.pid_v2_whitelist", "whitelist.recommendation.pid_v2"], + [ + "minute.bootstrap_label_content_types", + "whitelist.minute_bootstrap.label_content_types", + ], + ["minute.bootstrap_label_origin", "whitelist.minute_bootstrap.label_origin"], + [ + "minute.bootstrap_label_writers", + "whitelist.minute_bootstrap.label_writers", + ], + ["minute.bootstrap_tid_v2_allowlist", "whitelist.minute_bootstrap.tid_v2"], +] as const; + +export const whitelistSchema = z.object({ + video: z.object({ + typeIds: numberList.default([]), + copyrightTypes: numberList.default([]), + contentKeywords: z.array(z.string().trim().min(1)).default([]), + }), + recommendation: z.object({ + pidV2: z.array(z.number().int().positive()).default([]), + }), + minuteBootstrap: z.object({ + labelContentTypes: stringList.default(["vocaloid", "maybe_vocaloid"]), + labelOrigin: z.string().default("rule"), + labelWriters: stringList.default([ + "classification_apply", + "classification_trigger", + ]), + tidV2: z.array(z.coerce.number().int()).default([2022, 2061]), + }), +}); + +export type WhitelistConfig = z.infer; + +type ConfigValue = ( + tomlPath: string[], + envKey: string, + // biome-ignore lint/suspicious/noExplicitAny: TOML/env values are validated by zod + defaultValue?: any, + // biome-ignore lint/suspicious/noExplicitAny: TOML/env values are validated by zod +) => any; + +function listValue(value: unknown): unknown { + return typeof value === "string" + ? value.split(",").map((entry) => entry.trim()) + : value; +} + +function numberListValue(value: unknown): unknown { + const entries = listValue(value); + return Array.isArray(entries) && typeof value === "string" + ? entries.map((entry) => (entry === "" ? Number.NaN : Number(entry))) + : entries; +} + +export function createWhitelistConfig( + getConfigValue: ConfigValue, +): WhitelistConfig { + return whitelistSchema.parse({ + video: { + typeIds: numberListValue( + getConfigValue( + ["whitelist", "video", "type_ids"], + "TYPE_ID_WHITE_LIST", + [], + ), + ), + copyrightTypes: numberListValue( + getConfigValue( + ["whitelist", "video", "copyright_types"], + "COPYRIGHT_WHITE_LIST", + [], + ), + ), + contentKeywords: listValue( + getConfigValue( + ["whitelist", "video", "content_keywords"], + "CONTENT_WHITE_LIST", + [], + ), + ), + }, + recommendation: { + pidV2: numberListValue( + getConfigValue( + ["whitelist", "recommendation", "pid_v2"], + "UPDATE_INFO_PID_V2_WHITELIST", + [], + ), + ), + }, + minuteBootstrap: { + labelContentTypes: listValue( + getConfigValue( + ["whitelist", "minute_bootstrap", "label_content_types"], + "MINUTE_BOOTSTRAP_LABEL_CONTENT_TYPES", + ["vocaloid", "maybe_vocaloid"], + ), + ), + labelOrigin: getConfigValue( + ["whitelist", "minute_bootstrap", "label_origin"], + "MINUTE_BOOTSTRAP_LABEL_ORIGIN", + "rule", + ), + labelWriters: listValue( + getConfigValue( + ["whitelist", "minute_bootstrap", "label_writers"], + "MINUTE_BOOTSTRAP_LABEL_WRITERS", + ["classification_apply", "classification_trigger"], + ), + ), + tidV2: numberListValue( + getConfigValue( + ["whitelist", "minute_bootstrap", "tid_v2"], + "MINUTE_BOOTSTRAP_TID_V2_ALLOWLIST", + [2022, 2061], + ), + ), + }, + }); +} diff --git a/src/database/index.ts b/src/database/index.ts index 702c800..f3c982e 100644 --- a/src/database/index.ts +++ b/src/database/index.ts @@ -350,10 +350,11 @@ export class Database { { bootstrapPriority: config.minute.bootstrapPriority, bootstrapTtlHours: config.minute.bootstrapTtlHours, - bootstrapLabelContentTypes: config.minute.bootstrapLabelContentTypes, - bootstrapLabelOrigin: config.minute.bootstrapLabelOrigin, - bootstrapLabelWriters: config.minute.bootstrapLabelWriters, - bootstrapTidV2Allowlist: config.minute.bootstrapTidV2Allowlist, + bootstrapLabelContentTypes: + config.whitelist.minuteBootstrap.labelContentTypes, + bootstrapLabelOrigin: config.whitelist.minuteBootstrap.labelOrigin, + bootstrapLabelWriters: config.whitelist.minuteBootstrap.labelWriters, + bootstrapTidV2Allowlist: config.whitelist.minuteBootstrap.tidV2, processedBackfillNewVideoAgeDays: config.minute.processedBackfillNewVideoAgeDays, }, @@ -371,10 +372,11 @@ export class Database { { bootstrapPriority: config.minute.bootstrapPriority, bootstrapTtlHours: config.minute.bootstrapTtlHours, - bootstrapLabelContentTypes: config.minute.bootstrapLabelContentTypes, - bootstrapLabelOrigin: config.minute.bootstrapLabelOrigin, - bootstrapLabelWriters: config.minute.bootstrapLabelWriters, - bootstrapTidV2Allowlist: config.minute.bootstrapTidV2Allowlist, + bootstrapLabelContentTypes: + config.whitelist.minuteBootstrap.labelContentTypes, + bootstrapLabelOrigin: config.whitelist.minuteBootstrap.labelOrigin, + bootstrapLabelWriters: config.whitelist.minuteBootstrap.labelWriters, + bootstrapTidV2Allowlist: config.whitelist.minuteBootstrap.tidV2, processedBackfillNewVideoAgeDays: config.minute.processedBackfillNewVideoAgeDays, }, diff --git a/src/database/schema/collection_state.ts b/src/database/schema/collection_state.ts index ad2760b..3bd895b 100644 --- a/src/database/schema/collection_state.ts +++ b/src/database/schema/collection_state.ts @@ -483,10 +483,10 @@ export async function initCollectionStateSchema(pool: Pool): Promise { p_now timestamptz DEFAULT now(), p_bootstrap_priority integer DEFAULT ${config.minute.bootstrapPriority}, p_bootstrap_ttl_hours integer DEFAULT ${config.minute.bootstrapTtlHours}, - p_bootstrap_label_content_types text[] DEFAULT ARRAY[${sqlTextArray(config.minute.bootstrapLabelContentTypes)}]::text[], - p_bootstrap_label_origin text DEFAULT '${sqlText(config.minute.bootstrapLabelOrigin)}', - p_bootstrap_label_writers text[] DEFAULT ARRAY[${sqlTextArray(config.minute.bootstrapLabelWriters)}]::text[], - p_bootstrap_tid_v2_allowlist integer[] DEFAULT ARRAY[${sqlIntegerArray(config.minute.bootstrapTidV2Allowlist)}]::integer[], + p_bootstrap_label_content_types text[] DEFAULT ARRAY[${sqlTextArray(config.whitelist.minuteBootstrap.labelContentTypes)}]::text[], + p_bootstrap_label_origin text DEFAULT '${sqlText(config.whitelist.minuteBootstrap.labelOrigin)}', + p_bootstrap_label_writers text[] DEFAULT ARRAY[${sqlTextArray(config.whitelist.minuteBootstrap.labelWriters)}]::text[], + p_bootstrap_tid_v2_allowlist integer[] DEFAULT ARRAY[${sqlIntegerArray(config.whitelist.minuteBootstrap.tidV2)}]::integer[], p_processed_backfill_new_video_age_days integer DEFAULT ${config.minute.processedBackfillNewVideoAgeDays} ) RETURNS text AS $$ DECLARE @@ -660,10 +660,10 @@ export async function initCollectionStateSchema(pool: Pool): Promise { p_now, ${config.minute.bootstrapPriority}, ${config.minute.bootstrapTtlHours}, - ARRAY[${sqlTextArray(config.minute.bootstrapLabelContentTypes)}]::text[], - '${sqlText(config.minute.bootstrapLabelOrigin)}', - ARRAY[${sqlTextArray(config.minute.bootstrapLabelWriters)}]::text[], - ARRAY[${sqlIntegerArray(config.minute.bootstrapTidV2Allowlist)}]::integer[], + ARRAY[${sqlTextArray(config.whitelist.minuteBootstrap.labelContentTypes)}]::text[], + '${sqlText(config.whitelist.minuteBootstrap.labelOrigin)}', + ARRAY[${sqlTextArray(config.whitelist.minuteBootstrap.labelWriters)}]::text[], + ARRAY[${sqlIntegerArray(config.whitelist.minuteBootstrap.tidV2)}]::integer[], ${config.minute.processedBackfillNewVideoAgeDays} ); END; diff --git a/src/index.ts b/src/index.ts index f356a3a..9ae7e51 100644 --- a/src/index.ts +++ b/src/index.ts @@ -104,7 +104,7 @@ async function main() { } = await import("./scripts/update-info"); const whitelistValue = updateInfoWhitelistArgument(args) ?? - config.processing.filtering.pidV2Whitelist.join(","); + config.whitelist.recommendation.pidV2.join(","); const where = parseUpdateInfoPredicateArgument(args); await runUpdateInfo({ pidV2Whitelist: parsePidV2Whitelist(whitelistValue), diff --git a/src/services/minute/minuteHandler.ts b/src/services/minute/minuteHandler.ts index 8530749..aead4ef 100644 --- a/src/services/minute/minuteHandler.ts +++ b/src/services/minute/minuteHandler.ts @@ -111,7 +111,7 @@ export class MinuteHandler { typeof id === "number" ? fetchVideoFullDetail({ aid: id }) : fetchVideoFullDetail({ bvid: id }), - pidV2Whitelist: new Set(config.processing.filtering.pidV2Whitelist), + pidV2Whitelist: new Set(config.whitelist.recommendation.pidV2), }); } diff --git a/src/utils/filter.ts b/src/utils/filter.ts index 5e80c6f..5ec5991 100644 --- a/src/utils/filter.ts +++ b/src/utils/filter.ts @@ -12,18 +12,16 @@ export const filterVideo = async ( ].join(" "); if ( - Array.isArray(config.processing.filtering.typeIdWhitelist) && - config.processing.filtering.typeIdWhitelist.length > 0 + Array.isArray(config.whitelist.video.typeIds) && + config.whitelist.video.typeIds.length > 0 ) { - if ( - !config.processing.filtering.typeIdWhitelist.includes(videoData.type_id) - ) { + if (!config.whitelist.video.typeIds.includes(videoData.type_id)) { let inwhite = false; if ( - Array.isArray(config.processing.filtering.contentWhitelist) && - config.processing.filtering.contentWhitelist.length > 0 + Array.isArray(config.whitelist.video.contentKeywords) && + config.whitelist.video.contentKeywords.length > 0 ) { - for (const keyword of config.processing.filtering.contentWhitelist) { + for (const keyword of config.whitelist.video.contentKeywords) { if (contentToCheck.includes(keyword.toLowerCase())) { logger.debug( `${videoData.bvid} 包含白名单关键字 "${keyword}",忽略类型检查: ${videoData.title}`, @@ -44,21 +42,17 @@ export const filterVideo = async ( // Check copyright whitelist if ( - Array.isArray(config.processing.filtering.copyrightWhitelist) && - config.processing.filtering.copyrightWhitelist.length > 0 && + Array.isArray(config.whitelist.video.copyrightTypes) && + config.whitelist.video.copyrightTypes.length > 0 && videoData.copyright !== undefined ) { - if ( - !config.processing.filtering.copyrightWhitelist.includes( - videoData.copyright, - ) - ) { + if (!config.whitelist.video.copyrightTypes.includes(videoData.copyright)) { let inwhite = false; if ( - Array.isArray(config.processing.filtering.contentWhitelist) && - config.processing.filtering.contentWhitelist.length > 0 + Array.isArray(config.whitelist.video.contentKeywords) && + config.whitelist.video.contentKeywords.length > 0 ) { - for (const keyword of config.processing.filtering.contentWhitelist) { + for (const keyword of config.whitelist.video.contentKeywords) { if (contentToCheck.includes(keyword.toLowerCase())) { logger.debug( `${videoData.bvid} 包含白名单关键字 "${keyword}",忽略版权检查: ${videoData.title}`,