#!/usr/bin/env node /** * 批量 OCR 评价截图 → 匹配 DB 评价 → 写入 screenshot 字段 * * 流程: * 1. 扫描 public/uploads/reviews/ 的 jpg * 2. tesseract chi_sim OCR * 3. 过滤:包含手机号(1[3-9]\d{9})或负面关键词的截图丢弃 * 4. 规整文本,按 n-gram 交集相似度匹配 82 条 DB 评价 * 5. 命中 → UPDATE reviews SET screenshot=/uploads/reviews/xxx.jpg * 6. 未命中 → 报告供人工处理 * * 不自动创建新评价:避免 OCR 噪声/差评混入库。 */ import Database from 'better-sqlite3' import { execFileSync } from 'node:child_process' import { readdirSync, existsSync, writeFileSync } from 'node:fs' import { join, dirname } from 'node:path' import { fileURLToPath } from 'node:url' const __dirname = dirname(fileURLToPath(import.meta.url)) const ROOT = join(__dirname, '..') const DB_PATH = join(ROOT, 'server/database/hulai.db') const SHOTS_DIR = join(ROOT, 'public/uploads/reviews') const REL_PREFIX = '/uploads/reviews/' const PHONE_RE = /\b1[3-9]\d{9}\b/ const NEG_KEYWORDS = ['差评', '投诉', '退款', '退一赔', '不满意', '很失望', '骗人', '被骗', '黑心', '垃圾', '坑人', '太烂', '烂透', '糟糕', '欺诈', '维权', '曝光'] function ocr(imagePath) { try { const out = execFileSync('tesseract', [imagePath, 'stdout', '-l', 'chi_sim'], { encoding: 'utf8', stdio: ['ignore', 'pipe', 'ignore'], maxBuffer: 10 * 1024 * 1024, }) return out } catch { return '' } } // 清理 OCR 文本:去空白/标点/数字/半角字母,只留中文 function canonical(s) { return (s || '').replace(/[^\u4e00-\u9fa5]/g, '') } // 生成 bigram 集合 function bigrams(s) { const set = new Set() for (let i = 0; i < s.length - 1; i++) set.add(s.slice(i, i + 2)) return set } // bigram Jaccard 相似度(0~1) function similarity(a, b) { if (!a || !b) return 0 const A = bigrams(a), B = bigrams(b) if (A.size === 0 || B.size === 0) return 0 let inter = 0 for (const x of A) if (B.has(x)) inter++ return inter / Math.min(A.size, B.size) // use min instead of union: 截图文本往往比润色后的 DB 内容多噪声 } // 最长公共子串长度(粗粒度,用于 tie-break) function lcsLen(a, b) { // 限制长度避免 O(n²) 爆炸 if (a.length > 500) a = a.slice(0, 500) if (b.length > 500) b = b.slice(0, 500) const n = a.length, m = b.length if (!n || !m) return 0 let prev = new Array(m + 1).fill(0) let curr = new Array(m + 1).fill(0) let best = 0 for (let i = 1; i <= n; i++) { for (let j = 1; j <= m; j++) { curr[j] = a[i - 1] === b[j - 1] ? prev[j - 1] + 1 : 0 if (curr[j] > best) best = curr[j] } ;[prev, curr] = [curr, prev] } return best } const db = new Database(DB_PATH) const reviews = db.prepare('SELECT id, author, content, screenshot FROM reviews').all() console.log(`DB reviews: ${reviews.length}`) const reviewsIdx = reviews.map(r => ({ ...r, canon: canonical(r.content) })) const files = readdirSync(SHOTS_DIR).filter(f => /\.(jpe?g|png)$/i.test(f)) console.log(`screenshots: ${files.length}\n`) const SIM_THRESHOLD = 0.35 const LCS_MIN = 10 const results = { matched: [], filteredPhone: [], filteredNeg: [], unmatched: [], ocrFailed: [] } const assignedReviewIds = new Set() let i = 0 for (const file of files) { i++ const abs = join(SHOTS_DIR, file) process.stdout.write(`[${i}/${files.length}] ${file} ... `) const text = ocr(abs) if (!text || text.length < 20) { console.log('OCR 失败/太短') results.ocrFailed.push(file) continue } if (PHONE_RE.test(text)) { console.log('含手机号 → 跳过') results.filteredPhone.push({ file, snippet: text.match(PHONE_RE)[0] }) continue } const hitNeg = NEG_KEYWORDS.find(k => text.includes(k)) if (hitNeg) { console.log(`负面关键词「${hitNeg}」→ 跳过`) results.filteredNeg.push({ file, keyword: hitNeg }) continue } const canon = canonical(text) if (canon.length < 30) { console.log('有效中文太少') results.ocrFailed.push(file) continue } // 评分 let best = null for (const r of reviewsIdx) { const sim = similarity(canon, r.canon) if (!best || sim > best.sim) best = { ...r, sim } } if (best && best.sim >= SIM_THRESHOLD) { const lcs = lcsLen(canon, best.canon) if (lcs >= LCS_MIN) { if (assignedReviewIds.has(best.id)) { // 已分配给更高分的图 console.log(`评价#${best.id} 已占,放入未匹配`) results.unmatched.push({ file, bestId: best.id, sim: best.sim.toFixed(3), reason: 'dup' }) continue } assignedReviewIds.add(best.id) results.matched.push({ file, id: best.id, author: best.author, sim: best.sim.toFixed(3), lcs }) console.log(`→ #${best.id} ${best.author} sim=${best.sim.toFixed(3)} lcs=${lcs}`) continue } } console.log(`未匹配 (best sim=${best?.sim.toFixed(3)})`) results.unmatched.push({ file, bestId: best?.id, sim: best?.sim.toFixed(3) }) } // 写入 DB const update = db.prepare('UPDATE reviews SET screenshot = ? WHERE id = ?') const tx = db.transaction(rows => rows.forEach(r => update.run(REL_PREFIX + r.file, r.id))) tx(results.matched) // 过滤掉手机/负面的截图文件直接删除 import('node:fs').then(({ unlinkSync }) => { for (const { file } of [...results.filteredPhone, ...results.filteredNeg]) { try { unlinkSync(join(SHOTS_DIR, file)) } catch {} } }) writeFileSync(join(ROOT, 'scripts/.screenshot-import-report.json'), JSON.stringify(results, null, 2)) console.log('\n===== 汇总 =====') console.log(`命中 DB 并写入: ${results.matched.length}`) console.log(`含手机号已跳过删除: ${results.filteredPhone.length}`) console.log(`负面关键词已跳过删除: ${results.filteredNeg.length}`) console.log(`未匹配: ${results.unmatched.length}`) console.log(`OCR 失败: ${results.ocrFailed.length}`) console.log(`\n详细报告: scripts/.screenshot-import-report.json`)