import type { Token } from '../core/types'; import type { Breakpoint } from '../breakpoints/types'; import { joinTokens } from '../core/joinTokens'; import { measureSliceWidthCached } from '../measure/measureSliceWidthCached'; import { buildTieKey, scoreLayout } from '../scoring/index'; import { DEFAULT_TOO_LONG_THRESHOLDS, isLineStartPunctTC, isOverWidth, isTooLong } from './constraints'; import { insertTopK } from './topK'; import type { BestLayout, PartialLayout, SearchAppConfig, SearchAppInput, SearchAppResult } from './types'; const DEFAULT_CONFIG: SearchAppConfig = { topK: 10, tooLongThresholds: DEFAULT_TOO_LONG_THRESHOLDS, }; function mergeConfig(overrides?: Partial | null): SearchAppConfig { if (!overrides) return { ...DEFAULT_CONFIG }; return { topK: overrides.topK ?? DEFAULT_CONFIG.topK, tooLongThresholds: overrides.tooLongThresholds ?? DEFAULT_CONFIG.tooLongThresholds, }; } function uniqueSortedPositions(bps: Breakpoint[], n: number): number[] { const set = new Set(); for (const b of bps) { const p = b.pos | 0; if (p >= 1 && p <= n - 1) set.add(p); } const arr = Array.from(set); arr.sort((a, b) => a - b); return arr; } function rawSeparatorsForLang(tokensLen: number, lang: 'TC' | 'EN'): string[] | undefined { if (lang !== 'TC') return undefined; // TC:默认拼接不插入额外空格;空格作为 token 自身出现 return Array.from({ length: Math.max(0, tokensLen) }, () => ''); } function buildLineText(tokens: Token[], start: number, end: number, rawSeparators?: string[]): string { return joinTokens(tokens, start, end, rawSeparators); } function buildBestLayout(best: PartialLayout, debug?: boolean): BestLayout { const lines = best.lines.map((l) => l.text); const wrappedText = lines.join('\n'); const meta = debug ? { breaks: best.breaks, scoreTopTerms: best.scoreTopTerms, score: best.score } : { breaks: best.breaks }; return { breaks: best.breaks, lines, wrappedText, meta }; } export async function searchBestLayoutApp(input: SearchAppInput): Promise { const cfg = mergeConfig(input.config ?? null); const tokens = input.tokens ?? []; const lang = input.lang; const n = tokens.length; if (isTooLong(tokens, lang, cfg.tooLongThresholds)) { return { ok: false, reason: 'TOO_LONG', meta: { tokenCount: n } }; } const maxLines = Math.max(1, input.maxLines | 0); const availableWidth = input.availableWidth; const positions = uniqueSortedPositions(input.breakpoints ?? [], n); const rawSep = rawSeparatorsForLang(tokens.length, lang); // dp[pos][linesUsed] -> PartialLayout[] const dp: PartialLayout[][][] = Array.from({ length: n + 1 }, () => Array.from({ length: maxLines + 1 }, () => []) ); dp[0][0] = [ { pos: 0, breaks: [], lines: [], score: 0, tieKey: [0, 0, 0, 0, 0], // 空布局占位,后续会用 buildTieKey 覆盖 }, ]; const tcPunctuations = input.scoring.config.tcPunctuations ?? [',', '。', '!', '?', ';', ':', '、']; // DP 遍历顺序必须固定 for (let pos = 0; pos <= n; pos++) { for (let linesUsed = 0; linesUsed <= maxLines - 1; linesUsed++) { const states = dp[pos][linesUsed]; if (!states || states.length === 0) continue; // 枚举 nextPos:breakpoints 中 >pos 的位置(升序)+ N const nextList: number[] = []; for (const p of positions) if (p > pos) nextList.push(p); if (n > pos) nextList.push(n); for (const state of states) { for (const nextPos of nextList) { if (nextPos <= pos) continue; // 禁止空行 // H6:TC 禁止“行首标点”(断点导致下一行第一个 token 为标点) // 断点位置为 nextPos,因此要检查 tokens[nextPos] if (lang === 'TC' && nextPos < n && isLineStartPunctTC(tokens, nextPos, tcPunctuations)) { continue; } const lineText = buildLineText(tokens, pos, nextPos, rawSep); const widthRes = await measureSliceWidthCached({ tokens, start: pos, end: nextPos, context: 'APP', contextProfile: input.measure.contextProfile, fontSpec: input.measure.fontSpec, measureWidthImpl: input.measure.measureWidthImpl, rawSeparators: rawSep, }); if (widthRes.width === null) { // App 搜索必须依赖宽度派;宽度不可用交给 overflow-fallback 统一处理 return { ok: false, reason: widthRes.meta.reason === 'MEASURE_FAILED' ? 'MEASURE_FAILED' : 'WIDTH_UNKNOWN', meta: { tokenCount: n }, }; } const lineWidth = widthRes.width; if (isOverWidth(lineWidth, availableWidth)) continue; // H1 const tokenCount = nextPos - pos; const charCount = lang === 'TC' ? tokenCount : 0; const newLine = { start: pos, end: nextPos, text: lineText, width: lineWidth, tokenCount, charCount, }; const newLines = state.lines.concat([newLine]); const newBreaks = nextPos === n ? state.breaks : state.breaks.concat([nextPos]); // 构造 layoutCandidate 并评分(为保证确定性,首版直接全量评分) const layoutCandidate = { breaks: newBreaks, lines: newLines }; const scored = scoreLayout({ tokens, layoutCandidate: layoutCandidate as any, lang, context: 'APP', availableWidth, config: input.scoring.config, lexicons: input.scoring.lexicons, debug: Boolean(input.scoring.debug), }); const tieKey = buildTieKey({ scoredLayout: scored as any, layoutCandidate: layoutCandidate as any, lang, context: 'APP', tokenCount: n, availableWidth, idealWidthRatio: input.scoring.config.idealWidthRatio, }) as number[]; const cand: PartialLayout = { pos: nextPos, breaks: newBreaks, lines: newLines, score: scored.score, tieKey, scoreTopTerms: scored.scoreBreakdown?.terms, }; dp[nextPos][linesUsed + 1] = insertTopK(dp[nextPos][linesUsed + 1], cand, cfg.topK); } } } } // 结束选择 const collect: PartialLayout[] = []; if (input.lineMode === 'FIXED') { collect.push(...dp[n][maxLines]); } else { for (let linesUsed = 1; linesUsed <= maxLines; linesUsed++) { collect.push(...dp[n][linesUsed]); } } if (collect.length === 0) { return { ok: false, reason: 'NO_CANDIDATE', meta: { tokenCount: n } }; } // TopK 容器内本身已排序,但跨不同 linesUsed 需要再全局选最优 collect.sort((a, b) => { // 复用 insertTopK 的 compare 规则(在 topK.ts 内部)会更好,但这里直接再排序一次确保稳定 if (a.score !== b.score) return b.score - a.score; const na = a.tieKey; const nb = b.tieKey; const m = Math.min(na.length, nb.length); for (let i = 0; i < m; i++) { const av = na[i] ?? 0; const bv = nb[i] ?? 0; if (av < bv) return -1; if (av > bv) return 1; } return na.length - nb.length; }); const best = collect[0]!; return { ok: true, bestLayout: buildBestLayout(best, Boolean(input.scoring.debug)) }; }