import JSZip from 'jszip' /** 目录中的一行,匹配到正文后携带内容与体量。 */ export interface BookEntry { index: number title: string content: string charCount: number matched: boolean } /** AI 划分出的课时,引用其所辖的目录条目下标。 */ export interface ProposedLesson { title: string sourceIndexes: number[] } /** * 归一化标题/文件名以便匹配:去 .md 后缀、trim、折叠空白、 * 把 sanitizeFilename 会替换的非法字符统一成下划线(与服务端一致)。 */ export function normalizeTitle(value: string): string { return value .replace(/\.md$/i, '') .replace(/[\\/:*?"<>|]/g, '_') .replace(/\s+/g, ' ') .trim() } /** 解压 ZIP,按归一化文件名建立 { 标题 → 正文 } 映射(仅取 .md 文件)。 */ export async function parseZip(file: File | Blob): Promise> { const zip = await JSZip.loadAsync(file) const map = new Map() const entries = Object.values(zip.files).filter( (entry) => !entry.dir && /\.md$/i.test(entry.name), ) for (const entry of entries) { // 文件名可能带目录前缀,只取最后一段。 const base = entry.name.split('/').pop() ?? entry.name const key = normalizeTitle(base) if (key) { map.set(key, await entry.async('text')) } } return map } /** 把目录文本逐行与正文映射匹配,得到带内容与体量的条目列表。 */ export function matchToc(tocText: string, contentMap: Map): BookEntry[] { return tocText .split('\n') .map((line) => line.trim()) .filter(Boolean) .map((title, index) => { const content = contentMap.get(normalizeTitle(title)) ?? '' return { index, title, content, charCount: content.length, matched: content !== '', } }) } /** 按课时的 sourceIndexes 拼接其所辖目录条目的正文。 */ export function assembleContent(lesson: ProposedLesson, entries: readonly BookEntry[]): string { return lesson.sourceIndexes .map((i) => entries.find((entry) => entry.index === i)?.content ?? '') .filter(Boolean) .join('\n\n') }