feat(阶段3): 补齐用户侧依赖的三个 admin 端点
重判、提交统计、流程图统计这三条挂在旧后端的 admin 路由下,权限也确实是 teacher/super admin,但入口在用户侧页面里(提交列表页的重判按钮、两个统计面板)。 不做完,阶段 3 的出口标准「用户侧全部功能跑在新后端上」就不成立 —— 按 URL 前缀切阶段会漏掉它们。 - POST submissions/:id/rejudge ← GET admin/submission/rejudge - GET submissions/statistics ← GET admin/submission/statistics - GET flowcharts/statistics ← GET admin/flowchart/statistics 几处对齐旧后端的细节: - AST_CHECK_FAILED(10) 与 ACCEPTED(0) 同算通过 - 完成度先用原始花名册人数算、再修正 person_count,顺序照搬,兜住「学生已删号 但提交记录还在」 - 有提交但零通过的学生,两个名单里都不出现(旧后端同样口径) - 词云用 @node-rs/jieba,STOPWORDS 与 38 个自定义词逐词照搬;jieba@2 没有 insertWord,改用 loadDict 加载用户词典 - avgScore 分母是有分数的条数,对齐 Django Avg() 跳过 NULL rejudge 的 jobId 带时间戳。队列保留最近 100 个已完成任务,沿用 submissionId 做 jobId 的话 BullMQ 会认为任务已存在,重判会静默变成空操作。 correctRate 改成数值不带 %,展示格式化交给前端。stripClassPrefix 用 startsWith+slice 而不是 replace,前缀对不上时不会从中间截出乱码;site.ts 原有的 replace 写法一并改掉。 Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
This commit is contained in:
@@ -7,9 +7,10 @@ import {
|
||||
flowchartDetailSchema,
|
||||
flowchartListItemSchema,
|
||||
flowchartListSchema,
|
||||
flowchartStatisticsSchema,
|
||||
flowchartSubmissionSchema,
|
||||
} from "@oj2/contract"
|
||||
import { and, asc, count, desc, eq, ilike, sql } from "drizzle-orm"
|
||||
import { and, asc, count, desc, eq, ilike, isNull, sql } from "drizzle-orm"
|
||||
import { Hono } from "hono"
|
||||
|
||||
import { requireAuth, type AppEnv } from "../auth/middleware"
|
||||
@@ -17,7 +18,16 @@ import { config } from "../config"
|
||||
import { db, schema } from "../db"
|
||||
import { failure, success } from "../http"
|
||||
import { flowchartQueue } from "../queue"
|
||||
import { isAdminRole, objectValue, queryInteger, todayStart } from "./helpers"
|
||||
import { buildWordFrequencies } from "../services/word-frequency"
|
||||
import {
|
||||
isAdminRole,
|
||||
isTeacherOrAbove,
|
||||
objectValue,
|
||||
queryInteger,
|
||||
rounded,
|
||||
stripClassPrefix,
|
||||
todayStart,
|
||||
} from "./helpers"
|
||||
|
||||
export const flowchartRoutes = new Hono<AppEnv>()
|
||||
|
||||
@@ -127,6 +137,137 @@ flowchartRoutes.get("/flowcharts", requireAuth, async (c) => {
|
||||
}))
|
||||
})
|
||||
|
||||
const FLOWCHART_COMPLETED = 2
|
||||
|
||||
flowchartRoutes.get("/flowcharts/statistics", requireAuth, async (c) => {
|
||||
if (!isTeacherOrAbove(c.get("user"))) {
|
||||
return failure(c, 403, "permission-denied", "Teacher permission required")
|
||||
}
|
||||
const end = c.req.query("end")?.trim()
|
||||
if (!end) return failure(c, 400, "invalid-request", "end is required")
|
||||
const start = c.req.query("start")?.trim()
|
||||
|
||||
const filters = [
|
||||
eq(schema.flowchartSubmission.status, FLOWCHART_COMPLETED),
|
||||
sql`${schema.flowchartSubmission.createTime} <= ${end}`,
|
||||
]
|
||||
if (start) filters.push(sql`${schema.flowchartSubmission.createTime} >= ${start}`)
|
||||
|
||||
const displayId = c.req.query("problemId")?.trim()
|
||||
if (displayId) {
|
||||
const [problem] = await db
|
||||
.select({ id: schema.problem.id })
|
||||
.from(schema.problem)
|
||||
.where(and(
|
||||
sql`lower(${schema.problem.displayId}) = lower(${displayId})`,
|
||||
isNull(schema.problem.contestId),
|
||||
eq(schema.problem.visible, true),
|
||||
))
|
||||
.limit(1)
|
||||
if (!problem) return failure(c, 404, "problem-not-found", "Problem does not exist")
|
||||
filters.push(eq(schema.flowchartSubmission.problemId, problem.id))
|
||||
}
|
||||
|
||||
const username = c.req.query("username")?.trim()
|
||||
if (username) filters.push(ilike(schema.user.username, `%${username}%`))
|
||||
|
||||
// 只有指定了用户名才谈得上「班级人数」,不指定时分母无意义
|
||||
const roster = username
|
||||
? await db
|
||||
.select({ username: schema.user.username, className: schema.user.className })
|
||||
.from(schema.user)
|
||||
.where(and(
|
||||
ilike(schema.user.username, `%${username}%`),
|
||||
eq(schema.user.isDisabled, false),
|
||||
eq(schema.user.adminType, "Regular User"),
|
||||
))
|
||||
: []
|
||||
|
||||
const rows = await db
|
||||
.select({
|
||||
username: schema.user.username,
|
||||
score: schema.flowchartSubmission.aiScore,
|
||||
grade: schema.flowchartSubmission.aiGrade,
|
||||
criteria: schema.flowchartSubmission.aiCriteriaDetails,
|
||||
feedback: schema.flowchartSubmission.aiFeedback,
|
||||
suggestions: schema.flowchartSubmission.aiSuggestions,
|
||||
})
|
||||
.from(schema.flowchartSubmission)
|
||||
.innerJoin(schema.user, eq(schema.flowchartSubmission.userId, schema.user.id))
|
||||
.where(and(...filters))
|
||||
|
||||
const empty = {
|
||||
totalCount: 0,
|
||||
avgScore: 0,
|
||||
gradeDistribution: {},
|
||||
criteriaAverages: {},
|
||||
personCount: roster.length,
|
||||
completedCount: 0,
|
||||
wordFrequencies: [],
|
||||
dataUnaccepted: [],
|
||||
}
|
||||
if (rows.length === 0) return success(c, flowchartStatisticsSchema.parse(empty))
|
||||
|
||||
const gradeDistribution: Record<string, number> = {}
|
||||
const criteriaTotals = new Map<string, { sum: number; count: number; max: number }>()
|
||||
const texts: string[] = []
|
||||
const submitted = new Set<string>()
|
||||
let scoreSum = 0
|
||||
let scoreCount = 0
|
||||
|
||||
for (const row of rows) {
|
||||
submitted.add(row.username)
|
||||
// 旧后端用 values_list("ai_grade") 分组,null 也会成为一个桶;这里保持同样的口径
|
||||
const grade = row.grade ?? ""
|
||||
gradeDistribution[grade] = (gradeDistribution[grade] ?? 0) + 1
|
||||
if (row.score !== null) {
|
||||
scoreSum += row.score
|
||||
scoreCount += 1
|
||||
}
|
||||
for (const [key, value] of Object.entries(objectValue(row.criteria))) {
|
||||
const detail = objectValue(value)
|
||||
if (typeof detail.score !== "number") continue
|
||||
const bucket = criteriaTotals.get(key)
|
||||
if (bucket) {
|
||||
bucket.sum += detail.score
|
||||
bucket.count += 1
|
||||
} else {
|
||||
// max 取第一次见到的那条,与旧后端 `if key not in criteria_max` 一致
|
||||
criteriaTotals.set(key, {
|
||||
sum: detail.score,
|
||||
count: 1,
|
||||
max: typeof detail.max === "number" ? detail.max : 100,
|
||||
})
|
||||
}
|
||||
if (typeof detail.comment === "string" && detail.comment) texts.push(detail.comment)
|
||||
}
|
||||
if (row.feedback) texts.push(row.feedback)
|
||||
if (row.suggestions) texts.push(row.suggestions)
|
||||
}
|
||||
|
||||
const criteriaAverages: Record<string, { avg: number; max: number }> = {}
|
||||
for (const [key, bucket] of criteriaTotals) {
|
||||
criteriaAverages[key] = { avg: rounded(bucket.sum / bucket.count, 1), max: bucket.max }
|
||||
}
|
||||
|
||||
return success(c, flowchartStatisticsSchema.parse({
|
||||
totalCount: rows.length,
|
||||
// 分母是有分数的条数,不是总条数 —— 对齐 Django 的 Avg(),它跳过 NULL
|
||||
avgScore: scoreCount ? rounded(scoreSum / scoreCount, 1) : 0,
|
||||
gradeDistribution,
|
||||
criteriaAverages,
|
||||
personCount: roster.length,
|
||||
completedCount: submitted.size,
|
||||
wordFrequencies: buildWordFrequencies(texts),
|
||||
dataUnaccepted: roster
|
||||
.filter((row) => !submitted.has(row.username))
|
||||
.map((row) => ({
|
||||
username: row.username,
|
||||
realName: stripClassPrefix(row.username, row.className),
|
||||
})),
|
||||
}))
|
||||
})
|
||||
|
||||
flowchartRoutes.get("/flowcharts/:id", requireAuth, async (c) => {
|
||||
const [row] = await db.select({ flowchart: schema.flowchartSubmission, username: schema.user.username, problem: schema.problem })
|
||||
.from(schema.flowchartSubmission).innerJoin(schema.user, eq(schema.flowchartSubmission.userId, schema.user.id))
|
||||
|
||||
@@ -24,6 +24,22 @@ export function sampleUser(
|
||||
})
|
||||
}
|
||||
|
||||
/**
|
||||
* 去掉用户名里的 `ks<班级号>` 前缀,得到学生本人那一段:`ks251张三` + `251` → `张三`。
|
||||
* 对齐旧后端 `utils/shortcuts.py:52` 的 `strip_class_prefix`。
|
||||
*
|
||||
* 用 startsWith + slice 而不是 replace:replace 会删掉字符串中间的匹配,
|
||||
* 前缀对不上时从中间截出乱码。前缀不匹配就原样返回。
|
||||
*/
|
||||
export function stripClassPrefix(
|
||||
username: string,
|
||||
className: string | null | undefined,
|
||||
) {
|
||||
if (!className) return username
|
||||
const prefix = `ks${className}`
|
||||
return username.startsWith(prefix) ? username.slice(prefix.length) : username
|
||||
}
|
||||
|
||||
export function objectValue(value: unknown): Record<string, unknown> {
|
||||
return value && typeof value === "object" && !Array.isArray(value)
|
||||
? (value as Record<string, unknown>)
|
||||
|
||||
@@ -5,6 +5,7 @@ import { Hono } from "hono"
|
||||
import { db, schema } from "../db"
|
||||
import { failure, success } from "../http"
|
||||
import { getWebsiteOptions } from "../services/options"
|
||||
import { stripClassPrefix } from "./helpers"
|
||||
|
||||
export const siteRoutes = new Hono()
|
||||
|
||||
@@ -43,5 +44,6 @@ siteRoutes.get("/classes/:className/usernames", async (c) => {
|
||||
.from(schema.user)
|
||||
.where(eq(schema.user.className, className))
|
||||
.orderBy(desc(schema.user.createTime), asc(schema.user.id))
|
||||
return success(c, rows.map(({ username }) => username.replace(`ks${className}`, "")))
|
||||
// 用 stripClassPrefix 而不是 replace:replace 会把中间的匹配也删掉,前缀对不上时截出乱码
|
||||
return success(c, rows.map(({ username }) => stripClassPrefix(username, className)))
|
||||
})
|
||||
|
||||
@@ -9,6 +9,7 @@ import {
|
||||
submissionDetailSchema,
|
||||
submissionListItemSchema,
|
||||
submissionListSchema,
|
||||
submissionStatisticsSchema,
|
||||
} from "@oj2/contract"
|
||||
import { and, count, desc, eq, ilike, inArray, isNull, sql } from "drizzle-orm"
|
||||
import { Hono } from "hono"
|
||||
@@ -29,7 +30,15 @@ import {
|
||||
import { CodeFormatError, formatCode } from "../services/format-code"
|
||||
import { getBooleanOption } from "../services/options"
|
||||
import { consumeToken } from "../services/throttling"
|
||||
import { isAdminRole, queryInteger, todayStart } from "./helpers"
|
||||
import {
|
||||
isAdminRole,
|
||||
isSuperAdmin,
|
||||
isTeacherOrAbove,
|
||||
queryInteger,
|
||||
rounded,
|
||||
stripClassPrefix,
|
||||
todayStart,
|
||||
} from "./helpers"
|
||||
|
||||
export const submissionRoutes = new Hono<AppEnv>()
|
||||
|
||||
@@ -157,6 +166,196 @@ submissionRoutes.get("/submissions/today-count", async (c) => {
|
||||
return success(c, row?.value ?? 0)
|
||||
})
|
||||
|
||||
const ACCEPTED_RESULTS = [JudgeStatus.ACCEPTED, JudgeStatus.AST_CHECK_FAILED]
|
||||
|
||||
/**
|
||||
* 统计接口共用的时间窗解析。旧后端 `end` 必填、`start` 可选(不给就是「全部时段」)。
|
||||
*/
|
||||
function statisticsRange(c: { req: { query(name: string): string | undefined } }) {
|
||||
const end = c.req.query("end")?.trim()
|
||||
if (!end) return null
|
||||
const start = c.req.query("start")?.trim()
|
||||
return { start: start || null, end }
|
||||
}
|
||||
|
||||
/**
|
||||
* 按题号(展示用的 _id)定位公开题目。找不到时统计接口要报错而不是退化成「全部题目」,
|
||||
* 否则教师打错一个字就会看到全站数据还以为是本题的。
|
||||
*/
|
||||
async function findPublicProblemByDisplayId(displayId: string) {
|
||||
const [row] = await db
|
||||
.select({ id: schema.problem.id })
|
||||
.from(schema.problem)
|
||||
.where(
|
||||
and(
|
||||
sql`lower(${schema.problem.displayId}) = lower(${displayId})`,
|
||||
isNull(schema.problem.contestId),
|
||||
eq(schema.problem.visible, true),
|
||||
),
|
||||
)
|
||||
.limit(1)
|
||||
return row ?? null
|
||||
}
|
||||
|
||||
/**
|
||||
* 用户名模糊匹配到的在册学生,用来算「班级人数」和「谁没做」。
|
||||
* 只算未禁用的普通用户 —— 教师和管理员不该出现在完成度分母里。
|
||||
*/
|
||||
async function matchedStudents(username: string) {
|
||||
return db
|
||||
.select({ username: schema.user.username, className: schema.user.className })
|
||||
.from(schema.user)
|
||||
.where(
|
||||
and(
|
||||
ilike(schema.user.username, `%${username}%`),
|
||||
eq(schema.user.isDisabled, false),
|
||||
eq(schema.user.adminType, "Regular User"),
|
||||
),
|
||||
)
|
||||
}
|
||||
|
||||
submissionRoutes.get("/submissions/statistics", requireAuth, async (c) => {
|
||||
if (!isTeacherOrAbove(c.get("user"))) {
|
||||
return failure(c, 403, "permission-denied", "Teacher permission required")
|
||||
}
|
||||
const range = statisticsRange(c)
|
||||
if (!range) return failure(c, 400, "invalid-request", "end is required")
|
||||
|
||||
const filters = [
|
||||
isNull(schema.submission.contestId),
|
||||
sql`${schema.submission.createTime} <= ${range.end}`,
|
||||
]
|
||||
if (range.start) filters.push(sql`${schema.submission.createTime} >= ${range.start}`)
|
||||
|
||||
const displayId = c.req.query("problemId")?.trim()
|
||||
if (displayId) {
|
||||
const problem = await findPublicProblemByDisplayId(displayId)
|
||||
if (!problem) return failure(c, 404, "problem-not-found", "Problem does not exist")
|
||||
filters.push(eq(schema.submission.problemId, problem.id))
|
||||
}
|
||||
|
||||
const username = c.req.query("username")?.trim()
|
||||
if (username) filters.push(ilike(schema.submission.username, `%${username}%`))
|
||||
const where = and(...filters)
|
||||
|
||||
const acceptedFilter = sql`count(*) filter (where ${inArray(schema.submission.result, ACCEPTED_RESULTS)})`
|
||||
|
||||
const [[totals], perUser, rosterRows, items] = await Promise.all([
|
||||
db
|
||||
.select({ total: count(), accepted: acceptedFilter.mapWith(Number) })
|
||||
.from(schema.submission)
|
||||
.where(where),
|
||||
db
|
||||
.select({
|
||||
username: schema.submission.username,
|
||||
submissionCount: count(),
|
||||
acceptedCount: acceptedFilter.mapWith(Number),
|
||||
})
|
||||
.from(schema.submission)
|
||||
.where(where)
|
||||
.groupBy(schema.submission.username)
|
||||
.orderBy(desc(count())),
|
||||
// 只有指定了用户名才有「班级人数」这个概念;不指定时分母无意义,旧后端也返回 0
|
||||
username ? matchedStudents(username) : Promise.resolve([]),
|
||||
db
|
||||
.select({
|
||||
username: schema.submission.username,
|
||||
id: schema.submission.id,
|
||||
result: schema.submission.result,
|
||||
})
|
||||
.from(schema.submission)
|
||||
.where(where)
|
||||
.orderBy(desc(schema.submission.createTime)),
|
||||
])
|
||||
|
||||
const submissionCount = totals?.total ?? 0
|
||||
const acceptedCount = totals?.accepted ?? 0
|
||||
|
||||
const itemsByUser = new Map<string, { id: string; result: number }[]>()
|
||||
for (const item of items) {
|
||||
const bucket = itemsByUser.get(item.username)
|
||||
if (bucket) bucket.push({ id: item.id, result: item.result })
|
||||
else itemsByUser.set(item.username, [{ id: item.id, result: item.result }])
|
||||
}
|
||||
|
||||
const submittedUsernames = new Set(perUser.map((row) => row.username))
|
||||
const classNames = new Map<string, string | null>()
|
||||
if (submittedUsernames.size) {
|
||||
const rows = await db
|
||||
.select({ username: schema.user.username, className: schema.user.className })
|
||||
.from(schema.user)
|
||||
.where(inArray(schema.user.username, [...submittedUsernames]))
|
||||
for (const row of rows) classNames.set(row.username, row.className)
|
||||
}
|
||||
|
||||
// 只列出有正确提交的人。做了但一次没对的学生落在「未完成」那一栏
|
||||
const data = perUser
|
||||
.filter((row) => row.acceptedCount > 0)
|
||||
.map((row) => ({
|
||||
username: row.username,
|
||||
className: classNames.get(row.username) ?? null,
|
||||
submissionCount: row.submissionCount,
|
||||
acceptedCount: row.acceptedCount,
|
||||
correctRate: rounded((row.acceptedCount / row.submissionCount) * 100),
|
||||
submissionItems: itemsByUser.get(row.username) ?? [],
|
||||
}))
|
||||
|
||||
const dataUnaccepted = rosterRows
|
||||
.filter((row) => !submittedUsernames.has(row.username))
|
||||
.map((row) => ({
|
||||
username: row.username,
|
||||
realName: stripClassPrefix(row.username, row.className),
|
||||
}))
|
||||
|
||||
// 顺序照搬旧后端:先用原始 person_count 算完成度,再修正 person_count。
|
||||
// 修正是为了兜住「学生已删号但提交记录还在」——那时完成人数会大于花名册人数。
|
||||
let personCount = rosterRows.length
|
||||
let personRate = 0
|
||||
if (personCount) {
|
||||
personRate = Math.min(100, rounded((data.length / personCount) * 100))
|
||||
if (personCount < data.length) personCount = data.length
|
||||
}
|
||||
|
||||
return success(
|
||||
c,
|
||||
submissionStatisticsSchema.parse({
|
||||
submissionCount,
|
||||
acceptedCount,
|
||||
correctRate: submissionCount ? rounded((acceptedCount / submissionCount) * 100) : 0,
|
||||
personCount,
|
||||
personRate,
|
||||
data,
|
||||
dataUnaccepted,
|
||||
}),
|
||||
)
|
||||
})
|
||||
|
||||
submissionRoutes.post("/submissions/:id/rejudge", requireAuth, async (c) => {
|
||||
if (!isSuperAdmin(c.get("user"))) {
|
||||
return failure(c, 403, "permission-denied", "Super admin permission required")
|
||||
}
|
||||
const [row] = await db
|
||||
.select({ id: schema.submission.id, problemId: schema.submission.problemId })
|
||||
.from(schema.submission)
|
||||
.where(and(eq(schema.submission.id, c.req.param("id")), isNull(schema.submission.contestId)))
|
||||
.limit(1)
|
||||
if (!row) return failure(c, 404, "submission-not-found", "Submission does not exist")
|
||||
|
||||
await db
|
||||
.update(schema.submission)
|
||||
.set({ statisticInfo: {}, result: JudgeStatus.PENDING })
|
||||
.where(eq(schema.submission.id, row.id))
|
||||
|
||||
// jobId 必须带时间戳。队列保留最近 100 个已完成任务,沿用 submissionId 做 jobId 的话
|
||||
// BullMQ 会认为这个任务已经存在,重判静默变成空操作。与 flowcharts/:id/retry 同一处理。
|
||||
await judgeQueue.add(
|
||||
"judge",
|
||||
{ submissionId: row.id, problemId: row.problemId },
|
||||
{ jobId: `${row.id}:rejudge:${Date.now()}` },
|
||||
)
|
||||
return success(c, null)
|
||||
})
|
||||
|
||||
submissionRoutes.post("/code/format", requireAuth, async (c) => {
|
||||
const parsed = formatCodeRequestSchema.safeParse(await c.req.json().catch(() => null))
|
||||
if (!parsed.success) return failure(c, 400, "invalid-request", "Invalid format payload")
|
||||
|
||||
67
apps/api/src/services/word-frequency.ts
Normal file
67
apps/api/src/services/word-frequency.ts
Normal file
@@ -0,0 +1,67 @@
|
||||
import { Jieba } from "@node-rs/jieba"
|
||||
import { dict } from "@node-rs/jieba/dict"
|
||||
|
||||
/**
|
||||
* 流程图评语词云的分词。对齐旧后端 `flowchart/views/admin.py` 的
|
||||
* STOPWORDS / CUSTOM_WORDS / _build_word_frequencies 三段。
|
||||
*
|
||||
* 停用词表逐词照搬,改一个词就会让词云和旧后端对不上 —— 教师是拿它横向比不同班级的,
|
||||
* 词表变了历史截图就没法比。
|
||||
*/
|
||||
const STOPWORDS = new Set(
|
||||
(
|
||||
"的 了 是 在 和 有 就 不 也 都 要 会 这 那 到 说 上 为 与 及 等 " +
|
||||
"把 被 从 而 所 但 如 又 或 很 更 还 让 对 已 向 只 能 以 中 可以 " +
|
||||
"可能 需要 没有 使用 进行 注意 建议 应该 考虑 整体 基本 部分 " +
|
||||
"一个 一些 一下 一定 一种 这个 所有 其他 比较 存在 明确 " +
|
||||
"正确 良好 清晰 合理 较好 不错 符合 标准 "
|
||||
)
|
||||
.split(" ")
|
||||
.filter(Boolean),
|
||||
)
|
||||
|
||||
const CUSTOM_WORDS = [
|
||||
"循环结构", "条件判断", "判断条件", "结束条件", "循环条件",
|
||||
"异常处理", "边界条件", "输入输出", "输入验证", "开始结束",
|
||||
"结束节点", "开始节点", "判断节点", "流程走向", "逻辑错误",
|
||||
"逻辑缺陷", "逻辑不清", "缺少分支", "缺少步骤", "缺少判断",
|
||||
"缺少循环", "死循环", "无限循环", "循环出口", "循环体",
|
||||
"条件分支", "分支结构", "分支不全", "分支缺失", "符号使用",
|
||||
"符号不规范", "连线混乱", "变量初始化", "赋值操作", "累加操作",
|
||||
"终止条件", "退出条件", "返回值",
|
||||
]
|
||||
|
||||
/**
|
||||
* 词典加载有一次性开销(约 100ms),放在模块级会拖慢 API 冷启动,
|
||||
* 而词云只有教师偶尔点一次。改成首次调用时才建。
|
||||
*/
|
||||
let instance: Jieba | null = null
|
||||
|
||||
function jieba() {
|
||||
if (instance) return instance
|
||||
const built = Jieba.withDict(dict)
|
||||
// 对应旧后端的 jieba.add_word(w, freq=9999)。
|
||||
// @node-rs/jieba@2 没有导出 insertWord/addWord,改用用户词典缓冲区,格式为「词 词频」。
|
||||
built.loadDict(
|
||||
Buffer.from(CUSTOM_WORDS.map((word) => `${word} 9999`).join("\n") + "\n"),
|
||||
)
|
||||
instance = built
|
||||
return built
|
||||
}
|
||||
|
||||
export function buildWordFrequencies(texts: string[], topN = 80) {
|
||||
const counter = new Map<string, number>()
|
||||
const cutter = jieba()
|
||||
for (const raw of texts) {
|
||||
const text = raw.replaceAll("【重点】", "")
|
||||
for (const token of cutter.cut(text)) {
|
||||
const word = token.trim()
|
||||
if (word.length < 2 || STOPWORDS.has(word)) continue
|
||||
counter.set(word, (counter.get(word) ?? 0) + 1)
|
||||
}
|
||||
}
|
||||
return [...counter]
|
||||
.sort((a, b) => b[1] - a[1])
|
||||
.slice(0, topN)
|
||||
.map(([word, count]) => ({ word, count }))
|
||||
}
|
||||
Reference in New Issue
Block a user