Commit 62d9b71a by luoqi

merge: 残根目标类目 + 复查证据 + 含词大小写补词 + reparse 回溯修复 → main

parents 812d1613 7ad3aa93
Pipeline #3603 failed in 0 seconds
...@@ -23,7 +23,11 @@ keyword_mapping: ...@@ -23,7 +23,11 @@ keyword_mapping:
# 修复/备牙/取模:收 C.10 治疗痕迹(修复后/修复术后/全瓷桥修复后/备牙后/重新取模后)。 # 修复/备牙/取模:收 C.10 治疗痕迹(修复后/修复术后/全瓷桥修复后/备牙后/重新取模后)。
# 排在 implant/endo/ortho/cosmetic 之后 → 种植修复→implant、根管…修复→endo 仍被前面抢走。 # 排在 implant/endo/ortho/cosmetic 之后 → 种植修复→implant、根管…修复→endo 仍被前面抢走。
# "嵌体修复"等 restorative 已由上方精确 enum 先命中,不受裸"修复"影响。 # "嵌体修复"等 restorative 已由上方精确 enum 先命中,不受裸"修复"影响。
- { value: prosthodontic, any: [, , 义齿, 修复体, 修复, 备牙, 取模, 桩核, 桩冠, 戴牙, 全瓷, 烤瓷, 重新粘接] } # 加牙/活托(2026-08-27 过碧霞 TS0B006258 牙12;13;21):活动义齿在原基托上加牙 —— DW 实测
# 裸「加牙」58 条(50 带牙位)全部落 _default 被丢,患者整次修复在系统里不存在 → 误召缺失牙。
# 变体「活动义齿加牙/义齿加牙/取模加牙」已被 义齿/取模 收,只有裸词和「活托X」漏。
# 碰撞已验:「根管治疗加牙周治疗」含"加牙"但被上方 ② endodontic 的「根管」先吃,安全。
- { value: prosthodontic, any: [, , 义齿, 修复体, 修复, 备牙, 取模, 桩核, 桩冠, 戴牙, 全瓷, 烤瓷, 重新粘接, 加牙, 活托] }
# ⭐ 明确充填动作(充填/去腐/备洞/嵌体/垫底/补牙)压过 preventive: # ⭐ 明确充填动作(充填/去腐/备洞/嵌体/垫底/补牙)压过 preventive:
# "去腐,备洞,...树脂充填,窝沟封闭"(dispose 散文)是补牙(restorative),不能因含"窝沟封闭" # "去腐,备洞,...树脂充填,窝沟封闭"(dispose 散文)是补牙(restorative),不能因含"窝沟封闭"
# 被误判 preventive → 非 resolver → 误召(马思煦@46)。注:"开髓去腐"等根管语境已被上方 endodontic 先收。 # 被误判 preventive → 非 resolver → 误召(马思煦@46)。注:"开髓去腐"等根管语境已被上方 endodontic 先收。
...@@ -33,11 +37,25 @@ keyword_mapping: ...@@ -33,11 +37,25 @@ keyword_mapping:
# 材料弱词:放 preventive 后 —— "玻璃离子窝沟封闭" 已被上面 preventive 收;裸"树脂/玻璃离子"→ restorative # 材料弱词:放 preventive 后 —— "玻璃离子窝沟封闭" 已被上面 preventive 收;裸"树脂/玻璃离子"→ restorative
- { value: restorative, any: [树脂, 玻璃离子] } - { value: restorative, any: [树脂, 玻璃离子] }
# 牙周去裸"洁牙"(避开自由文本"清洁牙面"误命中);洁牙/全口洁牙等精确词由上方 enum_mapping 兜 # 牙周去裸"洁牙"(避开自由文本"清洁牙面"误命中);洁牙/全口洁牙等精确词由上方 enum_mapping 兜
- { value: periodontic, any: [洁治, 洗牙, 龈上, 龈下, 刮治, 牙周, 喷砂, 细洁] } # SRP = scaling & root planing(龈下刮治+根面平整)的国际通用缩写,正是 perio_no_srp 子场景
# 要认的那个治疗。DW 实测 SRP 648 + 局部SRP 40 + SRP+冲洗上药 76,此前全部落 _default 被丢 ——
# 患者做完牙周基础治疗仍被召「牙周炎未做基础治疗」。含 SRP 的 DW 全量已扫,无非牙周语义。
- { value: periodontic, any: [洁治, 洗牙, 龈上, 龈下, 刮治, 牙周, 喷砂, 细洁, SRP] }
- { value: preventive, any: [涂氟, 防龋, OHI, 口腔卫生宣教] } - { value: preventive, any: [涂氟, 防龋, OHI, 口腔卫生宣教] }
- { value: surgical, any: [拔除, 拔牙, 切开, 翻瓣, 切除, 系带, 脓肿, 囊肿, 植骨] } # 缝线:「拆线」已在宿主精确表(拆线→surgical,37,253 条正常落库),但措辞变体
# 「拆除缝线」699 条(665 带牙位)漏了 —— 同一临床动作。收「缝线」一并覆盖变体。
# 碰撞已验:「拆除」不含「拔除/切除」,不会被本行前面的词误命中。
- { value: surgical, any: [拔除, 拔牙, 切开, 翻瓣, 切除, 系带, 脓肿, 囊肿, 植骨, 缝线] }
- { value: pediatric, any: [乳牙, 儿童, 年轻恒牙] } - { value: pediatric, any: [乳牙, 儿童, 年轻恒牙] }
# ⛔ 刻意不收(2026-08-27 评估过,别"顺手补全"):
# 冲洗 / 换药 / 上药(~3,500 条)—— **对症处置 ≠ 根治**。阻生牙冠周炎「冲洗上药」之后那颗牙
# 仍然要拔;归 surgical 会把 K01 阻生牙缺口消掉 = 该召的不召(少召不报错,最难发现)。
# 调磨(817)/ 试戴(213)—— 姑息调整 / 修复流程中途,都不代表问题已解决。
# ⚠️ 根因:本表的 category 同时被两处消费 —— ①治疗史完整性 ②resolver 判"缺口已解"。
# 对这几个词我们想要①不想要②,而现在只有一个旋钮,只能二选一(现选"都不要"=丢弃)。
# 要两全需在 category 之外再加一维"是否根治",那是更大的改动,不在本表解决。
# actual 语义:这些词开头的从句 = 本次没做(条件/未来/建议)→ 切段丢弃后再匹配。 # actual 语义:这些词开头的从句 = 本次没做(条件/未来/建议)→ 切段丢弃后再匹配。
# 例 "充填,必要时根管治疗" → 丢"必要时根管治疗" → "充填" → restorative(不误判成 endodontic)。 # 例 "充填,必要时根管治疗" → 丢"必要时根管治疗" → "充填" → restorative(不误判成 endodontic)。
keyword_strip_clauses: [必要时, 如需, 择期, 建议, 推荐, 考虑, ] # 若…=条件从句(若牙面脱矿,建议充填),非本次实际治疗 keyword_strip_clauses: [必要时, 如需, 择期, 建议, 推荐, 考虑, ] # 若…=条件从句(若牙面脱矿,建议充填),非本次实际治疗
...@@ -8,6 +8,7 @@ import { ...@@ -8,6 +8,7 @@ import {
RESTORATION_INELIGIBLE_DX_NAMES, RESTORATION_INELIGIBLE_DX_NAMES,
STRUCTURAL_DX_CODE_LIST, STRUCTURAL_DX_CODE_LIST,
type DxTreatmentRule, type DxTreatmentRule,
REVIEW_IMPLIES_TREATMENT,
} from '@pac/types'; } from '@pac/types';
/** /**
...@@ -207,6 +208,30 @@ export function buildGapCore(input: GapCoreInput): GapCorePieces { ...@@ -207,6 +208,30 @@ export function buildGapCore(input: GapCoreInput): GapCorePieces {
? Prisma.sql`AND COALESCE(sig.content->>'name_zh','') NOT LIKE '%先天%'` ? Prisma.sql`AND COALESCE(sig.content->>'name_zh','') NOT LIKE '%先天%'`
: Prisma.empty; : Prisma.empty;
// (a''''') 「复查/复诊」= 修复体/矫治器在位的**证据**(不是治疗本身)。
// 宋志宏 TS0M001982 牙46;36;37:种植冠在位、按约复查,却被判"缺失牙未启动修复" ——
// 唯一带牙位的证据「种植复查」落在 review 类目,而 review 被刻意排除在 resolver 之外。
// ⛔ 只收 REVIEW_IMPLIES_TREATMENT(治疗词 + 复查/复诊),且**按类目过滤** ——
// 整类 review 放进来会误销:生产带牙位 4.4 万条里「观察」11,165 / 「无治疗」623,
// 那些恰恰是"还没治"。跨类目也不行(牙周复查 ≠ 补了龋)。
// 时间方向同治疗排除(afterDxFor):复查发生在信号之前不算数。
const reviewRules = REVIEW_IMPLIES_TREATMENT.filter((r) =>
(resolverCats as readonly string[]).includes(r.category),
);
const reviewImpliesBranch = reviewRules.length
? Prisma.sql`
UNION
SELECT rvt AS t
FROM patient_facts rvx
CROSS JOIN unnest(${toothArrSql(Prisma.sql`rvx.content->>'tooth_position'`)}) AS rvt
WHERE rvx.patient_id = p.id
AND rvx.type = 'treatment_record' AND rvx.kind = 'actual'
AND rvx.status IN ('active', 'fulfilled')
AND rvx.content->>'category' = 'review'
AND rvx.content->>'subtype' ~ ${reviewRules.map((r) => r.pattern).join('|')}
${afterDxFor('rvx')}`
: Prisma.empty;
const resolvedTeethSql = Prisma.sql` const resolvedTeethSql = Prisma.sql`
(SELECT COALESCE(array_agg(DISTINCT t), ARRAY[]::text[]) FROM ( (SELECT COALESCE(array_agg(DISTINCT t), ARRAY[]::text[]) FROM (
-- (a) 治疗家族 resolver(afterDx):同牙诊断后做了 resolverCats 家族里任一治疗 -- (a) 治疗家族 resolver(afterDx):同牙诊断后做了 resolverCats 家族里任一治疗
...@@ -304,6 +329,7 @@ export function buildGapCore(input: GapCoreInput): GapCorePieces { ...@@ -304,6 +329,7 @@ export function buildGapCore(input: GapCoreInput): GapCorePieces {
AND rdx.content->>'code' = ANY(${[...STRUCTURAL_DX_CODE_LIST]}::text[]) AND rdx.content->>'code' = ANY(${[...STRUCTURAL_DX_CODE_LIST]}::text[])
AND COALESCE(rdx.occurred_at, rdx.planned_for) >= COALESCE(sig.occurred_at, sig.planned_for) AND COALESCE(rdx.occurred_at, rdx.planned_for) >= COALESCE(sig.occurred_at, sig.planned_for)
AND sig.type = 'diagnosis_record' AND sig.type = 'diagnosis_record'
${reviewImpliesBranch}
${orthoExtractBranch} ${orthoExtractBranch}
${deferDxBranch} ${deferDxBranch}
) u)`; ) u)`;
......
...@@ -2476,16 +2476,24 @@ export function traceRawSourceTable(primaryTable: string, transforms: ReadonlyAr ...@@ -2476,16 +2476,24 @@ export function traceRawSourceTable(primaryTable: string, transforms: ReadonlyAr
input?: string; input?: string;
output?: string; output?: string;
inputs?: string[]; inputs?: string[];
outputs?: Array<{ output?: string }>; /// ⚠️ route_by_pattern 的字段名是 `routes`(见 transforms.schema.ts RouteByPatternOpSchema),
/// 不是 `outputs` —— 早先这里写成 outputs,恒为 undefined,导致**凡链路经过 route 的资源
/// 回溯都停在中间表**(_treatment_actual_raw_emr 等)。后果:reparse 把 rawPayload 灌进中间表,
/// 随即被 transform 链用空结果覆盖 → 治疗类 reparse 恒 0 变更**且报成功**(exit 0)。
/// diagnosis 没踩到只因它的链 split→derive→derive 不过 route。
routes?: Array<{ output?: string }>;
}; };
if (t.kind === 'union' && t.output && Array.isArray(t.inputs)) { if (t.kind === 'union' && t.output && Array.isArray(t.inputs)) {
unionInputs.set(t.output, t.inputs); unionInputs.set(t.output, t.inputs);
continue; continue;
} }
if (t.output && t.input) byOutput.set(t.output, t.input); // ⚠️ 跳过**原地 derive**(output === input,如 `_treat_plan_raw → _treat_plan_raw` 补 treat_name):
// 它不改变这张表的来源。登记进去会让 byOutput 指向自己 → resolve 撞环保护当场返回,
// 回溯同样停在中间表(与 routes 那条是**两个独立 bug**,只修一个仍然失效)。
if (t.output && t.input && t.output !== t.input) byOutput.set(t.output, t.input);
// route_by_pattern 多 output:每个 output 都回到同一 input // route_by_pattern 多 output:每个 output 都回到同一 input
if (Array.isArray(t.outputs) && t.input) { if (Array.isArray(t.routes) && t.input) {
for (const o of t.outputs) if (o?.output) byOutput.set(o.output, t.input); for (const o of t.routes) if (o?.output) byOutput.set(o.output, t.input);
} }
} }
const resolve = (tbl: string, seen: Set<string>): string => { const resolve = (tbl: string, seen: Set<string>): string => {
......
/**
* classifyByKeyword 大小写不敏感 + 2026-08-27 补的四个词(SRP / 加牙 / 活托 / 缝线)。
*
* 规则表按序裁决(首个命中即用),所以新词最大的风险不是"没命中",而是**抢了前面该赢的**。
* 下面每个新词都锁一条碰撞用例(生产 DW 里真实存在的串)。
*
* 跑:
* pnpm test -- keyword-case-and-terms
*/
import * as fs from 'node:fs';
import * as path from 'node:path';
import * as yaml from 'js-yaml';
import { classifyByKeyword } from '../src/modules/sync/assembler/assembler-engine';
import type { KeywordRule } from '../src/modules/sync/assembler/assembler.schema';
/// 直接读生产字典,不在测试里抄一份(抄了必漂)
const DICT = path.join(__dirname, '../data/_shared/dict/treatment-category-actual-rules.yaml');
const doc = yaml.load(fs.readFileSync(DICT, 'utf8')) as {
keyword_mapping: { category: KeywordRule[] };
keyword_strip_clauses?: string[];
};
const RULES = doc.keyword_mapping.category;
const STRIP = doc.keyword_strip_clauses;
const cat = (s: string) => classifyByKeyword(s, RULES, STRIP);
describe('大小写不敏感', () => {
it('RCT 大小写混写都归 endodontic(生产:RCT 3,122 落库 / rct 725 曾全丢)', () => {
for (const s of ['RCT', 'rct', 'Rct', 'rCT']) expect(cat(s)).toBe('endodontic');
});
it('复合串也不敏感', () => {
expect(cat('rct+冠修复')).toBe('endodontic'); // 根管优先于冠
expect(cat('局部srp')).toBe('periodontic');
});
it('中文不受影响(toLowerCase 对中文恒等)', () => {
expect(cat('根管治疗')).toBe('endodontic');
expect(cat('全口龈上洁治')).toBe('periodontic');
});
});
describe('SRP → periodontic', () => {
it.each(['SRP', '局部SRP', 'SRP+冲洗上药'])('%s', (s) => {
expect(cat(s)).toBe('periodontic');
});
});
describe('加牙 / 活托 → prosthodontic', () => {
it.each(['加牙', '活托加牙', '加牙后初戴', '活托加牙后初戴'])('%s', (s) => {
expect(cat(s)).toBe('prosthodontic');
});
// ⛔ 碰撞:「加牙」是「加牙周治疗」的子串,必须让根管先赢
it('「根管治疗加牙周治疗或酌情拔除」→ endodontic,不被加牙抢走', () => {
expect(cat('根管治疗加牙周治疗或酌情拔除')).toBe('endodontic');
});
it('已被旧词命中的变体行为不变', () => {
expect(cat('活动义齿加牙')).toBe('prosthodontic'); // 义齿
expect(cat('取模加牙')).toBe('prosthodontic'); // 取模
});
});
describe('缝线 → surgical', () => {
it.each(['拆除缝线', '拆除缝线,局部冲洗'])('%s', (s) => {
expect(cat(s)).toBe('surgical');
});
it('种植拆线仍归 implant(种植优先)', () => {
expect(cat('种植拆线')).toBe('implant');
});
});
describe('⛔ 刻意不收的词仍然不命中(收了会误销召回)', () => {
it.each(['冲洗', '换药', '上药', '调磨', '试戴', '观察', '定期观察'])('%s → undefined', (s) => {
expect(cat(s)).toBeUndefined();
});
it('「冲洗上药」不归 surgical —— 阻生牙冠周炎处置后那颗牙仍要拔', () => {
expect(cat('冲洗上药')).toBeUndefined();
});
});
describe('既有裁决顺序不被新词打乱', () => {
it.each([
['种植二期', 'implant'],
['根管治疗(磨牙)', 'endodontic'],
['二期隐形矫正', 'orthodontic'],
['冠修复(非美学区)', 'prosthodontic'],
['非美学区树脂充填', 'restorative'],
['玻璃离子窝沟封闭', 'preventive'],
['牙拔除术(松动乳牙)', 'surgical'],
])('%s → %s', (s, want) => {
expect(cat(s)).toBe(want);
});
});
/**
* refineCategoriesForDiagnosis —— K03 同码内主类目重排(残根/残冠 → 外科)。
*
* 守的是**「分配的潜在治疗项目」与「卡片上的目标」必须对应**这条口径:
* 分配矩阵把残根患者放进「拔牙治疗」列(POTENTIAL_LABEL_RULES: K03 + 含词 → extraction),
* 卡片的「目标 · X」就不能写「充填 / 嵌体」。两处共用 EXTRACTION_NAME_KEYWORDS 同一份词表。
*
* 跑:
* pnpm test -- refine-categories-k03
*/
import {
refineCategoriesForDiagnosis,
recommendedCategoriesForAge,
DiagnosisTreatmentMap,
EXTRACTION_NAME_KEYWORDS,
classifyCodeToLabel,
} from '@pac/types';
/** K03 在字典里的候选类目(单一真理源,不在测试里硬编码顺序) */
const K03_CATS = DiagnosisTreatmentMap.K03!.categories as readonly string[];
/** 复现 scenario 里 focusCategory 的算法:语义重排 → 年龄重排 → 取首项 */
const focusOf = (code: string, nameZh: string, age: number | null): string | null =>
recommendedCategoriesForAge(refineCategoriesForDiagnosis(code, nameZh, K03_CATS), age)[0] ?? null;
describe('refineCategoriesForDiagnosis · K03', () => {
it('K03 候选类目本身含 surgical(否则重排无处可挪)', () => {
expect(K03_CATS).toContain('surgical');
expect(K03_CATS).toContain('restorative');
});
describe('残根 / 残冠 → 外科打头', () => {
it.each(EXTRACTION_NAME_KEYWORDS)('「%s」→ surgical', (name) => {
expect(refineCategoriesForDiagnosis('K03', name, K03_CATS)[0]).toBe('surgical');
});
it('66 岁残根患者(孙海燕 TS0M012582 牙16;18;28;47)目标 = 外科,不是充填', () => {
expect(focusOf('K03', '残根', 66)).toBe('surgical');
});
it('年龄闸不会把它挪回去(K03 无 implant,recommendedCategoriesForAge 空转)', () => {
for (const age of [8, 18, 40, 66, 88, null]) {
expect(focusOf('K03', '残冠', age)).toBe('surgical');
}
});
});
describe('⛔ 只挪位,不增删', () => {
it('返回集合恒等于入参集合', () => {
const out = refineCategoriesForDiagnosis('K03', '残根', K03_CATS);
expect([...out].sort()).toEqual([...K03_CATS].sort());
});
it('其余相对顺序保持', () => {
const out = refineCategoriesForDiagnosis('K03', '残根', K03_CATS);
const rest = out.filter((c) => c !== 'surgical');
expect(rest).toEqual(K03_CATS.filter((c) => c !== 'surgical'));
});
});
describe('不该动的不动', () => {
it('楔状缺损 / 牙体缺损 → 保持原序(补得回来)', () => {
expect(refineCategoriesForDiagnosis('K03', '牙齿楔状缺损', K03_CATS)).toEqual(K03_CATS);
expect(refineCategoriesForDiagnosis('K03', '牙体缺损', K03_CATS)).toEqual(K03_CATS);
expect(focusOf('K03', '牙齿楔状缺损', 66)).toBe('restorative');
});
it('诊断名为空 → 原样返回(历史数据行为不变)', () => {
expect(refineCategoriesForDiagnosis('K03', null, K03_CATS)).toEqual(K03_CATS);
expect(refineCategoriesForDiagnosis('K03', ' ', K03_CATS)).toEqual(K03_CATS);
});
it('别的码不受影响', () => {
const k08 = DiagnosisTreatmentMap.K08!.categories as readonly string[];
expect(refineCategoriesForDiagnosis('K08', '残根', k08)).toEqual(k08);
expect(refineCategoriesForDiagnosis(null, '残根', K03_CATS)).toEqual(K03_CATS);
});
});
describe('K00 原有行为回归(表驱动改写后不能退化)', () => {
const k00 = DiagnosisTreatmentMap.K00!.categories as readonly string[];
it.each([
['乳牙滞留', 'surgical'],
['乳牙早失', 'orthodontic'],
['萌出障碍', 'orthodontic'],
['先天缺牙', 'prosthodontic'],
['釉质发育不全', 'prosthodontic'],
])('K00「%s」→ %s 打头', (name, lead) => {
expect(refineCategoriesForDiagnosis('K00', name, k00)[0]).toBe(lead);
});
});
describe('⭐ 标签与目标同源(本次修复的核心口径)', () => {
it.each(EXTRACTION_NAME_KEYWORDS)('「%s」:分配标签 extraction ⟺ 目标 surgical', (name) => {
expect(classifyCodeToLabel('K03', name, 66)).toBe('extraction');
expect(focusOf('K03', name, 66)).toBe('surgical');
});
it('非拔除类 K03:标签 restoration ⟺ 目标 restorative', () => {
expect(classifyCodeToLabel('K03', '牙齿楔状缺损', 66)).toBe('restoration');
expect(focusOf('K03', '牙齿楔状缺损', 66)).toBe('restorative');
});
});
});
/**
* REVIEW_IMPLIES_TREATMENT —— 「复查/复诊」作为修复体在位的**证据**。
*
* 守两条闸,任何一条松掉都会静默误销召回(少召不报错,一线只会觉得"系统没提醒过"):
* ① 词形闸:必须「治疗词 + 复查/复诊」——「正畸复诊」(在做)vs「正畸会诊」(还在谈)只差一字
* ② 类目闸:必须与 resolverCategoriesFor 交集 ——「牙周复查」不能解 K02 龋齿
*
* 正则用 PG 语义书写,这里用 JS RegExp 近似校验词形(两者对本表用到的语法等价)。
*
* 跑:
* pnpm test -- review-implies-treatment
*/
import { REVIEW_IMPLIES_TREATMENT, resolverCategoriesFor } from '@pac/types';
/** 某个 subtype 命中哪些规则 */
const rulesFor = (subtype: string) =>
REVIEW_IMPLIES_TREATMENT.filter((r) => new RegExp(r.pattern).test(subtype));
/** 模拟消费方:该诊断码下,这个 subtype 会不会解除缺口 */
const resolves = (code: string, subtype: string): boolean => {
const cats = resolverCategoriesFor(code) as readonly string[];
return rulesFor(subtype).some((r) => cats.includes(r.category));
};
describe('REVIEW_IMPLIES_TREATMENT · 词形闸', () => {
it.each([
['种植复查', 'implant'],
['牙周复查', 'periodontic'],
['正畸复诊', 'orthodontic'],
['正畸复查', 'orthodontic'],
['保持器复诊', 'orthodontic'],
])('生产实有词「%s」→ %s', (subtype, cat) => {
expect(rulesFor(subtype).map((r) => r.category)).toContain(cat);
});
// ⛔ 这些是生产带牙位 review 里的大头,收进来就是误销
it.each([
'观察', '暂观', '观察,必要时拔除', '暂观,必要时拔除', '随访观察', '观察随访', '随诊观察',
'无治疗', '初诊检查', '检查', '方案沟通', '沟通治疗方案', '听方案', '取资料', '缴费',
'转诊', '请全科医生会诊', '治疗中',
])('「%s」不命中任何规则', (subtype) => {
expect(rulesFor(subtype)).toHaveLength(0);
});
// ⛔ 泛指复查也不收 —— 它不指明治疗对象(宽口径 391 条里的争议区)
it.each(['复查', '常规复查', '定期复查'])('泛指「%s」不命中', (subtype) => {
expect(rulesFor(subtype)).toHaveLength(0);
});
// ⛔ 一字之差,语义相反
it.each(['正畸会诊', '正畸咨询', '正畸检查'])('「%s」不命中(还在谈,没在做)', (subtype) => {
expect(rulesFor(subtype)).toHaveLength(0);
});
});
describe('REVIEW_IMPLIES_TREATMENT · 类目闸', () => {
it('K08 缺失牙 + 种植复查 → 解除(宋志宏 TS0M001982 牙46;36;37)', () => {
expect(resolves('K08', '种植复查')).toBe(true);
});
it('K04 根尖周炎 + 种植复查 → 解除(种了牙就没牙髓了)', () => {
expect(resolves('K04', '种植复查')).toBe(true);
});
// ⛔ 生产实测的 9 条跨类目错配,逐条锁死
it.each([
['K02', '牙周复查', '洗牙不补龋'],
['K03', '牙周复查', '洗牙不修牙体'],
['K08', '牙周复查', '洗牙不修缺牙'],
['K06', '种植复查', 'K06 只认牙周/外科'],
['K01', '正畸复查', 'K01 走结构家族,不含正畸'],
])('%s + %s → 不解除(%s)', (code, subtype) => {
expect(resolves(code, subtype)).toBe(false);
});
});
describe('REVIEW_IMPLIES_TREATMENT · 表自身自洽', () => {
it('每条 pattern 都能编译成正则', () => {
for (const r of REVIEW_IMPLIES_TREATMENT) {
expect(() => new RegExp(r.pattern)).not.toThrow();
}
});
it('每条都写了 why(改表的人得说明临床依据)', () => {
for (const r of REVIEW_IMPLIES_TREATMENT) {
expect(r.why.length).toBeGreaterThan(4);
}
});
it('每条 pattern 都强制要求出现 复查/复诊', () => {
for (const r of REVIEW_IMPLIES_TREATMENT) {
expect(r.pattern).toMatch(/复查\|复诊/);
}
});
});
/**
* traceRawSourceTable —— reparse 的源表回溯。
*
* 回溯错了不会报错:rawPayload 被灌进**中间表**,随即被 transform 链用空结果覆盖 →
* reparse 恒 0 变更、exit 0、日志写"重衍完成"。2026-08-27 实测治疗类三个资源全中,
* 意味着任何治疗字典修补都无法回填存量,而且没人会发现。
*
* 跑:
* pnpm test -- trace-raw-source-table
*/
import * as fs from 'node:fs';
import * as path from 'node:path';
import * as yaml from 'js-yaml';
import { traceRawSourceTable } from '../src/modules/sync/cold-import/cold-import.service';
const MANIFEST = path.join(__dirname, '../data/jvs-dw/manifest.yaml');
const manifest = yaml.load(fs.readFileSync(MANIFEST, 'utf8')) as { transforms?: unknown[] };
const TF = manifest.transforms ?? [];
const trace = (t: string) => traceRawSourceTable(t, TF);
describe('traceRawSourceTable · 真实 manifest', () => {
// ⛔ 这四个必须都回溯到真正的 DW 原始表,任何一个停在中间表 = reparse 静默失效
it.each([
'treatment_actual_rows',
'treatment_planned_rows',
'treatment_review_rows',
'diagnosis_rows',
])('%s → fact_emr_treatment_out', (tbl) => {
expect(trace(tbl)).toBe('fact_emr_treatment_out');
});
it('⛔ 不能停在 route 的输出表上(本次 bug 的指纹)', () => {
for (const tbl of ['treatment_actual_rows', 'treatment_planned_rows', 'treatment_review_rows']) {
expect(trace(tbl)).not.toMatch(/^_/); // 下划线开头 = manifest 里的中间表
}
});
it('原始表回溯到自身(reparse 据此判定"非 transform 产出"并跳过)', () => {
expect(trace('fact_emr_treatment_out')).toBe('fact_emr_treatment_out');
});
});
describe('traceRawSourceTable · 合成用例', () => {
it('route_by_pattern 用的是 routes 字段,不是 outputs', () => {
const tf = [
{ kind: 'split_json_array', input: 'raw_src', output: '_mid' },
{ kind: 'route_by_pattern', input: '_mid', field: 'x', routes: [{ output: '_a' }, { output: '_b' }] },
{ kind: 'derive', input: '_a', output: 'final_rows' },
];
expect(traceRawSourceTable('final_rows', tf)).toBe('raw_src');
});
it('union 各分支回溯不一致 → 停在 union 输出(既有语义不变)', () => {
const tf = [
{ kind: 'derive', input: 'raw_a', output: 't1' },
{ kind: 'union', inputs: ['t1', 'raw_b'], output: 'merged' },
];
expect(traceRawSourceTable('merged', tf)).toBe('merged');
});
});
...@@ -17,7 +17,10 @@ const transforms = [ ...@@ -17,7 +17,10 @@ const transforms = [
{ kind: 'filter', input: 'patient_settlement_spec', output: '_refund_item_raw', where: {} }, { kind: 'filter', input: 'patient_settlement_spec', output: '_refund_item_raw', where: {} },
{ kind: 'lookup', input: '_refund_item_raw', output: 'refund_item_rows', from: 'patient_settlement', select: {} }, { kind: 'lookup', input: '_refund_item_raw', output: 'refund_item_rows', from: 'patient_settlement', select: {} },
// route_by_pattern 多输出回同一 input // route_by_pattern 多输出回同一 input
{ kind: 'route_by_pattern', input: '_treat_raw', outputs: [{ output: '_actual_raw' }, { output: '_rec_raw' }] }, // ⚠️ 2026-08-27 修正:字段名是 `routes`(见 transforms.schema.ts RouteByPatternOpSchema),
// 本 fixture 原先写 `outputs` —— 跟当时的实现犯了同一个错,于是测试一直绿、生产一直坏
// (治疗类三个资源 reparse 恒 0 变更且报成功)。fixture 必须照**契约**写,不能照实现写。
{ kind: 'route_by_pattern', input: '_treat_raw', routes: [{ output: '_actual_raw' }, { output: '_rec_raw' }] },
]; ];
describe('traceRawSourceTable', () => { describe('traceRawSourceTable', () => {
......
...@@ -10,6 +10,9 @@ ...@@ -10,6 +10,9 @@
* 详见 `apps/pac-docs/content/docs/architecture/data-ingestion.mdx` §三。 * 详见 `apps/pac-docs/content/docs/architecture/data-ingestion.mdx` §三。
*/ */
import { z } from 'zod'; import { z } from 'zod';
// ⭐ 单一真理源:「该拔不是该补」的诊断词表 —— 分配矩阵的 extraction 标签与
// 召回目标的 surgical 主类目共用同一份词,否则「潜在治疗项目」和「目标」会各说各话。
import { EXTRACTION_NAME_KEYWORDS } from './potential-label-rules';
// ============================================================= // =============================================================
// 诊断码(PACDiagnosisCode)— 借 ICD-10 K00-K14 大类 + 业务码 + 推荐码 // 诊断码(PACDiagnosisCode)— 借 ICD-10 K00-K14 大类 + 业务码 + 推荐码
...@@ -418,6 +421,46 @@ const STRUCTURAL_DX_CODES = new Set<string>([ ...@@ -418,6 +421,46 @@ const STRUCTURAL_DX_CODES = new Set<string>([
* *
* 单一真理源:召回 scenario 的 ⑤a 排除闸 + 牙位事实 oracle 对账 共用此函数,口径不漂移。 * 单一真理源:召回 scenario 的 ⑤a 排除闸 + 牙位事实 oracle 对账 共用此函数,口径不漂移。
*/ */
/**
* 「复查/复诊」→ 它蕴含的治疗类目(**证据,不是治疗**)。
*
* ── 为什么需要 ──
* 宋志宏 TS0M001982 牙46;36;37 三颗种植冠在位、正在按约复查,却被判「缺失牙未启动修复」。
* 他那次就诊里能证明修复体存在的四处表述(诊断「种植术后」/ 本次治疗「种植复查」/
* 检查所见「种植冠无松动」/ 主诉「种植戴牙术后1年余」)PAC 一处也没接住 ——
* `review` 类目被**刻意**排除在 [[STRUCTURAL_RESOLVER_CATEGORIES]] 之外(见其上方注释)。
* 那个排除是对的:review 是杂物抽屉,生产带牙位的 4.4 万条里「观察」11,165、
* 「观察,必要时拔除」928、「无治疗」623 —— 这些恰恰说明**还没治**,整类放进 resolver 会误销。
*
* ── 判据:复查的**对象**必须先存在 ──
* 「种植复查」蕴含种植体在位,「保持器复诊」蕴含正畸做过 —— 这是逻辑蕴含,不是统计推断。
* 生产验证:带牙位的「种植复查」覆盖 3,893 个牙位,其中 3,434(88.2%)在**同一颗牙**上
* 找得到 implant/prosthodontic 的 actual 治疗;患者级 95.2%。剩下 11.8% 正是本规则要救的
* (外院种的 / 摄入窗口之前 / 治疗记录漏牙位)。
*
* ⛔ **必须同时匹配「治疗词 + 复查/复诊」**:「正畸复诊」(在做正畸)与「正畸会诊」(还在谈)
* 只差一个字,语义相反;生产各 35 / 103 条。光匹配「正畸」会把会诊咨询一起网进来。
* ⛔ **必须按类目匹配**,不能一律放行:生产实测 19 条候选里 9 条是跨类目错配 ——
* 「牙周复查」去解 K02 龋齿(洗牙不补龋,4 条)、「种植复查」去解 K06 牙龈疾患(1 条)、
* 「正畸复查」去解 K01 阻生牙(1 条)。消费方须用 [[resolverCategoriesFor]] 过滤本表。
* ⛔ 「观察 / 暂观 / 无治疗 / 初诊检查 / 方案沟通」等**刻意不收** —— 它们是"还没治"的证据。
*
* 消费方:potential-treatment-gap.sql 的 resolvedTeethSql(牙位级)。
* 生产实测无一条落在 K05/K07 全口场景,故全口路径(gapWhere 的 NOT EXISTS)不接本表。
*/
export const REVIEW_IMPLIES_TREATMENT: ReadonlyArray<{
/// PG 正则(用于 `subtype ~ pattern`)
pattern: string;
/// 命中即视为该类目治疗已完成
category: PACTreatmentCategory;
why: string;
}> = [
{ pattern: '种植[^,,;;]*(复查|复诊)', category: 'implant', why: '种植体不存在就没有种植复查' },
{ pattern: '(修复|冠|桥|义齿|戴牙)[^,,;;]*(复查|复诊)', category: 'prosthodontic', why: '修复体在位才谈得上复查' },
{ pattern: '(正畸|保持器)[^,,;;]*(复查|复诊)', category: 'orthodontic', why: '矫治/保持阶段 ≠ 正畸会诊咨询' },
{ pattern: '牙周[^,,;;]*(复查|复诊)', category: 'periodontic', why: '牙周复查蕴含基础治疗做过' },
];
export function resolverCategoriesFor(code: string): readonly PACTreatmentCategory[] { export function resolverCategoriesFor(code: string): readonly PACTreatmentCategory[] {
if (STRUCTURAL_DX_CODES.has(code)) return STRUCTURAL_RESOLVER_CATEGORIES; if (STRUCTURAL_DX_CODES.has(code)) return STRUCTURAL_RESOLVER_CATEGORIES;
const rule = lookupDxTreatment(code); const rule = lookupDxTreatment(code);
...@@ -831,12 +874,45 @@ export const K00_LEAD_CATEGORY_RULES: ReadonlyArray<{ ...@@ -831,12 +874,45 @@ export const K00_LEAD_CATEGORY_RULES: ReadonlyArray<{
]; ];
/** /**
* 诊断词细分建议类目(目前只作用于 K00)—— 排在年龄适配**之前**的一道语义重排。 * K03 主类目细分(**同码内重排,不改诊断码**)。
*
* 背景(2026-08 孙海燕 TS0M012582 牙16;18;28;47):K03「牙体硬组织其他疾病」同样是大口袋 ——
* 楔状缺损 / 牙体缺损(补得回来,restorative)与 残根 / 残冠(补不回来,只能拔,surgical)
* 共用一份固定类目序 `[restorative, prosthodontic, surgical]` → focusCategory 恒为
* restorative,给残根患者打「目标 · 充填 / 嵌体」—— 残根是没法充填的。
*
* ⚠️ 词表**复用 [[EXTRACTION_NAME_KEYWORDS]]** —— 那张表已经在驱动分配矩阵的
* `K03 + 含词 → extraction(拔牙治疗)` 标签。两处同源,标签与目标才不会打架:
* 分配把人放进「拔牙治疗」列,卡片却写「目标 · 充填」,客服无从判断该说什么。
* ⚠️ 与 K00 同边界:只挪位,返回集合恒等于入参集合,排除闸不受影响。
*/
export const K03_LEAD_CATEGORY_RULES: ReadonlyArray<{
pattern: RegExp;
lead: string;
why: string;
}> = [
{
pattern: new RegExp(EXTRACTION_NAME_KEYWORDS.join('|')),
lead: 'surgical',
why: '残根 / 残冠补不回来,临床路径是拔除(或根管+桩核冠保留,仍非充填)',
},
];
/// 诊断码 → 同码内主类目重排规则(表驱动;未列的码原样返回)。
const LEAD_CATEGORY_RULES_BY_CODE: Readonly<
Record<string, ReadonlyArray<{ pattern: RegExp; lead: string; why: string }>>
> = {
K00: K00_LEAD_CATEGORY_RULES,
K03: K03_LEAD_CATEGORY_RULES,
};
/**
* 诊断词细分建议类目(K00 / K03,表驱动)—— 排在年龄适配**之前**的一道语义重排。
* *
* 顺序:`refineCategoriesForDiagnosis`(临床语义)→ `recommendedCategoriesForAge`(年龄)。 * 顺序:`refineCategoriesForDiagnosis`(临床语义)→ `recommendedCategoriesForAge`(年龄)。
* 前者定"这个病该往哪个方向治",后者定"这个岁数该先讲哪个"。两道都只挪位。 * 前者定"这个病该往哪个方向治",后者定"这个岁数该先讲哪个"。两道都只挪位。
* *
* @param code 诊断码(非 K00 一律原样返回) * @param code 诊断码(不在 LEAD_CATEGORY_RULES_BY_CODE 里的一律原样返回)
* @param nameZh 诊断原文词(DW diag message,如「乳牙早失」);空 → 原样返回 * @param nameZh 诊断原文词(DW diag message,如「乳牙早失」);空 → 原样返回
* @param categories 该诊断的候选治疗类目 * @param categories 该诊断的候选治疗类目
*/ */
...@@ -845,10 +921,11 @@ export function refineCategoriesForDiagnosis( ...@@ -845,10 +921,11 @@ export function refineCategoriesForDiagnosis(
nameZh: string | null | undefined, nameZh: string | null | undefined,
categories: readonly string[], categories: readonly string[],
): readonly string[] { ): readonly string[] {
if (code !== 'K00') return categories; const rules = code ? LEAD_CATEGORY_RULES_BY_CODE[code] : undefined;
if (!rules) return categories;
const name = (nameZh ?? '').trim(); const name = (nameZh ?? '').trim();
if (!name) return categories; if (!name) return categories;
const hit = K00_LEAD_CATEGORY_RULES.find((r) => r.pattern.test(name)); const hit = rules.find((r) => r.pattern.test(name));
if (!hit || !categories.includes(hit.lead)) return categories; if (!hit || !categories.includes(hit.lead)) return categories;
return [hit.lead, ...categories.filter((c) => c !== hit.lead)]; return [hit.lead, ...categories.filter((c) => c !== hit.lead)];
} }
......
Markdown is supported
0% or
You are about to add 0 people to the discussion. Proceed with caution.
Finish editing this message first!
Please register or to comment