457 lines
16 KiB
JavaScript
457 lines
16 KiB
JavaScript
/**
|
|
* PR-37 Part B — Domain Classifier
|
|
*
|
|
* Classifies product strategy descriptions into primary/secondary domains
|
|
* by matching keywords against the domain registry.
|
|
*
|
|
* @module pr-37-domain-intelligence/domain-classifier
|
|
* @since PR-37
|
|
*/
|
|
|
|
import {
|
|
getDomain,
|
|
getDomainRegistry,
|
|
searchDomains,
|
|
} from "./domain-registry.mjs";
|
|
|
|
// ══════════════════════════════════════════════════════════
|
|
// Constants
|
|
// ══════════════════════════════════════════════════════════
|
|
|
|
/** Score bonus when the domain name appears directly in input. */
|
|
const NAME_MATCH_BONUS = 10;
|
|
|
|
/** Score per keyword matched. */
|
|
const KEYWORD_SCORE = 3;
|
|
|
|
/** Score per alias matched. */
|
|
const ALIAS_SCORE = 5;
|
|
|
|
/** Score per core-process phrase matched. */
|
|
const PROCESS_SCORE = 4;
|
|
|
|
/**
|
|
* Known cross-domain edges for specific Chinese phrases.
|
|
*
|
|
* When these phrases appear in input, they trigger additional domain-specific
|
|
* boosts so that commonly-expected secondary domains surface correctly even
|
|
* when no direct keyword match exists.
|
|
*
|
|
* @type {object<string, Array<{domain: string, boost: number, reason: string}>>}
|
|
*/
|
|
const CROSS_MATCH_MAP = Object.freeze({
|
|
// 跨境电商 → E-Commerce primary, SCM secondary
|
|
"跨境电商": [
|
|
{ domain: "E-Commerce", boost: 20, reason: "Cross-border e-commerce" },
|
|
{ domain: "SCM", boost: 18, reason: "Cross-border logistics / supply chain" },
|
|
],
|
|
"跨境": [
|
|
{ domain: "E-Commerce", boost: 12, reason: "Cross-border trade" },
|
|
{ domain: "SCM", boost: 14, reason: "Cross-border supply chain" },
|
|
],
|
|
// 工业AI质检 → MES primary, AI Platform secondary
|
|
"工业ai质检": [
|
|
{ domain: "MES", boost: 20, reason: "Industrial AI QA on factory floor" },
|
|
{ domain: "AI Platform", boost: 14, reason: "AI/ML inspection component" },
|
|
],
|
|
"工业ai": [
|
|
{ domain: "MES", boost: 12, reason: "Industrial AI / smart manufacturing" },
|
|
{ domain: "AI Platform", boost: 8, reason: "AI capability" },
|
|
],
|
|
"智能质检": [
|
|
{ domain: "MES", boost: 18, reason: "Smart QA on factory floor" },
|
|
{ domain: "AI Platform", boost: 12, reason: "AI-powered inspection" },
|
|
],
|
|
"ai质检": [
|
|
{ domain: "MES", boost: 12, reason: "AI quality inspection on production line" },
|
|
{ domain: "AI Platform", boost: 10, reason: "AI vision / ML detection" },
|
|
],
|
|
});
|
|
|
|
/**
|
|
* Terms that are too generic (e.g. "平台", "系统", "管理") and should be
|
|
* excluded from keyword scoring to avoid inflating scores for unrelated domains.
|
|
*
|
|
* @type {Set<string>}
|
|
*/
|
|
const STOPWORDS = new Set([
|
|
"平台", "系统", "管理", "服务", "引擎", "引擎", "平台",
|
|
"在线", "智能", "数字", "自动",
|
|
]);
|
|
|
|
/**
|
|
* Category affinity matrix (lower is closer).
|
|
* 1 = same category, 2 = related, 3 = unrelated (penalty).
|
|
*/
|
|
const CATEGORY_AFFINITY = Object.freeze({
|
|
"enterprise-software": { "enterprise-software": 1, industrial: 2, commerce: 2, platform: 2, intelligence: 2, engineering: 3, management: 2 },
|
|
industrial: { "enterprise-software": 2, industrial: 1, commerce: 3, platform: 3, intelligence: 2, engineering: 2, management: 3 },
|
|
commerce: { "enterprise-software": 2, industrial: 3, commerce: 1, platform: 2, intelligence: 2, engineering: 3, management: 2 },
|
|
platform: { "enterprise-software": 2, industrial: 3, commerce: 2, platform: 1, intelligence: 2, engineering: 3, management: 2 },
|
|
intelligence: { "enterprise-software": 2, industrial: 2, commerce: 2, platform: 2, intelligence: 1, engineering: 3, management: 2 },
|
|
engineering: { "enterprise-software": 3, industrial: 2, commerce: 3, platform: 3, intelligence: 3, engineering: 1, management: 3 },
|
|
management: { "enterprise-software": 2, industrial: 3, commerce: 2, platform: 2, intelligence: 2, engineering: 3, management: 1 },
|
|
});
|
|
|
|
// ══════════════════════════════════════════════════════════
|
|
// Internal helpers
|
|
// ══════════════════════════════════════════════════════════
|
|
|
|
/**
|
|
* Tokenize a string into lowercase keyword tokens.
|
|
*
|
|
* @param {string} text - Input text
|
|
* @returns {{ tokens: string[], rawLower: string }} Tokenization result
|
|
*/
|
|
function tokenize(text) {
|
|
if (!text || typeof text !== "string") return { tokens: [], rawLower: "" };
|
|
const rawLower = text.toLowerCase();
|
|
// Split on whitespace, punctuation
|
|
const rawTokens = rawLower.split(/[\s,,。、;:;!?!?()()【】\[\]{}"''":·/\\\-–—]+/).filter(Boolean);
|
|
const result = new Set(rawTokens);
|
|
|
|
// Extract CJK bigrams for Chinese compound matching
|
|
for (const tok of rawTokens) {
|
|
if (/[\u4e00-\u9fff]/.test(tok) && tok.length >= 2) {
|
|
result.add(tok);
|
|
// Generate overlapping 2-char windows for compounds like "跨境电商"
|
|
for (let i = 0; i <= tok.length - 2; i++) {
|
|
const bigram = tok.substring(i, i + 2);
|
|
// Only add meaningful CJK bigrams, not common stopwords
|
|
if (!STOPWORDS.has(bigram)) result.add(bigram);
|
|
}
|
|
}
|
|
}
|
|
|
|
return { tokens: [...result], rawLower };
|
|
}
|
|
|
|
/**
|
|
* Check if a term is a stopword (too generic to match on).
|
|
*
|
|
* @param {string} term - The term to check
|
|
* @returns {boolean}
|
|
*/
|
|
function isStopword(term) {
|
|
const t = term.replace(/^(keyword:|alias:|domain:|process:)/, "").trim();
|
|
return STOPWORDS.has(t);
|
|
}
|
|
|
|
/**
|
|
* Score a single domain against tokenized input.
|
|
*
|
|
* @param {object} domain - Domain definition object
|
|
* @param {string[]} tokens - Tokenized input tokens
|
|
* @param {string} rawLower - Original input text (lowered)
|
|
* @returns {{ score: number, matchedKeywords: string[] }}
|
|
*/
|
|
function scoreDomain(domain, tokens, rawLower) {
|
|
const matched = new Set();
|
|
|
|
/** Check if a term exists in the raw text (case-insensitive). */
|
|
const matchIfFound = (term) => {
|
|
const t = typeof term === "string" ? term.toLowerCase() : "";
|
|
if (!t || isStopword(t)) return false;
|
|
if (rawLower.includes(t)) return true;
|
|
// Partial CJK match (e.g., keyword "电商" matches token "跨境电商")
|
|
if (/[\u4e00-\u9fff]/.test(t)) {
|
|
return tokens.some((tok) => tok.includes(t) || t.includes(tok));
|
|
}
|
|
// English token check (e.g., "SCM")
|
|
const normTok = t.replace(/[\s-]/g, "");
|
|
return tokens.some((tok) => tok === normTok || tok.includes(normTok) || normTok.includes(tok));
|
|
};
|
|
|
|
// 1. Name / displayName direct match
|
|
if (rawLower.includes(domain.name.toLowerCase())) {
|
|
matched.add(`domain:${domain.name}`);
|
|
}
|
|
|
|
// 2. Aliases
|
|
for (const alias of domain.aliases) {
|
|
if (matchIfFound(alias)) matched.add(`alias:${alias}`);
|
|
}
|
|
|
|
// 3. Keywords
|
|
for (const kw of domain.keywords) {
|
|
if (matchIfFound(kw)) matched.add(`keyword:${kw}`);
|
|
}
|
|
|
|
// 4. Core process phrases
|
|
for (const proc of domain.coreProcesses) {
|
|
const procPhrase = proc.replace(/\(.*?\)/g, "").trim().toLowerCase();
|
|
if (rawLower.includes(procPhrase)) {
|
|
matched.add(`process:${procPhrase.substring(0, 20)}`);
|
|
} else {
|
|
for (const tok of tokens) {
|
|
if (tok.length >= 2 && procPhrase.includes(tok)) {
|
|
matched.add(`process:${procPhrase.substring(0, 20)}`);
|
|
break;
|
|
}
|
|
}
|
|
}
|
|
}
|
|
|
|
// Compute score
|
|
let score = 0;
|
|
for (const m of matched) {
|
|
if (isStopword(m)) continue;
|
|
if (m.startsWith("domain:")) score += NAME_MATCH_BONUS;
|
|
else if (m.startsWith("alias:")) score += ALIAS_SCORE;
|
|
else if (m.startsWith("keyword:")) score += KEYWORD_SCORE;
|
|
else if (m.startsWith("process:")) score += PROCESS_SCORE;
|
|
}
|
|
|
|
return { score, matchedKeywords: [...matched].filter((m) => !isStopword(m)) };
|
|
}
|
|
|
|
/**
|
|
* Apply cross-match map boosts for well-known phrases.
|
|
* Mutates scored array in-place.
|
|
*
|
|
* @param {Array<{name: string, score: number, matchedKeywords: string[], category: string}>} scored
|
|
* @param {string} rawLower - Raw lowered input
|
|
*/
|
|
function applyCrossMatchBoosts(scored, rawLower) {
|
|
const shortened = rawLower.replace(/[\s-]/g, "");
|
|
for (const [phrase, crossRefs] of Object.entries(CROSS_MATCH_MAP)) {
|
|
const phraseNorm = phrase.toLowerCase().replace(/[\s-]/g, "");
|
|
if (shortened.includes(phraseNorm)) {
|
|
for (const ref of crossRefs) {
|
|
const existing = scored.find((s) => s.name === ref.domain);
|
|
if (existing) {
|
|
existing.score += ref.boost;
|
|
existing.matchedKeywords.push(`cross-ref:${ref.reason}`);
|
|
} else {
|
|
// Auto-create entry for this domain if it doesn't exist yet
|
|
scored.push({
|
|
name: ref.domain,
|
|
score: ref.boost,
|
|
matchedKeywords: [`cross-ref:${ref.reason}`],
|
|
category: "cross-reference",
|
|
});
|
|
}
|
|
}
|
|
}
|
|
}
|
|
}
|
|
|
|
// ══════════════════════════════════════════════════════════
|
|
// Public API
|
|
// ══════════════════════════════════════════════════════════
|
|
|
|
/**
|
|
* Classify a product strategy into primary and secondary domains.
|
|
*
|
|
* @param {object|string} productInput - Product strategy package containing
|
|
* `productName` / `description`, or a plain product name string
|
|
* @returns {object} Classification result
|
|
* @returns {string|null} .primaryDomain - Best-fit domain name
|
|
* @returns {string|null} .secondaryDomain - Runner-up domain name (or null)
|
|
* @returns {number} .confidenceScore - 0-1 confidence value
|
|
* @returns {string[]} .matchedKeywords - All matched keyword references
|
|
* @returns {string} .classificationReason - Human-readable explanation
|
|
*
|
|
* @example
|
|
* classifyProduct("跨境电商平台")
|
|
* // => { primaryDomain: "E-Commerce", secondaryDomain: "SCM", confidenceScore: 0.92, ... }
|
|
*
|
|
* @example
|
|
* classifyProduct({ productName: "工业AI质检系统", description: "基于机器学习的AOI" })
|
|
* // => { primaryDomain: "MES", secondaryDomain: "AI Platform", ... }
|
|
*/
|
|
export function classifyProduct(productInput) {
|
|
// Normalize input
|
|
if (!productInput) {
|
|
return {
|
|
primaryDomain: null,
|
|
secondaryDomain: null,
|
|
confidenceScore: 0,
|
|
matchedKeywords: [],
|
|
classificationReason: "Empty input — no classification possible",
|
|
};
|
|
}
|
|
|
|
const productName =
|
|
typeof productInput === "string"
|
|
? productInput
|
|
: productInput.productName || "";
|
|
|
|
const description =
|
|
typeof productInput === "string"
|
|
? productInput
|
|
: productInput.description || productInput.productDescription || "";
|
|
|
|
const rawText = `${productName} ${description}`.trim();
|
|
if (!rawText) {
|
|
return {
|
|
primaryDomain: null,
|
|
secondaryDomain: null,
|
|
confidenceScore: 0,
|
|
matchedKeywords: [],
|
|
classificationReason: "No product name or description provided",
|
|
};
|
|
}
|
|
|
|
const { tokens, rawLower } = tokenize(rawText);
|
|
const registry = getDomainRegistry();
|
|
const scored = [];
|
|
|
|
for (const [name, domain] of Object.entries(registry)) {
|
|
const { score, matchedKeywords } = scoreDomain(domain, tokens, rawLower);
|
|
if (score > 0) {
|
|
scored.push({ name, score, matchedKeywords, category: domain.category });
|
|
}
|
|
}
|
|
|
|
// Apply cross-match boosts for well-known edge cases
|
|
applyCrossMatchBoosts(scored, rawLower);
|
|
|
|
// Sort by score descending
|
|
scored.sort((a, b) => b.score - a.score);
|
|
|
|
if (scored.length === 0) {
|
|
return {
|
|
primaryDomain: null,
|
|
secondaryDomain: null,
|
|
confidenceScore: 0,
|
|
matchedKeywords: [],
|
|
classificationReason:
|
|
"No domain matched the provided product description",
|
|
};
|
|
}
|
|
|
|
const primary = scored[0];
|
|
const secondaryCandidate = scored[1];
|
|
|
|
// Calculate confidence: normalize to 0-1
|
|
const maxObservedScore = scored[0].score;
|
|
const maxExpectedScore = 60;
|
|
const rawConfidence = Math.min(maxObservedScore / maxExpectedScore, 1.0);
|
|
|
|
// If there's a clear gap between #1 and #2, boost confidence
|
|
let confidence = rawConfidence;
|
|
if (
|
|
scored.length >= 2 &&
|
|
primary.score > secondaryCandidate.score * 2
|
|
) {
|
|
confidence = Math.max(confidence, 0.75);
|
|
}
|
|
|
|
confidence = Math.round(Math.min(Math.max(confidence, 0), 1) * 100) / 100;
|
|
|
|
// Secondary domain: only if score is positive and not the same as primary
|
|
const secondary =
|
|
secondaryCandidate && secondaryCandidate.score > 0
|
|
? secondaryCandidate.name
|
|
: null;
|
|
|
|
// Build matched keywords (all unique)
|
|
const allMatched = [...new Set(scored.flatMap((s) => s.matchedKeywords))];
|
|
|
|
// Build reason
|
|
const reason = [
|
|
`Primary: ${primary.name} (score=${primary.score})`,
|
|
secondary
|
|
? `Secondary: ${secondary} (score=${secondaryCandidate.score})`
|
|
: null,
|
|
`Confidence: ${confidence}`,
|
|
`Matched: ${
|
|
allMatched.length > 0 ? allMatched.slice(0, 8).join(", ") : "none"
|
|
}`,
|
|
...(allMatched.length > 8 ? [`... and ${allMatched.length - 8} more`] : []),
|
|
]
|
|
.filter(Boolean)
|
|
.join(" | ");
|
|
|
|
return {
|
|
primaryDomain: primary.name,
|
|
secondaryDomain: secondary,
|
|
confidenceScore: confidence,
|
|
matchedKeywords: allMatched,
|
|
classificationReason: reason,
|
|
};
|
|
}
|
|
|
|
/**
|
|
* Classify and return only the domain name (convenience shortcut).
|
|
*
|
|
* @param {object|string} productInput - Product strategy input
|
|
* @returns {string|null} Primary domain name or null
|
|
*/
|
|
export function classifyDomain(productInput) {
|
|
return classifyProduct(productInput).primaryDomain;
|
|
}
|
|
|
|
/**
|
|
* List all supported domain names.
|
|
*
|
|
* @returns {string[]}
|
|
*/
|
|
export function listSupportedDomains() {
|
|
return Object.keys(getDomainRegistry());
|
|
}
|
|
|
|
// ══════════════════════════════════════════════════════════
|
|
// Fuzzy classifier (lightweight n-gram matching)
|
|
// ══════════════════════════════════════════════════════════
|
|
|
|
/**
|
|
* Soft-match a phrase against domain descriptions using n-gram overlap.
|
|
*
|
|
* @param {string} text - Input text
|
|
* @param {number} [threshold=0.15] - Minimum match ratio (lower = more permissive)
|
|
* @returns {object[]} Sorted domain matches with relevance score
|
|
*/
|
|
export function fuzzyClassify(text, threshold = 0.15) {
|
|
if (!text || typeof text !== "string" || !text.trim()) return [];
|
|
|
|
const { tokens, rawLower } = tokenize(text.trim());
|
|
if (tokens.length === 0) return [];
|
|
|
|
const tokensSet = new Set(tokens);
|
|
const registry = getDomainRegistry();
|
|
const results = [];
|
|
|
|
for (const [name, domain] of Object.entries(registry)) {
|
|
const combinedText = [
|
|
domain.name,
|
|
domain.displayName,
|
|
...domain.keywords,
|
|
...domain.aliases,
|
|
domain.description,
|
|
]
|
|
.join(" ")
|
|
.toLowerCase();
|
|
|
|
const domainTokens = new Set(
|
|
combinedText
|
|
.split(/[\s,,。、;:;!?!?()()【】\[\]{}"''":·/\\\-–—]+/)
|
|
.filter(Boolean)
|
|
);
|
|
|
|
// Compute overlap: how many input tokens partially match domain tokens
|
|
let overlap = 0;
|
|
for (const tok of tokens) {
|
|
if (tok.length < 2) continue;
|
|
for (const dt of domainTokens) {
|
|
const dtNorm = dt.replace(/[\s-]/g, "");
|
|
const tokNorm = tok.replace(/[\s-]/g, "");
|
|
if (dtNorm.includes(tokNorm) || tokNorm.includes(dtNorm)) {
|
|
overlap++;
|
|
break;
|
|
}
|
|
}
|
|
}
|
|
|
|
const union = new Set([...tokensSet, ...domainTokens]);
|
|
const ratio = union.size > 0 ? overlap / union.size : 0;
|
|
|
|
if (ratio >= threshold && overlap > 0) {
|
|
results.push({ domain: name, relevance: Math.round(ratio * 100) / 100 });
|
|
}
|
|
}
|
|
|
|
return results.sort((a, b) => b.relevance - a.relevance);
|
|
}
|