feat: port holistic node refinement contract
This commit is contained in:
@@ -53,10 +53,8 @@ function buildCanonicalMetadata({ existing, metadata }) {
|
||||
function mapNodeRow(row) {
|
||||
return {
|
||||
...row,
|
||||
dimensions: JSON.parse(row.dimensions_json || '[]'),
|
||||
metadata: row.metadata ? (typeof row.metadata === 'string' ? JSON.parse(row.metadata) : row.metadata) : null,
|
||||
context: row.context_json ? JSON.parse(row.context_json) : null,
|
||||
dimensions_json: undefined,
|
||||
context_json: undefined,
|
||||
};
|
||||
}
|
||||
@@ -65,13 +63,11 @@ function mapNodeRow(row) {
|
||||
* Get nodes with optional filtering.
|
||||
*/
|
||||
function getNodes(filters = {}) {
|
||||
const { dimensions, search, limit = 100, offset = 0, contextId } = filters;
|
||||
const { search, limit = 100, offset = 0, contextId } = filters;
|
||||
|
||||
let sql = `
|
||||
SELECT n.id, n.title, n.description, n.source, n.link, n.event_date, n.metadata,
|
||||
n.created_at, n.updated_at, n.context_id,
|
||||
COALESCE((SELECT JSON_GROUP_ARRAY(d.dimension)
|
||||
FROM node_dimensions d WHERE d.node_id = n.id), '[]') as dimensions_json,
|
||||
CASE
|
||||
WHEN c.id IS NULL THEN NULL
|
||||
ELSE json_object('id', c.id, 'name', c.name, 'description', c.description, 'icon', c.icon)
|
||||
@@ -82,16 +78,6 @@ function getNodes(filters = {}) {
|
||||
`;
|
||||
const params = [];
|
||||
|
||||
// Filter by dimensions
|
||||
if (dimensions && dimensions.length > 0) {
|
||||
sql += ` AND EXISTS (
|
||||
SELECT 1 FROM node_dimensions nd
|
||||
WHERE nd.node_id = n.id
|
||||
AND nd.dimension IN (${dimensions.map(() => '?').join(',')})
|
||||
)`;
|
||||
params.push(...dimensions);
|
||||
}
|
||||
|
||||
// Text search
|
||||
if (search) {
|
||||
sql += ` AND (n.title LIKE ? COLLATE NOCASE OR n.description LIKE ? COLLATE NOCASE OR n.source LIKE ? COLLATE NOCASE)`;
|
||||
@@ -135,8 +121,6 @@ function getNodeById(id) {
|
||||
const sql = `
|
||||
SELECT n.id, n.title, n.description, n.source, n.link, n.event_date, n.metadata,
|
||||
n.created_at, n.updated_at, n.context_id,
|
||||
COALESCE((SELECT JSON_GROUP_ARRAY(d.dimension)
|
||||
FROM node_dimensions d WHERE d.node_id = n.id), '[]') as dimensions_json,
|
||||
CASE
|
||||
WHEN c.id IS NULL THEN NULL
|
||||
ELSE json_object('id', c.id, 'name', c.name, 'description', c.description, 'icon', c.icon)
|
||||
@@ -164,134 +148,6 @@ function sanitizeTitle(title) {
|
||||
return clean.slice(0, 160);
|
||||
}
|
||||
|
||||
const STOP_WORDS = new Set([
|
||||
'a', 'an', 'and', 'are', 'as', 'at', 'be', 'by', 'for', 'from', 'has', 'i',
|
||||
'in', 'is', 'it', 'its', 'of', 'on', 'or', 'that', 'the', 'their', 'this',
|
||||
'to', 'was', 'with', 'you', 'your'
|
||||
]);
|
||||
|
||||
function normalizeText(value) {
|
||||
if (typeof value !== 'string') return '';
|
||||
return value.toLowerCase().replace(/[^a-z0-9\s]+/g, ' ').replace(/\s+/g, ' ').trim();
|
||||
}
|
||||
|
||||
function tokenize(value) {
|
||||
return normalizeText(value)
|
||||
.split(' ')
|
||||
.map((token) => token.trim())
|
||||
.filter((token) => token.length >= 2 && !STOP_WORDS.has(token));
|
||||
}
|
||||
|
||||
function uniqueTokens(values) {
|
||||
return [...new Set(values.flatMap((value) => tokenize(value || '')))];
|
||||
}
|
||||
|
||||
function safeStringify(value) {
|
||||
try {
|
||||
return JSON.stringify(value ?? {});
|
||||
} catch {
|
||||
return '';
|
||||
}
|
||||
}
|
||||
|
||||
function fetchContextCandidates() {
|
||||
return query(`
|
||||
WITH context_counts AS (
|
||||
SELECT c.id, c.name, c.description, COUNT(n.id) AS count
|
||||
FROM contexts c
|
||||
LEFT JOIN nodes n ON n.context_id = c.id
|
||||
GROUP BY c.id
|
||||
),
|
||||
ranked_anchors AS (
|
||||
SELECT
|
||||
c.id AS context_id,
|
||||
n.title AS anchor_title,
|
||||
n.description AS anchor_description,
|
||||
ROW_NUMBER() OVER (
|
||||
PARTITION BY c.id
|
||||
ORDER BY COUNT(e.id) DESC, n.updated_at DESC, n.id ASC
|
||||
) AS anchor_rank
|
||||
FROM contexts c
|
||||
LEFT JOIN nodes n ON n.context_id = c.id
|
||||
LEFT JOIN edges e ON (e.from_node_id = n.id OR e.to_node_id = n.id)
|
||||
GROUP BY c.id, n.id
|
||||
)
|
||||
SELECT
|
||||
cc.id,
|
||||
cc.name,
|
||||
cc.description,
|
||||
cc.count,
|
||||
ra.anchor_title,
|
||||
ra.anchor_description
|
||||
FROM context_counts cc
|
||||
LEFT JOIN ranked_anchors ra
|
||||
ON ra.context_id = cc.id
|
||||
AND ra.anchor_rank = 1
|
||||
ORDER BY cc.name COLLATE NOCASE ASC
|
||||
`).map((row) => ({
|
||||
id: Number(row.id),
|
||||
name: row.name,
|
||||
description: row.description ?? null,
|
||||
count: Number(row.count ?? 0),
|
||||
anchor_title: row.anchor_title ?? null,
|
||||
anchor_description: row.anchor_description ?? null,
|
||||
}));
|
||||
}
|
||||
|
||||
function scoreContextCandidate(candidate, input) {
|
||||
const titleText = normalizeText(input.title || '');
|
||||
const descriptionText = normalizeText(input.description || '');
|
||||
const sourceText = normalizeText(String(input.source || '').slice(0, 4000));
|
||||
const metadataText = normalizeText(safeStringify(input.metadata));
|
||||
const dimensionTokens = uniqueTokens(input.dimensions || []);
|
||||
const contextName = normalizeText(candidate.name);
|
||||
const contextNameTokens = tokenize(candidate.name);
|
||||
const contextDescriptorTokens = uniqueTokens([
|
||||
candidate.description,
|
||||
candidate.anchor_title,
|
||||
candidate.anchor_description,
|
||||
]);
|
||||
|
||||
let score = 0;
|
||||
if (contextName && (titleText.includes(contextName) || descriptionText.includes(contextName))) score += 80;
|
||||
if (contextName && sourceText.includes(contextName)) score += 40;
|
||||
|
||||
for (const token of contextNameTokens) {
|
||||
if (dimensionTokens.includes(token)) score += 30;
|
||||
if (titleText.includes(token)) score += 16;
|
||||
if (descriptionText.includes(token)) score += 12;
|
||||
if (sourceText.includes(token)) score += 6;
|
||||
if (metadataText.includes(token)) score += 4;
|
||||
}
|
||||
|
||||
for (const token of contextDescriptorTokens) {
|
||||
if (dimensionTokens.includes(token)) score += 8;
|
||||
if (titleText.includes(token)) score += 4;
|
||||
if (descriptionText.includes(token)) score += 3;
|
||||
if (sourceText.includes(token)) score += 2;
|
||||
}
|
||||
|
||||
return score;
|
||||
}
|
||||
|
||||
function inferBestContextIdForNode(input) {
|
||||
const contexts = fetchContextCandidates();
|
||||
if (contexts.length === 0) return null;
|
||||
|
||||
const ranked = contexts
|
||||
.map((context) => ({ context, score: scoreContextCandidate(context, input) }))
|
||||
.sort((a, b) => b.score - a.score || (b.context.count - a.context.count) || a.context.id - b.context.id);
|
||||
|
||||
const best = ranked[0];
|
||||
if (!best) return null;
|
||||
if (best.score > 0) return best.context.id;
|
||||
|
||||
const research = contexts.find((context) => context.name.trim().toLowerCase() === 'research');
|
||||
if (research) return research.id;
|
||||
|
||||
return best.context.id;
|
||||
}
|
||||
|
||||
/**
|
||||
* Create a new node.
|
||||
*/
|
||||
@@ -302,7 +158,6 @@ function createNode(nodeData) {
|
||||
source,
|
||||
link,
|
||||
event_date,
|
||||
dimensions = [],
|
||||
metadata = {},
|
||||
context_id
|
||||
} = nodeData;
|
||||
@@ -314,9 +169,7 @@ function createNode(nodeData) {
|
||||
const db = getDb();
|
||||
|
||||
const sourceToStore = source ?? ([title, description].filter(Boolean).join('\n\n').trim() || null);
|
||||
const effectiveContextId = context_id == null
|
||||
? inferBestContextIdForNode({ title, description, source: sourceToStore, dimensions, metadata: canonicalMetadata })
|
||||
: context_id;
|
||||
const effectiveContextId = context_id ?? null;
|
||||
|
||||
const nodeId = transaction(() => {
|
||||
const stmt = db.prepare(`
|
||||
@@ -338,16 +191,6 @@ function createNode(nodeData) {
|
||||
|
||||
const id = Number(result.lastInsertRowid);
|
||||
|
||||
// Insert dimensions
|
||||
if (dimensions.length > 0) {
|
||||
const dimStmt = db.prepare(
|
||||
'INSERT OR IGNORE INTO node_dimensions (node_id, dimension) VALUES (?, ?)'
|
||||
);
|
||||
for (const dimension of dimensions) {
|
||||
dimStmt.run(id, dimension);
|
||||
}
|
||||
}
|
||||
|
||||
return id;
|
||||
});
|
||||
|
||||
@@ -358,7 +201,7 @@ function createNode(nodeData) {
|
||||
* Update an existing node.
|
||||
*/
|
||||
function updateNode(id, updates, options = {}) {
|
||||
const { title, description, source, link, event_date, dimensions, metadata } = updates;
|
||||
const { title, description, source, link, event_date, metadata } = updates;
|
||||
const now = new Date().toISOString();
|
||||
const db = getDb();
|
||||
|
||||
@@ -415,14 +258,6 @@ function updateNode(id, updates, options = {}) {
|
||||
stmt.run(...params);
|
||||
}
|
||||
|
||||
// Handle dimensions separately
|
||||
if (Array.isArray(dimensions)) {
|
||||
db.prepare('DELETE FROM node_dimensions WHERE node_id = ?').run(id);
|
||||
const dimStmt = db.prepare('INSERT OR IGNORE INTO node_dimensions (node_id, dimension) VALUES (?, ?)');
|
||||
for (const dim of dimensions) {
|
||||
dimStmt.run(id, dim);
|
||||
}
|
||||
}
|
||||
});
|
||||
|
||||
return getNodeById(id);
|
||||
@@ -449,21 +284,15 @@ function getNodeCount() {
|
||||
|
||||
/**
|
||||
* Get knowledge graph context overview.
|
||||
* Returns stats, contexts, hub nodes, dimensions, and recent activity.
|
||||
* Returns stats, contexts, hub nodes, and recent activity.
|
||||
*/
|
||||
function getContext() {
|
||||
const nodeCount = query('SELECT COUNT(*) as count FROM nodes')[0].count;
|
||||
const edgeCount = query('SELECT COUNT(*) as count FROM edges')[0].count;
|
||||
|
||||
const dimensionService = require('./dimensionService');
|
||||
const dimensions = dimensionService.getDimensions();
|
||||
|
||||
const recentNodes = query(`
|
||||
SELECT n.id, n.title, n.description,
|
||||
GROUP_CONCAT(nd.dimension) as dimensions
|
||||
SELECT n.id, n.title, n.description
|
||||
FROM nodes n
|
||||
LEFT JOIN node_dimensions nd ON n.id = nd.node_id
|
||||
GROUP BY n.id
|
||||
ORDER BY n.created_at DESC
|
||||
LIMIT 5
|
||||
`);
|
||||
@@ -478,9 +307,8 @@ function getContext() {
|
||||
`);
|
||||
|
||||
return {
|
||||
stats: { nodeCount, edgeCount, dimensionCount: dimensions.length, contextCount: contextService.listContexts().length },
|
||||
stats: { nodeCount, edgeCount, dimensionCount: 0, contextCount: contextService.listContexts().length },
|
||||
contexts: contextService.listContexts(),
|
||||
dimensions,
|
||||
recentNodes,
|
||||
hubNodes
|
||||
};
|
||||
|
||||
Reference in New Issue
Block a user