diff --git a/backend/db/notes.dao.js b/backend/db/notes.dao.js index 5f79f64..b73a8bc 100644 --- a/backend/db/notes.dao.js +++ b/backend/db/notes.dao.js @@ -7,105 +7,104 @@ import { batch } from '../lib/streams.js'; const SELECT_NOTES = 'SELECT id, text, x, y, author, color FROM notes ORDER BY id'; // Postgres caps a statement at 65535 bind parameters; six columns per note -// leaves 10922 as the hard ceiling, so stay well under it. +// leaves 10922 as the hard ceiling. const INSERT_BATCH_SIZE = 1000; export const getAllNotes = async () => { - const { rows } = await query(SELECT_NOTES); - return rows; + const { rows } = await query(SELECT_NOTES); + return rows; }; /** - * Streams every note as an object-mode Readable. The pooled client is checked - * out for the life of the stream and released once it ends, errors, or is - * destroyed early by a consumer. + * Streams every note as an object-mode Readable. The pooled client is released on end, error, or + * destruction by consumer. * * @returns {Promise} */ export const streamAllNotes = async () => { - const client = await getPool().connect(); + const client = await getPool().connect(); - let released = false; - const release = () => { - if (released) return; - released = true; - client.release(); - }; + let released = false; + const release = () => { + if (released) return; + released = true; + client.release(); + }; - try { - const rows = client.query(new QueryStream(SELECT_NOTES)); - rows.once('end', release); - rows.once('error', release); - rows.once('close', release); - return rows; - } catch (err) { - release(); - throw err; - } + try { + const rows = client.query(new QueryStream(SELECT_NOTES)); + rows.once('end', release); + rows.once('error', release); + rows.once('close', release); + return rows; + } catch (err) { + release(); + throw err; + } }; export const getNoteById = async (id) => { - const { rows } = await query( - 'SELECT id, text, x, y, author, color FROM notes WHERE id = $1', - [id] - ); - return rows[0] || null; + const { rows } = await query( + 'SELECT id, text, x, y, author, color FROM notes WHERE id = $1', + [id] + ); + return rows[0] || null; }; export const createNote = async (note) => { - const { rows } = await query( - `INSERT INTO notes (id, text, x, y, author, color) + const { rows } = await query( + `INSERT INTO notes (id, text, x, y, author, color) VALUES ($1, $2, $3, $4, $5, $6) RETURNING id, text, x, y, author, color`, - [note.id, note.text, note.x ?? 0, note.y ?? 0, note.author, note.color ?? 'yellow'] - ); - return rows[0]; + [note.id, note.text, note.x ?? 0, note.y ?? 0, note.author, note.color ?? 'yellow'] + ); + return rows[0]; }; const insertNoteBatch = async (notes) => { - const values = []; - const placeholders = []; + const values = []; + const placeholders = []; - notes.forEach((note, i) => { - const offset = i * 6; - placeholders.push( - `($${offset + 1}, $${offset + 2}, $${offset + 3}, $${offset + 4}, $${offset + 5}, $${offset + 6})` - ); - values.push( - note.id, - note.text, - note.x ?? 0, - note.y ?? 0, - note.author, - note.color ?? 'yellow' - ); - }); + notes.forEach((note, i) => { + const offset = i * 6; + placeholders.push( + `($${offset + 1}, $${offset + 2}, $${offset + 3}, $${offset + 4}, $${offset + 5}, $${offset + 6})` + ); + values.push( + note.id, + note.text, + note.x ?? 0, + note.y ?? 0, + note.author, + note.color ?? 'yellow' + ); + }); - const { rows } = await query( - `INSERT INTO notes (id, text, x, y, author, color) + const { rows } = await query( + `INSERT INTO notes (id, text, x, y, author, color) VALUES ${placeholders.join(', ')} RETURNING id, text, x, y, author, color`, - values - ); - return rows; + values + ); + return rows; }; export const createNotes = async (notes) => { - if (!notes || notes.length === 0) { - return []; - } - - const inserted = []; - - await pipeline( - Readable.from(notes, { objectMode: true }), - batch(INSERT_BATCH_SIZE), - async (batches) => { - for await (const chunk of batches) { - inserted.push(...await insertNoteBatch(chunk)); - } + if (!notes || notes.length === 0) { + return []; } - ); - return inserted; + const inserted = []; + + await pipeline( + Readable.from(notes, { objectMode: true }), + batch(INSERT_BATCH_SIZE), + async (batches) => { + for await (const chunk of batches) { + inserted.push(...await insertNoteBatch(chunk)); + } + } + ); + + return inserted; }; diff --git a/backend/services/embedding.service.js b/backend/services/embedding.service.js index ffb21e1..e41cb36 100644 --- a/backend/services/embedding.service.js +++ b/backend/services/embedding.service.js @@ -4,10 +4,9 @@ import { Readable } from "node:stream"; import { batch } from "../lib/streams.js"; const client = new VoyageAIClient({ - apiKey: process.env.VOYAGEAI_API_KEY, + apiKey: process.env.VOYAGEAI_API_KEY, }); -// Well under Voyage's per-request input and token ceilings. const EMBED_BATCH_SIZE = 128; /** @@ -15,28 +14,28 @@ const EMBED_BATCH_SIZE = 128; * @returns {Promise>} noteId → embedding vector */ export const embedNotes = async (notes) => { - const embeddingMap = new Map(); + const embeddingMap = new Map(); - if (!notes || notes.length === 0) { - return embeddingMap; - } - - await pipeline( - Readable.from(notes, { objectMode: true }), - batch(EMBED_BATCH_SIZE), - async (batches) => { - for await (const chunk of batches) { - const response = await client.embed({ - input: chunk.map((n) => n.text), - model: "voyage-3", - }); - - response.data.forEach((item, i) => { - embeddingMap.set(chunk[i].id, item.embedding); - }); - } + if (!notes || notes.length === 0) { + return embeddingMap; } - ); - return embeddingMap; + await pipeline( + Readable.from(notes, { objectMode: true }), + batch(EMBED_BATCH_SIZE), + async (batches) => { + for await (const chunk of batches) { + const response = await client.embed({ + input: chunk.map((n) => n.text), + model: "voyage-3", + }); + + response.data.forEach((item, i) => { + embeddingMap.set(chunk[i].id, item.embedding); + }); + } + } + ); + + return embeddingMap; }; diff --git a/backend/services/validation.service.js b/backend/services/validation.service.js index 61cc3a3..9f99fea 100644 --- a/backend/services/validation.service.js +++ b/backend/services/validation.service.js @@ -1,68 +1,67 @@ /** - * Structural validation: confirms the LLM output is well-formed - * before it reaches the frontend. + * Structural validation for LLM output * * @param {Array<{label: string, noteIds: string[]}>} clusters * @param {string[]} inputNoteIds - the original note IDs that were sent to the LLM * @returns {{valid: boolean, reasons: string[]}} */ export const validateStructure = (clusters, inputNoteIds) => { - const reasons = []; + const reasons = []; - if (!Array.isArray(clusters) || clusters.length === 0) { - return { valid: false, reasons: ['Response is not a non-empty array'] }; - } - - const assignedIds = []; - for (const cluster of clusters) { - if (!cluster.label || typeof cluster.label !== 'string') { - reasons.push(`Cluster missing a valid label`); + if (!Array.isArray(clusters) || clusters.length === 0) { + return { valid: false, reasons: ['Response is not a non-empty array'] }; } - if (!Array.isArray(cluster.noteIds) || cluster.noteIds.length === 0) { - reasons.push(`Cluster "${cluster.label ?? '(unlabeled)'}" has no noteIds`); + + const assignedIds = []; + for (const cluster of clusters) { + if (!cluster.label || typeof cluster.label !== 'string') { + reasons.push(`Cluster missing a valid label`); + } + if (!Array.isArray(cluster.noteIds) || cluster.noteIds.length === 0) { + reasons.push(`Cluster "${cluster.label ?? '(unlabeled)'}" has no noteIds`); + } + assignedIds.push(...(cluster.noteIds ?? [])); } - assignedIds.push(...(cluster.noteIds ?? [])); - } - const inputSet = new Set(inputNoteIds); - const assignedSet = new Set(assignedIds); + const inputSet = new Set(inputNoteIds); + const assignedSet = new Set(assignedIds); - if (assignedIds.length !== assignedSet.size) { - reasons.push('One or more notes appear in multiple clusters'); - } + if (assignedIds.length !== assignedSet.size) { + reasons.push('One or more notes appear in multiple clusters'); + } - const missing = inputNoteIds.filter((id) => !assignedSet.has(id)); - if (missing.length > 0) { - reasons.push(`Notes missing from clusters: ${missing.join(', ')}`); - } + const missing = inputNoteIds.filter((id) => !assignedSet.has(id)); + if (missing.length > 0) { + reasons.push(`Notes missing from clusters: ${missing.join(', ')}`); + } - const extra = assignedIds.filter((id) => !inputSet.has(id)); - if (extra.length > 0) { - reasons.push(`Unknown noteIds in clusters: ${[...new Set(extra)].join(', ')}`); - } + const extra = assignedIds.filter((id) => !inputSet.has(id)); + if (extra.length > 0) { + reasons.push(`Unknown noteIds in clusters: ${[...new Set(extra)].join(', ')}`); + } - if (clusters.length > inputNoteIds.length) { - reasons.push(`More clusters (${clusters.length}) than notes (${inputNoteIds.length})`); - } + if (clusters.length > inputNoteIds.length) { + reasons.push(`More clusters (${clusters.length}) than notes (${inputNoteIds.length})`); + } - return { valid: reasons.length === 0, reasons }; + return { valid: reasons.length === 0, reasons }; }; const cosineSimilarity = (a, b) => { - let dot = 0; - let magA = 0; - let magB = 0; - for (let i = 0; i < a.length; i++) { - dot += a[i] * b[i]; - magA += a[i] * a[i]; - magB += b[i] * b[i]; - } - const denom = Math.sqrt(magA) * Math.sqrt(magB); - return denom === 0 ? 0 : dot / denom; + let dot = 0; + let magA = 0; + let magB = 0; + for (let i = 0; i < a.length; i++) { + dot += a[i] * b[i]; + magA += a[i] * a[i]; + magB += b[i] * b[i]; + } + const denom = Math.sqrt(magA) * Math.sqrt(magB); + return denom === 0 ? 0 : dot / denom; }; /** - * Computes a silhouette-style cohesion score for the clustering. + * Computes silhouette-style cohesion score for the clustering. * * For each note, measures how much more similar it is to its own cluster * versus the nearest neighboring cluster. Returns a score in [-1, 1] @@ -73,57 +72,57 @@ const cosineSimilarity = (a, b) => { * @returns {number} average silhouette score */ export const computeCohesionScore = (clusters, embeddingMap) => { - if (clusters.length <= 1) return 1.0; + if (clusters.length <= 1) return 1.0; - const scores = []; + const scores = []; - for (let ci = 0; ci < clusters.length; ci++) { - const clusterIds = clusters[ci].noteIds; - if (clusterIds.length <= 1) { - scores.push(0); - continue; + for (let ci = 0; ci < clusters.length; ci++) { + const clusterIds = clusters[ci].noteIds; + if (clusterIds.length <= 1) { + scores.push(0); + continue; + } + + for (const noteId of clusterIds) { + const vec = embeddingMap.get(noteId); + if (!vec) continue; + + // a(i): avg distance to other notes in same cluster + let intraSum = 0; + let intraCount = 0; + for (const otherId of clusterIds) { + if (otherId === noteId) continue; + const otherVec = embeddingMap.get(otherId); + if (!otherVec) continue; + intraSum += 1 - cosineSimilarity(vec, otherVec); + intraCount++; + } + const a = intraCount > 0 ? intraSum / intraCount : 0; + + // b(i): min avg distance to notes in any other cluster + let b = Infinity; + for (let oi = 0; oi < clusters.length; oi++) { + if (oi === ci) continue; + const otherClusterIds = clusters[oi].noteIds; + let interSum = 0; + let interCount = 0; + for (const otherId of otherClusterIds) { + const otherVec = embeddingMap.get(otherId); + if (!otherVec) continue; + interSum += 1 - cosineSimilarity(vec, otherVec); + interCount++; + } + if (interCount > 0) { + b = Math.min(b, interSum / interCount); + } + } + if (b === Infinity) b = 0; + + const max = Math.max(a, b); + scores.push(max === 0 ? 0 : (b - a) / max); + } } - for (const noteId of clusterIds) { - const vec = embeddingMap.get(noteId); - if (!vec) continue; - - // a(i): avg distance to other notes in same cluster - let intraSum = 0; - let intraCount = 0; - for (const otherId of clusterIds) { - if (otherId === noteId) continue; - const otherVec = embeddingMap.get(otherId); - if (!otherVec) continue; - intraSum += 1 - cosineSimilarity(vec, otherVec); - intraCount++; - } - const a = intraCount > 0 ? intraSum / intraCount : 0; - - // b(i): min avg distance to notes in any other cluster - let b = Infinity; - for (let oi = 0; oi < clusters.length; oi++) { - if (oi === ci) continue; - const otherClusterIds = clusters[oi].noteIds; - let interSum = 0; - let interCount = 0; - for (const otherId of otherClusterIds) { - const otherVec = embeddingMap.get(otherId); - if (!otherVec) continue; - interSum += 1 - cosineSimilarity(vec, otherVec); - interCount++; - } - if (interCount > 0) { - b = Math.min(b, interSum / interCount); - } - } - if (b === Infinity) b = 0; - - const max = Math.max(a, b); - scores.push(max === 0 ? 0 : (b - a) / max); - } - } - - if (scores.length === 0) return 0; - return scores.reduce((sum, s) => sum + s, 0) / scores.length; + if (scores.length === 0) return 0; + return scores.reduce((sum, s) => sum + s, 0) / scores.length; };