Files
experiments-database/frontend/src/lib/csvAnalysis.js
T
2026-05-01 07:51:22 -04:00

266 lines
9.4 KiB
JavaScript

/**
* CSV Analysis Library
*
* Method summary:
*
* 1. PARSING
* - parseCSVHeaders(text): reads only the first line, splits on comma
* (handles double-quoted fields with embedded commas).
* - parseCSV(text): full parse into array-of-objects. Each row is
* { [headerName]: value, ... }. Empty trailing fields are preserved as ''.
*
* 2. UNIQUE VALUES
* - getUniqueNoteValues(rows, noteCol): scans each row's noteCol,
* trims whitespace, ignores empty strings, returns sorted unique values.
*
* 3. ANALYSIS (runAnalysis)
* Input:
* rows — full parsed rows
* timestampCol — column name to sort by (numeric sort; falls back to row order)
* noteCol — column containing note values
* classifications — plain object: { noteValue: categoryName }
* unclassified notes are included in the sequence
* but excluded from category counts and run stats.
*
* Steps:
* a) Filter to rows where noteCol is non-empty.
* b) Sort by timestampCol (parseFloat; NaN values go to the end).
* c) Map each row to { note, category, timestamp }.
* d) CATEGORY COUNTS: for each category, count total occurrences
* and per-note-value breakdown.
* e) RUN-LENGTH ENCODING (consecutive stat):
* Walk the category sequence. Each time the category changes,
* start a new run. Uncategorised notes are skipped (do not break
* or continue a run). This gives an ordered list of
* { category, length } objects representing every streak.
* f) CONSECUTIVE STATS per category:
* From the runs list, group run lengths per category and compute
* totalRuns, maxRun, avgRun, and the full runLengths array.
*
* Output: { totalNoteRows, sequence, categoryCounts, runs, consecutiveStats }
*/
// ── CSV parsing ────────────────────────────────────────────────────────────────
function splitCSVLine(line) {
const fields = [];
let field = '';
let inQuotes = false;
for (let i = 0; i < line.length; i++) {
const ch = line[i];
if (ch === '"') {
if (inQuotes && line[i + 1] === '"') { field += '"'; i++; }
else { inQuotes = !inQuotes; }
} else if (ch === ',' && !inQuotes) {
fields.push(field);
field = '';
} else {
field += ch;
}
}
fields.push(field);
return fields;
}
/** Return column names from the first line of a CSV string. */
export function parseCSVHeaders(text) {
const firstLine = text.split(/\r?\n/)[0];
return splitCSVLine(firstLine);
}
/** Parse the full CSV into { headers, rows }. */
export function parseCSV(text) {
const lines = text.split(/\r?\n/);
const headers = splitCSVLine(lines[0]);
const rows = [];
for (let i = 1; i < lines.length; i++) {
const line = lines[i];
if (!line.trim()) continue;
const values = splitCSVLine(line);
const row = {};
for (let j = 0; j < headers.length; j++) {
row[headers[j]] = values[j] ?? '';
}
rows.push(row);
}
return { headers, rows };
}
// ── Unique note values ─────────────────────────────────────────────────────────
/** Return sorted unique non-empty values found in noteCol. */
export function getUniqueNoteValues(rows, noteCol) {
const seen = new Set();
for (const row of rows) {
const v = (row[noteCol] ?? '').trim();
if (v) seen.add(v);
}
return [...seen].sort();
}
// ── Frame gap detection ────────────────────────────────────────────────────────
/**
* Check whether a numeric column is strictly consecutive (no gaps, no duplicates).
* Returns { consecutive: bool, gaps: [{index, expected, actual, jump}], duplicates: [{index, value}] }
* Only examines rows where the column value parses as an integer.
*/
export function checkFrameConsecutive(rows, frameCol) {
const entries = [];
for (let i = 0; i < rows.length; i++) {
const v = rows[i][frameCol];
if (v == null || String(v).trim() === '') continue;
const n = parseInt(v, 10);
if (isNaN(n)) continue;
entries.push({ rowIndex: i, value: n });
}
const gaps = [];
const duplicates = [];
for (let i = 1; i < entries.length; i++) {
const diff = entries[i].value - entries[i - 1].value;
if (diff === 0) {
duplicates.push({ rowIndex: entries[i].rowIndex, value: entries[i].value });
} else if (diff !== 1) {
gaps.push({
rowIndex: entries[i].rowIndex,
expected: entries[i - 1].value + 1,
actual: entries[i].value,
jump: diff,
});
}
}
return { consecutive: gaps.length === 0 && duplicates.length === 0, gaps, duplicates, total: entries.length };
}
// ── Analysis ───────────────────────────────────────────────────────────────────
/**
* Run the full analysis.
*
* @param {object[]} rows - Parsed CSV rows
* @param {string} timestampCol - Column to sort by
* @param {string} noteCol - Column with note values
* @param {object} classifications - { noteValue: categoryName }
* @returns {object} Analysis result
*/
export function runAnalysis(rows, timestampCol, noteCol, classifications) {
// Compute session end from ALL rows (not just note rows)
let sessionEndTs = null;
for (const r of rows) {
const t = parseFloat(r[timestampCol]);
if (!isNaN(t) && (sessionEndTs === null || t > sessionEndTs)) sessionEndTs = t;
}
// a) Filter non-empty notes
const noteRows = rows.filter(r => (r[noteCol] ?? '').trim() !== '');
// b) Sort by timestamp (numeric); preserve original order for equal/NaN timestamps
noteRows.sort((a, b) => {
const ta = parseFloat(a[timestampCol]);
const tb = parseFloat(b[timestampCol]);
if (isNaN(ta) && isNaN(tb)) return 0;
if (isNaN(ta)) return 1;
if (isNaN(tb)) return -1;
return ta - tb;
});
// c) Build sequence
const sequence = noteRows.map(r => ({
note: r[noteCol].trim(),
category: classifications[r[noteCol].trim()] ?? null,
timestamp: r[timestampCol],
}));
// d) Category counts
const categoryCounts = {};
for (const { note, category } of sequence) {
if (!category) continue;
if (!categoryCounts[category]) categoryCounts[category] = { total: 0, byNote: {} };
categoryCounts[category].total++;
categoryCounts[category].byNote[note] = (categoryCounts[category].byNote[note] || 0) + 1;
}
// e) Run-length encoding (skip uncategorised entries — they neither break nor extend a run)
const runs = [];
for (const { category } of sequence) {
if (!category) continue;
if (runs.length === 0 || runs[runs.length - 1].category !== category) {
runs.push({ category, length: 1 });
} else {
runs[runs.length - 1].length++;
}
}
// f) Consecutive stats per category
const consecutiveStats = {};
for (const { category, length } of runs) {
if (!consecutiveStats[category]) {
consecutiveStats[category] = { runLengths: [], totalRuns: 0, maxRun: 0, avgRun: 0 };
}
const s = consecutiveStats[category];
s.runLengths.push(length);
s.totalRuns++;
if (length > s.maxRun) s.maxRun = length;
}
for (const cat of Object.keys(consecutiveStats)) {
const s = consecutiveStats[cat];
s.avgRun = s.runLengths.reduce((a, b) => a + b, 0) / s.runLengths.length;
}
return {
totalNoteRows: noteRows.length,
sessionEndTs,
sequence,
categoryCounts,
runs,
consecutiveStats,
};
}
// ── Numeric extraction (for numerical buckets) ────────────────────────────────
/**
* Extract the first signed decimal from a string. Returns null if none found.
* Examples: "L25.5" -> 25.5, "R-12" -> -12, "abc" -> null
*/
export function extractNumber(str) {
if (str == null) return null;
const m = String(str).match(/-?\d+(\.\d+)?/);
return m ? parseFloat(m[0]) : null;
}
/**
* Build a numeric-vs-x series for a numerical bucket.
*
* @param {object[]} rows - Parsed CSV rows
* @param {string} xColName - Column to plot on the x-axis (frame col, or timestamp as fallback)
* @param {string} noteCol - Column containing note values
* @param {object} bucket - { name, type:'numerical', notes: string[] }
* @returns {{points: {x:number,value:number,raw:string}[], avg:number|null, skipped:number, total:number, xLabel:string}}
*/
export function computeNumericSeries(rows, xColName, noteCol, bucket) {
const noteSet = new Set(bucket.notes);
const points = [];
let skipped = 0;
let total = 0;
for (const r of rows) {
const note = String(r[noteCol] ?? '').trim();
if (!noteSet.has(note)) continue;
total++;
const value = extractNumber(note);
const x = parseFloat(r[xColName]);
if (value == null || Number.isNaN(x)) { skipped++; continue; }
points.push({ x, value, raw: note });
}
points.sort((a, b) => a.x - b.x);
const avg = points.length === 0
? null
: points.reduce((s, p) => s + p.value, 0) / points.length;
return { points, avg, skipped, total, xLabel: xColName };
}