Files
2026-05-01 07:51:22 -04:00

362 lines
14 KiB
JavaScript
Raw Permalink Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
import {
parseCSVHeaders,
parseCSV,
getUniqueNoteValues,
runAnalysis,
extractNumber,
computeNumericSeries,
} from '../src/lib/csvAnalysis';
// ── Fixtures ───────────────────────────────────────────────────────────────────
// Minimal CSV mimicking the real experiment file structure
const FIXTURE_CSV = `timestamp,frame_number,frame_line_status,note
0.0,1,10,
1.0,2,10,f
2.0,3,10,s
3.0,4,10,s
4.0,5,10,f
5.0,6,10,
6.0,7,10,s
7.0,8,10,f
8.0,9,10,f
9.0,10,10,s`;
// Same data but with Windows-style line endings
const FIXTURE_CSV_CRLF = FIXTURE_CSV.replace(/\n/g, '\r\n');
// CSV with quoted field containing a comma
const FIXTURE_CSV_QUOTED = `timestamp,note\n1.0,"hello, world"\n2.0,plain`;
// Out-of-order timestamps (should be sorted)
const FIXTURE_UNSORTED = `timestamp,note
3.0,s
1.0,f
2.0,s`;
const CLASSIFICATIONS = { f: 'Failure', s: 'Success' };
// ── parseCSVHeaders ────────────────────────────────────────────────────────────
describe('parseCSVHeaders', () => {
test('returns column names from first line', () => {
expect(parseCSVHeaders(FIXTURE_CSV)).toEqual([
'timestamp', 'frame_number', 'frame_line_status', 'note',
]);
});
test('handles CRLF line endings', () => {
expect(parseCSVHeaders(FIXTURE_CSV_CRLF)).toEqual([
'timestamp', 'frame_number', 'frame_line_status', 'note',
]);
});
test('handles single-line CSV (no newline)', () => {
expect(parseCSVHeaders('a,b,c')).toEqual(['a', 'b', 'c']);
});
});
// ── parseCSV ──────────────────────────────────────────────────────────────────
describe('parseCSV', () => {
test('returns headers and rows', () => {
const { headers, rows } = parseCSV(FIXTURE_CSV);
expect(headers).toEqual(['timestamp', 'frame_number', 'frame_line_status', 'note']);
expect(rows).toHaveLength(10);
});
test('row values are keyed by header name', () => {
const { rows } = parseCSV(FIXTURE_CSV);
expect(rows[0]).toEqual({ timestamp: '0.0', frame_number: '1', frame_line_status: '10', note: '' });
expect(rows[1]).toEqual({ timestamp: '1.0', frame_number: '2', frame_line_status: '10', note: 'f' });
});
test('handles CRLF line endings', () => {
const { rows } = parseCSV(FIXTURE_CSV_CRLF);
expect(rows).toHaveLength(10);
expect(rows[1].note).toBe('f');
});
test('handles quoted fields containing commas', () => {
const { rows } = parseCSV(FIXTURE_CSV_QUOTED);
expect(rows[0].note).toBe('hello, world');
expect(rows[1].note).toBe('plain');
});
test('skips blank lines', () => {
const csv = 'a,b\n1,2\n\n3,4\n';
const { rows } = parseCSV(csv);
expect(rows).toHaveLength(2);
});
});
// ── getUniqueNoteValues ────────────────────────────────────────────────────────
describe('getUniqueNoteValues', () => {
test('returns sorted unique non-empty values', () => {
const { rows } = parseCSV(FIXTURE_CSV);
expect(getUniqueNoteValues(rows, 'note')).toEqual(['f', 's']);
});
test('ignores empty strings', () => {
const rows = [{ note: '' }, { note: 'a' }, { note: '' }, { note: 'b' }];
expect(getUniqueNoteValues(rows, 'note')).toEqual(['a', 'b']);
});
test('trims whitespace before deduplication', () => {
const rows = [{ note: ' a ' }, { note: 'a' }, { note: 'b' }];
expect(getUniqueNoteValues(rows, 'note')).toEqual(['a', 'b']);
});
test('returns empty array when no non-empty notes', () => {
const rows = [{ note: '' }, { note: '' }];
expect(getUniqueNoteValues(rows, 'note')).toEqual([]);
});
});
// ── runAnalysis ────────────────────────────────────────────────────────────────
describe('runAnalysis', () => {
let result;
beforeEach(() => {
const { rows } = parseCSV(FIXTURE_CSV);
result = runAnalysis(rows, 'timestamp', 'note', CLASSIFICATIONS);
});
// Fixture note sequence (sorted by timestamp, empty rows excluded):
// t=1 f, t=2 s, t=3 s, t=4 f, t=6 s, t=7 f, t=8 f, t=9 s
// Categories: F S S F S F F S
test('totalNoteRows counts only rows with non-empty note', () => {
expect(result.totalNoteRows).toBe(8);
});
test('sequence preserves order and maps to categories', () => {
const notes = result.sequence.map(r => r.note);
expect(notes).toEqual(['f', 's', 's', 'f', 's', 'f', 'f', 's']);
const cats = result.sequence.map(r => r.category);
expect(cats).toEqual(['Failure', 'Success', 'Success', 'Failure', 'Success', 'Failure', 'Failure', 'Success']);
});
test('categoryCounts totals are correct', () => {
expect(result.categoryCounts['Failure'].total).toBe(4);
expect(result.categoryCounts['Success'].total).toBe(4);
});
test('categoryCounts byNote breakdown is correct', () => {
expect(result.categoryCounts['Failure'].byNote).toEqual({ f: 4 });
expect(result.categoryCounts['Success'].byNote).toEqual({ s: 4 });
});
test('runs (run-length encoding) are correct', () => {
// F S S F S F F S → F×1, S×2, F×1, S×1, F×2, S×1
expect(result.runs).toEqual([
{ category: 'Failure', length: 1 },
{ category: 'Success', length: 2 },
{ category: 'Failure', length: 1 },
{ category: 'Success', length: 1 },
{ category: 'Failure', length: 2 },
{ category: 'Success', length: 1 },
]);
});
test('consecutiveStats totalRuns', () => {
expect(result.consecutiveStats['Failure'].totalRuns).toBe(3);
expect(result.consecutiveStats['Success'].totalRuns).toBe(3);
});
test('consecutiveStats maxRun', () => {
expect(result.consecutiveStats['Failure'].maxRun).toBe(2);
expect(result.consecutiveStats['Success'].maxRun).toBe(2);
});
test('consecutiveStats avgRun', () => {
// Failure runs: [1,1,2] → avg = 4/3
expect(result.consecutiveStats['Failure'].avgRun).toBeCloseTo(4 / 3);
// Success runs: [2,1,1] → avg = 4/3
expect(result.consecutiveStats['Success'].avgRun).toBeCloseTo(4 / 3);
});
test('consecutiveStats runLengths array', () => {
expect(result.consecutiveStats['Failure'].runLengths).toEqual([1, 1, 2]);
expect(result.consecutiveStats['Success'].runLengths).toEqual([2, 1, 1]);
});
});
describe('runAnalysis — sorting', () => {
test('sorts rows by timestamp before building sequence', () => {
const { rows } = parseCSV(FIXTURE_UNSORTED);
// Input order: s(3), f(1), s(2) — sorted order: f(1), s(2), s(3)
const result = runAnalysis(rows, 'timestamp', 'note', CLASSIFICATIONS);
expect(result.sequence.map(r => r.note)).toEqual(['f', 's', 's']);
});
});
describe('runAnalysis — unclassified notes', () => {
test('unclassified notes appear in sequence with null category', () => {
const { rows } = parseCSV(FIXTURE_CSV);
// Only classify 'f'; leave 's' unclassified
const result = runAnalysis(rows, 'timestamp', 'note', { f: 'Failure' });
const nullCats = result.sequence.filter(r => r.category === null);
expect(nullCats.length).toBe(4); // 4 's' notes
});
test('unclassified notes are excluded from categoryCounts', () => {
const { rows } = parseCSV(FIXTURE_CSV);
const result = runAnalysis(rows, 'timestamp', 'note', { f: 'Failure' });
expect(result.categoryCounts['Success']).toBeUndefined();
});
test('unclassified notes do not break or extend runs', () => {
// Sequence with 'x' unclassified: f x f → should still be one Failure run of 2
const csv = 'ts,note\n1.0,f\n2.0,x\n3.0,f';
const { rows } = parseCSV(csv);
// 'x' is unclassified, 'f' → Failure
// After skipping 'x': f, f → one run of Failure×2
const result = runAnalysis(rows, 'ts', 'note', { f: 'Failure' });
expect(result.runs).toEqual([{ category: 'Failure', length: 2 }]);
});
});
describe('runAnalysis — multiple note values per category', () => {
test('groups multiple note values under one category', () => {
// The user's example: ss and s both belong to Success
const csv = 'ts,note\n1.0,ss\n2.0,s\n3.0,s';
const { rows } = parseCSV(csv);
const result = runAnalysis(rows, 'ts', 'note', { ss: 'Success', s: 'Success' });
expect(result.categoryCounts['Success'].total).toBe(3);
expect(result.categoryCounts['Success'].byNote).toEqual({ ss: 1, s: 2 });
});
test('run-length encoding with multiple note values in same category', () => {
const csv = 'ts,note\n1.0,ss\n2.0,s\n3.0,f';
const { rows } = parseCSV(csv);
const result = runAnalysis(rows, 'ts', 'note', { ss: 'Success', s: 'Success', f: 'Failure' });
// ss and s are both Success → run of 2, then Failure run of 1
expect(result.runs).toEqual([
{ category: 'Success', length: 2 },
{ category: 'Failure', length: 1 },
]);
});
});
describe('runAnalysis — edge cases', () => {
test('empty rows returns zero totals', () => {
const result = runAnalysis([], 'timestamp', 'note', CLASSIFICATIONS);
expect(result.totalNoteRows).toBe(0);
expect(result.sequence).toEqual([]);
expect(result.runs).toEqual([]);
expect(result.categoryCounts).toEqual({});
expect(result.consecutiveStats).toEqual({});
});
test('all notes unclassified', () => {
const { rows } = parseCSV(FIXTURE_CSV);
const result = runAnalysis(rows, 'timestamp', 'note', {});
expect(result.categoryCounts).toEqual({});
expect(result.runs).toEqual([]);
});
test('single note row', () => {
const csv = 'ts,note\n1.0,f';
const { rows } = parseCSV(csv);
const result = runAnalysis(rows, 'ts', 'note', { f: 'Failure' });
expect(result.totalNoteRows).toBe(1);
expect(result.runs).toEqual([{ category: 'Failure', length: 1 }]);
expect(result.consecutiveStats['Failure'].maxRun).toBe(1);
expect(result.consecutiveStats['Failure'].avgRun).toBe(1);
});
test('all same category', () => {
const csv = 'ts,note\n1.0,f\n2.0,f\n3.0,f';
const { rows } = parseCSV(csv);
const result = runAnalysis(rows, 'ts', 'note', { f: 'Failure' });
expect(result.runs).toEqual([{ category: 'Failure', length: 3 }]);
expect(result.consecutiveStats['Failure'].maxRun).toBe(3);
expect(result.consecutiveStats['Failure'].totalRuns).toBe(1);
});
});
describe('extractNumber', () => {
test('extracts positive decimal from prefixed string', () => {
expect(extractNumber('L25.5')).toBe(25.5);
});
test('extracts negative integer with letter prefix', () => {
expect(extractNumber('R-12')).toBe(-12);
});
test('extracts integer from underscore-separated label', () => {
expect(extractNumber('trial_007')).toBe(7);
});
test('does not parse scientific notation as one number', () => {
// Accepted limitation: regex matches "3.2", strips "e2"
expect(extractNumber('3.2e2')).toBe(3.2);
});
test('strips trailing units', () => {
expect(extractNumber('42 ms')).toBe(42);
});
test('returns null for purely alphabetic string', () => {
expect(extractNumber('abc')).toBeNull();
});
test('returns null for empty string', () => {
expect(extractNumber('')).toBeNull();
});
test('returns null for null/undefined', () => {
expect(extractNumber(null)).toBeNull();
expect(extractNumber(undefined)).toBeNull();
});
});
describe('computeNumericSeries', () => {
const rows = [
{ frame: '1', timestamp: '0.0', note: 'L25' },
{ frame: '3', timestamp: '2.0', note: 'L30.5' },
{ frame: '2', timestamp: '1.0', note: 'L40' },
{ frame: '4', timestamp: '3.0', note: 'F' },
{ frame: '5', timestamp: '4.0', note: 'abc' }, // bucket-member but no number
{ frame: 'x', timestamp: '5.0', note: 'L99' }, // bad x
{ frame: '6', timestamp: '6.0', note: '' }, // not in bucket
];
const bucket = { name: 'Lick latency', type: 'numerical', notes: ['L25', 'L30.5', 'L40', 'abc', 'L99'] };
test('returns sorted points with average and counts', () => {
const out = computeNumericSeries(rows, 'frame', 'note', bucket);
expect(out.points).toEqual([
{ x: 1, value: 25, raw: 'L25' },
{ x: 2, value: 40, raw: 'L40' },
{ x: 3, value: 30.5, raw: 'L30.5' },
]);
expect(out.avg).toBeCloseTo((25 + 40 + 30.5) / 3);
expect(out.skipped).toBe(2); // 'abc' (no number) + bad x
expect(out.total).toBe(5); // bucket-member rows
expect(out.xLabel).toBe('frame');
});
test('falls back to timestamp x-column when frame is empty string', () => {
const out = computeNumericSeries(rows, 'timestamp', 'note', bucket);
expect(out.xLabel).toBe('timestamp');
expect(out.points[0].x).toBe(0);
});
test('returns avg=null and empty points when nothing plottable', () => {
const onlyBad = [
{ frame: '1', note: 'abc' },
];
const b = { name: 'Z', type: 'numerical', notes: ['abc'] };
const out = computeNumericSeries(onlyBad, 'frame', 'note', b);
expect(out.points).toEqual([]);
expect(out.avg).toBeNull();
expect(out.skipped).toBe(1);
expect(out.total).toBe(1);
});
test('returns total=0 when bucket notes match nothing in rows', () => {
const b = { name: 'Empty', type: 'numerical', notes: ['nope'] };
const out = computeNumericSeries(rows, 'frame', 'note', b);
expect(out.total).toBe(0);
expect(out.points).toEqual([]);
expect(out.avg).toBeNull();
});
});