diff --git a/src/App.tsx b/src/App.tsx index b690722..f0e41d2 100644 --- a/src/App.tsx +++ b/src/App.tsx @@ -8,8 +8,8 @@ import PlatesGrid from './components/PlatesGrid'; import QualityMetricsPanel from './components/QualityMetricsPanel'; import QualityLegend from './components/QualityLegend'; import SubjectPlacementPanel from './components/SubjectPlacementPanel'; -import { SearchData, RandomizationAlgorithm, GroupingConstraint, GroupValidationResult, RepeatedMeasuresConfig } from './utils/types'; -import { downloadCSV, buildProcessedSearches, getCovariateKey, getQualityLevelColor, formatScore, withTimestamp, buildLayoutFileName } from './utils/utils'; +import { SearchData, RandomizationAlgorithm, GroupingConstraint, GroupValidationResult, RepeatedMeasuresConfig, NaPolicy, DEFAULT_NA_POLICY } from './utils/types'; +import { downloadCSV, buildProcessedSearches, getCovariateKey, getQualityLevelColor, formatScore, withTimestamp, buildLayoutFileName, detectNaTypeValues } from './utils/utils'; import { exportToExcel } from './utils/excelExport'; import { serializeLayout, @@ -115,6 +115,9 @@ const App: React.FC = () => { const [qcColumn, setQcColumn] = useState(''); const [qcColumnValues, setQcColumnValues] = useState([]); const [selectedQcValues, setSelectedQcValues] = useState([]); + // Global N/A grouping choice. Set from the "N/A values" checklist when the data mixes + // spellings; the default folds nothing extra (blank stays distinct, spellings stay literal). + const [naPolicy, setNaPolicy] = useState(DEFAULT_NA_POLICY); // Algorithm selection const [selectedAlgorithm, setSelectedAlgorithm] = useState(defaultAlgorithm); @@ -217,6 +220,9 @@ const App: React.FC = () => { setGroupingConstraint('none'); setGroupValidation(null); + // N/A grouping (re-derived from the new data by the upload effect) + setNaPolicy(DEFAULT_NA_POLICY); + // Algorithm and plate dimensions (back to defaults) setSelectedAlgorithm(defaultAlgorithm); setKeepEmptyInLastPlate(false); @@ -254,6 +260,16 @@ const App: React.FC = () => { // IDs blank) still replaces the previous design instead of leaving it on screen. if (selectedFileName) { clearConfigAndLayout(); + // Initialize the N/A policy from the new data. When a column mixes spellings, default to + // folding every detected spelling (all checklist boxes checked); otherwise fold nothing. + setNaPolicy( + naDetection.hasAmbiguousColumn + ? { + foldBlank: naDetection.spellings.has(''), + foldSpellings: Array.from(naDetection.spellings).filter(t => t !== '' && t !== 'N/A'), + } + : DEFAULT_NA_POLICY + ); } // eslint-disable-next-line react-hooks/exhaustive-deps }, [selectedFileName, searches.length]); @@ -292,6 +308,31 @@ const App: React.FC = () => { setShowSubjectPlacements(false); }; + // Scan every metadata column of the uploaded data for N/A-type spellings. Drives the "N/A + // values" checklist and, at upload, the default policy. The ID column is not metadata, so it + // is excluded automatically. + const naDetection = useMemo( + () => detectNaTypeValues(searches, availableColumns.filter(col => col !== selectedIdColumn)), + [searches, availableColumns, selectedIdColumn] + ); + + // Toggle one entry of the "N/A values" checklist. Blank uses the empty-string token. The + // literal N/A is always folded (its box is disabled), so it never reaches here. Changing the + // policy invalidates the current layout, matching a covariate change, so the user re-Generates. + const handleNaPolicyToggle = (token: string) => { + setNaPolicy(prev => { + if (token === '') return { ...prev, foldBlank: !prev.foldBlank }; + const folded = prev.foldSpellings.includes(token); + return { + ...prev, + foldSpellings: folded + ? prev.foldSpellings.filter(s => s !== token) + : [...prev.foldSpellings, token], + }; + }); + resetCovariateState(); + }; + // Derive the available QC values for the chosen QC column. This only recomputes the // list of checkboxes to show; it must NOT reset the current selection. A layout load // sets qcColumn and searches together, which retriggers this effect, and resetting here @@ -422,6 +463,7 @@ const App: React.FC = () => { selectedCovariates, qcColumn, selectedQcValues, + naPolicy, }); }; @@ -447,7 +489,8 @@ const App: React.FC = () => { keepEmptyInLastPlate, plateRows, plateColumns, - repeatedMeasuresConfig + repeatedMeasuresConfig, + naPolicy ); if (success) { @@ -465,7 +508,8 @@ const App: React.FC = () => { searches, selectedCovariates, qcColumn, - selectedQcValues + selectedQcValues, + naPolicy ); } } catch (err: any) { @@ -495,6 +539,7 @@ const App: React.FC = () => { groupingConstraint, // Metadata columns in display order (every column except the ID column). metadataColumns: availableColumns.filter(col => col !== selectedIdColumn), + naPolicy, }); // Save layout handler - export the layout together with the settings that produced it. @@ -511,7 +556,8 @@ const App: React.FC = () => { appVersion: packageJson.version, }); } catch (e) { - // serializeLayout throws if a sample is not on the grid. Show it instead of downloading. + // serializeLayout throws if a sample is not on the grid, or if there are no covariate + // colors to save. Show the message instead of downloading. setLoadWarning((e as Error).message); return; } @@ -557,6 +603,7 @@ const App: React.FC = () => { setPlateColumns(settings.plateColumns); setSubjectColumn(settings.subjectColumn); setGroupingConstraint(settings.groupingConstraint); + setNaPolicy(settings.naPolicy); // Restore the plates directly (no re-randomization). restoreLayout(plates, plateAssignmentsToRestore); @@ -579,7 +626,8 @@ const App: React.FC = () => { loadedSearches, settings.selectedCovariates, settings.qcColumn, - settings.selectedQcValues + settings.selectedQcValues, + settings.naPolicy ); // Clear any stale highlight/error. @@ -619,11 +667,13 @@ const App: React.FC = () => { samples: loadedSearches, } = buildPlatesFromRows(parsed.rows, settings); - // Recompute covariate keys / QC flags (shared references update the plates too). + // Recompute covariate keys / QC flags (shared references update the plates too). Use the + // saved N/A policy so the derived keys match the stored covariate colors. buildProcessedSearches(loadedSearches, { selectedCovariates: settings.selectedCovariates, qcColumn: settings.qcColumn, selectedQcValues: settings.selectedQcValues, + naPolicy: settings.naPolicy, }); applyLoadedLayout( @@ -746,7 +796,8 @@ const App: React.FC = () => { numRows: plateRows, numColumns: plateColumns, inputFileName: selectedFileName, - qcColumn: qcColumn || undefined + qcColumn: qcColumn || undefined, + naPolicy }); }; @@ -772,7 +823,8 @@ const App: React.FC = () => { keepEmptyInLastPlate, plateRows, plateColumns, - repeatedMeasuresConfig + repeatedMeasuresConfig, + naPolicy ); } }; @@ -799,7 +851,8 @@ const App: React.FC = () => { keepEmptyInLastPlate, plateRows, plateColumns, - repeatedMeasuresConfig + repeatedMeasuresConfig, + naPolicy ); // Quality metrics will be recalculated automatically via useEffect } @@ -947,6 +1000,9 @@ const App: React.FC = () => { qcColumnValues={qcColumnValues} selectedQcValues={selectedQcValues} onQcValueToggle={handleQcValueToggle} + naDetection={naDetection} + naPolicy={naPolicy} + onNaPolicyToggle={handleNaPolicyToggle} selectedAlgorithm={selectedAlgorithm} onAlgorithmChange={handleAlgorithmChange} keepEmptyInLastPlate={keepEmptyInLastPlate} @@ -1097,6 +1153,7 @@ const App: React.FC = () => { onReRandomizePlate={handleReRandomizePlate} qualityMetrics={metrics ?? undefined} subjectColumn={subjectColumn || undefined} + qcColumn={qcColumn || undefined} /> )} @@ -1120,6 +1177,7 @@ const App: React.FC = () => { plateQuality={selectedPlateIndex !== null ? metrics?.plateDiversity.plateScores.find(score => score.plateIndex === selectedPlateIndex) : undefined} randomizedPlates={randomizedPlates} numPlates={randomizedPlates.length} + naPolicy={naPolicy} /> {/* Quality Assessment Modal */} diff --git a/src/algorithms/repeatedMeasuresDistribution.ts b/src/algorithms/repeatedMeasuresDistribution.ts index b07613d..25c1c60 100644 --- a/src/algorithms/repeatedMeasuresDistribution.ts +++ b/src/algorithms/repeatedMeasuresDistribution.ts @@ -1,4 +1,4 @@ -import { SearchData, SubjectGroup, GroupingConstraint, GroupValidationResult, RepeatedMeasuresConfig, BlockType } from '../utils/types'; +import { SearchData, SubjectGroup, GroupingConstraint, GroupValidationResult, RepeatedMeasuresConfig, BlockType, NaPolicy, DEFAULT_NA_POLICY } from '../utils/types'; import { shuffleArray, groupByCovariates, buildCovariateKey } from '../utils/utils'; import { greedyPlaceInRow } from './greedySpatialPlacement'; import { distributeToBlocks, calculateExpectedMinimums } from './balancedRandomization'; @@ -1165,7 +1165,8 @@ export function groupAwareRandomization( repeatedMeasuresConfig: RepeatedMeasuresConfig, keepEmptyInLastPlate: boolean, numRows: number, - numColumns: number + numColumns: number, + naPolicy: NaPolicy = DEFAULT_NA_POLICY ): { plates: (SearchData | undefined)[][][]; plateAssignments?: Map; @@ -1188,7 +1189,7 @@ export function groupAwareRandomization( // Use the single key builder so escaping can't drift from buildProcessedSearches. for (const sample of experimentalSamples) { if (!sample.covariateKey && selectedCovariates.length > 0) { - sample.covariateKey = buildCovariateKey(sample, { selectedCovariates }); + sample.covariateKey = buildCovariateKey(sample, { selectedCovariates, naPolicy }); } } diff --git a/src/components/ConfigurationForm.tsx b/src/components/ConfigurationForm.tsx index 60d5d97..6ecdaa6 100644 --- a/src/components/ConfigurationForm.tsx +++ b/src/components/ConfigurationForm.tsx @@ -1,5 +1,6 @@ import React, { useState } from 'react'; -import { RandomizationAlgorithm, getAlgorithmName, getAlgorithmDescription, getAlgorithmsInDisplayOrder, GroupingConstraint, GroupValidationResult, SubjectGroup } from '../utils/types'; +import { RandomizationAlgorithm, getAlgorithmName, getAlgorithmDescription, getAlgorithmsInDisplayOrder, GroupingConstraint, GroupValidationResult, SubjectGroup, NaPolicy } from '../utils/types'; +import { NaDetectionResult } from '../utils/utils'; interface ConfigurationFormProps { availableColumns: string[]; @@ -15,6 +16,9 @@ interface ConfigurationFormProps { qcColumnValues: string[]; selectedQcValues: string[]; onQcValueToggle: (value: string) => void; + naDetection: NaDetectionResult; + naPolicy: NaPolicy; + onNaPolicyToggle: (token: string) => void; selectedAlgorithm: RandomizationAlgorithm; onAlgorithmChange: (event: React.ChangeEvent) => void; keepEmptyInLastPlate: boolean; @@ -46,6 +50,9 @@ const ConfigurationForm: React.FC = ({ qcColumnValues: qcColumnValues, selectedQcValues: selectedQcValues, onQcValueToggle: onQcValueToggle, + naDetection, + naPolicy, + onNaPolicyToggle, selectedAlgorithm, onAlgorithmChange, keepEmptyInLastPlate, @@ -123,6 +130,16 @@ const ConfigurationForm: React.FC = ({ const groupSummary = subjectColumn ? buildGroupSummary() : null; + // "N/A values" checklist: shown only when some column mixes two or more N/A-type spellings. + // The union of spellings is listed with the literal N/A first, then blank, then the rest + // alphabetically. + const showNaValues = naDetection.hasAmbiguousColumn; + const naTokenRank = (token: string) => (token === 'N/A' ? 0 : token === '' ? 1 : 2); + const naTokens = Array.from(naDetection.spellings).sort((a, b) => { + const rankDiff = naTokenRank(a) - naTokenRank(b); + return rankDiff !== 0 ? rankDiff : a.localeCompare(b); + }); + return (
{/* Uploaded file name, shown above the configuration options */} @@ -424,6 +441,46 @@ const ConfigurationForm: React.FC = ({ )}
+ {/* N/A values: one global choice shown only when the data mixes not-applicable spellings */} + {showNaValues && ( +
+
+
+ + + The input contains the following values that could be interpreted as N/A. Checked + values will be grouped as N/A. Uncheck any that should be kept separate. + +
+ {naTokens.map((token) => { + const isBlank = token === ''; + const isLiteralNa = token === 'N/A'; + const label = isBlank ? '(blank)' : token; + const checked = isLiteralNa + ? true + : isBlank + ? naPolicy.foldBlank + : naPolicy.foldSpellings.includes(token); + return ( + + ); + })} +
+
+
+
+ )} + {/* Non-greedy Algorithm Options */} {selectedAlgorithm !== 'greedy' && (
@@ -659,6 +716,18 @@ const styles = { borderRadius: '4px', border: '1px solid #ddd', }, + naValuesContainer: { + padding: '12px 15px', + backgroundColor: '#fff', + borderRadius: '6px', + border: '1px solid #ddd', + }, + naCheckboxGroup: { + display: 'flex', + flexWrap: 'wrap' as const, + gap: '16px', + marginTop: '8px', + }, checkboxGroup: { display: 'flex', flexDirection: 'column' as const, diff --git a/src/components/Plate.tsx b/src/components/Plate.tsx index 87f46d3..12a46a0 100644 --- a/src/components/Plate.tsx +++ b/src/components/Plate.tsx @@ -55,18 +55,26 @@ const createTooltipText = ( rowIndex: number, columnIndex: number, selectedCovariates: string[], - subjectColumn?: string + subjectColumn?: string, + qcColumn?: string ): string => { const position = `${getRowLabel(rowIndex)}${columnIndex + 1}`; const subjectInfo = subjectColumn && search.metadata[subjectColumn] ? `\n${subjectColumn}: ${search.metadata[subjectColumn]}` : ''; + // A QC sample's QC value is part of its identity but is not a covariate, so it would otherwise + // not appear. Show it here, unless the QC column is already one of the selected covariates. + const qcInfo = qcColumn && search.isQC && !selectedCovariates.includes(qcColumn) + ? `\n${qcColumn}: ${search.metadata[qcColumn] ?? ''}` + : ''; + // Per-sample view: show the raw typed value. A blank cell shows blank, so a genuinely-missing + // value is not mislabeled as N/A (which now means a folded "not applicable" group). const covariateInfo = selectedCovariates.length > 0 ? '\n' + selectedCovariates - .map(cov => `${cov}: ${search.metadata[cov] || 'N/A'}`) + .map(cov => `${cov}: ${search.metadata[cov] ?? ''}`) .join(', ') : ''; - return `${search.name} (${position})${subjectInfo}${covariateInfo}`; + return `${search.name} (${position})${subjectInfo}${qcInfo}${covariateInfo}`; }; interface PlateProps { @@ -84,6 +92,7 @@ interface PlateProps { plateQuality?: PlateQualityScore; onReRandomizePlate?: (plateIndex: number) => void; subjectColumn?: string; + qcColumn?: string; numPlates?: number; } @@ -102,6 +111,7 @@ const Plate: React.FC = ({ plateQuality, onReRandomizePlate, subjectColumn, + qcColumn, numPlates = 1 }) => { @@ -227,17 +237,23 @@ const Plate: React.FC = ({ {`${subjectColumn}: ${search.metadata[subjectColumn]}`}
)} - {selectedCovariates.map((covariate: string) => - search.metadata[covariate] ? ( -
- {`${covariate}: ${search.metadata[covariate]}`} -
- ) : null + {qcColumn && search.isQC && !selectedCovariates.includes(qcColumn) && ( + // A QC sample's QC value is part of its identity but is not a covariate; show it here. +
+ {`${qcColumn}: ${search.metadata[qcColumn] ?? ''}`} +
)} + {selectedCovariates.map((covariate: string) => ( + // Per-sample view: always show the covariate row, with the raw typed value. A blank + // cell shows "Covariate:" with no value, matching the compact-view tooltip. +
+ {`${covariate}: ${search.metadata[covariate] ?? ''}`} +
+ ))} ); - }, [covariateColors, selectedCovariates, currentStyles, onDragStart, compact, subjectColumn]); + }, [covariateColors, selectedCovariates, currentStyles, onDragStart, compact, subjectColumn, qcColumn]); // Unified cell renderer const renderSearchCell = useCallback((search: SearchData, isHighlighted: boolean) => { @@ -354,7 +370,7 @@ const Plate: React.FC = ({ onDrop={(event) => handleDrop(event, rowIndex, columnIndex)} title={ compact && search - ? createTooltipText(search, rowIndex, columnIndex, selectedCovariates, subjectColumn) + ? createTooltipText(search, rowIndex, columnIndex, selectedCovariates, subjectColumn, qcColumn) : undefined } > diff --git a/src/components/PlateDetailsModal.tsx b/src/components/PlateDetailsModal.tsx index 57282b0..4b212f9 100644 --- a/src/components/PlateDetailsModal.tsx +++ b/src/components/PlateDetailsModal.tsx @@ -1,7 +1,7 @@ import React from 'react'; -import { SearchData, CovariateColorInfo, PlateQualityScore } from '../utils/types'; +import { SearchData, CovariateColorInfo, PlateQualityScore, NaPolicy, DEFAULT_NA_POLICY } from '../utils/types'; import { QUALITY_DISPLAY_CONFIG } from '../utils/configs'; -import { getCovariateKey, getQualityColor, getCompactQualityLevel, formatScore } from '../utils/utils'; +import { getCovariateKey, getQualityColor, getCompactQualityLevel, formatScore, effectiveDisplayValue } from '../utils/utils'; interface PlateDetailsModalProps { show: boolean; @@ -20,6 +20,7 @@ interface PlateDetailsModalProps { plateQuality?: PlateQualityScore; randomizedPlates?: (SearchData | undefined)[][][]; numPlates?: number; + naPolicy?: NaPolicy; } const PlateDetailsModal: React.FC = ({ @@ -39,6 +40,7 @@ const PlateDetailsModal: React.FC = ({ plateQuality, randomizedPlates, numPlates = 1, + naPolicy = DEFAULT_NA_POLICY, }) => { @@ -297,8 +299,9 @@ const PlateDetailsModal: React.FC = ({ {selectedCovariates.map((cov) => { // Read each covariate from the sample's metadata, // so a value containing '|' stays in its own row. + // Per-group rollup: a folded group shows N/A, a kept-distinct blank shows blank. const representative = representativeByKey.get(combination); - const value = representative ? (representative.metadata[cov] || 'N/A') : 'N/A'; + const value = effectiveDisplayValue(representative?.metadata[cov] ?? '', naPolicy); return (
{cov}: {value} diff --git a/src/components/PlatesGrid.tsx b/src/components/PlatesGrid.tsx index a94fa5b..2ca8f6b 100644 --- a/src/components/PlatesGrid.tsx +++ b/src/components/PlatesGrid.tsx @@ -30,6 +30,7 @@ interface PlatesGridProps { onReRandomizePlate?: (plateIndex: number) => void; qualityMetrics?: QualityMetrics; subjectColumn?: string; + qcColumn?: string; } const PlatesGrid: React.FC = ({ @@ -46,6 +47,7 @@ const PlatesGrid: React.FC = ({ onReRandomizePlate, qualityMetrics, subjectColumn, + qcColumn, }) => { if (randomizedPlates.length === 0) return null; @@ -68,6 +70,7 @@ const PlatesGrid: React.FC = ({ onReRandomizePlate={onReRandomizePlate} plateQuality={qualityMetrics?.plateDiversity.plateScores.find(score => score.plateIndex === plateIndex)} subjectColumn={subjectColumn} + qcColumn={qcColumn} numPlates={randomizedPlates.length} />
diff --git a/src/hooks/useCovariateColors.ts b/src/hooks/useCovariateColors.ts index 1880e3c..62e2bf2 100644 --- a/src/hooks/useCovariateColors.ts +++ b/src/hooks/useCovariateColors.ts @@ -1,6 +1,6 @@ import { useState, useCallback } from 'react'; -import { SearchData, SummaryItem, CovariateColorInfo } from '../utils/types'; -import { getTextColorForBackground, sortCombinationsByCountAndName, getCovariateKey } from '../utils/utils'; +import { SearchData, SummaryItem, CovariateColorInfo, NaPolicy, DEFAULT_NA_POLICY } from '../utils/types'; +import { getTextColorForBackground, sortCombinationsByCountAndName, getCovariateKey, effectiveDisplayValue } from '../utils/utils'; import { BRIGHT_COLOR_PALETTE, QC_COLOR_PALETTE } from '../utils/configs'; /** @@ -93,7 +93,8 @@ export function useCovariateColors() { searches: SearchData[], selectedCovariates: string[], qcColumn?: string, - selectedQcValues?: string[] + selectedQcValues?: string[], + naPolicy: NaPolicy = DEFAULT_NA_POLICY ) => { if (selectedCovariates.length > 0 && searches.length > 0) { // Group searches by their covariate combinations using getCovariateKey @@ -106,7 +107,8 @@ export function useCovariateColors() { searches.forEach((search) => { const covariateValues: { [key: string]: string } = {}; selectedCovariates.forEach((covariate) => { - covariateValues[covariate] = search.metadata[covariate] || 'N/A'; + // Summary is a per-group rollup: a folded group shows N/A, a kept-distinct blank shows blank. + covariateValues[covariate] = effectiveDisplayValue(search.metadata[covariate] ?? '', naPolicy); }); // covariateKey should always be set diff --git a/src/hooks/useRandomization.ts b/src/hooks/useRandomization.ts index 91b4acb..c9bc6bb 100644 --- a/src/hooks/useRandomization.ts +++ b/src/hooks/useRandomization.ts @@ -1,5 +1,5 @@ import { useState } from 'react'; -import { SearchData, RandomizationAlgorithm, RepeatedMeasuresConfig } from '../utils/types'; +import { SearchData, RandomizationAlgorithm, RepeatedMeasuresConfig, NaPolicy, DEFAULT_NA_POLICY } from '../utils/types'; import { randomizeSearches } from '../utils/utils'; import { greedyPlaceInRow } from '../algorithms/greedySpatialPlacement'; import { groupAwareRandomization } from '../algorithms/repeatedMeasuresDistribution'; @@ -17,7 +17,8 @@ export function useRandomization() { keepEmptyInLastPlate: boolean, plateRows: number, plateColumns: number, - repeatedMeasuresConfig?: RepeatedMeasuresConfig + repeatedMeasuresConfig?: RepeatedMeasuresConfig, + naPolicy: NaPolicy = DEFAULT_NA_POLICY ) => { if (searches.length > 0 && selectedCovariates.length > 0) { const result = randomizeSearches( @@ -27,7 +28,8 @@ export function useRandomization() { keepEmptyInLastPlate, plateRows, plateColumns, - repeatedMeasuresConfig + repeatedMeasuresConfig, + naPolicy ); setRandomizedPlates(result.plates); setPlateAssignments(result.plateAssignments); @@ -44,7 +46,8 @@ export function useRandomization() { keepEmptyInLastPlate: boolean, plateRows: number, plateColumns: number, - repeatedMeasuresConfig?: RepeatedMeasuresConfig + repeatedMeasuresConfig?: RepeatedMeasuresConfig, + naPolicy: NaPolicy = DEFAULT_NA_POLICY ) => { if (searches.length > 0 && selectedCovariates.length > 0) { const result = randomizeSearches( @@ -54,7 +57,8 @@ export function useRandomization() { keepEmptyInLastPlate, plateRows, plateColumns, - repeatedMeasuresConfig + repeatedMeasuresConfig, + naPolicy ); setRandomizedPlates(result.plates); setPlateAssignments(result.plateAssignments); @@ -149,7 +153,8 @@ export function useRandomization() { keepEmptyInLastPlate: boolean, plateRows: number, plateColumns: number, - repeatedMeasuresConfig?: RepeatedMeasuresConfig + repeatedMeasuresConfig?: RepeatedMeasuresConfig, + naPolicy: NaPolicy = DEFAULT_NA_POLICY ) => { if (!plateAssignments) return false; @@ -172,7 +177,8 @@ export function useRandomization() { repeatedMeasuresConfig, false, // Don't keep empty in last plate for single-plate re-randomization plateRows, - plateColumns + plateColumns, + naPolicy ); // The result should produce exactly 1 plate since we're passing just this plate's samples diff --git a/src/tests/ConfigurationForm.test.tsx b/src/tests/ConfigurationForm.test.tsx new file mode 100644 index 0000000..0938723 --- /dev/null +++ b/src/tests/ConfigurationForm.test.tsx @@ -0,0 +1,124 @@ +/** + * Component tests for the "N/A values" checklist in ConfigurationForm. + * + * The section appears only when a column mixes two or more N/A-type spellings. It lists the union + * of spellings (blank shown as "(blank)"), all checked by default, with the literal N/A checkbox + * checked and disabled. Unchecking a value calls onNaPolicyToggle with that value's token. + */ + +import React from 'react'; +import { render, screen, fireEvent } from '@testing-library/react'; +import ConfigurationForm from '../components/ConfigurationForm'; +import { detectNaTypeValues } from '../utils/utils'; +import { SearchData, NaPolicy, DEFAULT_NA_POLICY } from '../utils/types'; + +const mkSample = (dose: string): SearchData => ({ name: 's', metadata: { Dose: dose } }); + +const noop = () => {}; + +function renderForm(overrides: { + samples: SearchData[]; + naPolicy: NaPolicy; + onNaPolicyToggle?: (token: string) => void; +}) { + const { samples, naPolicy, onNaPolicyToggle = noop } = overrides; + const naDetection = detectNaTypeValues(samples, ['Dose']); + render( + + ); + return naDetection; +} + +describe('ConfigurationForm N/A values checklist', () => { + it('renders one entry per distinct spelling across the column, with blank labeled (blank)', () => { + renderForm({ + samples: [mkSample('na'), mkSample('NA'), mkSample('')], + naPolicy: { foldBlank: true, foldSpellings: ['na', 'NA'] }, + }); + expect(screen.getByText('N/A values:')).toBeInTheDocument(); + expect(screen.getByLabelText('Treat na as N/A')).toBeInTheDocument(); + expect(screen.getByLabelText('Treat NA as N/A')).toBeInTheDocument(); + expect(screen.getByLabelText('Treat (blank) as N/A')).toBeInTheDocument(); + }); + + it('checks the literal N/A entry and disables it', () => { + renderForm({ + samples: [mkSample('na'), mkSample('N/A')], + naPolicy: { foldBlank: false, foldSpellings: ['na'] }, + }); + const literal = screen.getByLabelText('Treat N/A as N/A') as HTMLInputElement; + expect(literal.checked).toBe(true); + expect(literal.disabled).toBe(true); + }); + + it('reflects the policy: a folded spelling is checked, an unfolded one is unchecked', () => { + renderForm({ + samples: [mkSample('na'), mkSample('NA')], + naPolicy: { foldBlank: false, foldSpellings: ['na'] }, + }); + expect((screen.getByLabelText('Treat na as N/A') as HTMLInputElement).checked).toBe(true); + expect((screen.getByLabelText('Treat NA as N/A') as HTMLInputElement).checked).toBe(false); + }); + + it('calls onNaPolicyToggle with the exact spelling when a box is clicked', () => { + const toggled: string[] = []; + renderForm({ + samples: [mkSample('na'), mkSample('NA')], + naPolicy: { foldBlank: false, foldSpellings: ['na', 'NA'] }, + onNaPolicyToggle: t => toggled.push(t), + }); + fireEvent.click(screen.getByLabelText('Treat NA as N/A')); + expect(toggled).toEqual(['NA']); + }); + + it('uses the empty-string token when the blank entry is clicked', () => { + const toggled: string[] = []; + renderForm({ + samples: [mkSample('na'), mkSample('')], + naPolicy: { foldBlank: true, foldSpellings: ['na'] }, + onNaPolicyToggle: t => toggled.push(t), + }); + fireEvent.click(screen.getByLabelText('Treat (blank) as N/A')); + expect(toggled).toEqual(['']); + }); + + it('does not render the section when no column mixes spellings', () => { + renderForm({ + samples: [mkSample('na'), mkSample('na'), mkSample('108')], + naPolicy: DEFAULT_NA_POLICY, + }); + expect(screen.queryByText('N/A values:')).not.toBeInTheDocument(); + }); +}); diff --git a/src/tests/PlateDetailsModal.test.tsx b/src/tests/PlateDetailsModal.test.tsx index 4114470..b27a91f 100644 --- a/src/tests/PlateDetailsModal.test.tsx +++ b/src/tests/PlateDetailsModal.test.tsx @@ -88,4 +88,42 @@ describe('PlateDetailsModal covariate decode', () => { expectCovariateRow('Dose', '5'); expectCovariateRow('Site', 'S2'); }); + + it('renders a folded spelling as N/A when the policy folds it', () => { + const naPolicy = { foldBlank: false, foldSpellings: ['na'] }; + const sample: SearchData = { name: 'na-sample', metadata: { Treatment: 'Ctrl', Dose: 'na', Site: 'S1' } }; + sample.covariateKey = buildCovariateKey(sample, { selectedCovariates: COVS, naPolicy }); + + render( + + ); + + // The folded Dose group shows the canonical N/A label. + expectCovariateRow('Dose', 'N/A'); + expectCovariateRow('Treatment', 'Ctrl'); + }); + + it('renders a genuinely-missing value as blank (not N/A) under the default policy', () => { + const sample = makeSample('blank-sample', { Treatment: 'Ctrl', Dose: '', Site: 'S1' }); + + render( + + ); + + // A missing group renders blank, so the label 'N/A' must not appear anywhere. + expect(screen.queryByText('N/A')).not.toBeInTheDocument(); + expectCovariateRow('Treatment', 'Ctrl'); + expectCovariateRow('Site', 'S1'); + }); }); diff --git a/src/tests/covariateKey.properties.test.ts b/src/tests/covariateKey.properties.test.ts index b781135..703089a 100644 --- a/src/tests/covariateKey.properties.test.ts +++ b/src/tests/covariateKey.properties.test.ts @@ -1,39 +1,55 @@ /** - * Property-based tests for buildCovariateKey (escape-encoded covariate key). + * Property-based tests for buildCovariateKey (escape-encoded covariate key) under the + * na-value-handling NaPolicy. * - * Complements the hand-picked cases in `utils.test.ts` by asserting the two - * correctness properties from the covariate-key-fragility design over a large - * random sample space, with an alphabet that includes the delimiter `|`, the - * escape character `\`, empty, `na`, and `N/A`. + * Complements the hand-picked cases in `utils.test.ts` by asserting the correctness + * properties over a large random sample space, with an alphabet that includes the + * delimiter `|`, the escape character `\`, blank, whitespace, and the sentinel spellings + * `na`, `NA`, `n/a`, `N/A`. * - * Property 1 (Injectivity): two samples with different value tuples produce - * different keys; two samples with the same effective tuple produce the same - * key (the `|| 'N/A'` fallback means empty and a literal `N/A` collide, which - * is the accepted B-min residual). - * Property 2 (Clean-data preservation): for tuples with no `|` and no `\`, the - * key is byte-identical to the legacy `values.map(v => v || 'N/A').join('|')`. - * Property 3 (QC prefix injectivity): with a QC column selected, a QC sample's - * key (QC value prepended) and a non-QC sample's key are equal iff their + * The key encodes an EFFECTIVE identity per value (see effectiveValue in utils): + * - blank folds to canonical N/A only when the policy sets foldBlank; otherwise it is a + * distinct "missing" identity (the lone-backslash marker), + * - the literal `N/A` always folds, + * - any other spelling folds only when the policy lists it, else it stays its own value. + * + * Property 1 (Injectivity): two samples whose effective identity tuples differ produce + * different keys; equal effective tuples produce equal keys. Checked across several + * policies (fold nothing, fold blank, fold spellings, fold all). + * Property 2 (Clean-data preservation): for tuples with NO N/A-type value, the key is + * byte-identical to the pre-feature escape-join, so grouping is unchanged (Requirement 6.1). + * Property 3 (QC prefix injectivity): with a QC column selected, a QC sample's key (QC value + * prepended, also run through the policy) and a non-QC sample's key are equal iff their * effective identity matches, and a QC key never collides with a non-QC key. - * Covers a QC value that itself contains the delimiter `|`. */ import * as fc from 'fast-check'; import { buildCovariateKey } from '../utils/utils'; -import { SearchData, CovariateConfig } from '../utils/types'; +import { SearchData, CovariateConfig, NaPolicy, DEFAULT_NA_POLICY } from '../utils/types'; + +// Alphabet exercising every hazard: delimiter, escape char, both together, blank, +// whitespace-only, and the sentinel spellings in several cases. +const VALUE = fc.constantFrom( + 'a', 'Drug', 'Hi|10', '|', '\\', '\\|', '|\\', 'na', 'NA', 'n/a', 'N/A', '', ' ', '0' +); -// Alphabet exercising every hazard: delimiter, escape char, both together, -// empty, and the sentinel spellings. -const VALUE = fc.constantFrom('a', 'Drug', 'Hi|10', '|', '\\', '\\|', '|\\', 'na', 'N/A', '', '0'); +// Clean values: no N/A-type value (no blank, no na/n/a spelling). Delimiter and escape +// characters are still exercised, since those are handled by escaping, not the policy. +const CLEAN_VALUE = fc.constantFrom('a', 'Drug', 'Hi|10', '|', '\\', '\\|', '|\\', '108', 'FA1', '0'); -// Clean values: no delimiter and no escape character (empty and sentinels are -// still allowed, since the legacy formula also maps them through `|| 'N/A'`). -const CLEAN_VALUE = fc.constantFrom('a', 'Drug', 'Training', '108', 'FA1', 'S1', '0', '', 'N/A', 'na'); +// Policies spanning the meaningful choices. +const POLICIES: NaPolicy[] = [ + DEFAULT_NA_POLICY, // fold nothing extra: blank distinct, spellings literal + { foldBlank: true, foldSpellings: [] }, // fold blank into N/A + { foldBlank: false, foldSpellings: ['na', 'NA', 'n/a'] }, // fold spellings, keep blank distinct + { foldBlank: true, foldSpellings: ['na', 'NA', 'n/a'] }, // fold everything +]; const tupleArb = (n: number) => fc.array(VALUE, { minLength: n, maxLength: n }); -const configFor = (n: number): CovariateConfig => ({ +const configFor = (n: number, naPolicy: NaPolicy): CovariateConfig => ({ selectedCovariates: Array.from({ length: n }, (_, i) => `c${i}`), + naPolicy, }); const sampleFor = (values: string[]): SearchData => ({ @@ -41,20 +57,36 @@ const sampleFor = (values: string[]): SearchData => ({ metadata: Object.fromEntries(values.map((v, i) => [`c${i}`, v])), }); -// The effective tuple is what the key builder actually encodes: an empty cell -// falls back to the literal 'N/A'. -const effective = (values: string[]): string[] => values.map(v => v || 'N/A'); +// A canonical identity token for a raw value under a policy. Distinct tokens correspond to +// distinct key parts and vice versa. Mirrors effectiveValue: a real value carries a "V:" tag +// so it can never coincide with the NA or MISSING tokens. +function identity(raw: string, policy: NaPolicy): string { + const isBlank = raw.trim().length === 0; + const lower = raw.toLowerCase(); + const isNa = isBlank || lower === 'na' || lower === 'n/a'; + if (!isNa) return `V:${raw}`; + if (isBlank) return policy.foldBlank ? 'NA' : 'MISSING'; + if (raw === 'N/A') return 'NA'; + return policy.foldSpellings.includes(raw) ? 'NA' : `V:${raw}`; +} + +const identityTuple = (values: string[], policy: NaPolicy): string[] => + values.map(v => identity(v, policy)); -// QC-prefix arbitraries. The selected QC values include one that carries the -// delimiter, and the value space also draws non-selected values, empty, and the -// sentinels, so a sample is QC only when its QC value is truthy and selected. +// The pre-feature escape used by covariate-key-fragility, replicated so clean-data keys can be +// checked to be byte-identical to that behavior. +const escape = (v: string) => v.replace(/\\/g, '\\\\').replace(/\|/g, '\\|'); + +// QC-prefix arbitraries. The selected QC values include one with the delimiter and one N/A-type +// spelling, and the value space also draws non-selected values, blank, and the sentinels. const QC_VALUES = ['Ref', 'Batch|QC', 'na']; const QC_VALUE = fc.constantFrom('Ref', 'Batch|QC', 'na', 'X', '', 'N/A', '\\', 'Ref|x'); -const qcConfigFor = (n: number): CovariateConfig => ({ +const qcConfigFor = (n: number, naPolicy: NaPolicy): CovariateConfig => ({ selectedCovariates: Array.from({ length: n }, (_, i) => `c${i}`), qcColumn: 'qc', selectedQcValues: QC_VALUES, + naPolicy, }); const qcSampleFor = (qcValue: string, values: string[]): SearchData => ({ @@ -62,29 +94,32 @@ const qcSampleFor = (qcValue: string, values: string[]): SearchData => ({ metadata: { qc: qcValue, ...Object.fromEntries(values.map((v, i) => [`c${i}`, v])) }, }); -// The effective identity the key encodes: the base tuple, with the QC value -// prepended as an extra leading segment only when the sample is QC. A QC identity -// has one more segment than a non-QC one, so the two can never coincide. -const qcIdentity = (qcValue: string, values: string[]): string[] => { - const eff = effective(values); +// The effective identity the key encodes: the base identity tuple, with the QC value's identity +// prepended as an extra leading segment only when the sample is QC. A QC identity has one more +// segment than a non-QC one, so the two can never coincide. +const qcIdentity = (qcValue: string, values: string[], policy: NaPolicy): string[] => { + const base = identityTuple(values, policy); const isQC = !!qcValue && QC_VALUES.includes(qcValue); - return isQC ? [qcValue, ...eff] : eff; + return isQC ? [identity(qcValue, policy), ...base] : base; }; -describe('buildCovariateKey property tests', () => { - it('maps distinct value tuples to distinct keys, equal tuples to equal keys (Property 1)', () => { +describe('buildCovariateKey property tests (NaPolicy)', () => { + it('maps distinct effective tuples to distinct keys, equal ones to equal keys (Property 1)', () => { fc.assert( fc.property( - fc.integer({ min: 1, max: 5 }).chain(n => fc.tuple(tupleArb(n), tupleArb(n))), - ([t1, t2]) => { - const config = configFor(t1.length); + fc.integer({ min: 1, max: 5 }).chain(n => + fc.tuple(tupleArb(n), tupleArb(n), fc.constantFrom(...POLICIES)) + ), + ([t1, t2, policy]) => { + const config = configFor(t1.length, policy); const k1 = buildCovariateKey(sampleFor(t1), config); const k2 = buildCovariateKey(sampleFor(t2), config); - const sameTuple = JSON.stringify(effective(t1)) === JSON.stringify(effective(t2)); - return sameTuple ? k1 === k2 : k1 !== k2; + const same = + JSON.stringify(identityTuple(t1, policy)) === JSON.stringify(identityTuple(t2, policy)); + return same ? k1 === k2 : k1 !== k2; } ), - { numRuns: 3000 } + { numRuns: 4000 } ); }); @@ -94,33 +129,36 @@ describe('buildCovariateKey property tests', () => { fc.integer({ min: 1, max: 4 }).chain(n => fc.tuple( fc.tuple(QC_VALUE, tupleArb(n)), - fc.tuple(QC_VALUE, tupleArb(n)) + fc.tuple(QC_VALUE, tupleArb(n)), + fc.constantFrom(...POLICIES) ) ), - ([[q1, t1], [q2, t2]]) => { - const config = qcConfigFor(t1.length); + ([[q1, t1], [q2, t2], policy]) => { + const config = qcConfigFor(t1.length, policy); const k1 = buildCovariateKey(qcSampleFor(q1, t1), config); const k2 = buildCovariateKey(qcSampleFor(q2, t2), config); - const sameIdentity = - JSON.stringify(qcIdentity(q1, t1)) === JSON.stringify(qcIdentity(q2, t2)); - return sameIdentity ? k1 === k2 : k1 !== k2; + const same = + JSON.stringify(qcIdentity(q1, t1, policy)) === JSON.stringify(qcIdentity(q2, t2, policy)); + return same ? k1 === k2 : k1 !== k2; } ), - { numRuns: 3000 } + { numRuns: 4000 } ); }); - it('is byte-identical to the legacy pipe-join for clean data (Property 2)', () => { + it('is byte-identical to the pre-feature escape-join for clean data (Property 2)', () => { fc.assert( fc.property( - fc.array(CLEAN_VALUE, { minLength: 1, maxLength: 5 }), - (values) => { - const config = configFor(values.length); - const legacy = values.map(v => v || 'N/A').join('|'); - return buildCovariateKey(sampleFor(values), config) === legacy; + fc.array(CLEAN_VALUE, { minLength: 1, maxLength: 5 }).chain(values => + fc.tuple(fc.constant(values), fc.constantFrom(...POLICIES)) + ), + ([values, policy]) => { + const config = configFor(values.length, policy); + const expected = values.map(escape).join('|'); + return buildCovariateKey(sampleFor(values), config) === expected; } ), - { numRuns: 2000 } + { numRuns: 3000 } ); }); }); diff --git a/src/tests/distributeToBlocks.test.ts b/src/tests/distributeToBlocks.test.ts index a67eb90..d06e861 100644 --- a/src/tests/distributeToBlocks.test.ts +++ b/src/tests/distributeToBlocks.test.ts @@ -108,8 +108,10 @@ describe('distributeToBlocks - Plate-Level Distribution', () => { // Verify proportional distribution const groupCounts = countSamplesPerGroup(result, selectedCovariates); - // Control group (8 samples) should be distributed 4 per plate - const controlCounts = groupCounts.get('N/A|N/A|Control'); + // Control group (8 samples) should be distributed 4 per plate. The blank Gender and Plate + // cells resolve to the missing marker under the default policy, so the rebuilt key is + // "\|\|Control" (not "N/A|N/A|Control" as before na-value-handling). + const controlCounts = groupCounts.get('\\|\\|Control'); expect(controlCounts?.size).toBe(2); expect(controlCounts!.get(0)).toBe(4); expect(controlCounts!.get(1)).toBe(4); diff --git a/src/tests/e2e/na-value-handling.spec.ts b/src/tests/e2e/na-value-handling.spec.ts new file mode 100644 index 0000000..944d9b9 --- /dev/null +++ b/src/tests/e2e/na-value-handling.spec.ts @@ -0,0 +1,70 @@ +import { test, expect } from '@playwright/test'; +import path from 'path'; +import { openExportMenu } from './helpers'; + +/** + * N/A value handling: save/load must preserve the N/A policy. + * + * Regression for a load-path bug: the covariate keys rebuilt on load ignored the saved N/A + * policy, so na/NA/n/a split back out into their own groups (14 instead of 11) and the stored + * covariate colors no longer matched (several cells rendered gray). The demo file has a Dose + * column that mixes na/NA/n/a/N/A and blank, so the "N/A values" checklist appears. + */ + +const DEMO = path.join(__dirname, '../../../test-data/na-value-handling-demo.csv'); + +// With na/NA/n/a folded into N/A and blank kept distinct, this data forms 11 covariate groups. +const FOLDED_GROUPS = 11; + +async function configureDemo(page: import('@playwright/test').Page): Promise { + await page.locator('#file-upload').setInputFiles(DEMO); + await page.locator('#qcColumn').selectOption('QC'); + await page.getByRole('checkbox', { name: 'BatchQC' }).check(); + await page.getByRole('checkbox', { name: 'BatchRef' }).check(); + await page.locator('#covariates').selectOption(['Treatment', 'Dose', 'Region']); +} + +test.describe('N/A values save/load', () => { + test.beforeEach(async ({ page }) => { + page.on('dialog', dialog => dialog.accept()); + await page.goto('http://localhost:3000'); + await expect(page.getByRole('heading', { name: 'Octopus' })).toBeVisible(); + }); + + test('a saved layout reloads with the same N/A grouping (folded spellings, distinct blank)', async ({ page }, testInfo) => { + await configureDemo(page); + + // Keep genuinely-blank cells distinct: uncheck (blank). na/NA/n/a still fold into N/A. + await page.getByRole('checkbox', { name: 'Treat (blank) as N/A' }).uncheck(); + await page.getByRole('button', { name: 'Generate Randomized Plates' }).click(); + await page.waitForTimeout(1000); + + // The folded policy yields 11 groups before saving. + await expect( + page.getByRole('button', { name: new RegExp(`Covariate Summary \\(${FOLDED_GROUPS} combinations\\)`) }) + ).toBeVisible(); + + // Save the layout. + await openExportMenu(page); + const downloadPromise = page.waitForEvent('download'); + await page.getByRole('menuitem', { name: 'Layout', exact: true }).click(); + const download = await downloadPromise; + const savedPath = testInfo.outputPath(download.suggestedFilename()); + await download.saveAs(savedPath); + + // Reload the page to clear all state, then load the saved layout. + await page.goto('http://localhost:3000'); + await expect(page.getByRole('heading', { name: 'Octopus' })).toBeVisible(); + await page.locator('#layout-upload').setInputFiles(savedPath); + await expect(page.getByText('Layout file')).toBeVisible(); + + // The reloaded layout must reproduce the same 11 groups. With the load path ignoring the + // saved policy, na/NA/n/a split back out to 14 groups and colors go gray. + await expect( + page.getByRole('button', { name: new RegExp(`Covariate Summary \\(${FOLDED_GROUPS} combinations\\)`) }) + ).toBeVisible(); + + // The saved policy is restored, so (blank) comes back unchecked. + await expect(page.getByRole('checkbox', { name: 'Treat (blank) as N/A' })).not.toBeChecked(); + }); +}); diff --git a/src/tests/excelExport.test.ts b/src/tests/excelExport.test.ts index b19b650..9777d2a 100644 --- a/src/tests/excelExport.test.ts +++ b/src/tests/excelExport.test.ts @@ -3,14 +3,16 @@ * * The legend must read covariate values from structured metadata (not by * splitting the key), classify QC by the sample's real QC status (not by the - * key's part count), and render a missing covariate value as "N/A" (matching the - * Plate Details modal), while a present "na" is rendered literally. + * key's part count), and render each group's value through the N/A policy + * (matching the Plate Details modal): under the default policy a genuinely + * missing value renders blank and a present "na" is literal, while a policy that + * folds blank or a spelling renders those groups as "N/A". */ import ExcelJS from 'exceljs'; import { buildLayoutWorkbook } from '../utils/excelExport'; import { buildProcessedSearches } from '../utils/utils'; -import { SearchData, CovariateColorInfo, CovariateConfig } from '../utils/types'; +import { SearchData, CovariateColorInfo, CovariateConfig, NaPolicy, DEFAULT_NA_POLICY } from '../utils/types'; const TREATMENT_COVARIATES = ['Treatment', 'Dose']; const QC_COLUMN = 'QC'; @@ -25,11 +27,12 @@ const color = (): CovariateColorInfo => ({ const mk = (metadata: { [k: string]: string }): SearchData => ({ name: 'sample', metadata }); -function buildLegendSheet(samples: SearchData[]): ExcelJS.Worksheet { +function buildLegendSheet(samples: SearchData[], naPolicy: NaPolicy = DEFAULT_NA_POLICY): ExcelJS.Worksheet { const config: CovariateConfig = { selectedCovariates: TREATMENT_COVARIATES, qcColumn: QC_COLUMN, selectedQcValues: SELECTED_QC_VALUES, + naPolicy, }; // Sets covariateKey (escape-encoded) and isQC on each sample, as the app does. buildProcessedSearches(samples, config); @@ -47,6 +50,7 @@ function buildLegendSheet(samples: SearchData[]): ExcelJS.Worksheet { numRows: 1, numColumns: samples.length, qcColumn: QC_COLUMN, + naPolicy, }); return workbook.getWorksheet('Legend')!; } @@ -86,7 +90,7 @@ function readLegendRows(sheet: ExcelJS.Worksheet): Array<{ [header: string]: str } describe('Excel legend structured decode and flag-based QC', () => { - it('reads values from metadata, classifies QC by status, and renders missing as N/A', () => { + it('reads values from metadata, classifies QC by status, and renders a missing value as blank by default', () => { const withPipe = mk({ Treatment: 'Drug|Hi', Dose: '10', QC: '' }); // non-QC, value has delimiter const qcNa = mk({ Treatment: 'na', Dose: '5', QC: 'Ref' }); // real QC, present 'na' value const missing = mk({ Treatment: 'Ctrl', Dose: '', QC: '' }); // non-QC, genuinely missing Dose @@ -100,20 +104,32 @@ describe('Excel legend structured decode and flag-based QC', () => { expect(a!['Treatment']).toBe('Drug|Hi'); expect(a!['QC']).toBe(''); - // A real QC sample is placed under the QC column; a present 'na' is literal. + // A real QC sample is placed under the QC column; a present 'na' is literal under the default policy. const b = rows.find(r => r['Dose'] === '5'); expect(b).toBeDefined(); expect(b!['QC']).toBe('Ref'); expect(b!['Treatment']).toBe('na'); - // A genuinely missing covariate value renders as 'N/A', matching the modal. - // The QC column stays blank for a non-QC sample. + // Under the default policy a genuinely missing value is a distinct group that renders blank + // (not 'N/A', which now means a folded "not applicable" group). The QC column stays blank. const c = rows.find(r => r['Treatment'] === 'Ctrl'); expect(c).toBeDefined(); - expect(c!['Dose']).toBe('N/A'); + expect(c!['Dose']).toBe(''); expect(c!['QC']).toBe(''); }); + it('renders a folded spelling and a folded blank as N/A when the policy folds them', () => { + const na = mk({ Treatment: 'Ctrl', Dose: 'na', QC: '' }); // 'na' folded into N/A + const blank = mk({ Treatment: 'Ctrl', Dose: '', QC: '' }); // blank folded into N/A + // Fold both 'na' and blank: their Dose groups collapse to the same N/A group. + const sheet = buildLegendSheet([na, blank], { foldBlank: true, foldSpellings: ['na'] }); + const rows = readLegendRows(sheet); + + const doseCells = rows.filter(r => r['Treatment'] === 'Ctrl').map(r => r['Dose']); + // Both fold to one N/A group, so there is a single 'Ctrl' legend row with Dose 'N/A'. + expect(doseCells).toEqual(['N/A']); + }); + it('keeps a QC value containing "|" intact and under the QC column', () => { // A QC value with the key delimiter must not shift or drop legend columns: // the QC cell shows the literal value and the treatment columns stay aligned. diff --git a/src/tests/layoutIO.test.ts b/src/tests/layoutIO.test.ts index 1721d35..a1b6bf3 100644 --- a/src/tests/layoutIO.test.ts +++ b/src/tests/layoutIO.test.ts @@ -57,6 +57,7 @@ const SETTINGS: LayoutSettings = { subjectColumn: '', groupingConstraint: 'none', metadataColumns: ['Treatment', 'Dose'], + naPolicy: { foldBlank: false, foldSpellings: [] }, }; // Every color key matches a covariate combination the samples actually produce, so the @@ -121,10 +122,11 @@ describe('serializeLayout', () => { expect(parseLayout(fullFile()).appVersion).toBeNull(); }); - it('omits covariateColors when the color map is empty', () => { - const text = serializeLayout({ searches: SEARCHES, randomizedPlates: PLATES, settings: SETTINGS, covariateColors: {} }); - expect('covariateColors' in (JSON.parse(text) as object)).toBe(false); - expect(parseLayout(text).covariateColors).toBeNull(); + it('refuses to save a layout with no covariate colors', () => { + // A layout must record a color for every covariate group, so saving without colors is refused. + expect(() => + serializeLayout({ searches: SEARCHES, randomizedPlates: PLATES, settings: SETTINGS, covariateColors: {} }) + ).toThrow(/covariate colors/i); }); it('throws when a sample is not on any plate instead of emitting an invalid plate', () => { @@ -215,6 +217,7 @@ describe('settings round-trip (one field varied at a time)', () => { { name: 'reversed covariate order', settings: { ...SETTINGS, selectedCovariates: ['Dose', 'Treatment'] } }, { name: 'subject column + same-row grouping', settings: { ...SETTINGS, subjectColumn: 'Dose', groupingConstraint: 'same-row' } }, { name: 'subject column + same-plate grouping', settings: { ...SETTINGS, subjectColumn: 'Dose', groupingConstraint: 'same-plate' } }, + { name: 'naPolicy folds blank and a spelling', settings: { ...SETTINGS, naPolicy: { foldBlank: true, foldSpellings: ['na'] } } }, ]; it.each(variants)('preserves: $name', ({ settings }) => { @@ -229,6 +232,30 @@ describe('settings round-trip (one field varied at a time)', () => { expect(parsed.structuralErrors).toEqual([]); expect(parsed.settings).toEqual(settings); }); + + it('validates covariate-color keys derived under the saved naPolicy', () => { + // Samples carry an N/A-type spelling. A layout saved with a policy that folds 'na' into N/A + // must store the folded color key 'N/A' and reload cleanly, since validateLayout re-derives + // keys through the saved policy. With the default policy the same file would instead expect + // the literal 'na' key, so this proves the policy actually drives the re-derivation. + const naSamples = [makeSample('N1', 'na', '0'), makeSample('N2', 'Drug', '0')]; + const naPlates: (SearchData | undefined)[][][] = [[[naSamples[0], naSamples[1], undefined], [undefined, undefined, undefined]]]; + const foldSettings: LayoutSettings = { + ...SETTINGS, + qcColumn: '', + selectedQcValues: [], + naPolicy: { foldBlank: false, foldSpellings: ['na'] }, + }; + // Under this policy 'na' folds to N/A, so the group key for N1 is 'N/A|0'. + const foldColors: CovariateColorMap = { + 'N/A|0': { color: '#111111', useOutline: false, useStripes: false, textColor: '#fff' }, + 'Drug|0': { color: '#222222', useOutline: false, useStripes: false, textColor: '#fff' }, + }; + const text = serializeLayout({ searches: naSamples, randomizedPlates: naPlates, settings: foldSettings, covariateColors: foldColors }); + const parsed = parseLayout(text); + expect(parsed.settings!.naPolicy).toEqual({ foldBlank: false, foldSpellings: ['na'] }); + expect(validateLayout(parsed)).toEqual([]); + }); }); describe('color and style round-trip', () => { @@ -293,11 +320,15 @@ describe('per-cell placement', () => { ], ]; const settings: LayoutSettings = { ...SETTINGS, qcColumn: '', selectedQcValues: [] }; + // The one sample forms the group "Drug, high|10 mg"; colors are required, so give it one. + const trickyColors: CovariateColorMap = { + 'Drug, high|10 mg': { color: '#111111', useOutline: false, useStripes: false, textColor: '#fff' }, + }; const text = serializeLayout({ searches: [tricky], randomizedPlates: trickyPlates, settings, - covariateColors: {}, + covariateColors: trickyColors, }); const { samples } = buildPlatesFromRows(parseLayout(text).rows, settings); expect(samples).toHaveLength(1); @@ -418,6 +449,9 @@ describe('parseLayout strict structural validation', () => { ['unknown groupingConstraint', (d: any) => { d.settings.groupingConstraint = 'same-galaxy'; }], ['covariates not an array', (d: any) => { d.settings.covariates = 'Treatment'; }], ['idColumn not a string', (d: any) => { d.settings.idColumn = 5; }], + ['naPolicy missing', (d: any) => { delete d.settings.naPolicy; }], + ['naPolicy.foldBlank not a boolean', (d: any) => { d.settings.naPolicy.foldBlank = 'yes'; }], + ['naPolicy.foldSpellings not a string array', (d: any) => { d.settings.naPolicy.foldSpellings = 'na'; }], ])('flags bad settings: %s', (_label, mutate) => { const d = doc(); mutate(d); @@ -429,6 +463,8 @@ describe('parseLayout strict structural validation', () => { it.each([ ['bad color hex', (d: any) => { d.covariateColors['Drug|0'].color = 'red'; }], ['bad fill token', (d: any) => { d.covariateColors['Drug|0'].fill = 'zebra'; }], + ['covariateColors missing', (d: any) => { delete d.covariateColors; }], + ['covariateColors empty', (d: any) => { d.covariateColors = {}; }], ])('flags bad covariateColors: %s', (_label, mutate) => { const d = doc(); mutate(d); diff --git a/src/tests/layoutRoundTrip.properties.test.ts b/src/tests/layoutRoundTrip.properties.test.ts index 483e7ec..6b1d103 100644 --- a/src/tests/layoutRoundTrip.properties.test.ts +++ b/src/tests/layoutRoundTrip.properties.test.ts @@ -153,6 +153,7 @@ function buildScenario(input: ScenarioInput): BuiltScenario { subjectColumn: base.subjectColumn, groupingConstraint: base.groupingConstraint, metadataColumns: base.metadataColumns, + naPolicy: { foldBlank: false, foldSpellings: [] }, }; const colors: CovariateColorMap = {}; diff --git a/src/tests/naDetection.test.ts b/src/tests/naDetection.test.ts new file mode 100644 index 0000000..7cf0677 --- /dev/null +++ b/src/tests/naDetection.test.ts @@ -0,0 +1,88 @@ +import { detectNaTypeValues } from '../utils/utils'; +import { SearchData } from '../utils/types'; + +const mkSample = (metadata: { [key: string]: string }): SearchData => ({ name: 's', metadata }); + +describe('detectNaTypeValues', () => { + it('returns the exact distinct spellings per column', () => { + const samples = [ + mkSample({ Dose: 'na', Site: 'FA1' }), + mkSample({ Dose: 'NA', Site: 'FA2' }), + mkSample({ Dose: '108', Site: 'na' }), + ]; + const result = detectNaTypeValues(samples, ['Dose', 'Site']); + expect(result.byColumn.get('Dose')).toEqual(new Set(['na', 'NA'])); + expect(result.byColumn.get('Site')).toEqual(new Set(['na'])); + expect(result.spellings).toEqual(new Set(['na', 'NA'])); + }); + + it('flags a column that mixes two or more distinct N/A-type spellings', () => { + const samples = [ + mkSample({ Dose: 'na' }), + mkSample({ Dose: 'NA' }), + ]; + expect(detectNaTypeValues(samples, ['Dose']).hasAmbiguousColumn).toBe(true); + }); + + it('does not flag a column with a single N/A-type spelling', () => { + const samples = [ + mkSample({ Dose: 'na' }), + mkSample({ Dose: 'na' }), + mkSample({ Dose: '108' }), + ]; + const result = detectNaTypeValues(samples, ['Dose']); + expect(result.byColumn.get('Dose')).toEqual(new Set(['na'])); + expect(result.hasAmbiguousColumn).toBe(false); + }); + + it('treats blank and whitespace-only cells as the one blank token, and mixing blank with a spelling is ambiguous', () => { + const samples = [ + mkSample({ Dose: '' }), + mkSample({ Dose: ' ' }), + mkSample({ Dose: 'na' }), + ]; + const result = detectNaTypeValues(samples, ['Dose']); + // '' and ' ' collapse to the single blank token '', so the column holds { blank, na }. + expect(result.byColumn.get('Dose')).toEqual(new Set(['', 'na'])); + expect(result.hasAmbiguousColumn).toBe(true); + }); + + it('does not flag a column whose only N/A-type value is blank', () => { + const samples = [ + mkSample({ Dose: '' }), + mkSample({ Dose: ' ' }), + mkSample({ Dose: '108' }), + ]; + const result = detectNaTypeValues(samples, ['Dose']); + expect(result.byColumn.get('Dose')).toEqual(new Set([''])); + expect(result.hasAmbiguousColumn).toBe(false); + }); + + it('omits columns that hold no N/A-type value', () => { + const samples = [mkSample({ Treatment: 'Drug', Dose: 'na' })]; + const result = detectNaTypeValues(samples, ['Treatment', 'Dose']); + expect(result.byColumn.has('Treatment')).toBe(false); + expect(result.byColumn.has('Dose')).toBe(true); + }); + + it('does not treat None, null, or - as N/A-type', () => { + const samples = [ + mkSample({ Dose: 'None' }), + mkSample({ Dose: 'null' }), + mkSample({ Dose: '-' }), + ]; + const result = detectNaTypeValues(samples, ['Dose']); + expect(result.byColumn.has('Dose')).toBe(false); + expect(result.hasAmbiguousColumn).toBe(false); + }); + + it('reports ambiguity across the whole set even if no single sample shows both spellings', () => { + const samples = [ + mkSample({ Focus: 'n/a' }), + mkSample({ Focus: 'N/A' }), + ]; + const result = detectNaTypeValues(samples, ['Focus']); + expect(result.byColumn.get('Focus')).toEqual(new Set(['n/a', 'N/A'])); + expect(result.hasAmbiguousColumn).toBe(true); + }); +}); diff --git a/src/tests/naPolicyRoundTrip.test.ts b/src/tests/naPolicyRoundTrip.test.ts new file mode 100644 index 0000000..7e20798 --- /dev/null +++ b/src/tests/naPolicyRoundTrip.test.ts @@ -0,0 +1,198 @@ +/** + * Exhaustive N/A-policy tests: every combination of the "N/A values" checklist choices. + * + * The checklist has four toggles: (blank), na, NA, n/a. The literal N/A is always folded. This + * suite permutes all 2^4 = 16 combinations and, for each policy: + * 1. asserts the covariate groups the key builder produces match an INDEPENDENT reference + * implementation (so a bug in effectiveValue cannot silently agree with itself), and + * 2. asserts a Save -> parse -> validate -> Load round-trip reproduces the exact same groups, + * with validateLayout accepting the file (colors match the groups one-to-one). + * + * A negative test confirms validateLayout now rejects a file whose colors do not cover every + * group (the class of drift that previously rendered uncolored cells gray instead of failing). + */ + +import { + serializeLayout, + parseLayout, + validateLayout, + buildPlatesFromRows, + LayoutSettings, + CovariateColorMap, +} from '../utils/layoutIO'; +import { buildProcessedSearches } from '../utils/utils'; +import { SearchData, NaPolicy, CovariateColorInfo } from '../utils/types'; + +const COVARIATES = ['Treatment', 'Dose']; +const QC_COLUMN = 'QC'; +const QC_VALUES = ['Ref']; +const METADATA_COLUMNS = ['Treatment', 'Dose', 'QC']; + +// Samples covering every N/A form in Dose, a second value in Treatment, plus a QC sample whose +// Dose is an N/A form (so folding must also apply to a QC-prefixed key's covariate part). +const RAW_SAMPLES: Array<{ name: string; Treatment: string; Dose: string; QC: string }> = [ + { name: 's0', Treatment: 'A', Dose: '0', QC: '' }, + { name: 's1', Treatment: 'A', Dose: 'na', QC: '' }, + { name: 's2', Treatment: 'A', Dose: 'NA', QC: '' }, + { name: 's3', Treatment: 'A', Dose: 'n/a', QC: '' }, + { name: 's4', Treatment: 'A', Dose: 'N/A', QC: '' }, + { name: 's5', Treatment: 'A', Dose: '', QC: '' }, // genuinely blank + { name: 's6', Treatment: 'B', Dose: 'na', QC: '' }, + { name: 's7', Treatment: 'B', Dose: '0', QC: '' }, + { name: 'q0', Treatment: 'A', Dose: 'na', QC: 'Ref' }, // QC sample +]; + +const makeSamples = (): SearchData[] => + RAW_SAMPLES.map(r => ({ name: r.name, metadata: { Treatment: r.Treatment, Dose: r.Dose, QC: r.QC } })); + +// --- Independent reference for the expected covariate key (mirrors the spec, not the code) --- + +const MARKER = '\\'; +function refEffective(raw: string, policy: NaPolicy): string { + const blank = raw.trim() === ''; + const lower = raw.toLowerCase(); + const isNa = blank || lower === 'na' || lower === 'n/a'; + if (!isNa) return raw; // no delimiters/escapes in this fixture, so raw === escaped + if (blank) return policy.foldBlank ? 'N/A' : MARKER; + if (raw === 'N/A') return 'N/A'; + return policy.foldSpellings.includes(raw) ? 'N/A' : raw; +} +function refKey(s: { Treatment: string; Dose: string; QC: string }, policy: NaPolicy): string { + const base = [s.Treatment, s.Dose].map(v => refEffective(v, policy)).join('|'); + const isQc = !!s.QC && QC_VALUES.includes(s.QC) && !COVARIATES.includes(QC_COLUMN); + return isQc ? `${refEffective(s.QC, policy)}|${base}` : base; +} + +// --- Enumerate all 16 policies (foldBlank x subset of {na, NA, n/a}) --- + +const TOGGLE_SPELLINGS = ['na', 'NA', 'n/a']; +function allPolicies(): NaPolicy[] { + const policies: NaPolicy[] = []; + for (const foldBlank of [false, true]) { + for (let mask = 0; mask < 1 << TOGGLE_SPELLINGS.length; mask++) { + const foldSpellings = TOGGLE_SPELLINGS.filter((_, i) => mask & (1 << i)); + policies.push({ foldBlank, foldSpellings }); + } + } + return policies; +} + +const sortedKeys = (keys: Iterable): string[] => Array.from(new Set(keys)).sort(); + +const DARK: CovariateColorInfo = { color: '#101010', useOutline: false, useStripes: false, textColor: '#fff' }; + +function colorsForGroups(groups: string[]): CovariateColorMap { + const map: CovariateColorMap = {}; + groups.forEach(g => { map[g] = { ...DARK }; }); + return map; +} + +function settingsFor(naPolicy: NaPolicy): LayoutSettings { + return { + selectedIdColumn: 'Sample ID', + selectedCovariates: COVARIATES, + qcColumn: QC_COLUMN, + selectedQcValues: QC_VALUES, + selectedAlgorithm: 'balanced', + keepEmptyInLastPlate: false, + plateRows: 3, + plateColumns: 4, + subjectColumn: '', + groupingConstraint: 'none', + metadataColumns: METADATA_COLUMNS, + naPolicy, + }; +} + +// Place the 9 samples row-major on a single 3x4 plate. +function buildPlates(samples: SearchData[]): (SearchData | undefined)[][][] { + const rows = 3, cols = 4; + const plate: (SearchData | undefined)[][] = Array.from({ length: rows }, () => + Array.from({ length: cols }, () => undefined as SearchData | undefined) + ); + samples.forEach((s, i) => { plate[Math.floor(i / cols)][i % cols] = s; }); + return [plate]; +} + +describe('N/A policy permutations: covariate groups and layout round-trip', () => { + const policies = allPolicies(); + + it('enumerates all 16 checklist combinations', () => { + expect(policies.length).toBe(16); + }); + + it.each(policies.map(p => [JSON.stringify(p), p] as const))( + 'produces the reference groups and round-trips exactly: %s', + (_label, policy) => { + // 1. Correctness: the key builder must match the independent reference, group for group. + const samples = makeSamples(); + buildProcessedSearches(samples, { + selectedCovariates: COVARIATES, + qcColumn: QC_COLUMN, + selectedQcValues: QC_VALUES, + naPolicy: policy, + }); + const producedGroups = sortedKeys(samples.map(s => s.covariateKey!)); + const referenceGroups = sortedKeys(RAW_SAMPLES.map(r => refKey(r, policy))); + expect(producedGroups).toEqual(referenceGroups); + + // 2. Round-trip: serialize with a color per group, then parse/validate/load and re-derive. + const settings = settingsFor(policy); + const covariateColors = colorsForGroups(producedGroups); + const text = serializeLayout({ + searches: samples, + randomizedPlates: buildPlates(samples), + settings, + covariateColors, + }); + + const parsed = parseLayout(text); + expect(parsed.structuralErrors).toEqual([]); + // Colors correspond one-to-one with the groups, so validation is clean. + expect(validateLayout(parsed)).toEqual([]); + // The saved policy round-trips verbatim. + expect(parsed.settings!.naPolicy).toEqual(policy); + + const { samples: loaded } = buildPlatesFromRows(parsed.rows, parsed.settings!); + buildProcessedSearches(loaded, { + selectedCovariates: parsed.settings!.selectedCovariates, + qcColumn: parsed.settings!.qcColumn, + selectedQcValues: parsed.settings!.selectedQcValues, + naPolicy: parsed.settings!.naPolicy, + }); + const reloadedGroups = sortedKeys(loaded.map(s => s.covariateKey!)); + + // Exact reproduction: same groups after load, matching the stored colors one-to-one. + expect(reloadedGroups).toEqual(producedGroups); + expect(reloadedGroups).toEqual(sortedKeys(Object.keys(covariateColors))); + } + ); + + it('rejects a layout whose colors do not cover every produced group', () => { + // Fold nothing extra: na/NA/n/a stay distinct, so the samples form more groups than a color + // map that only colors the folded (N/A) groups would cover. This is the drift that used to + // render uncolored cells gray; validateLayout must now reject it. + const foldNone: NaPolicy = { foldBlank: false, foldSpellings: [] }; + const samples = makeSamples(); + buildProcessedSearches(samples, { + selectedCovariates: COVARIATES, + qcColumn: QC_COLUMN, + selectedQcValues: QC_VALUES, + naPolicy: foldNone, + }); + const producedGroups = sortedKeys(samples.map(s => s.covariateKey!)); + + // Drop one group's color to simulate a colors/groups mismatch. + const droppedGroup = producedGroups[0]; + const covariateColors = colorsForGroups(producedGroups.filter(g => g !== droppedGroup)); + + const text = serializeLayout({ + searches: samples, + randomizedPlates: buildPlates(samples), + settings: settingsFor(foldNone), + covariateColors, + }); + const errors = validateLayout(parseLayout(text)); + expect(errors.some(e => e.fatal && e.message.includes(droppedGroup))).toBe(true); + }); +}); diff --git a/src/tests/utils.test.ts b/src/tests/utils.test.ts index 307bc80..f485edc 100644 --- a/src/tests/utils.test.ts +++ b/src/tests/utils.test.ts @@ -1,4 +1,4 @@ -import { formatTimestampForFilename, withTimestamp, buildLayoutFileName, buildCovariateKey } from '../utils/utils'; +import { formatTimestampForFilename, withTimestamp, buildLayoutFileName, buildCovariateKey, effectiveValue, effectiveDisplayValue } from '../utils/utils'; import { SearchData, CovariateConfig } from '../utils/types'; const mkSample = (metadata: { [key: string]: string }): SearchData => ({ name: 'sample', metadata }); @@ -102,9 +102,11 @@ describe('buildCovariateKey injectivity (escape encoding)', () => { expect(buildCovariateKey(clean, config)).toBe('Training|108|FA1'); }); - it('keeps the N/A fallback for a genuinely missing value (B-min)', () => { + it('represents a genuinely-missing value with the missing marker by default', () => { + // Under na-value-handling the default policy keeps a blank cell distinct from a literal + // N/A: it becomes the lone-backslash missing marker, not the string N/A. const missing = mkSample({ Treatment: 'Training', Dose: '', Site: 'FA1' }); - expect(buildCovariateKey(missing, config)).toBe('Training|N/A|FA1'); + expect(buildCovariateKey(missing, config)).toBe('Training|\\|FA1'); }); it('preserves the QC prefix for clean data', () => { @@ -117,3 +119,83 @@ describe('buildCovariateKey injectivity (escape encoding)', () => { expect(buildCovariateKey(qc, qcConfig)).toBe('BatchQC|108|FA1'); }); }); + +// na-value-handling: a blank cell can stay distinct from a literal N/A, folded spellings +// collapse into N/A, and na/NA stay distinct unless folded. These are the green side of the +// red-green flip captured pre-policy in the covariate-key-fragility residual. +describe('buildCovariateKey under NaPolicy', () => { + const cols = ['Treatment', 'Dose', 'Site']; + const blank = mkSample({ Treatment: 'Training', Dose: '', Site: 'FA1' }); + const literalNA = mkSample({ Treatment: 'Training', Dose: 'N/A', Site: 'FA1' }); + + it('keeps a blank cell distinct from a literal N/A by default', () => { + const config: CovariateConfig = { selectedCovariates: cols }; + expect(buildCovariateKey(blank, config)).toBe('Training|\\|FA1'); + expect(buildCovariateKey(literalNA, config)).toBe('Training|N/A|FA1'); + expect(buildCovariateKey(blank, config)).not.toBe(buildCovariateKey(literalNA, config)); + }); + + it('folds a blank cell into N/A when the policy folds blank', () => { + const config: CovariateConfig = { + selectedCovariates: cols, + naPolicy: { foldBlank: true, foldSpellings: [] }, + }; + expect(buildCovariateKey(blank, config)).toBe('Training|N/A|FA1'); + expect(buildCovariateKey(blank, config)).toBe(buildCovariateKey(literalNA, config)); + }); + + it('folds a listed spelling into N/A while a literal N/A always folds', () => { + const config: CovariateConfig = { + selectedCovariates: cols, + naPolicy: { foldBlank: false, foldSpellings: ['na'] }, + }; + const na = mkSample({ Treatment: 'Training', Dose: 'na', Site: 'FA1' }); + expect(buildCovariateKey(na, config)).toBe('Training|N/A|FA1'); + expect(buildCovariateKey(na, config)).toBe(buildCovariateKey(literalNA, config)); + }); + + it('keeps na and NA distinct when only na is folded (grouping is by exact text)', () => { + const config: CovariateConfig = { + selectedCovariates: cols, + naPolicy: { foldBlank: false, foldSpellings: ['na'] }, + }; + const na = mkSample({ Treatment: 'Training', Dose: 'na', Site: 'FA1' }); + const NA = mkSample({ Treatment: 'Training', Dose: 'NA', Site: 'FA1' }); + expect(buildCovariateKey(na, config)).toBe('Training|N/A|FA1'); + expect(buildCovariateKey(NA, config)).toBe('Training|NA|FA1'); + expect(buildCovariateKey(na, config)).not.toBe(buildCovariateKey(NA, config)); + }); + + it('leaves na and NA as their own literal groups under the default policy', () => { + const config: CovariateConfig = { selectedCovariates: cols }; + const na = mkSample({ Treatment: 'Training', Dose: 'na', Site: 'FA1' }); + const NA = mkSample({ Treatment: 'Training', Dose: 'NA', Site: 'FA1' }); + expect(buildCovariateKey(na, config)).toBe('Training|na|FA1'); + expect(buildCovariateKey(NA, config)).toBe('Training|NA|FA1'); + }); +}); + +describe('effectiveValue and effectiveDisplayValue', () => { + it('classifies raw values under the default policy', () => { + expect(effectiveValue('Drug')).toEqual({ kind: 'value', value: 'Drug' }); + expect(effectiveValue('')).toEqual({ kind: 'missing' }); + expect(effectiveValue(' ')).toEqual({ kind: 'missing' }); + expect(effectiveValue('N/A')).toEqual({ kind: 'na' }); + expect(effectiveValue('na')).toEqual({ kind: 'value', value: 'na' }); + }); + + it('folds blank and listed spellings when the policy says so', () => { + const policy = { foldBlank: true, foldSpellings: ['na', 'n/a'] }; + expect(effectiveValue('', policy)).toEqual({ kind: 'na' }); + expect(effectiveValue('na', policy)).toEqual({ kind: 'na' }); + expect(effectiveValue('NA', policy)).toEqual({ kind: 'value', value: 'NA' }); + }); + + it('maps effective values to display text: folded -> N/A, missing -> blank, value -> itself', () => { + expect(effectiveDisplayValue('Drug')).toBe('Drug'); + expect(effectiveDisplayValue('N/A')).toBe('N/A'); + expect(effectiveDisplayValue('')).toBe(''); // missing renders blank by default + expect(effectiveDisplayValue('', { foldBlank: true, foldSpellings: [] })).toBe('N/A'); + expect(effectiveDisplayValue('na', { foldBlank: false, foldSpellings: ['na'] })).toBe('N/A'); + }); +}); diff --git a/src/utils/excelExport.ts b/src/utils/excelExport.ts index 8a22245..2fee745 100644 --- a/src/utils/excelExport.ts +++ b/src/utils/excelExport.ts @@ -1,6 +1,6 @@ import ExcelJS from 'exceljs'; -import { SearchData, CovariateColorInfo } from './types'; -import { getCovariateKey, withTimestamp } from './utils'; +import { SearchData, CovariateColorInfo, NaPolicy, DEFAULT_NA_POLICY } from './types'; +import { getCovariateKey, withTimestamp, effectiveDisplayValue } from './utils'; interface ExcelExportOptions { searches: SearchData[]; @@ -12,6 +12,7 @@ interface ExcelExportOptions { numColumns: number; inputFileName?: string; qcColumn?: string; // QC/Reference column name (for legend display when not in treatmentCovariates) + naPolicy?: NaPolicy; // Global N/A grouping choice; the legend shows folded groups as N/A } // Style constants @@ -57,7 +58,7 @@ const createSolidFill = (argbColor: string): ExcelJS.Fill => ({ * Pure and DOM-free, so it is reusable and unit-testable; exportToExcel adds the download. */ export function buildLayoutWorkbook(options: ExcelExportOptions): ExcelJS.Workbook { - const { searches, randomizedPlates, covariateColors, treatmentCovariates, exportCovariates, numRows, numColumns, qcColumn } = options; + const { searches, randomizedPlates, covariateColors, treatmentCovariates, exportCovariates, numRows, numColumns, qcColumn, naPolicy = DEFAULT_NA_POLICY } = options; const workbook = new ExcelJS.Workbook(); workbook.creator = 'Octopus'; @@ -72,7 +73,7 @@ export function buildLayoutWorkbook(options: ExcelExportOptions): ExcelJS.Workbo }); // Create legend sheet (always uses treatment covariates for color grouping) - createLegendSheet(workbook, searches, covariateColors, treatmentCovariates, randomizedPlates, qcColumn); + createLegendSheet(workbook, searches, covariateColors, treatmentCovariates, randomizedPlates, qcColumn, naPolicy); // Create sample details sheet (uses treatment covariates for color lookup) createSampleDetailsSheet(workbook, searches, treatmentCovariates, randomizedPlates, covariateColors); @@ -125,7 +126,7 @@ function calculateOptimalColumnWidth( // Check covariate value lengths (format: "covariate: value") selectedCovariates.forEach(cov => { - const covariateText = `${cov}: ${sample.metadata[cov] || 'N/A'}`; + const covariateText = `${cov}: ${sample.metadata[cov] ?? ''}`; maxLength = Math.max(maxLength, covariateText.length); }); }); @@ -202,9 +203,10 @@ function createPlateSheet( cell.alignment = { horizontal: 'center', vertical: 'middle' }; cell.font = { bold: true }; } else { - // Row 3: Covariate values + // Row 3: Covariate values. Each plate cell shows one sample, so this is a per-sample + // view: show the raw typed value (a blank cell shows blank), not the folded N/A label. const covariateText = exportCovariates - .map(cov => `${cov}: ${sample.metadata[cov] || 'N/A'}`) + .map(cov => `${cov}: ${sample.metadata[cov] ?? ''}`) .join('\n'); cell.value = covariateText; cell.alignment = { horizontal: 'left', vertical: 'top', wrapText: true }; @@ -263,7 +265,8 @@ function createLegendSheet( covariateColors: { [key: string]: CovariateColorInfo }, treatmentCovariates: string[], randomizedPlates: (SearchData | undefined)[][][], - qcColumn?: string + qcColumn?: string, + naPolicy: NaPolicy = DEFAULT_NA_POLICY ): void { const sheet = workbook.addWorksheet('Legend'); @@ -331,7 +334,8 @@ function createLegendSheet( // A missing value falls back to "N/A", matching the Plate Details modal. // A present "na" shows as-is. const representative = representativeByKey.get(combination); - const covCell = (cov: string): string => representative?.metadata[cov] || 'N/A'; + // Legend is a per-group rollup: a folded group shows N/A, a kept-distinct blank group shows blank. + const covCell = (cov: string): string => effectiveDisplayValue(representative?.metadata[cov] ?? '', naPolicy); // Build display values aligned to legendCovHeaders. let displayValues: string[]; diff --git a/src/utils/layoutIO.ts b/src/utils/layoutIO.ts index 148d64a..ef804ad 100644 --- a/src/utils/layoutIO.ts +++ b/src/utils/layoutIO.ts @@ -3,6 +3,7 @@ import { RandomizationAlgorithm, GroupingConstraint, CovariateColorInfo, + NaPolicy, getAllAlgorithms, } from './types'; import { @@ -22,7 +23,7 @@ import { * "appVersion": "1.4.0", // optional, provenance only * "plateCount": 3, // number of plates, enforced on load * "settings": { ... }, // the user-chosen configuration (LayoutSettings) - * "covariateColors": { ... }, // optional; key -> { color, fill } + * "covariateColors": { ... }, // required; key -> { color, fill }, one per covariate group * "samples": [ { id, plate, well, metadata } ] * } * @@ -62,6 +63,8 @@ export interface LayoutSettings { groupingConstraint: GroupingConstraint; /** Metadata column names in display order, so re-exports keep stable column order. */ metadataColumns: string[]; + /** Global N/A grouping choice, so a reloaded layout re-derives the same covariate groups. */ + naPolicy: NaPolicy; } export type CovariateColorMap = { [key: string]: CovariateColorInfo }; @@ -89,7 +92,7 @@ export interface ParsedLayout { plateCount: number | null; /** Settings from the file, or null when the settings object is missing/invalid. */ settings: LayoutSettings | null; - /** Colors from the file, or null when none are present. */ + /** Colors from the file (required), or null when the covariateColors object is missing/invalid. */ covariateColors: CovariateColorMap | null; /** Placement rows parsed from `samples` (empty when samples are missing/invalid). */ rows: LayoutRow[]; @@ -190,6 +193,7 @@ function settingsToJson(s: LayoutSettings) { subjectColumn: s.subjectColumn, groupingConstraint: s.groupingConstraint, metadataColumns: s.metadataColumns, + naPolicy: { foldBlank: s.naPolicy.foldBlank, foldSpellings: s.naPolicy.foldSpellings }, }; } @@ -218,13 +222,15 @@ export function serializeLayout(options: { }): string { const { searches, randomizedPlates, settings, covariateColors, appVersion } = options; + // A layout must record a color for every covariate group, so grouping reloads exactly and the + // load-time color check has something to verify against. Refuse to save without colors. const colorEntries = Object.entries(covariateColors); - const covariateColorsJson = - colorEntries.length > 0 - ? Object.fromEntries( - colorEntries.map(([key, info]) => [key, { color: info.color, fill: fillToken(info) }]) - ) - : undefined; + if (colorEntries.length === 0) { + throw new Error('Cannot save layout: a layout must include the covariate colors, but none were provided.'); + } + const covariateColorsJson = Object.fromEntries( + colorEntries.map(([key, info]) => [key, { color: info.color, fill: fillToken(info) }]) + ); const doc = { format: LAYOUT_FORMAT, @@ -232,7 +238,7 @@ export function serializeLayout(options: { ...(appVersion ? { appVersion } : {}), plateCount: randomizedPlates.length, settings: settingsToJson(settings), - ...(covariateColorsJson ? { covariateColors: covariateColorsJson } : {}), + covariateColors: covariateColorsJson, samples: searches.map((search) => { // getPlateNumber/getWell return '' when the sample is not on the grid. That must never be // written (plate has to be a positive integer), so fail the save rather than emit a file the @@ -284,8 +290,14 @@ function parseSettings(raw: unknown): { value: LayoutSettings | null; errors: La `settings.groupingConstraint must be one of: ${GROUPING_CONSTRAINTS.join(', ')}.` ); need(isStringArray(r.metadataColumns), 'settings.metadataColumns must be an array of strings.'); + const naPolicy = r.naPolicy; + need( + isRecord(naPolicy) && typeof naPolicy.foldBlank === 'boolean' && isStringArray(naPolicy.foldSpellings), + 'settings.naPolicy must be an object with a boolean foldBlank and a string-array foldSpellings.' + ); if (errors.length) return { value: null, errors }; + const validNaPolicy = naPolicy as { foldBlank: boolean; foldSpellings: string[] }; return { value: { selectedIdColumn: r.idColumn as string, @@ -299,6 +311,7 @@ function parseSettings(raw: unknown): { value: LayoutSettings | null; errors: La subjectColumn: r.subjectColumn as string, groupingConstraint: r.groupingConstraint as GroupingConstraint, metadataColumns: r.metadataColumns as string[], + naPolicy: { foldBlank: validNaPolicy.foldBlank, foldSpellings: validNaPolicy.foldSpellings }, }, errors: [], }; @@ -433,10 +446,17 @@ export function parseLayout(fileText: string): ParsedLayout { if (settingsResult.errors.length) errors.push(...settingsResult.errors); else base.settings = settingsResult.value; - if (root.covariateColors !== undefined) { + // covariateColors is required: a saved layout always records a color for every covariate group. + if (root.covariateColors === undefined) { + errors.push(fatal('The layout file is missing its "covariateColors" object.')); + } else { const colorsResult = parseColors(root.covariateColors); if (colorsResult.errors.length) errors.push(...colorsResult.errors); - else base.covariateColors = Object.keys(colorsResult.value).length > 0 ? colorsResult.value : null; + else if (Object.keys(colorsResult.value).length === 0) { + errors.push(fatal('The layout file records no covariate colors; a layout must color every covariate group.')); + } else { + base.covariateColors = colorsResult.value; + } } const samplesResult = parseSamples(root.samples); @@ -612,17 +632,23 @@ export function validateLayout(parsed: ParsedLayout): LayoutValidationError[] { ); } - // Covariate-color consistency: every stored color must key a covariate combination that the - // samples actually produce under the selected covariates. + // Covariate-color consistency: the stored colors must correspond exactly to the covariate + // groups the samples produce under the selected covariates and the saved N/A policy. Both + // directions are enforced so a reloaded layout reproduces the file's grouping exactly: + // - every stored color must key a real group (no orphan colors), and + // - every produced group must have a stored color (no uncolored groups rendered gray). if (parsed.covariateColors) { + const colors = parsed.covariateColors; const samples: SearchData[] = parsed.rows.map((r) => ({ name: r.name, metadata: r.metadata })); buildProcessedSearches(samples, { selectedCovariates: s.selectedCovariates, qcColumn: s.qcColumn, selectedQcValues: s.selectedQcValues, + naPolicy: s.naPolicy, }); - const derivedKeys = new Set(samples.map((x) => x.covariateKey)); - Object.keys(parsed.covariateColors).forEach((key) => { + const derivedKeys = new Set(samples.map((x) => x.covariateKey as string)); + const colorKeys = Object.keys(colors); + colorKeys.forEach((key) => { if (!derivedKeys.has(key)) { errors.push( fatal( @@ -631,6 +657,16 @@ export function validateLayout(parsed: ParsedLayout): LayoutValidationError[] { ); } }); + derivedKeys.forEach((key) => { + if (!(key in colors)) { + errors.push( + fatal( + `The layout produces the covariate group "${key}" but stores no color for it ` + + `(the samples form ${derivedKeys.size} groups but the file records ${colorKeys.length} colors).` + ) + ); + } + }); } return errors; diff --git a/src/utils/types.ts b/src/utils/types.ts index 92a082c..b8cd500 100644 --- a/src/utils/types.ts +++ b/src/utils/types.ts @@ -13,10 +13,31 @@ export interface SearchData { isQC?: boolean; // Whether this sample is a QC/Reference sample } +/** + * The single global choice for how N/A-type covariate values are grouped. It applies to + * every column, not per column. + * - foldBlank: fold a genuinely-blank cell into the canonical N/A group. + * - foldSpellings: the exact (case-sensitive) N/A-type spellings that fold into N/A. + * The literal 'N/A' always folds and is never listed here. + */ +export interface NaPolicy { + foldBlank: boolean; + foldSpellings: string[]; +} + +/** + * Policy used when the N/A values setting is not shown (no column mixes spellings). + * A blank cell stays distinct via the missing marker and every N/A-type spelling stays its + * own literal value. A cell whose exact text is 'N/A' always maps to the canonical N/A group, + * regardless of policy (see effectiveValue). + */ +export const DEFAULT_NA_POLICY: NaPolicy = { foldBlank: false, foldSpellings: [] }; + export interface CovariateConfig { selectedCovariates: string[]; // Treatment covariate column names qcColumn?: string; // QC/Reference column name selectedQcValues?: string[]; // QC/Reference values + naPolicy?: NaPolicy; // Global N/A grouping choice; defaults to DEFAULT_NA_POLICY } export type RandomizationAlgorithm = keyof typeof ALGORITHM_CONFIG; diff --git a/src/utils/utils.ts b/src/utils/utils.ts index 83c047b..d51c210 100644 --- a/src/utils/utils.ts +++ b/src/utils/utils.ts @@ -1,4 +1,4 @@ -import { SearchData, RandomizationAlgorithm, CovariateConfig, RepeatedMeasuresConfig } from './types'; +import { SearchData, RandomizationAlgorithm, CovariateConfig, RepeatedMeasuresConfig, NaPolicy, DEFAULT_NA_POLICY } from './types'; import { QualityLevel, QUALITY_LEVEL_CONFIG } from './configs'; import Papa from 'papaparse'; import { balancedBlockRandomization } from '../algorithms/balancedRandomization'; @@ -40,27 +40,138 @@ function escapeCovariateValue(value: string): string { return value.replace(/\\/g, '\\\\').replace(/\|/g, '\\|'); } +/** + * Key part for a blank cell, kept distinct from a cell that literally says N/A. + * The marker is a single backslash, which is safe because: + * - We emit it raw, never through escapeCovariateValue, so it stays a lone backslash. + * - Escaping a real value doubles every backslash, so a real value never becomes a lone one. + * - It is not 'N/A', so it also cannot match a folded N/A cell. + * A lone backslash in a key therefore always means the cell was blank. + */ +export const MISSING_MARKER = '\\'; // '\\' is a JS string literal for one backslash character + +/** True when a value is empty or whitespace-only. */ +function isBlank(value: string): boolean { + return value.trim().length === 0; +} + +/** + * True when a value is a candidate for the N/A bucket: blank/whitespace-only, or any case + * spelling of NA or N/A. Nothing else (None, null, -) counts. + */ +export function isNaType(value: string): boolean { + if (isBlank(value)) return true; + const lower = value.toLowerCase(); + return lower === 'na' || lower === 'n/a'; +} + +/** + * What a raw covariate value means for grouping and display under the global policy: + * - value: an ordinary value, or an N/A-type spelling the user chose to keep distinct. + * - na: folds into the canonical N/A group. + * - missing: a genuinely-blank cell kept distinct from N/A (uses the missing marker). + */ +export type EffectiveValue = + | { kind: 'value'; value: string } + | { kind: 'na' } + | { kind: 'missing' }; + +/** + * Decide, for one raw covariate value under the global policy, what it means. Grouping (the key) + * and display both go through this one function so they cannot drift. See the na-value-handling + * design. The literal 'N/A' always folds, regardless of the policy. + */ +export function effectiveValue(raw: string, policy: NaPolicy = DEFAULT_NA_POLICY): EffectiveValue { + if (!isNaType(raw)) return { kind: 'value', value: raw }; + if (isBlank(raw)) return policy.foldBlank ? { kind: 'na' } : { kind: 'missing' }; + if (raw === 'N/A') return { kind: 'na' }; + return policy.foldSpellings.includes(raw) ? { kind: 'na' } : { kind: 'value', value: raw }; +} + +/** The key part for an effective value. The missing marker is emitted raw, never escaped. */ +function effectiveKeyPart(ev: EffectiveValue): string { + switch (ev.kind) { + case 'value': return escapeCovariateValue(ev.value); + case 'na': return escapeCovariateValue('N/A'); + case 'missing': return MISSING_MARKER; + } +} + +/** + * The display text for a raw covariate value under the policy. A folded group shows 'N/A', a + * kept-distinct missing (blank) group shows blank, and an ordinary value shows itself. Group + * rollups (legend, summary, modal, Excel legend) use this so their labels match the grouping. + */ +export function effectiveDisplayValue(raw: string, policy: NaPolicy = DEFAULT_NA_POLICY): string { + const ev = effectiveValue(raw, policy); + switch (ev.kind) { + case 'value': return ev.value; + case 'na': return 'N/A'; + case 'missing': return ''; + } +} + +export interface NaDetectionResult { + /** + * For each column that holds at least one N/A-type value, the distinct spellings found. A + * blank or whitespace-only cell is recorded as the empty string ''. A non-blank spelling is + * recorded with its exact text (for example 'na', 'NA', 'N/A'), since grouping is case-exact. + */ + byColumn: Map>; + /** Union of the distinct N/A-type spellings across all columns, with blank recorded as ''. */ + spellings: Set; + /** True when at least one column holds two or more distinct N/A-type spellings. */ + hasAmbiguousColumn: boolean; +} + +/** + * Scan every given column across all samples for N/A-type values (blank/whitespace-only, or any + * case spelling of NA or N/A). Records the distinct spellings per column, their union, and whether + * any single column mixes two or more. A mixing column is what makes the data ambiguous and + * surfaces the "N/A values" setting. Runs at upload, before covariate/QC selection, so it treats + * all columns uniformly. Whitespace-only cells all collapse to the one blank token ''. + */ +export function detectNaTypeValues(samples: SearchData[], columns: string[]): NaDetectionResult { + const byColumn = new Map>(); + const spellings = new Set(); + for (const col of columns) { + const found = new Set(); + for (const sample of samples) { + const raw = sample.metadata[col] ?? ''; + if (!isNaType(raw)) continue; + found.add(isBlank(raw) ? '' : raw); + } + if (found.size > 0) { + byColumn.set(col, found); + found.forEach(token => spellings.add(token)); + } + } + const hasAmbiguousColumn = Array.from(byColumn.values()).some(set => set.size >= 2); + return { byColumn, spellings, hasAmbiguousColumn }; +} + export function buildCovariateKey( search: SearchData, config: CovariateConfig ): string { - const { selectedCovariates, qcColumn: qcColumn, selectedQcValues: selectedQcValues } = config; + const { selectedCovariates, qcColumn, selectedQcValues, naPolicy = DEFAULT_NA_POLICY } = config; - // Build the base covariate key + // Build the base covariate key. Each value passes through the policy so a blank cell, a + // literal N/A, and folded spellings all resolve to the right key part. const baseKey = selectedCovariates - .map(cov => escapeCovariateValue(search.metadata[cov] || 'N/A')) + .map(cov => effectiveKeyPart(effectiveValue(search.metadata[cov] ?? '', naPolicy))) .join('|'); -// Check if this sample is a QC sample and QC column is not in covariates + // Check if this sample is a QC sample and QC column is not in covariates if (qcColumn && selectedQcValues && selectedQcValues.length > 0 - && !selectedCovariates.includes(qcColumn)) - { + && !selectedCovariates.includes(qcColumn)) { const sampleValue = search.metadata[qcColumn]; const isQC = sampleValue && selectedQcValues.includes(sampleValue); - // Prepend QC value if sample is QC and QC column is not a covariate + // Prepend QC value if sample is QC and QC column is not a covariate. The prepended value + // goes through the same policy so it can't drift from the covariate parts. if (isQC) { - return `${escapeCovariateValue(sampleValue)}|${baseKey}`; + return `${effectiveKeyPart(effectiveValue(sampleValue, naPolicy))}|${baseKey}`; } } @@ -138,7 +249,8 @@ export function randomizeSearches( keepEmptyInLastPlate: boolean = true, numRows: number = 8, numColumns: number = 12, - repeatedMeasuresConfig?: RepeatedMeasuresConfig + repeatedMeasuresConfig?: RepeatedMeasuresConfig, + naPolicy: NaPolicy = DEFAULT_NA_POLICY ): { plates: (SearchData | undefined)[][][]; plateAssignments?: Map; @@ -151,7 +263,8 @@ export function randomizeSearches( repeatedMeasuresConfig, keepEmptyInLastPlate, numRows, - numColumns + numColumns, + naPolicy ); } @@ -176,7 +289,7 @@ export function buildProcessedSearches( searches: SearchData[], config: CovariateConfig ): void { - const { selectedCovariates, qcColumn, selectedQcValues } = config; + const { selectedCovariates, qcColumn, selectedQcValues, naPolicy } = config; searches.forEach(search => { let isQC = false; if (qcColumn && selectedQcValues && selectedQcValues.length > 0) { @@ -190,6 +303,7 @@ export function buildProcessedSearches( selectedCovariates, qcColumn, selectedQcValues, + naPolicy, }); }); } diff --git a/test-data/na-value-handling-demo.csv b/test-data/na-value-handling-demo.csv new file mode 100644 index 0000000..f13919c --- /dev/null +++ b/test-data/na-value-handling-demo.csv @@ -0,0 +1,37 @@ +Sample ID,Treatment,Dose,Region,QC +S01,Drug,0,R1, +S02,Drug,0,R1, +S03,Drug,0,R1, +S04,Drug,100,R1, +S05,Drug,100,R1, +S06,Drug,100,R1, +S07,Drug,na,R1, +S08,Drug,na,R1, +S09,Drug,NA,R1, +S10,Drug,NA,R1, +S11,Drug,n/a,R1, +S12,Drug,N/A,R1, +S13,Drug,,R1, +S14,Drug,,R1, +S15,Placebo,0,R2, +S16,Placebo,0,R2, +S17,Placebo,0,R2, +S18,Placebo,100,R2, +S19,Placebo,100,R2, +S20,Placebo,100,R2, +S21,Placebo,na,R2, +S22,Placebo,na,R2, +S23,Control,0,na, +S24,Control,0,na, +S25,Control,0,na, +S26,Control,100,na, +S27,Control,100,na, +S28,Control,100,na, +S29,na,na,na,BatchQC +S30,na,na,na,BatchQC +S31,na,na,na,BatchQC +S32,na,na,na,BatchQC +S33,na,na,na,BatchRef +S34,na,na,na,BatchRef +S35,na,na,na,BatchRef +S36,na,na,na,BatchRef