diff --git a/src/lib/types/fileTypes.ts b/src/lib/types/fileTypes.ts index 5ba873ba1..d9cdb0078 100644 --- a/src/lib/types/fileTypes.ts +++ b/src/lib/types/fileTypes.ts @@ -65,6 +65,16 @@ export type FileProcessingResult = { columnNames?: string[]; sampleData?: string | unknown[]; hasEmptyColumns?: boolean; + /** Enhanced column metadata with type detection and statistics */ + columnMetadata?: CSVColumnMetadata[]; + /** Data quality warnings */ + dataQualityWarnings?: CSVDataQualityWarning[]; + /** Overall data quality score (0-100) */ + dataQualityScore?: number; + /** Whether headers were detected */ + hasHeaders?: boolean; + /** Detected delimiter */ + detectedDelimiter?: string; // PDF-specific metadata version?: string; estimatedPages?: number | null; @@ -93,6 +103,66 @@ export type FileProcessingResult = { */ export type SampleDataFormat = "object" | "json" | "csv" | "markdown"; +/** + * Detected data type for a CSV column + */ +export type CSVColumnDataType = + | "string" + | "number" + | "integer" + | "float" + | "boolean" + | "date" + | "datetime" + | "email" + | "url" + | "empty" + | "mixed"; + +/** + * Data quality warning for CSV columns + */ +export type CSVDataQualityWarning = { + column: string; + type: + | "empty_values" + | "invalid_name" + | "mixed_types" + | "high_null_rate" + | "duplicates" + | "inconsistent_format"; + message: string; + severity: "info" | "warning" | "error"; + affectedRows?: number; +}; + +/** + * Rich metadata for a single CSV column + */ +export type CSVColumnMetadata = { + name: string; + index: number; + detectedType: CSVColumnDataType; + /** Confidence of type detection (0-100) */ + typeConfidence: number; + /** Count of null/empty values */ + nullCount: number; + /** Count of unique values */ + uniqueCount: number; + /** Sample values from this column (up to 5) */ + sampleValues: string[]; + /** For numeric columns: min value */ + minValue?: number; + /** For numeric columns: max value */ + maxValue?: number; + /** For numeric columns: average value */ + avgValue?: number; + /** For date columns: detected format (e.g., 'YYYY-MM-DD', 'MM/DD/YYYY') */ + dateFormat?: string; + /** Column name validation issues */ + nameIssues?: string[]; +}; + /** * CSV processor options */ diff --git a/src/lib/utils/csvProcessor.ts b/src/lib/utils/csvProcessor.ts index 3345c63df..704dcc901 100644 --- a/src/lib/utils/csvProcessor.ts +++ b/src/lib/utils/csvProcessor.ts @@ -11,8 +11,542 @@ import type { FileProcessingResult, CSVProcessorOptions, SampleDataFormat, + CSVColumnDataType, + CSVColumnMetadata, + CSVDataQualityWarning, } from "../types/fileTypes.js"; +// ============================================================================ +// Data Type Detection Patterns +// ============================================================================ + +const DATE_PATTERNS = [ + { regex: /^\d{4}-\d{2}-\d{2}$/, format: "YYYY-MM-DD" }, + { regex: /^\d{2}\/\d{2}\/\d{4}$/, format: "MM/DD/YYYY" }, + { regex: /^\d{2}-\d{2}-\d{4}$/, format: "DD-MM-YYYY" }, + { regex: /^\d{2}\.\d{2}\.\d{4}$/, format: "DD.MM.YYYY" }, + { regex: /^\d{4}\/\d{2}\/\d{2}$/, format: "YYYY/MM/DD" }, +]; + +const DATETIME_PATTERNS = [ + { regex: /^\d{4}-\d{2}-\d{2}[T ]\d{2}:\d{2}:\d{2}/, format: "ISO8601" }, + { regex: /^\d{2}\/\d{2}\/\d{4} \d{2}:\d{2}/, format: "MM/DD/YYYY HH:mm" }, +]; + +const EMAIL_REGEX = /^[^\s@]+@[^\s@]+\.[^\s@]+$/; +const URL_REGEX = /^(https?:\/\/|www\.)[^\s]+$/i; +const INTEGER_REGEX = /^-?\d+$/; +const FLOAT_REGEX = /^-?\d+\.\d+$/; +const BOOLEAN_VALUES = new Set([ + "true", + "false", + "yes", + "no", + "1", + "0", + "t", + "f", + "y", + "n", +]); + +// ============================================================================ +// Column Name Validation +// ============================================================================ + +/** + * Validate column name and return issues + */ +function validateColumnName(name: string): string[] { + const issues: string[] = []; + + if (!name || name.trim() === "") { + issues.push("Empty or blank column name"); + return issues; + } + + if (name !== name.trim()) { + issues.push("Leading or trailing whitespace"); + } + + if (/^\d/.test(name)) { + issues.push("Starts with a number"); + } + + if (/[^a-zA-Z0-9_\- ]/.test(name)) { + issues.push("Contains special characters"); + } + + if (name.length > 64) { + issues.push("Name exceeds 64 characters"); + } + + if (/\s{2,}/.test(name)) { + issues.push("Contains multiple consecutive spaces"); + } + + return issues; +} + +// ============================================================================ +// Data Type Detection +// ============================================================================ + +/** + * Detect the data type of a single value + */ +function detectValueType(value: string): CSVColumnDataType { + if (value === "" || value === null || value === undefined) { + return "empty"; + } + + const trimmed = value.trim(); + + if (trimmed === "") { + return "empty"; + } + + // Check boolean first (before numbers since "1" and "0" could be both) + if (BOOLEAN_VALUES.has(trimmed.toLowerCase())) { + return "boolean"; + } + + // Check integer + if (INTEGER_REGEX.test(trimmed)) { + return "integer"; + } + + // Check float + if (FLOAT_REGEX.test(trimmed)) { + return "float"; + } + + // Check email + if (EMAIL_REGEX.test(trimmed)) { + return "email"; + } + + // Check URL + if (URL_REGEX.test(trimmed)) { + return "url"; + } + + // Check datetime (before date since datetime is more specific) + for (const pattern of DATETIME_PATTERNS) { + if (pattern.regex.test(trimmed)) { + return "datetime"; + } + } + + // Check date + for (const pattern of DATE_PATTERNS) { + if (pattern.regex.test(trimmed)) { + return "date"; + } + } + + return "string"; +} + +/** + * Detect date format from value + */ +function detectDateFormat(value: string): string | undefined { + const trimmed = value.trim(); + + for (const pattern of DATETIME_PATTERNS) { + if (pattern.regex.test(trimmed)) { + return pattern.format; + } + } + + for (const pattern of DATE_PATTERNS) { + if (pattern.regex.test(trimmed)) { + return pattern.format; + } + } + + return undefined; +} + +/** + * Determine the predominant type for a column based on sampled values + */ +function determineColumnType(types: CSVColumnDataType[]): { + type: CSVColumnDataType; + confidence: number; +} { + const nonEmpty = types.filter((t) => t !== "empty"); + + if (nonEmpty.length === 0) { + return { type: "empty", confidence: 100 }; + } + + // Count occurrences of each type + const typeCounts = new Map(); + for (const t of nonEmpty) { + typeCounts.set(t, (typeCounts.get(t) || 0) + 1); + } + + // Find the most common type + let maxType: CSVColumnDataType = "string"; + let maxCount = 0; + for (const [type, count] of typeCounts) { + if (count > maxCount) { + maxCount = count; + maxType = type; + } + } + + // Calculate confidence + const confidence = Math.round((maxCount / nonEmpty.length) * 100); + + // Consolidate integer and float into number if the column contains only numeric types + // This check must happen before the mixed-type check to avoid classifying numeric-only columns as mixed + if (typeCounts.has("integer") && typeCounts.has("float")) { + // Check if these are the only two types (purely numeric column) + if (typeCounts.size === 2) { + const totalNumeric = + (typeCounts.get("integer") || 0) + (typeCounts.get("float") || 0); + const numericConfidence = Math.round( + (totalNumeric / nonEmpty.length) * 100, + ); + return { type: "number", confidence: numericConfidence }; + } + } + + // If confidence is low and multiple types exist, mark as mixed + if (confidence < 70 && typeCounts.size > 1) { + return { type: "mixed", confidence }; + } + + return { type: maxType, confidence }; +} + +/** + * Analyze a single column and return rich metadata + */ +function analyzeColumn( + columnName: string, + columnIndex: number, + values: string[], +): CSVColumnMetadata { + const types: CSVColumnDataType[] = []; + const uniqueValues = new Set(); + const numericValues: number[] = []; + let nullCount = 0; + let dateFormat: string | undefined; + + for (const value of values) { + const trimmed = value?.trim() ?? ""; + + if (trimmed === "") { + nullCount++; + types.push("empty"); + continue; + } + + uniqueValues.add(trimmed); + const type = detectValueType(trimmed); + types.push(type); + + // Collect numeric values for statistics + if (type === "integer" || type === "float") { + const num = parseFloat(trimmed); + if (!isNaN(num)) { + numericValues.push(num); + } + } + + // Detect date format + if ((type === "date" || type === "datetime") && !dateFormat) { + dateFormat = detectDateFormat(trimmed); + } + } + + const { type: detectedType, confidence } = determineColumnType(types); + + // Get sample values (up to 5 unique non-empty) + const sampleValues = Array.from(uniqueValues).slice(0, 5); + + // Calculate numeric statistics + let minValue: number | undefined; + let maxValue: number | undefined; + let avgValue: number | undefined; + + if (numericValues.length > 0) { + minValue = Math.min(...numericValues); + maxValue = Math.max(...numericValues); + avgValue = + Math.round( + (numericValues.reduce((a, b) => a + b, 0) / numericValues.length) * 100, + ) / 100; + } + + // Validate column name + const nameIssues = validateColumnName(columnName); + + const metadata: CSVColumnMetadata = { + name: columnName, + index: columnIndex, + detectedType, + typeConfidence: confidence, + nullCount, + uniqueCount: uniqueValues.size, + sampleValues, + }; + + if (minValue !== undefined) { + metadata.minValue = minValue; + } + if (maxValue !== undefined) { + metadata.maxValue = maxValue; + } + if (avgValue !== undefined) { + metadata.avgValue = avgValue; + } + if (dateFormat) { + metadata.dateFormat = dateFormat; + } + if (nameIssues.length > 0) { + metadata.nameIssues = nameIssues; + } + + return metadata; +} + +/** + * Generate data quality warnings based on column analysis + */ +function generateDataQualityWarnings( + columns: CSVColumnMetadata[], + totalRows: number, +): CSVDataQualityWarning[] { + const warnings: CSVDataQualityWarning[] = []; + + for (const col of columns) { + // Check for high null rate (>20%) + const nullRate = totalRows > 0 ? col.nullCount / totalRows : 0; + if (nullRate > 0.2) { + warnings.push({ + column: col.name, + type: "high_null_rate", + message: `Column has ${Math.round(nullRate * 100)}% empty/null values (${col.nullCount} of ${totalRows} rows)`, + severity: nullRate > 0.5 ? "warning" : "info", + affectedRows: col.nullCount, + }); + } + + // Check for invalid column names + if (col.nameIssues && col.nameIssues.length > 0) { + warnings.push({ + column: col.name, + type: "invalid_name", + message: `Column name issues: ${col.nameIssues.join(", ")}`, + severity: col.name.trim() === "" ? "error" : "warning", + }); + } + + // Check for mixed types (low confidence) + if (col.detectedType === "mixed" || col.typeConfidence < 70) { + warnings.push({ + column: col.name, + type: "mixed_types", + message: `Column has inconsistent data types (${col.typeConfidence}% confidence for ${col.detectedType})`, + severity: "warning", + }); + } + + // Check for potential duplicates (very low unique count) + if (totalRows > 10 && col.uniqueCount === 1 && col.nullCount === 0) { + warnings.push({ + column: col.name, + type: "duplicates", + message: `All ${totalRows} rows have the same value`, + severity: "info", + affectedRows: totalRows, + }); + } + + // Check for all empty column + if (col.detectedType === "empty") { + warnings.push({ + column: col.name, + type: "empty_values", + message: "Column is entirely empty", + severity: "warning", + affectedRows: totalRows, + }); + } + } + + return warnings; +} + +/** + * Calculate overall data quality score + */ +function calculateDataQualityScore( + columns: CSVColumnMetadata[], + warnings: CSVDataQualityWarning[], + totalRows: number, +): number { + if (columns.length === 0 || totalRows === 0) { + return 0; + } + + let score = 100; + + // Deduct for warnings + for (const warning of warnings) { + switch (warning.severity) { + case "error": + score -= 15; + break; + case "warning": + score -= 8; + break; + case "info": + score -= 3; + break; + } + } + + // Deduct for overall null rate + const totalNulls = columns.reduce((sum, col) => sum + col.nullCount, 0); + const totalCells = columns.length * totalRows; + const overallNullRate = totalCells > 0 ? totalNulls / totalCells : 0; + score -= Math.round(overallNullRate * 30); + + // Deduct for low type confidence + const avgConfidence = + columns.reduce((sum, col) => sum + col.typeConfidence, 0) / columns.length; + if (avgConfidence < 80) { + score -= Math.round((80 - avgConfidence) / 2); + } + + return Math.max(0, Math.min(100, score)); +} + +/** + * Analyze all columns in parsed CSV data + */ +function analyzeColumns(rows: unknown[]): { + columnMetadata: CSVColumnMetadata[]; + dataQualityWarnings: CSVDataQualityWarning[]; + dataQualityScore: number; +} { + if (rows.length === 0) { + return { + columnMetadata: [], + dataQualityWarnings: [], + dataQualityScore: 0, + }; + } + + const columnNames = Object.keys(rows[0] as Record); + const columnMetadata: CSVColumnMetadata[] = []; + + for (let i = 0; i < columnNames.length; i++) { + const colName = columnNames[i]; + const values = rows.map((row) => + String((row as Record)[colName] ?? ""), + ); + columnMetadata.push(analyzeColumn(colName, i, values)); + } + + const dataQualityWarnings = generateDataQualityWarnings( + columnMetadata, + rows.length, + ); + const dataQualityScore = calculateDataQualityScore( + columnMetadata, + dataQualityWarnings, + rows.length, + ); + + return { + columnMetadata, + dataQualityWarnings, + dataQualityScore, + }; +} + +/** + * Detect if the first row appears to be a header row + * + * Heuristics used: + * 1. Header values should be text/string type (not numbers, dates, emails, etc.) + * 2. Header values should be unique (no duplicate column names) + * 3. If data rows exist, headers should have different type profile than data + * + * @param headerValues - The values from the first row (potential headers) + * @param dataRows - Sample of data rows for comparison (optional) + * @returns true if the first row appears to be headers + */ +function detectHasHeaders( + headerValues: string[], + dataRows?: Record[], +): boolean { + if (headerValues.length === 0) { + return false; + } + + // Check 1: All header values should look like text labels, not data values + let textLikeCount = 0; + for (const value of headerValues) { + const trimmed = value?.trim() ?? ""; + if (trimmed === "") { + continue; // Empty headers are allowed but don't count toward text-like + } + + const type = detectValueType(trimmed); + // Headers are typically strings - not numbers, dates, emails, URLs, or booleans + if (type === "string") { + textLikeCount++; + } + } + + // If most header values are text-like (not numeric/date/etc.), likely headers + const nonEmptyHeaders = headerValues.filter((v) => v?.trim()).length; + if (nonEmptyHeaders === 0) { + return false; + } + + const textRatio = textLikeCount / nonEmptyHeaders; + + // Check 2: Headers should be unique + const uniqueHeaders = new Set( + headerValues.map((v) => v?.trim().toLowerCase()), + ); + const hasUniqueHeaders = uniqueHeaders.size === headerValues.length; + + // Check 3: Compare with data rows if available + if (dataRows && dataRows.length > 0) { + // If first data row has different type profile than headers, likely has headers + const firstDataRow = Object.values(dataRows[0] || {}).map((v) => + String(v ?? ""), + ); + let dataTextCount = 0; + for (const value of firstDataRow) { + const type = detectValueType(value?.trim() ?? ""); + if (type === "string") { + dataTextCount++; + } + } + const dataTextRatio = + firstDataRow.length > 0 ? dataTextCount / firstDataRow.length : 0; + + // If headers are mostly text but data has more varied types, likely has headers + if (textRatio > 0.7 && dataTextRatio < textRatio - 0.2) { + return true; + } + } + + // Default: if >70% of header values are text-like and unique, assume headers + return textRatio >= 0.7 && hasUniqueHeaders; +} + /** * Detect if first line is CSV metadata (not actual data/headers) * Common patterns: @@ -149,6 +683,22 @@ export class CSVProcessor { truncated: wasTruncated, }); + // Parse a sample for enhanced metadata analysis (raw format still benefits from column analysis) + const sampleForAnalysis = await this.parseCSVString( + limitedCSV, + Math.min(rowCount, 500), + ); + const { columnMetadata, dataQualityWarnings, dataQualityScore } = + analyzeColumns(sampleForAnalysis); + + // Log data quality summary + if (dataQualityWarnings.length > 0) { + logger.debug("[CSVProcessor] Data quality warnings detected", { + warningCount: dataQualityWarnings.length, + score: dataQualityScore, + }); + } + return { type: "csv", content: limitedCSV, @@ -160,6 +710,14 @@ export class CSVProcessor { totalLines: limitedLines.length, columnCount: (limitedLines[0] || "").split(",").length, extension, + columnMetadata, + dataQualityWarnings, + dataQualityScore, + hasHeaders: detectHasHeaders( + (limitedLines[0] || "").split(","), + undefined, + ), + detectedDelimiter: ",", }, }; } @@ -217,6 +775,18 @@ export class CSVProcessor { logger.warn("[CSVProcessor] CSV file contains no data rows"); } + // Perform enhanced column analysis + const { columnMetadata, dataQualityWarnings, dataQualityScore } = + analyzeColumns(nonEmptyRows); + + // Log data quality summary + if (dataQualityWarnings.length > 0) { + logger.debug("[CSVProcessor] Data quality warnings detected", { + warningCount: dataQualityWarnings.length, + score: dataQualityScore, + }); + } + // Format parsed data logger.debug( `[CSVProcessor] Converting ${rowCount} rows to ${formatStyle} format`, @@ -233,6 +803,7 @@ export class CSVProcessor { columnCount, outputLength: formatted.length, hasEmptyColumns, + dataQualityScore, }); return { @@ -248,6 +819,14 @@ export class CSVProcessor { sampleData, hasEmptyColumns, extension, + columnMetadata, + dataQualityWarnings, + dataQualityScore, + hasHeaders: detectHasHeaders( + columnNames, + nonEmptyRows as Record[], + ), + detectedDelimiter: ",", }, }; } diff --git a/test/unit/utils/csvProcessor.test.ts b/test/unit/utils/csvProcessor.test.ts index d57ef8058..72f2cbe3a 100644 --- a/test/unit/utils/csvProcessor.test.ts +++ b/test/unit/utils/csvProcessor.test.ts @@ -546,4 +546,365 @@ Charlie,35,Chicago`; expect(rawResult.metadata.totalLines).toBe(5); // header + 2 data + 2 whitespace }); }); + + describe("Column metadata analysis", () => { + it("should detect integer column type", async () => { + const csvData = Buffer.from( + "id,name\n1,Alice\n2,Bob\n3,Charlie\n4,Diana\n5,Eve", + ); + const result = await CSVProcessor.process(csvData, { + formatStyle: "raw", + }); + + expect(result.metadata.columnMetadata).toBeDefined(); + const idColumn = result.metadata.columnMetadata?.find( + (col) => col.name === "id", + ); + expect(idColumn?.detectedType).toBe("integer"); + expect(idColumn?.typeConfidence).toBeGreaterThanOrEqual(70); + }); + + it("should detect float/number column type", async () => { + const csvData = Buffer.from( + "price,item\n19.99,Widget\n29.50,Gadget\n9.99,Tool", + ); + const result = await CSVProcessor.process(csvData, { + formatStyle: "raw", + }); + + const priceColumn = result.metadata.columnMetadata?.find( + (col) => col.name === "price", + ); + expect(priceColumn?.detectedType).toBe("float"); + expect(priceColumn?.minValue).toBe(9.99); + expect(priceColumn?.maxValue).toBe(29.5); + expect(priceColumn?.avgValue).toBeDefined(); + }); + + it("should consolidate integers and floats as number type", async () => { + // Mix of integers and floats should be classified as "number", not "mixed" + // Using values > 1 to avoid boolean detection (0 and 1 are treated as boolean) + const csvData = Buffer.from("value\n10\n2.5\n30\n4.5\n50\n6.5\n70\n8.5"); + const result = await CSVProcessor.process(csvData, { + formatStyle: "raw", + }); + + const valueColumn = result.metadata.columnMetadata?.find( + (col) => col.name === "value", + ); + expect(valueColumn?.detectedType).toBe("number"); + // Should NOT trigger mixed_types warning since it's all numeric + const mixedWarning = result.metadata.dataQualityWarnings?.find( + (w) => w.type === "mixed_types" && w.column === "value", + ); + expect(mixedWarning).toBeUndefined(); + }); + + it("should detect string column type", async () => { + const csvData = Buffer.from("name,city\nAlice,New York\nBob,Los Angeles"); + const result = await CSVProcessor.process(csvData, { + formatStyle: "raw", + }); + + const nameColumn = result.metadata.columnMetadata?.find( + (col) => col.name === "name", + ); + expect(nameColumn?.detectedType).toBe("string"); + }); + + it("should detect date column type with format", async () => { + const csvData = Buffer.from( + "date,event\n2024-01-15,Meeting\n2024-02-20,Conference", + ); + const result = await CSVProcessor.process(csvData, { + formatStyle: "raw", + }); + + const dateColumn = result.metadata.columnMetadata?.find( + (col) => col.name === "date", + ); + expect(dateColumn?.detectedType).toBe("date"); + expect(dateColumn?.dateFormat).toBe("YYYY-MM-DD"); + }); + + it("should detect email column type", async () => { + const csvData = Buffer.from( + "email,name\nalice@example.com,Alice\nbob@test.org,Bob", + ); + const result = await CSVProcessor.process(csvData, { + formatStyle: "raw", + }); + + const emailColumn = result.metadata.columnMetadata?.find( + (col) => col.name === "email", + ); + expect(emailColumn?.detectedType).toBe("email"); + }); + + it("should detect boolean column type", async () => { + const csvData = Buffer.from( + "active,name\ntrue,Alice\nfalse,Bob\nyes,Charlie", + ); + const result = await CSVProcessor.process(csvData, { + formatStyle: "raw", + }); + + const activeColumn = result.metadata.columnMetadata?.find( + (col) => col.name === "active", + ); + expect(activeColumn?.detectedType).toBe("boolean"); + }); + + it("should detect mixed types with low confidence", async () => { + const csvData = Buffer.from("value\n123\nhello\n456\nworld\ntrue"); + const result = await CSVProcessor.process(csvData, { + formatStyle: "raw", + }); + + const valueColumn = result.metadata.columnMetadata?.find( + (col) => col.name === "value", + ); + expect(valueColumn?.detectedType).toBe("mixed"); + expect(valueColumn?.typeConfidence).toBeLessThan(70); + }); + + it("should calculate null count correctly", async () => { + const csvData = Buffer.from( + "name,value\nAlice,10\nBob,\nCharlie,30\nDiana,", + ); + const result = await CSVProcessor.process(csvData, { + formatStyle: "raw", + }); + + const valueColumn = result.metadata.columnMetadata?.find( + (col) => col.name === "value", + ); + expect(valueColumn?.nullCount).toBe(2); + }); + + it("should calculate unique count correctly", async () => { + const csvData = Buffer.from( + "status\nactive\npending\nactive\nclosed\nactive", + ); + const result = await CSVProcessor.process(csvData, { + formatStyle: "raw", + }); + + const statusColumn = result.metadata.columnMetadata?.find( + (col) => col.name === "status", + ); + expect(statusColumn?.uniqueCount).toBe(3); // active, pending, closed + }); + + it("should include sample values", async () => { + const csvData = Buffer.from( + "color\nred\nblue\ngreen\nyellow\norange\npurple", + ); + const result = await CSVProcessor.process(csvData, { + formatStyle: "raw", + }); + + const colorColumn = result.metadata.columnMetadata?.find( + (col) => col.name === "color", + ); + expect(colorColumn?.sampleValues).toBeDefined(); + expect(colorColumn?.sampleValues?.length).toBeLessThanOrEqual(5); + }); + + it("should detect column name issues", async () => { + const csvData = Buffer.from(" name with spaces ,123starts\nAlice,test"); + const result = await CSVProcessor.process(csvData, { + formatStyle: "raw", + }); + + const col1 = result.metadata.columnMetadata?.find( + (col) => col.name === " name with spaces ", + ); + const col2 = result.metadata.columnMetadata?.find( + (col) => col.name === "123starts", + ); + + expect(col1?.nameIssues).toContain("Leading or trailing whitespace"); + expect(col2?.nameIssues).toContain("Starts with a number"); + }); + }); + + describe("Data quality warnings", () => { + it("should warn about high null rate", async () => { + const csvData = Buffer.from( + "name,value\nAlice,\nBob,\nCharlie,\nDiana,10\nEve,", + ); + const result = await CSVProcessor.process(csvData, { + formatStyle: "raw", + }); + + expect(result.metadata.dataQualityWarnings).toBeDefined(); + const nullWarning = result.metadata.dataQualityWarnings?.find( + (w) => w.type === "high_null_rate" && w.column === "value", + ); + expect(nullWarning).toBeDefined(); + expect(nullWarning?.message).toContain("empty/null values"); + }); + + it("should warn about mixed types", async () => { + const csvData = Buffer.from("data\n123\nhello\n456\nworld\ntrue"); + const result = await CSVProcessor.process(csvData, { + formatStyle: "raw", + }); + + const mixedWarning = result.metadata.dataQualityWarnings?.find( + (w) => w.type === "mixed_types", + ); + expect(mixedWarning).toBeDefined(); + expect(mixedWarning?.severity).toBe("warning"); + }); + + it("should warn about duplicate values (all same)", async () => { + const csvData = Buffer.from( + "status\nactive\nactive\nactive\nactive\nactive\nactive\nactive\nactive\nactive\nactive\nactive", + ); + const result = await CSVProcessor.process(csvData, { + formatStyle: "raw", + }); + + const dupWarning = result.metadata.dataQualityWarnings?.find( + (w) => w.type === "duplicates", + ); + expect(dupWarning).toBeDefined(); + expect(dupWarning?.message).toContain("same value"); + }); + + it("should warn about entirely empty columns", async () => { + const csvData = Buffer.from("name,empty_col\nAlice,\nBob,\nCharlie,"); + const result = await CSVProcessor.process(csvData, { + formatStyle: "raw", + }); + + const emptyWarning = result.metadata.dataQualityWarnings?.find( + (w) => w.type === "empty_values" && w.column === "empty_col", + ); + expect(emptyWarning).toBeDefined(); + expect(emptyWarning?.message).toContain("entirely empty"); + }); + + it("should warn about invalid column names", async () => { + const csvData = Buffer.from("123invalid,name\ntest,Alice"); + const result = await CSVProcessor.process(csvData, { + formatStyle: "raw", + }); + + const nameWarning = result.metadata.dataQualityWarnings?.find( + (w) => w.type === "invalid_name", + ); + expect(nameWarning).toBeDefined(); + expect(nameWarning?.column).toBe("123invalid"); + }); + }); + + describe("Data quality score", () => { + it("should return high score for clean data", async () => { + const csvData = Buffer.from( + "id,name,age\n1,Alice,30\n2,Bob,25\n3,Charlie,35", + ); + const result = await CSVProcessor.process(csvData, { + formatStyle: "raw", + }); + + expect(result.metadata.dataQualityScore).toBeGreaterThanOrEqual(80); + }); + + it("should return lower score for data with issues", async () => { + const csvData = Buffer.from( + "id,value,empty\n1,,\n2,hello,\n3,123,\n4,,\n5,true,", + ); + const result = await CSVProcessor.process(csvData, { + formatStyle: "raw", + }); + + expect(result.metadata.dataQualityScore).toBeLessThan(80); + }); + + it("should return 0 for empty CSV", async () => { + const csvData = Buffer.from("name,value"); + const result = await CSVProcessor.process(csvData, { + formatStyle: "raw", + }); + + expect(result.metadata.dataQualityScore).toBe(0); + }); + + it("should include dataQualityScore in metadata", async () => { + const csvData = Buffer.from("name,age\nAlice,30\nBob,25"); + const result = await CSVProcessor.process(csvData, { + formatStyle: "raw", + }); + + expect(result.metadata.dataQualityScore).toBeDefined(); + expect(typeof result.metadata.dataQualityScore).toBe("number"); + }); + }); + + describe("Header detection", () => { + it("should detect headers when first row is text and data has numbers", async () => { + const csvData = Buffer.from("name,age,score\nAlice,30,95.5\nBob,25,88.0"); + const result = await CSVProcessor.process(csvData, { + formatStyle: "json", + }); + + expect(result.metadata.hasHeaders).toBe(true); + }); + + it("should detect headers when column names are descriptive text", async () => { + const csvData = Buffer.from( + "first_name,last_name,email\nJohn,Doe,john@example.com\nJane,Smith,jane@test.org", + ); + const result = await CSVProcessor.process(csvData, { + formatStyle: "json", + }); + + expect(result.metadata.hasHeaders).toBe(true); + }); + + it("should detect no headers when first row looks like data", async () => { + // All rows are numeric data - no clear header + const csvData = Buffer.from("1,2,3\n4,5,6\n7,8,9"); + const result = await CSVProcessor.process(csvData, { + formatStyle: "json", + }); + + expect(result.metadata.hasHeaders).toBe(false); + }); + + it("should detect no headers when first row contains dates like data", async () => { + // All rows look like date/numeric data - no clear header + const csvData = Buffer.from( + "2024-01-01,100,50\n2024-01-02,200,75\n2024-01-03,150,60", + ); + const result = await CSVProcessor.process(csvData, { + formatStyle: "json", + }); + + expect(result.metadata.hasHeaders).toBe(false); + }); + + it("should detect headers in raw format", async () => { + const csvData = Buffer.from( + "product,price,quantity\nWidget,19.99,100\nGadget,29.99,50", + ); + const result = await CSVProcessor.process(csvData, { + formatStyle: "raw", + }); + + expect(result.metadata.hasHeaders).toBe(true); + }); + + it("should handle empty CSV for header detection", async () => { + const csvData = Buffer.from(""); + const result = await CSVProcessor.process(csvData, { + formatStyle: "raw", + }); + + expect(result.metadata.hasHeaders).toBe(false); + }); + }); });