UNPKG

datapilot-cli

Version:

Enterprise-grade streaming multi-format data analysis with comprehensive statistical insights and intelligent relationship detection - supports CSV, JSON, Excel, TSV, Parquet - memory-efficient, cross-platform

247 lines 9.46 kB
"use strict"; /** * Parsing Metadata Tracker - Enhanced CSV parsing with detailed analytics * Wraps our CSV parser with confidence scoring and method documentation */ Object.defineProperty(exports, "__esModule", { value: true }); exports.ParsingMetadataTracker = void 0; const csv_parser_1 = require("../../parsers/csv-parser"); const csv_detector_1 = require("../../parsers/csv-detector"); const encoding_detector_1 = require("../../parsers/encoding-detector"); const fs_1 = require("fs"); class ParsingMetadataTracker { warnings = []; parser; constructor(_config) { this.parser = new csv_parser_1.CSVParser({ autoDetect: true }); } /** * Perform enhanced CSV parsing with detailed metadata collection */ async parseWithMetadata(filePath) { const startTime = Date.now(); try { // Sample file for detection analysis const { sample, sampleStats } = await this.createFileSample(filePath); // Enhanced detection with confidence scoring const encodingDetection = this.analyzeEncoding(sample); const delimiterDetection = this.analyzeDelimiter(sample); // Parse the file const rows = await this.parser.parseFile(filePath); const parseEndTime = Date.now(); // Get parser options const parserOptions = this.parser.getOptions(); // Analyze header detection const headerAnalysis = this.analyzeHeaderDetection(rows); // Count empty lines const emptyLinesCount = this.countEmptyLines(sample); const metadata = { dataSourceType: 'Local File System', parsingEngine: `DataPilot Advanced CSV Parser v${this.getParserVersion()}`, parsingTimeSeconds: Number(((parseEndTime - startTime) / 1000).toFixed(3)), encoding: encodingDetection, delimiter: delimiterDetection, lineEnding: this.detectLineEndings(sample), quotingCharacter: parserOptions.quote || 'None Detected', emptyLinesEncountered: emptyLinesCount, headerProcessing: headerAnalysis, initialScanLimit: sampleStats, }; // Add performance warnings this.addPerformanceWarnings(parseEndTime - startTime, rows.length); return { rows, metadata }; } catch (error) { const message = error instanceof Error ? error.message : 'Unknown parsing error'; throw new Error(`Enhanced parsing failed: ${message}`); } } /** * Create optimized file sample for analysis */ async createFileSample(filePath) { const maxSampleSize = 1024 * 1024; // 1MB const maxLines = 1000; try { // Read sample from beginning of file const buffer = (0, fs_1.readFileSync)(filePath); const sampleSize = Math.min(buffer.length, maxSampleSize); const sample = buffer.slice(0, sampleSize); // Count lines in sample const lineCount = (sample.toString('utf8').match(/\n/g) || []).length; const actualLines = Math.min(lineCount, maxLines); return { sample, sampleStats: { method: `First ${sampleSize} bytes or ${maxLines} lines`, linesScanned: actualLines, bytesScanned: sampleSize, }, }; } catch (error) { throw new Error(`Failed to create file sample: ${error instanceof Error ? error.message : 'Unknown error'}`); } } /** * Enhanced encoding detection with confidence analysis */ analyzeEncoding(sample) { const result = encoding_detector_1.EncodingDetector.detect(sample); return { encoding: result.encoding, detectionMethod: result.hasBOM ? 'Byte Order Mark (BOM) Detection' : 'Statistical Character Pattern Analysis', confidence: Math.round(result.confidence * 100), bomDetected: result.hasBOM, bomType: result.hasBOM ? this.identifyBomType(sample) : undefined, }; } /** * Enhanced delimiter detection with alternatives */ analyzeDelimiter(sample) { const detected = csv_detector_1.CSVDetector.detect(sample); const alternatives = this.getDelimiterAlternatives(sample.toString('utf8')); return { delimiter: detected.delimiter, detectionMethod: 'Character Frequency Analysis with Field Consistency Scoring', confidence: Math.round(detected.delimiterConfidence * 100), alternativesConsidered: alternatives, }; } /** * Analyze delimiter alternatives with scoring */ getDelimiterAlternatives(text) { const delimiters = [',', '\t', ';', '|', ':']; const lines = text.split('\n').slice(0, 20); return delimiters .map((delimiter) => { const fieldCounts = lines.map((line) => line.split(delimiter).length); const consistency = this.calculateConsistencyScore(fieldCounts); const frequency = (text.match(new RegExp(delimiter.replace(/[.*+?^${}()|[\]\\]/g, '\\$&'), 'g')) || []).length; return { delimiter: delimiter === '\t' ? 'TAB' : delimiter, score: Math.round((consistency * 0.7 + Math.min(frequency / 100, 1) * 0.3) * 100), }; }) .sort((a, b) => b.score - a.score); } /** * Calculate consistency score for field counts */ calculateConsistencyScore(counts) { if (counts.length === 0) return 0; const mean = counts.reduce((sum, c) => sum + c, 0) / counts.length; const variance = counts.reduce((sum, c) => sum + Math.pow(c - mean, 2), 0) / counts.length; // Lower variance = higher consistency return Math.max(0, 1 - variance / mean); } /** * Detect line ending format */ detectLineEndings(sample) { const text = sample.toString('utf8', 0, Math.min(sample.length, 10000)); const crlfCount = (text.match(/\r\n/g) || []).length; const lfCount = (text.match(/(?<!\r)\n/g) || []).length; return crlfCount > lfCount ? 'CRLF' : 'LF'; } /** * Analyze header detection with confidence */ analyzeHeaderDetection(rows) { if (rows.length === 0) { return { headerPresence: 'Not Detected', headerRowNumbers: [], columnNamesSource: 'None - empty dataset', }; } const parserOptions = this.parser.getOptions(); if (parserOptions.hasHeader) { return { headerPresence: 'Detected', headerRowNumbers: [1], columnNamesSource: 'First row interpreted as column headers', }; } else { return { headerPresence: 'Not Detected', headerRowNumbers: [], columnNamesSource: 'Generated column indices (Col_0, Col_1, etc.)', }; } } /** * Count empty lines in sample */ countEmptyLines(sample) { const text = sample.toString('utf8'); const lines = text.split(/\r?\n/); return lines.filter((line) => line.trim().length === 0).length; } /** * Identify BOM type */ identifyBomType(sample) { if (sample.length >= 3 && sample[0] === 0xef && sample[1] === 0xbb && sample[2] === 0xbf) { return 'UTF-8 BOM'; } if (sample.length >= 2 && sample[0] === 0xff && sample[1] === 0xfe) { return 'UTF-16LE BOM'; } if (sample.length >= 2 && sample[0] === 0xfe && sample[1] === 0xff) { return 'UTF-16BE BOM'; } return 'Unknown BOM'; } /** * Add performance-related warnings */ addPerformanceWarnings(parseTimeMs, rowCount) { const parseTimeSeconds = parseTimeMs / 1000; const rowsPerSecond = rowCount / parseTimeSeconds; if (parseTimeSeconds > 30) { this.warnings.push({ category: 'parsing', severity: 'medium', message: `Parsing took ${parseTimeSeconds.toFixed(1)} seconds`, impact: 'Longer analysis time', suggestion: 'Consider using sampling for very large datasets', }); } if (rowsPerSecond < 1000) { this.warnings.push({ category: 'parsing', severity: 'low', message: `Processing rate: ${Math.round(rowsPerSecond)} rows/second`, impact: 'Slower than optimal performance', }); } } /** * Get parser version */ getParserVersion() { // In a real implementation, this would come from package.json return '1.0.0'; } /** * Get collected warnings */ getWarnings() { return [...this.warnings]; } /** * Clear warnings */ clearWarnings() { this.warnings = []; } } exports.ParsingMetadataTracker = ParsingMetadataTracker; //# sourceMappingURL=parsing-metadata-tracker.js.map