@loaders.gl/csv
Version:
Framework-independent loader for CSV and DSV table formats
293 lines • 11.7 kB
JavaScript
// loaders.gl
// SPDX-License-Identifier: MIT
// Copyright (c) vis.gl contributors
import { log, toArrayBufferIterator } from '@loaders.gl/loader-utils';
import { AsyncQueue, deduceTableSchema, TableBatchBuilder, convertToArrayRow, convertToObjectRow } from '@loaders.gl/schema-utils';
import Papa from "./papaparse/papaparse.js";
import AsyncIteratorStreamer from "./papaparse/async-iterator-streamer.js";
import { CSVFormat } from "./csv-format.js";
// __VERSION__ is injected by babel-plugin-version-inline
// @ts-ignore TS2304: Cannot find name '__VERSION__'.
const VERSION = typeof "4.4.4" !== 'undefined' ? "4.4.4" : 'latest';
const DEFAULT_CSV_SHAPE = 'object-row-table';
export const CSVLoader = {
...CSVFormat,
dataType: null,
batchType: null,
version: VERSION,
parse: async (arrayBuffer, options) => parseCSV(new TextDecoder().decode(arrayBuffer), options),
parseText: (text, options) => parseCSV(text, options),
parseInBatches: parseCSVInBatches,
// @ts-ignore
// testText: null,
options: {
csv: {
shape: DEFAULT_CSV_SHAPE, // 'object-row-table'
optimizeMemoryUsage: false,
// CSV options
header: 'auto',
columnPrefix: 'column',
// delimiter: auto
// newline: auto
quoteChar: '"',
escapeChar: '"',
dynamicTyping: true,
comments: false,
skipEmptyLines: true,
// transform: null?
delimitersToGuess: [',', '\t', '|', ';']
// fastMode: auto
}
}
};
async function parseCSV(csvText, options) {
// Apps can call the parse method directly, so we apply default options here
const csvOptions = { ...CSVLoader.options.csv, ...options?.csv };
const firstRow = readFirstRow(csvText);
const header = csvOptions.header === 'auto' ? isHeaderRow(firstRow) : Boolean(csvOptions.header);
const parseWithHeader = header;
const papaparseConfig = {
// dynamicTyping: true,
...csvOptions,
header: parseWithHeader,
download: false, // We handle loading, no need for papaparse to do it for us
transformHeader: parseWithHeader ? duplicateColumnTransformer() : undefined,
error: (e) => {
throw new Error(e);
}
};
const result = Papa.parse(csvText, papaparseConfig);
const rows = result.data;
const headerRow = result.meta.fields || generateHeader(csvOptions.columnPrefix, firstRow.length);
const shape = csvOptions.shape || DEFAULT_CSV_SHAPE;
let table;
switch (shape) {
case 'object-row-table':
table = {
shape: 'object-row-table',
data: rows.map((row) => (Array.isArray(row) ? convertToObjectRow(row, headerRow) : row))
};
break;
case 'array-row-table':
table = {
shape: 'array-row-table',
data: rows.map((row) => (Array.isArray(row) ? row : convertToArrayRow(row, headerRow)))
};
break;
default:
throw new Error(shape);
}
table.schema = deduceTableSchema(table);
return table;
}
// TODO - support batch size 0 = no batching/single batch?
function parseCSVInBatches(asyncIterator, options) {
// Papaparse does not support standard batch size handling
// TODO - investigate papaparse chunks mode
options = { ...options };
if (options?.core?.batchSize === 'auto') {
options.core.batchSize = 4000;
}
// Apps can call the parse method directly, we so apply default options here
const csvOptions = { ...CSVLoader.options.csv, ...options?.csv };
const asyncQueue = new AsyncQueue();
let isFirstRow = true;
let headerRow = null;
let tableBatchBuilder = null;
let schema = null;
const config = {
// dynamicTyping: true, // Convert numbers and boolean values in rows from strings,
...csvOptions,
header: false, // Unfortunately, header detection is not automatic and does not infer shapes
download: false, // We handle loading, no need for papaparse to do it for us
// chunkSize is set to 5MB explicitly (same as Papaparse default) due to a bug where the
// streaming parser gets stuck if skipEmptyLines and a step callback are both supplied.
// See https://github.com/mholt/PapaParse/issues/465
chunkSize: 1024 * 1024 * 5,
// skipEmptyLines is set to a boolean value if supplied. Greedy is set to true
// skipEmptyLines is handled manually given two bugs where the streaming parser gets stuck if
// both of the skipEmptyLines and step callback options are provided:
// - true doesn't work unless chunkSize is set: https://github.com/mholt/PapaParse/issues/465
// - greedy doesn't work: https://github.com/mholt/PapaParse/issues/825
skipEmptyLines: false,
// step is called on every row
// eslint-disable-next-line complexity, max-statements
step(results) {
let row = results.data;
if (csvOptions.skipEmptyLines) {
// Manually reject lines that are empty
const collapsedRow = row.flat().join('').trim();
if (collapsedRow === '') {
return;
}
}
const bytesUsed = results.meta.cursor;
// Check if we need to save a header row
if (isFirstRow && !headerRow) {
// Auto detects or can be forced with csvOptions.header
const header = csvOptions.header === 'auto' ? isHeaderRow(row) : Boolean(csvOptions.header);
if (header) {
headerRow = row.map(duplicateColumnTransformer());
return;
}
}
// If first data row, we can deduce the schema
if (isFirstRow) {
isFirstRow = false;
if (!headerRow) {
headerRow = generateHeader(csvOptions.columnPrefix, row.length);
}
schema = deduceCSVSchema(row, headerRow);
}
if (csvOptions.optimizeMemoryUsage) {
// A workaround to allocate new strings and don't retain pointers to original strings.
// https://bugs.chromium.org/p/v8/issues/detail?id=2869
row = JSON.parse(JSON.stringify(row));
}
const shape = options?.shape || csvOptions.shape || DEFAULT_CSV_SHAPE;
// Add the row
tableBatchBuilder =
tableBatchBuilder ||
new TableBatchBuilder(
// @ts-expect-error TODO this is not a proper schema
schema, {
shape,
...(options?.core || {})
});
try {
tableBatchBuilder.addRow(row);
// If a batch has been completed, emit it
const batch = tableBatchBuilder && tableBatchBuilder.getFullBatch({ bytesUsed });
if (batch) {
asyncQueue.enqueue(batch);
}
}
catch (error) {
asyncQueue.enqueue(error);
}
},
// complete is called when all rows have been read
complete(results) {
try {
const bytesUsed = results.meta.cursor;
// Ensure any final (partial) batch gets emitted
const batch = tableBatchBuilder && tableBatchBuilder.getFinalBatch({ bytesUsed });
if (batch) {
asyncQueue.enqueue(batch);
}
}
catch (error) {
asyncQueue.enqueue(error);
}
asyncQueue.close();
}
};
Papa.parse(toArrayBufferIterator(asyncIterator), config, AsyncIteratorStreamer);
// TODO - Does it matter if we return asyncIterable or asyncIterator
// return asyncQueue[Symbol.asyncIterator]();
return asyncQueue;
}
/**
* Checks if a certain row is a header row
* @param row the row to check
* @returns true if the row looks like a header
*/
function isHeaderRow(row) {
return row && row.every((value) => typeof value === 'string');
}
/**
* Reads, parses, and returns the first row of a CSV text
* @param csvText the csv text to parse
* @returns the first row
*/
function readFirstRow(csvText) {
const result = Papa.parse(csvText, {
dynamicTyping: true,
preview: 1
});
return result.data[0];
}
/**
* Creates a transformer that renames duplicate columns. This is needed as Papaparse doesn't handle
* duplicate header columns and would use the latest occurrence by default.
* See the header option in https://www.papaparse.com/docs#config
* @returns a transform function that returns sanitized names for duplicate fields
*/
function duplicateColumnTransformer() {
const observedColumns = new Set();
return (col) => {
let colName = col;
let counter = 1;
while (observedColumns.has(colName)) {
colName = `${col}.${counter}`;
counter++;
}
observedColumns.add(colName);
return colName;
};
}
/**
* Generates the header of a CSV given a prefix and a column count
* @param columnPrefix the columnPrefix to use
* @param count the count of column names to generate
* @returns an array of column names
*/
function generateHeader(columnPrefix, count = 0) {
const headers = [];
for (let i = 0; i < count; i++) {
headers.push(`${columnPrefix}${i + 1}`);
}
return headers;
}
function deduceCSVSchema(row, headerRow) {
const fields = [];
for (let i = 0; i < row.length; i++) {
const columnName = (headerRow && headerRow[i]) || i;
const value = row[i];
switch (typeof value) {
case 'number':
fields.push({ name: String(columnName), type: 'float64', nullable: true });
break;
case 'boolean':
fields.push({ name: String(columnName), type: 'bool', nullable: true });
break;
case 'string':
fields.push({ name: String(columnName), type: 'utf8', nullable: true });
break;
default:
log.warn(`CSV: Unknown column type: ${typeof value}`)();
fields.push({ name: String(columnName), type: 'utf8', nullable: true });
}
}
return {
fields,
metadata: {
'loaders.gl#format': 'csv',
'loaders.gl#loader': 'CSVLoader'
}
};
}
// TODO - remove
// type ObjectField = {name: string; index: number; type: any};
// type ObjectSchema = {[key: string]: ObjectField} | ObjectField[];
// function deduceObjectSchema(row, headerRow): ObjectSchema {
// const schema: ObjectSchema = headerRow ? {} : [];
// for (let i = 0; i < row.length; i++) {
// const columnName = (headerRow && headerRow[i]) || i;
// const value = row[i];
// switch (typeof value) {
// case 'number':
// case 'boolean':
// // TODO - booleans could be handled differently...
// schema[columnName] = {name: String(columnName), index: i, type: Float32Array};
// break;
// case 'string':
// default:
// schema[columnName] = {name: String(columnName), index: i, type: Array};
// // We currently only handle numeric rows
// // TODO we could offer a function to map strings to numbers?
// }
// }
// return schema;
// }
//# sourceMappingURL=csv-loader.js.map