From 667f14aa75b8495bd35f74f666a81307162d58e9 Mon Sep 17 00:00:00 2001 From: x0Lazarus <113273587+x0Lazarus@users.noreply.github.com> Date: Sat, 26 Sep 2026 11:44:41 -0700 Subject: [PATCH 1/3] fix(csv-parse): accept stream options in TypeScript --- packages/csv-parse/dist/cjs/index.d.cts | 11 ++++----- packages/csv-parse/dist/esm/index.d.ts | 11 ++++----- packages/csv-parse/lib/index.d.ts | 11 ++++----- packages/csv-parse/test/api.types.ts | 33 +++++++++++++++++++++++-- 4 files changed, 46 insertions(+), 20 deletions(-) diff --git a/packages/csv-parse/dist/cjs/index.d.cts b/packages/csv-parse/dist/cjs/index.d.cts index 1a524cc7..4e87a4dc 100644 --- a/packages/csv-parse/dist/cjs/index.d.cts +++ b/packages/csv-parse/dist/cjs/index.d.cts @@ -297,12 +297,11 @@ export interface OptionsNormalized { trim: boolean; } -/* -Note, could not `extends stream.TransformOptions` because encoding can be -BufferEncoding and undefined as well as null which is not defined in the -extended type. -*/ -export interface Options { +// Keep the parser's encoding options instead of the narrower stream encoding. +export interface Options extends Omit< + stream.TransformOptions, + "encoding" +> { /** * If true, the parser will attempt to convert read data types to native types. * @deprecated Use {@link cast} diff --git a/packages/csv-parse/dist/esm/index.d.ts b/packages/csv-parse/dist/esm/index.d.ts index 1a524cc7..4e87a4dc 100644 --- a/packages/csv-parse/dist/esm/index.d.ts +++ b/packages/csv-parse/dist/esm/index.d.ts @@ -297,12 +297,11 @@ export interface OptionsNormalized { trim: boolean; } -/* -Note, could not `extends stream.TransformOptions` because encoding can be -BufferEncoding and undefined as well as null which is not defined in the -extended type. -*/ -export interface Options { +// Keep the parser's encoding options instead of the narrower stream encoding. +export interface Options extends Omit< + stream.TransformOptions, + "encoding" +> { /** * If true, the parser will attempt to convert read data types to native types. * @deprecated Use {@link cast} diff --git a/packages/csv-parse/lib/index.d.ts b/packages/csv-parse/lib/index.d.ts index 1a524cc7..4e87a4dc 100644 --- a/packages/csv-parse/lib/index.d.ts +++ b/packages/csv-parse/lib/index.d.ts @@ -297,12 +297,11 @@ export interface OptionsNormalized { trim: boolean; } -/* -Note, could not `extends stream.TransformOptions` because encoding can be -BufferEncoding and undefined as well as null which is not defined in the -extended type. -*/ -export interface Options { +// Keep the parser's encoding options instead of the narrower stream encoding. +export interface Options extends Omit< + stream.TransformOptions, + "encoding" +> { /** * If true, the parser will attempt to convert read data types to native types. * @deprecated Use {@link cast} diff --git a/packages/csv-parse/test/api.types.ts b/packages/csv-parse/test/api.types.ts index 521d565b..6e08ff3a 100644 --- a/packages/csv-parse/test/api.types.ts +++ b/packages/csv-parse/test/api.types.ts @@ -1,11 +1,40 @@ import "should"; -import { parse, CsvError, normalize_options } from "../lib/index.js"; -import type { Info, InfoField, Options, Parser } from "../lib/index.js"; +import { parse, CsvError, normalize_options, Parser } from "../lib/index.js"; +import type { Info, InfoField, Options } from "../lib/index.js"; describe("API Types", function () { type Person = { name: string; age: number }; describe("stream/callback API", function () { + it("accepts stream buffer options in the constructor", function () { + const parser = new Parser({ + readableHighWaterMark: 2, + writableHighWaterMark: 32, + }); + parser.readableHighWaterMark.should.eql(2); + parser.writableHighWaterMark.should.eql(32); + parser.destroy(); + }); + + it("accepts stream options with a callback", function (next) { + parse("a,b\n", { highWaterMark: 16 }, (error, records) => { + if (error) return next(error); + records.should.eql([["a", "b"]]); + next(); + }); + }); + + for (const encoding of [null, false]) { + it(`keeps encoding ${encoding} with stream options`, function (next) { + const options: Options = { highWaterMark: 1, encoding }; + parse("a,b\n", options, (error, records) => { + if (error) return next(error); + records.should.eql([[Buffer.from("a"), Buffer.from("b")]]); + next(); + }); + }); + } + it("respect parse signature", function () { // No argument parse(); From 86fff09653f05c4dffd2c404be47c845b429a84c Mon Sep 17 00:00:00 2001 From: David Worms Date: Wed, 30 Sep 2026 16:53:57 +0200 Subject: [PATCH 2/3] refactor(csv-parse): isolate error and option types --- .../csv-parse/dist/cjs/api/CsvError.d.cts | 35 ++ packages/csv-parse/dist/cjs/index.d.cts | 530 +----------------- packages/csv-parse/dist/cjs/options.d.cts | 484 ++++++++++++++++ packages/csv-parse/dist/esm/api/CsvError.d.ts | 35 ++ packages/csv-parse/dist/esm/index.d.ts | 530 +----------------- packages/csv-parse/dist/esm/options.d.ts | 484 ++++++++++++++++ packages/csv-parse/lib/api/CsvError.d.ts | 35 ++ packages/csv-parse/lib/index.d.ts | 530 +----------------- packages/csv-parse/lib/options.d.ts | 484 ++++++++++++++++ packages/csv-parse/package.json | 2 +- 10 files changed, 1597 insertions(+), 1552 deletions(-) create mode 100644 packages/csv-parse/dist/cjs/api/CsvError.d.cts create mode 100644 packages/csv-parse/dist/cjs/options.d.cts create mode 100644 packages/csv-parse/dist/esm/api/CsvError.d.ts create mode 100644 packages/csv-parse/dist/esm/options.d.ts create mode 100644 packages/csv-parse/lib/api/CsvError.d.ts create mode 100644 packages/csv-parse/lib/options.d.ts diff --git a/packages/csv-parse/dist/cjs/api/CsvError.d.cts b/packages/csv-parse/dist/cjs/api/CsvError.d.cts new file mode 100644 index 00000000..f86ed330 --- /dev/null +++ b/packages/csv-parse/dist/cjs/api/CsvError.d.cts @@ -0,0 +1,35 @@ +import type { OptionsNormalized } from "../options.cjs"; + +export type CsvErrorCode = + | "CSV_INVALID_ARGUMENT" + | "CSV_INVALID_CLOSING_QUOTE" + | "CSV_INVALID_COLUMN_DEFINITION" + | "CSV_INVALID_COLUMN_MAPPING" + | "CSV_INVALID_OPTION_BOM" + | "CSV_INVALID_OPTION_CAST" + | "CSV_INVALID_OPTION_CAST_DATE" + | "CSV_INVALID_OPTION_COLUMNS" + | "CSV_INVALID_OPTION_COMMENT" + | "CSV_INVALID_OPTION_DELIMITER" + | "CSV_INVALID_OPTION_GROUP_COLUMNS_BY_NAME" + | "CSV_INVALID_OPTION_ON_RECORD" + | "CSV_MAX_RECORD_SIZE" + | "CSV_NON_TRIMABLE_CHAR_AFTER_CLOSING_QUOTE" + | "CSV_OPTION_COLUMNS_MISSING_NAME" + | "CSV_QUOTE_NOT_CLOSED" + | "CSV_RECORD_INCONSISTENT_FIELDS_LENGTH" + | "CSV_RECORD_INCONSISTENT_COLUMNS" + | "CSV_UNKNOWN_ERROR" + | "INVALID_OPENING_QUOTE"; + +export class CsvError extends Error { + readonly code: CsvErrorCode; + [key: string]: unknown; + + constructor( + code: CsvErrorCode, + message: string | string[], + options?: OptionsNormalized, + ...contexts: unknown[] + ); +} diff --git a/packages/csv-parse/dist/cjs/index.d.cts b/packages/csv-parse/dist/cjs/index.d.cts index 4e87a4dc..17f50217 100644 --- a/packages/csv-parse/dist/cjs/index.d.cts +++ b/packages/csv-parse/dist/cjs/index.d.cts @@ -3,6 +3,19 @@ /// import * as stream from "stream"; +import type { CsvError } from "./api/CsvError.cjs"; +import type { + // Info + Info, + InfoCallback, + // Options + OptionsWithColumns, + OptionsNormalized, + Options, +} from "./options.cjs"; + +export * from "./api/CsvError.cjs"; +export type * from "./options.cjs"; export type Callback = ( err: CsvError | undefined, @@ -23,487 +36,6 @@ export class Parser extends stream.Transform { readonly info: Info; } -export interface Info { - /** - * The number of processed bytes. - */ - readonly bytes: number; - /** - * The number of processed bytes until the last successfully parsed and emitted records. - */ - readonly bytes_records: number; - /** - * The number of lines being fully commented. - */ - readonly comment_lines: number; - /** - * The number of processed empty lines. - */ - readonly empty_lines: number; - /** - * The number of non uniform records when `relax_column_count` is true. - */ - readonly invalid_field_length: number; - /** - * The number of lines encountered in the source dataset, start at 1 for the first line. - */ - readonly lines: number; - /** - * The number of processed records. - */ - readonly records: number; -} - -export interface InfoCallback extends Info { - /** - * Normalized version of `options.columns` when `options.columns` is true, boolean otherwise. - */ - readonly columns: boolean | { name: string }[] | { disabled: true }[]; -} - -export interface InfoDataSet extends Info { - readonly column: number | string; -} - -export interface InfoRecord extends InfoDataSet { - readonly error: CsvError; - readonly header: boolean; - readonly index: number; - readonly raw: string | undefined; -} - -export interface InfoField extends InfoRecord { - readonly quoting: boolean; -} - -/** - * @deprecated Use the InfoField interface instead, the interface will disappear in future versions. - */ -// eslint-disable-next-line -export interface CastingContext extends InfoField {} - -export type CastingFunction = (value: string, context: InfoField) => unknown; - -export type CastingDateFunction = (value: string, context: InfoField) => Date; - -export type ColumnOption = - K | undefined | null | false | { name: K }; - -type ColumnKey = T extends string[] - ? string - : unknown extends T - ? string - : string | keyof T; - -// Keep columns from overriding record types inferred from options such as raw. -type NoInferColumnRecord = [T][T extends unknown ? 0 : never]; - -export interface ScoringFunctionInfo { - /** - * The character code of the delimiter candidate being scored. - */ - readonly char_code: number; - /** - * The number of occurrences of the candidate in each line. - */ - readonly lines: number[]; - /** - * Whether the candidate is listed in the `preferred` option. - */ - readonly preferred: boolean; - /** - * The standard deviation of the occurrences across the lines. - */ - readonly std: number; - /** - * The total number of occurrences of the candidate. - */ - readonly total: number; -} - -export type ScoringFunction = ( - info: ScoringFunctionInfo, - options: ScoringFunctionOptions, -) => number; - -export interface ScoringFunctionOptions { - preferred: Record; - score: ScoringFunction; - size: number; -} - -export interface OptionsNormalized { - auto_parse?: boolean | CastingFunction; - auto_parse_date?: boolean | CastingDateFunction; - /** - * If true, detect and exclude the byte order mark (BOM) from the CSV input if present. - */ - bom?: boolean; - /** - * If true, the parser will attempt to convert input string to native types. - * If a function, receive the value as first argument, a context as second argument and return a new value. More information about the context properties is available below. - */ - cast?: boolean | CastingFunction; - /** - * If true, the parser will attempt to convert input string to dates. - * If a function, receive the value as argument and return a new value. It requires the "auto_parse" option. Be careful, it relies on Date.parse. - */ - cast_date?: boolean | CastingDateFunction; - /** - * Internal property string the function to - */ - cast_first_line_to_header?: ( - record: string[], - ) => ColumnOption>[]; - /** - * List of fields as an array, a user defined callback accepting the first - * line and returning the column names or true if autodiscovered in the first - * CSV line, default to null, affect the result data set in the sense that - * records will be objects instead of arrays. The callback receives the raw - * header fields as strings, while returned names may use keys from the typed - * input record. - */ - columns: boolean | ColumnOption>[]; - /** - * Treat all the characters after this one as a comment, default to '' (disabled). - */ - comment: string | null; - /** - * Restrict the definition of comments to a full line. Comment characters - * defined in the middle of the line are not interpreted as such. The - * option require the activation of comments. - */ - comment_no_infix: boolean; - /** - * Set the field delimiter. One character only, defaults to comma. - */ - delimiter: Buffer[]; - /** - * Discover the field delimiter. - */ - delimiter_auto: ScoringFunctionOptions; - /** - * Set the source and destination encoding, a value of `null` returns buffer instead of strings. - */ - encoding: BufferEncoding | null; - /** - * Set the escape character, one character only, defaults to double quotes. - */ - escape: null | Buffer; - /** - * Start handling records from the requested number of records. - */ - from: number; - /** - * Start handling records from the requested line number. - */ - from_line: number; - /** - * Convert values into an array of values when columns are activated and - * when multiple columns of the same name are found. - */ - group_columns_by_name: boolean; - /** - * Don't interpret delimiters as such in the last field according to the number of fields calculated from the number of columns, the option require the presence of the `column` option when `true`. - */ - ignore_last_delimiters: boolean | number; - /** - * Generate two properties `info` and `record` where `info` is a snapshot of the info object at the time the record was created and `record` is the parsed array or object. - */ - info: boolean; - /** - * If true, ignore whitespace immediately following the delimiter (i.e. left-trim all fields), defaults to false. - * Does not remove whitespace in a quoted field. - */ - ltrim: boolean; - /** - * Maximum number of characters to be contained in the field and line buffers before an exception is raised, - * used to guard against a wrong delimiter or record_delimiter, - * default to 128000 characters. - */ - max_record_size: number; - /** - * Name of header-record title to name objects by. - */ - objname: number | string | undefined; - /** - * Alter and filter records by executing a user defined function. - */ - on_record?: (record: U, context: InfoRecord) => T | null | undefined; - /** - * Function called when an error occurred if the `skip_records_with_error` - * option is activated. - */ - on_skip?: (err: CsvError | undefined, raw: string | undefined) => undefined; - /** - * Optional character surrounding a field, one character only, defaults to double quotes. - */ - quote?: Buffer | null; - /** - * Generate two properties raw and row where raw is the original CSV row content and row is the parsed array or object. - */ - raw: boolean; - /** - * One or multiple characters used to delimit record rows; defaults to auto discovery if not provided. - * Supported auto discovery method are Linux ("\n"), Apple ("\r") and Windows ("\r\n") row delimiters. - */ - record_delimiter: Buffer[]; - /** - * Discard inconsistent columns count, default to false. - */ - relax_column_count: boolean; - /** - * Discard inconsistent columns count when the record contains less fields than expected, default to false. - */ - relax_column_count_less: boolean; - /** - * Discard inconsistent columns count when the record contains more fields than expected, default to false. - */ - relax_column_count_more: boolean; - /** - * Preserve quotes inside unquoted field. - */ - relax_quotes: boolean; - /** - * If true, ignore whitespace immediately preceding the delimiter (i.e. right-trim all fields), defaults to false. - * Does not remove whitespace in a quoted field. - */ - rtrim: boolean; - /** - * Dont generate empty values for empty lines. - * Defaults to false - */ - skip_empty_lines: boolean; - /** - * Don't generate records for lines containing empty column values (column matching /\s*\/), defaults to false. - */ - skip_records_with_empty_values: boolean; - /** - * Skip a line with error found inside and directly go process the next line. - */ - skip_records_with_error: boolean; - /** - * Stop handling records after the requested number of records. - */ - to: number; - /** - * Stop handling records after the requested line number. - */ - to_line: number; - /** - * If true, ignore whitespace immediately around the delimiter, defaults to false. - * Does not remove whitespace in a quoted field. - */ - trim: boolean; -} - -// Keep the parser's encoding options instead of the narrower stream encoding. -export interface Options extends Omit< - stream.TransformOptions, - "encoding" -> { - /** - * If true, the parser will attempt to convert read data types to native types. - * @deprecated Use {@link cast} - */ - auto_parse?: boolean | CastingFunction; - autoParse?: boolean | CastingFunction; - /** - * If true, the parser will attempt to convert read data types to dates. It requires the "auto_parse" option. - * @deprecated Use {@link cast_date} - */ - auto_parse_date?: boolean | CastingDateFunction; - autoParseDate?: boolean | CastingDateFunction; - /** - * If true, detect and exclude the byte order mark (BOM) from the CSV input if present. - */ - bom?: OptionsNormalized["bom"]; - /** - * If true, the parser will attempt to convert input string to native types. - * If a function, receive the value as first argument, a context as second argument and return a new value. More information about the context properties is available below. - */ - cast?: OptionsNormalized["cast"]; - /** - * If true, the parser will attempt to convert input string to dates. - * If a function, receive the value as argument and return a new value. It requires the "auto_parse" option. Be careful, it relies on Date.parse. - */ - cast_date?: OptionsNormalized["cast_date"]; - castDate?: OptionsNormalized["cast_date"]; - /** - * List of fields as an array, - * a user defined callback accepting the first line and returning the column names or true if autodiscovered in the first CSV line, - * default to null, - * affect the result data set in the sense that records will be objects instead of arrays. The callback receives the raw header fields as strings, while returned names may use keys from the typed input record. - */ - columns?: - | OptionsNormalized["columns"] - | ((record: string[]) => ColumnOption>[]); - /** - * Treat all the characters after this one as a comment, default to '' (disabled). - */ - comment?: OptionsNormalized["comment"] | boolean; - /** - * Restrict the definition of comments to a full line. Comment characters - * defined in the middle of the line are not interpreted as such. The - * option require the activation of comments. - */ - comment_no_infix?: OptionsNormalized["comment_no_infix"] | null; - /** - * Set the field delimiter. One character only, defaults to comma. - */ - delimiter?: OptionsNormalized["delimiter"] | string | string[] | Buffer; - /** - * Discover the field delimiter - */ - delimiter_auto?: boolean | Partial; - /** - * Set the source and destination encoding, a value of `null` returns buffer instead of strings. - */ - encoding?: OptionsNormalized["encoding"] | boolean | undefined; - /** - * Set the escape character, one character only, defaults to double quotes. - */ - escape?: OptionsNormalized["escape"] | string | boolean; - /** - * Start handling records from the requested number of records. - */ - from?: OptionsNormalized["from"] | string; - /** - * Start handling records from the requested line number. - */ - from_line?: OptionsNormalized["from_line"] | null | string; - fromLine?: OptionsNormalized["from_line"] | null | string; - /** - * Convert values into an array of values when columns are activated and - * when multiple columns of the same name are found. - */ - group_columns_by_name?: OptionsNormalized["group_columns_by_name"]; - groupColumnsByName?: OptionsNormalized["group_columns_by_name"]; - /** - * Don't interpret delimiters as such in the last field according to the number of fields calculated from the number of columns, the option require the presence of the `column` option when `true`. - */ - ignore_last_delimiters?: OptionsNormalized["ignore_last_delimiters"]; - /** - * Generate two properties `info` and `record` where `info` is a snapshot of the info object at the time the record was created and `record` is the parsed array or object. - */ - info?: OptionsNormalized["info"]; - /** - * If true, ignore whitespace immediately following the delimiter (i.e. left-trim all fields), defaults to false. - * Does not remove whitespace in a quoted field. - */ - ltrim?: OptionsNormalized["ltrim"] | null; - /** - * Maximum number of characters to be contained in the field and line buffers before an exception is raised, - * used to guard against a wrong delimiter or record_delimiter, - * default to 128000 characters. - */ - max_record_size?: OptionsNormalized["max_record_size"] | null | string; - maxRecordSize?: OptionsNormalized["max_record_size"]; - /** - * Name of header-record title to name objects by. - */ - objname?: OptionsNormalized["objname"] | Buffer | null; - /** - * Alter and filter records by executing a user defined function. - */ - on_record?: (record: U, context: InfoRecord) => T | null | undefined | U; - onRecord?: (record: U, context: InfoRecord) => T | null | undefined | U; - /** - * Function called when an error occurred if the `skip_records_with_error` - * option is activated. - */ - on_skip?: OptionsNormalized["on_skip"]; - onSkip?: OptionsNormalized["on_skip"]; - /** - * Optional character surrounding a field, one character only, defaults to double quotes. - */ - quote?: OptionsNormalized["quote"] | string | boolean; - /** - * Generate two properties raw and row where raw is the original CSV row content and row is the parsed array or object. - */ - raw?: OptionsNormalized["raw"] | null; - /** - * One or multiple characters used to delimit record rows; defaults to auto discovery if not provided. - * Supported auto discovery method are Linux ("\n"), Apple ("\r") and Windows ("\r\n") row delimiters. - */ - record_delimiter?: - | OptionsNormalized["record_delimiter"] - | string - | Buffer - | null - | (string | null)[]; - recordDelimiter?: - | OptionsNormalized["record_delimiter"] - | string - | Buffer - | null - | (string | null)[]; - /** - * Discard inconsistent columns count, default to false. - */ - relax_column_count?: OptionsNormalized["relax_column_count"] | null; - relaxColumnCount?: OptionsNormalized["relax_column_count"] | null; - /** - * Discard inconsistent columns count when the record contains less fields than expected, default to false. - */ - relax_column_count_less?: OptionsNormalized["relax_column_count_less"] | null; - relaxColumnCountLess?: OptionsNormalized["relax_column_count_less"] | null; - /** - * Discard inconsistent columns count when the record contains more fields than expected, default to false. - */ - relax_column_count_more?: OptionsNormalized["relax_column_count_more"] | null; - relaxColumnCountMore?: OptionsNormalized["relax_column_count_more"] | null; - /** - * Preserve quotes inside unquoted field. - */ - relax_quotes?: OptionsNormalized["relax_quotes"] | null; - relaxQuotes?: OptionsNormalized["relax_quotes"] | null; - /** - * If true, ignore whitespace immediately preceding the delimiter (i.e. right-trim all fields), defaults to false. - * Does not remove whitespace in a quoted field. - */ - rtrim?: OptionsNormalized["rtrim"] | null; - /** - * Dont generate empty values for empty lines. - * Defaults to false - */ - skip_empty_lines?: OptionsNormalized["skip_empty_lines"] | null; - skipEmptyLines?: OptionsNormalized["skip_empty_lines"] | null; - /** - * Don't generate records for lines containing empty column values (column matching /\s*\/), defaults to false. - */ - skip_records_with_empty_values?: - OptionsNormalized["skip_records_with_empty_values"] | null; - skipRecordsWithEmptyValues?: - OptionsNormalized["skip_records_with_empty_values"] | null; - /** - * Skip a line with error found inside and directly go process the next line. - */ - skip_records_with_error?: OptionsNormalized["skip_records_with_error"] | null; - skipRecordsWithError?: OptionsNormalized["skip_records_with_error"] | null; - /** - * Stop handling records after the requested number of records. - */ - to?: OptionsNormalized["to"] | null | string; - /** - * Stop handling records after the requested line number. - */ - to_line?: OptionsNormalized["to_line"] | null | string; - toLine?: OptionsNormalized["to_line"] | null | string; - /** - * If true, ignore whitespace immediately around the delimiter, defaults to false. - * Does not remove whitespace in a quoted field. - */ - trim?: OptionsNormalized["trim"] | null; -} - -export type OptionsWithColumns = Omit, "columns"> & { - columns: Exclude< - Options, NoInferColumnRecord>["columns"], - undefined | false - >; -}; - declare function parse( input: string | Buffer | Uint8Array, options: OptionsWithColumns, @@ -529,42 +61,6 @@ declare function parse(callback?: Callback): Parser; export { parse }; -/////////////////////////////////////////////////////// CsvError - -export type CsvErrorCode = - | "CSV_INVALID_ARGUMENT" - | "CSV_INVALID_CLOSING_QUOTE" - | "CSV_INVALID_COLUMN_DEFINITION" - | "CSV_INVALID_COLUMN_MAPPING" - | "CSV_INVALID_OPTION_BOM" - | "CSV_INVALID_OPTION_CAST" - | "CSV_INVALID_OPTION_CAST_DATE" - | "CSV_INVALID_OPTION_COLUMNS" - | "CSV_INVALID_OPTION_COMMENT" - | "CSV_INVALID_OPTION_DELIMITER" - | "CSV_INVALID_OPTION_GROUP_COLUMNS_BY_NAME" - | "CSV_INVALID_OPTION_ON_RECORD" - | "CSV_MAX_RECORD_SIZE" - | "CSV_NON_TRIMABLE_CHAR_AFTER_CLOSING_QUOTE" - | "CSV_OPTION_COLUMNS_MISSING_NAME" - | "CSV_QUOTE_NOT_CLOSED" - | "CSV_RECORD_INCONSISTENT_FIELDS_LENGTH" - | "CSV_RECORD_INCONSISTENT_COLUMNS" - | "CSV_UNKNOWN_ERROR" - | "INVALID_OPENING_QUOTE"; - -export class CsvError extends Error { - readonly code: CsvErrorCode; - [key: string]: unknown; - - constructor( - code: CsvErrorCode, - message: string | string[], - options?: OptionsNormalized, - ...contexts: unknown[] - ); -} - /////////////////////////////////////////////////////// normalize_options declare function normalize_options(opts: Options): OptionsNormalized; diff --git a/packages/csv-parse/dist/cjs/options.d.cts b/packages/csv-parse/dist/cjs/options.d.cts new file mode 100644 index 00000000..daf04a5c --- /dev/null +++ b/packages/csv-parse/dist/cjs/options.d.cts @@ -0,0 +1,484 @@ +import * as stream from "stream"; + +import { CsvError } from "./api/CsvError.cjs"; + +export interface Info { + /** + * The number of processed bytes. + */ + readonly bytes: number; + /** + * The number of processed bytes until the last successfully parsed and emitted records. + */ + readonly bytes_records: number; + /** + * The number of lines being fully commented. + */ + readonly comment_lines: number; + /** + * The number of processed empty lines. + */ + readonly empty_lines: number; + /** + * The number of non uniform records when `relax_column_count` is true. + */ + readonly invalid_field_length: number; + /** + * The number of lines encountered in the source dataset, start at 1 for the first line. + */ + readonly lines: number; + /** + * The number of processed records. + */ + readonly records: number; +} + +export interface InfoCallback extends Info { + /** + * Normalized version of `options.columns` when `options.columns` is true, boolean otherwise. + */ + readonly columns: boolean | { name: string }[] | { disabled: true }[]; +} + +export interface InfoDataSet extends Info { + readonly column: number | string; +} + +export interface InfoRecord extends InfoDataSet { + readonly error: CsvError; + readonly header: boolean; + readonly index: number; + readonly raw: string | undefined; +} + +export interface InfoField extends InfoRecord { + readonly quoting: boolean; +} + +/** + * @deprecated Use the InfoField interface instead, the interface will disappear in future versions. + */ +// eslint-disable-next-line +export interface CastingContext extends InfoField {} + +export type CastingFunction = (value: string, context: InfoField) => unknown; + +export type CastingDateFunction = (value: string, context: InfoField) => Date; + +export type ColumnOption = + K | undefined | null | false | { name: K }; + +type ColumnKey = T extends string[] + ? string + : unknown extends T + ? string + : string | keyof T; + +// Keep columns from overriding record types inferred from options such as raw. +type NoInferColumnRecord = [T][T extends unknown ? 0 : never]; + +export interface ScoringFunctionInfo { + /** + * The character code of the delimiter candidate being scored. + */ + readonly char_code: number; + /** + * The number of occurrences of the candidate in each line. + */ + readonly lines: number[]; + /** + * Whether the candidate is listed in the `preferred` option. + */ + readonly preferred: boolean; + /** + * The standard deviation of the occurrences across the lines. + */ + readonly std: number; + /** + * The total number of occurrences of the candidate. + */ + readonly total: number; +} + +export type ScoringFunction = ( + info: ScoringFunctionInfo, + options: ScoringFunctionOptions, +) => number; + +export interface ScoringFunctionOptions { + preferred: Record; + score: ScoringFunction; + size: number; +} + +export interface OptionsNormalized { + auto_parse?: boolean | CastingFunction; + auto_parse_date?: boolean | CastingDateFunction; + /** + * If true, detect and exclude the byte order mark (BOM) from the CSV input if present. + */ + bom?: boolean; + /** + * If true, the parser will attempt to convert input string to native types. + * If a function, receive the value as first argument, a context as second argument and return a new value. More information about the context properties is available below. + */ + cast?: boolean | CastingFunction; + /** + * If true, the parser will attempt to convert input string to dates. + * If a function, receive the value as argument and return a new value. It requires the "auto_parse" option. Be careful, it relies on Date.parse. + */ + cast_date?: boolean | CastingDateFunction; + /** + * Internal property string the function to + */ + cast_first_line_to_header?: ( + record: string[], + ) => ColumnOption>[]; + /** + * List of fields as an array, a user defined callback accepting the first + * line and returning the column names or true if autodiscovered in the first + * CSV line, default to null, affect the result data set in the sense that + * records will be objects instead of arrays. The callback receives the raw + * header fields as strings, while returned names may use keys from the typed + * input record. + */ + columns: boolean | ColumnOption>[]; + /** + * Treat all the characters after this one as a comment, default to '' (disabled). + */ + comment: string | null; + /** + * Restrict the definition of comments to a full line. Comment characters + * defined in the middle of the line are not interpreted as such. The + * option require the activation of comments. + */ + comment_no_infix: boolean; + /** + * Set the field delimiter. One character only, defaults to comma. + */ + delimiter: Buffer[]; + /** + * Discover the field delimiter. + */ + delimiter_auto: ScoringFunctionOptions; + /** + * Set the source and destination encoding, a value of `null` returns buffer instead of strings. + */ + encoding: BufferEncoding | null; + /** + * Set the escape character, one character only, defaults to double quotes. + */ + escape: null | Buffer; + /** + * Start handling records from the requested number of records. + */ + from: number; + /** + * Start handling records from the requested line number. + */ + from_line: number; + /** + * Convert values into an array of values when columns are activated and + * when multiple columns of the same name are found. + */ + group_columns_by_name: boolean; + /** + * Don't interpret delimiters as such in the last field according to the number of fields calculated from the number of columns, the option require the presence of the `column` option when `true`. + */ + ignore_last_delimiters: boolean | number; + /** + * Generate two properties `info` and `record` where `info` is a snapshot of the info object at the time the record was created and `record` is the parsed array or object. + */ + info: boolean; + /** + * If true, ignore whitespace immediately following the delimiter (i.e. left-trim all fields), defaults to false. + * Does not remove whitespace in a quoted field. + */ + ltrim: boolean; + /** + * Maximum number of characters to be contained in the field and line buffers before an exception is raised, + * used to guard against a wrong delimiter or record_delimiter, + * default to 128000 characters. + */ + max_record_size: number; + /** + * Name of header-record title to name objects by. + */ + objname: number | string | undefined; + /** + * Alter and filter records by executing a user defined function. + */ + on_record?: (record: U, context: InfoRecord) => T | null | undefined; + /** + * Function called when an error occurred if the `skip_records_with_error` + * option is activated. + */ + on_skip?: (err: CsvError | undefined, raw: string | undefined) => undefined; + /** + * Optional character surrounding a field, one character only, defaults to double quotes. + */ + quote?: Buffer | null; + /** + * Generate two properties raw and row where raw is the original CSV row content and row is the parsed array or object. + */ + raw: boolean; + /** + * One or multiple characters used to delimit record rows; defaults to auto discovery if not provided. + * Supported auto discovery method are Linux ("\n"), Apple ("\r") and Windows ("\r\n") row delimiters. + */ + record_delimiter: Buffer[]; + /** + * Discard inconsistent columns count, default to false. + */ + relax_column_count: boolean; + /** + * Discard inconsistent columns count when the record contains less fields than expected, default to false. + */ + relax_column_count_less: boolean; + /** + * Discard inconsistent columns count when the record contains more fields than expected, default to false. + */ + relax_column_count_more: boolean; + /** + * Preserve quotes inside unquoted field. + */ + relax_quotes: boolean; + /** + * If true, ignore whitespace immediately preceding the delimiter (i.e. right-trim all fields), defaults to false. + * Does not remove whitespace in a quoted field. + */ + rtrim: boolean; + /** + * Dont generate empty values for empty lines. + * Defaults to false + */ + skip_empty_lines: boolean; + /** + * Don't generate records for lines containing empty column values (column matching /\s*\/), defaults to false. + */ + skip_records_with_empty_values: boolean; + /** + * Skip a line with error found inside and directly go process the next line. + */ + skip_records_with_error: boolean; + /** + * Stop handling records after the requested number of records. + */ + to: number; + /** + * Stop handling records after the requested line number. + */ + to_line: number; + /** + * If true, ignore whitespace immediately around the delimiter, defaults to false. + * Does not remove whitespace in a quoted field. + */ + trim: boolean; +} + +// Keep the parser's encoding options instead of the narrower stream encoding. +export interface Options extends Omit< + stream.TransformOptions, + "encoding" +> { + /** + * If true, the parser will attempt to convert read data types to native types. + * @deprecated Use {@link cast} + */ + auto_parse?: boolean | CastingFunction; + autoParse?: boolean | CastingFunction; + /** + * If true, the parser will attempt to convert read data types to dates. It requires the "auto_parse" option. + * @deprecated Use {@link cast_date} + */ + auto_parse_date?: boolean | CastingDateFunction; + autoParseDate?: boolean | CastingDateFunction; + /** + * If true, detect and exclude the byte order mark (BOM) from the CSV input if present. + */ + bom?: OptionsNormalized["bom"]; + /** + * If true, the parser will attempt to convert input string to native types. + * If a function, receive the value as first argument, a context as second argument and return a new value. More information about the context properties is available below. + */ + cast?: OptionsNormalized["cast"]; + /** + * If true, the parser will attempt to convert input string to dates. + * If a function, receive the value as argument and return a new value. It requires the "auto_parse" option. Be careful, it relies on Date.parse. + */ + cast_date?: OptionsNormalized["cast_date"]; + castDate?: OptionsNormalized["cast_date"]; + /** + * List of fields as an array, + * a user defined callback accepting the first line and returning the column names or true if autodiscovered in the first CSV line, + * default to null, + * affect the result data set in the sense that records will be objects instead of arrays. The callback receives the raw header fields as strings, while returned names may use keys from the typed input record. + */ + columns?: + | OptionsNormalized["columns"] + | ((record: string[]) => ColumnOption>[]); + /** + * Treat all the characters after this one as a comment, default to '' (disabled). + */ + comment?: OptionsNormalized["comment"] | boolean; + /** + * Restrict the definition of comments to a full line. Comment characters + * defined in the middle of the line are not interpreted as such. The + * option require the activation of comments. + */ + comment_no_infix?: OptionsNormalized["comment_no_infix"] | null; + /** + * Set the field delimiter. One character only, defaults to comma. + */ + delimiter?: OptionsNormalized["delimiter"] | string | string[] | Buffer; + /** + * Discover the field delimiter + */ + delimiter_auto?: boolean | Partial; + /** + * Set the source and destination encoding, a value of `null` returns buffer instead of strings. + */ + encoding?: OptionsNormalized["encoding"] | boolean | undefined; + /** + * Set the escape character, one character only, defaults to double quotes. + */ + escape?: OptionsNormalized["escape"] | string | boolean; + /** + * Start handling records from the requested number of records. + */ + from?: OptionsNormalized["from"] | string; + /** + * Start handling records from the requested line number. + */ + from_line?: OptionsNormalized["from_line"] | null | string; + fromLine?: OptionsNormalized["from_line"] | null | string; + /** + * Convert values into an array of values when columns are activated and + * when multiple columns of the same name are found. + */ + group_columns_by_name?: OptionsNormalized["group_columns_by_name"]; + groupColumnsByName?: OptionsNormalized["group_columns_by_name"]; + /** + * Don't interpret delimiters as such in the last field according to the number of fields calculated from the number of columns, the option require the presence of the `column` option when `true`. + */ + ignore_last_delimiters?: OptionsNormalized["ignore_last_delimiters"]; + /** + * Generate two properties `info` and `record` where `info` is a snapshot of the info object at the time the record was created and `record` is the parsed array or object. + */ + info?: OptionsNormalized["info"]; + /** + * If true, ignore whitespace immediately following the delimiter (i.e. left-trim all fields), defaults to false. + * Does not remove whitespace in a quoted field. + */ + ltrim?: OptionsNormalized["ltrim"] | null; + /** + * Maximum number of characters to be contained in the field and line buffers before an exception is raised, + * used to guard against a wrong delimiter or record_delimiter, + * default to 128000 characters. + */ + max_record_size?: OptionsNormalized["max_record_size"] | null | string; + maxRecordSize?: OptionsNormalized["max_record_size"]; + /** + * Name of header-record title to name objects by. + */ + objname?: OptionsNormalized["objname"] | Buffer | null; + /** + * Alter and filter records by executing a user defined function. + */ + on_record?: (record: U, context: InfoRecord) => T | null | undefined | U; + onRecord?: (record: U, context: InfoRecord) => T | null | undefined | U; + /** + * Function called when an error occurred if the `skip_records_with_error` + * option is activated. + */ + on_skip?: OptionsNormalized["on_skip"]; + onSkip?: OptionsNormalized["on_skip"]; + /** + * Optional character surrounding a field, one character only, defaults to double quotes. + */ + quote?: OptionsNormalized["quote"] | string | boolean; + /** + * Generate two properties raw and row where raw is the original CSV row content and row is the parsed array or object. + */ + raw?: OptionsNormalized["raw"] | null; + /** + * One or multiple characters used to delimit record rows; defaults to auto discovery if not provided. + * Supported auto discovery method are Linux ("\n"), Apple ("\r") and Windows ("\r\n") row delimiters. + */ + record_delimiter?: + | OptionsNormalized["record_delimiter"] + | string + | Buffer + | null + | (string | null)[]; + recordDelimiter?: + | OptionsNormalized["record_delimiter"] + | string + | Buffer + | null + | (string | null)[]; + /** + * Discard inconsistent columns count, default to false. + */ + relax_column_count?: OptionsNormalized["relax_column_count"] | null; + relaxColumnCount?: OptionsNormalized["relax_column_count"] | null; + /** + * Discard inconsistent columns count when the record contains less fields than expected, default to false. + */ + relax_column_count_less?: OptionsNormalized["relax_column_count_less"] | null; + relaxColumnCountLess?: OptionsNormalized["relax_column_count_less"] | null; + /** + * Discard inconsistent columns count when the record contains more fields than expected, default to false. + */ + relax_column_count_more?: OptionsNormalized["relax_column_count_more"] | null; + relaxColumnCountMore?: OptionsNormalized["relax_column_count_more"] | null; + /** + * Preserve quotes inside unquoted field. + */ + relax_quotes?: OptionsNormalized["relax_quotes"] | null; + relaxQuotes?: OptionsNormalized["relax_quotes"] | null; + /** + * If true, ignore whitespace immediately preceding the delimiter (i.e. right-trim all fields), defaults to false. + * Does not remove whitespace in a quoted field. + */ + rtrim?: OptionsNormalized["rtrim"] | null; + /** + * Dont generate empty values for empty lines. + * Defaults to false + */ + skip_empty_lines?: OptionsNormalized["skip_empty_lines"] | null; + skipEmptyLines?: OptionsNormalized["skip_empty_lines"] | null; + /** + * Don't generate records for lines containing empty column values (column matching /\s*\/), defaults to false. + */ + skip_records_with_empty_values?: + OptionsNormalized["skip_records_with_empty_values"] | null; + skipRecordsWithEmptyValues?: + OptionsNormalized["skip_records_with_empty_values"] | null; + /** + * Skip a line with error found inside and directly go process the next line. + */ + skip_records_with_error?: OptionsNormalized["skip_records_with_error"] | null; + skipRecordsWithError?: OptionsNormalized["skip_records_with_error"] | null; + /** + * Stop handling records after the requested number of records. + */ + to?: OptionsNormalized["to"] | null | string; + /** + * Stop handling records after the requested line number. + */ + to_line?: OptionsNormalized["to_line"] | null | string; + toLine?: OptionsNormalized["to_line"] | null | string; + /** + * If true, ignore whitespace immediately around the delimiter, defaults to false. + * Does not remove whitespace in a quoted field. + */ + trim?: OptionsNormalized["trim"] | null; +} + +export type OptionsWithColumns = Omit, "columns"> & { + columns: Exclude< + Options, NoInferColumnRecord>["columns"], + undefined | false + >; +}; diff --git a/packages/csv-parse/dist/esm/api/CsvError.d.ts b/packages/csv-parse/dist/esm/api/CsvError.d.ts new file mode 100644 index 00000000..4e384857 --- /dev/null +++ b/packages/csv-parse/dist/esm/api/CsvError.d.ts @@ -0,0 +1,35 @@ +import type { OptionsNormalized } from "../options.js"; + +export type CsvErrorCode = + | "CSV_INVALID_ARGUMENT" + | "CSV_INVALID_CLOSING_QUOTE" + | "CSV_INVALID_COLUMN_DEFINITION" + | "CSV_INVALID_COLUMN_MAPPING" + | "CSV_INVALID_OPTION_BOM" + | "CSV_INVALID_OPTION_CAST" + | "CSV_INVALID_OPTION_CAST_DATE" + | "CSV_INVALID_OPTION_COLUMNS" + | "CSV_INVALID_OPTION_COMMENT" + | "CSV_INVALID_OPTION_DELIMITER" + | "CSV_INVALID_OPTION_GROUP_COLUMNS_BY_NAME" + | "CSV_INVALID_OPTION_ON_RECORD" + | "CSV_MAX_RECORD_SIZE" + | "CSV_NON_TRIMABLE_CHAR_AFTER_CLOSING_QUOTE" + | "CSV_OPTION_COLUMNS_MISSING_NAME" + | "CSV_QUOTE_NOT_CLOSED" + | "CSV_RECORD_INCONSISTENT_FIELDS_LENGTH" + | "CSV_RECORD_INCONSISTENT_COLUMNS" + | "CSV_UNKNOWN_ERROR" + | "INVALID_OPENING_QUOTE"; + +export class CsvError extends Error { + readonly code: CsvErrorCode; + [key: string]: unknown; + + constructor( + code: CsvErrorCode, + message: string | string[], + options?: OptionsNormalized, + ...contexts: unknown[] + ); +} diff --git a/packages/csv-parse/dist/esm/index.d.ts b/packages/csv-parse/dist/esm/index.d.ts index 4e87a4dc..8799c00d 100644 --- a/packages/csv-parse/dist/esm/index.d.ts +++ b/packages/csv-parse/dist/esm/index.d.ts @@ -3,6 +3,19 @@ /// import * as stream from "stream"; +import type { CsvError } from "./api/CsvError.js"; +import type { + // Info + Info, + InfoCallback, + // Options + OptionsWithColumns, + OptionsNormalized, + Options, +} from "./options.js"; + +export * from "./api/CsvError.js"; +export type * from "./options.js"; export type Callback = ( err: CsvError | undefined, @@ -23,487 +36,6 @@ export class Parser extends stream.Transform { readonly info: Info; } -export interface Info { - /** - * The number of processed bytes. - */ - readonly bytes: number; - /** - * The number of processed bytes until the last successfully parsed and emitted records. - */ - readonly bytes_records: number; - /** - * The number of lines being fully commented. - */ - readonly comment_lines: number; - /** - * The number of processed empty lines. - */ - readonly empty_lines: number; - /** - * The number of non uniform records when `relax_column_count` is true. - */ - readonly invalid_field_length: number; - /** - * The number of lines encountered in the source dataset, start at 1 for the first line. - */ - readonly lines: number; - /** - * The number of processed records. - */ - readonly records: number; -} - -export interface InfoCallback extends Info { - /** - * Normalized version of `options.columns` when `options.columns` is true, boolean otherwise. - */ - readonly columns: boolean | { name: string }[] | { disabled: true }[]; -} - -export interface InfoDataSet extends Info { - readonly column: number | string; -} - -export interface InfoRecord extends InfoDataSet { - readonly error: CsvError; - readonly header: boolean; - readonly index: number; - readonly raw: string | undefined; -} - -export interface InfoField extends InfoRecord { - readonly quoting: boolean; -} - -/** - * @deprecated Use the InfoField interface instead, the interface will disappear in future versions. - */ -// eslint-disable-next-line -export interface CastingContext extends InfoField {} - -export type CastingFunction = (value: string, context: InfoField) => unknown; - -export type CastingDateFunction = (value: string, context: InfoField) => Date; - -export type ColumnOption = - K | undefined | null | false | { name: K }; - -type ColumnKey = T extends string[] - ? string - : unknown extends T - ? string - : string | keyof T; - -// Keep columns from overriding record types inferred from options such as raw. -type NoInferColumnRecord = [T][T extends unknown ? 0 : never]; - -export interface ScoringFunctionInfo { - /** - * The character code of the delimiter candidate being scored. - */ - readonly char_code: number; - /** - * The number of occurrences of the candidate in each line. - */ - readonly lines: number[]; - /** - * Whether the candidate is listed in the `preferred` option. - */ - readonly preferred: boolean; - /** - * The standard deviation of the occurrences across the lines. - */ - readonly std: number; - /** - * The total number of occurrences of the candidate. - */ - readonly total: number; -} - -export type ScoringFunction = ( - info: ScoringFunctionInfo, - options: ScoringFunctionOptions, -) => number; - -export interface ScoringFunctionOptions { - preferred: Record; - score: ScoringFunction; - size: number; -} - -export interface OptionsNormalized { - auto_parse?: boolean | CastingFunction; - auto_parse_date?: boolean | CastingDateFunction; - /** - * If true, detect and exclude the byte order mark (BOM) from the CSV input if present. - */ - bom?: boolean; - /** - * If true, the parser will attempt to convert input string to native types. - * If a function, receive the value as first argument, a context as second argument and return a new value. More information about the context properties is available below. - */ - cast?: boolean | CastingFunction; - /** - * If true, the parser will attempt to convert input string to dates. - * If a function, receive the value as argument and return a new value. It requires the "auto_parse" option. Be careful, it relies on Date.parse. - */ - cast_date?: boolean | CastingDateFunction; - /** - * Internal property string the function to - */ - cast_first_line_to_header?: ( - record: string[], - ) => ColumnOption>[]; - /** - * List of fields as an array, a user defined callback accepting the first - * line and returning the column names or true if autodiscovered in the first - * CSV line, default to null, affect the result data set in the sense that - * records will be objects instead of arrays. The callback receives the raw - * header fields as strings, while returned names may use keys from the typed - * input record. - */ - columns: boolean | ColumnOption>[]; - /** - * Treat all the characters after this one as a comment, default to '' (disabled). - */ - comment: string | null; - /** - * Restrict the definition of comments to a full line. Comment characters - * defined in the middle of the line are not interpreted as such. The - * option require the activation of comments. - */ - comment_no_infix: boolean; - /** - * Set the field delimiter. One character only, defaults to comma. - */ - delimiter: Buffer[]; - /** - * Discover the field delimiter. - */ - delimiter_auto: ScoringFunctionOptions; - /** - * Set the source and destination encoding, a value of `null` returns buffer instead of strings. - */ - encoding: BufferEncoding | null; - /** - * Set the escape character, one character only, defaults to double quotes. - */ - escape: null | Buffer; - /** - * Start handling records from the requested number of records. - */ - from: number; - /** - * Start handling records from the requested line number. - */ - from_line: number; - /** - * Convert values into an array of values when columns are activated and - * when multiple columns of the same name are found. - */ - group_columns_by_name: boolean; - /** - * Don't interpret delimiters as such in the last field according to the number of fields calculated from the number of columns, the option require the presence of the `column` option when `true`. - */ - ignore_last_delimiters: boolean | number; - /** - * Generate two properties `info` and `record` where `info` is a snapshot of the info object at the time the record was created and `record` is the parsed array or object. - */ - info: boolean; - /** - * If true, ignore whitespace immediately following the delimiter (i.e. left-trim all fields), defaults to false. - * Does not remove whitespace in a quoted field. - */ - ltrim: boolean; - /** - * Maximum number of characters to be contained in the field and line buffers before an exception is raised, - * used to guard against a wrong delimiter or record_delimiter, - * default to 128000 characters. - */ - max_record_size: number; - /** - * Name of header-record title to name objects by. - */ - objname: number | string | undefined; - /** - * Alter and filter records by executing a user defined function. - */ - on_record?: (record: U, context: InfoRecord) => T | null | undefined; - /** - * Function called when an error occurred if the `skip_records_with_error` - * option is activated. - */ - on_skip?: (err: CsvError | undefined, raw: string | undefined) => undefined; - /** - * Optional character surrounding a field, one character only, defaults to double quotes. - */ - quote?: Buffer | null; - /** - * Generate two properties raw and row where raw is the original CSV row content and row is the parsed array or object. - */ - raw: boolean; - /** - * One or multiple characters used to delimit record rows; defaults to auto discovery if not provided. - * Supported auto discovery method are Linux ("\n"), Apple ("\r") and Windows ("\r\n") row delimiters. - */ - record_delimiter: Buffer[]; - /** - * Discard inconsistent columns count, default to false. - */ - relax_column_count: boolean; - /** - * Discard inconsistent columns count when the record contains less fields than expected, default to false. - */ - relax_column_count_less: boolean; - /** - * Discard inconsistent columns count when the record contains more fields than expected, default to false. - */ - relax_column_count_more: boolean; - /** - * Preserve quotes inside unquoted field. - */ - relax_quotes: boolean; - /** - * If true, ignore whitespace immediately preceding the delimiter (i.e. right-trim all fields), defaults to false. - * Does not remove whitespace in a quoted field. - */ - rtrim: boolean; - /** - * Dont generate empty values for empty lines. - * Defaults to false - */ - skip_empty_lines: boolean; - /** - * Don't generate records for lines containing empty column values (column matching /\s*\/), defaults to false. - */ - skip_records_with_empty_values: boolean; - /** - * Skip a line with error found inside and directly go process the next line. - */ - skip_records_with_error: boolean; - /** - * Stop handling records after the requested number of records. - */ - to: number; - /** - * Stop handling records after the requested line number. - */ - to_line: number; - /** - * If true, ignore whitespace immediately around the delimiter, defaults to false. - * Does not remove whitespace in a quoted field. - */ - trim: boolean; -} - -// Keep the parser's encoding options instead of the narrower stream encoding. -export interface Options extends Omit< - stream.TransformOptions, - "encoding" -> { - /** - * If true, the parser will attempt to convert read data types to native types. - * @deprecated Use {@link cast} - */ - auto_parse?: boolean | CastingFunction; - autoParse?: boolean | CastingFunction; - /** - * If true, the parser will attempt to convert read data types to dates. It requires the "auto_parse" option. - * @deprecated Use {@link cast_date} - */ - auto_parse_date?: boolean | CastingDateFunction; - autoParseDate?: boolean | CastingDateFunction; - /** - * If true, detect and exclude the byte order mark (BOM) from the CSV input if present. - */ - bom?: OptionsNormalized["bom"]; - /** - * If true, the parser will attempt to convert input string to native types. - * If a function, receive the value as first argument, a context as second argument and return a new value. More information about the context properties is available below. - */ - cast?: OptionsNormalized["cast"]; - /** - * If true, the parser will attempt to convert input string to dates. - * If a function, receive the value as argument and return a new value. It requires the "auto_parse" option. Be careful, it relies on Date.parse. - */ - cast_date?: OptionsNormalized["cast_date"]; - castDate?: OptionsNormalized["cast_date"]; - /** - * List of fields as an array, - * a user defined callback accepting the first line and returning the column names or true if autodiscovered in the first CSV line, - * default to null, - * affect the result data set in the sense that records will be objects instead of arrays. The callback receives the raw header fields as strings, while returned names may use keys from the typed input record. - */ - columns?: - | OptionsNormalized["columns"] - | ((record: string[]) => ColumnOption>[]); - /** - * Treat all the characters after this one as a comment, default to '' (disabled). - */ - comment?: OptionsNormalized["comment"] | boolean; - /** - * Restrict the definition of comments to a full line. Comment characters - * defined in the middle of the line are not interpreted as such. The - * option require the activation of comments. - */ - comment_no_infix?: OptionsNormalized["comment_no_infix"] | null; - /** - * Set the field delimiter. One character only, defaults to comma. - */ - delimiter?: OptionsNormalized["delimiter"] | string | string[] | Buffer; - /** - * Discover the field delimiter - */ - delimiter_auto?: boolean | Partial; - /** - * Set the source and destination encoding, a value of `null` returns buffer instead of strings. - */ - encoding?: OptionsNormalized["encoding"] | boolean | undefined; - /** - * Set the escape character, one character only, defaults to double quotes. - */ - escape?: OptionsNormalized["escape"] | string | boolean; - /** - * Start handling records from the requested number of records. - */ - from?: OptionsNormalized["from"] | string; - /** - * Start handling records from the requested line number. - */ - from_line?: OptionsNormalized["from_line"] | null | string; - fromLine?: OptionsNormalized["from_line"] | null | string; - /** - * Convert values into an array of values when columns are activated and - * when multiple columns of the same name are found. - */ - group_columns_by_name?: OptionsNormalized["group_columns_by_name"]; - groupColumnsByName?: OptionsNormalized["group_columns_by_name"]; - /** - * Don't interpret delimiters as such in the last field according to the number of fields calculated from the number of columns, the option require the presence of the `column` option when `true`. - */ - ignore_last_delimiters?: OptionsNormalized["ignore_last_delimiters"]; - /** - * Generate two properties `info` and `record` where `info` is a snapshot of the info object at the time the record was created and `record` is the parsed array or object. - */ - info?: OptionsNormalized["info"]; - /** - * If true, ignore whitespace immediately following the delimiter (i.e. left-trim all fields), defaults to false. - * Does not remove whitespace in a quoted field. - */ - ltrim?: OptionsNormalized["ltrim"] | null; - /** - * Maximum number of characters to be contained in the field and line buffers before an exception is raised, - * used to guard against a wrong delimiter or record_delimiter, - * default to 128000 characters. - */ - max_record_size?: OptionsNormalized["max_record_size"] | null | string; - maxRecordSize?: OptionsNormalized["max_record_size"]; - /** - * Name of header-record title to name objects by. - */ - objname?: OptionsNormalized["objname"] | Buffer | null; - /** - * Alter and filter records by executing a user defined function. - */ - on_record?: (record: U, context: InfoRecord) => T | null | undefined | U; - onRecord?: (record: U, context: InfoRecord) => T | null | undefined | U; - /** - * Function called when an error occurred if the `skip_records_with_error` - * option is activated. - */ - on_skip?: OptionsNormalized["on_skip"]; - onSkip?: OptionsNormalized["on_skip"]; - /** - * Optional character surrounding a field, one character only, defaults to double quotes. - */ - quote?: OptionsNormalized["quote"] | string | boolean; - /** - * Generate two properties raw and row where raw is the original CSV row content and row is the parsed array or object. - */ - raw?: OptionsNormalized["raw"] | null; - /** - * One or multiple characters used to delimit record rows; defaults to auto discovery if not provided. - * Supported auto discovery method are Linux ("\n"), Apple ("\r") and Windows ("\r\n") row delimiters. - */ - record_delimiter?: - | OptionsNormalized["record_delimiter"] - | string - | Buffer - | null - | (string | null)[]; - recordDelimiter?: - | OptionsNormalized["record_delimiter"] - | string - | Buffer - | null - | (string | null)[]; - /** - * Discard inconsistent columns count, default to false. - */ - relax_column_count?: OptionsNormalized["relax_column_count"] | null; - relaxColumnCount?: OptionsNormalized["relax_column_count"] | null; - /** - * Discard inconsistent columns count when the record contains less fields than expected, default to false. - */ - relax_column_count_less?: OptionsNormalized["relax_column_count_less"] | null; - relaxColumnCountLess?: OptionsNormalized["relax_column_count_less"] | null; - /** - * Discard inconsistent columns count when the record contains more fields than expected, default to false. - */ - relax_column_count_more?: OptionsNormalized["relax_column_count_more"] | null; - relaxColumnCountMore?: OptionsNormalized["relax_column_count_more"] | null; - /** - * Preserve quotes inside unquoted field. - */ - relax_quotes?: OptionsNormalized["relax_quotes"] | null; - relaxQuotes?: OptionsNormalized["relax_quotes"] | null; - /** - * If true, ignore whitespace immediately preceding the delimiter (i.e. right-trim all fields), defaults to false. - * Does not remove whitespace in a quoted field. - */ - rtrim?: OptionsNormalized["rtrim"] | null; - /** - * Dont generate empty values for empty lines. - * Defaults to false - */ - skip_empty_lines?: OptionsNormalized["skip_empty_lines"] | null; - skipEmptyLines?: OptionsNormalized["skip_empty_lines"] | null; - /** - * Don't generate records for lines containing empty column values (column matching /\s*\/), defaults to false. - */ - skip_records_with_empty_values?: - OptionsNormalized["skip_records_with_empty_values"] | null; - skipRecordsWithEmptyValues?: - OptionsNormalized["skip_records_with_empty_values"] | null; - /** - * Skip a line with error found inside and directly go process the next line. - */ - skip_records_with_error?: OptionsNormalized["skip_records_with_error"] | null; - skipRecordsWithError?: OptionsNormalized["skip_records_with_error"] | null; - /** - * Stop handling records after the requested number of records. - */ - to?: OptionsNormalized["to"] | null | string; - /** - * Stop handling records after the requested line number. - */ - to_line?: OptionsNormalized["to_line"] | null | string; - toLine?: OptionsNormalized["to_line"] | null | string; - /** - * If true, ignore whitespace immediately around the delimiter, defaults to false. - * Does not remove whitespace in a quoted field. - */ - trim?: OptionsNormalized["trim"] | null; -} - -export type OptionsWithColumns = Omit, "columns"> & { - columns: Exclude< - Options, NoInferColumnRecord>["columns"], - undefined | false - >; -}; - declare function parse( input: string | Buffer | Uint8Array, options: OptionsWithColumns, @@ -529,42 +61,6 @@ declare function parse(callback?: Callback): Parser; export { parse }; -/////////////////////////////////////////////////////// CsvError - -export type CsvErrorCode = - | "CSV_INVALID_ARGUMENT" - | "CSV_INVALID_CLOSING_QUOTE" - | "CSV_INVALID_COLUMN_DEFINITION" - | "CSV_INVALID_COLUMN_MAPPING" - | "CSV_INVALID_OPTION_BOM" - | "CSV_INVALID_OPTION_CAST" - | "CSV_INVALID_OPTION_CAST_DATE" - | "CSV_INVALID_OPTION_COLUMNS" - | "CSV_INVALID_OPTION_COMMENT" - | "CSV_INVALID_OPTION_DELIMITER" - | "CSV_INVALID_OPTION_GROUP_COLUMNS_BY_NAME" - | "CSV_INVALID_OPTION_ON_RECORD" - | "CSV_MAX_RECORD_SIZE" - | "CSV_NON_TRIMABLE_CHAR_AFTER_CLOSING_QUOTE" - | "CSV_OPTION_COLUMNS_MISSING_NAME" - | "CSV_QUOTE_NOT_CLOSED" - | "CSV_RECORD_INCONSISTENT_FIELDS_LENGTH" - | "CSV_RECORD_INCONSISTENT_COLUMNS" - | "CSV_UNKNOWN_ERROR" - | "INVALID_OPENING_QUOTE"; - -export class CsvError extends Error { - readonly code: CsvErrorCode; - [key: string]: unknown; - - constructor( - code: CsvErrorCode, - message: string | string[], - options?: OptionsNormalized, - ...contexts: unknown[] - ); -} - /////////////////////////////////////////////////////// normalize_options declare function normalize_options(opts: Options): OptionsNormalized; diff --git a/packages/csv-parse/dist/esm/options.d.ts b/packages/csv-parse/dist/esm/options.d.ts new file mode 100644 index 00000000..6a5ae688 --- /dev/null +++ b/packages/csv-parse/dist/esm/options.d.ts @@ -0,0 +1,484 @@ +import * as stream from "stream"; + +import { CsvError } from "./api/CsvError.js"; + +export interface Info { + /** + * The number of processed bytes. + */ + readonly bytes: number; + /** + * The number of processed bytes until the last successfully parsed and emitted records. + */ + readonly bytes_records: number; + /** + * The number of lines being fully commented. + */ + readonly comment_lines: number; + /** + * The number of processed empty lines. + */ + readonly empty_lines: number; + /** + * The number of non uniform records when `relax_column_count` is true. + */ + readonly invalid_field_length: number; + /** + * The number of lines encountered in the source dataset, start at 1 for the first line. + */ + readonly lines: number; + /** + * The number of processed records. + */ + readonly records: number; +} + +export interface InfoCallback extends Info { + /** + * Normalized version of `options.columns` when `options.columns` is true, boolean otherwise. + */ + readonly columns: boolean | { name: string }[] | { disabled: true }[]; +} + +export interface InfoDataSet extends Info { + readonly column: number | string; +} + +export interface InfoRecord extends InfoDataSet { + readonly error: CsvError; + readonly header: boolean; + readonly index: number; + readonly raw: string | undefined; +} + +export interface InfoField extends InfoRecord { + readonly quoting: boolean; +} + +/** + * @deprecated Use the InfoField interface instead, the interface will disappear in future versions. + */ +// eslint-disable-next-line +export interface CastingContext extends InfoField {} + +export type CastingFunction = (value: string, context: InfoField) => unknown; + +export type CastingDateFunction = (value: string, context: InfoField) => Date; + +export type ColumnOption = + K | undefined | null | false | { name: K }; + +type ColumnKey = T extends string[] + ? string + : unknown extends T + ? string + : string | keyof T; + +// Keep columns from overriding record types inferred from options such as raw. +type NoInferColumnRecord = [T][T extends unknown ? 0 : never]; + +export interface ScoringFunctionInfo { + /** + * The character code of the delimiter candidate being scored. + */ + readonly char_code: number; + /** + * The number of occurrences of the candidate in each line. + */ + readonly lines: number[]; + /** + * Whether the candidate is listed in the `preferred` option. + */ + readonly preferred: boolean; + /** + * The standard deviation of the occurrences across the lines. + */ + readonly std: number; + /** + * The total number of occurrences of the candidate. + */ + readonly total: number; +} + +export type ScoringFunction = ( + info: ScoringFunctionInfo, + options: ScoringFunctionOptions, +) => number; + +export interface ScoringFunctionOptions { + preferred: Record; + score: ScoringFunction; + size: number; +} + +export interface OptionsNormalized { + auto_parse?: boolean | CastingFunction; + auto_parse_date?: boolean | CastingDateFunction; + /** + * If true, detect and exclude the byte order mark (BOM) from the CSV input if present. + */ + bom?: boolean; + /** + * If true, the parser will attempt to convert input string to native types. + * If a function, receive the value as first argument, a context as second argument and return a new value. More information about the context properties is available below. + */ + cast?: boolean | CastingFunction; + /** + * If true, the parser will attempt to convert input string to dates. + * If a function, receive the value as argument and return a new value. It requires the "auto_parse" option. Be careful, it relies on Date.parse. + */ + cast_date?: boolean | CastingDateFunction; + /** + * Internal property string the function to + */ + cast_first_line_to_header?: ( + record: string[], + ) => ColumnOption>[]; + /** + * List of fields as an array, a user defined callback accepting the first + * line and returning the column names or true if autodiscovered in the first + * CSV line, default to null, affect the result data set in the sense that + * records will be objects instead of arrays. The callback receives the raw + * header fields as strings, while returned names may use keys from the typed + * input record. + */ + columns: boolean | ColumnOption>[]; + /** + * Treat all the characters after this one as a comment, default to '' (disabled). + */ + comment: string | null; + /** + * Restrict the definition of comments to a full line. Comment characters + * defined in the middle of the line are not interpreted as such. The + * option require the activation of comments. + */ + comment_no_infix: boolean; + /** + * Set the field delimiter. One character only, defaults to comma. + */ + delimiter: Buffer[]; + /** + * Discover the field delimiter. + */ + delimiter_auto: ScoringFunctionOptions; + /** + * Set the source and destination encoding, a value of `null` returns buffer instead of strings. + */ + encoding: BufferEncoding | null; + /** + * Set the escape character, one character only, defaults to double quotes. + */ + escape: null | Buffer; + /** + * Start handling records from the requested number of records. + */ + from: number; + /** + * Start handling records from the requested line number. + */ + from_line: number; + /** + * Convert values into an array of values when columns are activated and + * when multiple columns of the same name are found. + */ + group_columns_by_name: boolean; + /** + * Don't interpret delimiters as such in the last field according to the number of fields calculated from the number of columns, the option require the presence of the `column` option when `true`. + */ + ignore_last_delimiters: boolean | number; + /** + * Generate two properties `info` and `record` where `info` is a snapshot of the info object at the time the record was created and `record` is the parsed array or object. + */ + info: boolean; + /** + * If true, ignore whitespace immediately following the delimiter (i.e. left-trim all fields), defaults to false. + * Does not remove whitespace in a quoted field. + */ + ltrim: boolean; + /** + * Maximum number of characters to be contained in the field and line buffers before an exception is raised, + * used to guard against a wrong delimiter or record_delimiter, + * default to 128000 characters. + */ + max_record_size: number; + /** + * Name of header-record title to name objects by. + */ + objname: number | string | undefined; + /** + * Alter and filter records by executing a user defined function. + */ + on_record?: (record: U, context: InfoRecord) => T | null | undefined; + /** + * Function called when an error occurred if the `skip_records_with_error` + * option is activated. + */ + on_skip?: (err: CsvError | undefined, raw: string | undefined) => undefined; + /** + * Optional character surrounding a field, one character only, defaults to double quotes. + */ + quote?: Buffer | null; + /** + * Generate two properties raw and row where raw is the original CSV row content and row is the parsed array or object. + */ + raw: boolean; + /** + * One or multiple characters used to delimit record rows; defaults to auto discovery if not provided. + * Supported auto discovery method are Linux ("\n"), Apple ("\r") and Windows ("\r\n") row delimiters. + */ + record_delimiter: Buffer[]; + /** + * Discard inconsistent columns count, default to false. + */ + relax_column_count: boolean; + /** + * Discard inconsistent columns count when the record contains less fields than expected, default to false. + */ + relax_column_count_less: boolean; + /** + * Discard inconsistent columns count when the record contains more fields than expected, default to false. + */ + relax_column_count_more: boolean; + /** + * Preserve quotes inside unquoted field. + */ + relax_quotes: boolean; + /** + * If true, ignore whitespace immediately preceding the delimiter (i.e. right-trim all fields), defaults to false. + * Does not remove whitespace in a quoted field. + */ + rtrim: boolean; + /** + * Dont generate empty values for empty lines. + * Defaults to false + */ + skip_empty_lines: boolean; + /** + * Don't generate records for lines containing empty column values (column matching /\s*\/), defaults to false. + */ + skip_records_with_empty_values: boolean; + /** + * Skip a line with error found inside and directly go process the next line. + */ + skip_records_with_error: boolean; + /** + * Stop handling records after the requested number of records. + */ + to: number; + /** + * Stop handling records after the requested line number. + */ + to_line: number; + /** + * If true, ignore whitespace immediately around the delimiter, defaults to false. + * Does not remove whitespace in a quoted field. + */ + trim: boolean; +} + +// Keep the parser's encoding options instead of the narrower stream encoding. +export interface Options extends Omit< + stream.TransformOptions, + "encoding" +> { + /** + * If true, the parser will attempt to convert read data types to native types. + * @deprecated Use {@link cast} + */ + auto_parse?: boolean | CastingFunction; + autoParse?: boolean | CastingFunction; + /** + * If true, the parser will attempt to convert read data types to dates. It requires the "auto_parse" option. + * @deprecated Use {@link cast_date} + */ + auto_parse_date?: boolean | CastingDateFunction; + autoParseDate?: boolean | CastingDateFunction; + /** + * If true, detect and exclude the byte order mark (BOM) from the CSV input if present. + */ + bom?: OptionsNormalized["bom"]; + /** + * If true, the parser will attempt to convert input string to native types. + * If a function, receive the value as first argument, a context as second argument and return a new value. More information about the context properties is available below. + */ + cast?: OptionsNormalized["cast"]; + /** + * If true, the parser will attempt to convert input string to dates. + * If a function, receive the value as argument and return a new value. It requires the "auto_parse" option. Be careful, it relies on Date.parse. + */ + cast_date?: OptionsNormalized["cast_date"]; + castDate?: OptionsNormalized["cast_date"]; + /** + * List of fields as an array, + * a user defined callback accepting the first line and returning the column names or true if autodiscovered in the first CSV line, + * default to null, + * affect the result data set in the sense that records will be objects instead of arrays. The callback receives the raw header fields as strings, while returned names may use keys from the typed input record. + */ + columns?: + | OptionsNormalized["columns"] + | ((record: string[]) => ColumnOption>[]); + /** + * Treat all the characters after this one as a comment, default to '' (disabled). + */ + comment?: OptionsNormalized["comment"] | boolean; + /** + * Restrict the definition of comments to a full line. Comment characters + * defined in the middle of the line are not interpreted as such. The + * option require the activation of comments. + */ + comment_no_infix?: OptionsNormalized["comment_no_infix"] | null; + /** + * Set the field delimiter. One character only, defaults to comma. + */ + delimiter?: OptionsNormalized["delimiter"] | string | string[] | Buffer; + /** + * Discover the field delimiter + */ + delimiter_auto?: boolean | Partial; + /** + * Set the source and destination encoding, a value of `null` returns buffer instead of strings. + */ + encoding?: OptionsNormalized["encoding"] | boolean | undefined; + /** + * Set the escape character, one character only, defaults to double quotes. + */ + escape?: OptionsNormalized["escape"] | string | boolean; + /** + * Start handling records from the requested number of records. + */ + from?: OptionsNormalized["from"] | string; + /** + * Start handling records from the requested line number. + */ + from_line?: OptionsNormalized["from_line"] | null | string; + fromLine?: OptionsNormalized["from_line"] | null | string; + /** + * Convert values into an array of values when columns are activated and + * when multiple columns of the same name are found. + */ + group_columns_by_name?: OptionsNormalized["group_columns_by_name"]; + groupColumnsByName?: OptionsNormalized["group_columns_by_name"]; + /** + * Don't interpret delimiters as such in the last field according to the number of fields calculated from the number of columns, the option require the presence of the `column` option when `true`. + */ + ignore_last_delimiters?: OptionsNormalized["ignore_last_delimiters"]; + /** + * Generate two properties `info` and `record` where `info` is a snapshot of the info object at the time the record was created and `record` is the parsed array or object. + */ + info?: OptionsNormalized["info"]; + /** + * If true, ignore whitespace immediately following the delimiter (i.e. left-trim all fields), defaults to false. + * Does not remove whitespace in a quoted field. + */ + ltrim?: OptionsNormalized["ltrim"] | null; + /** + * Maximum number of characters to be contained in the field and line buffers before an exception is raised, + * used to guard against a wrong delimiter or record_delimiter, + * default to 128000 characters. + */ + max_record_size?: OptionsNormalized["max_record_size"] | null | string; + maxRecordSize?: OptionsNormalized["max_record_size"]; + /** + * Name of header-record title to name objects by. + */ + objname?: OptionsNormalized["objname"] | Buffer | null; + /** + * Alter and filter records by executing a user defined function. + */ + on_record?: (record: U, context: InfoRecord) => T | null | undefined | U; + onRecord?: (record: U, context: InfoRecord) => T | null | undefined | U; + /** + * Function called when an error occurred if the `skip_records_with_error` + * option is activated. + */ + on_skip?: OptionsNormalized["on_skip"]; + onSkip?: OptionsNormalized["on_skip"]; + /** + * Optional character surrounding a field, one character only, defaults to double quotes. + */ + quote?: OptionsNormalized["quote"] | string | boolean; + /** + * Generate two properties raw and row where raw is the original CSV row content and row is the parsed array or object. + */ + raw?: OptionsNormalized["raw"] | null; + /** + * One or multiple characters used to delimit record rows; defaults to auto discovery if not provided. + * Supported auto discovery method are Linux ("\n"), Apple ("\r") and Windows ("\r\n") row delimiters. + */ + record_delimiter?: + | OptionsNormalized["record_delimiter"] + | string + | Buffer + | null + | (string | null)[]; + recordDelimiter?: + | OptionsNormalized["record_delimiter"] + | string + | Buffer + | null + | (string | null)[]; + /** + * Discard inconsistent columns count, default to false. + */ + relax_column_count?: OptionsNormalized["relax_column_count"] | null; + relaxColumnCount?: OptionsNormalized["relax_column_count"] | null; + /** + * Discard inconsistent columns count when the record contains less fields than expected, default to false. + */ + relax_column_count_less?: OptionsNormalized["relax_column_count_less"] | null; + relaxColumnCountLess?: OptionsNormalized["relax_column_count_less"] | null; + /** + * Discard inconsistent columns count when the record contains more fields than expected, default to false. + */ + relax_column_count_more?: OptionsNormalized["relax_column_count_more"] | null; + relaxColumnCountMore?: OptionsNormalized["relax_column_count_more"] | null; + /** + * Preserve quotes inside unquoted field. + */ + relax_quotes?: OptionsNormalized["relax_quotes"] | null; + relaxQuotes?: OptionsNormalized["relax_quotes"] | null; + /** + * If true, ignore whitespace immediately preceding the delimiter (i.e. right-trim all fields), defaults to false. + * Does not remove whitespace in a quoted field. + */ + rtrim?: OptionsNormalized["rtrim"] | null; + /** + * Dont generate empty values for empty lines. + * Defaults to false + */ + skip_empty_lines?: OptionsNormalized["skip_empty_lines"] | null; + skipEmptyLines?: OptionsNormalized["skip_empty_lines"] | null; + /** + * Don't generate records for lines containing empty column values (column matching /\s*\/), defaults to false. + */ + skip_records_with_empty_values?: + OptionsNormalized["skip_records_with_empty_values"] | null; + skipRecordsWithEmptyValues?: + OptionsNormalized["skip_records_with_empty_values"] | null; + /** + * Skip a line with error found inside and directly go process the next line. + */ + skip_records_with_error?: OptionsNormalized["skip_records_with_error"] | null; + skipRecordsWithError?: OptionsNormalized["skip_records_with_error"] | null; + /** + * Stop handling records after the requested number of records. + */ + to?: OptionsNormalized["to"] | null | string; + /** + * Stop handling records after the requested line number. + */ + to_line?: OptionsNormalized["to_line"] | null | string; + toLine?: OptionsNormalized["to_line"] | null | string; + /** + * If true, ignore whitespace immediately around the delimiter, defaults to false. + * Does not remove whitespace in a quoted field. + */ + trim?: OptionsNormalized["trim"] | null; +} + +export type OptionsWithColumns = Omit, "columns"> & { + columns: Exclude< + Options, NoInferColumnRecord>["columns"], + undefined | false + >; +}; diff --git a/packages/csv-parse/lib/api/CsvError.d.ts b/packages/csv-parse/lib/api/CsvError.d.ts new file mode 100644 index 00000000..4e384857 --- /dev/null +++ b/packages/csv-parse/lib/api/CsvError.d.ts @@ -0,0 +1,35 @@ +import type { OptionsNormalized } from "../options.js"; + +export type CsvErrorCode = + | "CSV_INVALID_ARGUMENT" + | "CSV_INVALID_CLOSING_QUOTE" + | "CSV_INVALID_COLUMN_DEFINITION" + | "CSV_INVALID_COLUMN_MAPPING" + | "CSV_INVALID_OPTION_BOM" + | "CSV_INVALID_OPTION_CAST" + | "CSV_INVALID_OPTION_CAST_DATE" + | "CSV_INVALID_OPTION_COLUMNS" + | "CSV_INVALID_OPTION_COMMENT" + | "CSV_INVALID_OPTION_DELIMITER" + | "CSV_INVALID_OPTION_GROUP_COLUMNS_BY_NAME" + | "CSV_INVALID_OPTION_ON_RECORD" + | "CSV_MAX_RECORD_SIZE" + | "CSV_NON_TRIMABLE_CHAR_AFTER_CLOSING_QUOTE" + | "CSV_OPTION_COLUMNS_MISSING_NAME" + | "CSV_QUOTE_NOT_CLOSED" + | "CSV_RECORD_INCONSISTENT_FIELDS_LENGTH" + | "CSV_RECORD_INCONSISTENT_COLUMNS" + | "CSV_UNKNOWN_ERROR" + | "INVALID_OPENING_QUOTE"; + +export class CsvError extends Error { + readonly code: CsvErrorCode; + [key: string]: unknown; + + constructor( + code: CsvErrorCode, + message: string | string[], + options?: OptionsNormalized, + ...contexts: unknown[] + ); +} diff --git a/packages/csv-parse/lib/index.d.ts b/packages/csv-parse/lib/index.d.ts index 4e87a4dc..8799c00d 100644 --- a/packages/csv-parse/lib/index.d.ts +++ b/packages/csv-parse/lib/index.d.ts @@ -3,6 +3,19 @@ /// import * as stream from "stream"; +import type { CsvError } from "./api/CsvError.js"; +import type { + // Info + Info, + InfoCallback, + // Options + OptionsWithColumns, + OptionsNormalized, + Options, +} from "./options.js"; + +export * from "./api/CsvError.js"; +export type * from "./options.js"; export type Callback = ( err: CsvError | undefined, @@ -23,487 +36,6 @@ export class Parser extends stream.Transform { readonly info: Info; } -export interface Info { - /** - * The number of processed bytes. - */ - readonly bytes: number; - /** - * The number of processed bytes until the last successfully parsed and emitted records. - */ - readonly bytes_records: number; - /** - * The number of lines being fully commented. - */ - readonly comment_lines: number; - /** - * The number of processed empty lines. - */ - readonly empty_lines: number; - /** - * The number of non uniform records when `relax_column_count` is true. - */ - readonly invalid_field_length: number; - /** - * The number of lines encountered in the source dataset, start at 1 for the first line. - */ - readonly lines: number; - /** - * The number of processed records. - */ - readonly records: number; -} - -export interface InfoCallback extends Info { - /** - * Normalized version of `options.columns` when `options.columns` is true, boolean otherwise. - */ - readonly columns: boolean | { name: string }[] | { disabled: true }[]; -} - -export interface InfoDataSet extends Info { - readonly column: number | string; -} - -export interface InfoRecord extends InfoDataSet { - readonly error: CsvError; - readonly header: boolean; - readonly index: number; - readonly raw: string | undefined; -} - -export interface InfoField extends InfoRecord { - readonly quoting: boolean; -} - -/** - * @deprecated Use the InfoField interface instead, the interface will disappear in future versions. - */ -// eslint-disable-next-line -export interface CastingContext extends InfoField {} - -export type CastingFunction = (value: string, context: InfoField) => unknown; - -export type CastingDateFunction = (value: string, context: InfoField) => Date; - -export type ColumnOption = - K | undefined | null | false | { name: K }; - -type ColumnKey = T extends string[] - ? string - : unknown extends T - ? string - : string | keyof T; - -// Keep columns from overriding record types inferred from options such as raw. -type NoInferColumnRecord = [T][T extends unknown ? 0 : never]; - -export interface ScoringFunctionInfo { - /** - * The character code of the delimiter candidate being scored. - */ - readonly char_code: number; - /** - * The number of occurrences of the candidate in each line. - */ - readonly lines: number[]; - /** - * Whether the candidate is listed in the `preferred` option. - */ - readonly preferred: boolean; - /** - * The standard deviation of the occurrences across the lines. - */ - readonly std: number; - /** - * The total number of occurrences of the candidate. - */ - readonly total: number; -} - -export type ScoringFunction = ( - info: ScoringFunctionInfo, - options: ScoringFunctionOptions, -) => number; - -export interface ScoringFunctionOptions { - preferred: Record; - score: ScoringFunction; - size: number; -} - -export interface OptionsNormalized { - auto_parse?: boolean | CastingFunction; - auto_parse_date?: boolean | CastingDateFunction; - /** - * If true, detect and exclude the byte order mark (BOM) from the CSV input if present. - */ - bom?: boolean; - /** - * If true, the parser will attempt to convert input string to native types. - * If a function, receive the value as first argument, a context as second argument and return a new value. More information about the context properties is available below. - */ - cast?: boolean | CastingFunction; - /** - * If true, the parser will attempt to convert input string to dates. - * If a function, receive the value as argument and return a new value. It requires the "auto_parse" option. Be careful, it relies on Date.parse. - */ - cast_date?: boolean | CastingDateFunction; - /** - * Internal property string the function to - */ - cast_first_line_to_header?: ( - record: string[], - ) => ColumnOption>[]; - /** - * List of fields as an array, a user defined callback accepting the first - * line and returning the column names or true if autodiscovered in the first - * CSV line, default to null, affect the result data set in the sense that - * records will be objects instead of arrays. The callback receives the raw - * header fields as strings, while returned names may use keys from the typed - * input record. - */ - columns: boolean | ColumnOption>[]; - /** - * Treat all the characters after this one as a comment, default to '' (disabled). - */ - comment: string | null; - /** - * Restrict the definition of comments to a full line. Comment characters - * defined in the middle of the line are not interpreted as such. The - * option require the activation of comments. - */ - comment_no_infix: boolean; - /** - * Set the field delimiter. One character only, defaults to comma. - */ - delimiter: Buffer[]; - /** - * Discover the field delimiter. - */ - delimiter_auto: ScoringFunctionOptions; - /** - * Set the source and destination encoding, a value of `null` returns buffer instead of strings. - */ - encoding: BufferEncoding | null; - /** - * Set the escape character, one character only, defaults to double quotes. - */ - escape: null | Buffer; - /** - * Start handling records from the requested number of records. - */ - from: number; - /** - * Start handling records from the requested line number. - */ - from_line: number; - /** - * Convert values into an array of values when columns are activated and - * when multiple columns of the same name are found. - */ - group_columns_by_name: boolean; - /** - * Don't interpret delimiters as such in the last field according to the number of fields calculated from the number of columns, the option require the presence of the `column` option when `true`. - */ - ignore_last_delimiters: boolean | number; - /** - * Generate two properties `info` and `record` where `info` is a snapshot of the info object at the time the record was created and `record` is the parsed array or object. - */ - info: boolean; - /** - * If true, ignore whitespace immediately following the delimiter (i.e. left-trim all fields), defaults to false. - * Does not remove whitespace in a quoted field. - */ - ltrim: boolean; - /** - * Maximum number of characters to be contained in the field and line buffers before an exception is raised, - * used to guard against a wrong delimiter or record_delimiter, - * default to 128000 characters. - */ - max_record_size: number; - /** - * Name of header-record title to name objects by. - */ - objname: number | string | undefined; - /** - * Alter and filter records by executing a user defined function. - */ - on_record?: (record: U, context: InfoRecord) => T | null | undefined; - /** - * Function called when an error occurred if the `skip_records_with_error` - * option is activated. - */ - on_skip?: (err: CsvError | undefined, raw: string | undefined) => undefined; - /** - * Optional character surrounding a field, one character only, defaults to double quotes. - */ - quote?: Buffer | null; - /** - * Generate two properties raw and row where raw is the original CSV row content and row is the parsed array or object. - */ - raw: boolean; - /** - * One or multiple characters used to delimit record rows; defaults to auto discovery if not provided. - * Supported auto discovery method are Linux ("\n"), Apple ("\r") and Windows ("\r\n") row delimiters. - */ - record_delimiter: Buffer[]; - /** - * Discard inconsistent columns count, default to false. - */ - relax_column_count: boolean; - /** - * Discard inconsistent columns count when the record contains less fields than expected, default to false. - */ - relax_column_count_less: boolean; - /** - * Discard inconsistent columns count when the record contains more fields than expected, default to false. - */ - relax_column_count_more: boolean; - /** - * Preserve quotes inside unquoted field. - */ - relax_quotes: boolean; - /** - * If true, ignore whitespace immediately preceding the delimiter (i.e. right-trim all fields), defaults to false. - * Does not remove whitespace in a quoted field. - */ - rtrim: boolean; - /** - * Dont generate empty values for empty lines. - * Defaults to false - */ - skip_empty_lines: boolean; - /** - * Don't generate records for lines containing empty column values (column matching /\s*\/), defaults to false. - */ - skip_records_with_empty_values: boolean; - /** - * Skip a line with error found inside and directly go process the next line. - */ - skip_records_with_error: boolean; - /** - * Stop handling records after the requested number of records. - */ - to: number; - /** - * Stop handling records after the requested line number. - */ - to_line: number; - /** - * If true, ignore whitespace immediately around the delimiter, defaults to false. - * Does not remove whitespace in a quoted field. - */ - trim: boolean; -} - -// Keep the parser's encoding options instead of the narrower stream encoding. -export interface Options extends Omit< - stream.TransformOptions, - "encoding" -> { - /** - * If true, the parser will attempt to convert read data types to native types. - * @deprecated Use {@link cast} - */ - auto_parse?: boolean | CastingFunction; - autoParse?: boolean | CastingFunction; - /** - * If true, the parser will attempt to convert read data types to dates. It requires the "auto_parse" option. - * @deprecated Use {@link cast_date} - */ - auto_parse_date?: boolean | CastingDateFunction; - autoParseDate?: boolean | CastingDateFunction; - /** - * If true, detect and exclude the byte order mark (BOM) from the CSV input if present. - */ - bom?: OptionsNormalized["bom"]; - /** - * If true, the parser will attempt to convert input string to native types. - * If a function, receive the value as first argument, a context as second argument and return a new value. More information about the context properties is available below. - */ - cast?: OptionsNormalized["cast"]; - /** - * If true, the parser will attempt to convert input string to dates. - * If a function, receive the value as argument and return a new value. It requires the "auto_parse" option. Be careful, it relies on Date.parse. - */ - cast_date?: OptionsNormalized["cast_date"]; - castDate?: OptionsNormalized["cast_date"]; - /** - * List of fields as an array, - * a user defined callback accepting the first line and returning the column names or true if autodiscovered in the first CSV line, - * default to null, - * affect the result data set in the sense that records will be objects instead of arrays. The callback receives the raw header fields as strings, while returned names may use keys from the typed input record. - */ - columns?: - | OptionsNormalized["columns"] - | ((record: string[]) => ColumnOption>[]); - /** - * Treat all the characters after this one as a comment, default to '' (disabled). - */ - comment?: OptionsNormalized["comment"] | boolean; - /** - * Restrict the definition of comments to a full line. Comment characters - * defined in the middle of the line are not interpreted as such. The - * option require the activation of comments. - */ - comment_no_infix?: OptionsNormalized["comment_no_infix"] | null; - /** - * Set the field delimiter. One character only, defaults to comma. - */ - delimiter?: OptionsNormalized["delimiter"] | string | string[] | Buffer; - /** - * Discover the field delimiter - */ - delimiter_auto?: boolean | Partial; - /** - * Set the source and destination encoding, a value of `null` returns buffer instead of strings. - */ - encoding?: OptionsNormalized["encoding"] | boolean | undefined; - /** - * Set the escape character, one character only, defaults to double quotes. - */ - escape?: OptionsNormalized["escape"] | string | boolean; - /** - * Start handling records from the requested number of records. - */ - from?: OptionsNormalized["from"] | string; - /** - * Start handling records from the requested line number. - */ - from_line?: OptionsNormalized["from_line"] | null | string; - fromLine?: OptionsNormalized["from_line"] | null | string; - /** - * Convert values into an array of values when columns are activated and - * when multiple columns of the same name are found. - */ - group_columns_by_name?: OptionsNormalized["group_columns_by_name"]; - groupColumnsByName?: OptionsNormalized["group_columns_by_name"]; - /** - * Don't interpret delimiters as such in the last field according to the number of fields calculated from the number of columns, the option require the presence of the `column` option when `true`. - */ - ignore_last_delimiters?: OptionsNormalized["ignore_last_delimiters"]; - /** - * Generate two properties `info` and `record` where `info` is a snapshot of the info object at the time the record was created and `record` is the parsed array or object. - */ - info?: OptionsNormalized["info"]; - /** - * If true, ignore whitespace immediately following the delimiter (i.e. left-trim all fields), defaults to false. - * Does not remove whitespace in a quoted field. - */ - ltrim?: OptionsNormalized["ltrim"] | null; - /** - * Maximum number of characters to be contained in the field and line buffers before an exception is raised, - * used to guard against a wrong delimiter or record_delimiter, - * default to 128000 characters. - */ - max_record_size?: OptionsNormalized["max_record_size"] | null | string; - maxRecordSize?: OptionsNormalized["max_record_size"]; - /** - * Name of header-record title to name objects by. - */ - objname?: OptionsNormalized["objname"] | Buffer | null; - /** - * Alter and filter records by executing a user defined function. - */ - on_record?: (record: U, context: InfoRecord) => T | null | undefined | U; - onRecord?: (record: U, context: InfoRecord) => T | null | undefined | U; - /** - * Function called when an error occurred if the `skip_records_with_error` - * option is activated. - */ - on_skip?: OptionsNormalized["on_skip"]; - onSkip?: OptionsNormalized["on_skip"]; - /** - * Optional character surrounding a field, one character only, defaults to double quotes. - */ - quote?: OptionsNormalized["quote"] | string | boolean; - /** - * Generate two properties raw and row where raw is the original CSV row content and row is the parsed array or object. - */ - raw?: OptionsNormalized["raw"] | null; - /** - * One or multiple characters used to delimit record rows; defaults to auto discovery if not provided. - * Supported auto discovery method are Linux ("\n"), Apple ("\r") and Windows ("\r\n") row delimiters. - */ - record_delimiter?: - | OptionsNormalized["record_delimiter"] - | string - | Buffer - | null - | (string | null)[]; - recordDelimiter?: - | OptionsNormalized["record_delimiter"] - | string - | Buffer - | null - | (string | null)[]; - /** - * Discard inconsistent columns count, default to false. - */ - relax_column_count?: OptionsNormalized["relax_column_count"] | null; - relaxColumnCount?: OptionsNormalized["relax_column_count"] | null; - /** - * Discard inconsistent columns count when the record contains less fields than expected, default to false. - */ - relax_column_count_less?: OptionsNormalized["relax_column_count_less"] | null; - relaxColumnCountLess?: OptionsNormalized["relax_column_count_less"] | null; - /** - * Discard inconsistent columns count when the record contains more fields than expected, default to false. - */ - relax_column_count_more?: OptionsNormalized["relax_column_count_more"] | null; - relaxColumnCountMore?: OptionsNormalized["relax_column_count_more"] | null; - /** - * Preserve quotes inside unquoted field. - */ - relax_quotes?: OptionsNormalized["relax_quotes"] | null; - relaxQuotes?: OptionsNormalized["relax_quotes"] | null; - /** - * If true, ignore whitespace immediately preceding the delimiter (i.e. right-trim all fields), defaults to false. - * Does not remove whitespace in a quoted field. - */ - rtrim?: OptionsNormalized["rtrim"] | null; - /** - * Dont generate empty values for empty lines. - * Defaults to false - */ - skip_empty_lines?: OptionsNormalized["skip_empty_lines"] | null; - skipEmptyLines?: OptionsNormalized["skip_empty_lines"] | null; - /** - * Don't generate records for lines containing empty column values (column matching /\s*\/), defaults to false. - */ - skip_records_with_empty_values?: - OptionsNormalized["skip_records_with_empty_values"] | null; - skipRecordsWithEmptyValues?: - OptionsNormalized["skip_records_with_empty_values"] | null; - /** - * Skip a line with error found inside and directly go process the next line. - */ - skip_records_with_error?: OptionsNormalized["skip_records_with_error"] | null; - skipRecordsWithError?: OptionsNormalized["skip_records_with_error"] | null; - /** - * Stop handling records after the requested number of records. - */ - to?: OptionsNormalized["to"] | null | string; - /** - * Stop handling records after the requested line number. - */ - to_line?: OptionsNormalized["to_line"] | null | string; - toLine?: OptionsNormalized["to_line"] | null | string; - /** - * If true, ignore whitespace immediately around the delimiter, defaults to false. - * Does not remove whitespace in a quoted field. - */ - trim?: OptionsNormalized["trim"] | null; -} - -export type OptionsWithColumns = Omit, "columns"> & { - columns: Exclude< - Options, NoInferColumnRecord>["columns"], - undefined | false - >; -}; - declare function parse( input: string | Buffer | Uint8Array, options: OptionsWithColumns, @@ -529,42 +61,6 @@ declare function parse(callback?: Callback): Parser; export { parse }; -/////////////////////////////////////////////////////// CsvError - -export type CsvErrorCode = - | "CSV_INVALID_ARGUMENT" - | "CSV_INVALID_CLOSING_QUOTE" - | "CSV_INVALID_COLUMN_DEFINITION" - | "CSV_INVALID_COLUMN_MAPPING" - | "CSV_INVALID_OPTION_BOM" - | "CSV_INVALID_OPTION_CAST" - | "CSV_INVALID_OPTION_CAST_DATE" - | "CSV_INVALID_OPTION_COLUMNS" - | "CSV_INVALID_OPTION_COMMENT" - | "CSV_INVALID_OPTION_DELIMITER" - | "CSV_INVALID_OPTION_GROUP_COLUMNS_BY_NAME" - | "CSV_INVALID_OPTION_ON_RECORD" - | "CSV_MAX_RECORD_SIZE" - | "CSV_NON_TRIMABLE_CHAR_AFTER_CLOSING_QUOTE" - | "CSV_OPTION_COLUMNS_MISSING_NAME" - | "CSV_QUOTE_NOT_CLOSED" - | "CSV_RECORD_INCONSISTENT_FIELDS_LENGTH" - | "CSV_RECORD_INCONSISTENT_COLUMNS" - | "CSV_UNKNOWN_ERROR" - | "INVALID_OPENING_QUOTE"; - -export class CsvError extends Error { - readonly code: CsvErrorCode; - [key: string]: unknown; - - constructor( - code: CsvErrorCode, - message: string | string[], - options?: OptionsNormalized, - ...contexts: unknown[] - ); -} - /////////////////////////////////////////////////////// normalize_options declare function normalize_options(opts: Options): OptionsNormalized; diff --git a/packages/csv-parse/lib/options.d.ts b/packages/csv-parse/lib/options.d.ts new file mode 100644 index 00000000..6a5ae688 --- /dev/null +++ b/packages/csv-parse/lib/options.d.ts @@ -0,0 +1,484 @@ +import * as stream from "stream"; + +import { CsvError } from "./api/CsvError.js"; + +export interface Info { + /** + * The number of processed bytes. + */ + readonly bytes: number; + /** + * The number of processed bytes until the last successfully parsed and emitted records. + */ + readonly bytes_records: number; + /** + * The number of lines being fully commented. + */ + readonly comment_lines: number; + /** + * The number of processed empty lines. + */ + readonly empty_lines: number; + /** + * The number of non uniform records when `relax_column_count` is true. + */ + readonly invalid_field_length: number; + /** + * The number of lines encountered in the source dataset, start at 1 for the first line. + */ + readonly lines: number; + /** + * The number of processed records. + */ + readonly records: number; +} + +export interface InfoCallback extends Info { + /** + * Normalized version of `options.columns` when `options.columns` is true, boolean otherwise. + */ + readonly columns: boolean | { name: string }[] | { disabled: true }[]; +} + +export interface InfoDataSet extends Info { + readonly column: number | string; +} + +export interface InfoRecord extends InfoDataSet { + readonly error: CsvError; + readonly header: boolean; + readonly index: number; + readonly raw: string | undefined; +} + +export interface InfoField extends InfoRecord { + readonly quoting: boolean; +} + +/** + * @deprecated Use the InfoField interface instead, the interface will disappear in future versions. + */ +// eslint-disable-next-line +export interface CastingContext extends InfoField {} + +export type CastingFunction = (value: string, context: InfoField) => unknown; + +export type CastingDateFunction = (value: string, context: InfoField) => Date; + +export type ColumnOption = + K | undefined | null | false | { name: K }; + +type ColumnKey = T extends string[] + ? string + : unknown extends T + ? string + : string | keyof T; + +// Keep columns from overriding record types inferred from options such as raw. +type NoInferColumnRecord = [T][T extends unknown ? 0 : never]; + +export interface ScoringFunctionInfo { + /** + * The character code of the delimiter candidate being scored. + */ + readonly char_code: number; + /** + * The number of occurrences of the candidate in each line. + */ + readonly lines: number[]; + /** + * Whether the candidate is listed in the `preferred` option. + */ + readonly preferred: boolean; + /** + * The standard deviation of the occurrences across the lines. + */ + readonly std: number; + /** + * The total number of occurrences of the candidate. + */ + readonly total: number; +} + +export type ScoringFunction = ( + info: ScoringFunctionInfo, + options: ScoringFunctionOptions, +) => number; + +export interface ScoringFunctionOptions { + preferred: Record; + score: ScoringFunction; + size: number; +} + +export interface OptionsNormalized { + auto_parse?: boolean | CastingFunction; + auto_parse_date?: boolean | CastingDateFunction; + /** + * If true, detect and exclude the byte order mark (BOM) from the CSV input if present. + */ + bom?: boolean; + /** + * If true, the parser will attempt to convert input string to native types. + * If a function, receive the value as first argument, a context as second argument and return a new value. More information about the context properties is available below. + */ + cast?: boolean | CastingFunction; + /** + * If true, the parser will attempt to convert input string to dates. + * If a function, receive the value as argument and return a new value. It requires the "auto_parse" option. Be careful, it relies on Date.parse. + */ + cast_date?: boolean | CastingDateFunction; + /** + * Internal property string the function to + */ + cast_first_line_to_header?: ( + record: string[], + ) => ColumnOption>[]; + /** + * List of fields as an array, a user defined callback accepting the first + * line and returning the column names or true if autodiscovered in the first + * CSV line, default to null, affect the result data set in the sense that + * records will be objects instead of arrays. The callback receives the raw + * header fields as strings, while returned names may use keys from the typed + * input record. + */ + columns: boolean | ColumnOption>[]; + /** + * Treat all the characters after this one as a comment, default to '' (disabled). + */ + comment: string | null; + /** + * Restrict the definition of comments to a full line. Comment characters + * defined in the middle of the line are not interpreted as such. The + * option require the activation of comments. + */ + comment_no_infix: boolean; + /** + * Set the field delimiter. One character only, defaults to comma. + */ + delimiter: Buffer[]; + /** + * Discover the field delimiter. + */ + delimiter_auto: ScoringFunctionOptions; + /** + * Set the source and destination encoding, a value of `null` returns buffer instead of strings. + */ + encoding: BufferEncoding | null; + /** + * Set the escape character, one character only, defaults to double quotes. + */ + escape: null | Buffer; + /** + * Start handling records from the requested number of records. + */ + from: number; + /** + * Start handling records from the requested line number. + */ + from_line: number; + /** + * Convert values into an array of values when columns are activated and + * when multiple columns of the same name are found. + */ + group_columns_by_name: boolean; + /** + * Don't interpret delimiters as such in the last field according to the number of fields calculated from the number of columns, the option require the presence of the `column` option when `true`. + */ + ignore_last_delimiters: boolean | number; + /** + * Generate two properties `info` and `record` where `info` is a snapshot of the info object at the time the record was created and `record` is the parsed array or object. + */ + info: boolean; + /** + * If true, ignore whitespace immediately following the delimiter (i.e. left-trim all fields), defaults to false. + * Does not remove whitespace in a quoted field. + */ + ltrim: boolean; + /** + * Maximum number of characters to be contained in the field and line buffers before an exception is raised, + * used to guard against a wrong delimiter or record_delimiter, + * default to 128000 characters. + */ + max_record_size: number; + /** + * Name of header-record title to name objects by. + */ + objname: number | string | undefined; + /** + * Alter and filter records by executing a user defined function. + */ + on_record?: (record: U, context: InfoRecord) => T | null | undefined; + /** + * Function called when an error occurred if the `skip_records_with_error` + * option is activated. + */ + on_skip?: (err: CsvError | undefined, raw: string | undefined) => undefined; + /** + * Optional character surrounding a field, one character only, defaults to double quotes. + */ + quote?: Buffer | null; + /** + * Generate two properties raw and row where raw is the original CSV row content and row is the parsed array or object. + */ + raw: boolean; + /** + * One or multiple characters used to delimit record rows; defaults to auto discovery if not provided. + * Supported auto discovery method are Linux ("\n"), Apple ("\r") and Windows ("\r\n") row delimiters. + */ + record_delimiter: Buffer[]; + /** + * Discard inconsistent columns count, default to false. + */ + relax_column_count: boolean; + /** + * Discard inconsistent columns count when the record contains less fields than expected, default to false. + */ + relax_column_count_less: boolean; + /** + * Discard inconsistent columns count when the record contains more fields than expected, default to false. + */ + relax_column_count_more: boolean; + /** + * Preserve quotes inside unquoted field. + */ + relax_quotes: boolean; + /** + * If true, ignore whitespace immediately preceding the delimiter (i.e. right-trim all fields), defaults to false. + * Does not remove whitespace in a quoted field. + */ + rtrim: boolean; + /** + * Dont generate empty values for empty lines. + * Defaults to false + */ + skip_empty_lines: boolean; + /** + * Don't generate records for lines containing empty column values (column matching /\s*\/), defaults to false. + */ + skip_records_with_empty_values: boolean; + /** + * Skip a line with error found inside and directly go process the next line. + */ + skip_records_with_error: boolean; + /** + * Stop handling records after the requested number of records. + */ + to: number; + /** + * Stop handling records after the requested line number. + */ + to_line: number; + /** + * If true, ignore whitespace immediately around the delimiter, defaults to false. + * Does not remove whitespace in a quoted field. + */ + trim: boolean; +} + +// Keep the parser's encoding options instead of the narrower stream encoding. +export interface Options extends Omit< + stream.TransformOptions, + "encoding" +> { + /** + * If true, the parser will attempt to convert read data types to native types. + * @deprecated Use {@link cast} + */ + auto_parse?: boolean | CastingFunction; + autoParse?: boolean | CastingFunction; + /** + * If true, the parser will attempt to convert read data types to dates. It requires the "auto_parse" option. + * @deprecated Use {@link cast_date} + */ + auto_parse_date?: boolean | CastingDateFunction; + autoParseDate?: boolean | CastingDateFunction; + /** + * If true, detect and exclude the byte order mark (BOM) from the CSV input if present. + */ + bom?: OptionsNormalized["bom"]; + /** + * If true, the parser will attempt to convert input string to native types. + * If a function, receive the value as first argument, a context as second argument and return a new value. More information about the context properties is available below. + */ + cast?: OptionsNormalized["cast"]; + /** + * If true, the parser will attempt to convert input string to dates. + * If a function, receive the value as argument and return a new value. It requires the "auto_parse" option. Be careful, it relies on Date.parse. + */ + cast_date?: OptionsNormalized["cast_date"]; + castDate?: OptionsNormalized["cast_date"]; + /** + * List of fields as an array, + * a user defined callback accepting the first line and returning the column names or true if autodiscovered in the first CSV line, + * default to null, + * affect the result data set in the sense that records will be objects instead of arrays. The callback receives the raw header fields as strings, while returned names may use keys from the typed input record. + */ + columns?: + | OptionsNormalized["columns"] + | ((record: string[]) => ColumnOption>[]); + /** + * Treat all the characters after this one as a comment, default to '' (disabled). + */ + comment?: OptionsNormalized["comment"] | boolean; + /** + * Restrict the definition of comments to a full line. Comment characters + * defined in the middle of the line are not interpreted as such. The + * option require the activation of comments. + */ + comment_no_infix?: OptionsNormalized["comment_no_infix"] | null; + /** + * Set the field delimiter. One character only, defaults to comma. + */ + delimiter?: OptionsNormalized["delimiter"] | string | string[] | Buffer; + /** + * Discover the field delimiter + */ + delimiter_auto?: boolean | Partial; + /** + * Set the source and destination encoding, a value of `null` returns buffer instead of strings. + */ + encoding?: OptionsNormalized["encoding"] | boolean | undefined; + /** + * Set the escape character, one character only, defaults to double quotes. + */ + escape?: OptionsNormalized["escape"] | string | boolean; + /** + * Start handling records from the requested number of records. + */ + from?: OptionsNormalized["from"] | string; + /** + * Start handling records from the requested line number. + */ + from_line?: OptionsNormalized["from_line"] | null | string; + fromLine?: OptionsNormalized["from_line"] | null | string; + /** + * Convert values into an array of values when columns are activated and + * when multiple columns of the same name are found. + */ + group_columns_by_name?: OptionsNormalized["group_columns_by_name"]; + groupColumnsByName?: OptionsNormalized["group_columns_by_name"]; + /** + * Don't interpret delimiters as such in the last field according to the number of fields calculated from the number of columns, the option require the presence of the `column` option when `true`. + */ + ignore_last_delimiters?: OptionsNormalized["ignore_last_delimiters"]; + /** + * Generate two properties `info` and `record` where `info` is a snapshot of the info object at the time the record was created and `record` is the parsed array or object. + */ + info?: OptionsNormalized["info"]; + /** + * If true, ignore whitespace immediately following the delimiter (i.e. left-trim all fields), defaults to false. + * Does not remove whitespace in a quoted field. + */ + ltrim?: OptionsNormalized["ltrim"] | null; + /** + * Maximum number of characters to be contained in the field and line buffers before an exception is raised, + * used to guard against a wrong delimiter or record_delimiter, + * default to 128000 characters. + */ + max_record_size?: OptionsNormalized["max_record_size"] | null | string; + maxRecordSize?: OptionsNormalized["max_record_size"]; + /** + * Name of header-record title to name objects by. + */ + objname?: OptionsNormalized["objname"] | Buffer | null; + /** + * Alter and filter records by executing a user defined function. + */ + on_record?: (record: U, context: InfoRecord) => T | null | undefined | U; + onRecord?: (record: U, context: InfoRecord) => T | null | undefined | U; + /** + * Function called when an error occurred if the `skip_records_with_error` + * option is activated. + */ + on_skip?: OptionsNormalized["on_skip"]; + onSkip?: OptionsNormalized["on_skip"]; + /** + * Optional character surrounding a field, one character only, defaults to double quotes. + */ + quote?: OptionsNormalized["quote"] | string | boolean; + /** + * Generate two properties raw and row where raw is the original CSV row content and row is the parsed array or object. + */ + raw?: OptionsNormalized["raw"] | null; + /** + * One or multiple characters used to delimit record rows; defaults to auto discovery if not provided. + * Supported auto discovery method are Linux ("\n"), Apple ("\r") and Windows ("\r\n") row delimiters. + */ + record_delimiter?: + | OptionsNormalized["record_delimiter"] + | string + | Buffer + | null + | (string | null)[]; + recordDelimiter?: + | OptionsNormalized["record_delimiter"] + | string + | Buffer + | null + | (string | null)[]; + /** + * Discard inconsistent columns count, default to false. + */ + relax_column_count?: OptionsNormalized["relax_column_count"] | null; + relaxColumnCount?: OptionsNormalized["relax_column_count"] | null; + /** + * Discard inconsistent columns count when the record contains less fields than expected, default to false. + */ + relax_column_count_less?: OptionsNormalized["relax_column_count_less"] | null; + relaxColumnCountLess?: OptionsNormalized["relax_column_count_less"] | null; + /** + * Discard inconsistent columns count when the record contains more fields than expected, default to false. + */ + relax_column_count_more?: OptionsNormalized["relax_column_count_more"] | null; + relaxColumnCountMore?: OptionsNormalized["relax_column_count_more"] | null; + /** + * Preserve quotes inside unquoted field. + */ + relax_quotes?: OptionsNormalized["relax_quotes"] | null; + relaxQuotes?: OptionsNormalized["relax_quotes"] | null; + /** + * If true, ignore whitespace immediately preceding the delimiter (i.e. right-trim all fields), defaults to false. + * Does not remove whitespace in a quoted field. + */ + rtrim?: OptionsNormalized["rtrim"] | null; + /** + * Dont generate empty values for empty lines. + * Defaults to false + */ + skip_empty_lines?: OptionsNormalized["skip_empty_lines"] | null; + skipEmptyLines?: OptionsNormalized["skip_empty_lines"] | null; + /** + * Don't generate records for lines containing empty column values (column matching /\s*\/), defaults to false. + */ + skip_records_with_empty_values?: + OptionsNormalized["skip_records_with_empty_values"] | null; + skipRecordsWithEmptyValues?: + OptionsNormalized["skip_records_with_empty_values"] | null; + /** + * Skip a line with error found inside and directly go process the next line. + */ + skip_records_with_error?: OptionsNormalized["skip_records_with_error"] | null; + skipRecordsWithError?: OptionsNormalized["skip_records_with_error"] | null; + /** + * Stop handling records after the requested number of records. + */ + to?: OptionsNormalized["to"] | null | string; + /** + * Stop handling records after the requested line number. + */ + to_line?: OptionsNormalized["to_line"] | null | string; + toLine?: OptionsNormalized["to_line"] | null | string; + /** + * If true, ignore whitespace immediately around the delimiter, defaults to false. + * Does not remove whitespace in a quoted field. + */ + trim?: OptionsNormalized["trim"] | null; +} + +export type OptionsWithColumns = Omit, "columns"> & { + columns: Exclude< + Options, NoInferColumnRecord>["columns"], + undefined | false + >; +}; diff --git a/packages/csv-parse/package.json b/packages/csv-parse/package.json index 7232e89f..9aafb77c 100644 --- a/packages/csv-parse/package.json +++ b/packages/csv-parse/package.json @@ -111,7 +111,7 @@ "scripts": { "build": "npm run build:rollup && npm run build:ts", "build:rollup": "rollup -c", - "build:ts": "cp lib/index.d.ts dist/cjs/index.d.cts && cp lib/sync.d.ts dist/cjs/sync.d.cts && cp lib/stream.d.ts dist/cjs/stream.d.cts && cp lib/*.ts dist/esm", + "build:ts": "mkdir -p dist/cjs/api dist/esm/api && cp lib/index.d.ts dist/cjs/index.d.cts && cp lib/options.d.ts dist/cjs/options.d.cts && cp lib/sync.d.ts dist/cjs/sync.d.cts && cp lib/stream.d.ts dist/cjs/stream.d.cts && cp lib/api/CsvError.d.ts dist/cjs/api/CsvError.d.cts && cp lib/*.ts dist/esm && cp lib/api/*.ts dist/esm/api", "postbuild:ts": "find dist/cjs -name '*.d.cts' -exec sh -c \"sed -i \"s/\\.js'/\\.cjs'/g\" {} || sed -i '' \"s/\\.js'/\\.cjs'/g\" {}\" \\;", "lint:check": "eslint", "lint:fix": "eslint --fix", From 8c42729043a007bc4663f990b67575343384677e Mon Sep 17 00:00:00 2001 From: David Worms Date: Thu, 1 Oct 2026 00:05:03 +0200 Subject: [PATCH 3/3] feat(csv-parse): local stream type --- packages/csv-parse/dist/cjs/index.d.cts | 22 +++++++++++++++++----- packages/csv-parse/dist/cjs/options.d.cts | 7 +------ packages/csv-parse/dist/esm/index.d.ts | 22 +++++++++++++++++----- packages/csv-parse/dist/esm/options.d.ts | 7 +------ packages/csv-parse/lib/index.d.ts | 22 +++++++++++++++++----- packages/csv-parse/lib/options.d.ts | 7 +------ 6 files changed, 54 insertions(+), 33 deletions(-) diff --git a/packages/csv-parse/dist/cjs/index.d.cts b/packages/csv-parse/dist/cjs/index.d.cts index 17f50217..0cbcac0f 100644 --- a/packages/csv-parse/dist/cjs/index.d.cts +++ b/packages/csv-parse/dist/cjs/index.d.cts @@ -4,18 +4,31 @@ import * as stream from "stream"; import type { CsvError } from "./api/CsvError.cjs"; +export type * from "./options.cjs"; import type { // Info Info, InfoCallback, // Options - OptionsWithColumns, - OptionsNormalized, - Options, + Options as OptionsOriginal, + OptionsNormalized as OptionsNormalizedOriginal, + OptionsWithColumns as OptionsWithColumnsOriginal, } from "./options.cjs"; export * from "./api/CsvError.cjs"; -export type * from "./options.cjs"; + +export interface Options + extends OptionsOriginal, Omit {} + +export interface OptionsNormalized + extends + OptionsNormalizedOriginal, + Omit {} + +export interface OptionsWithColumns + extends + OptionsWithColumnsOriginal, + Omit {} export type Callback = ( err: CsvError | undefined, @@ -35,7 +48,6 @@ export class Parser extends stream.Transform { readonly info: Info; } - declare function parse( input: string | Buffer | Uint8Array, options: OptionsWithColumns, diff --git a/packages/csv-parse/dist/cjs/options.d.cts b/packages/csv-parse/dist/cjs/options.d.cts index daf04a5c..75c14567 100644 --- a/packages/csv-parse/dist/cjs/options.d.cts +++ b/packages/csv-parse/dist/cjs/options.d.cts @@ -1,5 +1,3 @@ -import * as stream from "stream"; - import { CsvError } from "./api/CsvError.cjs"; export interface Info { @@ -277,10 +275,7 @@ export interface OptionsNormalized { } // Keep the parser's encoding options instead of the narrower stream encoding. -export interface Options extends Omit< - stream.TransformOptions, - "encoding" -> { +export interface Options { /** * If true, the parser will attempt to convert read data types to native types. * @deprecated Use {@link cast} diff --git a/packages/csv-parse/dist/esm/index.d.ts b/packages/csv-parse/dist/esm/index.d.ts index 8799c00d..53e6e7d7 100644 --- a/packages/csv-parse/dist/esm/index.d.ts +++ b/packages/csv-parse/dist/esm/index.d.ts @@ -4,18 +4,31 @@ import * as stream from "stream"; import type { CsvError } from "./api/CsvError.js"; +export type * from "./options.js"; import type { // Info Info, InfoCallback, // Options - OptionsWithColumns, - OptionsNormalized, - Options, + Options as OptionsOriginal, + OptionsNormalized as OptionsNormalizedOriginal, + OptionsWithColumns as OptionsWithColumnsOriginal, } from "./options.js"; export * from "./api/CsvError.js"; -export type * from "./options.js"; + +export interface Options + extends OptionsOriginal, Omit {} + +export interface OptionsNormalized + extends + OptionsNormalizedOriginal, + Omit {} + +export interface OptionsWithColumns + extends + OptionsWithColumnsOriginal, + Omit {} export type Callback = ( err: CsvError | undefined, @@ -35,7 +48,6 @@ export class Parser extends stream.Transform { readonly info: Info; } - declare function parse( input: string | Buffer | Uint8Array, options: OptionsWithColumns, diff --git a/packages/csv-parse/dist/esm/options.d.ts b/packages/csv-parse/dist/esm/options.d.ts index 6a5ae688..81f77b94 100644 --- a/packages/csv-parse/dist/esm/options.d.ts +++ b/packages/csv-parse/dist/esm/options.d.ts @@ -1,5 +1,3 @@ -import * as stream from "stream"; - import { CsvError } from "./api/CsvError.js"; export interface Info { @@ -277,10 +275,7 @@ export interface OptionsNormalized { } // Keep the parser's encoding options instead of the narrower stream encoding. -export interface Options extends Omit< - stream.TransformOptions, - "encoding" -> { +export interface Options { /** * If true, the parser will attempt to convert read data types to native types. * @deprecated Use {@link cast} diff --git a/packages/csv-parse/lib/index.d.ts b/packages/csv-parse/lib/index.d.ts index 8799c00d..53e6e7d7 100644 --- a/packages/csv-parse/lib/index.d.ts +++ b/packages/csv-parse/lib/index.d.ts @@ -4,18 +4,31 @@ import * as stream from "stream"; import type { CsvError } from "./api/CsvError.js"; +export type * from "./options.js"; import type { // Info Info, InfoCallback, // Options - OptionsWithColumns, - OptionsNormalized, - Options, + Options as OptionsOriginal, + OptionsNormalized as OptionsNormalizedOriginal, + OptionsWithColumns as OptionsWithColumnsOriginal, } from "./options.js"; export * from "./api/CsvError.js"; -export type * from "./options.js"; + +export interface Options + extends OptionsOriginal, Omit {} + +export interface OptionsNormalized + extends + OptionsNormalizedOriginal, + Omit {} + +export interface OptionsWithColumns + extends + OptionsWithColumnsOriginal, + Omit {} export type Callback = ( err: CsvError | undefined, @@ -35,7 +48,6 @@ export class Parser extends stream.Transform { readonly info: Info; } - declare function parse( input: string | Buffer | Uint8Array, options: OptionsWithColumns, diff --git a/packages/csv-parse/lib/options.d.ts b/packages/csv-parse/lib/options.d.ts index 6a5ae688..81f77b94 100644 --- a/packages/csv-parse/lib/options.d.ts +++ b/packages/csv-parse/lib/options.d.ts @@ -1,5 +1,3 @@ -import * as stream from "stream"; - import { CsvError } from "./api/CsvError.js"; export interface Info { @@ -277,10 +275,7 @@ export interface OptionsNormalized { } // Keep the parser's encoding options instead of the narrower stream encoding. -export interface Options extends Omit< - stream.TransformOptions, - "encoding" -> { +export interface Options { /** * If true, the parser will attempt to convert read data types to native types. * @deprecated Use {@link cast}