Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
3 changes: 2 additions & 1 deletion README.md
Original file line number Diff line number Diff line change
Expand Up @@ -30,8 +30,9 @@ Parse a `Content-Type` header. This will return an object with the following pro

- `type`: The media type. Example: `'image/svg+xml'`.
- `parameters`: An object of the parameters in the media type (parameter name is always lower case). Example: `{charset: 'utf-8'}`.
- `index`: The index where parsing stopped. Example: `33`.

The parser is lenient and does not error. You should validate `type` and `parameters` before trusting them.
The parser is lenient and does not validate or throw on malformed input.

#### Options

Expand Down
25 changes: 25 additions & 0 deletions src/index.bench.ts
Original file line number Diff line number Diff line change
Expand Up @@ -3,7 +3,12 @@ import { format, parse } from "./index.js";

describe("parse", () => {
const BASIC_HEADER = "text/html";
const SINGLE_PARAM_HEADER = "application/json; charset=utf-8";
const UPPERCASE_PARAM_HEADER = "APPLICATION/JSON; CHARSET=utf-8";
const ACCEPT_HEADER =
"text/html; charset=utf-8, application/json;q=0.9, */*;q=0.8";
const PARAMS_HEADER = "application/json; charset=utf-8; foo=bar; version=1";
const QUOTED_SIMPLE_HEADER = 'application/json; charset="utf-8"';
const QUOTED_HEADER =
'text/plain; filename="report\\"-2026.csv"; foo=bar; version=1';
const OWS_HEADER =
Expand All @@ -21,6 +26,26 @@ describe("parse", () => {
parse(PARAMS_HEADER);
});

bench("single parameter", () => {
parse(SINGLE_PARAM_HEADER);
});

bench("uppercase single parameter", () => {
parse(UPPERCASE_PARAM_HEADER);
});

bench("accept header (comma = true)", () => {
parse(ACCEPT_HEADER, { comma: true });
});

bench("accept header (comma = false)", () => {
parse(ACCEPT_HEADER);
});

bench("simple quoted parameter", () => {
parse(QUOTED_SIMPLE_HEADER);
});

bench("simple parameters (options.parameters = false)", () => {
parse(PARAMS_HEADER, { parameters: false });
});
Expand Down
223 changes: 148 additions & 75 deletions src/index.ts
Original file line number Diff line number Diff line change
Expand Up @@ -22,6 +22,39 @@ const QUOTE_REGEXP = /[\\"]/g;
const TYPE_REGEXP =
/^[!#$%&'*+.^_`|~0-9A-Za-z-]+\/[!#$%&'*+.^_`|~0-9A-Za-z-]+$/;

const SP = 32; // " "
const HTAB = 9; // "\t"
const SEMI = 59; // ";"
const EQ = 61; // "="
const DQUOTE = 34; // '"'
const BSLASH = 92; // "\\"
const COMMA = 44; // ","

const LOWER_CASE = 1;
const OWS = 2;
const SEMI_FLAG = 4;
const COMMA_FLAG = 8;
const NON_ASCII = 0xff00;
const CASE_FLAGS = LOWER_CASE | NON_ASCII;

/**
* Character flags used to normalize HTTP field values while scanning.
* Out-of-range reads intentionally coerce to zero in bitwise expressions.
*/
const CHAR_MAP = new Uint8Array(0x100);

for (let code = 0x41 /* A */; code <= 0x5a /* Z */; code++) {
CHAR_MAP[code] |= LOWER_CASE;
}

CHAR_MAP[HTAB] |= OWS;
CHAR_MAP[SP] |= OWS;
CHAR_MAP[SEMI] |= SEMI_FLAG;
CHAR_MAP[COMMA] |= COMMA_FLAG;
for (let code = 0x80 /* non-ASCII */; code <= 0xff; code++) {
CHAR_MAP[code] |= LOWER_CASE;
}

/**
* Null object perf optimization. Faster than `Object.create(null)` and `{ __proto__: null }`.
*/
Expand Down Expand Up @@ -95,30 +128,45 @@ export interface ParseOptions {
* Parse a `Content-Type` header.
*/
export function parse(header: string, options?: ParseOptions): ContentType {
const stopChar = options?.comma === true ? COMMA : 65_536; // Sentinel for "no stop char".
const stopFlags = SEMI_FLAG | (options?.comma === true ? COMMA_FLAG : 0);
const len = header.length;
let index = skipOWS(header, options?.start ?? 0, len);
let valueStart = options?.start ?? 0;
while ((CHAR_MAP[header.charCodeAt(valueStart)] & OWS) !== 0) {
valueStart++;
}

const valueStart = index;
index = skipValue(header, index, len, stopChar);
const valueEnd = trailingOWS(header, valueStart, index);
const type = header.slice(valueStart, valueEnd).toLowerCase();
let index = valueStart;
let typeFlags = 0;
let whitespace = -1;
let stop = options?.parameters === false ? COMMA_FLAG : 0;
while (index < len) {
const code = header.charCodeAt(index);
const flags = CHAR_MAP[code];
if ((flags & stopFlags) !== 0) {
stop |= flags & COMMA_FLAG;
break;
}

if ((flags & OWS) !== 0) {
if (whitespace === -1) whitespace = index;
} else {
whitespace = -1;
}

if (options?.parameters === false) {
typeFlags |= (code & NON_ASCII) | flags;
index++;
}
const valueEnd = whitespace === -1 ? index : whitespace;
const value = header.slice(valueStart, valueEnd);
const type = (typeFlags & CASE_FLAGS) === 0 ? value : value.toLowerCase();

Copy link
Copy Markdown
Member Author

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

Skipping toLowerCase was the largest perf win when the value is already lower cased.


if (index === len || stop !== 0) {
return { type, index, parameters: new NullObject() };
}

return parseParameters(header, type, index, len, stopChar);
return parseParameters(header, type, index, len, stopFlags);
}

const SP = 32; // " "
const HTAB = 9; // "\t"
const SEMI = 59; // ";"
const EQ = 61; // "="
const DQUOTE = 34; // '"'
const BSLASH = 92; // "\\"
const COMMA = 44; // ","

/**
* Parses the parameters of a `Content-Type` header starting at the given index.
*/
Expand All @@ -127,63 +175,117 @@ function parseParameters(
type: string,
index: number,
len: number,
stopChar: number,
stopFlags: number,
): ContentType {
const parameters: Record<string, string> = new NullObject();

parameter: while (index < len) {
if (header.charCodeAt(index) === stopChar) break;

index = skipOWS(header, index + 1 /* Skip over ; */, len);
index++; // Skip over ;
while ((CHAR_MAP[header.charCodeAt(index)] & OWS) !== 0) {
index++;
}

const keyStart = index;
let keyFlags = 0;
let keyWhitespace = -1;

while (index < len) {
const code = header.charCodeAt(index);
if (code === stopChar) break parameter;

if (code === SEMI) continue parameter;
const flags = CHAR_MAP[code];
if ((flags & stopFlags) !== 0) {
if (flags === COMMA_FLAG) break parameter;
continue parameter;
}

if (code === EQ) {
const keyEnd = trailingOWS(header, keyStart, index);
const key = header.slice(keyStart, keyEnd).toLowerCase();
const keyEnd = keyWhitespace === -1 ? index : keyWhitespace;
const value = header.slice(keyStart, keyEnd);
const key = (keyFlags & CASE_FLAGS) === 0 ? value : value.toLowerCase();

index = skipOWS(header, index + 1, len);
index++;
while ((CHAR_MAP[header.charCodeAt(index)] & OWS) !== 0) {
index++;
}

if (index < len && header.charCodeAt(index) === DQUOTE) {
index++;
const quotedStart = ++index;
let escaped = false;

let value = "";
while (index < len) {
const code = header.charCodeAt(index++);
const code = header.charCodeAt(index);
if (code === DQUOTE) {
index = skipValue(header, index, len, stopChar);
if (parameters[key] === undefined) parameters[key] = value;
break;
if (parameters[key] === undefined) {
parameters[key] = escaped
? unescapeQuotedPairs(header, quotedStart, index)
: header.slice(quotedStart, index);
}

index++;
let stop = 0;

// Discard characters between quote and delimiter.
while (index < len) {
const code = header.charCodeAt(index);
const flags = CHAR_MAP[code];
if ((flags & stopFlags) !== 0) {
stop = flags & COMMA_FLAG;
break;
}
index++;
}

if (stop !== 0) break parameter;
continue parameter;
}

if (code === BSLASH && index < len) {
value += header[index++];
if (code === BSLASH && index + 1 < len) {
escaped = true;
index += 2;
continue;
}

value += String.fromCharCode(code);

Copy link
Copy Markdown
Member Author

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

Unquoting seemed like a large perf improvement with the right conditions (most won't have escapes, turns into a single slice).

index++;
}

continue parameter;
}

const valueStart = index;
index = skipValue(header, index, len, stopChar);
let stop = 0;
let valueWhitespace = -1;
while (index < len) {
const code = header.charCodeAt(index);
const flags = CHAR_MAP[code];
if ((flags & stopFlags) !== 0) {
stop = flags & COMMA_FLAG;
break;
}

if ((flags & OWS) !== 0) {
if (valueWhitespace === -1) valueWhitespace = index;
} else {
valueWhitespace = -1;
}

index++;
}

if (parameters[key] === undefined) {
const valueEnd = trailingOWS(header, valueStart, index);
const valueEnd = valueWhitespace === -1 ? index : valueWhitespace;
parameters[key] = header.slice(valueStart, valueEnd);
}

if (stop !== 0) break parameter;
continue parameter;
}

if ((flags & OWS) !== 0) {
if (keyWhitespace === -1) keyWhitespace = index;
} else {
keyWhitespace = -1;
}

keyFlags |= (code & NON_ASCII) | flags;
index++;
}
}
Expand All @@ -192,48 +294,19 @@ function parseParameters(
}

/**
* Skip over characters until a semicolon or other exit character.
* Remove backslashes from quoted pairs in a known-terminated quoted string body.
*/
function skipValue(

Copy link
Copy Markdown
Member Author

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

Inlined skipValue, skipOWS and removing trailingOWS in favor of tracking the last whitespace showed marginal improvements to performance (a few % each).

str: string,
index: number,
len: number,
stopChar: number,
): number {
while (index < len) {
const code = str.charCodeAt(index);
if (code === SEMI || code === stopChar) break;
index++;
}
return index;
}
function unescapeQuotedPairs(str: string, start: number, end: number): string {
let result = "";

/**
* Skip optional whitespace (OWS) in an HTTP header value.
*
* OWS is defined in RFC 9110 sec 5.6.3 as SP (" ") or HTAB ("\t").
*/
function skipOWS(header: string, index: number, len: number): number {
while (index < len) {
const char = header.charCodeAt(index);
if (char !== SP && char !== HTAB) break;
index++;
for (let index = start; index < end; index++) {
if (str.charCodeAt(index) === BSLASH) {
result += str.slice(start, index);
start = ++index;
}
}
return index;
}

/**
* Trim optional whitespace (OWS) from the end of a substring.
*
* OWS is defined in RFC 9110 sec 5.6.3 as SP (" ") or HTAB ("\t").
*/
function trailingOWS(header: string, start: number, end: number): number {
while (end > start) {
const char = header.charCodeAt(end - 1);
if (char !== SP && char !== HTAB) break;
end--;
}
return end;
return result + str.slice(start, end);
}

/**
Expand Down
Loading
Loading