diff --git a/packages/bundle-size/README.md b/packages/bundle-size/README.md index 592779d98..64070dc8c 100644 --- a/packages/bundle-size/README.md +++ b/packages/bundle-size/README.md @@ -16,11 +16,11 @@ usually do. We repeat this for an increasing number of files. | code generator | files | bundle size | minified | compressed | | ------------------- | ----: | ----------: | --------: | ---------: | -| Protobuf-ES | 1 | 136,220 b | 67,700 b | 15,940 b | -| Protobuf-ES | 4 | 138,409 b | 69,204 b | 16,599 b | -| Protobuf-ES | 8 | 141,167 b | 70,975 b | 17,177 b | -| Protobuf-ES | 16 | 151,601 b | 78,958 b | 19,449 b | -| Protobuf-ES | 32 | 179,388 b | 100,981 b | 24,966 b | +| Protobuf-ES | 1 | 136,343 b | 67,792 b | 15,926 b | +| Protobuf-ES | 4 | 138,532 b | 69,296 b | 16,641 b | +| Protobuf-ES | 8 | 141,290 b | 71,067 b | 17,155 b | +| Protobuf-ES | 16 | 151,724 b | 79,050 b | 19,498 b | +| Protobuf-ES | 32 | 179,511 b | 101,073 b | 24,962 b | | protobuf-javascript | 1 | 314,172 b | 244,057 b | 36,091 b | | protobuf-javascript | 4 | 340,189 b | 259,029 b | 37,458 b | | protobuf-javascript | 8 | 360,983 b | 270,606 b | 38,596 b | diff --git a/packages/bundle-size/chart.svg b/packages/bundle-size/chart.svg index cf453fa26..772091dc3 100644 --- a/packages/bundle-size/chart.svg +++ b/packages/bundle-size/chart.svg @@ -44,14 +44,14 @@ 0 KiB - + Protobuf-ES -Protobuf-ES 15.57 KiB for 1 files -Protobuf-ES 16.21 KiB for 4 files -Protobuf-ES 16.77 KiB for 8 files -Protobuf-ES 18.99 KiB for 16 files -Protobuf-ES 24.38 KiB for 32 files +Protobuf-ES 15.55 KiB for 1 files +Protobuf-ES 16.25 KiB for 4 files +Protobuf-ES 16.75 KiB for 8 files +Protobuf-ES 19.04 KiB for 16 files +Protobuf-ES 24.38 KiB for 32 files diff --git a/packages/protobuf-test/src/wire/binary-encoding.test.ts b/packages/protobuf-test/src/wire/binary-encoding.test.ts index f84708654..757d90b8a 100644 --- a/packages/protobuf-test/src/wire/binary-encoding.test.ts +++ b/packages/protobuf-test/src/wire/binary-encoding.test.ts @@ -327,4 +327,53 @@ void suite("BinaryReader", () => { assert.strictEqual(wireType, WireType.Varint); }); }); + void suite("string", () => { + // The ASCII fast path covers strings of up to 32 bytes; longer ones and + // any string with a byte above 0x7f go to the UTF-8 decoder. + const cases: [string, string][] = [ + ["an empty string", ""], + ["a one-byte ASCII string", "a"], + ["the longest ASCII string on the fast path", "x".repeat(32)], + ["the shortest ASCII string past the fast path", "x".repeat(33)], + ["a non-ASCII byte at the end of the fast path", `${"x".repeat(30)}é`], + ["a non-ASCII byte at the start", `é${"x".repeat(29)}`], + ["multi-byte characters only", "日本語"], + ["a long ASCII string", "x".repeat(200)], + ]; + for (const [name, value] of cases) { + void test(`reads ${name}`, () => { + const bytes = new BinaryWriter().string(value).string("next").finish(); + const reader = new BinaryReader(bytes); + assert.strictEqual(reader.string(), value); + // The position must land exactly after the string, also when the + // fast path bails out to the decoder part-way through. + assert.strictEqual(reader.string(), "next"); + assert.strictEqual(reader.pos, reader.len); + }); + } + void test("reads from a view that does not start at offset 0", () => { + // A reader over a subarray indexes relative to the view, not to the + // underlying buffer. + const inner = new BinaryWriter().string("abc").string("dé").finish(); + const outer = new Uint8Array(inner.length + 7).fill(0x41); + outer.set(inner, 5); + const reader = new BinaryReader(outer.subarray(5, 5 + inner.length)); + assert.strictEqual(reader.string(), "abc"); + assert.strictEqual(reader.string(), "dé"); + }); + void test("rejects a length beyond the end of the data", () => { + // Length prefix 5, but only 3 bytes follow. + const reader = new BinaryReader(new Uint8Array([5, 0x61, 0x62, 0x63])); + assert.throws(() => reader.string(), { + name: "RangeError", + message: "premature EOF", + }); + }); + void test("throws on invalid UTF-8 in strict mode only", () => { + // 0xff never occurs in UTF-8; short, so it starts on the fast path. + const bytes = new Uint8Array([2, 0x61, 0xff]); + assert.strictEqual(new BinaryReader(bytes).string(), "a�"); + assert.throws(() => new BinaryReader(bytes).string(true)); + }); + }); }); diff --git a/packages/protobuf/src/wire/binary-encoding.ts b/packages/protobuf/src/wire/binary-encoding.ts index fbd4fe5a7..66ea84be3 100644 --- a/packages/protobuf/src/wire/binary-encoding.ts +++ b/packages/protobuf/src/wire/binary-encoding.ts @@ -780,23 +780,27 @@ export class BinaryReader { * `strict` is true, throw on invalid UTF-8 instead of substituting U+FFFD. */ string(strict?: boolean): string { - const bytes = this.bytes(); - const len = bytes.length; + // Like bytes(), but the ASCII fast path reads the buffer without a view. + const len = this.uint32(); + const start = this.pos; + this.pos += len; + this.assertBounds(); + const buf = this.buf; // Fast path for ASCII. if (len <= ASCII_MAX_LENGTH) { const codes = new Array(len); for (let i = 0; i < len; i++) { - const byte = bytes[i]; + const byte = buf[start + i]; if (byte > 0x7f) { - return this.decodeUtf8(bytes, strict); + return this.decodeUtf8(buf.subarray(start, this.pos), strict); } codes[i] = byte; } return String.fromCharCode.apply(String, codes); } - return this.decodeUtf8(bytes, strict); + return this.decodeUtf8(buf.subarray(start, this.pos), strict); } }