diff --git a/packages/bundle-size/README.md b/packages/bundle-size/README.md
index 592779d98..64070dc8c 100644
--- a/packages/bundle-size/README.md
+++ b/packages/bundle-size/README.md
@@ -16,11 +16,11 @@ usually do. We repeat this for an increasing number of files.
| code generator | files | bundle size | minified | compressed |
| ------------------- | ----: | ----------: | --------: | ---------: |
-| Protobuf-ES | 1 | 136,220 b | 67,700 b | 15,940 b |
-| Protobuf-ES | 4 | 138,409 b | 69,204 b | 16,599 b |
-| Protobuf-ES | 8 | 141,167 b | 70,975 b | 17,177 b |
-| Protobuf-ES | 16 | 151,601 b | 78,958 b | 19,449 b |
-| Protobuf-ES | 32 | 179,388 b | 100,981 b | 24,966 b |
+| Protobuf-ES | 1 | 136,343 b | 67,792 b | 15,926 b |
+| Protobuf-ES | 4 | 138,532 b | 69,296 b | 16,641 b |
+| Protobuf-ES | 8 | 141,290 b | 71,067 b | 17,155 b |
+| Protobuf-ES | 16 | 151,724 b | 79,050 b | 19,498 b |
+| Protobuf-ES | 32 | 179,511 b | 101,073 b | 24,962 b |
| protobuf-javascript | 1 | 314,172 b | 244,057 b | 36,091 b |
| protobuf-javascript | 4 | 340,189 b | 259,029 b | 37,458 b |
| protobuf-javascript | 8 | 360,983 b | 270,606 b | 38,596 b |
diff --git a/packages/bundle-size/chart.svg b/packages/bundle-size/chart.svg
index cf453fa26..772091dc3 100644
--- a/packages/bundle-size/chart.svg
+++ b/packages/bundle-size/chart.svg
@@ -44,14 +44,14 @@
0 KiB
-
+
Protobuf-ES
-Protobuf-ES 15.57 KiB for 1 files
-Protobuf-ES 16.21 KiB for 4 files
-Protobuf-ES 16.77 KiB for 8 files
-Protobuf-ES 18.99 KiB for 16 files
-Protobuf-ES 24.38 KiB for 32 files
+Protobuf-ES 15.55 KiB for 1 files
+Protobuf-ES 16.25 KiB for 4 files
+Protobuf-ES 16.75 KiB for 8 files
+Protobuf-ES 19.04 KiB for 16 files
+Protobuf-ES 24.38 KiB for 32 files
diff --git a/packages/protobuf-test/src/wire/binary-encoding.test.ts b/packages/protobuf-test/src/wire/binary-encoding.test.ts
index f84708654..757d90b8a 100644
--- a/packages/protobuf-test/src/wire/binary-encoding.test.ts
+++ b/packages/protobuf-test/src/wire/binary-encoding.test.ts
@@ -327,4 +327,53 @@ void suite("BinaryReader", () => {
assert.strictEqual(wireType, WireType.Varint);
});
});
+ void suite("string", () => {
+ // The ASCII fast path covers strings of up to 32 bytes; longer ones and
+ // any string with a byte above 0x7f go to the UTF-8 decoder.
+ const cases: [string, string][] = [
+ ["an empty string", ""],
+ ["a one-byte ASCII string", "a"],
+ ["the longest ASCII string on the fast path", "x".repeat(32)],
+ ["the shortest ASCII string past the fast path", "x".repeat(33)],
+ ["a non-ASCII byte at the end of the fast path", `${"x".repeat(30)}é`],
+ ["a non-ASCII byte at the start", `é${"x".repeat(29)}`],
+ ["multi-byte characters only", "日本語"],
+ ["a long ASCII string", "x".repeat(200)],
+ ];
+ for (const [name, value] of cases) {
+ void test(`reads ${name}`, () => {
+ const bytes = new BinaryWriter().string(value).string("next").finish();
+ const reader = new BinaryReader(bytes);
+ assert.strictEqual(reader.string(), value);
+ // The position must land exactly after the string, also when the
+ // fast path bails out to the decoder part-way through.
+ assert.strictEqual(reader.string(), "next");
+ assert.strictEqual(reader.pos, reader.len);
+ });
+ }
+ void test("reads from a view that does not start at offset 0", () => {
+ // A reader over a subarray indexes relative to the view, not to the
+ // underlying buffer.
+ const inner = new BinaryWriter().string("abc").string("dé").finish();
+ const outer = new Uint8Array(inner.length + 7).fill(0x41);
+ outer.set(inner, 5);
+ const reader = new BinaryReader(outer.subarray(5, 5 + inner.length));
+ assert.strictEqual(reader.string(), "abc");
+ assert.strictEqual(reader.string(), "dé");
+ });
+ void test("rejects a length beyond the end of the data", () => {
+ // Length prefix 5, but only 3 bytes follow.
+ const reader = new BinaryReader(new Uint8Array([5, 0x61, 0x62, 0x63]));
+ assert.throws(() => reader.string(), {
+ name: "RangeError",
+ message: "premature EOF",
+ });
+ });
+ void test("throws on invalid UTF-8 in strict mode only", () => {
+ // 0xff never occurs in UTF-8; short, so it starts on the fast path.
+ const bytes = new Uint8Array([2, 0x61, 0xff]);
+ assert.strictEqual(new BinaryReader(bytes).string(), "a�");
+ assert.throws(() => new BinaryReader(bytes).string(true));
+ });
+ });
});
diff --git a/packages/protobuf/src/wire/binary-encoding.ts b/packages/protobuf/src/wire/binary-encoding.ts
index fbd4fe5a7..66ea84be3 100644
--- a/packages/protobuf/src/wire/binary-encoding.ts
+++ b/packages/protobuf/src/wire/binary-encoding.ts
@@ -780,23 +780,27 @@ export class BinaryReader {
* `strict` is true, throw on invalid UTF-8 instead of substituting U+FFFD.
*/
string(strict?: boolean): string {
- const bytes = this.bytes();
- const len = bytes.length;
+ // Like bytes(), but the ASCII fast path reads the buffer without a view.
+ const len = this.uint32();
+ const start = this.pos;
+ this.pos += len;
+ this.assertBounds();
+ const buf = this.buf;
// Fast path for ASCII.
if (len <= ASCII_MAX_LENGTH) {
const codes = new Array(len);
for (let i = 0; i < len; i++) {
- const byte = bytes[i];
+ const byte = buf[start + i];
if (byte > 0x7f) {
- return this.decodeUtf8(bytes, strict);
+ return this.decodeUtf8(buf.subarray(start, this.pos), strict);
}
codes[i] = byte;
}
return String.fromCharCode.apply(String, codes);
}
- return this.decodeUtf8(bytes, strict);
+ return this.decodeUtf8(buf.subarray(start, this.pos), strict);
}
}