Count WTF-8 byte size without encoding

In `ExpandableTextPropertyValue`, we’re doing a full WTF-8 encode
just to get the byte size of the result. Instead, we can compute
the total byte size without the overhead of doing any encoding.
For the string `'Iñtërnâtiônàlizætiøn☃💩'.repeat(1_000_000)` this
reduces the runtime cost from 166 ms to 98 ms. For the string
`'ASCII-only'.repeat(1_000_000)` the runtime cost increases only
slightly, from 42 ms to 45 ms.

This patch introduces `StringUtilities.countWtf8Bytes` to do exactly
that. Note that WTF-8 [1] is like UTF-8 with additional support for
lone surrogates. We need to support lone surrogates in this case,
since `ExpandableTextPropertyValue` is potentially dealing with
JavaScript strings.

[1]: https://simonsapin.github.io/wtf-8/

Bug: chromium:1024721
Change-Id: I00d0cd8ecd328daddb43b2476229d5e4e47038e8
Reviewed-on: https://chromium-review.googlesource.com/c/devtools/devtools-frontend/+/2134308
Commit-Queue: Mathias Bynens <mathias@chromium.org>
Reviewed-by: Paul Lewis <aerotwist@chromium.org>
This commit is contained in:
Mathias Bynens
2020-04-03 11:08:53 +00:00
committed by Commit Bot
parent 76336cdd04
commit 2af328044c
3 changed files with 53 additions and 7 deletions
@@ -1644,15 +1644,14 @@ export class ExpandableTextPropertyValue extends ObjectPropertyValue {
this._text = text;
this._maxLength = maxLength;
container.textContent = text.slice(0, maxLength);
container.title = `${text.slice(0, maxLength)}...`;
container.title = `${text.slice(0, maxLength)}…`;
this._expandElement = container.createChild('span');
this._maxDisplayableTextLength = 10000000;
const encoder = new TextEncoder();
const buffer = encoder.encode(text);
const totalBytes = Number.bytesToString(buffer.byteLength);
const byteCount = Platform.StringUtilities.countWtf8Bytes(text);
const totalBytesText = Number.bytesToString(byteCount);
if (this._text.length < this._maxDisplayableTextLength) {
this._expandElementText = ls`Show more (${totalBytes})`;
this._expandElementText = ls`Show more (${totalBytesText})`;
this._expandElement.setAttribute('data-text', this._expandElementText);
this._expandElement.classList.add('expandable-inline-button');
this._expandElement.addEventListener('click', this._expandText.bind(this));
@@ -1662,9 +1661,8 @@ export class ExpandableTextPropertyValue extends ObjectPropertyValue {
}
});
UI.ARIAUtils.markAsButton(this._expandElement);
} else {
this._expandElement.setAttribute('data-text', ls`long text was truncated (${totalBytes})`);
this._expandElement.setAttribute('data-text', ls`long text was truncated (${totalBytesText})`);
this._expandElement.classList.add('undisplayable-text');
}
+33
View File
@@ -354,3 +354,36 @@ export const replaceControlCharacters = inputString => {
// Do not replace '\t', \n' and '\r'.
return inputString.replace(/[\0-\x08\x0B\f\x0E-\x1F\x80-\x9F]/g, '\uFFFD');
};
/**
* @param {string} inputString
* @return {number}
*/
export const countWtf8Bytes = inputString => {
let count = 0;
for (let i = 0; i < inputString.length; i++) {
const c = inputString.charCodeAt(i);
if (c <= 0x7F) {
count++;
} else if (c <= 0x07FF) {
count += 2;
} else if (c < 0xD800 || c > 0xDFFF) {
count += 3;
} else {
// The current character is a leading surrogate, and there is a
// next character.
if (c <= 0xDBFF && i + 1 < inputString.length) {
const next = inputString.charCodeAt(i + 1);
if (next >= 0xDC00 && next <= 0xDFFF) {
// The next character is a trailing surrogate, meaning this
// is a surrogate pair.
count += 4;
i++;
continue;
}
}
count += 3;
}
}
return count;
};
@@ -135,4 +135,19 @@ describe('StringUtilities', () => {
assert.equal(inputString, outputString);
});
});
describe('countWtf8Bytes', () => {
it('produces the correct WTF-8 byte size', () => {
assert.equal(StringUtilities.countWtf8Bytes('a'), 1);
assert.equal(StringUtilities.countWtf8Bytes('\x7F'), 1);
assert.equal(StringUtilities.countWtf8Bytes('\u07FF'), 2);
assert.equal(StringUtilities.countWtf8Bytes('\uD800'), 3);
assert.equal(StringUtilities.countWtf8Bytes('\uDBFF'), 3);
assert.equal(StringUtilities.countWtf8Bytes('\uDC00'), 3);
assert.equal(StringUtilities.countWtf8Bytes('\uDFFF'), 3);
assert.equal(StringUtilities.countWtf8Bytes('\uFFFF'), 3);
assert.equal(StringUtilities.countWtf8Bytes('\u{10FFFF}'), 4);
assert.equal(StringUtilities.countWtf8Bytes('Iñtërnâtiônàlizætiøn☃💩'), 34);
});
});
});