ladybird/Libraries/LibJS/SourceCode.h
Andreas Kling 7c19719946 LibJS: Avoid repeated lazy source decoding
Function.prototype.toString() can ask for the same cached function
source text repeatedly after a script was materialized from the
bytecode cache. In that path SourceCode still owns only the encoded
source bytes, so every request decoded the requested source range
again. Large ASCII bundles made that path expensive enough to stall
Speedometer 2.1's Ember debug test.

Keep the byte-backed SourceCode representation lazy, but let ASCII
UTF-8 source ranges slice the source bytes directly. For other ASCII
byte-backed encodings, first ask the decoder to prove that the source
bytes map to the same UTF-16 code units at the same positions.

Cache extracted function source text on shared function data after the
first request so repeated toString() calls do not keep going back to
SourceCode.

Add bytecode cache coverage for toString() on lazy UTF-8 source bytes,
keep malformed UTF-8 and UTF-16 edge cases on the decoder path, and
cover PDFDocEncoding bytes that are not identity-mapped.
2026-05-19 17:58:08 +02:00

65 lines
2.3 KiB
C++

/*
* Copyright (c) 2022-2023, Andreas Kling <andreas@ladybird.org>
*
* SPDX-License-Identifier: BSD-2-Clause
*/
#pragma once
#include <AK/Optional.h>
#include <AK/String.h>
#include <AK/Utf16String.h>
#include <AK/Vector.h>
#include <LibCore/ImmutableBytes.h>
#include <LibJS/Export.h>
#include <LibJS/Forward.h>
#include <LibJS/Position.h>
namespace JS {
class JS_API SourceCode : public RefCounted<SourceCode> {
public:
static NonnullRefPtr<SourceCode const> create(String filename, Utf16String code);
static NonnullRefPtr<SourceCode const> create(String filename, size_t length_in_code_units, String source_encoding, Core::ImmutableBytes source_bytes);
String const& filename() const { return m_filename; }
Utf16String const& code() const;
Utf16View const& code_view() const;
size_t length_in_code_units() const { return m_length_in_code_units; }
u16 const* utf16_data() const;
Utf16String source_text_from_offsets(size_t start_offset, size_t length) const;
SourceRange range_from_offsets(u32 start_offset, u32 end_offset) const;
private:
SourceCode(String filename, Utf16String code);
SourceCode(String filename, size_t length_in_code_units, String source_encoding, Core::ImmutableBytes source_bytes);
void ensure_code() const;
Utf16String decode_source_range(size_t start_offset, size_t length) const;
bool source_bytes_can_be_sliced_by_code_unit_offsets() const;
String m_filename;
Optional<Utf16String> mutable m_code;
String m_source_encoding;
Core::ImmutableBytes mutable m_source_bytes;
Utf16View mutable m_code_view;
size_t m_length_in_code_units { 0 };
// For fast mapping of offsets to line/column numbers, we build a list of
// starting points (with byte offsets into the source string) and which
// line:column they map to. This can then be binary-searched.
void fill_position_cache() const;
struct CachedPosition {
Position position;
u32 offset { 0 };
};
Vector<CachedPosition> mutable m_cached_positions;
// Cached UTF-16 widening of ASCII source data, lazily populated by
// utf16_data() for use by the Rust compilation pipeline.
Vector<u16> mutable m_utf16_data_cache;
Optional<bool> mutable m_source_bytes_can_be_sliced_by_code_unit_offsets;
};
}