Skip to main content

layout/flow/inline/
text_transform.rs

1/* This Source Code Form is subject to the terms of the Mozilla Public
2 * License, v. 2.0. If a copy of the MPL was not distributed with this
3 * file, You can obtain one at https://mozilla.org/MPL/2.0/. */
4
5//! # Logic for text transform in inline formatting contexts
6//!
7//! Inline formatting contexts do a variety of text transformations on their text content
8//! including white space collapsing, application of the `text-transform` CSS property,
9//! and application of the `-webkit-text-security` property. This module contains code to
10//! handle this as well as code to map from offsets in the original DOM node to the final
11//! IFC text and vice-versa.
12
13use arrayvec::ArrayVec;
14use icu_properties::props::{EnumeratedProperty, GeneralCategory, GeneralCategoryGroup};
15use icu_segmenter::WordSegmenter;
16use icu_segmenter::options::WordBreakInvariantOptions;
17use malloc_size_of_derive::MallocSizeOf;
18#[cfg(test)]
19use servo_base::text::AssumeUnder4GB;
20use servo_base::text::Utf32CodeUnits;
21use style::computed_values::_webkit_text_security::T as WebKitTextSecurity;
22use style::computed_values::white_space_collapse::T as WhiteSpaceCollapse;
23use style::properties::ComputedValues;
24use style::values::specified::text::{TextTransform, TextTransformCase};
25
26use crate::flow::inline::construct::InlineFormattingContextBuilder;
27
28/// <https://github.com/rust-lang/rust/blob/1.97.1/library/core/src/char/mod.rs#L523>
29///
30/// This is the maximum amount of characters that can be produced from case mapping,
31/// and by consequence the maximum amount of characters that can be produced during
32/// inline formatting context text transformation.
33const MAX_CASE_MAPPING_LENGTH: usize = 3;
34
35/// A single iteration in a pipeline of character iterators, that handle things like
36/// whitespace collapse and `text-transform` processing for text in an
37/// [`InlineFormattingContext`]. Each iteration can consume multiple characters and
38/// produce zero or more characters (up to 3). Consumption of characters greater than the
39/// characters produced by [`CharacterTransformIteration`] indicate that those characters
40/// have been collapsed.
41#[derive(Clone)]
42pub struct CharacterTransformIteration {
43    /// The number of characters consumed during this iteration of character transformation.
44    consumed: Utf32CodeUnits,
45    /// The characters that were produced during this iteration.
46    characters: ArrayVec<char, MAX_CASE_MAPPING_LENGTH>,
47}
48
49impl CharacterTransformIteration {
50    fn case_mapped(iterator: impl ExactSizeIterator<Item = char>) -> Self {
51        debug_assert!(iterator.len() <= MAX_CASE_MAPPING_LENGTH);
52        Self {
53            consumed: Utf32CodeUnits(1),
54            characters: iterator.collect(),
55        }
56    }
57
58    fn one_to_one(character: char) -> Self {
59        Self {
60            consumed: Utf32CodeUnits(1),
61            characters: std::iter::once(character).collect(),
62        }
63    }
64
65    fn collapse(amount_collapsed: Utf32CodeUnits, character: Option<char>) -> Self {
66        Self {
67            consumed: amount_collapsed,
68            characters: character.into_iter().collect(),
69        }
70    }
71
72    fn is_one_to_one(&self) -> bool {
73        self.characters.len() == 1 && self.consumed.0 == 1
74    }
75
76    pub fn characters(&self) -> &[char] {
77        &self.characters
78    }
79}
80
81pub struct WhitespaceCollapse<InputIterator> {
82    input_iterator: InputIterator,
83    white_space_collapse: WhiteSpaceCollapse,
84
85    /// Whether or not we are in the process of collapse leading white space. This is true
86    /// when the last character handled in our owning [`super::InlineFormattingContext`]
87    /// was collapsible white space and we have not seen any non-whitespace characters
88    /// during processing of this iterator's input.
89    trimming_leading_white_space: bool,
90
91    /// Whether or not the last character produced was newline. There is special behavior
92    /// we do after each newline.
93    following_newline: bool,
94
95    /// When whitespace collapses before a non-whitespace character, the iterator returns
96    /// the collapsed whitespace and in the next iteration the non-whitespace character
97    /// must be returned. This value caches it until the next iteration.
98    character_pending_to_return: Option<char>,
99}
100
101impl<InputIterator: Iterator<Item = char>> WhitespaceCollapse<InputIterator> {
102    pub fn new(
103        input_iterator: InputIterator,
104        white_space_collapse: WhiteSpaceCollapse,
105        should_trim_leading_white_space: bool,
106    ) -> Self {
107        Self {
108            input_iterator,
109            white_space_collapse,
110            following_newline: false,
111            trimming_leading_white_space: should_trim_leading_white_space,
112            character_pending_to_return: None,
113        }
114    }
115
116    /// In some cases, white space is replaced by a single character (when not
117    /// following a newline and when leading whitespace is not being trimmed). In all
118    /// other cases, the white space is simply removed. This method handles that.
119    fn iteration_for_collapsed_whitespace(
120        &self,
121        collapsed_whitespace: Utf32CodeUnits,
122    ) -> CharacterTransformIteration {
123        if !self.following_newline && !self.trimming_leading_white_space {
124            CharacterTransformIteration::collapse(collapsed_whitespace, Some(' '))
125        } else {
126            CharacterTransformIteration::collapse(collapsed_whitespace, None)
127        }
128    }
129
130    fn iteration_for_collected_white_space(
131        &self,
132        collected_whitespace: Utf32CodeUnits,
133    ) -> Option<CharacterTransformIteration> {
134        (collected_whitespace.0 != 0)
135            .then(|| self.iteration_for_collapsed_whitespace(collected_whitespace))
136    }
137}
138
139impl<InputIterator: Iterator<Item = char>> Iterator for WhitespaceCollapse<InputIterator> {
140    type Item = CharacterTransformIteration;
141
142    fn next(&mut self) -> Option<Self::Item> {
143        // Point 4.1.1 first bullet:
144        // > If white-space is set to normal, nowrap, or pre-line, whitespace
145        // > characters are considered collapsible
146        // If whitespace is not considered collapsible, it is preserved entirely, which
147        // means that we can simply return the input string exactly.
148        if self.white_space_collapse == WhiteSpaceCollapse::Preserve ||
149            self.white_space_collapse == WhiteSpaceCollapse::BreakSpaces
150        {
151            // From <https://drafts.csswg.org/css-text-3/#white-space-processing>:
152            // > Carriage returns (U+000D) are treated identically to spaces (U+0020) in all respects.
153            //
154            // In the non-preserved case these are converted to space below.
155            return match self.input_iterator.next() {
156                Some('\r') => Some(CharacterTransformIteration::one_to_one(' ')),
157                next => next.map(CharacterTransformIteration::one_to_one),
158            };
159        }
160
161        if let Some(character) = self.character_pending_to_return.take() {
162            // Once we produce a non-whitespace character, we are no longer trimming leading whitespace.
163            self.trimming_leading_white_space = false;
164            self.following_newline = false;
165            return Some(CharacterTransformIteration::one_to_one(character));
166        }
167
168        // When we enter a collapsible white space region, we may need to wait to produce
169        // a single white space character as soon as we encounter a non-white space
170        // character. When that happens we queue up the non-white space character for the
171        // next iterator call.
172        let mut collected_whitespace = Utf32CodeUnits(0);
173
174        while let Some(character) = self.input_iterator.next() {
175            // Don't push non-newline whitespace immediately. Instead wait to push it until we
176            // know that it isn't followed by a newline. See `push_pending_whitespace_if_needed`
177            // above.
178            if InlineFormattingContextBuilder::is_document_white_space(character) &&
179                character != '\n'
180            {
181                collected_whitespace += Utf32CodeUnits(1);
182                continue;
183            }
184
185            // Point 4.1.1:
186            // > 2. Collapsible segment breaks are transformed for rendering according to the
187            // >    segment break transformation rules.
188            if character == '\n' {
189                // From <https://drafts.csswg.org/css-text-3/#line-break-transform>
190                // (4.1.3 -- the segment break transformation rules):
191                //
192                // > When white-space is pre, pre-wrap, or pre-line, segment breaks are not
193                // > collapsible and are instead transformed into a preserved line feed"
194                //
195                // > 1. First, any collapsible segment break immediately following another
196                // >    collapsible segment break is removed.
197                // > 2. Then any remaining segment break is either transformed into a space (U+0020)
198                // >    or removed depending on the context before and after the break.
199                let iteration = if self.white_space_collapse != WhiteSpaceCollapse::Collapse {
200                    CharacterTransformIteration::collapse(
201                        collected_whitespace + Utf32CodeUnits(1),
202                        Some('\n'),
203                    )
204                } else {
205                    self.iteration_for_collapsed_whitespace(
206                        collected_whitespace + Utf32CodeUnits(1),
207                    )
208                };
209
210                self.following_newline = true;
211                return Some(iteration);
212            }
213
214            // Non-whitespace character
215
216            // Point 4.1.1:
217            // > 2. Any sequence of collapsible spaces and tabs immediately preceding or
218            // >    following a segment break is removed.
219            // > 3. Every collapsible tab is converted to a collapsible space (U+0020).
220            // > 4. Any collapsible space immediately following another collapsible space—even
221            // >    one outside the boundary of the inline containing that space, provided both
222            // >    spaces are within the same inline formatting context—is collapsed to have zero
223            // >    advance width.
224            if let Some(iteration) = self.iteration_for_collected_white_space(collected_whitespace)
225            {
226                self.character_pending_to_return = Some(character);
227                return Some(iteration);
228            }
229
230            // Once we produce a non-whitespace character, we are no longer trimming leading whitespace.
231            self.trimming_leading_white_space = false;
232            self.following_newline = false;
233            return Some(CharacterTransformIteration::one_to_one(character));
234        }
235
236        self.iteration_for_collected_white_space(collected_whitespace)
237    }
238}
239
240pub(crate) struct TextTransformationIterator<'a> {
241    case_map_iterator: Box<dyn Iterator<Item = CharacterTransformIteration> + 'a>,
242    full_width: bool,
243    full_size_kana: bool,
244}
245
246impl<'a> TextTransformationIterator<'a> {
247    pub(crate) fn new(
248        mut text: &'a str,
249        style: &'a ComputedValues,
250        trim_leading_white_space: bool,
251        on_word_boundary: bool,
252    ) -> Self {
253        let text_security = style.get__webkit_text_security();
254
255        // <https://drafts.csswg.org/css-text-4/#text-transform-property>
256        let text_transform = style.get_text_transform();
257
258        if text_transform.intersects(TextTransform::MATH_AUTO) {
259            // `math-auto` only does anything “on text nodes containing a single character” per
260            // https://w3c.github.io/mathml-core/#math-auto-transform
261            //
262            // TODO: should this be single character after whitespace collapsing?
263            // TODO: does `::first-letter` mess with this check?
264            let mut char_iter = text.chars();
265            if let Some(first_char) = char_iter.next() &&
266                let None = char_iter.next() &&
267                let Some(&mapping) = super::mathml_italics::ITALICS_MAPPINGS.get(&first_char)
268            {
269                text = mapping
270            }
271        }
272
273        let chars = text
274            .chars()
275            .map(move |character| map_character_for_webkit_text_security(text_security, character));
276        let white_space_collapse = style.slow_clone_white_space_collapse();
277        let iterator =
278            WhitespaceCollapse::new(chars, white_space_collapse, trim_leading_white_space);
279
280        // https://drafts.csswg.org/css-text-4/#text-transform-order
281        // > When multiple transformations need to be applied,
282        // > they are applied in the following order:
283        // >
284        // > * `word-space-transform`
285        // > * `capitalize`, `uppercase`, and `lowercase`
286        // > * `full-width`
287        // > * `full-size-kana`
288        // >
289        // > Word space transformation and text transformation happen after
290        // > § 4.3.1 Phase I: Collapsing and Transformation but before
291        // > § 4.3.2 Phase II: Trimming and Positioning. This means for instance that full-width
292        // > only transforms spaces (U+0020) to U+3000 IDEOGRAPHIC SPACE within
293        // > preserved white space.
294
295        let case_map_iterator = match text_transform.case() {
296            TextTransformCase::None | TextTransformCase::MathAuto => {
297                Box::new(iterator) as Box<dyn Iterator<Item = CharacterTransformIteration>>
298            },
299            TextTransformCase::Lowercase => {
300                Box::new(simple_case_transform_iterator(iterator, |character| {
301                    CharacterTransformIteration::case_mapped(character.to_lowercase())
302                }))
303            },
304            TextTransformCase::Uppercase => {
305                Box::new(simple_case_transform_iterator(iterator, |character| {
306                    CharacterTransformIteration::case_mapped(character.to_uppercase())
307                }))
308            },
309            TextTransformCase::Capitalize => Box::new(capitalization_iterator(
310                iterator,
311                text.len(),
312                on_word_boundary,
313            )),
314        };
315
316        Self {
317            case_map_iterator,
318            full_width: text_transform.intersects(TextTransform::FULL_WIDTH),
319            full_size_kana: text_transform.intersects(TextTransform::FULL_SIZE_KANA),
320        }
321    }
322}
323
324impl Iterator for TextTransformationIterator<'_> {
325    type Item = CharacterTransformIteration;
326
327    fn next(&mut self) -> Option<Self::Item> {
328        // https://drafts.csswg.org/css-text-4/#text-transform-order
329        // > When multiple transformations need to be applied,
330        // > they are applied in the following order:
331        // >
332        // > * `word-space-transform`
333        // > * `capitalize`, `uppercase`, and `lowercase`
334        // > * `full-width`
335        // > * `full-size-kana`
336        let mut iteration = self.case_map_iterator.next()?;
337        map_characters_with_phf(
338            self.full_width,
339            &mut iteration.characters,
340            &super::full_width::FULL_WIDTH_MAPPINGS,
341        );
342        map_characters_with_phf(
343            self.full_size_kana,
344            &mut iteration.characters,
345            &super::small_kana::SMALL_KANA_MAPPINGS,
346        );
347        Some(iteration)
348    }
349}
350
351fn simple_case_transform_iterator(
352    input_iterator: impl Iterator<Item = CharacterTransformIteration>,
353    mapping: impl Fn(char) -> CharacterTransformIteration,
354) -> impl Iterator<Item = CharacterTransformIteration> {
355    input_iterator.map(move |iteration| {
356        if iteration.is_one_to_one() {
357            mapping(iteration.characters[0])
358        } else {
359            iteration
360        }
361    })
362}
363
364/// From <https://drafts.csswg.org/css-text-4/#typographic-letter-unit>:
365/// > A typographic letter unit (or letter for the purpose of this specification) is a
366/// > typographic character unit belonging to one of the Letter or Number general categories. See
367/// > Appendix E: Characters and Properties for how to determine the Unicode properties of a
368/// > typographic character unit.
369fn is_typographic_letter_unit(character: char) -> bool {
370    let category = GeneralCategory::for_char(character);
371    GeneralCategoryGroup::Letter.contains(category) ||
372        GeneralCategoryGroup::Number.contains(category)
373}
374
375/// Given an input iterator, a size hint for the number items in the iterator,
376/// and a boolean determining whether the start of the input represents a word
377/// boundary, return an iterator that capitalizes one-to-one mapped characters
378/// from the input iterator.
379pub(crate) fn capitalization_iterator(
380    input_iterator: impl Iterator<Item = CharacterTransformIteration>,
381    size_hint: usize,
382    allow_word_at_start: bool,
383) -> impl Iterator<Item = CharacterTransformIteration> {
384    let mut iterations: Vec<_> = input_iterator.collect();
385    let mut string = String::with_capacity(size_hint);
386    for iteration in &iterations {
387        string.extend(iteration.characters());
388    }
389
390    let word_segmenter = WordSegmenter::new_auto(WordBreakInvariantOptions::default());
391    let mut bounds = word_segmenter.segment_str(&string).peekable();
392    let mut current_byte_index = 0;
393    let mut pending_word_start = false;
394    for iteration in iterations.iter_mut() {
395        let bytes_to_advance: usize = iteration
396            .characters()
397            .iter()
398            .map(|character| character.len_utf8())
399            .sum();
400        if bytes_to_advance == 0 {
401            continue;
402        }
403
404        if bounds.peek() == Some(&current_byte_index) {
405            pending_word_start = current_byte_index != 0 || allow_word_at_start;
406            bounds.next();
407        }
408
409        // From <https://drafts.csswg.org/css-text-4/#text-transform-property>:
410        // > Puts the first typographic letter unit of each word, if lowercase, in titlecase;
411        // > other characters are unaffected.
412        if iteration.is_one_to_one() &&
413            pending_word_start &&
414            is_typographic_letter_unit(iteration.characters[0])
415        {
416            if iteration.characters[0].is_lowercase() {
417                // TODO: Replace this with a call to `character.to_titlecase()` when available:
418                // See: https://github.com/rust-lang/rust/issues/153892
419                // See: https://doc.rust-lang.org/stable/std/primitive.char.html#difference-from-uppercase
420                *iteration = CharacterTransformIteration::case_mapped(
421                    iteration.characters[0].to_uppercase(),
422                );
423            }
424
425            pending_word_start = false;
426        }
427
428        current_byte_index += bytes_to_advance;
429    }
430
431    iterations.into_iter()
432}
433
434/// Map a character according to the rules of the `-webkit-text-security` CSS property.
435///
436/// Note: The behavior of `-webkit-text-security` isn't specified, so we have some
437/// flexibility in the implementation. We just need to maintain a rough compatibility with
438/// other browsers.
439fn map_character_for_webkit_text_security(mode: &WebKitTextSecurity, character: char) -> char {
440    if let WebKitTextSecurity::None = mode {
441        return character;
442    }
443
444    // TODO: When MSRV is 1.95+ use std::hint::cold_path().
445    match character {
446        // Newlines are preserved, so that `<br>` keeps working as expected.
447        '\n' => '\n',
448        _ => match mode {
449            WebKitTextSecurity::None => character, // unreachable
450            WebKitTextSecurity::Circle => '○',
451            WebKitTextSecurity::Disc => '●',
452            WebKitTextSecurity::Square => '■',
453        },
454    }
455}
456
457fn map_characters_with_phf(enabled: bool, characters: &mut [char], map: &phf::Map<char, char>) {
458    if enabled {
459        // TODO: When MSRV is 1.95+ use std::hint::cold_path().
460
461        for character in characters {
462            if let Some(mapping) = map.get(character) {
463                *character = *mapping
464            }
465        }
466    }
467}
468
469#[derive(MallocSizeOf, Clone, Copy)]
470struct OffsetMapKnownPosition {
471    original_offset: Utf32CodeUnits,
472    final_offset: Utf32CodeUnits,
473}
474
475#[derive(Default, MallocSizeOf)]
476pub struct OffsetMap {
477    /// Not including `IMPLICIT_KNOWN_POSITION_AT_START`
478    known_positions: Vec<OffsetMapKnownPosition>,
479    /// `Default` initializes to `false`
480    last_range_maps_one_to_one: bool,
481}
482
483impl std::fmt::Debug for OffsetMap {
484    fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result {
485        f.debug_struct("OffsetMap")
486            .field("total_original_size", &self.total_original_size())
487            .field("total_final_size", &self.total_final_size())
488            .finish()
489    }
490}
491
492static IMPLICIT_KNOWN_POSITION_AT_START: OffsetMapKnownPosition = OffsetMapKnownPosition {
493    original_offset: Utf32CodeUnits(0),
494    final_offset: Utf32CodeUnits(0),
495};
496
497impl OffsetMap {
498    fn last_known_position(&self) -> &OffsetMapKnownPosition {
499        self.known_positions
500            .last()
501            .unwrap_or(&IMPLICIT_KNOWN_POSITION_AT_START)
502    }
503
504    pub fn total_original_size(&self) -> Utf32CodeUnits {
505        self.last_known_position().original_offset
506    }
507
508    pub fn total_final_size(&self) -> Utf32CodeUnits {
509        self.last_known_position().final_offset
510    }
511
512    pub fn push_range(
513        &mut self,
514        additional_original_length: Utf32CodeUnits,
515        additional_final_length: Utf32CodeUnits,
516    ) {
517        let this_range_maps_one_to_one = additional_original_length == additional_final_length;
518        if this_range_maps_one_to_one &&
519            self.last_range_maps_one_to_one &&
520            let Some(last) = self.known_positions.last_mut()
521        {
522            last.original_offset += additional_original_length;
523            last.final_offset += additional_final_length;
524        } else {
525            let last = self.last_known_position();
526            self.known_positions.push(OffsetMapKnownPosition {
527                original_offset: last.original_offset + additional_original_length,
528                final_offset: last.final_offset + additional_final_length,
529            });
530        }
531        self.last_range_maps_one_to_one = this_range_maps_one_to_one;
532    }
533
534    pub(crate) fn push_iteration(&mut self, iteration: &CharacterTransformIteration) {
535        self.push_range(
536            iteration.consumed,
537            Utf32CodeUnits(iteration.characters.len() as u32),
538        );
539    }
540
541    pub fn map(&self, target_original_offset: Utf32CodeUnits) -> Utf32CodeUnits {
542        self.map_common(
543            target_original_offset,
544            |position| position.original_offset,
545            |position| position.final_offset,
546        )
547    }
548
549    pub fn reverse_map(&self, target_final_offset: Utf32CodeUnits) -> Utf32CodeUnits {
550        self.map_common(
551            target_final_offset,
552            |position| position.final_offset,
553            |position| position.original_offset,
554        )
555    }
556
557    fn map_common(
558        &self,
559        target_offset: Utf32CodeUnits,
560        get_input_offset: impl Copy + Fn(&OffsetMapKnownPosition) -> Utf32CodeUnits,
561        get_output_offset: impl Fn(&OffsetMapKnownPosition) -> Utf32CodeUnits,
562    ) -> Utf32CodeUnits {
563        if target_offset.0 == 0 {
564            // Implict known position
565            return Utf32CodeUnits(0);
566        }
567        match self
568            .known_positions
569            .binary_search_by_key(&target_offset, get_input_offset)
570        {
571            Ok(index) => {
572                // Exact known position
573                get_output_offset(&self.known_positions[index])
574            },
575            Err(index) => {
576                // `index` is where inserting a new position would keep the `Vec` sorted
577                if let Some(position_after) = self.known_positions.get(index) {
578                    let position_before = if index > 0 {
579                        &self.known_positions[index - 1]
580                    } else {
581                        &IMPLICIT_KNOWN_POSITION_AT_START
582                    };
583                    debug_assert!(target_offset > get_input_offset(position_before));
584                    debug_assert!(target_offset < get_input_offset(position_after));
585                    let offset_within_range = target_offset - get_input_offset(position_before);
586                    let candidate = get_output_offset(position_before) + offset_within_range;
587                    // If the output range is shorter, to go beyond it
588                    let upper_bound = get_output_offset(position_after);
589                    upper_bound.min(candidate)
590                } else {
591                    // `target_offset` at or past the end of the text covered by this map
592                    get_output_offset(self.last_known_position())
593                }
594            },
595        }
596    }
597}
598
599#[test]
600fn test_offsetmap_basic_expansion() {
601    let original_string = "aßΰb";
602    let final_string = "ASS\u{3a5}\u{308}\u{301}B";
603    assert_eq!(original_string.to_uppercase(), final_string);
604
605    let mut offset_map = OffsetMap::default();
606    offset_map.push_iteration(&CharacterTransformIteration::case_mapped(
607        'a'.to_uppercase(),
608    ));
609    offset_map.push_iteration(&CharacterTransformIteration::case_mapped(
610        'ß'.to_uppercase(),
611    ));
612    offset_map.push_iteration(&CharacterTransformIteration::case_mapped(
613        'ΰ'.to_uppercase(),
614    ));
615    offset_map.push_iteration(&CharacterTransformIteration::case_mapped(
616        'b'.to_uppercase(),
617    ));
618
619    assert_eq!(offset_map.map(Utf32CodeUnits(0)).0, 0);
620    assert_eq!(offset_map.map(Utf32CodeUnits(1)).0, 1);
621    assert_eq!(offset_map.map(Utf32CodeUnits(2)).0, 3);
622    assert_eq!(offset_map.map(Utf32CodeUnits(3)).0, 6);
623    assert_eq!(offset_map.map(Utf32CodeUnits(4)).0, 7);
624
625    // Beyond the last index should always map to the index after the last character
626    // (for handling selections).
627    assert_eq!(offset_map.map(Utf32CodeUnits(5)).0, 7);
628    assert_eq!(offset_map.map(Utf32CodeUnits(100)).0, 7);
629
630    let map_substring = |offset: u32, length: u32| {
631        let start = usize::from(
632            offset_map
633                .map(Utf32CodeUnits(offset))
634                .to_utf8_code_units_in(AssumeUnder4GB, final_string),
635        );
636        let end = usize::from(
637            offset_map
638                .map(Utf32CodeUnits(offset + length))
639                .to_utf8_code_units_in(AssumeUnder4GB, final_string),
640        );
641        &final_string[start..end]
642    };
643    assert_eq!(map_substring(0, 1), "A");
644    assert_eq!(map_substring(0, 2), "ASS");
645    assert_eq!(map_substring(0, 3), "ASS\u{3a5}\u{308}\u{301}");
646    assert_eq!(map_substring(0, 4), "ASS\u{3a5}\u{308}\u{301}B");
647    assert_eq!(map_substring(1, 1), "SS");
648}
649
650#[test]
651fn test_offsetmap_basic_collapse() {
652    let _original_string = "  aaa  b \nc";
653    let final_string = "aaa b\nc";
654
655    let mut offset_map = OffsetMap::default();
656    offset_map.push_iteration(&CharacterTransformIteration::collapse(
657        Utf32CodeUnits(2),
658        None,
659    ));
660    offset_map.push_iteration(&CharacterTransformIteration::one_to_one('a'));
661    offset_map.push_iteration(&CharacterTransformIteration::one_to_one('a'));
662    offset_map.push_iteration(&CharacterTransformIteration::one_to_one('a'));
663    assert_eq!(
664        offset_map.known_positions.len(),
665        2,
666        "Consecutive one-to-one mappings are merged"
667    );
668
669    offset_map.push_iteration(&CharacterTransformIteration::collapse(
670        Utf32CodeUnits(2),
671        Some(' '),
672    ));
673    offset_map.push_iteration(&CharacterTransformIteration::one_to_one('b'));
674    offset_map.push_iteration(&CharacterTransformIteration::collapse(
675        Utf32CodeUnits(2),
676        Some('\n'),
677    ));
678    offset_map.push_iteration(&CharacterTransformIteration::one_to_one('c'));
679
680    assert_eq!(offset_map.map(Utf32CodeUnits(0)).0, 0);
681    assert_eq!(offset_map.map(Utf32CodeUnits(1)).0, 0);
682    assert_eq!(offset_map.map(Utf32CodeUnits(2)).0, 0);
683    assert_eq!(offset_map.map(Utf32CodeUnits(3)).0, 1);
684    assert_eq!(offset_map.map(Utf32CodeUnits(4)).0, 2);
685    assert_eq!(offset_map.map(Utf32CodeUnits(5)).0, 3);
686    // Mapping from the middle of the collapsed sequence should map to after the replacement.
687    assert_eq!(offset_map.map(Utf32CodeUnits(6)).0, 4);
688    assert_eq!(offset_map.map(Utf32CodeUnits(7)).0, 4);
689    assert_eq!(offset_map.map(Utf32CodeUnits(8)).0, 5);
690    // Mapping from the middle of the collapsed sequence should map to after the replacement.
691    assert_eq!(offset_map.map(Utf32CodeUnits(9)).0, 6);
692    assert_eq!(offset_map.map(Utf32CodeUnits(10)).0, 6);
693    assert_eq!(offset_map.map(Utf32CodeUnits(11)).0, 7);
694
695    // Beyond the last index should always map to the index after the last character
696    // (for handling selections).
697    assert_eq!(offset_map.map(Utf32CodeUnits(12)).0, 7);
698    assert_eq!(offset_map.map(Utf32CodeUnits(100)).0, 7);
699
700    let map_substring = |offset: u32, length: u32| {
701        let start = usize::from(offset_map.map(Utf32CodeUnits(offset)));
702        let end = usize::from(offset_map.map(Utf32CodeUnits(offset + length)));
703        &final_string[start..end]
704    };
705    assert_eq!(map_substring(0, 1), "");
706    assert_eq!(map_substring(0, 3), "a");
707    assert_eq!(map_substring(0, 5), "aaa");
708    assert_eq!(map_substring(0, 6), "aaa ");
709    assert_eq!(map_substring(0, 7), "aaa ");
710    assert_eq!(map_substring(0, 8), "aaa b");
711    assert_eq!(map_substring(0, 11), "aaa b\nc");
712}