Skip to main content

layout/flow/inline/
shaping_queue.rs

1/* This Source Code Form is subject to the terms of the Mozilla Public
2 * License, v. 2.0. If a copy of the MPL was not distributed with this
3 * file, You can obtain one at https://mozilla.org/MPL/2.0/. */
4
5use std::ops::Range;
6use std::sync::Arc;
7
8use fonts::{ShapedText, ShapedTextSlice, ShapedTextSlicer, ShapingOptions, TrailingWhiteSpace};
9use icu_properties::props::{EnumeratedProperty, GeneralCategory, LineBreak};
10use icu_segmenter::options::LineBreakOptions;
11use servo_base::text::{AssumeUnder4GB, Utf8CodeUnits, Utf32CodeUnits};
12use style::computed_values::text_wrap_mode::T as TextWrapMode;
13use style::computed_values::white_space_collapse::T as WhiteSpaceCollapse;
14use style::computed_values::word_break::T as WordBreak;
15use style::properties::ComputedValues;
16use style::properties::style_structs::InheritedText;
17use unicode_script::Script;
18
19use crate::ArcRefCell;
20use crate::flow::inline::line_breaker::LineBreaker;
21use crate::flow::inline::text_run::{FontAndScriptInfo, TextRun, TextRunItem, script_is_specific};
22
23/// An entry on the shaping queue that represents text that needs to be shaped.
24/// This contains a lot of duplicated data from `TextRunSegment` so that
25/// it can outlive a mutable borrow on the owning `TextRun`.
26pub(crate) struct ShapingQueueText {
27    info: FontAndScriptInfo,
28    byte_range: Range<Utf8CodeUnits>,
29    character_range: Range<Utf32CodeUnits>,
30    text_run: ArcRefCell<TextRun>,
31    index_in_text_run: usize,
32    old_shaped_text: Option<Arc<ShapedText>>,
33}
34
35/// A new entry for the [`ShapingQueue`].
36pub(crate) enum ShapingQueueEntry {
37    PreservedTabOrNewline,
38    Text(ShapingQueueText),
39}
40
41impl ShapingQueueEntry {
42    pub(crate) fn new(
43        text_run: ArcRefCell<TextRun>,
44        text_run_item: &TextRunItem,
45        index_in_text_run: usize,
46        old_text_run_line_item: Option<TextRunItem>,
47    ) -> Self {
48        let text_segment = match text_run_item {
49            TextRunItem::LineBreak { .. } | TextRunItem::Tab { .. } => {
50                return Self::PreservedTabOrNewline;
51            },
52            TextRunItem::TextSegment(text_run_segment) => text_run_segment,
53        };
54
55        let old_shaped_text = old_text_run_line_item.and_then(|old_text_run_line_item| {
56            let TextRunItem::TextSegment(old_text_segment) = old_text_run_line_item else {
57                return None;
58            };
59            if !text_segment.is_compatible_with_old_shaping_result(&old_text_segment) {
60                return None;
61            }
62            old_text_segment.shaped_text
63        });
64
65        Self::Text(ShapingQueueText {
66            info: text_segment.info.clone(),
67            byte_range: text_segment.byte_range.clone(),
68            character_range: text_segment.character_range.clone(),
69            text_run,
70            index_in_text_run,
71            old_shaped_text,
72        })
73    }
74}
75
76struct BatchSlicer<'a> {
77    slicer: ShapedTextSlicer,
78    text: &'a str,
79    line_breaker: &'a mut LineBreaker,
80    character_offset_origin: Utf32CodeUnits,
81}
82
83impl BatchSlicer<'_> {
84    fn slice_shaped_text_at_line_break_opportunities(
85        &mut self,
86        segment: &ShapingQueueText,
87        parent_style: &ComputedValues,
88    ) -> (Vec<Arc<ShapedTextSlice>>, bool) {
89        // Gather the linebreaks that apply to this segment from the inline formatting context's collection
90        // of line breaks. Also add a simulated break at the end of the segment in order to ensure the final
91        // piece of text is processed.
92        let range = segment.byte_range.clone();
93        let text_style = parent_style.get_inherited_text();
94        let mut break_at_start =
95            self.line_breaker.take_additional_break_at_start() == Some(range.start);
96        let linebreaks = self
97            .line_breaker
98            .advance_to_linebreaks_in_range(segment.byte_range.clone());
99        let linebreak_iter = linebreaks.iter().chain(std::iter::once(&range.end));
100
101        let mut current_character_offset =
102            segment.character_range.start - self.character_offset_origin;
103
104        let mut slices = Vec::with_capacity(linebreaks.len());
105        let mut maybe_push_slice_and_update_character_offset = |slice_text: &str| {
106            current_character_offset += Utf32CodeUnits::length_of(AssumeUnder4GB, slice_text);
107            let (trailing_white_space, all_white_space) =
108                trailing_white_space_of(slice_text, parent_style);
109            if let Some(slice) = self.slicer.slice_until_character_offset(
110                current_character_offset,
111                trailing_white_space,
112                all_white_space,
113            ) {
114                slices.push(slice);
115            }
116        };
117
118        let mut last_slice_end = segment.byte_range.start;
119        for break_index in linebreak_iter {
120            if *break_index == segment.byte_range.start &&
121                !line_break_ignored_for_keep_all(self.text, text_style, *break_index)
122            {
123                break_at_start = true;
124                continue;
125            }
126
127            let slice = last_slice_end..*break_index;
128
129            // `keep-all` might suppress line breaks between certain characters and that check
130            // is done here, but not in the case that the line break is the last one for this
131            // segment (in order to push the rest of the text).
132            if *break_index != segment.byte_range.end &&
133                line_break_ignored_for_keep_all(self.text, text_style, *break_index)
134            {
135                continue;
136            }
137
138            last_slice_end = *break_index;
139            if slice.is_empty() {
140                continue;
141            }
142
143            let full_slice_text = &self.text[Utf8CodeUnits::to_usize_range(&slice)];
144            if text_style.white_space_collapse != WhiteSpaceCollapse::BreakSpaces {
145                maybe_push_slice_and_update_character_offset(full_slice_text)
146            } else {
147                for slice_text in full_slice_text.split_inclusive(breaks_for_break_spaces) {
148                    maybe_push_slice_and_update_character_offset(slice_text)
149                }
150            }
151        }
152
153        if text_style.white_space_collapse == WhiteSpaceCollapse::BreakSpaces &&
154            self.text[Utf8CodeUnits::to_usize_range(&segment.byte_range)]
155                .chars()
156                .last()
157                .is_some_and(breaks_for_break_spaces)
158        {
159            self.line_breaker
160                .set_additional_break_at_start(segment.byte_range.end);
161        }
162
163        (slices, break_at_start)
164    }
165}
166
167fn line_break_ignored_for_keep_all(
168    text: &str,
169    text_style: &InheritedText,
170    break_index: Utf8CodeUnits,
171) -> bool {
172    if text_style.word_break != WordBreak::KeepAll {
173        return false;
174    }
175
176    let break_index = break_index.0 as usize;
177    let text_before = &text[..break_index];
178    let Some(character_before) = text_before.chars().rev().find(|character| {
179        !matches!(
180            LineBreak::for_char(*character),
181            LineBreak::CombiningMark | LineBreak::ZWJ,
182        )
183    }) else {
184        return false;
185    };
186
187    if !suppresses_line_break_for_keep_all(character_before) {
188        return false;
189    }
190
191    text[break_index..]
192        .chars()
193        .next()
194        .is_some_and(suppresses_line_break_for_keep_all)
195}
196
197/// From <https://drafts.csswg.org/css-text-4/#valdef-word-break-keep-all>:
198/// > Breaking is forbidden within “words”: implicit soft wrap opportunities between
199/// > typographic letter units (or other typographic character units belonging to the NU,
200/// > AL, AI, or ID Unicode line breaking classes [UAX14]) are suppressed, i.e. breaks are
201/// > prohibited between pairs of such characters (regardless of line-break settings other
202/// > than anywhere) except where opportunities exist due to § 6.1.1.1 Lexical Word
203/// > Breaking. Otherwise this option is equivalent to normal. In this style, sequences of
204/// > CJK characters do not break.
205///
206/// From <https://drafts.csswg.org/css-text-4/#typographic-letter-unit>:
207/// > A typographic letter unit (or letter for the purpose of this specification) is a
208/// > typographic character unit belonging to one of the Letter or Number general
209/// > categories. See Appendix E: Characters and Properties for how to determine the Unicode
210/// > properties of a typographic character unit.
211fn suppresses_line_break_for_keep_all(character: char) -> bool {
212    let line_break_class = LineBreak::for_char(character);
213    (matches!(
214        GeneralCategory::for_char(character),
215        GeneralCategory::UppercaseLetter |
216            GeneralCategory::LowercaseLetter |
217            GeneralCategory::TitlecaseLetter |
218            GeneralCategory::ModifierLetter |
219            GeneralCategory::OtherLetter |
220            GeneralCategory::DecimalNumber |
221            GeneralCategory::LetterNumber |
222            GeneralCategory::OtherNumber
223    ) || matches!(
224        line_break_class,
225        LineBreak::Numeric | LineBreak::Alphabetic | LineBreak::Ambiguous | LineBreak::Ideographic
226    )) &&
227    // From <https://drafts.csswg.org/css-text-4/#lexical-breaking>:
228    // > To provide the expected normal behavior for Southeast Asian languages, typographic
229    // > character units with line breaking class SA in [UAX14] must be treated as if they
230    // > had class AL. However, the user agent must additionally analyze the content of a
231    // > run of such characters to detect word boundaries and treat each boundary as a soft
232    // > wrap opportunities.
233    //
234    // `ComplexContext` is the SA class here. `break-all` must not suppress dictionary-based
235    // word breaks inside SA text.
236    !matches!(line_break_class, LineBreak::ComplexContext)
237}
238
239fn breaks_for_break_spaces(character: char) -> bool {
240    match CssTextType::from(character) {
241        CssTextType::NonWhiteSpace => false,
242        CssTextType::DocumentWhiteSpace => true,
243        // From <https://www.unicode.org/reports/tr14/tr14-57.html#GL>:
244        // > Non-breaking characters prohibit breaks on either side, but that prohibition
245        // > can be overridden by SP or ZW.
246        //
247        // The specification also marks this class of characters as non-tailorable.
248        CssTextType::OtherSpaceSeparator => LineBreak::for_char(character) != LineBreak::Glue,
249    }
250}
251
252/// Returns a tuple containing the [`TrailingWhiteSpace`] values for the text and boolean
253/// indicating whether all of the content was white space.
254fn trailing_white_space_of(
255    text: &str,
256    style: &ComputedValues,
257) -> (TrailingWhiteSpace<Utf32CodeUnits>, bool) {
258    let (anything_hangable, anything_removable) =
259        match (style.get_white_space_collapse(), style.get_text_wrap_mode()) {
260            (WhiteSpaceCollapse::BreakSpaces, _) |
261            (WhiteSpaceCollapse::Preserve, TextWrapMode::Nowrap) => (false, false),
262            (WhiteSpaceCollapse::Preserve, TextWrapMode::Wrap) => (true, false),
263            _ => (true, true),
264        };
265
266    let mut removable = 0;
267    let mut hangable = 0;
268    let mut all_white_space = true;
269    for character in text.chars().rev() {
270        match CssTextType::from(character) {
271            CssTextType::NonWhiteSpace => {
272                all_white_space = false;
273                break;
274            },
275            CssTextType::DocumentWhiteSpace if hangable == 0 && anything_removable => {
276                removable += 1;
277            },
278            CssTextType::DocumentWhiteSpace | CssTextType::OtherSpaceSeparator
279                if anything_hangable =>
280            {
281                hangable += 1;
282            },
283            _ => {},
284        }
285    }
286
287    (
288        TrailingWhiteSpace {
289            hangable: Utf32CodeUnits(hangable),
290            removable: Utf32CodeUnits(removable),
291        },
292        all_white_space,
293    )
294}
295
296/// The [`ShapingQueue`] is responsible for shaping text during inline formatting context
297/// construction. It allows for shaping text across inline box boundaries. When pushing
298/// items to the queue, if the items are compatible pieces of text that can be shaped
299/// together, they are accumulated. The queue may be flushed in the given situations:
300///
301/// - An incompatible piece of text (different fonts or certain style properties) is
302///   pushed to the queue.
303/// - A preserved newline or tab is pushed to the queue.
304/// - An inline box breaks shaping via padding, border, margins or a non-`baseline`
305///   `vertical-align` property.
306/// - Atomic content in the inline formatting context
307///
308/// Upon flushing, the [`ShapingQueue`] will shape any pending text and assign the
309/// resulting [`ShapedTextSlice`]s to the originating [`TextRun`]s.
310pub(crate) struct ShapingQueue<'a> {
311    /// The queue of items in the current batch that will be shaped together.
312    queue: Vec<ShapingQueueText>,
313    /// The text that will be used for shaping.
314    text: &'a str,
315    /// The line breaker that will be used to slice shaping results across on line break boundaries.
316    line_breaker: LineBreaker,
317    /// The byte range of the text to shape in [`Self::text`] for the current batch.
318    /// Only contiguous ranges can be shaped together.
319    byte_range: Range<Utf8CodeUnits>,
320    /// The character range of the text to shape in [`Self::text`] for the current batch.
321    /// Only contiguous ranges can be shaped together.
322    character_range: Range<Utf32CodeUnits>,
323    /// The resolved script for the current batch. This is used to gradually turn non-specific
324    /// scripts into a resolved value for shaping.
325    resolved_script: Option<Script>,
326}
327
328impl<'a> ShapingQueue<'a> {
329    pub(crate) fn new(text: &'a str, line_break_options: LineBreakOptions<'_>) -> Self {
330        Self {
331            queue: Default::default(),
332            text,
333            line_breaker: LineBreaker::new(text, line_break_options),
334            byte_range: Default::default(),
335            character_range: Default::default(),
336            resolved_script: None,
337        }
338    }
339
340    fn compatible_old_shaping_result(
341        &self,
342        character_count: Utf32CodeUnits,
343    ) -> Option<Arc<ShapedText>> {
344        let old_shaped_text = self.queue.first()?.old_shaped_text.as_ref()?;
345        if old_shaped_text.character_count() != character_count {
346            return None;
347        }
348
349        if !self.queue.iter().all(|entry| {
350            entry
351                .old_shaped_text
352                .as_ref()
353                .is_some_and(|entry_old_shaped_text| {
354                    Arc::ptr_eq(old_shaped_text, entry_old_shaped_text)
355                })
356        }) {
357            return None;
358        }
359        Some(old_shaped_text.clone())
360    }
361
362    fn shape_batch(&self) -> Option<Arc<ShapedText>> {
363        let first = self.queue.first()?;
364
365        let character_count = self.character_range.end - self.character_range.start;
366        if let Some(old_shaping_result) = self.compatible_old_shaping_result(character_count) {
367            return Some(old_shaping_result);
368        };
369
370        let mut options: ShapingOptions = (&first.info).into();
371        options.script = self.resolved_script.unwrap_or(first.info.script);
372
373        let font = &first.info.font_info.font;
374        Some(font.shape_text(
375            &self.text[Utf8CodeUnits::to_usize_range(&self.byte_range)],
376            &options,
377        ))
378    }
379
380    /// Flush this [`ShapingQueue`]. If any content had been collected up to this point,
381    /// it will be shaped and the resulting [`ShapedTextSlice`]s will be assigned to their
382    /// originating [`TextRun`]s.
383    pub(crate) fn flush(&mut self) {
384        let Some(shaped_text) = self.shape_batch() else {
385            return;
386        };
387
388        let mut slicer = BatchSlicer {
389            slicer: ShapedTextSlicer::new(shaped_text.clone()),
390            text: self.text,
391            line_breaker: &mut self.line_breaker,
392            character_offset_origin: self.character_range.start,
393        };
394
395        for entry in self.queue.drain(..) {
396            let mut text_run = entry.text_run.borrow_mut();
397            let style = text_run.inline_styles().style.borrow().clone();
398            let (runs, break_at_start) =
399                slicer.slice_shaped_text_at_line_break_opportunities(&entry, &style);
400
401            if let TextRunItem::TextSegment(text_segment) =
402                &mut text_run.items[entry.index_in_text_run]
403            {
404                text_segment.shaped_text = Some(shaped_text.clone());
405                text_segment.runs = runs;
406                text_segment.break_at_start = break_at_start;
407            }
408        }
409    }
410
411    fn compatible_with_batch(&self, text: &ShapingQueueText) -> bool {
412        // If the queue is empty, we can always add new text to the batch.
413        let Some(last) = self.queue.last() else {
414            return true;
415        };
416
417        // The new text is only compatible with the current batch if their character and
418        // text byte boundaries are contiguous.
419        if last.character_range.end != text.character_range.start ||
420            last.byte_range.end != text.byte_range.start
421        {
422            return false;
423        }
424
425        // The `FontInfo`s of the batch and the new text need to match exactly to shape
426        // together.
427        if !Arc::ptr_eq(&last.info.font_info, &text.info.font_info) &&
428            *last.info.font_info != *text.info.font_info
429        {
430            return false;
431        }
432
433        // Any resolved `Script` has to be compatible with any new specific `Script`.
434        !script_is_specific(text.info.script) ||
435            self.resolved_script
436                .is_none_or(|resolved_script| resolved_script == text.info.script)
437    }
438
439    fn push_text(&mut self, text: ShapingQueueText) {
440        if !self.compatible_with_batch(&text) {
441            self.flush();
442        }
443
444        if self.queue.is_empty() {
445            self.character_range = text.character_range.clone();
446            self.byte_range = text.byte_range.clone();
447            self.resolved_script = None;
448        } else {
449            self.character_range.end = text.character_range.end;
450            self.byte_range.end = text.byte_range.end;
451        }
452        if self.resolved_script.is_none() && script_is_specific(text.info.script) {
453            self.resolved_script = Some(text.info.script);
454        }
455
456        self.queue.push(text);
457    }
458
459    /// Push a new [`ShapingQueueEntry`] on to this [`ShapingQueue`], maybe flushing
460    /// previously collected entries.
461    pub(crate) fn push(&mut self, entry: ShapingQueueEntry) {
462        match entry {
463            ShapingQueueEntry::PreservedTabOrNewline => self.flush(),
464            ShapingQueueEntry::Text(shaping_queue_text) => self.push_text(shaping_queue_text),
465        }
466    }
467}
468
469#[derive(Clone, Copy, Debug, PartialEq, Eq)]
470pub(crate) enum CssTextType {
471    /// A [`char`] that is non-space content.
472    NonWhiteSpace,
473    /// <https://drafts.csswg.org/css-text-3/#white-space>.
474    DocumentWhiteSpace,
475    /// <https://drafts.csswg.org/css-text-3/#other-space-separators>.
476    OtherSpaceSeparator,
477}
478
479impl From<char> for CssTextType {
480    fn from(character: char) -> Self {
481        match character {
482            // See <https://drafts.csswg.org/css-text-3/#white-space>:
483            // > Except where specified otherwise, white space processing in CSS affects only the
484            // > document white space characters: spaces (U+0020), tabs (U+0009), and segment breaks.
485            ' ' | '\t' | '\n' | '\r' => Self::DocumentWhiteSpace,
486            // This is a fast path to avoid having to do Unicode category classification for ASCII
487            // characters.
488            _ if character.is_ascii() => Self::NonWhiteSpace,
489            // See <https://drafts.csswg.org/css-text-3/#other-space-separators>:
490            // > Besides space (U+0020) and no-break space (U+00A0), Unicode defines a number of
491            // > additional space separator characters. [UNICODE] In this specification all characters
492            // > in the Unicode general category Zs except space (U+0020) and no-break space (U+00A0)
493            // > are collectively referred to as other space separators.
494            //
495            // Note: ' ' (space) is handled above.
496            _ if GeneralCategory::for_char(character) == GeneralCategory::SpaceSeparator &&
497                character != '\u{00a0}' =>
498            {
499                Self::OtherSpaceSeparator
500            },
501            _ => Self::NonWhiteSpace,
502        }
503    }
504}