Skip to main content

turnframe_understand/
words.rs

1//! The user's words, numbered, so a task can point at them.
2//!
3//! A message is split on whitespace, punctuation staying attached to its word, and
4//! grouped into sentences after terminal punctuation. A pointer is a closed range of
5//! word indices; code slices the exact words and their byte offsets. A value's copied
6//! words may narrow a pointer to the words inside it, never move it: nothing a model
7//! writes is looked for outside the words it pointed at.
8
9use serde::{Deserialize, Serialize};
10use turnframe_core::understanding::WordRange;
11
12/// One word of a message and where it sits.
13#[derive(Debug, Clone, PartialEq, Eq)]
14pub struct Word {
15    /// The word as written, punctuation included.
16    pub text: String,
17    /// Byte offset of its first character.
18    pub start: usize,
19    /// Byte offset just past its last character.
20    pub end: usize,
21}
22
23/// A message split into numbered words.
24#[derive(Debug, Clone, PartialEq, Eq)]
25pub struct Words {
26    text: String,
27    words: Vec<Word>,
28}
29
30/// A closed range of word indices, `from` to `to` inclusive, counted from 0.
31///
32/// A model is shown and answers word numbers counted from 1, the way people count;
33/// [`Span::shown`] and the serde form are that numbering, and nothing else is.
34#[derive(Debug, Clone, Copy, PartialEq, Eq, Hash, PartialOrd, Ord)]
35pub struct Span {
36    /// First word.
37    pub from: usize,
38    /// Last word.
39    pub to: usize,
40}
41
42impl Span {
43    /// A span from `from` to `to` inclusive.
44    #[must_use]
45    pub const fn new(from: usize, to: usize) -> Self {
46        Self { from, to }
47    }
48
49    /// The span a model wrote as words `from` to `to`, counted from 1. A 0 does not
50    /// fit any message, so it stays out of range for the check to report.
51    #[must_use]
52    pub const fn from_shown(from: usize, to: usize) -> Self {
53        Self::new(from.wrapping_sub(1), to.wrapping_sub(1))
54    }
55
56    /// The word numbers a model is shown, counted from 1.
57    #[must_use]
58    pub const fn shown(self) -> (usize, usize) {
59        (self.from.wrapping_add(1), self.to.wrapping_add(1))
60    }
61}
62
63/// How a span travels to and from a model: word numbers counted from 1.
64#[derive(Serialize, Deserialize)]
65struct ShownSpan {
66    from: usize,
67    to: usize,
68}
69
70impl Serialize for Span {
71    fn serialize<S: serde::Serializer>(&self, serializer: S) -> Result<S::Ok, S::Error> {
72        let (from, to) = self.shown();
73        ShownSpan { from, to }.serialize(serializer)
74    }
75}
76
77impl<'de> Deserialize<'de> for Span {
78    fn deserialize<D: serde::Deserializer<'de>>(deserializer: D) -> Result<Self, D::Error> {
79        let shown = ShownSpan::deserialize(deserializer)?;
80        Ok(Self::from_shown(shown.from, shown.to))
81    }
82}
83
84/// A span that does not fit the message it points into.
85#[derive(Debug, Clone, PartialEq, Eq, thiserror::Error)]
86#[error("words {from} to {to} are not in a message of {count} words")]
87pub struct SpanError {
88    /// First word asked for.
89    pub from: usize,
90    /// Last word asked for.
91    pub to: usize,
92    /// Words in the message.
93    pub count: usize,
94}
95
96impl Words {
97    /// Splits `text`.
98    #[must_use]
99    pub fn split(text: &str) -> Self {
100        let mut words = Vec::new();
101        let mut start = None;
102        for (index, character) in text.char_indices() {
103            match (character.is_whitespace(), start) {
104                (true, Some(begin)) => {
105                    words.push(Word {
106                        text: text[begin..index].to_owned(),
107                        start: begin,
108                        end: index,
109                    });
110                    start = None;
111                }
112                (false, None) => start = Some(index),
113                _ => {}
114            }
115        }
116        if let Some(begin) = start {
117            words.push(Word {
118                text: text[begin..].to_owned(),
119                start: begin,
120                end: text.len(),
121            });
122        }
123        Self {
124            text: text.to_owned(),
125            words,
126        }
127    }
128
129    /// The message as given.
130    #[must_use]
131    pub fn text(&self) -> &str {
132        &self.text
133    }
134
135    /// How many words it has.
136    #[must_use]
137    pub fn len(&self) -> usize {
138        self.words.len()
139    }
140
141    /// Whether it has none.
142    #[must_use]
143    pub fn is_empty(&self) -> bool {
144        self.words.is_empty()
145    }
146
147    /// The words.
148    #[must_use]
149    pub fn words(&self) -> &[Word] {
150        &self.words
151    }
152
153    /// Checks that `span` is inside the message.
154    ///
155    /// # Errors
156    ///
157    /// [`SpanError`] when it is not.
158    pub const fn check(&self, span: Span) -> Result<(), SpanError> {
159        if span.from <= span.to && span.to < self.words.len() {
160            Ok(())
161        } else {
162            Err(SpanError {
163                from: span.from,
164                to: span.to,
165                count: self.words.len(),
166            })
167        }
168    }
169
170    /// The exact text a span covers, from its first word's start to its last word's end.
171    ///
172    /// # Errors
173    ///
174    /// [`SpanError`] when the span is not inside the message.
175    pub fn slice(&self, span: Span) -> Result<&str, SpanError> {
176        let (start, end) = self.bytes(span)?;
177        Ok(&self.text[start..end])
178    }
179
180    /// The byte range a span covers.
181    ///
182    /// # Errors
183    ///
184    /// [`SpanError`] when the span is not inside the message.
185    pub fn bytes(&self, span: Span) -> Result<(usize, usize), SpanError> {
186        self.check(span)?;
187        Ok((self.words[span.from].start, self.words[span.to].end))
188    }
189
190    /// The words inside `span` that `copied` repeats, compared without case and without
191    /// the punctuation at either end of a word; `None` when they are not there.
192    #[must_use]
193    pub fn narrow(&self, span: Span, copied: &str) -> Option<Span> {
194        self.places(span, copied)?.into_iter().next()
195    }
196
197    /// Where in the whole message `copied` is repeated, when it is repeated in one place only.
198    #[must_use]
199    pub fn only_place(&self, copied: &str) -> Option<Span> {
200        let whole = Span::new(0, self.words.len().checked_sub(1)?);
201        match self.places(whole, copied)?.as_slice() {
202            [place] => Some(*place),
203            _ => None,
204        }
205    }
206
207    /// Every run of words inside `span` that repeats `copied`, first to last.
208    fn places(&self, span: Span, copied: &str) -> Option<Vec<Span>> {
209        let wanted: Vec<String> = copied.split_whitespace().map(bare).collect();
210        let wanted: Vec<&String> = wanted.iter().filter(|word| !word.is_empty()).collect();
211        self.check(span).ok()?;
212        if wanted.is_empty() || wanted.len() > span.to - span.from + 1 {
213            return None;
214        }
215        let inside: Vec<String> = self.words[span.from..=span.to]
216            .iter()
217            .map(|word| bare(&word.text))
218            .collect();
219        Some(
220            (0..=inside.len() - wanted.len())
221                .filter(|start| {
222                    wanted
223                        .iter()
224                        .zip(&inside[*start..])
225                        .all(|(want, have)| *want == have)
226                })
227                .map(|start| Span::new(span.from + start, span.from + start + wanted.len() - 1))
228                .collect(),
229        )
230    }
231
232    /// The words a span covers, with their byte range.
233    ///
234    /// # Errors
235    ///
236    /// [`SpanError`] when the span is not inside the message.
237    pub fn range(&self, span: Span) -> Result<WordRange, SpanError> {
238        let (start, end) = self.bytes(span)?;
239        Ok(WordRange {
240            first: span.from,
241            last: span.to,
242            start,
243            end,
244        })
245    }
246
247    /// The message rendered for a model: one line per sentence, each word prefixed
248    /// with its number counted from 1, `S1: [1]I [2]want`.
249    #[must_use]
250    pub fn render(&self) -> String {
251        let mut out = String::new();
252        let mut sentence = 1;
253        let mut open = false;
254        for (index, word) in self.words.iter().enumerate() {
255            if !open {
256                if sentence > 1 {
257                    out.push('\n');
258                }
259                out.push_str(&format!("S{sentence}:"));
260                open = true;
261            }
262            out.push_str(&format!(" [{}]{}", index + 1, word.text));
263            if ends_sentence(&word.text) {
264                sentence += 1;
265                open = false;
266            }
267        }
268        out
269    }
270}
271
272/// Whether a word closes a sentence: it ends with `.`, `!`, `?` or `…`, possibly
273/// followed by closing quotes or brackets.
274fn ends_sentence(word: &str) -> bool {
275    word.trim_end_matches(['"', '\'', ')', ']', '»', '”', '’'])
276        .ends_with(['.', '!', '?', '…'])
277}
278
279/// A word without case and without the punctuation at either end.
280fn bare(word: &str) -> String {
281    word.trim_matches(|c: char| !c.is_alphanumeric())
282        .to_lowercase()
283}
284
285#[cfg(test)]
286mod tests {
287    use super::*;
288
289    #[test]
290    fn words_keep_their_punctuation_and_their_byte_offsets() {
291        let words = Words::split("  set the date, then «ciao»!  ");
292        let texts: Vec<&str> = words.words().iter().map(|w| w.text.as_str()).collect();
293        assert_eq!(texts, vec!["set", "the", "date,", "then", "«ciao»!"]);
294        assert_eq!(words.slice(Span::new(1, 2)).unwrap(), "the date,");
295        assert_eq!(words.slice(Span::new(4, 4)).unwrap(), "«ciao»!");
296        assert!(words.slice(Span::new(3, 5)).is_err());
297        assert!(words.slice(Span::new(2, 1)).is_err());
298    }
299
300    #[test]
301    fn rendering_numbers_words_and_breaks_sentences() {
302        let words = Words::split("Add a bag. Then set the name? Thanks");
303        assert_eq!(
304            words.render(),
305            "S1: [1]Add [2]a [3]bag.\nS2: [4]Then [5]set [6]the [7]name?\nS3: [8]Thanks"
306        );
307    }
308
309    #[test]
310    fn a_copy_has_one_place_only_when_the_message_repeats_it_once() {
311        let words = Words::split("the Lisbon offsite, not the Porto one");
312        assert_eq!(words.only_place("Lisbon offsite"), Some(Span::new(1, 2)));
313        assert_eq!(words.only_place("the"), None);
314        assert_eq!(words.only_place("Madrid"), None);
315    }
316
317    #[test]
318    fn multi_byte_characters_slice_on_their_boundaries() {
319        let words = Words::split("perché l'aereo è «puntuale»");
320        assert_eq!(words.slice(Span::new(2, 3)).unwrap(), "è «puntuale»");
321        let (start, end) = words.bytes(Span::new(0, 0)).unwrap();
322        assert_eq!(&words.text()[start..end], "perché");
323    }
324
325    #[test]
326    fn copied_words_narrow_a_pointer_and_never_move_it() {
327        let words = Words::split("nome: Offsite Lisbona, partenza domani");
328        let all = Span::new(0, 4);
329        assert_eq!(words.narrow(all, "offsite lisbona"), Some(Span::new(1, 2)));
330        assert_eq!(
331            words.narrow(all, "«Offsite Lisbona»"),
332            Some(Span::new(1, 2))
333        );
334        assert_eq!(
335            words.narrow(Span::new(3, 4), "offsite"),
336            None,
337            "outside the pointer"
338        );
339        assert_eq!(
340            words.narrow(all, "offsite roma"),
341            None,
342            "not the user's words"
343        );
344        assert_eq!(words.narrow(all, "  "), None);
345    }
346
347    #[test]
348    fn a_model_counts_words_from_one() {
349        let span: Span = serde_json::from_value(serde_json::json!({"from": 1, "to": 3})).unwrap();
350        assert_eq!(span, Span::new(0, 2));
351        assert_eq!(
352            serde_json::to_value(span).unwrap(),
353            serde_json::json!({"from": 1, "to": 3})
354        );
355        let zero: Span = serde_json::from_value(serde_json::json!({"from": 0, "to": 0})).unwrap();
356        assert!(Words::split("one two").check(zero).is_err(), "0 is no word");
357    }
358}