Skip to main content

rustpython_vm/builtins/
str.rs

1use super::{
2    PositionIterInternal, PyBytesRef, PyDict, PyList, PyTuple, PyTupleRef, PyType, PyTypeRef,
3    int::{PyInt, PyIntRef},
4    iter::{IterStatus, builtins_iter},
5};
6use crate::{
7    AsObject, Context, Py, PyExact, PyObject, PyObjectRef, PyPayload, PyRef, PyRefExact, PyResult,
8    TryFromBorrowedObject, TryFromObject, VirtualMachine,
9    anystr::{self, AnyStr, AnyStrContainer, AnyStrWrapper, StringRange, adjust_indices},
10    atomic_func,
11    bytes_inner::{swapcase_ascii, title_ascii},
12    cformat::cformat_string,
13    class::{PyClassDef, PyClassImpl},
14    common::{
15        lock::LazyLock,
16        str::{PyKindStr, StrData, StrKind},
17    },
18    convert::{IntoPyException, ToPyException, ToPyObject, ToPyResult},
19    format::{format, format_map},
20    function::{ArgIterable, FuncArgs, OptionalArg, PyComparisonValue, PySsize},
21    intern::PyInterned,
22    object::{MaybeTraverse, Traverse, TraverseFn},
23    protocol::{
24        BufferFlags, PyBuffer, PyIterReturn, PyMappingMethods, PyNumberMethods, PySequenceMethods,
25    },
26    sequence::SequenceExt,
27    sliceable::{SequenceIndex, SliceableSequenceOp},
28    types::{
29        AsMapping, AsNumber, AsSequence, Comparable, Constructor, Hashable, IterNext, Iterable,
30        PyComparisonOp, Representable, SelfIter,
31    },
32};
33use alloc::{borrow::Cow, fmt};
34use ascii::{AsciiChar, AsciiStr, AsciiString};
35use bstr::ByteSlice;
36use core::ffi::CStr;
37use core::{char, mem, ops::Range};
38use itertools::Itertools;
39use memchr::memchr;
40use num_traits::ToPrimitive;
41use rustpython_common::{
42    ascii,
43    atomic::{self, PyAtomic, Radium},
44    format::{FormatSpec, FormatString, FromTemplate},
45    hash,
46    lock::PyMutex,
47    str::DeduceStrKind,
48    wtf8::{CodePoint, Wtf8, Wtf8Buf, Wtf8Concat},
49};
50
51use rustpython_unicode::{self as unicode, case};
52
53impl<'a> TryFromBorrowedObject<'a> for String {
54    fn try_from_borrowed_object(vm: &VirtualMachine, obj: &'a PyObject) -> PyResult<Self> {
55        obj.try_value_with(|pystr: &Py<PyUtf8Str>| Ok(pystr.as_str().to_owned()), vm)
56    }
57}
58
59impl<'a> TryFromBorrowedObject<'a> for &'a str {
60    fn try_from_borrowed_object(vm: &VirtualMachine, obj: &'a PyObject) -> PyResult<Self> {
61        let pystr: &Py<PyUtf8Str> = TryFromBorrowedObject::try_from_borrowed_object(vm, obj)?;
62        Ok(pystr.as_str())
63    }
64}
65
66impl<'a> TryFromBorrowedObject<'a> for &'a Wtf8 {
67    fn try_from_borrowed_object(vm: &VirtualMachine, obj: &'a PyObject) -> PyResult<Self> {
68        let pystr: &Py<PyStr> = TryFromBorrowedObject::try_from_borrowed_object(vm, obj)?;
69        Ok(pystr.as_wtf8())
70    }
71}
72
73pub type PyStrRef = PyRef<PyStr>;
74pub type PyUtf8StrRef = PyRef<PyUtf8Str>;
75
76#[pyclass(module = false, name = "str")]
77pub struct PyStr {
78    data: StrData,
79    hash: PyAtomic<hash::PyHash>,
80}
81
82impl fmt::Debug for PyStr {
83    fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result {
84        f.debug_struct("PyStr")
85            .field("value", &self.as_wtf8())
86            .field("kind", &self.data.kind())
87            .field("hash", &self.hash)
88            .finish()
89    }
90}
91
92impl AsRef<str> for PyStr {
93    #[track_caller] // <- can remove this once it doesn't panic
94    fn as_ref(&self) -> &str {
95        self.to_str().expect("str has surrogates")
96    }
97}
98
99impl AsRef<str> for Py<PyStr> {
100    #[track_caller] // <- can remove this once it doesn't panic
101    fn as_ref(&self) -> &str {
102        self.to_str().expect("str has surrogates")
103    }
104}
105
106impl AsRef<str> for PyStrRef {
107    #[track_caller] // <- can remove this once it doesn't panic
108    fn as_ref(&self) -> &str {
109        self.to_str().expect("str has surrogates")
110    }
111}
112
113impl AsRef<Wtf8> for PyStr {
114    fn as_ref(&self) -> &Wtf8 {
115        self.as_wtf8()
116    }
117}
118
119impl AsRef<Wtf8> for Py<PyStr> {
120    fn as_ref(&self) -> &Wtf8 {
121        self.as_wtf8()
122    }
123}
124
125impl AsRef<Wtf8> for PyStrRef {
126    fn as_ref(&self) -> &Wtf8 {
127        self.as_wtf8()
128    }
129}
130
131impl Wtf8Concat for PyStr {
132    #[inline]
133    fn fmt_wtf8(&self, buf: &mut Wtf8Buf) {
134        buf.push_wtf8(self.as_wtf8());
135    }
136}
137
138impl Wtf8Concat for Py<PyStr> {
139    #[inline]
140    fn fmt_wtf8(&self, buf: &mut Wtf8Buf) {
141        buf.push_wtf8(self.as_wtf8());
142    }
143}
144
145impl<'a> From<&'a AsciiStr> for PyStr {
146    fn from(s: &'a AsciiStr) -> Self {
147        s.to_owned().into()
148    }
149}
150
151impl From<AsciiString> for PyStr {
152    fn from(s: AsciiString) -> Self {
153        s.into_boxed_ascii_str().into()
154    }
155}
156
157impl From<Box<AsciiStr>> for PyStr {
158    fn from(s: Box<AsciiStr>) -> Self {
159        StrData::from(s).into()
160    }
161}
162
163impl From<AsciiChar> for PyStr {
164    fn from(ch: AsciiChar) -> Self {
165        AsciiString::from(ch).into()
166    }
167}
168
169impl<'a> From<&'a str> for PyStr {
170    fn from(s: &'a str) -> Self {
171        s.to_owned().into()
172    }
173}
174
175impl<'a> From<&'a Wtf8> for PyStr {
176    fn from(s: &'a Wtf8) -> Self {
177        s.to_owned().into()
178    }
179}
180
181impl From<String> for PyStr {
182    fn from(s: String) -> Self {
183        s.into_boxed_str().into()
184    }
185}
186
187impl From<Wtf8Buf> for PyStr {
188    fn from(w: Wtf8Buf) -> Self {
189        w.into_box().into()
190    }
191}
192
193impl From<char> for PyStr {
194    fn from(ch: char) -> Self {
195        StrData::from(ch).into()
196    }
197}
198
199impl From<CodePoint> for PyStr {
200    fn from(ch: CodePoint) -> Self {
201        StrData::from(ch).into()
202    }
203}
204
205impl From<StrData> for PyStr {
206    fn from(data: StrData) -> Self {
207        Self {
208            data,
209            hash: Radium::new(hash::SENTINEL),
210        }
211    }
212}
213
214impl<'a> From<alloc::borrow::Cow<'a, str>> for PyStr {
215    fn from(s: alloc::borrow::Cow<'a, str>) -> Self {
216        s.into_owned().into()
217    }
218}
219
220impl From<Box<str>> for PyStr {
221    #[inline]
222    fn from(value: Box<str>) -> Self {
223        StrData::from(value).into()
224    }
225}
226
227impl From<Box<Wtf8>> for PyStr {
228    #[inline]
229    fn from(value: Box<Wtf8>) -> Self {
230        StrData::from(value).into()
231    }
232}
233
234impl Default for PyStr {
235    fn default() -> Self {
236        Self {
237            data: StrData::default(),
238            hash: Radium::new(hash::SENTINEL),
239        }
240    }
241}
242
243impl fmt::Display for PyStr {
244    #[inline]
245    fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result {
246        self.as_wtf8().fmt(f)
247    }
248}
249
250pub trait AsPyStr<'a>
251where
252    Self: 'a,
253{
254    #[allow(
255        clippy::wrong_self_convention,
256        reason = "this trait is intentionally implemented for references"
257    )]
258    fn as_pystr(self, ctx: &Context) -> &'a Py<PyStr>;
259}
260
261impl<'a> AsPyStr<'a> for &'a Py<PyStr> {
262    #[inline]
263    fn as_pystr(self, _ctx: &Context) -> &'a Py<PyStr> {
264        self
265    }
266}
267
268impl<'a> AsPyStr<'a> for &'a Py<PyUtf8Str> {
269    #[inline]
270    fn as_pystr(self, _ctx: &Context) -> &'a Py<PyStr> {
271        Py::<PyUtf8Str>::as_pystr(self)
272    }
273}
274
275impl<'a> AsPyStr<'a> for &'a PyStrRef {
276    #[inline]
277    fn as_pystr(self, _ctx: &Context) -> &'a Py<PyStr> {
278        self
279    }
280}
281
282impl<'a> AsPyStr<'a> for &'a PyUtf8StrRef {
283    #[inline]
284    fn as_pystr(self, _ctx: &Context) -> &'a Py<PyStr> {
285        Py::<PyUtf8Str>::as_pystr(self)
286    }
287}
288
289impl AsPyStr<'static> for &'static str {
290    #[inline]
291    fn as_pystr(self, ctx: &Context) -> &'static Py<PyStr> {
292        ctx.intern_str(self)
293    }
294}
295
296impl<'a> AsPyStr<'a> for &'a PyStrInterned {
297    #[inline]
298    fn as_pystr(self, _ctx: &Context) -> &'a Py<PyStr> {
299        self
300    }
301}
302
303impl<'a> AsPyStr<'a> for &'a PyUtf8StrInterned {
304    #[inline]
305    fn as_pystr(self, _ctx: &Context) -> &'a Py<PyStr> {
306        Py::<PyUtf8Str>::as_pystr(self)
307    }
308}
309
310#[pyclass(module = false, name = "str_iterator", traverse = "manual")]
311#[derive(Debug)]
312pub(crate) struct PyStrIterator {
313    internal: PyMutex<(PositionIterInternal<PyStrRef>, usize)>,
314}
315
316unsafe impl Traverse for PyStrIterator {
317    fn traverse(&self, tracer: &mut TraverseFn<'_>) {
318        // No need to worry about deadlock, for inner is a PyStr and can't make ref cycle
319        self.internal.lock().0.traverse(tracer);
320    }
321}
322
323impl PyPayload for PyStrIterator {
324    fn class(ctx: &Context) -> &'static Py<PyType> {
325        ctx.types.str_iterator_type
326    }
327}
328
329#[pyclass(flags(DISALLOW_INSTANTIATION), with(IterNext, Iterable))]
330impl Py<PyStrIterator> {
331    #[pymethod]
332    fn __length_hint__(&self) -> usize {
333        self.internal.lock().0.length_hint(|obj| obj.char_len())
334    }
335
336    #[pymethod]
337    fn __setstate__(&self, object: PyObjectRef, vm: &VirtualMachine) -> PyResult<()> {
338        let mut internal = self.internal.lock();
339        internal.1 = usize::MAX;
340        internal
341            .0
342            .set_state(&object, |obj, pos| pos.min(obj.char_len()), vm)
343    }
344
345    #[pymethod]
346    fn __reduce__(&self, vm: &VirtualMachine) -> PyResult<PyTupleRef> {
347        let func = builtins_iter(vm)?;
348        Ok(self.internal.lock().0.reduce(
349            func,
350            |x| x.clone().into(),
351            |vm| vm.ctx.empty_str.to_owned().into(),
352            vm,
353        ))
354    }
355}
356
357impl SelfIter for PyStrIterator {}
358
359impl IterNext for PyStrIterator {
360    fn next(zelf: &Py<Self>, vm: &VirtualMachine) -> PyResult<PyIterReturn> {
361        let mut internal = zelf.internal.lock();
362
363        if let IterStatus::Active(s) = &internal.0.status {
364            let value = s.as_wtf8();
365
366            if internal.1 == usize::MAX {
367                if let Some((offset, ch)) = value.code_point_indices().nth(internal.0.position) {
368                    internal.0.position += 1;
369                    internal.1 = offset + ch.len_wtf8();
370                    return Ok(PyIterReturn::Return(ch.to_pyobject(vm)));
371                }
372            } else if let Some(value) = value.get(internal.1..)
373                && let Some(ch) = value.code_points().next()
374            {
375                internal.0.position += 1;
376                internal.1 += ch.len_wtf8();
377                return Ok(PyIterReturn::Return(ch.to_pyobject(vm)));
378            }
379            let released = internal.0.exhaust();
380            // The string is released after the lock. A `__del__` that iterates
381            // again would otherwise reach for a lock this call still holds.
382            drop(internal);
383            drop(released);
384        }
385        Ok(PyIterReturn::StopIteration(None))
386    }
387}
388
389#[derive(FromArgs)]
390pub struct StrArgs {
391    #[pyarg(any, optional)]
392    object: OptionalArg<PyObjectRef>,
393    #[pyarg(any, optional)]
394    encoding: OptionalArg<PyUtf8StrRef>,
395    #[pyarg(any, optional)]
396    errors: OptionalArg<PyUtf8StrRef>,
397}
398
399impl Constructor for PyStr {
400    type Args = StrArgs;
401
402    fn slot_new(cls: PyTypeRef, func_args: FuncArgs, vm: &VirtualMachine) -> PyResult {
403        // Optimization: return exact str as-is (only when no encoding/errors provided)
404        if cls.is(vm.ctx.types.str_type)
405            && func_args.args.len() == 1
406            && func_args.kwargs.is_empty()
407            && func_args.args[0].class().is(vm.ctx.types.str_type)
408        {
409            return Ok(func_args.args[0].clone());
410        }
411
412        let args: Self::Args = func_args.bind_for(vm, Self::NAME)?;
413
414        // CPython parity: when cls is exactly str, return the __str__ / __repr__
415        // result as-is so any str subclass type the user returned is preserved
416        // (matches unicode_new_impl which only invokes unicode_subtype_new when
417        // type != &PyUnicode_Type).
418        // CPython parity: `errors` without `encoding` also triggers decode
419        // mode (with default UTF-8). The fast-path repr only applies when
420        // BOTH `encoding` and `errors` are missing.
421        if cls.is(vm.ctx.types.str_type)
422            && args.encoding.is_missing()
423            && args.errors.is_missing()
424            && let OptionalArg::Present(input) = &args.object
425        {
426            return Ok(input.str(vm)?.into());
427        }
428
429        let payload = Self::py_new(&cls, args, vm)?;
430        payload.into_ref_with_type(vm, cls).map(Into::into)
431    }
432
433    fn py_new(_cls: &Py<PyType>, args: Self::Args, vm: &VirtualMachine) -> PyResult<Self> {
434        match args.object {
435            OptionalArg::Present(input) => {
436                let encoding = args.encoding.into_option();
437                let errors = args.errors.into_option();
438                // CPython parity: presence of `encoding` OR `errors` triggers
439                // decode mode. When `errors` is given alone, the encoding
440                // defaults to UTF-8.
441                if encoding.is_some() || errors.is_some() {
442                    // CPython rejects str / non-bytes-like input early with
443                    // specific TypeError wording (unicode_new_impl).
444                    if input.fast_isinstance(vm.ctx.types.str_type) {
445                        return Err(vm.new_type_error("decoding str is not supported"));
446                    }
447                    let input = if input.fast_isinstance(vm.ctx.types.bytes_type)
448                        || input.fast_isinstance(vm.ctx.types.bytearray_type)
449                    {
450                        input
451                    } else {
452                        // PyUnicode_FromEncodedObject: whatever an exporter
453                        // complains about, the argument is simply not bytes-like.
454                        let buffer = PyBuffer::from_object(vm, &input, BufferFlags::SIMPLE)
455                            .map_err(|_| {
456                                vm.new_type_error(format!(
457                                    "decoding to str: need a bytes-like object, {} found",
458                                    input.class().name()
459                                ))
460                            })?;
461                        vm.ctx
462                            .new_bytes(buffer.contiguous_or_collect(<[u8]>::to_vec))
463                            .into()
464                    };
465                    let enc_str = encoding.as_ref().map_or("utf-8", |e| e.as_str());
466                    let s = vm
467                        .state
468                        .codec_registry
469                        .decode_text_object(input, enc_str, errors, vm)?;
470                    Ok(Self::from(s.as_wtf8().to_owned()))
471                } else {
472                    let s = input.str(vm)?;
473                    Ok(Self::from(s.as_wtf8().to_owned()))
474                }
475            }
476            OptionalArg::Missing => Ok(Self::from(String::new())),
477        }
478    }
479}
480
481impl PyStr {
482    /// # Safety: Given `bytes` must be valid data for given `kind`
483    unsafe fn new_str_unchecked(data: Box<Wtf8>, kind: StrKind) -> Self {
484        unsafe { StrData::new_str_unchecked(data, kind) }.into()
485    }
486
487    unsafe fn new_with_char_len<T: DeduceStrKind + Into<Box<Wtf8>>>(s: T, char_len: usize) -> Self {
488        let kind = s.str_kind();
489        unsafe { StrData::new_with_char_len(s.into(), kind, char_len) }.into()
490    }
491
492    /// # Safety
493    /// Given `bytes` must be ascii
494    #[must_use]
495    pub unsafe fn new_ascii_unchecked(bytes: Vec<u8>) -> Self {
496        unsafe { AsciiString::from_ascii_unchecked(bytes) }.into()
497    }
498
499    #[deprecated(note = "use PyStr::from(...).into_ref() instead")]
500    pub fn new_ref(zelf: impl Into<Self>, ctx: &Context) -> PyRef<Self> {
501        let zelf = zelf.into();
502        zelf.into_ref(ctx)
503    }
504
505    fn new_substr(&self, s: Wtf8Buf) -> Self {
506        let kind = if self.kind().is_ascii() || s.is_ascii() {
507            StrKind::Ascii
508        } else if self.kind().is_utf8() || s.is_utf8() {
509            StrKind::Utf8
510        } else {
511            StrKind::Wtf8
512        };
513        unsafe {
514            // SAFETY: kind is properly decided for substring
515            Self::new_str_unchecked(s.into(), kind)
516        }
517    }
518
519    #[inline]
520    pub const fn as_wtf8(&self) -> &Wtf8 {
521        self.data.as_wtf8()
522    }
523
524    pub const fn as_bytes(&self) -> &[u8] {
525        self.data.as_wtf8().as_bytes()
526    }
527
528    pub fn to_str(&self) -> Option<&str> {
529        self.data.as_str()
530    }
531
532    /// Returns `&str`
533    ///
534    /// # Panic
535    /// If the string contains surrogates.
536    #[inline]
537    #[track_caller]
538    pub fn expect_str(&self) -> &str {
539        self.to_str().expect("PyStr contains surrogates")
540    }
541
542    pub(crate) fn ensure_valid_utf8(&self, vm: &VirtualMachine) -> PyResult<()> {
543        if self.is_utf8() {
544            Ok(())
545        } else {
546            let start = self
547                .as_wtf8()
548                .code_points()
549                .position(|c| c.to_char().is_none())
550                .unwrap();
551            Err(vm.new_unicode_encode_error(
552                identifier!(vm, utf_8).to_owned(),
553                vm.ctx.new_str(self.data.clone()),
554                start,
555                start + 1,
556                vm.ctx.new_str("surrogates not allowed"),
557            ))
558        }
559    }
560
561    /// Check string bytes for interior NULs.
562    #[inline]
563    #[must_use]
564    pub fn contains_nuls(&self) -> bool {
565        memchr(b'\0', self.as_bytes()).is_some()
566    }
567
568    pub fn to_string_lossy(&self) -> Cow<'_, str> {
569        self.to_str()
570            .map_or_else(|| self.as_wtf8().to_string_lossy(), Cow::Borrowed)
571    }
572
573    pub const fn kind(&self) -> StrKind {
574        self.data.kind()
575    }
576
577    #[inline]
578    pub fn as_str_kind(&self) -> PyKindStr<'_> {
579        self.data.as_str_kind()
580    }
581
582    pub const fn is_utf8(&self) -> bool {
583        self.kind().is_utf8()
584    }
585
586    fn char_all<F>(&self, test: F) -> bool
587    where
588        F: Fn(char) -> bool,
589    {
590        match self.as_str_kind() {
591            PyKindStr::Ascii(s) => s.chars().all(|ch| test(ch.into())),
592            PyKindStr::Utf8(s) => s.chars().all(test),
593            PyKindStr::Wtf8(w) => w.code_points().all(|ch| ch.is_char_and(&test)),
594        }
595    }
596
597    fn repeat(zelf: PyRef<Self>, value: isize, vm: &VirtualMachine) -> PyResult<PyRef<Self>> {
598        if value == 0 && zelf.class().is(vm.ctx.types.str_type) {
599            // Special case: when some `str` is multiplied by `0`,
600            // returns the empty `str`.
601            return Ok(vm.ctx.empty_str.to_owned());
602        }
603        if (value == 1 || zelf.is_empty()) && zelf.class().is(vm.ctx.types.str_type) {
604            // Special case: when some `str` is multiplied by `1` or is the empty `str`,
605            // nothing really happens, we need to return an object itself
606            // with the same `id()` to be compatible with CPython.
607            // This only works for `str` itself, not its subclasses.
608            return Ok(zelf);
609        }
610        zelf.as_wtf8()
611            .as_bytes()
612            .mul(vm, value)
613            .map(|x| Self::from(unsafe { Wtf8Buf::from_bytes_unchecked(x) }).into_ref(&vm.ctx))
614    }
615
616    pub fn as_utf8(&self) -> Option<&PyUtf8Str> {
617        if self.is_utf8() {
618            // SAFETY: is_utf8() guarantees the PyUtf8Str invariant.
619            Some(unsafe { &*(self as *const Self as *const PyUtf8Str) })
620        } else {
621            None
622        }
623    }
624
625    pub fn try_as_utf8<'a>(&'a self, vm: &VirtualMachine) -> PyResult<&'a PyUtf8Str> {
626        self.as_utf8()
627            .ok_or_else(|| self.ensure_valid_utf8(vm).unwrap_err())
628    }
629}
630
631impl Py<PyStr> {
632    #[inline]
633    pub fn as_wtf8(&self) -> &Wtf8 {
634        self.payload().as_wtf8()
635    }
636
637    #[inline]
638    pub fn as_bytes(&self) -> &[u8] {
639        self.payload().as_bytes()
640    }
641
642    pub fn as_utf8(&self) -> Option<&Py<PyUtf8Str>> {
643        if self.is_utf8() {
644            // SAFETY: is_utf8() guarantees the PyUtf8Str invariant.
645            Some(unsafe { &*(self as *const Self as *const Py<PyUtf8Str>) })
646        } else {
647            None
648        }
649    }
650
651    pub fn try_as_utf8<'a>(&'a self, vm: &VirtualMachine) -> PyResult<&'a Py<PyUtf8Str>> {
652        self.as_utf8()
653            .ok_or_else(|| self.ensure_valid_utf8(vm).unwrap_err())
654    }
655}
656
657impl PyStr {
658    fn __add__(zelf: PyRef<Self>, other: &PyObject, vm: &VirtualMachine) -> PyResult {
659        if let Some(other) = other.downcast_ref::<Self>() {
660            let bytes = zelf.as_wtf8().py_add(other.as_wtf8());
661            Ok(unsafe {
662                // SAFETY: `kind` is safely decided
663                let kind = zelf.kind() | other.kind();
664                Self::new_str_unchecked(bytes.into(), kind)
665            }
666            .to_pyobject(vm))
667        } else {
668            Err(vm.new_type_error(format!(
669                r#"can only concatenate str (not "{}") to str"#,
670                other.class().slot_name()
671            )))
672        }
673    }
674
675    fn _contains(&self, needle: &PyObject, vm: &VirtualMachine) -> PyResult<bool> {
676        if let Some(needle) = needle.downcast_ref::<Self>() {
677            Ok(memchr::memmem::find(self.as_bytes(), needle.as_bytes()).is_some())
678        } else {
679            Err(vm.new_type_error(format!(
680                "'in <string>' requires string as left operand, not {}",
681                needle.class().slot_name()
682            )))
683        }
684    }
685
686    fn __contains__(&self, needle: &PyObject, vm: &VirtualMachine) -> PyResult<bool> {
687        self._contains(needle, vm)
688    }
689
690    fn _getitem(&self, needle: &PyObject, vm: &VirtualMachine) -> PyResult {
691        let item = match SequenceIndex::try_from_str_subscript(vm, needle)? {
692            SequenceIndex::Int(i) => self.getitem_by_index(vm, i)?.to_pyobject(vm),
693            SequenceIndex::Slice(slice) => self.getitem_by_slice(vm, slice)?.to_pyobject(vm),
694        };
695        Ok(item)
696    }
697
698    fn __getitem__(&self, needle: &PyObject, vm: &VirtualMachine) -> PyResult {
699        self._getitem(needle, vm)
700    }
701
702    #[inline]
703    pub(crate) fn hash(&self, vm: &VirtualMachine) -> hash::PyHash {
704        match self.hash.load(atomic::Ordering::Relaxed) {
705            hash::SENTINEL => self._compute_hash(vm),
706            hash => hash,
707        }
708    }
709
710    #[cold]
711    fn _compute_hash(&self, _vm: &VirtualMachine) -> hash::PyHash {
712        let hash_val = crate::vm::hash_secret().hash_bytes(self.as_bytes());
713        debug_assert_ne!(hash_val, hash::SENTINEL);
714        // spell-checker:ignore cmpxchg
715        // like with char_len, we don't need a cmpxchg loop, since it'll always be the same value
716        self.hash.store(hash_val, atomic::Ordering::Relaxed);
717        hash_val
718    }
719
720    #[inline]
721    pub fn byte_len(&self) -> usize {
722        self.data.len()
723    }
724
725    #[inline]
726    pub fn is_empty(&self) -> bool {
727        self.data.is_empty()
728    }
729
730    #[inline]
731    pub fn char_len(&self) -> usize {
732        self.data.char_len()
733    }
734
735    /// The byte offset the `index`-th character starts at, or the string's byte
736    /// length if `index` is at or past its end.
737    #[inline]
738    pub fn char_index_to_byte(&self, index: usize) -> usize {
739        self.data.char_index_to_byte(index)
740    }
741
742    /// The character index of the character starting at byte offset `bytepos`,
743    /// which must be a character boundary at or before the end.
744    #[inline]
745    pub fn byte_to_char_index(&self, bytepos: usize) -> usize {
746        self.data.byte_to_char_index(bytepos)
747    }
748
749    fn __mul__(zelf: PyRef<Self>, value: PySsize, vm: &VirtualMachine) -> PyResult<PyRef<Self>> {
750        Self::repeat(zelf, value, vm)
751    }
752
753    #[inline]
754    pub(crate) fn repr(&self, vm: &VirtualMachine) -> PyResult<String> {
755        use crate::literal::escape::UnicodeEscape;
756        UnicodeEscape::new_repr(self.as_wtf8())
757            .str_repr()
758            .to_string()
759            .ok_or_else(|| vm.new_overflow_error("string is too long to generate repr"))
760    }
761
762    /// Return `zelf` when it is an exact str; otherwise a new str copy.
763    fn result_unchanged(zelf: PyRef<Self>, vm: &VirtualMachine) -> PyRef<Self> {
764        if zelf.class().is(vm.ctx.types.str_type) {
765            zelf
766        } else {
767            vm.ctx.new_str(zelf.as_wtf8())
768        }
769    }
770
771    pub fn __mod__(&self, values: PyObjectRef, vm: &VirtualMachine) -> PyResult<Wtf8Buf> {
772        cformat_string(vm, self.as_wtf8(), &values)
773    }
774
775    /// `join` over already-materialized items: checks them and sizes the result before copying.
776    fn join_items(
777        zelf: &Py<Self>,
778        items: &[PyObjectRef],
779        vm: &VirtualMachine,
780    ) -> PyResult<PyStrRef> {
781        fn item_str<'a>(
782            i: usize,
783            obj: &'a PyObject,
784            vm: &VirtualMachine,
785        ) -> PyResult<&'a Py<PyStr>> {
786            obj.downcast_ref::<PyStr>().ok_or_else(|| {
787                vm.new_type_error(format!(
788                    "sequence item {i}: expected str instance, {} found",
789                    obj.class().slot_name()
790                ))
791            })
792        }
793        let sep = zelf.as_wtf8();
794        let mut len = sep.len().saturating_mul(items.len().saturating_sub(1));
795        for (i, obj) in items.iter().enumerate() {
796            len = len.saturating_add(item_str(i, obj, vm)?.as_wtf8().len());
797        }
798        if let [only] = items {
799            let only = item_str(0, only, vm)?;
800            if only.class().is(vm.ctx.types.str_type) {
801                return Ok(only.to_owned());
802            }
803        }
804        let mut joined = Wtf8Buf::with_capacity(len);
805        for (i, obj) in items.iter().enumerate() {
806            if i > 0 {
807                joined.push_wtf8(sep);
808            }
809            joined.push_wtf8(item_str(i, obj, vm)?.as_wtf8());
810        }
811        Ok(vm.ctx.new_str(joined))
812    }
813
814    /// The bytes the character range `range` spans and the byte offset it
815    /// starts at, or `None` if the range is inverted.
816    ///
817    /// The bounds go through the string's character index, so reaching a range
818    /// deep in the subject costs a lookup rather than a walk to it.
819    #[inline]
820    fn char_range_bytes(&self, range: Range<usize>) -> Option<(usize, &Wtf8)> {
821        if !range.is_normal() {
822            return None;
823        }
824        let bytes = self.data.char_range_to_bytes(range);
825        Some((bytes.start, &self.as_wtf8()[bytes]))
826    }
827
828    /// Searches the character range `range` with `find`, which answers in bytes
829    /// relative to the range, and reports the hit as a character index.
830    #[inline]
831    fn _find<F>(&self, args: FindArgs, find: F) -> Option<usize>
832    where
833        F: Fn(&Wtf8, &Wtf8) -> Option<usize>,
834    {
835        let (sub, range) = args.get_value(self.len());
836        let (start, haystack) = self.char_range_bytes(range)?;
837        let found = find(haystack, sub.as_wtf8())?;
838        Some(self.byte_to_char_index(start + found))
839    }
840
841    #[inline]
842    fn _pad(
843        &self,
844        width: isize,
845        fillchar: PyStrRef,
846        pad: fn(&Wtf8, usize, CodePoint, usize) -> Option<Wtf8Buf>,
847        vm: &VirtualMachine,
848    ) -> PyResult<Wtf8Buf> {
849        let fillchar = fillchar
850            .as_wtf8()
851            .code_points()
852            .exactly_one()
853            .map_err(|_| {
854                vm.new_type_error("The fill character must be exactly one character long")
855            })?;
856        if self.len() as isize >= width {
857            return Ok(self.as_wtf8().to_owned());
858        }
859        pad(self.as_wtf8(), width as usize, fillchar, self.len())
860            .ok_or_else(|| vm.no_memory_error())
861    }
862}
863
864#[pyclass(
865    flags(BASETYPE, _MATCH_SELF),
866    with(
867        AsMapping,
868        AsNumber,
869        AsSequence,
870        Representable,
871        Hashable,
872        Comparable,
873        Iterable,
874        Constructor
875    )
876)]
877impl Py<PyStr> {
878    #[pymethod]
879    #[inline(always)]
880    pub fn isascii(&self) -> bool {
881        matches!(self.kind(), StrKind::Ascii)
882    }
883
884    #[pymethod]
885    fn __sizeof__(&self) -> usize {
886        core::mem::size_of::<PyStr>() + self.byte_len() * core::mem::size_of::<u8>()
887    }
888
889    #[pymethod]
890    fn lower(&self) -> PyStr {
891        match self.as_str_kind() {
892            PyKindStr::Ascii(s) => s.to_ascii_lowercase().into(),
893            PyKindStr::Utf8(s) => s.to_lowercase().into(),
894            PyKindStr::Wtf8(w) => w.to_lowercase().into(),
895        }
896    }
897
898    // Case folding is a Unicode standard operation to erase case differences.
899    //
900    // Lower, upper, and title case are special properties. Case folding erases those
901    // differences. For ASCII, case folding is the same as lower case but other scripts have
902    // their own, well-defined mappings.
903    #[pymethod]
904    fn casefold(&self) -> PyStr {
905        match self.as_str_kind() {
906            PyKindStr::Ascii(s) => s.to_ascii_lowercase().into(),
907            PyKindStr::Utf8(s) => unicode::case::casefold_str(s).into(),
908            PyKindStr::Wtf8(w) => unicode::case::casefold_wtf8(w).into(),
909        }
910    }
911
912    #[pymethod]
913    fn upper(&self) -> PyStr {
914        match self.as_str_kind() {
915            PyKindStr::Ascii(s) => s.to_ascii_uppercase().into(),
916            PyKindStr::Utf8(s) => s.to_uppercase().into(),
917            PyKindStr::Wtf8(w) => w.to_uppercase().into(),
918        }
919    }
920
921    #[pymethod]
922    fn capitalize(&self) -> Wtf8Buf {
923        match self.as_str_kind() {
924            PyKindStr::Ascii(s) => {
925                let mut s = s.to_owned();
926                if let [first, rest @ ..] = s.as_mut_slice() {
927                    first.make_ascii_uppercase();
928                    ascii::AsciiStr::make_ascii_lowercase(rest.into());
929                }
930                s.into()
931            }
932            PyKindStr::Utf8(s) => case::capitalize_str(s).into(),
933            PyKindStr::Wtf8(s) => case::capitalize_wtf8(s),
934        }
935    }
936
937    #[pymethod]
938    fn split(zelf: &Self, args: SplitArgs, vm: &VirtualMachine) -> PyResult<Vec<PyObjectRef>> {
939        let elements = match zelf.as_str_kind() {
940            PyKindStr::Ascii(s) => s.py_split(
941                args,
942                vm,
943                || zelf.as_object().to_owned(),
944                |v, s, vm| {
945                    v.as_bytes()
946                        .split_str(s)
947                        .map(|s| unsafe { AsciiStr::from_ascii_unchecked(s) }.to_pyobject(vm))
948                        .collect()
949                },
950                |v, s, n, vm| {
951                    v.as_bytes()
952                        .splitn_str(n, s)
953                        .map(|s| unsafe { AsciiStr::from_ascii_unchecked(s) }.to_pyobject(vm))
954                        .collect()
955                },
956                |v, n, vm| {
957                    v.as_str().py_split_whitespace(n, |s| {
958                        unsafe { AsciiStr::from_ascii_unchecked(s.as_bytes()) }.to_pyobject(vm)
959                    })
960                },
961            ),
962            PyKindStr::Utf8(s) => s.py_split(
963                args,
964                vm,
965                || zelf.as_object().to_owned(),
966                |v, s, vm| v.split(s).map(|s| vm.ctx.new_str(s).into()).collect(),
967                |v, s, n, vm| v.splitn(n, s).map(|s| vm.ctx.new_str(s).into()).collect(),
968                |v, n, vm| v.py_split_whitespace(n, |s| vm.ctx.new_str(s).into()),
969            ),
970            PyKindStr::Wtf8(w) => w.py_split(
971                args,
972                vm,
973                || zelf.as_object().to_owned(),
974                |v, s, vm| v.split(s).map(|s| vm.ctx.new_str(s).into()).collect(),
975                |v, s, n, vm| v.splitn(n, s).map(|s| vm.ctx.new_str(s).into()).collect(),
976                |v, n, vm| v.py_split_whitespace(n, |s| vm.ctx.new_str(s).into()),
977            ),
978        }?;
979        Ok(elements)
980    }
981
982    #[pymethod]
983    fn rsplit(zelf: &Self, args: SplitArgs, vm: &VirtualMachine) -> PyResult<Vec<PyObjectRef>> {
984        let mut elements = zelf.as_wtf8().py_split(
985            args,
986            vm,
987            || zelf.as_object().to_owned(),
988            |v, s, vm| v.rsplit(s).map(|s| vm.ctx.new_str(s).into()).collect(),
989            |v, s, n, vm| v.rsplitn(n, s).map(|s| vm.ctx.new_str(s).into()).collect(),
990            |v, n, vm| v.py_rsplit_whitespace(n, |s| vm.ctx.new_str(s).into()),
991        )?;
992        // Unlike Python rsplit, Rust rsplitn returns an iterator that
993        // starts from the end of the string.
994        elements.reverse();
995        Ok(elements)
996    }
997
998    #[pymethod]
999    fn strip(zelf: PyRef<PyStr>, args: StripArgs, vm: &VirtualMachine) -> PyRef<PyStr> {
1000        let chars = args.chars;
1001        let stripped: &Wtf8 = match zelf.as_str_kind() {
1002            PyKindStr::Ascii(s) if chars.as_ref().is_none_or(|c| c.kind().is_ascii()) => s
1003                .py_strip(
1004                    chars,
1005                    |s, chars| {
1006                        let s = s
1007                            .as_str()
1008                            .trim_matches(|c| memchr::memchr(c as _, chars.as_bytes()).is_some());
1009                        unsafe { AsciiStr::from_ascii_unchecked(s.as_bytes()) }
1010                    },
1011                    |s| {
1012                        let s = s.as_str().trim_matches(unicode::classify::is_space);
1013                        unsafe { AsciiStr::from_ascii_unchecked(s.as_bytes()) }
1014                    },
1015                )
1016                .as_str()
1017                .into(),
1018            PyKindStr::Utf8(s) if chars.as_ref().is_none_or(|c| c.kind().is_utf8()) => s
1019                .py_strip(
1020                    chars,
1021                    |s, chars| s.trim_matches(|c| chars.contains(c)),
1022                    |s| s.trim_matches(unicode::classify::is_space),
1023                )
1024                .into(),
1025            _ => zelf.as_wtf8().py_strip(
1026                chars,
1027                |s, chars| s.trim_matches(|c| chars.code_points().contains(&c)),
1028                |s| s.trim_matches(|c: CodePoint| c.is_char_and(unicode::classify::is_space)),
1029            ),
1030        };
1031        if zelf.byte_len() == stripped.len() {
1032            PyStr::result_unchanged(zelf, vm)
1033        } else {
1034            vm.ctx.new_str(zelf.new_substr(stripped.to_owned()))
1035        }
1036    }
1037
1038    #[pymethod]
1039    fn lstrip(zelf: PyRef<PyStr>, args: StripArgs, vm: &VirtualMachine) -> PyRef<PyStr> {
1040        let chars = args.chars;
1041        let s = zelf.as_wtf8();
1042        let stripped = s.py_strip(
1043            chars,
1044            |s, chars| s.trim_start_matches(|c| chars.contains_code_point(c)),
1045            |s| s.trim_start_matches(|c: CodePoint| c.is_char_and(unicode::classify::is_space)),
1046        );
1047        if s.len() == stripped.len() {
1048            PyStr::result_unchanged(zelf, vm)
1049        } else {
1050            vm.ctx.new_str(stripped)
1051        }
1052    }
1053
1054    #[pymethod]
1055    fn rstrip(zelf: PyRef<PyStr>, args: StripArgs, vm: &VirtualMachine) -> PyRef<PyStr> {
1056        let chars = args.chars;
1057        let s = zelf.as_wtf8();
1058        let stripped = s.py_strip(
1059            chars,
1060            |s, chars| s.trim_end_matches(|c| chars.contains_code_point(c)),
1061            |s| s.trim_end_matches(|c: CodePoint| c.is_char_and(unicode::classify::is_space)),
1062        );
1063        if s.len() == stripped.len() {
1064            PyStr::result_unchanged(zelf, vm)
1065        } else {
1066            vm.ctx.new_str(stripped)
1067        }
1068    }
1069
1070    #[pymethod]
1071    fn endswith(&self, options: anystr::StartsEndsWithArgs, vm: &VirtualMachine) -> PyResult<bool> {
1072        let (affix, substr) = match options.prepare(self.as_wtf8(), self.len(), |s, r| {
1073            &s[self.data.char_range_to_bytes(r)]
1074        }) {
1075            Some(x) => x,
1076            None => return Ok(false),
1077        };
1078        substr.py_starts_ends_with(
1079            &affix,
1080            "endswith",
1081            "str",
1082            |s, x: &Self| s.ends_with(x.as_wtf8()),
1083            vm,
1084        )
1085    }
1086
1087    #[pymethod]
1088    fn startswith(
1089        &self,
1090        options: anystr::StartsEndsWithArgs,
1091        vm: &VirtualMachine,
1092    ) -> PyResult<bool> {
1093        let (affix, substr) = match options.prepare(self.as_wtf8(), self.len(), |s, r| {
1094            &s[self.data.char_range_to_bytes(r)]
1095        }) {
1096            Some(x) => x,
1097            None => return Ok(false),
1098        };
1099        substr.py_starts_ends_with(
1100            &affix,
1101            "startswith",
1102            "str",
1103            |s, x: &Self| s.starts_with(x.as_wtf8()),
1104            vm,
1105        )
1106    }
1107
1108    #[pymethod]
1109    fn removeprefix(&self, prefix: PyStrRef) -> Wtf8Buf {
1110        self.as_wtf8()
1111            .py_removeprefix(prefix.as_wtf8(), prefix.byte_len(), |s, p| s.starts_with(p))
1112            .to_owned()
1113    }
1114
1115    #[pymethod]
1116    fn removesuffix(&self, suffix: PyStrRef) -> Wtf8Buf {
1117        self.as_wtf8()
1118            .py_removesuffix(suffix.as_wtf8(), suffix.byte_len(), |s, p| s.ends_with(p))
1119            .to_owned()
1120    }
1121
1122    #[pymethod]
1123    fn isalnum(&self) -> bool {
1124        !self.data.is_empty() && self.char_all(unicode::classify::is_alnum)
1125    }
1126
1127    #[pymethod]
1128    fn isnumeric(&self) -> bool {
1129        !self.data.is_empty() && self.char_all(unicode::classify::is_numeric)
1130    }
1131
1132    #[pymethod]
1133    fn isdigit(&self) -> bool {
1134        !self.data.is_empty() && self.char_all(unicode::classify::is_digit)
1135    }
1136
1137    #[pymethod]
1138    fn isdecimal(&self) -> bool {
1139        !self.data.is_empty() && self.char_all(unicode::classify::is_decimal)
1140    }
1141
1142    #[pymethod]
1143    fn format(&self, args: FuncArgs, vm: &VirtualMachine) -> PyResult<Wtf8Buf> {
1144        let format_str =
1145            FormatString::from_str(self.as_wtf8()).map_err(|e| e.to_pyexception(vm))?;
1146        format(&format_str, &args, vm)
1147    }
1148
1149    #[pymethod]
1150    fn format_map(&self, mapping: PyObjectRef, vm: &VirtualMachine) -> PyResult<Wtf8Buf> {
1151        let format_string =
1152            FormatString::from_str(self.as_wtf8()).map_err(|err| err.to_pyexception(vm))?;
1153        format_map(&format_string, &mapping, vm)
1154    }
1155
1156    #[pymethod]
1157    fn __format__(
1158        zelf: PyRef<PyStr>,
1159        format_spec: PyUtf8StrRef,
1160        vm: &VirtualMachine,
1161    ) -> PyResult<PyRef<PyStr>> {
1162        if format_spec.is_empty() {
1163            return if zelf.class().is(vm.ctx.types.str_type) {
1164                Ok(zelf)
1165            } else {
1166                zelf.as_object().str(vm)
1167            };
1168        }
1169        let zelf = zelf.try_into_utf8(vm)?;
1170        let s = FormatSpec::parse(format_spec.as_str())
1171            .and_then(|format_spec| {
1172                format_spec.format_string(&CharLenStr(zelf.as_str(), zelf.char_len()))
1173            })
1174            .map_err(|err| err.into_pyexception(vm))?;
1175        Ok(vm.ctx.new_str(s))
1176    }
1177
1178    #[pymethod]
1179    fn title(&self) -> Wtf8Buf {
1180        match self.as_str_kind() {
1181            PyKindStr::Ascii(_) => unsafe {
1182                Wtf8Buf::from_bytes_unchecked(title_ascii(self.as_bytes()))
1183            },
1184            PyKindStr::Utf8(s) => case::title_str(s).into(),
1185            PyKindStr::Wtf8(s) => case::title_wtf8(s),
1186        }
1187    }
1188
1189    #[pymethod]
1190    fn swapcase(&self) -> Wtf8Buf {
1191        match self.as_str_kind() {
1192            PyKindStr::Ascii(s) => unsafe {
1193                // SAFETY: ASCII is valid Unicode and swapcase_ascii does not produce non-ASCII.
1194                Wtf8Buf::from_bytes_unchecked(swapcase_ascii(s.as_bytes()))
1195            },
1196            PyKindStr::Utf8(s) => case::swapcase_str(s).into(),
1197            PyKindStr::Wtf8(s) => case::swapcase_wtf8(s),
1198        }
1199    }
1200
1201    #[pymethod]
1202    fn isalpha(&self) -> bool {
1203        !self.data.is_empty() && self.char_all(unicode::classify::is_alpha)
1204    }
1205
1206    #[pymethod]
1207    fn replace(zelf: PyRef<PyStr>, args: ReplaceArgs, vm: &VirtualMachine) -> PyRef<PyStr> {
1208        let ReplaceArgs { old, new, count } = args;
1209        if count == 0 || old.byte_len() > zelf.byte_len() || old.as_wtf8() == new.as_wtf8() {
1210            return PyStr::result_unchanged(zelf, vm);
1211        }
1212
1213        let s = zelf.as_wtf8();
1214        let replaced = if count < 0 {
1215            s.replace(old.as_wtf8(), new.as_wtf8())
1216        } else {
1217            let s_is_empty = s.is_empty();
1218            let old_is_empty = old.is_empty();
1219
1220            if s_is_empty && !old_is_empty {
1221                s.to_owned()
1222            } else if s_is_empty && old_is_empty {
1223                new.as_wtf8().to_owned()
1224            } else {
1225                s.replacen(old.as_wtf8(), new.as_wtf8(), count as usize)
1226            }
1227        };
1228        vm.ctx.new_str(replaced)
1229    }
1230
1231    #[pymethod]
1232    fn isprintable(&self) -> bool {
1233        self.char_all(unicode::classify::is_printable)
1234    }
1235
1236    #[pymethod]
1237    fn isspace(&self) -> bool {
1238        !self.data.is_empty() && self.char_all(unicode::classify::is_space)
1239    }
1240
1241    // Return true if all cased characters in the string are lowercase and there is at least one cased character, false otherwise.
1242    #[pymethod]
1243    fn islower(&self) -> bool {
1244        match self.as_str_kind() {
1245            PyKindStr::Ascii(s) => s.py_islower(),
1246            PyKindStr::Utf8(s) => s.py_islower(),
1247            PyKindStr::Wtf8(w) => w.py_islower(),
1248        }
1249    }
1250
1251    // Return true if all cased characters in the string are uppercase and there is at least one cased character, false otherwise.
1252    #[pymethod]
1253    fn isupper(&self) -> bool {
1254        match self.as_str_kind() {
1255            PyKindStr::Ascii(s) => s.py_isupper(),
1256            PyKindStr::Utf8(s) => s.py_isupper(),
1257            PyKindStr::Wtf8(w) => w.py_isupper(),
1258        }
1259    }
1260
1261    #[pymethod]
1262    pub(crate) fn splitlines(
1263        &self,
1264        args: anystr::SplitLinesArgs,
1265        vm: &VirtualMachine,
1266    ) -> Vec<PyObjectRef> {
1267        let into_wrapper = |s: &Wtf8| self.new_substr(s.to_owned()).to_pyobject(vm);
1268        let mut elements = Vec::new();
1269        let mut last_i = 0;
1270        let self_str = self.as_wtf8();
1271        let mut enumerated = self_str.code_point_indices().peekable();
1272        while let Some((i, ch)) = enumerated.next() {
1273            let end_len = match ch.to_char_lossy() {
1274                '\n' => 1,
1275                '\r' => {
1276                    let is_rn = enumerated.next_if(|(_, ch)| *ch == '\n').is_some();
1277                    if is_rn { 2 } else { 1 }
1278                }
1279                '\x0b' | '\x0c' | '\x1c' | '\x1d' | '\x1e' | '\u{0085}' | '\u{2028}'
1280                | '\u{2029}' => ch.len_wtf8(),
1281                _ => continue,
1282            };
1283            let range = if args.keepends {
1284                last_i..i + end_len
1285            } else {
1286                last_i..i
1287            };
1288            last_i = i + end_len;
1289            elements.push(into_wrapper(&self_str[range]));
1290        }
1291        if last_i != self_str.len() {
1292            elements.push(into_wrapper(&self_str[last_i..]));
1293        }
1294        elements
1295    }
1296
1297    #[pymethod]
1298    fn join(zelf: PyRef<PyStr>, iterable: PyObjectRef, vm: &VirtualMachine) -> PyResult<PyStrRef> {
1299        // `PySequence_Fast()` hands a list or tuple over as-is.
1300        if let Some(list) = iterable.downcast_ref_if_exact::<PyList>(vm) {
1301            return PyStr::join_items(&zelf, &list.borrow_vec(), vm);
1302        }
1303        if let Some(tuple) = iterable.downcast_ref_if_exact::<PyTuple>(vm) {
1304            return PyStr::join_items(&zelf, tuple.as_slice(), vm);
1305        }
1306        // `PyUnicode_Join()` reaches its elements through `PySequence_Fast()`,
1307        // which fills a list from the iterator and so asks it how long it is,
1308        // and which has its own wording for what it cannot iterate.
1309        let iterable = ArgIterable::<PyObjectRef>::try_from_object(vm, iterable)
1310            .map_err(|_| vm.new_type_error("can only join an iterable"))?;
1311        let items = iterable.iter_sized(vm)?.collect::<PyResult<Vec<_>>>()?;
1312        PyStr::join_items(&zelf, &items, vm)
1313    }
1314
1315    #[pymethod]
1316    fn find(&self, args: FindArgs) -> isize {
1317        self._find(args, Wtf8::find).map_or(-1, |v| v as isize)
1318    }
1319
1320    #[pymethod]
1321    fn rfind(&self, args: FindArgs) -> isize {
1322        self._find(args, Wtf8::rfind).map_or(-1, |v| v as isize)
1323    }
1324
1325    #[pymethod]
1326    fn index(&self, args: FindArgs, vm: &VirtualMachine) -> PyResult<usize> {
1327        self._find(args, Wtf8::find)
1328            .ok_or_else(|| vm.new_value_error("substring not found"))
1329    }
1330
1331    #[pymethod]
1332    fn rindex(&self, args: FindArgs, vm: &VirtualMachine) -> PyResult<usize> {
1333        self._find(args, Wtf8::rfind)
1334            .ok_or_else(|| vm.new_value_error("substring not found"))
1335    }
1336
1337    #[pymethod]
1338    pub fn partition(&self, sep: PyStrRef, vm: &VirtualMachine) -> PyResult {
1339        let (front, has_mid, back) = self.as_wtf8().py_partition(
1340            sep.as_wtf8(),
1341            || self.as_wtf8().splitn(2, sep.as_wtf8()),
1342            vm,
1343        )?;
1344        let partition = (
1345            self.new_substr(front),
1346            if has_mid {
1347                sep
1348            } else {
1349                vm.ctx.new_str(ascii!(""))
1350            },
1351            self.new_substr(back),
1352        );
1353        Ok(partition.to_pyobject(vm))
1354    }
1355
1356    #[pymethod]
1357    pub fn rpartition(&self, sep: PyStrRef, vm: &VirtualMachine) -> PyResult {
1358        let (back, has_mid, front) = self.as_wtf8().py_partition(
1359            sep.as_wtf8(),
1360            || self.as_wtf8().rsplitn(2, sep.as_wtf8()),
1361            vm,
1362        )?;
1363        Ok((
1364            self.new_substr(front),
1365            if has_mid {
1366                sep
1367            } else {
1368                vm.ctx.empty_str.to_owned()
1369            },
1370            self.new_substr(back),
1371        )
1372            .to_pyobject(vm))
1373    }
1374
1375    #[pymethod]
1376    fn istitle(&self) -> bool {
1377        if self.data.is_empty() {
1378            return false;
1379        }
1380
1381        let mut cased = false;
1382        let mut previous_is_cased = false;
1383        for c in self.as_wtf8().code_points().map(CodePoint::to_char_lossy) {
1384            if c.is_uppercase() || case::is_titlecase(c) {
1385                if previous_is_cased {
1386                    return false;
1387                }
1388                previous_is_cased = true;
1389                cased = true;
1390            } else if c.is_lowercase() {
1391                if !previous_is_cased {
1392                    return false;
1393                }
1394                previous_is_cased = true;
1395                cased = true;
1396            } else {
1397                previous_is_cased = false;
1398            }
1399        }
1400        cased
1401    }
1402
1403    #[pymethod]
1404    fn count(&self, args: FindArgs) -> usize {
1405        let (needle, range) = args.get_value(self.len());
1406        let chars = range.len();
1407        self.char_range_bytes(range).map_or(0, |(_, haystack)| {
1408            if needle.is_empty() {
1409                // An empty needle sits between every pair of characters and at
1410                // both ends, so it occurs once more than the range holds
1411                // characters. Counting it in the bytes would answer in encoded
1412                // positions instead.
1413                chars + 1
1414            } else {
1415                haystack.find_iter(needle.as_wtf8()).count()
1416            }
1417        })
1418    }
1419
1420    #[pymethod]
1421    fn zfill(&self, width: PySsize, vm: &VirtualMachine) -> PyResult<Wtf8Buf> {
1422        let filled = self
1423            .as_wtf8()
1424            .py_zfill(width)
1425            .ok_or_else(|| vm.no_memory_error())?;
1426        // SAFETY: this is safe-guaranteed because the original self.as_wtf8() is valid wtf8
1427        Ok(unsafe { Wtf8Buf::from_bytes_unchecked(filled) })
1428    }
1429
1430    #[pymethod]
1431    fn center(&self, args: PadArgs, vm: &VirtualMachine) -> PyResult<Wtf8Buf> {
1432        self._pad(args.width, args.fillchar, AnyStr::py_center, vm)
1433    }
1434
1435    #[pymethod]
1436    fn ljust(&self, args: PadArgs, vm: &VirtualMachine) -> PyResult<Wtf8Buf> {
1437        self._pad(args.width, args.fillchar, AnyStr::py_ljust, vm)
1438    }
1439
1440    #[pymethod]
1441    fn rjust(&self, args: PadArgs, vm: &VirtualMachine) -> PyResult<Wtf8Buf> {
1442        self._pad(args.width, args.fillchar, AnyStr::py_rjust, vm)
1443    }
1444
1445    #[pymethod]
1446    fn expandtabs(&self, args: anystr::ExpandTabsArgs) -> Wtf8Buf {
1447        rustpython_common::str::expandtabs(self.as_wtf8(), args.tabsize())
1448    }
1449
1450    #[pymethod]
1451    pub fn isidentifier(&self) -> bool {
1452        let Some(s) = self.to_str() else { return false };
1453        let mut chars = s.chars();
1454
1455        let is_identifier_start = chars.next().is_some_and(unicode::identifier::is_start);
1456
1457        // a string is not an identifier if it has whitespace or starts with a number
1458        is_identifier_start && chars.all(unicode::identifier::is_continue)
1459    }
1460
1461    // https://docs.python.org/3/library/stdtypes.html#str.translate
1462    #[pymethod]
1463    pub fn translate(&self, table: PyObjectRef, vm: &VirtualMachine) -> PyResult<Wtf8Buf> {
1464        let dict = table.downcast_ref_if_exact::<PyDict>(vm);
1465        let mut translated = Wtf8Buf::with_capacity(self.as_wtf8().len());
1466        for cp in self.as_wtf8().code_points() {
1467            let key = cp.to_u32().to_pyobject(vm);
1468            // `charmaptranslate_lookup`: a missing key or any `LookupError` leaves `cp` unchanged.
1469            let value = match dict {
1470                Some(dict) => dict.get_item_opt(&*key, vm)?,
1471                None => match table.get_item(&*key, vm) {
1472                    Ok(value) => Some(value),
1473                    Err(e) if e.fast_isinstance(vm.ctx.exceptions.lookup_error) => None,
1474                    Err(e) => return Err(e),
1475                },
1476            };
1477            let Some(value) = value else {
1478                translated.push(cp);
1479                continue;
1480            };
1481            if let Some(text) = value.downcast_ref::<PyStr>() {
1482                translated.push_wtf8(text.as_wtf8());
1483            } else if let Some(bigint) = value.downcast_ref::<PyInt>() {
1484                let mapped = bigint
1485                    .as_bigint()
1486                    .to_u32()
1487                    .and_then(CodePoint::from_u32)
1488                    .ok_or_else(|| {
1489                        vm.new_value_error("character mapping must be in range(0x110000)")
1490                    })?;
1491                translated.push(mapped);
1492            } else if !vm.is_none(&value) {
1493                return Err(vm.new_type_error("character mapping must return integer, None or str"));
1494            }
1495        }
1496        Ok(translated)
1497    }
1498
1499    #[pystaticmethod]
1500    fn maketrans(
1501        dict_or_str: PyObjectRef,
1502        to_str: OptionalArg<PyStrRef>,
1503        none_str: OptionalArg<PyStrRef>,
1504        vm: &VirtualMachine,
1505    ) -> PyResult {
1506        let new_dict = vm.ctx.new_dict();
1507        if let OptionalArg::Present(to_str) = to_str {
1508            match dict_or_str.downcast::<PyStr>() {
1509                Ok(from_str) => {
1510                    if to_str.len() == from_str.len() {
1511                        for (c1, c2) in from_str
1512                            .as_wtf8()
1513                            .code_points()
1514                            .zip(to_str.as_wtf8().code_points())
1515                        {
1516                            new_dict.set_item(
1517                                &*vm.new_pyobj(c1.to_u32()),
1518                                vm.new_pyobj(c2.to_u32()),
1519                                vm,
1520                            )?;
1521                        }
1522                        if let OptionalArg::Present(none_str) = none_str {
1523                            for c in none_str.as_wtf8().code_points() {
1524                                new_dict.set_item(&*vm.new_pyobj(c.to_u32()), vm.ctx.none(), vm)?;
1525                            }
1526                        }
1527                        Ok(new_dict.to_pyobject(vm))
1528                    } else {
1529                        Err(vm.new_value_error(
1530                            "the first two maketrans arguments must have equal length",
1531                        ))
1532                    }
1533                }
1534                _ => Err(vm.new_type_error(
1535                    "first maketrans argument must be a string if there is a second argument",
1536                )),
1537            }
1538        } else {
1539            // dict_str must be a dict
1540            match dict_or_str.downcast::<PyDict>() {
1541                Ok(dict) => {
1542                    for (key, val) in dict {
1543                        // FIXME: ints are key-compatible
1544                        if let Some(num) = key.downcast_ref::<PyInt>() {
1545                            new_dict.set_item(
1546                                &*num.as_bigint().to_i32().to_pyobject(vm),
1547                                val,
1548                                vm,
1549                            )?;
1550                        } else if let Some(string) = key.downcast_ref::<PyStr>() {
1551                            if string.len() == 1 {
1552                                let num_value =
1553                                    string.as_wtf8().code_points().next().unwrap().to_u32();
1554                                new_dict.set_item(&*num_value.to_pyobject(vm), val, vm)?;
1555                            } else {
1556                                return Err(vm.new_value_error(
1557                                    "string keys in translate table must be of length 1",
1558                                ));
1559                            }
1560                        } else {
1561                            return Err(vm.new_type_error(
1562                                "keys in translate table must be strings or integers",
1563                            ));
1564                        }
1565                    }
1566                    Ok(new_dict.to_pyobject(vm))
1567                }
1568                _ => Err(vm.new_value_error(
1569                    "if you give only one argument to maketrans it must be a dict",
1570                )),
1571            }
1572        }
1573    }
1574
1575    #[pymethod]
1576    fn encode(zelf: PyRef<PyStr>, args: EncodeArgs, vm: &VirtualMachine) -> PyResult<PyBytesRef> {
1577        encode_string(zelf, args.encoding.as_deref(), args.errors, vm)
1578    }
1579
1580    #[pymethod]
1581    fn __getnewargs__(zelf: PyRef<PyStr>, vm: &VirtualMachine) -> PyObjectRef {
1582        (zelf.as_wtf8(),).to_pyobject(vm)
1583    }
1584
1585    #[pymethod]
1586    fn __str__(zelf: &Self, vm: &VirtualMachine) -> PyStrRef {
1587        if zelf.class().is(vm.ctx.types.str_type) {
1588            // Already exact str, just return a reference
1589            zelf.to_owned()
1590        } else {
1591            // Subclass, create a new exact str
1592            PyStr::from(zelf.data.clone()).into_ref(&vm.ctx)
1593        }
1594    }
1595}
1596
1597impl PyRef<PyStr> {
1598    #[must_use]
1599    pub fn is_empty(&self) -> bool {
1600        (**self).is_empty()
1601    }
1602
1603    pub fn concat_in_place(&mut self, other: &Wtf8, vm: &VirtualMachine) {
1604        if other.is_empty() {
1605            return;
1606        }
1607        let mut s = Wtf8Buf::with_capacity(self.byte_len() + other.len());
1608        s.push_wtf8(self.as_ref());
1609        s.push_wtf8(other);
1610        if self.as_object().strong_count() == 1 {
1611            // SAFETY: strong_count()==1 guarantees unique ownership of this PyStr.
1612            // Mutating payload in place preserves semantics while avoiding PyObject reallocation.
1613            unsafe {
1614                let payload = self.payload() as *const PyStr as *mut PyStr;
1615                (*payload).data = PyStr::from(s).data;
1616                (*payload)
1617                    .hash
1618                    .store(hash::SENTINEL, atomic::Ordering::Relaxed);
1619            }
1620        } else {
1621            *self = PyStr::from(s).into_ref(&vm.ctx);
1622        }
1623    }
1624
1625    pub fn try_into_utf8(self, vm: &VirtualMachine) -> PyResult<PyRef<PyUtf8Str>> {
1626        self.ensure_valid_utf8(vm)?;
1627        Ok(unsafe { mem::transmute::<Self, PyRef<PyUtf8Str>>(self) })
1628    }
1629}
1630
1631struct CharLenStr<'a>(&'a str, usize);
1632impl core::ops::Deref for CharLenStr<'_> {
1633    type Target = str;
1634
1635    fn deref(&self) -> &Self::Target {
1636        self.0
1637    }
1638}
1639impl crate::common::format::CharLen for CharLenStr<'_> {
1640    fn char_len(&self) -> usize {
1641        self.1
1642    }
1643}
1644
1645impl Representable for PyStr {
1646    #[inline]
1647    fn repr_str(zelf: &Py<Self>, vm: &VirtualMachine) -> PyResult<String> {
1648        zelf.repr(vm)
1649    }
1650}
1651
1652impl Hashable for PyStr {
1653    #[inline]
1654    fn hash(zelf: &Py<Self>, vm: &VirtualMachine) -> PyResult<hash::PyHash> {
1655        Ok(zelf.hash(vm))
1656    }
1657}
1658
1659impl Comparable for PyStr {
1660    fn cmp(
1661        zelf: &Py<Self>,
1662        other: &PyObject,
1663        op: PyComparisonOp,
1664        _vm: &VirtualMachine,
1665    ) -> PyResult<PyComparisonValue> {
1666        if let Some(res) = op.identical_optimization(zelf, other) {
1667            return Ok(res.into());
1668        }
1669        let other = class_or_notimplemented!(Self, other);
1670        // Equality does not need the ordering, and answers two strings of
1671        // different length without reading either.
1672        if let Some(res) = op.eval_eq(|| zelf.as_wtf8() == other.as_wtf8()) {
1673            return Ok(res.into());
1674        }
1675        Ok(op.eval_ord(zelf.as_wtf8().cmp(other.as_wtf8())).into())
1676    }
1677}
1678
1679impl Iterable for PyStr {
1680    fn iter(zelf: PyRef<Self>, vm: &VirtualMachine) -> PyResult {
1681        Ok(PyStrIterator {
1682            internal: PyMutex::new((PositionIterInternal::new(zelf, 0), 0)),
1683        }
1684        .into_pyobject(vm))
1685    }
1686}
1687
1688impl AsMapping for PyStr {
1689    fn as_mapping() -> &'static PyMappingMethods {
1690        static AS_MAPPING: LazyLock<PyMappingMethods> = LazyLock::new(|| PyMappingMethods {
1691            length: atomic_func!(|mapping, _vm| Ok(PyStr::mapping_downcast(mapping).len())),
1692            subscript: atomic_func!(
1693                |mapping, needle, vm| PyStr::mapping_downcast(mapping)._getitem(needle, vm)
1694            ),
1695            ..PyMappingMethods::NOT_IMPLEMENTED
1696        });
1697        &AS_MAPPING
1698    }
1699}
1700
1701impl AsNumber for PyStr {
1702    fn as_number() -> &'static PyNumberMethods {
1703        static AS_NUMBER: PyNumberMethods = PyNumberMethods {
1704            remainder: Some(|a, b, vm| {
1705                if let Some(a) = a.downcast_ref::<PyStr>() {
1706                    a.__mod__(b.to_owned(), vm).to_pyresult(vm)
1707                } else {
1708                    Ok(vm.ctx.not_implemented())
1709                }
1710            }),
1711            ..PyNumberMethods::NOT_IMPLEMENTED
1712        };
1713        &AS_NUMBER
1714    }
1715}
1716
1717impl AsSequence for PyStr {
1718    fn as_sequence() -> &'static PySequenceMethods {
1719        static AS_SEQUENCE: LazyLock<PySequenceMethods> = LazyLock::new(|| PySequenceMethods {
1720            length: atomic_func!(|seq, _vm| Ok(PyStr::sequence_downcast(seq).len())),
1721            concat: atomic_func!(|seq, other, vm| {
1722                let zelf = PyStr::sequence_downcast(seq);
1723                PyStr::__add__(zelf.to_owned(), other, vm)
1724            }),
1725            repeat: atomic_func!(|seq, n, vm| {
1726                let zelf = PyStr::sequence_downcast(seq);
1727                PyStr::repeat(zelf.to_owned(), n, vm).map(|x| x.into())
1728            }),
1729            item: atomic_func!(|seq, i, vm| {
1730                let zelf = PyStr::sequence_downcast(seq);
1731                zelf.getitem_by_index(vm, i).to_pyresult(vm)
1732            }),
1733            contains: atomic_func!(
1734                |seq, needle, vm| PyStr::sequence_downcast(seq)._contains(needle, vm)
1735            ),
1736            ..PySequenceMethods::NOT_IMPLEMENTED
1737        });
1738        &AS_SEQUENCE
1739    }
1740}
1741
1742#[derive(FromArgs)]
1743struct EncodeArgs {
1744    // None is filled in as utf-8 when encoding.
1745    #[pyarg(any, optional, py_default = "'utf-8'")]
1746    encoding: Option<PyUtf8StrRef>,
1747    // None is filled in as strict when encoding.
1748    #[pyarg(any, optional, py_default = "'strict'")]
1749    errors: Option<PyUtf8StrRef>,
1750}
1751
1752#[derive(FromArgs)]
1753struct StripArgs {
1754    #[pyarg(positional, optional)]
1755    chars: Option<PyStrRef>,
1756}
1757
1758#[derive(FromArgs)]
1759struct PadArgs {
1760    #[pyarg(positional)]
1761    width: PySsize,
1762    #[pyarg(positional, default = " ")]
1763    fillchar: PyStrRef,
1764}
1765
1766pub(crate) fn encode_string(
1767    s: PyStrRef,
1768    encoding: Option<&Py<PyUtf8Str>>,
1769    errors: Option<PyUtf8StrRef>,
1770    vm: &VirtualMachine,
1771) -> PyResult<PyBytesRef> {
1772    let encoding = match encoding {
1773        None => crate::codecs::DEFAULT_ENCODING,
1774        Some(s) => s.as_str(),
1775    };
1776    vm.state.codec_registry.encode_text(s, encoding, errors, vm)
1777}
1778
1779impl PyPayload for PyStr {
1780    #[inline]
1781    fn class(ctx: &Context) -> &'static Py<PyType> {
1782        ctx.types.str_type
1783    }
1784}
1785
1786impl ToPyObject for String {
1787    fn to_pyobject(self, vm: &VirtualMachine) -> PyObjectRef {
1788        vm.ctx.new_str(self).into()
1789    }
1790}
1791
1792impl ToPyObject for Wtf8Buf {
1793    fn to_pyobject(self, vm: &VirtualMachine) -> PyObjectRef {
1794        vm.ctx.new_str(self).into()
1795    }
1796}
1797
1798impl ToPyObject for char {
1799    fn to_pyobject(self, vm: &VirtualMachine) -> PyObjectRef {
1800        let cp = self as u32;
1801        u8::try_from(cp).map_or_else(
1802            |_| vm.ctx.new_str(self).into(),
1803            |v| vm.ctx.latin1_char(v).into(),
1804        )
1805    }
1806}
1807
1808impl ToPyObject for CodePoint {
1809    fn to_pyobject(self, vm: &VirtualMachine) -> PyObjectRef {
1810        let cp = self.to_u32();
1811        u8::try_from(cp).map_or_else(
1812            |_| vm.ctx.new_str(self).into(),
1813            |v| vm.ctx.latin1_char(v).into(),
1814        )
1815    }
1816}
1817
1818impl ToPyObject for &str {
1819    fn to_pyobject(self, vm: &VirtualMachine) -> PyObjectRef {
1820        vm.ctx.new_str(self).into()
1821    }
1822}
1823
1824impl ToPyObject for &String {
1825    fn to_pyobject(self, vm: &VirtualMachine) -> PyObjectRef {
1826        vm.ctx.new_str(self.clone()).into()
1827    }
1828}
1829
1830impl ToPyObject for &CStr {
1831    fn to_pyobject(self, vm: &VirtualMachine) -> PyObjectRef {
1832        let s = self.to_str().expect("ToPyObject expects utf-8 CStr");
1833        vm.ctx.new_str(s).into()
1834    }
1835}
1836
1837impl ToPyObject for &Wtf8 {
1838    fn to_pyobject(self, vm: &VirtualMachine) -> PyObjectRef {
1839        vm.ctx.new_str(self).into()
1840    }
1841}
1842
1843impl ToPyObject for &Wtf8Buf {
1844    fn to_pyobject(self, vm: &VirtualMachine) -> PyObjectRef {
1845        vm.ctx.new_str(self.clone()).into()
1846    }
1847}
1848
1849impl ToPyObject for &AsciiStr {
1850    fn to_pyobject(self, vm: &VirtualMachine) -> PyObjectRef {
1851        vm.ctx.new_str(self).into()
1852    }
1853}
1854
1855impl ToPyObject for AsciiString {
1856    fn to_pyobject(self, vm: &VirtualMachine) -> PyObjectRef {
1857        vm.ctx.new_str(self).into()
1858    }
1859}
1860
1861impl ToPyObject for AsciiChar {
1862    fn to_pyobject(self, vm: &VirtualMachine) -> PyObjectRef {
1863        vm.ctx.latin1_char(u8::from(self)).into()
1864    }
1865}
1866
1867type SplitArgs = anystr::SplitArgs<PyStrRef>;
1868
1869#[derive(FromArgs)]
1870pub(crate) struct FindArgs {
1871    #[pyarg(positional)]
1872    sub: PyStrRef,
1873    #[pyarg(positional, default)]
1874    start: Option<PyIntRef>,
1875    #[pyarg(positional, default)]
1876    end: Option<PyIntRef>,
1877}
1878
1879impl FindArgs {
1880    fn get_value(self, len: usize) -> (PyStrRef, core::ops::Range<usize>) {
1881        let range = adjust_indices(self.start.as_deref(), self.end.as_deref(), len);
1882        (self.sub, range)
1883    }
1884}
1885
1886#[derive(FromArgs)]
1887struct ReplaceArgs {
1888    #[pyarg(positional)]
1889    old: PyStrRef,
1890
1891    #[pyarg(positional)]
1892    new: PyStrRef,
1893
1894    #[pyarg(any, default = -1)]
1895    count: isize,
1896}
1897
1898fn vectorcall_str(
1899    zelf_obj: &PyObject,
1900    args: Vec<PyObjectRef>,
1901    nargs: usize,
1902    kwnames: Option<&[PyObjectRef]>,
1903    vm: &VirtualMachine,
1904) -> PyResult {
1905    let zelf: &Py<PyType> = zelf_obj.downcast_ref().unwrap();
1906    let func_args = FuncArgs::from_vectorcall_owned(args, nargs, kwnames);
1907    (zelf.slots.new.load().unwrap())(zelf.to_owned(), func_args, vm)
1908}
1909
1910pub(crate) fn init(ctx: &'static Context) {
1911    PyStr::extend_class(ctx, ctx.types.str_type);
1912    ctx.types
1913        .str_type
1914        .slots
1915        .vectorcall
1916        .store(Some(vectorcall_str));
1917
1918    PyStrIterator::extend_class(ctx, ctx.types.str_iterator_type);
1919}
1920
1921impl PyStr {
1922    /// The code points at `indices`, in that order, as a new string.
1923    ///
1924    /// Each index is resolved through the string's own index table, so the
1925    /// cost is one lookup per collected character rather than a walk to the
1926    /// furthest one. The iterator's length is the result's character count,
1927    /// which is why it has to be exact.
1928    fn gather_chars(&self, indices: impl ExactSizeIterator<Item = usize>) -> Self {
1929        let char_len = indices.len();
1930        // Not ascii, so the code points are at least two bytes each.
1931        let mut out = Wtf8Buf::with_capacity(2 * char_len);
1932        let s = self.as_wtf8();
1933        for index in indices {
1934            out.push(
1935                s[self.data.char_index_to_byte(index)..]
1936                    .code_points()
1937                    .next()
1938                    .expect("index is below the character count"),
1939            );
1940        }
1941        // SAFETY: char_len is accurate
1942        unsafe { Self::new_with_char_len(out, char_len) }
1943    }
1944}
1945
1946impl SliceableSequenceOp for PyStr {
1947    type Item = CodePoint;
1948    type Sliced = Self;
1949
1950    fn do_get(&self, index: usize) -> Self::Item {
1951        self.data.nth_char(index)
1952    }
1953
1954    fn getitem_by_index(&self, vm: &VirtualMachine, index: isize) -> PyResult<Self::Item> {
1955        let pos = self
1956            .wrap_index(index)
1957            .ok_or_else(|| vm.new_index_error("string index out of range"))?;
1958        Ok(self.do_get(pos))
1959    }
1960
1961    fn do_slice(&self, range: Range<usize>) -> Self::Sliced {
1962        if let PyKindStr::Ascii(s) = self.as_str_kind() {
1963            return s[range].into();
1964        }
1965        // Both ends resolve through the string's own index, so the slice is a
1966        // byte reslice rather than a walk to `range.start` and another to
1967        // `range.end`.
1968        let char_len = range.len();
1969        let bytes = self.data.char_range_to_bytes(range);
1970        let out = &self.as_wtf8()[bytes];
1971        // SAFETY: char_len is accurate
1972        unsafe { Self::new_with_char_len(out.to_owned(), char_len) }
1973    }
1974
1975    fn do_slice_reverse(&self, range: Range<usize>) -> Self::Sliced {
1976        if let PyKindStr::Ascii(s) = self.as_str_kind() {
1977            let mut out = s[range].to_owned();
1978            out.as_mut_slice().reverse();
1979            return out.into();
1980        }
1981        let char_len = range.len();
1982        let bytes = self.data.char_range_to_bytes(range);
1983        let mut out = Wtf8Buf::with_capacity(bytes.len());
1984        out.extend(self.as_wtf8()[bytes].code_points().rev());
1985        // SAFETY: char_len is accurate
1986        unsafe { Self::new_with_char_len(out, char_len) }
1987    }
1988
1989    fn do_stepped_slice(&self, range: Range<usize>, step: usize) -> Self::Sliced {
1990        if let PyKindStr::Ascii(s) = self.as_str_kind() {
1991            return s[range]
1992                .as_slice()
1993                .iter()
1994                .copied()
1995                .step_by(step)
1996                .collect::<AsciiString>()
1997                .into();
1998        }
1999        self.gather_chars(range.step_by(step))
2000    }
2001
2002    fn do_stepped_slice_reverse(&self, range: Range<usize>, step: usize) -> Self::Sliced {
2003        if let PyKindStr::Ascii(s) = self.as_str_kind() {
2004            return s[range]
2005                .chars()
2006                .rev()
2007                .step_by(step)
2008                .collect::<AsciiString>()
2009                .into();
2010        }
2011        self.gather_chars(range.rev().step_by(step))
2012    }
2013
2014    fn empty() -> Self::Sliced {
2015        Self::default()
2016    }
2017
2018    fn len(&self) -> usize {
2019        self.char_len()
2020    }
2021}
2022
2023impl AsRef<str> for PyRefExact<PyStr> {
2024    #[track_caller]
2025    fn as_ref(&self) -> &str {
2026        self.to_str().expect("str has surrogates")
2027    }
2028}
2029
2030impl AsRef<str> for PyExact<PyStr> {
2031    #[track_caller]
2032    fn as_ref(&self) -> &str {
2033        self.to_str().expect("str has surrogates")
2034    }
2035}
2036
2037impl AsRef<Wtf8> for PyRefExact<PyStr> {
2038    fn as_ref(&self) -> &Wtf8 {
2039        self.as_wtf8()
2040    }
2041}
2042
2043impl AsRef<Wtf8> for PyExact<PyStr> {
2044    fn as_ref(&self) -> &Wtf8 {
2045        self.as_wtf8()
2046    }
2047}
2048
2049impl AnyStrWrapper<Wtf8> for PyStrRef {
2050    fn as_ref(&self) -> Option<&Wtf8> {
2051        Some(self.as_wtf8())
2052    }
2053
2054    fn is_empty(&self) -> bool {
2055        self.data.is_empty()
2056    }
2057}
2058
2059impl AnyStrWrapper<str> for PyStrRef {
2060    fn as_ref(&self) -> Option<&str> {
2061        self.data.as_str()
2062    }
2063
2064    fn is_empty(&self) -> bool {
2065        self.data.is_empty()
2066    }
2067}
2068
2069impl AnyStrWrapper<AsciiStr> for PyStrRef {
2070    fn as_ref(&self) -> Option<&AsciiStr> {
2071        self.data.as_ascii()
2072    }
2073
2074    fn is_empty(&self) -> bool {
2075        self.data.is_empty()
2076    }
2077}
2078
2079#[repr(transparent)]
2080#[derive(Debug)]
2081pub struct PyUtf8Str(PyStr);
2082
2083impl fmt::Display for PyUtf8Str {
2084    #[inline]
2085    fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result {
2086        self.0.fmt(f)
2087    }
2088}
2089
2090impl MaybeTraverse for PyUtf8Str {
2091    const HAS_TRAVERSE: bool = true;
2092    const HAS_CLEAR: bool = false;
2093
2094    fn try_traverse(&self, traverse_fn: &mut TraverseFn<'_>) {
2095        self.0.try_traverse(traverse_fn);
2096    }
2097
2098    fn try_clear(&mut self, _out: &mut Vec<PyObjectRef>) {
2099        // No clear needed for PyUtf8Str
2100    }
2101}
2102
2103impl PyPayload for PyUtf8Str {
2104    #[inline]
2105    fn class(ctx: &Context) -> &'static Py<PyType> {
2106        ctx.types.str_type
2107    }
2108
2109    const PAYLOAD_TYPE_ID: core::any::TypeId = core::any::TypeId::of::<PyStr>();
2110
2111    unsafe fn validate_downcastable_from(obj: &PyObject) -> bool {
2112        // SAFETY: we know the object is a PyStr in this context
2113        let wtf8 = unsafe { obj.downcast_unchecked_ref::<PyStr>() };
2114        wtf8.is_utf8()
2115    }
2116
2117    fn try_downcast_from(obj: &PyObject, vm: &VirtualMachine) -> PyResult<()> {
2118        let str = obj.try_downcast_ref::<PyStr>(vm)?;
2119        str.ensure_valid_utf8(vm)
2120    }
2121}
2122
2123impl<'a> From<&'a AsciiStr> for PyUtf8Str {
2124    fn from(s: &'a AsciiStr) -> Self {
2125        s.to_owned().into()
2126    }
2127}
2128
2129impl From<AsciiString> for PyUtf8Str {
2130    fn from(s: AsciiString) -> Self {
2131        s.into_boxed_ascii_str().into()
2132    }
2133}
2134
2135impl From<Box<AsciiStr>> for PyUtf8Str {
2136    fn from(s: Box<AsciiStr>) -> Self {
2137        let data = StrData::from(s);
2138        unsafe { Self::from_str_data_unchecked(data) }
2139    }
2140}
2141
2142impl From<AsciiChar> for PyUtf8Str {
2143    fn from(ch: AsciiChar) -> Self {
2144        AsciiString::from(ch).into()
2145    }
2146}
2147
2148impl<'a> From<&'a str> for PyUtf8Str {
2149    fn from(s: &'a str) -> Self {
2150        s.to_owned().into()
2151    }
2152}
2153
2154impl From<String> for PyUtf8Str {
2155    fn from(s: String) -> Self {
2156        s.into_boxed_str().into()
2157    }
2158}
2159
2160impl From<char> for PyUtf8Str {
2161    fn from(ch: char) -> Self {
2162        let data = StrData::from(ch);
2163        unsafe { Self::from_str_data_unchecked(data) }
2164    }
2165}
2166
2167impl<'a> From<alloc::borrow::Cow<'a, str>> for PyUtf8Str {
2168    fn from(s: alloc::borrow::Cow<'a, str>) -> Self {
2169        s.into_owned().into()
2170    }
2171}
2172
2173impl From<Box<str>> for PyUtf8Str {
2174    #[inline]
2175    fn from(value: Box<str>) -> Self {
2176        let data = StrData::from(value);
2177        unsafe { Self::from_str_data_unchecked(data) }
2178    }
2179}
2180
2181impl AsRef<Wtf8> for PyUtf8Str {
2182    #[inline]
2183    fn as_ref(&self) -> &Wtf8 {
2184        self.0.as_wtf8()
2185    }
2186}
2187
2188impl AsRef<str> for PyUtf8Str {
2189    #[inline]
2190    fn as_ref(&self) -> &str {
2191        self.as_str()
2192    }
2193}
2194
2195impl PyUtf8Str {
2196    // Create a new `PyUtf8Str` from `StrData` without validation.
2197    // This function must be only used in this module to create conversions.
2198    // # Safety: must be called with a valid UTF-8 string data.
2199    unsafe fn from_str_data_unchecked(data: StrData) -> Self {
2200        Self(PyStr::from(data))
2201    }
2202
2203    /// Returns the underlying WTF-8 slice (always valid UTF-8 for this type).
2204    #[inline]
2205    pub fn as_wtf8(&self) -> &Wtf8 {
2206        self.0.as_wtf8()
2207    }
2208
2209    /// Returns the underlying string slice.
2210    pub fn as_str(&self) -> &str {
2211        debug_assert!(
2212            self.0.is_utf8(),
2213            "PyUtf8Str invariant violated: inner string is not valid UTF-8"
2214        );
2215        // Safety: This is safe because the type invariant guarantees UTF-8 validity.
2216        unsafe { self.0.to_str().unwrap_unchecked() }
2217    }
2218
2219    #[inline]
2220    pub fn as_bytes(&self) -> &[u8] {
2221        self.as_str().as_bytes()
2222    }
2223
2224    #[inline]
2225    pub fn byte_len(&self) -> usize {
2226        self.0.byte_len()
2227    }
2228
2229    #[inline]
2230    pub fn is_empty(&self) -> bool {
2231        self.0.is_empty()
2232    }
2233
2234    #[inline]
2235    pub fn char_len(&self) -> usize {
2236        self.0.char_len()
2237    }
2238}
2239
2240impl Py<PyUtf8Str> {
2241    /// Upcast to PyStr.
2242    pub const fn as_pystr(&self) -> &Py<PyStr> {
2243        unsafe {
2244            // Safety: PyUtf8Str is a wrapper around PyStr, so this cast is safe.
2245            &*(self as *const Self as *const Py<PyStr>)
2246        }
2247    }
2248
2249    /// Returns the underlying `&str`.
2250    #[inline]
2251    pub fn as_str(&self) -> &str {
2252        self.as_pystr().to_str().unwrap_or_else(|| {
2253            debug_assert!(false, "PyUtf8Str invariant violated");
2254            // Safety: PyUtf8Str guarantees valid UTF-8
2255            unsafe { core::hint::unreachable_unchecked() }
2256        })
2257    }
2258}
2259
2260impl PyRef<PyUtf8Str> {
2261    /// Convert to PyStrRef. Safe because PyUtf8Str is a subtype of PyStr.
2262    #[must_use]
2263    pub fn into_wtf8(self) -> PyStrRef {
2264        unsafe { mem::transmute::<Self, PyStrRef>(self) }
2265    }
2266}
2267
2268impl From<PyRef<PyUtf8Str>> for PyRef<PyStr> {
2269    fn from(s: PyRef<PyUtf8Str>) -> Self {
2270        s.into_wtf8()
2271    }
2272}
2273
2274impl PartialEq for PyUtf8Str {
2275    fn eq(&self, other: &Self) -> bool {
2276        self.as_str() == other.as_str()
2277    }
2278}
2279impl Eq for PyUtf8Str {}
2280
2281impl AnyStrContainer<str> for String {
2282    fn new() -> Self {
2283        Self::new()
2284    }
2285
2286    fn with_capacity(capacity: usize) -> Self {
2287        Self::with_capacity(capacity)
2288    }
2289
2290    fn try_with_capacity(capacity: usize) -> Option<Self> {
2291        let mut s = Self::new();
2292        s.try_reserve_exact(capacity).ok()?;
2293        Some(s)
2294    }
2295
2296    fn push_str(&mut self, other: &str) {
2297        Self::push_str(self, other)
2298    }
2299}
2300
2301impl anystr::AnyChar for char {
2302    fn bytes_len(self) -> usize {
2303        self.len_utf8()
2304    }
2305}
2306
2307impl AnyStr for str {
2308    type Char = char;
2309    type Container = String;
2310
2311    fn to_container(&self) -> Self::Container {
2312        self.to_owned()
2313    }
2314
2315    fn as_bytes(&self) -> &[u8] {
2316        self.as_bytes()
2317    }
2318
2319    fn elements(&self) -> impl Iterator<Item = char> {
2320        Self::chars(self)
2321    }
2322
2323    fn get_bytes(&self, range: core::ops::Range<usize>) -> &Self {
2324        &self[range]
2325    }
2326
2327    fn get_chars(&self, range: core::ops::Range<usize>) -> &Self {
2328        rustpython_common::str::get_chars(self, range)
2329    }
2330
2331    fn is_empty(&self) -> bool {
2332        Self::is_empty(self)
2333    }
2334
2335    fn bytes_len(&self) -> usize {
2336        Self::len(self)
2337    }
2338
2339    fn py_split_whitespace<F>(&self, maxsplit: isize, convert: F) -> Vec<PyObjectRef>
2340    where
2341        F: Fn(&Self) -> PyObjectRef,
2342    {
2343        // CPython split_whitespace
2344        let mut splits = Vec::new();
2345        let mut last_offset = 0;
2346        let mut count = maxsplit;
2347        for (offset, separator) in self.match_indices(unicode::classify::is_space) {
2348            if last_offset == offset {
2349                last_offset += separator.len();
2350                continue;
2351            }
2352            if count == 0 {
2353                break;
2354            }
2355            splits.push(convert(&self[last_offset..offset]));
2356            last_offset = offset + separator.len();
2357            count -= 1;
2358        }
2359        if last_offset != self.len() {
2360            splits.push(convert(&self[last_offset..]));
2361        }
2362        splits
2363    }
2364
2365    fn py_rsplit_whitespace<F>(&self, maxsplit: isize, convert: F) -> Vec<PyObjectRef>
2366    where
2367        F: Fn(&Self) -> PyObjectRef,
2368    {
2369        // CPython rsplit_whitespace
2370        let mut splits = Vec::new();
2371        let mut last_offset = self.len();
2372        let mut count = maxsplit;
2373        for (offset, separator) in self.rmatch_indices(unicode::classify::is_space) {
2374            if last_offset == offset + separator.len() {
2375                last_offset = offset;
2376                continue;
2377            }
2378            if count == 0 {
2379                break;
2380            }
2381            splits.push(convert(&self[offset + separator.len()..last_offset]));
2382            last_offset = offset;
2383            count -= 1;
2384        }
2385        if last_offset != 0 {
2386            splits.push(convert(&self[..last_offset]));
2387        }
2388        splits
2389    }
2390
2391    fn py_islower(&self) -> bool {
2392        self.is_cased(case::is_lowercase, case::is_uppercase)
2393    }
2394
2395    fn py_isupper(&self) -> bool {
2396        self.is_cased(case::is_uppercase, case::is_lowercase)
2397    }
2398}
2399
2400impl AnyStrContainer<Wtf8> for Wtf8Buf {
2401    fn new() -> Self {
2402        Self::new()
2403    }
2404
2405    fn with_capacity(capacity: usize) -> Self {
2406        Self::with_capacity(capacity)
2407    }
2408
2409    fn try_with_capacity(capacity: usize) -> Option<Self> {
2410        let mut s = Self::new();
2411        s.try_reserve_exact(capacity).ok()?;
2412        Some(s)
2413    }
2414
2415    fn push_str(&mut self, other: &Wtf8) {
2416        self.push_wtf8(other)
2417    }
2418}
2419
2420impl anystr::AnyChar for CodePoint {
2421    fn bytes_len(self) -> usize {
2422        self.len_wtf8()
2423    }
2424}
2425
2426impl AnyStr for Wtf8 {
2427    type Char = CodePoint;
2428    type Container = Wtf8Buf;
2429
2430    fn to_container(&self) -> Self::Container {
2431        self.to_owned()
2432    }
2433
2434    fn as_bytes(&self) -> &[u8] {
2435        self.as_bytes()
2436    }
2437
2438    fn elements(&self) -> impl Iterator<Item = Self::Char> {
2439        self.code_points()
2440    }
2441
2442    fn get_bytes(&self, range: core::ops::Range<usize>) -> &Self {
2443        &self[range]
2444    }
2445
2446    fn get_chars(&self, range: core::ops::Range<usize>) -> &Self {
2447        rustpython_common::str::get_codepoints(self, range)
2448    }
2449
2450    fn bytes_len(&self) -> usize {
2451        self.len()
2452    }
2453
2454    fn is_empty(&self) -> bool {
2455        self.is_empty()
2456    }
2457
2458    fn py_split_whitespace<F>(&self, maxsplit: isize, convert: F) -> Vec<PyObjectRef>
2459    where
2460        F: Fn(&Self) -> PyObjectRef,
2461    {
2462        // CPython split_whitespace
2463        let mut splits = Vec::new();
2464        let mut last_offset = 0;
2465        let mut count = maxsplit;
2466        for (offset, separator) in self
2467            .code_point_indices()
2468            .filter(|(_, c)| c.is_char_and(unicode::classify::is_space))
2469        {
2470            if last_offset == offset {
2471                last_offset += separator.len_wtf8();
2472                continue;
2473            }
2474            if count == 0 {
2475                break;
2476            }
2477            splits.push(convert(&self[last_offset..offset]));
2478            last_offset = offset + separator.len_wtf8();
2479            count -= 1;
2480        }
2481        if last_offset != self.len() {
2482            splits.push(convert(&self[last_offset..]));
2483        }
2484        splits
2485    }
2486
2487    fn py_rsplit_whitespace<F>(&self, maxsplit: isize, convert: F) -> Vec<PyObjectRef>
2488    where
2489        F: Fn(&Self) -> PyObjectRef,
2490    {
2491        // CPython rsplit_whitespace
2492        let mut splits = Vec::new();
2493        let mut last_offset = self.len();
2494        let mut count = maxsplit;
2495        for (offset, separator) in self
2496            .code_point_indices()
2497            .rev()
2498            .filter(|(_, c)| c.is_char_and(unicode::classify::is_space))
2499        {
2500            if last_offset == offset + separator.len_wtf8() {
2501                last_offset = offset;
2502                continue;
2503            }
2504            if count == 0 {
2505                break;
2506            }
2507            splits.push(convert(&self[offset + separator.len_wtf8()..last_offset]));
2508            last_offset = offset;
2509            count -= 1;
2510        }
2511        if last_offset != 0 {
2512            splits.push(convert(&self[..last_offset]));
2513        }
2514        splits
2515    }
2516
2517    fn py_islower(&self) -> bool {
2518        self.is_cased(case::is_lowercase, case::is_uppercase)
2519    }
2520
2521    fn py_isupper(&self) -> bool {
2522        self.is_cased(case::is_uppercase, case::is_lowercase)
2523    }
2524}
2525
2526impl AnyStrContainer<AsciiStr> for AsciiString {
2527    fn new() -> Self {
2528        Self::new()
2529    }
2530
2531    fn with_capacity(capacity: usize) -> Self {
2532        Self::with_capacity(capacity)
2533    }
2534
2535    fn try_with_capacity(capacity: usize) -> Option<Self> {
2536        let mut v = Vec::new();
2537        v.try_reserve_exact(capacity).ok()?;
2538        Some(Self::from(v))
2539    }
2540
2541    fn push_str(&mut self, other: &AsciiStr) {
2542        Self::push_str(self, other)
2543    }
2544}
2545
2546impl anystr::AnyChar for ascii::AsciiChar {
2547    fn bytes_len(self) -> usize {
2548        1
2549    }
2550}
2551
2552const ASCII_WHITESPACES: [u8; 6] = [0x20, 0x09, 0x0a, 0x0c, 0x0d, 0x0b];
2553
2554impl AnyStr for AsciiStr {
2555    type Char = AsciiChar;
2556    type Container = AsciiString;
2557
2558    fn to_container(&self) -> Self::Container {
2559        self.to_ascii_string()
2560    }
2561
2562    fn as_bytes(&self) -> &[u8] {
2563        self.as_bytes()
2564    }
2565
2566    fn elements(&self) -> impl Iterator<Item = Self::Char> {
2567        self.chars()
2568    }
2569
2570    fn get_bytes(&self, range: core::ops::Range<usize>) -> &Self {
2571        &self[range]
2572    }
2573
2574    fn get_chars(&self, range: core::ops::Range<usize>) -> &Self {
2575        &self[range]
2576    }
2577
2578    fn bytes_len(&self) -> usize {
2579        self.len()
2580    }
2581
2582    fn is_empty(&self) -> bool {
2583        self.is_empty()
2584    }
2585
2586    fn py_split_whitespace<F>(&self, maxsplit: isize, convert: F) -> Vec<PyObjectRef>
2587    where
2588        F: Fn(&Self) -> PyObjectRef,
2589    {
2590        let mut splits = Vec::new();
2591        let mut count = maxsplit;
2592        let mut haystack = self;
2593        while let Some(offset) = haystack.as_bytes().find_byteset(ASCII_WHITESPACES) {
2594            if offset != 0 {
2595                if count == 0 {
2596                    break;
2597                }
2598                splits.push(convert(&haystack[..offset]));
2599                count -= 1;
2600            }
2601            haystack = &haystack[offset + 1..];
2602        }
2603        if !haystack.is_empty() {
2604            splits.push(convert(haystack));
2605        }
2606        splits
2607    }
2608
2609    fn py_rsplit_whitespace<F>(&self, maxsplit: isize, convert: F) -> Vec<PyObjectRef>
2610    where
2611        F: Fn(&Self) -> PyObjectRef,
2612    {
2613        // CPython rsplit_whitespace
2614        let mut splits = Vec::new();
2615        let mut count = maxsplit;
2616        let mut haystack = self;
2617        while let Some(offset) = haystack.as_bytes().rfind_byteset(ASCII_WHITESPACES) {
2618            if offset + 1 != haystack.len() {
2619                if count == 0 {
2620                    break;
2621                }
2622                splits.push(convert(&haystack[offset + 1..]));
2623                count -= 1;
2624            }
2625            haystack = &haystack[..offset];
2626        }
2627        if !haystack.is_empty() {
2628            splits.push(convert(haystack));
2629        }
2630        splits
2631    }
2632}
2633
2634/// The unique reference of interned PyStr
2635/// Always intended to be used as a static reference
2636pub type PyStrInterned = PyInterned<PyStr>;
2637
2638impl PyStrInterned {
2639    #[inline]
2640    pub fn to_exact(&'static self) -> PyRefExact<PyStr> {
2641        unsafe { PyRefExact::new_unchecked(self.to_owned()) }
2642    }
2643}
2644
2645impl core::fmt::Display for PyStrInterned {
2646    fn fmt(&self, f: &mut core::fmt::Formatter<'_>) -> core::fmt::Result {
2647        self.data.fmt(f)
2648    }
2649}
2650
2651impl AsRef<str> for PyStrInterned {
2652    #[inline(always)]
2653    fn as_ref(&self) -> &str {
2654        self.to_str()
2655            .expect("Interned PyStr should always be valid UTF-8")
2656    }
2657}
2658
2659/// Interned PyUtf8Str — guaranteed UTF-8 at type level.
2660/// Same layout as `PyStrInterned` due to `#[repr(transparent)]` on both
2661/// `PyInterned<T>` and `PyUtf8Str`.
2662pub type PyUtf8StrInterned = PyInterned<PyUtf8Str>;
2663
2664impl PyUtf8StrInterned {
2665    /// Returns the underlying `&str`.
2666    #[inline]
2667    pub fn as_str(&self) -> &str {
2668        Py::<PyUtf8Str>::as_str(self)
2669    }
2670
2671    /// View as `PyStrInterned` (widening: UTF-8 → WTF-8).
2672    #[inline]
2673    pub fn as_interned_str(&self) -> &PyStrInterned {
2674        // Safety: PyUtf8Str is #[repr(transparent)] over PyStr,
2675        // so PyInterned<PyUtf8Str> has the same layout as PyInterned<PyStr>.
2676        unsafe { &*(self as *const Self as *const PyStrInterned) }
2677    }
2678
2679    /// Narrow a `PyStrInterned` to `PyUtf8StrInterned`.
2680    ///
2681    /// # Safety
2682    /// The caller must ensure that the interned string is valid UTF-8.
2683    #[inline]
2684    pub unsafe fn from_str_interned_unchecked(s: &PyStrInterned) -> &Self {
2685        unsafe { &*(s as *const PyStrInterned as *const Self) }
2686    }
2687}
2688
2689impl core::fmt::Display for PyUtf8StrInterned {
2690    fn fmt(&self, f: &mut core::fmt::Formatter<'_>) -> core::fmt::Result {
2691        f.write_str(self.as_str())
2692    }
2693}
2694
2695impl AsRef<str> for PyUtf8StrInterned {
2696    #[inline(always)]
2697    fn as_ref(&self) -> &str {
2698        self.as_str()
2699    }
2700}
2701
2702#[cfg(test)]
2703mod tests {
2704    use super::*;
2705    use crate::{Context, Interpreter, Py};
2706    use rustpython_common::wtf8::Wtf8Buf;
2707
2708    #[test]
2709    fn str_title() {
2710        let tests = vec![
2711            (" Hello ", " hello "),
2712            ("Hello ", "hello "),
2713            ("Hello ", "Hello "),
2714            ("Format This As Title String", "fOrMaT thIs aS titLe String"),
2715            ("Format,This-As*Title;String", "fOrMaT,thIs-aS*titLe;String"),
2716            ("Getint", "getInt"),
2717            // spell-checker:disable-next-line
2718            ("Greek Ωppercases ...", "greek ωppercases ..."),
2719            // spell-checker:disable-next-line
2720            ("Greek ῼitlecases ...", "greek ῳitlecases ..."),
2721            // Latin Extended-B digraphs: uppercase forms map to titlecase forms
2722            // (e.g. U+01F1 'DZ' -> U+01F2 'Dz', U+01C4 'DŽ' -> U+01C5 'Dž').
2723            ("\u{01F2}", "\u{01F1}"),
2724            ("\u{01C5}", "\u{01C4}"),
2725        ];
2726        for (title, input) in tests {
2727            assert_eq!(
2728                Context::genesis().new_str(input).title().as_str(),
2729                Ok(title)
2730            );
2731        }
2732    }
2733
2734    #[test]
2735    fn str_istitle() {
2736        let pos = vec![
2737            "A",
2738            "A Titlecased Line",
2739            "A\nTitlecased Line",
2740            "A Titlecased, Line",
2741            // spell-checker:disable-next-line
2742            "Greek Ωppercases ...",
2743            // spell-checker:disable-next-line
2744            "Greek ῼitlecases ...",
2745        ];
2746
2747        for s in pos {
2748            assert!(Context::genesis().new_str(s).istitle());
2749        }
2750
2751        let neg = vec![
2752            "",
2753            "a",
2754            "\n",
2755            "Not a capitalized String",
2756            "Not\ta Titlecase String",
2757            "Not--a Titlecase String",
2758            "NOT",
2759        ];
2760        for s in neg {
2761            assert!(!Context::genesis().new_str(s).istitle());
2762        }
2763    }
2764
2765    #[test]
2766    fn str_maketrans_and_translate() {
2767        Interpreter::without_stdlib(Default::default()).enter(|vm| {
2768            let table = vm.ctx.new_dict();
2769            table
2770                .set_item("a", vm.ctx.new_str("🎅").into(), vm)
2771                .unwrap();
2772            table.set_item("b", vm.ctx.none(), vm).unwrap();
2773            table
2774                .set_item("c", vm.ctx.new_str(ascii!("xda")).into(), vm)
2775                .unwrap();
2776            let translated = Py::<PyStr>::maketrans(
2777                table.into(),
2778                OptionalArg::Missing,
2779                OptionalArg::Missing,
2780                vm,
2781            )
2782            .unwrap();
2783            let text = vm.ctx.new_str("abc");
2784            let translated = text.translate(translated, vm).unwrap();
2785            assert_eq!(translated, Wtf8Buf::from("🎅xda"));
2786            let translated = text.translate(vm.ctx.new_int(3).into(), vm);
2787            assert_eq!("TypeError", &*translated.unwrap_err().class().name(),);
2788        })
2789    }
2790
2791    #[test]
2792    fn str_isprintable_unicode15() {
2793        // Regression test for https://github.com/RustPython/RustPython/issues/7525
2794        // At the time of the issue, RustPython used unic_ucd_category which had
2795        // outdated Unicode data, causing U+0B55 to be misclassified as Unassigned.
2796        // Now fixed by migrating to icu_properties with up-to-date Unicode data.
2797
2798        // Characters that should be printable
2799        assert!(Context::genesis().new_str("\u{0B55}").isprintable());
2800        assert!(Context::genesis().new_str("A").isprintable());
2801        assert!(Context::genesis().new_str(" ").isprintable());
2802        assert!(Context::genesis().new_str("").isprintable());
2803
2804        // Characters that should NOT be printable
2805        assert!(!Context::genesis().new_str("\x00").isprintable());
2806        assert!(!Context::genesis().new_str("\u{200B}").isprintable());
2807        assert!(!Context::genesis().new_str("\u{E000}").isprintable());
2808        assert!(!Context::genesis().new_str("\u{00A0}").isprintable());
2809    }
2810}