1use super::{
2 PositionIterInternal, PyBytesRef, PyDict, PyList, PyTuple, PyTupleRef, PyType, PyTypeRef,
3 int::{PyInt, PyIntRef},
4 iter::{IterStatus, builtins_iter},
5};
6use crate::{
7 AsObject, Context, Py, PyExact, PyObject, PyObjectRef, PyPayload, PyRef, PyRefExact, PyResult,
8 TryFromBorrowedObject, TryFromObject, VirtualMachine,
9 anystr::{self, AnyStr, AnyStrContainer, AnyStrWrapper, StringRange, adjust_indices},
10 atomic_func,
11 bytes_inner::{swapcase_ascii, title_ascii},
12 cformat::cformat_string,
13 class::{PyClassDef, PyClassImpl},
14 common::{
15 lock::LazyLock,
16 str::{PyKindStr, StrData, StrKind},
17 },
18 convert::{IntoPyException, ToPyException, ToPyObject, ToPyResult},
19 format::{format, format_map},
20 function::{ArgIterable, FuncArgs, OptionalArg, PyComparisonValue, PySsize},
21 intern::PyInterned,
22 object::{MaybeTraverse, Traverse, TraverseFn},
23 protocol::{
24 BufferFlags, PyBuffer, PyIterReturn, PyMappingMethods, PyNumberMethods, PySequenceMethods,
25 },
26 sequence::SequenceExt,
27 sliceable::{SequenceIndex, SliceableSequenceOp},
28 types::{
29 AsMapping, AsNumber, AsSequence, Comparable, Constructor, Hashable, IterNext, Iterable,
30 PyComparisonOp, Representable, SelfIter,
31 },
32};
33use alloc::{borrow::Cow, fmt};
34use ascii::{AsciiChar, AsciiStr, AsciiString};
35use bstr::ByteSlice;
36use core::ffi::CStr;
37use core::{char, mem, ops::Range};
38use itertools::Itertools;
39use memchr::memchr;
40use num_traits::ToPrimitive;
41use rustpython_common::{
42 ascii,
43 atomic::{self, PyAtomic, Radium},
44 format::{FormatSpec, FormatString, FromTemplate},
45 hash,
46 lock::PyMutex,
47 str::DeduceStrKind,
48 wtf8::{CodePoint, Wtf8, Wtf8Buf, Wtf8Concat},
49};
50
51use rustpython_unicode::{self as unicode, case};
52
53impl<'a> TryFromBorrowedObject<'a> for String {
54 fn try_from_borrowed_object(vm: &VirtualMachine, obj: &'a PyObject) -> PyResult<Self> {
55 obj.try_value_with(|pystr: &Py<PyUtf8Str>| Ok(pystr.as_str().to_owned()), vm)
56 }
57}
58
59impl<'a> TryFromBorrowedObject<'a> for &'a str {
60 fn try_from_borrowed_object(vm: &VirtualMachine, obj: &'a PyObject) -> PyResult<Self> {
61 let pystr: &Py<PyUtf8Str> = TryFromBorrowedObject::try_from_borrowed_object(vm, obj)?;
62 Ok(pystr.as_str())
63 }
64}
65
66impl<'a> TryFromBorrowedObject<'a> for &'a Wtf8 {
67 fn try_from_borrowed_object(vm: &VirtualMachine, obj: &'a PyObject) -> PyResult<Self> {
68 let pystr: &Py<PyStr> = TryFromBorrowedObject::try_from_borrowed_object(vm, obj)?;
69 Ok(pystr.as_wtf8())
70 }
71}
72
73pub type PyStrRef = PyRef<PyStr>;
74pub type PyUtf8StrRef = PyRef<PyUtf8Str>;
75
76#[pyclass(module = false, name = "str")]
77pub struct PyStr {
78 data: StrData,
79 hash: PyAtomic<hash::PyHash>,
80}
81
82impl fmt::Debug for PyStr {
83 fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result {
84 f.debug_struct("PyStr")
85 .field("value", &self.as_wtf8())
86 .field("kind", &self.data.kind())
87 .field("hash", &self.hash)
88 .finish()
89 }
90}
91
92impl AsRef<str> for PyStr {
93 #[track_caller] fn as_ref(&self) -> &str {
95 self.to_str().expect("str has surrogates")
96 }
97}
98
99impl AsRef<str> for Py<PyStr> {
100 #[track_caller] fn as_ref(&self) -> &str {
102 self.to_str().expect("str has surrogates")
103 }
104}
105
106impl AsRef<str> for PyStrRef {
107 #[track_caller] fn as_ref(&self) -> &str {
109 self.to_str().expect("str has surrogates")
110 }
111}
112
113impl AsRef<Wtf8> for PyStr {
114 fn as_ref(&self) -> &Wtf8 {
115 self.as_wtf8()
116 }
117}
118
119impl AsRef<Wtf8> for Py<PyStr> {
120 fn as_ref(&self) -> &Wtf8 {
121 self.as_wtf8()
122 }
123}
124
125impl AsRef<Wtf8> for PyStrRef {
126 fn as_ref(&self) -> &Wtf8 {
127 self.as_wtf8()
128 }
129}
130
131impl Wtf8Concat for PyStr {
132 #[inline]
133 fn fmt_wtf8(&self, buf: &mut Wtf8Buf) {
134 buf.push_wtf8(self.as_wtf8());
135 }
136}
137
138impl Wtf8Concat for Py<PyStr> {
139 #[inline]
140 fn fmt_wtf8(&self, buf: &mut Wtf8Buf) {
141 buf.push_wtf8(self.as_wtf8());
142 }
143}
144
145impl<'a> From<&'a AsciiStr> for PyStr {
146 fn from(s: &'a AsciiStr) -> Self {
147 s.to_owned().into()
148 }
149}
150
151impl From<AsciiString> for PyStr {
152 fn from(s: AsciiString) -> Self {
153 s.into_boxed_ascii_str().into()
154 }
155}
156
157impl From<Box<AsciiStr>> for PyStr {
158 fn from(s: Box<AsciiStr>) -> Self {
159 StrData::from(s).into()
160 }
161}
162
163impl From<AsciiChar> for PyStr {
164 fn from(ch: AsciiChar) -> Self {
165 AsciiString::from(ch).into()
166 }
167}
168
169impl<'a> From<&'a str> for PyStr {
170 fn from(s: &'a str) -> Self {
171 s.to_owned().into()
172 }
173}
174
175impl<'a> From<&'a Wtf8> for PyStr {
176 fn from(s: &'a Wtf8) -> Self {
177 s.to_owned().into()
178 }
179}
180
181impl From<String> for PyStr {
182 fn from(s: String) -> Self {
183 s.into_boxed_str().into()
184 }
185}
186
187impl From<Wtf8Buf> for PyStr {
188 fn from(w: Wtf8Buf) -> Self {
189 w.into_box().into()
190 }
191}
192
193impl From<char> for PyStr {
194 fn from(ch: char) -> Self {
195 StrData::from(ch).into()
196 }
197}
198
199impl From<CodePoint> for PyStr {
200 fn from(ch: CodePoint) -> Self {
201 StrData::from(ch).into()
202 }
203}
204
205impl From<StrData> for PyStr {
206 fn from(data: StrData) -> Self {
207 Self {
208 data,
209 hash: Radium::new(hash::SENTINEL),
210 }
211 }
212}
213
214impl<'a> From<alloc::borrow::Cow<'a, str>> for PyStr {
215 fn from(s: alloc::borrow::Cow<'a, str>) -> Self {
216 s.into_owned().into()
217 }
218}
219
220impl From<Box<str>> for PyStr {
221 #[inline]
222 fn from(value: Box<str>) -> Self {
223 StrData::from(value).into()
224 }
225}
226
227impl From<Box<Wtf8>> for PyStr {
228 #[inline]
229 fn from(value: Box<Wtf8>) -> Self {
230 StrData::from(value).into()
231 }
232}
233
234impl Default for PyStr {
235 fn default() -> Self {
236 Self {
237 data: StrData::default(),
238 hash: Radium::new(hash::SENTINEL),
239 }
240 }
241}
242
243impl fmt::Display for PyStr {
244 #[inline]
245 fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result {
246 self.as_wtf8().fmt(f)
247 }
248}
249
250pub trait AsPyStr<'a>
251where
252 Self: 'a,
253{
254 #[allow(
255 clippy::wrong_self_convention,
256 reason = "this trait is intentionally implemented for references"
257 )]
258 fn as_pystr(self, ctx: &Context) -> &'a Py<PyStr>;
259}
260
261impl<'a> AsPyStr<'a> for &'a Py<PyStr> {
262 #[inline]
263 fn as_pystr(self, _ctx: &Context) -> &'a Py<PyStr> {
264 self
265 }
266}
267
268impl<'a> AsPyStr<'a> for &'a Py<PyUtf8Str> {
269 #[inline]
270 fn as_pystr(self, _ctx: &Context) -> &'a Py<PyStr> {
271 Py::<PyUtf8Str>::as_pystr(self)
272 }
273}
274
275impl<'a> AsPyStr<'a> for &'a PyStrRef {
276 #[inline]
277 fn as_pystr(self, _ctx: &Context) -> &'a Py<PyStr> {
278 self
279 }
280}
281
282impl<'a> AsPyStr<'a> for &'a PyUtf8StrRef {
283 #[inline]
284 fn as_pystr(self, _ctx: &Context) -> &'a Py<PyStr> {
285 Py::<PyUtf8Str>::as_pystr(self)
286 }
287}
288
289impl AsPyStr<'static> for &'static str {
290 #[inline]
291 fn as_pystr(self, ctx: &Context) -> &'static Py<PyStr> {
292 ctx.intern_str(self)
293 }
294}
295
296impl<'a> AsPyStr<'a> for &'a PyStrInterned {
297 #[inline]
298 fn as_pystr(self, _ctx: &Context) -> &'a Py<PyStr> {
299 self
300 }
301}
302
303impl<'a> AsPyStr<'a> for &'a PyUtf8StrInterned {
304 #[inline]
305 fn as_pystr(self, _ctx: &Context) -> &'a Py<PyStr> {
306 Py::<PyUtf8Str>::as_pystr(self)
307 }
308}
309
310#[pyclass(module = false, name = "str_iterator", traverse = "manual")]
311#[derive(Debug)]
312pub(crate) struct PyStrIterator {
313 internal: PyMutex<(PositionIterInternal<PyStrRef>, usize)>,
314}
315
316unsafe impl Traverse for PyStrIterator {
317 fn traverse(&self, tracer: &mut TraverseFn<'_>) {
318 self.internal.lock().0.traverse(tracer);
320 }
321}
322
323impl PyPayload for PyStrIterator {
324 fn class(ctx: &Context) -> &'static Py<PyType> {
325 ctx.types.str_iterator_type
326 }
327}
328
329#[pyclass(flags(DISALLOW_INSTANTIATION), with(IterNext, Iterable))]
330impl Py<PyStrIterator> {
331 #[pymethod]
332 fn __length_hint__(&self) -> usize {
333 self.internal.lock().0.length_hint(|obj| obj.char_len())
334 }
335
336 #[pymethod]
337 fn __setstate__(&self, object: PyObjectRef, vm: &VirtualMachine) -> PyResult<()> {
338 let mut internal = self.internal.lock();
339 internal.1 = usize::MAX;
340 internal
341 .0
342 .set_state(&object, |obj, pos| pos.min(obj.char_len()), vm)
343 }
344
345 #[pymethod]
346 fn __reduce__(&self, vm: &VirtualMachine) -> PyResult<PyTupleRef> {
347 let func = builtins_iter(vm)?;
348 Ok(self.internal.lock().0.reduce(
349 func,
350 |x| x.clone().into(),
351 |vm| vm.ctx.empty_str.to_owned().into(),
352 vm,
353 ))
354 }
355}
356
357impl SelfIter for PyStrIterator {}
358
359impl IterNext for PyStrIterator {
360 fn next(zelf: &Py<Self>, vm: &VirtualMachine) -> PyResult<PyIterReturn> {
361 let mut internal = zelf.internal.lock();
362
363 if let IterStatus::Active(s) = &internal.0.status {
364 let value = s.as_wtf8();
365
366 if internal.1 == usize::MAX {
367 if let Some((offset, ch)) = value.code_point_indices().nth(internal.0.position) {
368 internal.0.position += 1;
369 internal.1 = offset + ch.len_wtf8();
370 return Ok(PyIterReturn::Return(ch.to_pyobject(vm)));
371 }
372 } else if let Some(value) = value.get(internal.1..)
373 && let Some(ch) = value.code_points().next()
374 {
375 internal.0.position += 1;
376 internal.1 += ch.len_wtf8();
377 return Ok(PyIterReturn::Return(ch.to_pyobject(vm)));
378 }
379 let released = internal.0.exhaust();
380 drop(internal);
383 drop(released);
384 }
385 Ok(PyIterReturn::StopIteration(None))
386 }
387}
388
389#[derive(FromArgs)]
390pub struct StrArgs {
391 #[pyarg(any, optional)]
392 object: OptionalArg<PyObjectRef>,
393 #[pyarg(any, optional)]
394 encoding: OptionalArg<PyUtf8StrRef>,
395 #[pyarg(any, optional)]
396 errors: OptionalArg<PyUtf8StrRef>,
397}
398
399impl Constructor for PyStr {
400 type Args = StrArgs;
401
402 fn slot_new(cls: PyTypeRef, func_args: FuncArgs, vm: &VirtualMachine) -> PyResult {
403 if cls.is(vm.ctx.types.str_type)
405 && func_args.args.len() == 1
406 && func_args.kwargs.is_empty()
407 && func_args.args[0].class().is(vm.ctx.types.str_type)
408 {
409 return Ok(func_args.args[0].clone());
410 }
411
412 let args: Self::Args = func_args.bind_for(vm, Self::NAME)?;
413
414 if cls.is(vm.ctx.types.str_type)
422 && args.encoding.is_missing()
423 && args.errors.is_missing()
424 && let OptionalArg::Present(input) = &args.object
425 {
426 return Ok(input.str(vm)?.into());
427 }
428
429 let payload = Self::py_new(&cls, args, vm)?;
430 payload.into_ref_with_type(vm, cls).map(Into::into)
431 }
432
433 fn py_new(_cls: &Py<PyType>, args: Self::Args, vm: &VirtualMachine) -> PyResult<Self> {
434 match args.object {
435 OptionalArg::Present(input) => {
436 let encoding = args.encoding.into_option();
437 let errors = args.errors.into_option();
438 if encoding.is_some() || errors.is_some() {
442 if input.fast_isinstance(vm.ctx.types.str_type) {
445 return Err(vm.new_type_error("decoding str is not supported"));
446 }
447 let input = if input.fast_isinstance(vm.ctx.types.bytes_type)
448 || input.fast_isinstance(vm.ctx.types.bytearray_type)
449 {
450 input
451 } else {
452 let buffer = PyBuffer::from_object(vm, &input, BufferFlags::SIMPLE)
455 .map_err(|_| {
456 vm.new_type_error(format!(
457 "decoding to str: need a bytes-like object, {} found",
458 input.class().name()
459 ))
460 })?;
461 vm.ctx
462 .new_bytes(buffer.contiguous_or_collect(<[u8]>::to_vec))
463 .into()
464 };
465 let enc_str = encoding.as_ref().map_or("utf-8", |e| e.as_str());
466 let s = vm
467 .state
468 .codec_registry
469 .decode_text_object(input, enc_str, errors, vm)?;
470 Ok(Self::from(s.as_wtf8().to_owned()))
471 } else {
472 let s = input.str(vm)?;
473 Ok(Self::from(s.as_wtf8().to_owned()))
474 }
475 }
476 OptionalArg::Missing => Ok(Self::from(String::new())),
477 }
478 }
479}
480
481impl PyStr {
482 unsafe fn new_str_unchecked(data: Box<Wtf8>, kind: StrKind) -> Self {
484 unsafe { StrData::new_str_unchecked(data, kind) }.into()
485 }
486
487 unsafe fn new_with_char_len<T: DeduceStrKind + Into<Box<Wtf8>>>(s: T, char_len: usize) -> Self {
488 let kind = s.str_kind();
489 unsafe { StrData::new_with_char_len(s.into(), kind, char_len) }.into()
490 }
491
492 #[must_use]
495 pub unsafe fn new_ascii_unchecked(bytes: Vec<u8>) -> Self {
496 unsafe { AsciiString::from_ascii_unchecked(bytes) }.into()
497 }
498
499 #[deprecated(note = "use PyStr::from(...).into_ref() instead")]
500 pub fn new_ref(zelf: impl Into<Self>, ctx: &Context) -> PyRef<Self> {
501 let zelf = zelf.into();
502 zelf.into_ref(ctx)
503 }
504
505 fn new_substr(&self, s: Wtf8Buf) -> Self {
506 let kind = if self.kind().is_ascii() || s.is_ascii() {
507 StrKind::Ascii
508 } else if self.kind().is_utf8() || s.is_utf8() {
509 StrKind::Utf8
510 } else {
511 StrKind::Wtf8
512 };
513 unsafe {
514 Self::new_str_unchecked(s.into(), kind)
516 }
517 }
518
519 #[inline]
520 pub const fn as_wtf8(&self) -> &Wtf8 {
521 self.data.as_wtf8()
522 }
523
524 pub const fn as_bytes(&self) -> &[u8] {
525 self.data.as_wtf8().as_bytes()
526 }
527
528 pub fn to_str(&self) -> Option<&str> {
529 self.data.as_str()
530 }
531
532 #[inline]
537 #[track_caller]
538 pub fn expect_str(&self) -> &str {
539 self.to_str().expect("PyStr contains surrogates")
540 }
541
542 pub(crate) fn ensure_valid_utf8(&self, vm: &VirtualMachine) -> PyResult<()> {
543 if self.is_utf8() {
544 Ok(())
545 } else {
546 let start = self
547 .as_wtf8()
548 .code_points()
549 .position(|c| c.to_char().is_none())
550 .unwrap();
551 Err(vm.new_unicode_encode_error(
552 identifier!(vm, utf_8).to_owned(),
553 vm.ctx.new_str(self.data.clone()),
554 start,
555 start + 1,
556 vm.ctx.new_str("surrogates not allowed"),
557 ))
558 }
559 }
560
561 #[inline]
563 #[must_use]
564 pub fn contains_nuls(&self) -> bool {
565 memchr(b'\0', self.as_bytes()).is_some()
566 }
567
568 pub fn to_string_lossy(&self) -> Cow<'_, str> {
569 self.to_str()
570 .map_or_else(|| self.as_wtf8().to_string_lossy(), Cow::Borrowed)
571 }
572
573 pub const fn kind(&self) -> StrKind {
574 self.data.kind()
575 }
576
577 #[inline]
578 pub fn as_str_kind(&self) -> PyKindStr<'_> {
579 self.data.as_str_kind()
580 }
581
582 pub const fn is_utf8(&self) -> bool {
583 self.kind().is_utf8()
584 }
585
586 fn char_all<F>(&self, test: F) -> bool
587 where
588 F: Fn(char) -> bool,
589 {
590 match self.as_str_kind() {
591 PyKindStr::Ascii(s) => s.chars().all(|ch| test(ch.into())),
592 PyKindStr::Utf8(s) => s.chars().all(test),
593 PyKindStr::Wtf8(w) => w.code_points().all(|ch| ch.is_char_and(&test)),
594 }
595 }
596
597 fn repeat(zelf: PyRef<Self>, value: isize, vm: &VirtualMachine) -> PyResult<PyRef<Self>> {
598 if value == 0 && zelf.class().is(vm.ctx.types.str_type) {
599 return Ok(vm.ctx.empty_str.to_owned());
602 }
603 if (value == 1 || zelf.is_empty()) && zelf.class().is(vm.ctx.types.str_type) {
604 return Ok(zelf);
609 }
610 zelf.as_wtf8()
611 .as_bytes()
612 .mul(vm, value)
613 .map(|x| Self::from(unsafe { Wtf8Buf::from_bytes_unchecked(x) }).into_ref(&vm.ctx))
614 }
615
616 pub fn as_utf8(&self) -> Option<&PyUtf8Str> {
617 if self.is_utf8() {
618 Some(unsafe { &*(self as *const Self as *const PyUtf8Str) })
620 } else {
621 None
622 }
623 }
624
625 pub fn try_as_utf8<'a>(&'a self, vm: &VirtualMachine) -> PyResult<&'a PyUtf8Str> {
626 self.as_utf8()
627 .ok_or_else(|| self.ensure_valid_utf8(vm).unwrap_err())
628 }
629}
630
631impl Py<PyStr> {
632 #[inline]
633 pub fn as_wtf8(&self) -> &Wtf8 {
634 self.payload().as_wtf8()
635 }
636
637 #[inline]
638 pub fn as_bytes(&self) -> &[u8] {
639 self.payload().as_bytes()
640 }
641
642 pub fn as_utf8(&self) -> Option<&Py<PyUtf8Str>> {
643 if self.is_utf8() {
644 Some(unsafe { &*(self as *const Self as *const Py<PyUtf8Str>) })
646 } else {
647 None
648 }
649 }
650
651 pub fn try_as_utf8<'a>(&'a self, vm: &VirtualMachine) -> PyResult<&'a Py<PyUtf8Str>> {
652 self.as_utf8()
653 .ok_or_else(|| self.ensure_valid_utf8(vm).unwrap_err())
654 }
655}
656
657impl PyStr {
658 fn __add__(zelf: PyRef<Self>, other: &PyObject, vm: &VirtualMachine) -> PyResult {
659 if let Some(other) = other.downcast_ref::<Self>() {
660 let bytes = zelf.as_wtf8().py_add(other.as_wtf8());
661 Ok(unsafe {
662 let kind = zelf.kind() | other.kind();
664 Self::new_str_unchecked(bytes.into(), kind)
665 }
666 .to_pyobject(vm))
667 } else {
668 Err(vm.new_type_error(format!(
669 r#"can only concatenate str (not "{}") to str"#,
670 other.class().slot_name()
671 )))
672 }
673 }
674
675 fn _contains(&self, needle: &PyObject, vm: &VirtualMachine) -> PyResult<bool> {
676 if let Some(needle) = needle.downcast_ref::<Self>() {
677 Ok(memchr::memmem::find(self.as_bytes(), needle.as_bytes()).is_some())
678 } else {
679 Err(vm.new_type_error(format!(
680 "'in <string>' requires string as left operand, not {}",
681 needle.class().slot_name()
682 )))
683 }
684 }
685
686 fn __contains__(&self, needle: &PyObject, vm: &VirtualMachine) -> PyResult<bool> {
687 self._contains(needle, vm)
688 }
689
690 fn _getitem(&self, needle: &PyObject, vm: &VirtualMachine) -> PyResult {
691 let item = match SequenceIndex::try_from_str_subscript(vm, needle)? {
692 SequenceIndex::Int(i) => self.getitem_by_index(vm, i)?.to_pyobject(vm),
693 SequenceIndex::Slice(slice) => self.getitem_by_slice(vm, slice)?.to_pyobject(vm),
694 };
695 Ok(item)
696 }
697
698 fn __getitem__(&self, needle: &PyObject, vm: &VirtualMachine) -> PyResult {
699 self._getitem(needle, vm)
700 }
701
702 #[inline]
703 pub(crate) fn hash(&self, vm: &VirtualMachine) -> hash::PyHash {
704 match self.hash.load(atomic::Ordering::Relaxed) {
705 hash::SENTINEL => self._compute_hash(vm),
706 hash => hash,
707 }
708 }
709
710 #[cold]
711 fn _compute_hash(&self, _vm: &VirtualMachine) -> hash::PyHash {
712 let hash_val = crate::vm::hash_secret().hash_bytes(self.as_bytes());
713 debug_assert_ne!(hash_val, hash::SENTINEL);
714 self.hash.store(hash_val, atomic::Ordering::Relaxed);
717 hash_val
718 }
719
720 #[inline]
721 pub fn byte_len(&self) -> usize {
722 self.data.len()
723 }
724
725 #[inline]
726 pub fn is_empty(&self) -> bool {
727 self.data.is_empty()
728 }
729
730 #[inline]
731 pub fn char_len(&self) -> usize {
732 self.data.char_len()
733 }
734
735 #[inline]
738 pub fn char_index_to_byte(&self, index: usize) -> usize {
739 self.data.char_index_to_byte(index)
740 }
741
742 #[inline]
745 pub fn byte_to_char_index(&self, bytepos: usize) -> usize {
746 self.data.byte_to_char_index(bytepos)
747 }
748
749 fn __mul__(zelf: PyRef<Self>, value: PySsize, vm: &VirtualMachine) -> PyResult<PyRef<Self>> {
750 Self::repeat(zelf, value, vm)
751 }
752
753 #[inline]
754 pub(crate) fn repr(&self, vm: &VirtualMachine) -> PyResult<String> {
755 use crate::literal::escape::UnicodeEscape;
756 UnicodeEscape::new_repr(self.as_wtf8())
757 .str_repr()
758 .to_string()
759 .ok_or_else(|| vm.new_overflow_error("string is too long to generate repr"))
760 }
761
762 fn result_unchanged(zelf: PyRef<Self>, vm: &VirtualMachine) -> PyRef<Self> {
764 if zelf.class().is(vm.ctx.types.str_type) {
765 zelf
766 } else {
767 vm.ctx.new_str(zelf.as_wtf8())
768 }
769 }
770
771 pub fn __mod__(&self, values: PyObjectRef, vm: &VirtualMachine) -> PyResult<Wtf8Buf> {
772 cformat_string(vm, self.as_wtf8(), &values)
773 }
774
775 fn join_items(
777 zelf: &Py<Self>,
778 items: &[PyObjectRef],
779 vm: &VirtualMachine,
780 ) -> PyResult<PyStrRef> {
781 fn item_str<'a>(
782 i: usize,
783 obj: &'a PyObject,
784 vm: &VirtualMachine,
785 ) -> PyResult<&'a Py<PyStr>> {
786 obj.downcast_ref::<PyStr>().ok_or_else(|| {
787 vm.new_type_error(format!(
788 "sequence item {i}: expected str instance, {} found",
789 obj.class().slot_name()
790 ))
791 })
792 }
793 let sep = zelf.as_wtf8();
794 let mut len = sep.len().saturating_mul(items.len().saturating_sub(1));
795 for (i, obj) in items.iter().enumerate() {
796 len = len.saturating_add(item_str(i, obj, vm)?.as_wtf8().len());
797 }
798 if let [only] = items {
799 let only = item_str(0, only, vm)?;
800 if only.class().is(vm.ctx.types.str_type) {
801 return Ok(only.to_owned());
802 }
803 }
804 let mut joined = Wtf8Buf::with_capacity(len);
805 for (i, obj) in items.iter().enumerate() {
806 if i > 0 {
807 joined.push_wtf8(sep);
808 }
809 joined.push_wtf8(item_str(i, obj, vm)?.as_wtf8());
810 }
811 Ok(vm.ctx.new_str(joined))
812 }
813
814 #[inline]
820 fn char_range_bytes(&self, range: Range<usize>) -> Option<(usize, &Wtf8)> {
821 if !range.is_normal() {
822 return None;
823 }
824 let bytes = self.data.char_range_to_bytes(range);
825 Some((bytes.start, &self.as_wtf8()[bytes]))
826 }
827
828 #[inline]
831 fn _find<F>(&self, args: FindArgs, find: F) -> Option<usize>
832 where
833 F: Fn(&Wtf8, &Wtf8) -> Option<usize>,
834 {
835 let (sub, range) = args.get_value(self.len());
836 let (start, haystack) = self.char_range_bytes(range)?;
837 let found = find(haystack, sub.as_wtf8())?;
838 Some(self.byte_to_char_index(start + found))
839 }
840
841 #[inline]
842 fn _pad(
843 &self,
844 width: isize,
845 fillchar: PyStrRef,
846 pad: fn(&Wtf8, usize, CodePoint, usize) -> Option<Wtf8Buf>,
847 vm: &VirtualMachine,
848 ) -> PyResult<Wtf8Buf> {
849 let fillchar = fillchar
850 .as_wtf8()
851 .code_points()
852 .exactly_one()
853 .map_err(|_| {
854 vm.new_type_error("The fill character must be exactly one character long")
855 })?;
856 if self.len() as isize >= width {
857 return Ok(self.as_wtf8().to_owned());
858 }
859 pad(self.as_wtf8(), width as usize, fillchar, self.len())
860 .ok_or_else(|| vm.no_memory_error())
861 }
862}
863
864#[pyclass(
865 flags(BASETYPE, _MATCH_SELF),
866 with(
867 AsMapping,
868 AsNumber,
869 AsSequence,
870 Representable,
871 Hashable,
872 Comparable,
873 Iterable,
874 Constructor
875 )
876)]
877impl Py<PyStr> {
878 #[pymethod]
879 #[inline(always)]
880 pub fn isascii(&self) -> bool {
881 matches!(self.kind(), StrKind::Ascii)
882 }
883
884 #[pymethod]
885 fn __sizeof__(&self) -> usize {
886 core::mem::size_of::<PyStr>() + self.byte_len() * core::mem::size_of::<u8>()
887 }
888
889 #[pymethod]
890 fn lower(&self) -> PyStr {
891 match self.as_str_kind() {
892 PyKindStr::Ascii(s) => s.to_ascii_lowercase().into(),
893 PyKindStr::Utf8(s) => s.to_lowercase().into(),
894 PyKindStr::Wtf8(w) => w.to_lowercase().into(),
895 }
896 }
897
898 #[pymethod]
904 fn casefold(&self) -> PyStr {
905 match self.as_str_kind() {
906 PyKindStr::Ascii(s) => s.to_ascii_lowercase().into(),
907 PyKindStr::Utf8(s) => unicode::case::casefold_str(s).into(),
908 PyKindStr::Wtf8(w) => unicode::case::casefold_wtf8(w).into(),
909 }
910 }
911
912 #[pymethod]
913 fn upper(&self) -> PyStr {
914 match self.as_str_kind() {
915 PyKindStr::Ascii(s) => s.to_ascii_uppercase().into(),
916 PyKindStr::Utf8(s) => s.to_uppercase().into(),
917 PyKindStr::Wtf8(w) => w.to_uppercase().into(),
918 }
919 }
920
921 #[pymethod]
922 fn capitalize(&self) -> Wtf8Buf {
923 match self.as_str_kind() {
924 PyKindStr::Ascii(s) => {
925 let mut s = s.to_owned();
926 if let [first, rest @ ..] = s.as_mut_slice() {
927 first.make_ascii_uppercase();
928 ascii::AsciiStr::make_ascii_lowercase(rest.into());
929 }
930 s.into()
931 }
932 PyKindStr::Utf8(s) => case::capitalize_str(s).into(),
933 PyKindStr::Wtf8(s) => case::capitalize_wtf8(s),
934 }
935 }
936
937 #[pymethod]
938 fn split(zelf: &Self, args: SplitArgs, vm: &VirtualMachine) -> PyResult<Vec<PyObjectRef>> {
939 let elements = match zelf.as_str_kind() {
940 PyKindStr::Ascii(s) => s.py_split(
941 args,
942 vm,
943 || zelf.as_object().to_owned(),
944 |v, s, vm| {
945 v.as_bytes()
946 .split_str(s)
947 .map(|s| unsafe { AsciiStr::from_ascii_unchecked(s) }.to_pyobject(vm))
948 .collect()
949 },
950 |v, s, n, vm| {
951 v.as_bytes()
952 .splitn_str(n, s)
953 .map(|s| unsafe { AsciiStr::from_ascii_unchecked(s) }.to_pyobject(vm))
954 .collect()
955 },
956 |v, n, vm| {
957 v.as_str().py_split_whitespace(n, |s| {
958 unsafe { AsciiStr::from_ascii_unchecked(s.as_bytes()) }.to_pyobject(vm)
959 })
960 },
961 ),
962 PyKindStr::Utf8(s) => s.py_split(
963 args,
964 vm,
965 || zelf.as_object().to_owned(),
966 |v, s, vm| v.split(s).map(|s| vm.ctx.new_str(s).into()).collect(),
967 |v, s, n, vm| v.splitn(n, s).map(|s| vm.ctx.new_str(s).into()).collect(),
968 |v, n, vm| v.py_split_whitespace(n, |s| vm.ctx.new_str(s).into()),
969 ),
970 PyKindStr::Wtf8(w) => w.py_split(
971 args,
972 vm,
973 || zelf.as_object().to_owned(),
974 |v, s, vm| v.split(s).map(|s| vm.ctx.new_str(s).into()).collect(),
975 |v, s, n, vm| v.splitn(n, s).map(|s| vm.ctx.new_str(s).into()).collect(),
976 |v, n, vm| v.py_split_whitespace(n, |s| vm.ctx.new_str(s).into()),
977 ),
978 }?;
979 Ok(elements)
980 }
981
982 #[pymethod]
983 fn rsplit(zelf: &Self, args: SplitArgs, vm: &VirtualMachine) -> PyResult<Vec<PyObjectRef>> {
984 let mut elements = zelf.as_wtf8().py_split(
985 args,
986 vm,
987 || zelf.as_object().to_owned(),
988 |v, s, vm| v.rsplit(s).map(|s| vm.ctx.new_str(s).into()).collect(),
989 |v, s, n, vm| v.rsplitn(n, s).map(|s| vm.ctx.new_str(s).into()).collect(),
990 |v, n, vm| v.py_rsplit_whitespace(n, |s| vm.ctx.new_str(s).into()),
991 )?;
992 elements.reverse();
995 Ok(elements)
996 }
997
998 #[pymethod]
999 fn strip(zelf: PyRef<PyStr>, args: StripArgs, vm: &VirtualMachine) -> PyRef<PyStr> {
1000 let chars = args.chars;
1001 let stripped: &Wtf8 = match zelf.as_str_kind() {
1002 PyKindStr::Ascii(s) if chars.as_ref().is_none_or(|c| c.kind().is_ascii()) => s
1003 .py_strip(
1004 chars,
1005 |s, chars| {
1006 let s = s
1007 .as_str()
1008 .trim_matches(|c| memchr::memchr(c as _, chars.as_bytes()).is_some());
1009 unsafe { AsciiStr::from_ascii_unchecked(s.as_bytes()) }
1010 },
1011 |s| {
1012 let s = s.as_str().trim_matches(unicode::classify::is_space);
1013 unsafe { AsciiStr::from_ascii_unchecked(s.as_bytes()) }
1014 },
1015 )
1016 .as_str()
1017 .into(),
1018 PyKindStr::Utf8(s) if chars.as_ref().is_none_or(|c| c.kind().is_utf8()) => s
1019 .py_strip(
1020 chars,
1021 |s, chars| s.trim_matches(|c| chars.contains(c)),
1022 |s| s.trim_matches(unicode::classify::is_space),
1023 )
1024 .into(),
1025 _ => zelf.as_wtf8().py_strip(
1026 chars,
1027 |s, chars| s.trim_matches(|c| chars.code_points().contains(&c)),
1028 |s| s.trim_matches(|c: CodePoint| c.is_char_and(unicode::classify::is_space)),
1029 ),
1030 };
1031 if zelf.byte_len() == stripped.len() {
1032 PyStr::result_unchanged(zelf, vm)
1033 } else {
1034 vm.ctx.new_str(zelf.new_substr(stripped.to_owned()))
1035 }
1036 }
1037
1038 #[pymethod]
1039 fn lstrip(zelf: PyRef<PyStr>, args: StripArgs, vm: &VirtualMachine) -> PyRef<PyStr> {
1040 let chars = args.chars;
1041 let s = zelf.as_wtf8();
1042 let stripped = s.py_strip(
1043 chars,
1044 |s, chars| s.trim_start_matches(|c| chars.contains_code_point(c)),
1045 |s| s.trim_start_matches(|c: CodePoint| c.is_char_and(unicode::classify::is_space)),
1046 );
1047 if s.len() == stripped.len() {
1048 PyStr::result_unchanged(zelf, vm)
1049 } else {
1050 vm.ctx.new_str(stripped)
1051 }
1052 }
1053
1054 #[pymethod]
1055 fn rstrip(zelf: PyRef<PyStr>, args: StripArgs, vm: &VirtualMachine) -> PyRef<PyStr> {
1056 let chars = args.chars;
1057 let s = zelf.as_wtf8();
1058 let stripped = s.py_strip(
1059 chars,
1060 |s, chars| s.trim_end_matches(|c| chars.contains_code_point(c)),
1061 |s| s.trim_end_matches(|c: CodePoint| c.is_char_and(unicode::classify::is_space)),
1062 );
1063 if s.len() == stripped.len() {
1064 PyStr::result_unchanged(zelf, vm)
1065 } else {
1066 vm.ctx.new_str(stripped)
1067 }
1068 }
1069
1070 #[pymethod]
1071 fn endswith(&self, options: anystr::StartsEndsWithArgs, vm: &VirtualMachine) -> PyResult<bool> {
1072 let (affix, substr) = match options.prepare(self.as_wtf8(), self.len(), |s, r| {
1073 &s[self.data.char_range_to_bytes(r)]
1074 }) {
1075 Some(x) => x,
1076 None => return Ok(false),
1077 };
1078 substr.py_starts_ends_with(
1079 &affix,
1080 "endswith",
1081 "str",
1082 |s, x: &Self| s.ends_with(x.as_wtf8()),
1083 vm,
1084 )
1085 }
1086
1087 #[pymethod]
1088 fn startswith(
1089 &self,
1090 options: anystr::StartsEndsWithArgs,
1091 vm: &VirtualMachine,
1092 ) -> PyResult<bool> {
1093 let (affix, substr) = match options.prepare(self.as_wtf8(), self.len(), |s, r| {
1094 &s[self.data.char_range_to_bytes(r)]
1095 }) {
1096 Some(x) => x,
1097 None => return Ok(false),
1098 };
1099 substr.py_starts_ends_with(
1100 &affix,
1101 "startswith",
1102 "str",
1103 |s, x: &Self| s.starts_with(x.as_wtf8()),
1104 vm,
1105 )
1106 }
1107
1108 #[pymethod]
1109 fn removeprefix(&self, prefix: PyStrRef) -> Wtf8Buf {
1110 self.as_wtf8()
1111 .py_removeprefix(prefix.as_wtf8(), prefix.byte_len(), |s, p| s.starts_with(p))
1112 .to_owned()
1113 }
1114
1115 #[pymethod]
1116 fn removesuffix(&self, suffix: PyStrRef) -> Wtf8Buf {
1117 self.as_wtf8()
1118 .py_removesuffix(suffix.as_wtf8(), suffix.byte_len(), |s, p| s.ends_with(p))
1119 .to_owned()
1120 }
1121
1122 #[pymethod]
1123 fn isalnum(&self) -> bool {
1124 !self.data.is_empty() && self.char_all(unicode::classify::is_alnum)
1125 }
1126
1127 #[pymethod]
1128 fn isnumeric(&self) -> bool {
1129 !self.data.is_empty() && self.char_all(unicode::classify::is_numeric)
1130 }
1131
1132 #[pymethod]
1133 fn isdigit(&self) -> bool {
1134 !self.data.is_empty() && self.char_all(unicode::classify::is_digit)
1135 }
1136
1137 #[pymethod]
1138 fn isdecimal(&self) -> bool {
1139 !self.data.is_empty() && self.char_all(unicode::classify::is_decimal)
1140 }
1141
1142 #[pymethod]
1143 fn format(&self, args: FuncArgs, vm: &VirtualMachine) -> PyResult<Wtf8Buf> {
1144 let format_str =
1145 FormatString::from_str(self.as_wtf8()).map_err(|e| e.to_pyexception(vm))?;
1146 format(&format_str, &args, vm)
1147 }
1148
1149 #[pymethod]
1150 fn format_map(&self, mapping: PyObjectRef, vm: &VirtualMachine) -> PyResult<Wtf8Buf> {
1151 let format_string =
1152 FormatString::from_str(self.as_wtf8()).map_err(|err| err.to_pyexception(vm))?;
1153 format_map(&format_string, &mapping, vm)
1154 }
1155
1156 #[pymethod]
1157 fn __format__(
1158 zelf: PyRef<PyStr>,
1159 format_spec: PyUtf8StrRef,
1160 vm: &VirtualMachine,
1161 ) -> PyResult<PyRef<PyStr>> {
1162 if format_spec.is_empty() {
1163 return if zelf.class().is(vm.ctx.types.str_type) {
1164 Ok(zelf)
1165 } else {
1166 zelf.as_object().str(vm)
1167 };
1168 }
1169 let zelf = zelf.try_into_utf8(vm)?;
1170 let s = FormatSpec::parse(format_spec.as_str())
1171 .and_then(|format_spec| {
1172 format_spec.format_string(&CharLenStr(zelf.as_str(), zelf.char_len()))
1173 })
1174 .map_err(|err| err.into_pyexception(vm))?;
1175 Ok(vm.ctx.new_str(s))
1176 }
1177
1178 #[pymethod]
1179 fn title(&self) -> Wtf8Buf {
1180 match self.as_str_kind() {
1181 PyKindStr::Ascii(_) => unsafe {
1182 Wtf8Buf::from_bytes_unchecked(title_ascii(self.as_bytes()))
1183 },
1184 PyKindStr::Utf8(s) => case::title_str(s).into(),
1185 PyKindStr::Wtf8(s) => case::title_wtf8(s),
1186 }
1187 }
1188
1189 #[pymethod]
1190 fn swapcase(&self) -> Wtf8Buf {
1191 match self.as_str_kind() {
1192 PyKindStr::Ascii(s) => unsafe {
1193 Wtf8Buf::from_bytes_unchecked(swapcase_ascii(s.as_bytes()))
1195 },
1196 PyKindStr::Utf8(s) => case::swapcase_str(s).into(),
1197 PyKindStr::Wtf8(s) => case::swapcase_wtf8(s),
1198 }
1199 }
1200
1201 #[pymethod]
1202 fn isalpha(&self) -> bool {
1203 !self.data.is_empty() && self.char_all(unicode::classify::is_alpha)
1204 }
1205
1206 #[pymethod]
1207 fn replace(zelf: PyRef<PyStr>, args: ReplaceArgs, vm: &VirtualMachine) -> PyRef<PyStr> {
1208 let ReplaceArgs { old, new, count } = args;
1209 if count == 0 || old.byte_len() > zelf.byte_len() || old.as_wtf8() == new.as_wtf8() {
1210 return PyStr::result_unchanged(zelf, vm);
1211 }
1212
1213 let s = zelf.as_wtf8();
1214 let replaced = if count < 0 {
1215 s.replace(old.as_wtf8(), new.as_wtf8())
1216 } else {
1217 let s_is_empty = s.is_empty();
1218 let old_is_empty = old.is_empty();
1219
1220 if s_is_empty && !old_is_empty {
1221 s.to_owned()
1222 } else if s_is_empty && old_is_empty {
1223 new.as_wtf8().to_owned()
1224 } else {
1225 s.replacen(old.as_wtf8(), new.as_wtf8(), count as usize)
1226 }
1227 };
1228 vm.ctx.new_str(replaced)
1229 }
1230
1231 #[pymethod]
1232 fn isprintable(&self) -> bool {
1233 self.char_all(unicode::classify::is_printable)
1234 }
1235
1236 #[pymethod]
1237 fn isspace(&self) -> bool {
1238 !self.data.is_empty() && self.char_all(unicode::classify::is_space)
1239 }
1240
1241 #[pymethod]
1243 fn islower(&self) -> bool {
1244 match self.as_str_kind() {
1245 PyKindStr::Ascii(s) => s.py_islower(),
1246 PyKindStr::Utf8(s) => s.py_islower(),
1247 PyKindStr::Wtf8(w) => w.py_islower(),
1248 }
1249 }
1250
1251 #[pymethod]
1253 fn isupper(&self) -> bool {
1254 match self.as_str_kind() {
1255 PyKindStr::Ascii(s) => s.py_isupper(),
1256 PyKindStr::Utf8(s) => s.py_isupper(),
1257 PyKindStr::Wtf8(w) => w.py_isupper(),
1258 }
1259 }
1260
1261 #[pymethod]
1262 pub(crate) fn splitlines(
1263 &self,
1264 args: anystr::SplitLinesArgs,
1265 vm: &VirtualMachine,
1266 ) -> Vec<PyObjectRef> {
1267 let into_wrapper = |s: &Wtf8| self.new_substr(s.to_owned()).to_pyobject(vm);
1268 let mut elements = Vec::new();
1269 let mut last_i = 0;
1270 let self_str = self.as_wtf8();
1271 let mut enumerated = self_str.code_point_indices().peekable();
1272 while let Some((i, ch)) = enumerated.next() {
1273 let end_len = match ch.to_char_lossy() {
1274 '\n' => 1,
1275 '\r' => {
1276 let is_rn = enumerated.next_if(|(_, ch)| *ch == '\n').is_some();
1277 if is_rn { 2 } else { 1 }
1278 }
1279 '\x0b' | '\x0c' | '\x1c' | '\x1d' | '\x1e' | '\u{0085}' | '\u{2028}'
1280 | '\u{2029}' => ch.len_wtf8(),
1281 _ => continue,
1282 };
1283 let range = if args.keepends {
1284 last_i..i + end_len
1285 } else {
1286 last_i..i
1287 };
1288 last_i = i + end_len;
1289 elements.push(into_wrapper(&self_str[range]));
1290 }
1291 if last_i != self_str.len() {
1292 elements.push(into_wrapper(&self_str[last_i..]));
1293 }
1294 elements
1295 }
1296
1297 #[pymethod]
1298 fn join(zelf: PyRef<PyStr>, iterable: PyObjectRef, vm: &VirtualMachine) -> PyResult<PyStrRef> {
1299 if let Some(list) = iterable.downcast_ref_if_exact::<PyList>(vm) {
1301 return PyStr::join_items(&zelf, &list.borrow_vec(), vm);
1302 }
1303 if let Some(tuple) = iterable.downcast_ref_if_exact::<PyTuple>(vm) {
1304 return PyStr::join_items(&zelf, tuple.as_slice(), vm);
1305 }
1306 let iterable = ArgIterable::<PyObjectRef>::try_from_object(vm, iterable)
1310 .map_err(|_| vm.new_type_error("can only join an iterable"))?;
1311 let items = iterable.iter_sized(vm)?.collect::<PyResult<Vec<_>>>()?;
1312 PyStr::join_items(&zelf, &items, vm)
1313 }
1314
1315 #[pymethod]
1316 fn find(&self, args: FindArgs) -> isize {
1317 self._find(args, Wtf8::find).map_or(-1, |v| v as isize)
1318 }
1319
1320 #[pymethod]
1321 fn rfind(&self, args: FindArgs) -> isize {
1322 self._find(args, Wtf8::rfind).map_or(-1, |v| v as isize)
1323 }
1324
1325 #[pymethod]
1326 fn index(&self, args: FindArgs, vm: &VirtualMachine) -> PyResult<usize> {
1327 self._find(args, Wtf8::find)
1328 .ok_or_else(|| vm.new_value_error("substring not found"))
1329 }
1330
1331 #[pymethod]
1332 fn rindex(&self, args: FindArgs, vm: &VirtualMachine) -> PyResult<usize> {
1333 self._find(args, Wtf8::rfind)
1334 .ok_or_else(|| vm.new_value_error("substring not found"))
1335 }
1336
1337 #[pymethod]
1338 pub fn partition(&self, sep: PyStrRef, vm: &VirtualMachine) -> PyResult {
1339 let (front, has_mid, back) = self.as_wtf8().py_partition(
1340 sep.as_wtf8(),
1341 || self.as_wtf8().splitn(2, sep.as_wtf8()),
1342 vm,
1343 )?;
1344 let partition = (
1345 self.new_substr(front),
1346 if has_mid {
1347 sep
1348 } else {
1349 vm.ctx.new_str(ascii!(""))
1350 },
1351 self.new_substr(back),
1352 );
1353 Ok(partition.to_pyobject(vm))
1354 }
1355
1356 #[pymethod]
1357 pub fn rpartition(&self, sep: PyStrRef, vm: &VirtualMachine) -> PyResult {
1358 let (back, has_mid, front) = self.as_wtf8().py_partition(
1359 sep.as_wtf8(),
1360 || self.as_wtf8().rsplitn(2, sep.as_wtf8()),
1361 vm,
1362 )?;
1363 Ok((
1364 self.new_substr(front),
1365 if has_mid {
1366 sep
1367 } else {
1368 vm.ctx.empty_str.to_owned()
1369 },
1370 self.new_substr(back),
1371 )
1372 .to_pyobject(vm))
1373 }
1374
1375 #[pymethod]
1376 fn istitle(&self) -> bool {
1377 if self.data.is_empty() {
1378 return false;
1379 }
1380
1381 let mut cased = false;
1382 let mut previous_is_cased = false;
1383 for c in self.as_wtf8().code_points().map(CodePoint::to_char_lossy) {
1384 if c.is_uppercase() || case::is_titlecase(c) {
1385 if previous_is_cased {
1386 return false;
1387 }
1388 previous_is_cased = true;
1389 cased = true;
1390 } else if c.is_lowercase() {
1391 if !previous_is_cased {
1392 return false;
1393 }
1394 previous_is_cased = true;
1395 cased = true;
1396 } else {
1397 previous_is_cased = false;
1398 }
1399 }
1400 cased
1401 }
1402
1403 #[pymethod]
1404 fn count(&self, args: FindArgs) -> usize {
1405 let (needle, range) = args.get_value(self.len());
1406 let chars = range.len();
1407 self.char_range_bytes(range).map_or(0, |(_, haystack)| {
1408 if needle.is_empty() {
1409 chars + 1
1414 } else {
1415 haystack.find_iter(needle.as_wtf8()).count()
1416 }
1417 })
1418 }
1419
1420 #[pymethod]
1421 fn zfill(&self, width: PySsize, vm: &VirtualMachine) -> PyResult<Wtf8Buf> {
1422 let filled = self
1423 .as_wtf8()
1424 .py_zfill(width)
1425 .ok_or_else(|| vm.no_memory_error())?;
1426 Ok(unsafe { Wtf8Buf::from_bytes_unchecked(filled) })
1428 }
1429
1430 #[pymethod]
1431 fn center(&self, args: PadArgs, vm: &VirtualMachine) -> PyResult<Wtf8Buf> {
1432 self._pad(args.width, args.fillchar, AnyStr::py_center, vm)
1433 }
1434
1435 #[pymethod]
1436 fn ljust(&self, args: PadArgs, vm: &VirtualMachine) -> PyResult<Wtf8Buf> {
1437 self._pad(args.width, args.fillchar, AnyStr::py_ljust, vm)
1438 }
1439
1440 #[pymethod]
1441 fn rjust(&self, args: PadArgs, vm: &VirtualMachine) -> PyResult<Wtf8Buf> {
1442 self._pad(args.width, args.fillchar, AnyStr::py_rjust, vm)
1443 }
1444
1445 #[pymethod]
1446 fn expandtabs(&self, args: anystr::ExpandTabsArgs) -> Wtf8Buf {
1447 rustpython_common::str::expandtabs(self.as_wtf8(), args.tabsize())
1448 }
1449
1450 #[pymethod]
1451 pub fn isidentifier(&self) -> bool {
1452 let Some(s) = self.to_str() else { return false };
1453 let mut chars = s.chars();
1454
1455 let is_identifier_start = chars.next().is_some_and(unicode::identifier::is_start);
1456
1457 is_identifier_start && chars.all(unicode::identifier::is_continue)
1459 }
1460
1461 #[pymethod]
1463 pub fn translate(&self, table: PyObjectRef, vm: &VirtualMachine) -> PyResult<Wtf8Buf> {
1464 let dict = table.downcast_ref_if_exact::<PyDict>(vm);
1465 let mut translated = Wtf8Buf::with_capacity(self.as_wtf8().len());
1466 for cp in self.as_wtf8().code_points() {
1467 let key = cp.to_u32().to_pyobject(vm);
1468 let value = match dict {
1470 Some(dict) => dict.get_item_opt(&*key, vm)?,
1471 None => match table.get_item(&*key, vm) {
1472 Ok(value) => Some(value),
1473 Err(e) if e.fast_isinstance(vm.ctx.exceptions.lookup_error) => None,
1474 Err(e) => return Err(e),
1475 },
1476 };
1477 let Some(value) = value else {
1478 translated.push(cp);
1479 continue;
1480 };
1481 if let Some(text) = value.downcast_ref::<PyStr>() {
1482 translated.push_wtf8(text.as_wtf8());
1483 } else if let Some(bigint) = value.downcast_ref::<PyInt>() {
1484 let mapped = bigint
1485 .as_bigint()
1486 .to_u32()
1487 .and_then(CodePoint::from_u32)
1488 .ok_or_else(|| {
1489 vm.new_value_error("character mapping must be in range(0x110000)")
1490 })?;
1491 translated.push(mapped);
1492 } else if !vm.is_none(&value) {
1493 return Err(vm.new_type_error("character mapping must return integer, None or str"));
1494 }
1495 }
1496 Ok(translated)
1497 }
1498
1499 #[pystaticmethod]
1500 fn maketrans(
1501 dict_or_str: PyObjectRef,
1502 to_str: OptionalArg<PyStrRef>,
1503 none_str: OptionalArg<PyStrRef>,
1504 vm: &VirtualMachine,
1505 ) -> PyResult {
1506 let new_dict = vm.ctx.new_dict();
1507 if let OptionalArg::Present(to_str) = to_str {
1508 match dict_or_str.downcast::<PyStr>() {
1509 Ok(from_str) => {
1510 if to_str.len() == from_str.len() {
1511 for (c1, c2) in from_str
1512 .as_wtf8()
1513 .code_points()
1514 .zip(to_str.as_wtf8().code_points())
1515 {
1516 new_dict.set_item(
1517 &*vm.new_pyobj(c1.to_u32()),
1518 vm.new_pyobj(c2.to_u32()),
1519 vm,
1520 )?;
1521 }
1522 if let OptionalArg::Present(none_str) = none_str {
1523 for c in none_str.as_wtf8().code_points() {
1524 new_dict.set_item(&*vm.new_pyobj(c.to_u32()), vm.ctx.none(), vm)?;
1525 }
1526 }
1527 Ok(new_dict.to_pyobject(vm))
1528 } else {
1529 Err(vm.new_value_error(
1530 "the first two maketrans arguments must have equal length",
1531 ))
1532 }
1533 }
1534 _ => Err(vm.new_type_error(
1535 "first maketrans argument must be a string if there is a second argument",
1536 )),
1537 }
1538 } else {
1539 match dict_or_str.downcast::<PyDict>() {
1541 Ok(dict) => {
1542 for (key, val) in dict {
1543 if let Some(num) = key.downcast_ref::<PyInt>() {
1545 new_dict.set_item(
1546 &*num.as_bigint().to_i32().to_pyobject(vm),
1547 val,
1548 vm,
1549 )?;
1550 } else if let Some(string) = key.downcast_ref::<PyStr>() {
1551 if string.len() == 1 {
1552 let num_value =
1553 string.as_wtf8().code_points().next().unwrap().to_u32();
1554 new_dict.set_item(&*num_value.to_pyobject(vm), val, vm)?;
1555 } else {
1556 return Err(vm.new_value_error(
1557 "string keys in translate table must be of length 1",
1558 ));
1559 }
1560 } else {
1561 return Err(vm.new_type_error(
1562 "keys in translate table must be strings or integers",
1563 ));
1564 }
1565 }
1566 Ok(new_dict.to_pyobject(vm))
1567 }
1568 _ => Err(vm.new_value_error(
1569 "if you give only one argument to maketrans it must be a dict",
1570 )),
1571 }
1572 }
1573 }
1574
1575 #[pymethod]
1576 fn encode(zelf: PyRef<PyStr>, args: EncodeArgs, vm: &VirtualMachine) -> PyResult<PyBytesRef> {
1577 encode_string(zelf, args.encoding.as_deref(), args.errors, vm)
1578 }
1579
1580 #[pymethod]
1581 fn __getnewargs__(zelf: PyRef<PyStr>, vm: &VirtualMachine) -> PyObjectRef {
1582 (zelf.as_wtf8(),).to_pyobject(vm)
1583 }
1584
1585 #[pymethod]
1586 fn __str__(zelf: &Self, vm: &VirtualMachine) -> PyStrRef {
1587 if zelf.class().is(vm.ctx.types.str_type) {
1588 zelf.to_owned()
1590 } else {
1591 PyStr::from(zelf.data.clone()).into_ref(&vm.ctx)
1593 }
1594 }
1595}
1596
1597impl PyRef<PyStr> {
1598 #[must_use]
1599 pub fn is_empty(&self) -> bool {
1600 (**self).is_empty()
1601 }
1602
1603 pub fn concat_in_place(&mut self, other: &Wtf8, vm: &VirtualMachine) {
1604 if other.is_empty() {
1605 return;
1606 }
1607 let mut s = Wtf8Buf::with_capacity(self.byte_len() + other.len());
1608 s.push_wtf8(self.as_ref());
1609 s.push_wtf8(other);
1610 if self.as_object().strong_count() == 1 {
1611 unsafe {
1614 let payload = self.payload() as *const PyStr as *mut PyStr;
1615 (*payload).data = PyStr::from(s).data;
1616 (*payload)
1617 .hash
1618 .store(hash::SENTINEL, atomic::Ordering::Relaxed);
1619 }
1620 } else {
1621 *self = PyStr::from(s).into_ref(&vm.ctx);
1622 }
1623 }
1624
1625 pub fn try_into_utf8(self, vm: &VirtualMachine) -> PyResult<PyRef<PyUtf8Str>> {
1626 self.ensure_valid_utf8(vm)?;
1627 Ok(unsafe { mem::transmute::<Self, PyRef<PyUtf8Str>>(self) })
1628 }
1629}
1630
1631struct CharLenStr<'a>(&'a str, usize);
1632impl core::ops::Deref for CharLenStr<'_> {
1633 type Target = str;
1634
1635 fn deref(&self) -> &Self::Target {
1636 self.0
1637 }
1638}
1639impl crate::common::format::CharLen for CharLenStr<'_> {
1640 fn char_len(&self) -> usize {
1641 self.1
1642 }
1643}
1644
1645impl Representable for PyStr {
1646 #[inline]
1647 fn repr_str(zelf: &Py<Self>, vm: &VirtualMachine) -> PyResult<String> {
1648 zelf.repr(vm)
1649 }
1650}
1651
1652impl Hashable for PyStr {
1653 #[inline]
1654 fn hash(zelf: &Py<Self>, vm: &VirtualMachine) -> PyResult<hash::PyHash> {
1655 Ok(zelf.hash(vm))
1656 }
1657}
1658
1659impl Comparable for PyStr {
1660 fn cmp(
1661 zelf: &Py<Self>,
1662 other: &PyObject,
1663 op: PyComparisonOp,
1664 _vm: &VirtualMachine,
1665 ) -> PyResult<PyComparisonValue> {
1666 if let Some(res) = op.identical_optimization(zelf, other) {
1667 return Ok(res.into());
1668 }
1669 let other = class_or_notimplemented!(Self, other);
1670 if let Some(res) = op.eval_eq(|| zelf.as_wtf8() == other.as_wtf8()) {
1673 return Ok(res.into());
1674 }
1675 Ok(op.eval_ord(zelf.as_wtf8().cmp(other.as_wtf8())).into())
1676 }
1677}
1678
1679impl Iterable for PyStr {
1680 fn iter(zelf: PyRef<Self>, vm: &VirtualMachine) -> PyResult {
1681 Ok(PyStrIterator {
1682 internal: PyMutex::new((PositionIterInternal::new(zelf, 0), 0)),
1683 }
1684 .into_pyobject(vm))
1685 }
1686}
1687
1688impl AsMapping for PyStr {
1689 fn as_mapping() -> &'static PyMappingMethods {
1690 static AS_MAPPING: LazyLock<PyMappingMethods> = LazyLock::new(|| PyMappingMethods {
1691 length: atomic_func!(|mapping, _vm| Ok(PyStr::mapping_downcast(mapping).len())),
1692 subscript: atomic_func!(
1693 |mapping, needle, vm| PyStr::mapping_downcast(mapping)._getitem(needle, vm)
1694 ),
1695 ..PyMappingMethods::NOT_IMPLEMENTED
1696 });
1697 &AS_MAPPING
1698 }
1699}
1700
1701impl AsNumber for PyStr {
1702 fn as_number() -> &'static PyNumberMethods {
1703 static AS_NUMBER: PyNumberMethods = PyNumberMethods {
1704 remainder: Some(|a, b, vm| {
1705 if let Some(a) = a.downcast_ref::<PyStr>() {
1706 a.__mod__(b.to_owned(), vm).to_pyresult(vm)
1707 } else {
1708 Ok(vm.ctx.not_implemented())
1709 }
1710 }),
1711 ..PyNumberMethods::NOT_IMPLEMENTED
1712 };
1713 &AS_NUMBER
1714 }
1715}
1716
1717impl AsSequence for PyStr {
1718 fn as_sequence() -> &'static PySequenceMethods {
1719 static AS_SEQUENCE: LazyLock<PySequenceMethods> = LazyLock::new(|| PySequenceMethods {
1720 length: atomic_func!(|seq, _vm| Ok(PyStr::sequence_downcast(seq).len())),
1721 concat: atomic_func!(|seq, other, vm| {
1722 let zelf = PyStr::sequence_downcast(seq);
1723 PyStr::__add__(zelf.to_owned(), other, vm)
1724 }),
1725 repeat: atomic_func!(|seq, n, vm| {
1726 let zelf = PyStr::sequence_downcast(seq);
1727 PyStr::repeat(zelf.to_owned(), n, vm).map(|x| x.into())
1728 }),
1729 item: atomic_func!(|seq, i, vm| {
1730 let zelf = PyStr::sequence_downcast(seq);
1731 zelf.getitem_by_index(vm, i).to_pyresult(vm)
1732 }),
1733 contains: atomic_func!(
1734 |seq, needle, vm| PyStr::sequence_downcast(seq)._contains(needle, vm)
1735 ),
1736 ..PySequenceMethods::NOT_IMPLEMENTED
1737 });
1738 &AS_SEQUENCE
1739 }
1740}
1741
1742#[derive(FromArgs)]
1743struct EncodeArgs {
1744 #[pyarg(any, optional, py_default = "'utf-8'")]
1746 encoding: Option<PyUtf8StrRef>,
1747 #[pyarg(any, optional, py_default = "'strict'")]
1749 errors: Option<PyUtf8StrRef>,
1750}
1751
1752#[derive(FromArgs)]
1753struct StripArgs {
1754 #[pyarg(positional, optional)]
1755 chars: Option<PyStrRef>,
1756}
1757
1758#[derive(FromArgs)]
1759struct PadArgs {
1760 #[pyarg(positional)]
1761 width: PySsize,
1762 #[pyarg(positional, default = " ")]
1763 fillchar: PyStrRef,
1764}
1765
1766pub(crate) fn encode_string(
1767 s: PyStrRef,
1768 encoding: Option<&Py<PyUtf8Str>>,
1769 errors: Option<PyUtf8StrRef>,
1770 vm: &VirtualMachine,
1771) -> PyResult<PyBytesRef> {
1772 let encoding = match encoding {
1773 None => crate::codecs::DEFAULT_ENCODING,
1774 Some(s) => s.as_str(),
1775 };
1776 vm.state.codec_registry.encode_text(s, encoding, errors, vm)
1777}
1778
1779impl PyPayload for PyStr {
1780 #[inline]
1781 fn class(ctx: &Context) -> &'static Py<PyType> {
1782 ctx.types.str_type
1783 }
1784}
1785
1786impl ToPyObject for String {
1787 fn to_pyobject(self, vm: &VirtualMachine) -> PyObjectRef {
1788 vm.ctx.new_str(self).into()
1789 }
1790}
1791
1792impl ToPyObject for Wtf8Buf {
1793 fn to_pyobject(self, vm: &VirtualMachine) -> PyObjectRef {
1794 vm.ctx.new_str(self).into()
1795 }
1796}
1797
1798impl ToPyObject for char {
1799 fn to_pyobject(self, vm: &VirtualMachine) -> PyObjectRef {
1800 let cp = self as u32;
1801 u8::try_from(cp).map_or_else(
1802 |_| vm.ctx.new_str(self).into(),
1803 |v| vm.ctx.latin1_char(v).into(),
1804 )
1805 }
1806}
1807
1808impl ToPyObject for CodePoint {
1809 fn to_pyobject(self, vm: &VirtualMachine) -> PyObjectRef {
1810 let cp = self.to_u32();
1811 u8::try_from(cp).map_or_else(
1812 |_| vm.ctx.new_str(self).into(),
1813 |v| vm.ctx.latin1_char(v).into(),
1814 )
1815 }
1816}
1817
1818impl ToPyObject for &str {
1819 fn to_pyobject(self, vm: &VirtualMachine) -> PyObjectRef {
1820 vm.ctx.new_str(self).into()
1821 }
1822}
1823
1824impl ToPyObject for &String {
1825 fn to_pyobject(self, vm: &VirtualMachine) -> PyObjectRef {
1826 vm.ctx.new_str(self.clone()).into()
1827 }
1828}
1829
1830impl ToPyObject for &CStr {
1831 fn to_pyobject(self, vm: &VirtualMachine) -> PyObjectRef {
1832 let s = self.to_str().expect("ToPyObject expects utf-8 CStr");
1833 vm.ctx.new_str(s).into()
1834 }
1835}
1836
1837impl ToPyObject for &Wtf8 {
1838 fn to_pyobject(self, vm: &VirtualMachine) -> PyObjectRef {
1839 vm.ctx.new_str(self).into()
1840 }
1841}
1842
1843impl ToPyObject for &Wtf8Buf {
1844 fn to_pyobject(self, vm: &VirtualMachine) -> PyObjectRef {
1845 vm.ctx.new_str(self.clone()).into()
1846 }
1847}
1848
1849impl ToPyObject for &AsciiStr {
1850 fn to_pyobject(self, vm: &VirtualMachine) -> PyObjectRef {
1851 vm.ctx.new_str(self).into()
1852 }
1853}
1854
1855impl ToPyObject for AsciiString {
1856 fn to_pyobject(self, vm: &VirtualMachine) -> PyObjectRef {
1857 vm.ctx.new_str(self).into()
1858 }
1859}
1860
1861impl ToPyObject for AsciiChar {
1862 fn to_pyobject(self, vm: &VirtualMachine) -> PyObjectRef {
1863 vm.ctx.latin1_char(u8::from(self)).into()
1864 }
1865}
1866
1867type SplitArgs = anystr::SplitArgs<PyStrRef>;
1868
1869#[derive(FromArgs)]
1870pub(crate) struct FindArgs {
1871 #[pyarg(positional)]
1872 sub: PyStrRef,
1873 #[pyarg(positional, default)]
1874 start: Option<PyIntRef>,
1875 #[pyarg(positional, default)]
1876 end: Option<PyIntRef>,
1877}
1878
1879impl FindArgs {
1880 fn get_value(self, len: usize) -> (PyStrRef, core::ops::Range<usize>) {
1881 let range = adjust_indices(self.start.as_deref(), self.end.as_deref(), len);
1882 (self.sub, range)
1883 }
1884}
1885
1886#[derive(FromArgs)]
1887struct ReplaceArgs {
1888 #[pyarg(positional)]
1889 old: PyStrRef,
1890
1891 #[pyarg(positional)]
1892 new: PyStrRef,
1893
1894 #[pyarg(any, default = -1)]
1895 count: isize,
1896}
1897
1898fn vectorcall_str(
1899 zelf_obj: &PyObject,
1900 args: Vec<PyObjectRef>,
1901 nargs: usize,
1902 kwnames: Option<&[PyObjectRef]>,
1903 vm: &VirtualMachine,
1904) -> PyResult {
1905 let zelf: &Py<PyType> = zelf_obj.downcast_ref().unwrap();
1906 let func_args = FuncArgs::from_vectorcall_owned(args, nargs, kwnames);
1907 (zelf.slots.new.load().unwrap())(zelf.to_owned(), func_args, vm)
1908}
1909
1910pub(crate) fn init(ctx: &'static Context) {
1911 PyStr::extend_class(ctx, ctx.types.str_type);
1912 ctx.types
1913 .str_type
1914 .slots
1915 .vectorcall
1916 .store(Some(vectorcall_str));
1917
1918 PyStrIterator::extend_class(ctx, ctx.types.str_iterator_type);
1919}
1920
1921impl PyStr {
1922 fn gather_chars(&self, indices: impl ExactSizeIterator<Item = usize>) -> Self {
1929 let char_len = indices.len();
1930 let mut out = Wtf8Buf::with_capacity(2 * char_len);
1932 let s = self.as_wtf8();
1933 for index in indices {
1934 out.push(
1935 s[self.data.char_index_to_byte(index)..]
1936 .code_points()
1937 .next()
1938 .expect("index is below the character count"),
1939 );
1940 }
1941 unsafe { Self::new_with_char_len(out, char_len) }
1943 }
1944}
1945
1946impl SliceableSequenceOp for PyStr {
1947 type Item = CodePoint;
1948 type Sliced = Self;
1949
1950 fn do_get(&self, index: usize) -> Self::Item {
1951 self.data.nth_char(index)
1952 }
1953
1954 fn getitem_by_index(&self, vm: &VirtualMachine, index: isize) -> PyResult<Self::Item> {
1955 let pos = self
1956 .wrap_index(index)
1957 .ok_or_else(|| vm.new_index_error("string index out of range"))?;
1958 Ok(self.do_get(pos))
1959 }
1960
1961 fn do_slice(&self, range: Range<usize>) -> Self::Sliced {
1962 if let PyKindStr::Ascii(s) = self.as_str_kind() {
1963 return s[range].into();
1964 }
1965 let char_len = range.len();
1969 let bytes = self.data.char_range_to_bytes(range);
1970 let out = &self.as_wtf8()[bytes];
1971 unsafe { Self::new_with_char_len(out.to_owned(), char_len) }
1973 }
1974
1975 fn do_slice_reverse(&self, range: Range<usize>) -> Self::Sliced {
1976 if let PyKindStr::Ascii(s) = self.as_str_kind() {
1977 let mut out = s[range].to_owned();
1978 out.as_mut_slice().reverse();
1979 return out.into();
1980 }
1981 let char_len = range.len();
1982 let bytes = self.data.char_range_to_bytes(range);
1983 let mut out = Wtf8Buf::with_capacity(bytes.len());
1984 out.extend(self.as_wtf8()[bytes].code_points().rev());
1985 unsafe { Self::new_with_char_len(out, char_len) }
1987 }
1988
1989 fn do_stepped_slice(&self, range: Range<usize>, step: usize) -> Self::Sliced {
1990 if let PyKindStr::Ascii(s) = self.as_str_kind() {
1991 return s[range]
1992 .as_slice()
1993 .iter()
1994 .copied()
1995 .step_by(step)
1996 .collect::<AsciiString>()
1997 .into();
1998 }
1999 self.gather_chars(range.step_by(step))
2000 }
2001
2002 fn do_stepped_slice_reverse(&self, range: Range<usize>, step: usize) -> Self::Sliced {
2003 if let PyKindStr::Ascii(s) = self.as_str_kind() {
2004 return s[range]
2005 .chars()
2006 .rev()
2007 .step_by(step)
2008 .collect::<AsciiString>()
2009 .into();
2010 }
2011 self.gather_chars(range.rev().step_by(step))
2012 }
2013
2014 fn empty() -> Self::Sliced {
2015 Self::default()
2016 }
2017
2018 fn len(&self) -> usize {
2019 self.char_len()
2020 }
2021}
2022
2023impl AsRef<str> for PyRefExact<PyStr> {
2024 #[track_caller]
2025 fn as_ref(&self) -> &str {
2026 self.to_str().expect("str has surrogates")
2027 }
2028}
2029
2030impl AsRef<str> for PyExact<PyStr> {
2031 #[track_caller]
2032 fn as_ref(&self) -> &str {
2033 self.to_str().expect("str has surrogates")
2034 }
2035}
2036
2037impl AsRef<Wtf8> for PyRefExact<PyStr> {
2038 fn as_ref(&self) -> &Wtf8 {
2039 self.as_wtf8()
2040 }
2041}
2042
2043impl AsRef<Wtf8> for PyExact<PyStr> {
2044 fn as_ref(&self) -> &Wtf8 {
2045 self.as_wtf8()
2046 }
2047}
2048
2049impl AnyStrWrapper<Wtf8> for PyStrRef {
2050 fn as_ref(&self) -> Option<&Wtf8> {
2051 Some(self.as_wtf8())
2052 }
2053
2054 fn is_empty(&self) -> bool {
2055 self.data.is_empty()
2056 }
2057}
2058
2059impl AnyStrWrapper<str> for PyStrRef {
2060 fn as_ref(&self) -> Option<&str> {
2061 self.data.as_str()
2062 }
2063
2064 fn is_empty(&self) -> bool {
2065 self.data.is_empty()
2066 }
2067}
2068
2069impl AnyStrWrapper<AsciiStr> for PyStrRef {
2070 fn as_ref(&self) -> Option<&AsciiStr> {
2071 self.data.as_ascii()
2072 }
2073
2074 fn is_empty(&self) -> bool {
2075 self.data.is_empty()
2076 }
2077}
2078
2079#[repr(transparent)]
2080#[derive(Debug)]
2081pub struct PyUtf8Str(PyStr);
2082
2083impl fmt::Display for PyUtf8Str {
2084 #[inline]
2085 fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result {
2086 self.0.fmt(f)
2087 }
2088}
2089
2090impl MaybeTraverse for PyUtf8Str {
2091 const HAS_TRAVERSE: bool = true;
2092 const HAS_CLEAR: bool = false;
2093
2094 fn try_traverse(&self, traverse_fn: &mut TraverseFn<'_>) {
2095 self.0.try_traverse(traverse_fn);
2096 }
2097
2098 fn try_clear(&mut self, _out: &mut Vec<PyObjectRef>) {
2099 }
2101}
2102
2103impl PyPayload for PyUtf8Str {
2104 #[inline]
2105 fn class(ctx: &Context) -> &'static Py<PyType> {
2106 ctx.types.str_type
2107 }
2108
2109 const PAYLOAD_TYPE_ID: core::any::TypeId = core::any::TypeId::of::<PyStr>();
2110
2111 unsafe fn validate_downcastable_from(obj: &PyObject) -> bool {
2112 let wtf8 = unsafe { obj.downcast_unchecked_ref::<PyStr>() };
2114 wtf8.is_utf8()
2115 }
2116
2117 fn try_downcast_from(obj: &PyObject, vm: &VirtualMachine) -> PyResult<()> {
2118 let str = obj.try_downcast_ref::<PyStr>(vm)?;
2119 str.ensure_valid_utf8(vm)
2120 }
2121}
2122
2123impl<'a> From<&'a AsciiStr> for PyUtf8Str {
2124 fn from(s: &'a AsciiStr) -> Self {
2125 s.to_owned().into()
2126 }
2127}
2128
2129impl From<AsciiString> for PyUtf8Str {
2130 fn from(s: AsciiString) -> Self {
2131 s.into_boxed_ascii_str().into()
2132 }
2133}
2134
2135impl From<Box<AsciiStr>> for PyUtf8Str {
2136 fn from(s: Box<AsciiStr>) -> Self {
2137 let data = StrData::from(s);
2138 unsafe { Self::from_str_data_unchecked(data) }
2139 }
2140}
2141
2142impl From<AsciiChar> for PyUtf8Str {
2143 fn from(ch: AsciiChar) -> Self {
2144 AsciiString::from(ch).into()
2145 }
2146}
2147
2148impl<'a> From<&'a str> for PyUtf8Str {
2149 fn from(s: &'a str) -> Self {
2150 s.to_owned().into()
2151 }
2152}
2153
2154impl From<String> for PyUtf8Str {
2155 fn from(s: String) -> Self {
2156 s.into_boxed_str().into()
2157 }
2158}
2159
2160impl From<char> for PyUtf8Str {
2161 fn from(ch: char) -> Self {
2162 let data = StrData::from(ch);
2163 unsafe { Self::from_str_data_unchecked(data) }
2164 }
2165}
2166
2167impl<'a> From<alloc::borrow::Cow<'a, str>> for PyUtf8Str {
2168 fn from(s: alloc::borrow::Cow<'a, str>) -> Self {
2169 s.into_owned().into()
2170 }
2171}
2172
2173impl From<Box<str>> for PyUtf8Str {
2174 #[inline]
2175 fn from(value: Box<str>) -> Self {
2176 let data = StrData::from(value);
2177 unsafe { Self::from_str_data_unchecked(data) }
2178 }
2179}
2180
2181impl AsRef<Wtf8> for PyUtf8Str {
2182 #[inline]
2183 fn as_ref(&self) -> &Wtf8 {
2184 self.0.as_wtf8()
2185 }
2186}
2187
2188impl AsRef<str> for PyUtf8Str {
2189 #[inline]
2190 fn as_ref(&self) -> &str {
2191 self.as_str()
2192 }
2193}
2194
2195impl PyUtf8Str {
2196 unsafe fn from_str_data_unchecked(data: StrData) -> Self {
2200 Self(PyStr::from(data))
2201 }
2202
2203 #[inline]
2205 pub fn as_wtf8(&self) -> &Wtf8 {
2206 self.0.as_wtf8()
2207 }
2208
2209 pub fn as_str(&self) -> &str {
2211 debug_assert!(
2212 self.0.is_utf8(),
2213 "PyUtf8Str invariant violated: inner string is not valid UTF-8"
2214 );
2215 unsafe { self.0.to_str().unwrap_unchecked() }
2217 }
2218
2219 #[inline]
2220 pub fn as_bytes(&self) -> &[u8] {
2221 self.as_str().as_bytes()
2222 }
2223
2224 #[inline]
2225 pub fn byte_len(&self) -> usize {
2226 self.0.byte_len()
2227 }
2228
2229 #[inline]
2230 pub fn is_empty(&self) -> bool {
2231 self.0.is_empty()
2232 }
2233
2234 #[inline]
2235 pub fn char_len(&self) -> usize {
2236 self.0.char_len()
2237 }
2238}
2239
2240impl Py<PyUtf8Str> {
2241 pub const fn as_pystr(&self) -> &Py<PyStr> {
2243 unsafe {
2244 &*(self as *const Self as *const Py<PyStr>)
2246 }
2247 }
2248
2249 #[inline]
2251 pub fn as_str(&self) -> &str {
2252 self.as_pystr().to_str().unwrap_or_else(|| {
2253 debug_assert!(false, "PyUtf8Str invariant violated");
2254 unsafe { core::hint::unreachable_unchecked() }
2256 })
2257 }
2258}
2259
2260impl PyRef<PyUtf8Str> {
2261 #[must_use]
2263 pub fn into_wtf8(self) -> PyStrRef {
2264 unsafe { mem::transmute::<Self, PyStrRef>(self) }
2265 }
2266}
2267
2268impl From<PyRef<PyUtf8Str>> for PyRef<PyStr> {
2269 fn from(s: PyRef<PyUtf8Str>) -> Self {
2270 s.into_wtf8()
2271 }
2272}
2273
2274impl PartialEq for PyUtf8Str {
2275 fn eq(&self, other: &Self) -> bool {
2276 self.as_str() == other.as_str()
2277 }
2278}
2279impl Eq for PyUtf8Str {}
2280
2281impl AnyStrContainer<str> for String {
2282 fn new() -> Self {
2283 Self::new()
2284 }
2285
2286 fn with_capacity(capacity: usize) -> Self {
2287 Self::with_capacity(capacity)
2288 }
2289
2290 fn try_with_capacity(capacity: usize) -> Option<Self> {
2291 let mut s = Self::new();
2292 s.try_reserve_exact(capacity).ok()?;
2293 Some(s)
2294 }
2295
2296 fn push_str(&mut self, other: &str) {
2297 Self::push_str(self, other)
2298 }
2299}
2300
2301impl anystr::AnyChar for char {
2302 fn bytes_len(self) -> usize {
2303 self.len_utf8()
2304 }
2305}
2306
2307impl AnyStr for str {
2308 type Char = char;
2309 type Container = String;
2310
2311 fn to_container(&self) -> Self::Container {
2312 self.to_owned()
2313 }
2314
2315 fn as_bytes(&self) -> &[u8] {
2316 self.as_bytes()
2317 }
2318
2319 fn elements(&self) -> impl Iterator<Item = char> {
2320 Self::chars(self)
2321 }
2322
2323 fn get_bytes(&self, range: core::ops::Range<usize>) -> &Self {
2324 &self[range]
2325 }
2326
2327 fn get_chars(&self, range: core::ops::Range<usize>) -> &Self {
2328 rustpython_common::str::get_chars(self, range)
2329 }
2330
2331 fn is_empty(&self) -> bool {
2332 Self::is_empty(self)
2333 }
2334
2335 fn bytes_len(&self) -> usize {
2336 Self::len(self)
2337 }
2338
2339 fn py_split_whitespace<F>(&self, maxsplit: isize, convert: F) -> Vec<PyObjectRef>
2340 where
2341 F: Fn(&Self) -> PyObjectRef,
2342 {
2343 let mut splits = Vec::new();
2345 let mut last_offset = 0;
2346 let mut count = maxsplit;
2347 for (offset, separator) in self.match_indices(unicode::classify::is_space) {
2348 if last_offset == offset {
2349 last_offset += separator.len();
2350 continue;
2351 }
2352 if count == 0 {
2353 break;
2354 }
2355 splits.push(convert(&self[last_offset..offset]));
2356 last_offset = offset + separator.len();
2357 count -= 1;
2358 }
2359 if last_offset != self.len() {
2360 splits.push(convert(&self[last_offset..]));
2361 }
2362 splits
2363 }
2364
2365 fn py_rsplit_whitespace<F>(&self, maxsplit: isize, convert: F) -> Vec<PyObjectRef>
2366 where
2367 F: Fn(&Self) -> PyObjectRef,
2368 {
2369 let mut splits = Vec::new();
2371 let mut last_offset = self.len();
2372 let mut count = maxsplit;
2373 for (offset, separator) in self.rmatch_indices(unicode::classify::is_space) {
2374 if last_offset == offset + separator.len() {
2375 last_offset = offset;
2376 continue;
2377 }
2378 if count == 0 {
2379 break;
2380 }
2381 splits.push(convert(&self[offset + separator.len()..last_offset]));
2382 last_offset = offset;
2383 count -= 1;
2384 }
2385 if last_offset != 0 {
2386 splits.push(convert(&self[..last_offset]));
2387 }
2388 splits
2389 }
2390
2391 fn py_islower(&self) -> bool {
2392 self.is_cased(case::is_lowercase, case::is_uppercase)
2393 }
2394
2395 fn py_isupper(&self) -> bool {
2396 self.is_cased(case::is_uppercase, case::is_lowercase)
2397 }
2398}
2399
2400impl AnyStrContainer<Wtf8> for Wtf8Buf {
2401 fn new() -> Self {
2402 Self::new()
2403 }
2404
2405 fn with_capacity(capacity: usize) -> Self {
2406 Self::with_capacity(capacity)
2407 }
2408
2409 fn try_with_capacity(capacity: usize) -> Option<Self> {
2410 let mut s = Self::new();
2411 s.try_reserve_exact(capacity).ok()?;
2412 Some(s)
2413 }
2414
2415 fn push_str(&mut self, other: &Wtf8) {
2416 self.push_wtf8(other)
2417 }
2418}
2419
2420impl anystr::AnyChar for CodePoint {
2421 fn bytes_len(self) -> usize {
2422 self.len_wtf8()
2423 }
2424}
2425
2426impl AnyStr for Wtf8 {
2427 type Char = CodePoint;
2428 type Container = Wtf8Buf;
2429
2430 fn to_container(&self) -> Self::Container {
2431 self.to_owned()
2432 }
2433
2434 fn as_bytes(&self) -> &[u8] {
2435 self.as_bytes()
2436 }
2437
2438 fn elements(&self) -> impl Iterator<Item = Self::Char> {
2439 self.code_points()
2440 }
2441
2442 fn get_bytes(&self, range: core::ops::Range<usize>) -> &Self {
2443 &self[range]
2444 }
2445
2446 fn get_chars(&self, range: core::ops::Range<usize>) -> &Self {
2447 rustpython_common::str::get_codepoints(self, range)
2448 }
2449
2450 fn bytes_len(&self) -> usize {
2451 self.len()
2452 }
2453
2454 fn is_empty(&self) -> bool {
2455 self.is_empty()
2456 }
2457
2458 fn py_split_whitespace<F>(&self, maxsplit: isize, convert: F) -> Vec<PyObjectRef>
2459 where
2460 F: Fn(&Self) -> PyObjectRef,
2461 {
2462 let mut splits = Vec::new();
2464 let mut last_offset = 0;
2465 let mut count = maxsplit;
2466 for (offset, separator) in self
2467 .code_point_indices()
2468 .filter(|(_, c)| c.is_char_and(unicode::classify::is_space))
2469 {
2470 if last_offset == offset {
2471 last_offset += separator.len_wtf8();
2472 continue;
2473 }
2474 if count == 0 {
2475 break;
2476 }
2477 splits.push(convert(&self[last_offset..offset]));
2478 last_offset = offset + separator.len_wtf8();
2479 count -= 1;
2480 }
2481 if last_offset != self.len() {
2482 splits.push(convert(&self[last_offset..]));
2483 }
2484 splits
2485 }
2486
2487 fn py_rsplit_whitespace<F>(&self, maxsplit: isize, convert: F) -> Vec<PyObjectRef>
2488 where
2489 F: Fn(&Self) -> PyObjectRef,
2490 {
2491 let mut splits = Vec::new();
2493 let mut last_offset = self.len();
2494 let mut count = maxsplit;
2495 for (offset, separator) in self
2496 .code_point_indices()
2497 .rev()
2498 .filter(|(_, c)| c.is_char_and(unicode::classify::is_space))
2499 {
2500 if last_offset == offset + separator.len_wtf8() {
2501 last_offset = offset;
2502 continue;
2503 }
2504 if count == 0 {
2505 break;
2506 }
2507 splits.push(convert(&self[offset + separator.len_wtf8()..last_offset]));
2508 last_offset = offset;
2509 count -= 1;
2510 }
2511 if last_offset != 0 {
2512 splits.push(convert(&self[..last_offset]));
2513 }
2514 splits
2515 }
2516
2517 fn py_islower(&self) -> bool {
2518 self.is_cased(case::is_lowercase, case::is_uppercase)
2519 }
2520
2521 fn py_isupper(&self) -> bool {
2522 self.is_cased(case::is_uppercase, case::is_lowercase)
2523 }
2524}
2525
2526impl AnyStrContainer<AsciiStr> for AsciiString {
2527 fn new() -> Self {
2528 Self::new()
2529 }
2530
2531 fn with_capacity(capacity: usize) -> Self {
2532 Self::with_capacity(capacity)
2533 }
2534
2535 fn try_with_capacity(capacity: usize) -> Option<Self> {
2536 let mut v = Vec::new();
2537 v.try_reserve_exact(capacity).ok()?;
2538 Some(Self::from(v))
2539 }
2540
2541 fn push_str(&mut self, other: &AsciiStr) {
2542 Self::push_str(self, other)
2543 }
2544}
2545
2546impl anystr::AnyChar for ascii::AsciiChar {
2547 fn bytes_len(self) -> usize {
2548 1
2549 }
2550}
2551
2552const ASCII_WHITESPACES: [u8; 6] = [0x20, 0x09, 0x0a, 0x0c, 0x0d, 0x0b];
2553
2554impl AnyStr for AsciiStr {
2555 type Char = AsciiChar;
2556 type Container = AsciiString;
2557
2558 fn to_container(&self) -> Self::Container {
2559 self.to_ascii_string()
2560 }
2561
2562 fn as_bytes(&self) -> &[u8] {
2563 self.as_bytes()
2564 }
2565
2566 fn elements(&self) -> impl Iterator<Item = Self::Char> {
2567 self.chars()
2568 }
2569
2570 fn get_bytes(&self, range: core::ops::Range<usize>) -> &Self {
2571 &self[range]
2572 }
2573
2574 fn get_chars(&self, range: core::ops::Range<usize>) -> &Self {
2575 &self[range]
2576 }
2577
2578 fn bytes_len(&self) -> usize {
2579 self.len()
2580 }
2581
2582 fn is_empty(&self) -> bool {
2583 self.is_empty()
2584 }
2585
2586 fn py_split_whitespace<F>(&self, maxsplit: isize, convert: F) -> Vec<PyObjectRef>
2587 where
2588 F: Fn(&Self) -> PyObjectRef,
2589 {
2590 let mut splits = Vec::new();
2591 let mut count = maxsplit;
2592 let mut haystack = self;
2593 while let Some(offset) = haystack.as_bytes().find_byteset(ASCII_WHITESPACES) {
2594 if offset != 0 {
2595 if count == 0 {
2596 break;
2597 }
2598 splits.push(convert(&haystack[..offset]));
2599 count -= 1;
2600 }
2601 haystack = &haystack[offset + 1..];
2602 }
2603 if !haystack.is_empty() {
2604 splits.push(convert(haystack));
2605 }
2606 splits
2607 }
2608
2609 fn py_rsplit_whitespace<F>(&self, maxsplit: isize, convert: F) -> Vec<PyObjectRef>
2610 where
2611 F: Fn(&Self) -> PyObjectRef,
2612 {
2613 let mut splits = Vec::new();
2615 let mut count = maxsplit;
2616 let mut haystack = self;
2617 while let Some(offset) = haystack.as_bytes().rfind_byteset(ASCII_WHITESPACES) {
2618 if offset + 1 != haystack.len() {
2619 if count == 0 {
2620 break;
2621 }
2622 splits.push(convert(&haystack[offset + 1..]));
2623 count -= 1;
2624 }
2625 haystack = &haystack[..offset];
2626 }
2627 if !haystack.is_empty() {
2628 splits.push(convert(haystack));
2629 }
2630 splits
2631 }
2632}
2633
2634pub type PyStrInterned = PyInterned<PyStr>;
2637
2638impl PyStrInterned {
2639 #[inline]
2640 pub fn to_exact(&'static self) -> PyRefExact<PyStr> {
2641 unsafe { PyRefExact::new_unchecked(self.to_owned()) }
2642 }
2643}
2644
2645impl core::fmt::Display for PyStrInterned {
2646 fn fmt(&self, f: &mut core::fmt::Formatter<'_>) -> core::fmt::Result {
2647 self.data.fmt(f)
2648 }
2649}
2650
2651impl AsRef<str> for PyStrInterned {
2652 #[inline(always)]
2653 fn as_ref(&self) -> &str {
2654 self.to_str()
2655 .expect("Interned PyStr should always be valid UTF-8")
2656 }
2657}
2658
2659pub type PyUtf8StrInterned = PyInterned<PyUtf8Str>;
2663
2664impl PyUtf8StrInterned {
2665 #[inline]
2667 pub fn as_str(&self) -> &str {
2668 Py::<PyUtf8Str>::as_str(self)
2669 }
2670
2671 #[inline]
2673 pub fn as_interned_str(&self) -> &PyStrInterned {
2674 unsafe { &*(self as *const Self as *const PyStrInterned) }
2677 }
2678
2679 #[inline]
2684 pub unsafe fn from_str_interned_unchecked(s: &PyStrInterned) -> &Self {
2685 unsafe { &*(s as *const PyStrInterned as *const Self) }
2686 }
2687}
2688
2689impl core::fmt::Display for PyUtf8StrInterned {
2690 fn fmt(&self, f: &mut core::fmt::Formatter<'_>) -> core::fmt::Result {
2691 f.write_str(self.as_str())
2692 }
2693}
2694
2695impl AsRef<str> for PyUtf8StrInterned {
2696 #[inline(always)]
2697 fn as_ref(&self) -> &str {
2698 self.as_str()
2699 }
2700}
2701
2702#[cfg(test)]
2703mod tests {
2704 use super::*;
2705 use crate::{Context, Interpreter, Py};
2706 use rustpython_common::wtf8::Wtf8Buf;
2707
2708 #[test]
2709 fn str_title() {
2710 let tests = vec![
2711 (" Hello ", " hello "),
2712 ("Hello ", "hello "),
2713 ("Hello ", "Hello "),
2714 ("Format This As Title String", "fOrMaT thIs aS titLe String"),
2715 ("Format,This-As*Title;String", "fOrMaT,thIs-aS*titLe;String"),
2716 ("Getint", "getInt"),
2717 ("Greek Ωppercases ...", "greek ωppercases ..."),
2719 ("Greek ῼitlecases ...", "greek ῳitlecases ..."),
2721 ("\u{01F2}", "\u{01F1}"),
2724 ("\u{01C5}", "\u{01C4}"),
2725 ];
2726 for (title, input) in tests {
2727 assert_eq!(
2728 Context::genesis().new_str(input).title().as_str(),
2729 Ok(title)
2730 );
2731 }
2732 }
2733
2734 #[test]
2735 fn str_istitle() {
2736 let pos = vec![
2737 "A",
2738 "A Titlecased Line",
2739 "A\nTitlecased Line",
2740 "A Titlecased, Line",
2741 "Greek Ωppercases ...",
2743 "Greek ῼitlecases ...",
2745 ];
2746
2747 for s in pos {
2748 assert!(Context::genesis().new_str(s).istitle());
2749 }
2750
2751 let neg = vec![
2752 "",
2753 "a",
2754 "\n",
2755 "Not a capitalized String",
2756 "Not\ta Titlecase String",
2757 "Not--a Titlecase String",
2758 "NOT",
2759 ];
2760 for s in neg {
2761 assert!(!Context::genesis().new_str(s).istitle());
2762 }
2763 }
2764
2765 #[test]
2766 fn str_maketrans_and_translate() {
2767 Interpreter::without_stdlib(Default::default()).enter(|vm| {
2768 let table = vm.ctx.new_dict();
2769 table
2770 .set_item("a", vm.ctx.new_str("🎅").into(), vm)
2771 .unwrap();
2772 table.set_item("b", vm.ctx.none(), vm).unwrap();
2773 table
2774 .set_item("c", vm.ctx.new_str(ascii!("xda")).into(), vm)
2775 .unwrap();
2776 let translated = Py::<PyStr>::maketrans(
2777 table.into(),
2778 OptionalArg::Missing,
2779 OptionalArg::Missing,
2780 vm,
2781 )
2782 .unwrap();
2783 let text = vm.ctx.new_str("abc");
2784 let translated = text.translate(translated, vm).unwrap();
2785 assert_eq!(translated, Wtf8Buf::from("🎅xda"));
2786 let translated = text.translate(vm.ctx.new_int(3).into(), vm);
2787 assert_eq!("TypeError", &*translated.unwrap_err().class().name(),);
2788 })
2789 }
2790
2791 #[test]
2792 fn str_isprintable_unicode15() {
2793 assert!(Context::genesis().new_str("\u{0B55}").isprintable());
2800 assert!(Context::genesis().new_str("A").isprintable());
2801 assert!(Context::genesis().new_str(" ").isprintable());
2802 assert!(Context::genesis().new_str("").isprintable());
2803
2804 assert!(!Context::genesis().new_str("\x00").isprintable());
2806 assert!(!Context::genesis().new_str("\u{200B}").isprintable());
2807 assert!(!Context::genesis().new_str("\u{E000}").isprintable());
2808 assert!(!Context::genesis().new_str("\u{00A0}").isprintable());
2809 }
2810}