Skip to main content

samp_sdk/
encoding.rs

1//! Global encoding for Rust <-> AMX conversion (only with the `encoding` feature).
2//!
3//! The original SA-MP operates on 8-bit encodings (Western Windows-1252 by
4//! default, Windows-1251 for Cyrillic on Russian servers). This module lets the
5//! plugin configure the encoding once in `on_load` — after that, `AmxString`
6//! decodes and [`Buffer::write_str`] encodes using it automatically.
7//!
8//! [`set_default_encoding`] takes any `&'static Encoding`, so every encoding
9//! `encoding_rs` implements is available; the ones SA-MP and Open Multiplayer
10//! servers actually run are re-exported here, and
11//! [`set_default_encoding_by_label`] resolves one by name for a plugin that
12//! reads it from configuration.
13//!
14//! ## Multi-byte encodings
15//!
16//! Pawn assumes one cell holds one character. That holds for every 8-bit
17//! encoding above, and **not** for UTF-8, GBK, Big5, Shift_JIS or EUC-KR, where
18//! one character may take several bytes. Converting in and out of Rust stays
19//! correct, but on the Pawn side `strlen` then counts bytes rather than
20//! characters and `text[3]` is the fourth byte, not the fourth character. Use a
21//! multi-byte encoding only when the script is written for it.
22//!
23//! [`Buffer::write_str`]: crate::cell::Buffer::write_str
24
25use encoding_rs::Encoding;
26use std::sync::atomic::{AtomicPtr, Ordering};
27
28// Re-exported so a plugin does not have to depend on `encoding_rs` directly to
29// name one. `set_default_encoding` takes any `&'static Encoding`, so the ones
30// missing here — the whole WHATWG set — still work via `encoding_rs`, or by
31// label through [`set_default_encoding_by_label`].
32//
33// | Encoding       | Where it is used                                    |
34// | -------------- | --------------------------------------------------- |
35// | `WINDOWS_1250` | Polish, Czech, Slovak, Hungarian, Romanian, Croatian |
36// | `WINDOWS_1251` | Russian and other Cyrillic scripts                   |
37// | `WINDOWS_1252` | Western Europe, Latin America (the default)          |
38// | `WINDOWS_1253` | Greek                                                |
39// | `WINDOWS_1254` | Turkish                                              |
40// | `WINDOWS_1256` | Arabic                                               |
41// | `WINDOWS_1257` | Baltic — Lithuanian, Latvian, Estonian               |
42// | `ISO_8859_2`   | Central Europe, where a script predates the CP1250 era |
43// | `UTF_8`        | Open Multiplayer scripts written in UTF-8            |
44pub use encoding_rs::{
45    ISO_8859_2, UTF_8, WINDOWS_1250, WINDOWS_1251, WINDOWS_1252, WINDOWS_1253, WINDOWS_1254,
46    WINDOWS_1256, WINDOWS_1257,
47};
48
49static DEFAULT_ENCODING: AtomicPtr<Encoding> =
50    AtomicPtr::new(std::ptr::from_ref::<Encoding>(WINDOWS_1252).cast_mut());
51
52/// Sets the global encoding used in every AMX string conversion.
53///
54/// Call once during plugin initialization (`on_load`). Later changes are
55/// visible immediately to any thread (`Ordering::Release`/`Acquire`).
56pub fn set_default_encoding(encoding: &'static Encoding) {
57    DEFAULT_ENCODING.store(
58        std::ptr::from_ref::<Encoding>(encoding).cast_mut(),
59        Ordering::Release,
60    );
61}
62
63/// Sets the global encoding from a label, the way a configuration file names it.
64///
65/// Accepts every label the [WHATWG Encoding Standard] defines, so
66/// `"windows-1251"`, `"cp1251"` and `"cyrillic"` all resolve to the same
67/// encoding, case and surrounding whitespace ignored. That is what lets a
68/// plugin read the encoding from the server's own configuration instead of
69/// compiling it in.
70///
71/// Returns the encoding that was set, or `None` when the label matches nothing
72/// — in which case the current encoding is left alone, and the caller should
73/// say so rather than silently run in the wrong encoding.
74///
75/// ```rust
76/// # use samp_sdk::encoding::{set_default_encoding, set_default_encoding_by_label, WINDOWS_1252};
77/// let chosen = set_default_encoding_by_label("windows-1254");
78/// assert_eq!(chosen.map(Encoding::name), Some("windows-1254"));
79/// assert!(set_default_encoding_by_label("not-an-encoding").is_none());
80/// # set_default_encoding(WINDOWS_1252);
81/// # use encoding_rs::Encoding;
82/// ```
83///
84/// [WHATWG Encoding Standard]: https://encoding.spec.whatwg.org/#names-and-labels
85pub fn set_default_encoding_by_label(label: &str) -> Option<&'static Encoding> {
86    let encoding = Encoding::for_label(label.as_bytes())?;
87    set_default_encoding(encoding);
88    Some(encoding)
89}
90
91/// Encodes `text` with the configured encoding, reporting whether anything was
92/// lost on the way.
93///
94/// Encoding is not always possible: Windows-1252 has no Cyrillic, Windows-1251
95/// has no `ç`, and no 8-bit code page has an emoji. `encoding_rs` substitutes
96/// what it cannot represent — a `?`, or an HTML numeric reference — and says
97/// nothing, so a player named `Ковальски` becomes `?????` on a Western server
98/// with the plugin none the wiser.
99///
100/// The returned flag is that warning. Pair it with [`unmappable_chars`] when it
101/// is `true` to say in the log which characters were lost.
102///
103/// ```rust
104/// # use samp_sdk::encoding::{encode_checked, set_default_encoding, WINDOWS_1252};
105/// set_default_encoding(WINDOWS_1252);
106/// let (bytes, lost) = encode_checked("cafe");
107/// assert!(!lost);
108/// assert_eq!(bytes.as_ref(), b"cafe");
109///
110/// let (_, lost) = encode_checked("Привет");
111/// assert!(lost, "Cyrillic does not fit in Windows-1252");
112/// ```
113#[must_use]
114pub fn encode_checked(text: &str) -> (std::borrow::Cow<'_, [u8]>, bool) {
115    let (bytes, _, had_unmappable) = get().encode(text);
116    (bytes, had_unmappable)
117}
118
119/// The characters of `text` the configured encoding cannot represent, in order
120/// of first appearance and without repeats.
121///
122/// Meant for the diagnostic that follows an [`encode_checked`] flag, so a log
123/// line can name the characters instead of only reporting that something was
124/// lost. It encodes character by character, so call it on the failure path
125/// rather than on every string.
126#[must_use]
127pub fn unmappable_chars(text: &str) -> Vec<char> {
128    let encoding = get();
129    let mut lost = Vec::new();
130    let mut buffer = [0u8; 4];
131
132    for ch in text.chars() {
133        let (_, _, had_unmappable) = encoding.encode(ch.encode_utf8(&mut buffer));
134        if had_unmappable && !lost.contains(&ch) {
135            lost.push(ch);
136        }
137    }
138    lost
139}
140
141pub(crate) fn get() -> &'static Encoding {
142    unsafe { &*DEFAULT_ENCODING.load(Ordering::Acquire) }
143}
144
145#[cfg(test)]
146mod tests {
147    use super::*;
148    use std::sync::{Mutex, PoisonError};
149
150    /// The encoding is process-wide, so the tests take turns and each one puts
151    /// the default back.
152    static TEST_LOCK: Mutex<()> = Mutex::new(());
153
154    #[test]
155    fn default_encoding_is_windows_1252() {
156        let _g = TEST_LOCK.lock().unwrap_or_else(PoisonError::into_inner);
157        let enc = get();
158        assert_eq!(enc.name(), WINDOWS_1252.name());
159    }
160
161    #[test]
162    fn set_and_get_encoding() {
163        let _g = TEST_LOCK.lock().unwrap_or_else(PoisonError::into_inner);
164        set_default_encoding(WINDOWS_1251);
165        let enc = get();
166        assert_eq!(enc.name(), WINDOWS_1251.name());
167
168        // restore default
169        set_default_encoding(WINDOWS_1252);
170        let enc = get();
171        assert_eq!(enc.name(), WINDOWS_1252.name());
172    }
173
174    #[test]
175    fn label_resolves_and_sets() {
176        let _g = TEST_LOCK.lock().unwrap_or_else(PoisonError::into_inner);
177        let chosen = set_default_encoding_by_label("windows-1254");
178        assert_eq!(chosen.map(Encoding::name), Some("windows-1254"));
179        assert_eq!(get().name(), "windows-1254");
180        set_default_encoding(WINDOWS_1252);
181    }
182
183    #[test]
184    fn label_matching_follows_the_whatwg_aliases() {
185        let _g = TEST_LOCK.lock().unwrap_or_else(PoisonError::into_inner);
186        // Same encoding under three spellings a config file might carry.
187        // Note `cyrillic` is *not* one of them — WHATWG maps that label to
188        // ISO-8859-5, a different encoding.
189        for label in ["cp1251", "WINDOWS-1251", "  x-cp1251  "] {
190            let chosen = set_default_encoding_by_label(label);
191            assert_eq!(
192                chosen.map(Encoding::name),
193                Some("windows-1251"),
194                "label {label:?} should resolve to windows-1251"
195            );
196        }
197        set_default_encoding(WINDOWS_1252);
198    }
199
200    #[test]
201    fn an_unknown_label_changes_nothing() {
202        let _g = TEST_LOCK.lock().unwrap_or_else(PoisonError::into_inner);
203        set_default_encoding(WINDOWS_1251);
204        assert!(set_default_encoding_by_label("not-an-encoding").is_none());
205        assert_eq!(get().name(), WINDOWS_1251.name(), "the encoding must stay");
206        set_default_encoding(WINDOWS_1252);
207    }
208
209    #[test]
210    fn encode_checked_flags_what_the_encoding_cannot_represent() {
211        let _g = TEST_LOCK.lock().unwrap_or_else(PoisonError::into_inner);
212        set_default_encoding(WINDOWS_1252);
213
214        let (bytes, lost) = encode_checked("cafe");
215        assert!(!lost);
216        assert_eq!(bytes.as_ref(), b"cafe");
217
218        let (_, lost) = encode_checked("Привет");
219        assert!(lost, "Cyrillic does not fit in Windows-1252");
220
221        // The same text under an encoding that does have it.
222        set_default_encoding(WINDOWS_1251);
223        let (_, lost) = encode_checked("Привет");
224        assert!(!lost);
225
226        set_default_encoding(WINDOWS_1252);
227    }
228
229    #[test]
230    fn unmappable_chars_names_them_once_and_in_order() {
231        let _g = TEST_LOCK.lock().unwrap_or_else(PoisonError::into_inner);
232        set_default_encoding(WINDOWS_1252);
233
234        assert_eq!(unmappable_chars("ok, tudo cabe: áéç"), Vec::<char>::new());
235        // `ж` twice, but reported once; the emoji keeps its position after it.
236        assert_eq!(unmappable_chars("aжbжc😀"), vec!['ж', '😀']);
237
238        set_default_encoding(WINDOWS_1252);
239    }
240
241    #[test]
242    fn the_re_exports_are_the_encodings_they_claim() {
243        assert_eq!(WINDOWS_1250.name(), "windows-1250");
244        assert_eq!(WINDOWS_1253.name(), "windows-1253");
245        assert_eq!(WINDOWS_1254.name(), "windows-1254");
246        assert_eq!(WINDOWS_1256.name(), "windows-1256");
247        assert_eq!(WINDOWS_1257.name(), "windows-1257");
248        assert_eq!(ISO_8859_2.name(), "ISO-8859-2");
249        assert_eq!(UTF_8.name(), "UTF-8");
250    }
251}