Skip to main content

samp_sdk/
encoding.rs

1//! Global encoding for Rust <-> AMX conversion (only with the `encoding` feature).
2//!
3//! The original SA-MP operates on 8-bit encodings (Western Windows-1252 by
4//! default, Windows-1251 for Cyrillic on Russian servers). This module lets the
5//! plugin configure the encoding once in `on_load` — after that, `AmxString`
6//! decodes and [`Buffer::write_str`] encodes using it automatically.
7//!
8//! [`set_default_encoding`] takes any `&'static Encoding`, so every encoding
9//! `encoding_rs` implements is available; the ones SA-MP and Open Multiplayer
10//! servers actually run are re-exported here, and
11//! [`set_default_encoding_by_label`] resolves one by name for a plugin that
12//! reads it from configuration.
13//!
14//! ## Multi-byte encodings
15//!
16//! Pawn assumes one cell holds one character. That holds for every 8-bit
17//! encoding above, and **not** for UTF-8, GBK, Big5, Shift_JIS or EUC-KR, where
18//! one character may take several bytes. Converting in and out of Rust stays
19//! correct, but on the Pawn side `strlen` then counts bytes rather than
20//! characters and `text[3]` is the fourth byte, not the fourth character. Use a
21//! multi-byte encoding only when the script is written for it.
22//!
23//! [`Buffer::write_str`]: crate::cell::Buffer::write_str
24
25use encoding_rs::Encoding;
26use std::sync::atomic::{AtomicPtr, Ordering};
27
28// Re-exported so a plugin does not have to depend on `encoding_rs` directly to
29// name one. `set_default_encoding` takes any `&'static Encoding`, so the ones
30// missing here — the whole WHATWG set — still work via `encoding_rs`, or by
31// label through [`set_default_encoding_by_label`].
32//
33// | Encoding       | Where it is used                                    |
34// | -------------- | --------------------------------------------------- |
35// | `WINDOWS_1250` | Polish, Czech, Slovak, Hungarian, Romanian, Croatian |
36// | `WINDOWS_1251` | Russian and other Cyrillic scripts                   |
37// | `WINDOWS_1252` | Western Europe, Latin America (the default)          |
38// | `WINDOWS_1253` | Greek                                                |
39// | `WINDOWS_1254` | Turkish                                              |
40// | `WINDOWS_1256` | Arabic                                               |
41// | `WINDOWS_1257` | Baltic — Lithuanian, Latvian, Estonian               |
42// | `ISO_8859_2`   | Central Europe, where a script predates the CP1250 era |
43// | `UTF_8`        | Open Multiplayer scripts written in UTF-8            |
44pub use encoding_rs::{
45    ISO_8859_2, UTF_8, WINDOWS_1250, WINDOWS_1251, WINDOWS_1252, WINDOWS_1253, WINDOWS_1254,
46    WINDOWS_1256, WINDOWS_1257,
47};
48
49static DEFAULT_ENCODING: AtomicPtr<Encoding> =
50    AtomicPtr::new(std::ptr::from_ref::<Encoding>(WINDOWS_1252).cast_mut());
51
52/// Sets the global encoding used in every AMX string conversion.
53///
54/// Call once during plugin initialization (`on_load`). Later changes are
55/// visible immediately to any thread (`Ordering::Release`/`Acquire`).
56pub fn set_default_encoding(encoding: &'static Encoding) {
57    DEFAULT_ENCODING.store(
58        std::ptr::from_ref::<Encoding>(encoding).cast_mut(),
59        Ordering::Release,
60    );
61}
62
63/// Sets the global encoding from a label, the way a configuration file names it.
64///
65/// Accepts every label the [WHATWG Encoding Standard] defines, so
66/// `"windows-1251"`, `"cp1251"` and `"cyrillic"` all resolve to the same
67/// encoding, case and surrounding whitespace ignored. That is what lets a
68/// plugin read the encoding from the server's own configuration instead of
69/// compiling it in.
70///
71/// Returns the encoding that was set, or `None` when the label matches nothing
72/// — in which case the current encoding is left alone, and the caller should
73/// say so rather than silently run in the wrong encoding.
74///
75/// ```rust
76/// # use samp_sdk::encoding::{set_default_encoding, set_default_encoding_by_label, WINDOWS_1252};
77/// let chosen = set_default_encoding_by_label("windows-1254");
78/// assert_eq!(chosen.map(Encoding::name), Some("windows-1254"));
79/// assert!(set_default_encoding_by_label("not-an-encoding").is_none());
80/// # set_default_encoding(WINDOWS_1252);
81/// # use encoding_rs::Encoding;
82/// ```
83///
84/// [WHATWG Encoding Standard]: https://encoding.spec.whatwg.org/#names-and-labels
85pub fn set_default_encoding_by_label(label: &str) -> Option<&'static Encoding> {
86    let encoding = Encoding::for_label(label.as_bytes())?;
87    set_default_encoding(encoding);
88    Some(encoding)
89}
90
91/// Encodes `text` with the configured encoding, reporting whether anything was
92/// lost on the way.
93///
94/// Encoding is not always possible: Windows-1252 has no Cyrillic, Windows-1251
95/// has no `ç`, and no 8-bit code page has an emoji. `encoding_rs` substitutes
96/// what it cannot represent — a `?`, or an HTML numeric reference — and says
97/// nothing, so a player named `Ковальски` becomes `?????` on a Western server
98/// with the plugin none the wiser.
99///
100/// The returned flag is that warning. Pair it with [`unmappable_chars`] when it
101/// is `true` to say in the log which characters were lost.
102///
103/// ```rust
104/// # use samp_sdk::encoding::{encode_checked, set_default_encoding, WINDOWS_1252};
105/// set_default_encoding(WINDOWS_1252);
106/// let (bytes, lost) = encode_checked("cafe");
107/// assert!(!lost);
108/// assert_eq!(bytes.as_ref(), b"cafe");
109///
110/// let (_, lost) = encode_checked("Привет");
111/// assert!(lost, "Cyrillic does not fit in Windows-1252");
112/// ```
113#[must_use]
114pub fn encode_checked(text: &str) -> (std::borrow::Cow<'_, [u8]>, bool) {
115    let (bytes, _, had_unmappable) = get().encode(text);
116    (bytes, had_unmappable)
117}
118
119/// The characters of `text` the configured encoding cannot represent, in order
120/// of first appearance and without repeats.
121///
122/// Meant for the diagnostic that follows an [`encode_checked`] flag, so a log
123/// line can name the characters instead of only reporting that something was
124/// lost. It encodes character by character, so call it on the failure path
125/// rather than on every string.
126#[must_use]
127pub fn unmappable_chars(text: &str) -> Vec<char> {
128    let encoding = get();
129    let mut lost = Vec::new();
130    let mut buffer = [0u8; 4];
131
132    for ch in text.chars() {
133        let (_, _, had_unmappable) = encoding.encode(ch.encode_utf8(&mut buffer));
134        if had_unmappable && !lost.contains(&ch) {
135            lost.push(ch);
136        }
137    }
138    lost
139}
140
141pub(crate) fn get() -> &'static Encoding {
142    unsafe { &*DEFAULT_ENCODING.load(Ordering::Acquire) }
143}
144
145/// The encoding every AMX string conversion currently uses.
146///
147/// A plugin that lets the server owner choose the encoding needs to be able to
148/// report back which one is in force — in a diagnostics native, or in the line
149/// it logs at startup.
150///
151/// ```rust
152/// assert_eq!(samp_sdk::encoding::current().name(), "windows-1252");
153/// ```
154#[must_use]
155pub fn current() -> &'static Encoding {
156    get()
157}
158
159/// The encoding is process-wide, so tests that change it take turns and each
160/// one puts the default back.
161#[cfg(test)]
162pub(crate) fn tests_lock() -> std::sync::MutexGuard<'static, ()> {
163    static TEST_LOCK: std::sync::Mutex<()> = std::sync::Mutex::new(());
164    TEST_LOCK
165        .lock()
166        .unwrap_or_else(std::sync::PoisonError::into_inner)
167}
168
169#[cfg(test)]
170mod tests {
171    use super::*;
172
173    #[test]
174    fn default_encoding_is_windows_1252() {
175        let _g = tests_lock();
176        let enc = get();
177        assert_eq!(enc.name(), WINDOWS_1252.name());
178    }
179
180    #[test]
181    fn set_and_get_encoding() {
182        let _g = tests_lock();
183        set_default_encoding(WINDOWS_1251);
184        let enc = get();
185        assert_eq!(enc.name(), WINDOWS_1251.name());
186
187        // restore default
188        set_default_encoding(WINDOWS_1252);
189        let enc = get();
190        assert_eq!(enc.name(), WINDOWS_1252.name());
191    }
192
193    #[test]
194    fn label_resolves_and_sets() {
195        let _g = tests_lock();
196        let chosen = set_default_encoding_by_label("windows-1254");
197        assert_eq!(chosen.map(Encoding::name), Some("windows-1254"));
198        assert_eq!(get().name(), "windows-1254");
199        set_default_encoding(WINDOWS_1252);
200    }
201
202    #[test]
203    fn label_matching_follows_the_whatwg_aliases() {
204        let _g = tests_lock();
205        // Same encoding under three spellings a config file might carry.
206        // Note `cyrillic` is *not* one of them — WHATWG maps that label to
207        // ISO-8859-5, a different encoding.
208        for label in ["cp1251", "WINDOWS-1251", "  x-cp1251  "] {
209            let chosen = set_default_encoding_by_label(label);
210            assert_eq!(
211                chosen.map(Encoding::name),
212                Some("windows-1251"),
213                "label {label:?} should resolve to windows-1251"
214            );
215        }
216        set_default_encoding(WINDOWS_1252);
217    }
218
219    #[test]
220    fn an_unknown_label_changes_nothing() {
221        let _g = tests_lock();
222        set_default_encoding(WINDOWS_1251);
223        assert!(set_default_encoding_by_label("not-an-encoding").is_none());
224        assert_eq!(get().name(), WINDOWS_1251.name(), "the encoding must stay");
225        set_default_encoding(WINDOWS_1252);
226    }
227
228    #[test]
229    fn encode_checked_flags_what_the_encoding_cannot_represent() {
230        let _g = tests_lock();
231        set_default_encoding(WINDOWS_1252);
232
233        let (bytes, lost) = encode_checked("cafe");
234        assert!(!lost);
235        assert_eq!(bytes.as_ref(), b"cafe");
236
237        let (_, lost) = encode_checked("Привет");
238        assert!(lost, "Cyrillic does not fit in Windows-1252");
239
240        // The same text under an encoding that does have it.
241        set_default_encoding(WINDOWS_1251);
242        let (_, lost) = encode_checked("Привет");
243        assert!(!lost);
244
245        set_default_encoding(WINDOWS_1252);
246    }
247
248    #[test]
249    fn unmappable_chars_names_them_once_and_in_order() {
250        let _g = tests_lock();
251        set_default_encoding(WINDOWS_1252);
252
253        assert_eq!(unmappable_chars("ok, tudo cabe: áéç"), Vec::<char>::new());
254        // `ж` twice, but reported once; the emoji keeps its position after it.
255        assert_eq!(unmappable_chars("aжbжc😀"), vec!['ж', '😀']);
256
257        set_default_encoding(WINDOWS_1252);
258    }
259
260    #[test]
261    fn the_re_exports_are_the_encodings_they_claim() {
262        assert_eq!(WINDOWS_1250.name(), "windows-1250");
263        assert_eq!(WINDOWS_1253.name(), "windows-1253");
264        assert_eq!(WINDOWS_1254.name(), "windows-1254");
265        assert_eq!(WINDOWS_1256.name(), "windows-1256");
266        assert_eq!(WINDOWS_1257.name(), "windows-1257");
267        assert_eq!(ISO_8859_2.name(), "ISO-8859-2");
268        assert_eq!(UTF_8.name(), "UTF-8");
269    }
270}