samp_sdk/encoding.rs
1//! Global encoding for Rust <-> AMX conversion (only with the `encoding` feature).
2//!
3//! The original SA-MP operates on 8-bit encodings (Western Windows-1252 by
4//! default, Windows-1251 for Cyrillic on Russian servers). This module lets the
5//! plugin configure the encoding once in `on_load` — after that, `AmxString`
6//! decodes and [`Buffer::write_str`] encodes using it automatically.
7//!
8//! [`set_default_encoding`] takes any `&'static Encoding`, so every encoding
9//! `encoding_rs` implements is available; the ones SA-MP and Open Multiplayer
10//! servers actually run are re-exported here, and
11//! [`set_default_encoding_by_label`] resolves one by name for a plugin that
12//! reads it from configuration.
13//!
14//! ## Multi-byte encodings
15//!
16//! Pawn assumes one cell holds one character. That holds for every 8-bit
17//! encoding above, and **not** for UTF-8, GBK, Big5, Shift_JIS or EUC-KR, where
18//! one character may take several bytes. Converting in and out of Rust stays
19//! correct, but on the Pawn side `strlen` then counts bytes rather than
20//! characters and `text[3]` is the fourth byte, not the fourth character. Use a
21//! multi-byte encoding only when the script is written for it.
22//!
23//! [`Buffer::write_str`]: crate::cell::Buffer::write_str
24
25use encoding_rs::Encoding;
26use std::sync::atomic::{AtomicPtr, Ordering};
27
28// Re-exported so a plugin does not have to depend on `encoding_rs` directly to
29// name one. `set_default_encoding` takes any `&'static Encoding`, so the ones
30// missing here — the whole WHATWG set — still work via `encoding_rs`, or by
31// label through [`set_default_encoding_by_label`].
32//
33// | Encoding | Where it is used |
34// | -------------- | --------------------------------------------------- |
35// | `WINDOWS_1250` | Polish, Czech, Slovak, Hungarian, Romanian, Croatian |
36// | `WINDOWS_1251` | Russian and other Cyrillic scripts |
37// | `WINDOWS_1252` | Western Europe, Latin America (the default) |
38// | `WINDOWS_1253` | Greek |
39// | `WINDOWS_1254` | Turkish |
40// | `WINDOWS_1256` | Arabic |
41// | `WINDOWS_1257` | Baltic — Lithuanian, Latvian, Estonian |
42// | `ISO_8859_2` | Central Europe, where a script predates the CP1250 era |
43// | `UTF_8` | Open Multiplayer scripts written in UTF-8 |
44pub use encoding_rs::{
45 ISO_8859_2, UTF_8, WINDOWS_1250, WINDOWS_1251, WINDOWS_1252, WINDOWS_1253, WINDOWS_1254,
46 WINDOWS_1256, WINDOWS_1257,
47};
48
49static DEFAULT_ENCODING: AtomicPtr<Encoding> =
50 AtomicPtr::new(std::ptr::from_ref::<Encoding>(WINDOWS_1252).cast_mut());
51
52/// Sets the global encoding used in every AMX string conversion.
53///
54/// Call once during plugin initialization (`on_load`). Later changes are
55/// visible immediately to any thread (`Ordering::Release`/`Acquire`).
56pub fn set_default_encoding(encoding: &'static Encoding) {
57 DEFAULT_ENCODING.store(
58 std::ptr::from_ref::<Encoding>(encoding).cast_mut(),
59 Ordering::Release,
60 );
61}
62
63/// Sets the global encoding from a label, the way a configuration file names it.
64///
65/// Accepts every label the [WHATWG Encoding Standard] defines, so
66/// `"windows-1251"`, `"cp1251"` and `"cyrillic"` all resolve to the same
67/// encoding, case and surrounding whitespace ignored. That is what lets a
68/// plugin read the encoding from the server's own configuration instead of
69/// compiling it in.
70///
71/// Returns the encoding that was set, or `None` when the label matches nothing
72/// — in which case the current encoding is left alone, and the caller should
73/// say so rather than silently run in the wrong encoding.
74///
75/// ```rust
76/// # use samp_sdk::encoding::{set_default_encoding, set_default_encoding_by_label, WINDOWS_1252};
77/// let chosen = set_default_encoding_by_label("windows-1254");
78/// assert_eq!(chosen.map(Encoding::name), Some("windows-1254"));
79/// assert!(set_default_encoding_by_label("not-an-encoding").is_none());
80/// # set_default_encoding(WINDOWS_1252);
81/// # use encoding_rs::Encoding;
82/// ```
83///
84/// [WHATWG Encoding Standard]: https://encoding.spec.whatwg.org/#names-and-labels
85pub fn set_default_encoding_by_label(label: &str) -> Option<&'static Encoding> {
86 let encoding = Encoding::for_label(label.as_bytes())?;
87 set_default_encoding(encoding);
88 Some(encoding)
89}
90
91/// Encodes `text` with the configured encoding, reporting whether anything was
92/// lost on the way.
93///
94/// Encoding is not always possible: Windows-1252 has no Cyrillic, Windows-1251
95/// has no `ç`, and no 8-bit code page has an emoji. `encoding_rs` substitutes
96/// what it cannot represent — a `?`, or an HTML numeric reference — and says
97/// nothing, so a player named `Ковальски` becomes `?????` on a Western server
98/// with the plugin none the wiser.
99///
100/// The returned flag is that warning. Pair it with [`unmappable_chars`] when it
101/// is `true` to say in the log which characters were lost.
102///
103/// ```rust
104/// # use samp_sdk::encoding::{encode_checked, set_default_encoding, WINDOWS_1252};
105/// set_default_encoding(WINDOWS_1252);
106/// let (bytes, lost) = encode_checked("cafe");
107/// assert!(!lost);
108/// assert_eq!(bytes.as_ref(), b"cafe");
109///
110/// let (_, lost) = encode_checked("Привет");
111/// assert!(lost, "Cyrillic does not fit in Windows-1252");
112/// ```
113#[must_use]
114pub fn encode_checked(text: &str) -> (std::borrow::Cow<'_, [u8]>, bool) {
115 let (bytes, _, had_unmappable) = get().encode(text);
116 (bytes, had_unmappable)
117}
118
119/// The characters of `text` the configured encoding cannot represent, in order
120/// of first appearance and without repeats.
121///
122/// Meant for the diagnostic that follows an [`encode_checked`] flag, so a log
123/// line can name the characters instead of only reporting that something was
124/// lost. It encodes character by character, so call it on the failure path
125/// rather than on every string.
126#[must_use]
127pub fn unmappable_chars(text: &str) -> Vec<char> {
128 let encoding = get();
129 let mut lost = Vec::new();
130 let mut buffer = [0u8; 4];
131
132 for ch in text.chars() {
133 let (_, _, had_unmappable) = encoding.encode(ch.encode_utf8(&mut buffer));
134 if had_unmappable && !lost.contains(&ch) {
135 lost.push(ch);
136 }
137 }
138 lost
139}
140
141pub(crate) fn get() -> &'static Encoding {
142 unsafe { &*DEFAULT_ENCODING.load(Ordering::Acquire) }
143}
144
145/// The encoding every AMX string conversion currently uses.
146///
147/// A plugin that lets the server owner choose the encoding needs to be able to
148/// report back which one is in force — in a diagnostics native, or in the line
149/// it logs at startup.
150///
151/// ```rust
152/// assert_eq!(samp_sdk::encoding::current().name(), "windows-1252");
153/// ```
154#[must_use]
155pub fn current() -> &'static Encoding {
156 get()
157}
158
159/// The encoding is process-wide, so tests that change it take turns and each
160/// one puts the default back.
161#[cfg(test)]
162pub(crate) fn tests_lock() -> std::sync::MutexGuard<'static, ()> {
163 static TEST_LOCK: std::sync::Mutex<()> = std::sync::Mutex::new(());
164 TEST_LOCK
165 .lock()
166 .unwrap_or_else(std::sync::PoisonError::into_inner)
167}
168
169#[cfg(test)]
170mod tests {
171 use super::*;
172
173 #[test]
174 fn default_encoding_is_windows_1252() {
175 let _g = tests_lock();
176 let enc = get();
177 assert_eq!(enc.name(), WINDOWS_1252.name());
178 }
179
180 #[test]
181 fn set_and_get_encoding() {
182 let _g = tests_lock();
183 set_default_encoding(WINDOWS_1251);
184 let enc = get();
185 assert_eq!(enc.name(), WINDOWS_1251.name());
186
187 // restore default
188 set_default_encoding(WINDOWS_1252);
189 let enc = get();
190 assert_eq!(enc.name(), WINDOWS_1252.name());
191 }
192
193 #[test]
194 fn label_resolves_and_sets() {
195 let _g = tests_lock();
196 let chosen = set_default_encoding_by_label("windows-1254");
197 assert_eq!(chosen.map(Encoding::name), Some("windows-1254"));
198 assert_eq!(get().name(), "windows-1254");
199 set_default_encoding(WINDOWS_1252);
200 }
201
202 #[test]
203 fn label_matching_follows_the_whatwg_aliases() {
204 let _g = tests_lock();
205 // Same encoding under three spellings a config file might carry.
206 // Note `cyrillic` is *not* one of them — WHATWG maps that label to
207 // ISO-8859-5, a different encoding.
208 for label in ["cp1251", "WINDOWS-1251", " x-cp1251 "] {
209 let chosen = set_default_encoding_by_label(label);
210 assert_eq!(
211 chosen.map(Encoding::name),
212 Some("windows-1251"),
213 "label {label:?} should resolve to windows-1251"
214 );
215 }
216 set_default_encoding(WINDOWS_1252);
217 }
218
219 #[test]
220 fn an_unknown_label_changes_nothing() {
221 let _g = tests_lock();
222 set_default_encoding(WINDOWS_1251);
223 assert!(set_default_encoding_by_label("not-an-encoding").is_none());
224 assert_eq!(get().name(), WINDOWS_1251.name(), "the encoding must stay");
225 set_default_encoding(WINDOWS_1252);
226 }
227
228 #[test]
229 fn encode_checked_flags_what_the_encoding_cannot_represent() {
230 let _g = tests_lock();
231 set_default_encoding(WINDOWS_1252);
232
233 let (bytes, lost) = encode_checked("cafe");
234 assert!(!lost);
235 assert_eq!(bytes.as_ref(), b"cafe");
236
237 let (_, lost) = encode_checked("Привет");
238 assert!(lost, "Cyrillic does not fit in Windows-1252");
239
240 // The same text under an encoding that does have it.
241 set_default_encoding(WINDOWS_1251);
242 let (_, lost) = encode_checked("Привет");
243 assert!(!lost);
244
245 set_default_encoding(WINDOWS_1252);
246 }
247
248 #[test]
249 fn unmappable_chars_names_them_once_and_in_order() {
250 let _g = tests_lock();
251 set_default_encoding(WINDOWS_1252);
252
253 assert_eq!(unmappable_chars("ok, tudo cabe: áéç"), Vec::<char>::new());
254 // `ж` twice, but reported once; the emoji keeps its position after it.
255 assert_eq!(unmappable_chars("aжbжc😀"), vec!['ж', '😀']);
256
257 set_default_encoding(WINDOWS_1252);
258 }
259
260 #[test]
261 fn the_re_exports_are_the_encodings_they_claim() {
262 assert_eq!(WINDOWS_1250.name(), "windows-1250");
263 assert_eq!(WINDOWS_1253.name(), "windows-1253");
264 assert_eq!(WINDOWS_1254.name(), "windows-1254");
265 assert_eq!(WINDOWS_1256.name(), "windows-1256");
266 assert_eq!(WINDOWS_1257.name(), "windows-1257");
267 assert_eq!(ISO_8859_2.name(), "ISO-8859-2");
268 assert_eq!(UTF_8.name(), "UTF-8");
269 }
270}