samp_sdk/encoding.rs
1//! Global encoding for Rust <-> AMX conversion (only with the `encoding` feature).
2//!
3//! The original SA-MP operates on 8-bit encodings (Western Windows-1252 by
4//! default, Windows-1251 for Cyrillic on Russian servers). This module lets the
5//! plugin configure the encoding once in `on_load` — after that, `AmxString`
6//! decodes and [`Buffer::write_str`] encodes using it automatically.
7//!
8//! [`set_default_encoding`] takes any `&'static Encoding`, so every encoding
9//! `encoding_rs` implements is available; the ones SA-MP and Open Multiplayer
10//! servers actually run are re-exported here, and
11//! [`set_default_encoding_by_label`] resolves one by name for a plugin that
12//! reads it from configuration.
13//!
14//! ## Multi-byte encodings
15//!
16//! Pawn assumes one cell holds one character. That holds for every 8-bit
17//! encoding above, and **not** for UTF-8, GBK, Big5, Shift_JIS or EUC-KR, where
18//! one character may take several bytes. Converting in and out of Rust stays
19//! correct, but on the Pawn side `strlen` then counts bytes rather than
20//! characters and `text[3]` is the fourth byte, not the fourth character. Use a
21//! multi-byte encoding only when the script is written for it.
22//!
23//! [`Buffer::write_str`]: crate::cell::Buffer::write_str
24
25use encoding_rs::Encoding;
26use std::sync::atomic::{AtomicPtr, Ordering};
27
28// Re-exported so a plugin does not have to depend on `encoding_rs` directly to
29// name one. `set_default_encoding` takes any `&'static Encoding`, so the ones
30// missing here — the whole WHATWG set — still work via `encoding_rs`, or by
31// label through [`set_default_encoding_by_label`].
32//
33// | Encoding | Where it is used |
34// | -------------- | --------------------------------------------------- |
35// | `WINDOWS_1250` | Polish, Czech, Slovak, Hungarian, Romanian, Croatian |
36// | `WINDOWS_1251` | Russian and other Cyrillic scripts |
37// | `WINDOWS_1252` | Western Europe, Latin America (the default) |
38// | `WINDOWS_1253` | Greek |
39// | `WINDOWS_1254` | Turkish |
40// | `WINDOWS_1256` | Arabic |
41// | `WINDOWS_1257` | Baltic — Lithuanian, Latvian, Estonian |
42// | `ISO_8859_2` | Central Europe, where a script predates the CP1250 era |
43// | `UTF_8` | Open Multiplayer scripts written in UTF-8 |
44pub use encoding_rs::{
45 ISO_8859_2, UTF_8, WINDOWS_1250, WINDOWS_1251, WINDOWS_1252, WINDOWS_1253, WINDOWS_1254,
46 WINDOWS_1256, WINDOWS_1257,
47};
48
49static DEFAULT_ENCODING: AtomicPtr<Encoding> =
50 AtomicPtr::new(std::ptr::from_ref::<Encoding>(WINDOWS_1252).cast_mut());
51
52/// Sets the global encoding used in every AMX string conversion.
53///
54/// Call once during plugin initialization (`on_load`). Later changes are
55/// visible immediately to any thread (`Ordering::Release`/`Acquire`).
56pub fn set_default_encoding(encoding: &'static Encoding) {
57 DEFAULT_ENCODING.store(
58 std::ptr::from_ref::<Encoding>(encoding).cast_mut(),
59 Ordering::Release,
60 );
61}
62
63/// Sets the global encoding from a label, the way a configuration file names it.
64///
65/// Accepts every label the [WHATWG Encoding Standard] defines, so
66/// `"windows-1251"`, `"cp1251"` and `"cyrillic"` all resolve to the same
67/// encoding, case and surrounding whitespace ignored. That is what lets a
68/// plugin read the encoding from the server's own configuration instead of
69/// compiling it in.
70///
71/// Returns the encoding that was set, or `None` when the label matches nothing
72/// — in which case the current encoding is left alone, and the caller should
73/// say so rather than silently run in the wrong encoding.
74///
75/// ```rust
76/// # use samp_sdk::encoding::{set_default_encoding, set_default_encoding_by_label, WINDOWS_1252};
77/// let chosen = set_default_encoding_by_label("windows-1254");
78/// assert_eq!(chosen.map(Encoding::name), Some("windows-1254"));
79/// assert!(set_default_encoding_by_label("not-an-encoding").is_none());
80/// # set_default_encoding(WINDOWS_1252);
81/// # use encoding_rs::Encoding;
82/// ```
83///
84/// [WHATWG Encoding Standard]: https://encoding.spec.whatwg.org/#names-and-labels
85pub fn set_default_encoding_by_label(label: &str) -> Option<&'static Encoding> {
86 let encoding = Encoding::for_label(label.as_bytes())?;
87 set_default_encoding(encoding);
88 Some(encoding)
89}
90
91/// Encodes `text` with the configured encoding, reporting whether anything was
92/// lost on the way.
93///
94/// Encoding is not always possible: Windows-1252 has no Cyrillic, Windows-1251
95/// has no `ç`, and no 8-bit code page has an emoji. `encoding_rs` substitutes
96/// what it cannot represent — a `?`, or an HTML numeric reference — and says
97/// nothing, so a player named `Ковальски` becomes `?????` on a Western server
98/// with the plugin none the wiser.
99///
100/// The returned flag is that warning. Pair it with [`unmappable_chars`] when it
101/// is `true` to say in the log which characters were lost.
102///
103/// ```rust
104/// # use samp_sdk::encoding::{encode_checked, set_default_encoding, WINDOWS_1252};
105/// set_default_encoding(WINDOWS_1252);
106/// let (bytes, lost) = encode_checked("cafe");
107/// assert!(!lost);
108/// assert_eq!(bytes.as_ref(), b"cafe");
109///
110/// let (_, lost) = encode_checked("Привет");
111/// assert!(lost, "Cyrillic does not fit in Windows-1252");
112/// ```
113#[must_use]
114pub fn encode_checked(text: &str) -> (std::borrow::Cow<'_, [u8]>, bool) {
115 let (bytes, _, had_unmappable) = get().encode(text);
116 (bytes, had_unmappable)
117}
118
119/// The characters of `text` the configured encoding cannot represent, in order
120/// of first appearance and without repeats.
121///
122/// Meant for the diagnostic that follows an [`encode_checked`] flag, so a log
123/// line can name the characters instead of only reporting that something was
124/// lost. It encodes character by character, so call it on the failure path
125/// rather than on every string.
126#[must_use]
127pub fn unmappable_chars(text: &str) -> Vec<char> {
128 let encoding = get();
129 let mut lost = Vec::new();
130 let mut buffer = [0u8; 4];
131
132 for ch in text.chars() {
133 let (_, _, had_unmappable) = encoding.encode(ch.encode_utf8(&mut buffer));
134 if had_unmappable && !lost.contains(&ch) {
135 lost.push(ch);
136 }
137 }
138 lost
139}
140
141pub(crate) fn get() -> &'static Encoding {
142 unsafe { &*DEFAULT_ENCODING.load(Ordering::Acquire) }
143}
144
145#[cfg(test)]
146mod tests {
147 use super::*;
148 use std::sync::{Mutex, PoisonError};
149
150 /// The encoding is process-wide, so the tests take turns and each one puts
151 /// the default back.
152 static TEST_LOCK: Mutex<()> = Mutex::new(());
153
154 #[test]
155 fn default_encoding_is_windows_1252() {
156 let _g = TEST_LOCK.lock().unwrap_or_else(PoisonError::into_inner);
157 let enc = get();
158 assert_eq!(enc.name(), WINDOWS_1252.name());
159 }
160
161 #[test]
162 fn set_and_get_encoding() {
163 let _g = TEST_LOCK.lock().unwrap_or_else(PoisonError::into_inner);
164 set_default_encoding(WINDOWS_1251);
165 let enc = get();
166 assert_eq!(enc.name(), WINDOWS_1251.name());
167
168 // restore default
169 set_default_encoding(WINDOWS_1252);
170 let enc = get();
171 assert_eq!(enc.name(), WINDOWS_1252.name());
172 }
173
174 #[test]
175 fn label_resolves_and_sets() {
176 let _g = TEST_LOCK.lock().unwrap_or_else(PoisonError::into_inner);
177 let chosen = set_default_encoding_by_label("windows-1254");
178 assert_eq!(chosen.map(Encoding::name), Some("windows-1254"));
179 assert_eq!(get().name(), "windows-1254");
180 set_default_encoding(WINDOWS_1252);
181 }
182
183 #[test]
184 fn label_matching_follows_the_whatwg_aliases() {
185 let _g = TEST_LOCK.lock().unwrap_or_else(PoisonError::into_inner);
186 // Same encoding under three spellings a config file might carry.
187 // Note `cyrillic` is *not* one of them — WHATWG maps that label to
188 // ISO-8859-5, a different encoding.
189 for label in ["cp1251", "WINDOWS-1251", " x-cp1251 "] {
190 let chosen = set_default_encoding_by_label(label);
191 assert_eq!(
192 chosen.map(Encoding::name),
193 Some("windows-1251"),
194 "label {label:?} should resolve to windows-1251"
195 );
196 }
197 set_default_encoding(WINDOWS_1252);
198 }
199
200 #[test]
201 fn an_unknown_label_changes_nothing() {
202 let _g = TEST_LOCK.lock().unwrap_or_else(PoisonError::into_inner);
203 set_default_encoding(WINDOWS_1251);
204 assert!(set_default_encoding_by_label("not-an-encoding").is_none());
205 assert_eq!(get().name(), WINDOWS_1251.name(), "the encoding must stay");
206 set_default_encoding(WINDOWS_1252);
207 }
208
209 #[test]
210 fn encode_checked_flags_what_the_encoding_cannot_represent() {
211 let _g = TEST_LOCK.lock().unwrap_or_else(PoisonError::into_inner);
212 set_default_encoding(WINDOWS_1252);
213
214 let (bytes, lost) = encode_checked("cafe");
215 assert!(!lost);
216 assert_eq!(bytes.as_ref(), b"cafe");
217
218 let (_, lost) = encode_checked("Привет");
219 assert!(lost, "Cyrillic does not fit in Windows-1252");
220
221 // The same text under an encoding that does have it.
222 set_default_encoding(WINDOWS_1251);
223 let (_, lost) = encode_checked("Привет");
224 assert!(!lost);
225
226 set_default_encoding(WINDOWS_1252);
227 }
228
229 #[test]
230 fn unmappable_chars_names_them_once_and_in_order() {
231 let _g = TEST_LOCK.lock().unwrap_or_else(PoisonError::into_inner);
232 set_default_encoding(WINDOWS_1252);
233
234 assert_eq!(unmappable_chars("ok, tudo cabe: áéç"), Vec::<char>::new());
235 // `ж` twice, but reported once; the emoji keeps its position after it.
236 assert_eq!(unmappable_chars("aжbжc😀"), vec!['ж', '😀']);
237
238 set_default_encoding(WINDOWS_1252);
239 }
240
241 #[test]
242 fn the_re_exports_are_the_encodings_they_claim() {
243 assert_eq!(WINDOWS_1250.name(), "windows-1250");
244 assert_eq!(WINDOWS_1253.name(), "windows-1253");
245 assert_eq!(WINDOWS_1254.name(), "windows-1254");
246 assert_eq!(WINDOWS_1256.name(), "windows-1256");
247 assert_eq!(WINDOWS_1257.name(), "windows-1257");
248 assert_eq!(ISO_8859_2.name(), "ISO-8859-2");
249 assert_eq!(UTF_8.name(), "UTF-8");
250 }
251}