mdwire/lib.rs
1//! mdwire — safely deliver agent-generated Markdown to chat channels.
2//!
3//! It does three things in one pipeline. The order is the design.
4//!
5//! 1. **Normalize** — LLM output is not valid CommonMark. Unbalanced emphasis,
6//! emphasis that spans lines, and unclosed code fences are routine. Repair them first.
7//! 2. **Render for the channel** — translate into the syntax the target accepts. Syntax the
8//! target cannot accept is not worth parsing in the first place (see the `Channel` docs below).
9//! 3. **Split safely** — at channel limits and streaming boundaries, never cut through markup.
10//!
11//! # Why no dependencies
12//!
13//! Existing parsers are all **batch** parsers — they take the whole document, build an AST,
14//! then render. That does not fit a problem where output must go out as tokens stream in.
15//! And the CJK-adjacent emphasis policy is baked into the parser, so with someone else's
16//! parser you cannot change it. That is exactly the problem this library sets out to fix.
17
18#![forbid(unsafe_code)]
19
20mod block;
21mod inline;
22mod sink;
23mod vocab;
24pub mod width;
25
26use block::Engine;
27use sink::{PartsSink, StringSink};
28use vocab::Vocab;
29
30/// Target channel. Each channel accepts a different syntax, and **the narrower output decides the
31/// parsing scope**.
32#[derive(Debug, Clone, Copy, PartialEq, Eq)]
33pub enum Channel {
34 /// Telegram `parse_mode=HTML`. Nine allowed tags:
35 /// `b i u s code pre a blockquote tg-spoiler`. No tables or headings. 4096 characters.
36 TelegramHtml,
37 /// Slack `markdown_text`. Slack converts standard Markdown itself. 12,000 characters.
38 /// Almost no conversion is needed; what is left is normalization and splitting.
39 SlackMarkdown,
40 /// GitHub comments and PR bodies (GFM). Renders tables, headings, and strikethrough.
41 /// 65,536 characters. Emits the same Markdown as Slack, but escapes the two characters GFM
42 /// reads as syntax (`~` `<`).
43 GithubMarkdown,
44 /// Notion page body (Notion-flavored Markdown — the API `markdown` field and connectors).
45 /// Four heading levels. Emits the same Markdown as GitHub, but strips inline HTML Notion
46 /// cannot render (it would show as text), and writes autolinks `<url>` as `[url](url)` (the
47 /// angle brackets would remain as text). Emphasis before a Korean particle is rendered by
48 /// Notion as is, so it is not turned into `<strong>` — doing so would make the tag show as text.
49 NotionMarkdown,
50 /// Strips all markup. The fallback path.
51 Plain,
52 /// An HTML fragment for the browser. Renders headings, lists, tables, and code blocks as tags.
53 /// No limit.
54 ///
55 /// Assumes the output goes straight into `innerHTML` — all text is escaped, HTML from the
56 /// source survives only as inline tags with their attributes dropped, and only `http(s)` and
57 /// `mailto` links become `<a>`. Appending [`Streamer::close_open`] to the streaming
58 /// accumulated output gives a shape that can be inserted as is.
59 Html,
60}
61
62impl Channel {
63 /// The name used for corpus directories and CLI arguments. The source of truth wherever a
64 /// channel is handled as a string.
65 pub fn name(self) -> &'static str {
66 match self {
67 Channel::TelegramHtml => "telegram-html",
68 Channel::SlackMarkdown => "slack-markdown",
69 Channel::GithubMarkdown => "github-markdown",
70 Channel::NotionMarkdown => "notion-markdown",
71 Channel::Plain => "plain",
72 Channel::Html => "html",
73 }
74 }
75
76 /// Every supported channel. The corpus and the harness iterate over this list.
77 pub fn all() -> [Channel; 6] {
78 [
79 Channel::TelegramHtml,
80 Channel::SlackMarkdown,
81 Channel::GithubMarkdown,
82 Channel::NotionMarkdown,
83 Channel::Plain,
84 Channel::Html,
85 ]
86 }
87
88 /// Looks up a channel by name.
89 pub fn parse(name: &str) -> Option<Channel> {
90 Self::all().into_iter().find(|c| c.name() == name)
91 }
92
93 /// This channel's message length limit, in characters. Splitting is based on it.
94 pub fn limit(self) -> usize {
95 match self {
96 Channel::TelegramHtml => 4096,
97 Channel::SlackMarkdown | Channel::Plain => 12_000,
98 // 코멘트 본문의 한도다. 넘기면 API 가 422 로 거절한다("Body is too long").
99 Channel::GithubMarkdown => 65_536,
100 // **재지 않았다.** 페이지 하나에 들어갈 본문이라 GitHub 과 같은 값을 둔다. 보내는 쪽
101 // 한도가 따로 있으면 [`Options::limit`] 으로 준다.
102 Channel::NotionMarkdown => 65_536,
103 // 브라우저에는 메시지 한도가 없다. 나누지 않는다.
104 Channel::Html => usize::MAX,
105 }
106 }
107}
108
109/// The smallest part limit a caller can set.
110///
111/// **Each part needs room to close and reopen its markup.** If the limit is smaller than a tag,
112/// the splitter cuts between the tag's characters — splitting Telegram `**x**` with a limit of 1
113/// produced `<` · `b` · `></b>` (found in review). This value fits several nested opening tags
114/// (quote, bold, code, `<pre><code class="language-…">`) and still leaves room for content. It is
115/// far from real use (subtracting a header's share from Telegram's 4096).
116pub const MIN_LIMIT: usize = 256;
117
118/// Conversion options — the part limit and the browser channel's policy.
119///
120/// **There is no input-syntax choice.** Accepting LLM output even when it strays from the standard
121/// is the job of the default reader.
122///
123/// Fields may be added, so build it as `Options { limit, ..Default::default() }`.
124#[derive(Debug, Clone, PartialEq, Eq, Default)]
125pub struct Options {
126 /// The limit for one part (characters of rendered output). `None` means [`Channel::limit`].
127 ///
128 /// **The sender decides the limit.** Sometimes the channel cannot — plain is a fallback with
129 /// no known destination, so sending it to Telegram needs 4096 (a 7,153-character part split at
130 /// 12,000 got a 400), and a sender that prepends a title must use that much less.
131 /// Streaming ([`Streamer`]) does not split, so it ignores this value — except for Notion
132 /// tables. A table over the limit comes out as several tables that repeat the header row; that
133 /// is the table's shape, so streaming emits it the same way. The browser channel
134 /// ([`Channel::Html`]) does not split either — the splitter does not close and reopen block
135 /// tags, so it would cut through the middle of a tag. Values below [`MIN_LIMIT`] are raised
136 /// to it.
137 pub limit: Option<usize>,
138 /// The browser channel's ([`Channel::Html`]) policy. Other channels ignore it.
139 pub html: HtmlOptions,
140}
141
142/// The browser channel's policy. The defaults are the most conservative — `<br>` line breaks,
143/// images as links only, and only `http`, `https`, and `mailto` links.
144#[derive(Debug, Clone, PartialEq, Eq, Default)]
145pub struct HtmlOptions {
146 pub line_breaks: LineBreaks,
147 pub images: Images,
148 /// Schemes accepted for link and image URLs (without the colon, like `"https"`). `None`
149 /// means `http`, `https`, and `mailto`. A list accepts **only** those — it does not add to the
150 /// defaults.
151 pub schemes: Option<Vec<String>>,
152}
153
154/// How to emit line breaks inside a block.
155#[derive(Debug, Clone, Copy, PartialEq, Eq, Default)]
156pub enum LineBreaks {
157 /// `<br>` — for text where the author's line breaks carry meaning, like chat or notes. Every
158 /// other channel keeps line breaks.
159 #[default]
160 Br,
161 /// The newline character only — the browser folds it into a space. For reading a document
162 /// wrapped at 80 columns as paragraphs.
163 Space,
164}
165
166/// How to emit an image ``.
167#[derive(Debug, Clone, Copy, PartialEq, Eq, Default)]
168pub enum Images {
169 /// `<a href>alt</a>` — nothing loads until it is clicked (no tracking pixels).
170 #[default]
171 Link,
172 /// `<img src alt>` — only when the URL has an allowed scheme. Otherwise emitted like `Link`.
173 Load,
174}
175
176/// Counts of what normalization repaired and what was changed to fit the channel. The first four
177/// (repairs) measure **how often the model breaks formatting**; the last six (changes) measure
178/// **what a channel changes, before adopting it** — to ask about each separately, use
179/// [`Repairs::any`] and [`Repairs::changed`].
180#[derive(Debug, Clone, Copy, PartialEq, Eq, Default)]
181pub struct Repairs {
182 /// Emphasis left unclosed at the end of a block and closed for it (like `**영향 범위`).
183 pub closed_emphasis: usize,
184 /// Code fences left unclosed at the end of the document and closed for it.
185 pub closed_fence: usize,
186 /// Unmatched backtick runs turned back into text instead of code.
187 pub reverted_code_span: usize,
188 /// Orphaned `**` dropped (preceded by text, like `꼬리**`).
189 pub dropped_marker: usize,
190 /// Characters escaped because the channel would read them as syntax (GitHub's `\~`, `\<`,
191 /// `\*`).
192 pub escaped_char: usize,
193 /// Emphasis emitted differently because the channel cannot read the markers in that position
194 /// (`**「설정」**가`) — `<strong>` on GitHub, U+2060 inserted inside the markers on Slack.
195 pub tag_emphasis: usize,
196 /// Source HTML stripped — tags the channel cannot render, comments, and `<br>` turned into
197 /// line breaks.
198 pub stripped_html: usize,
199 /// List markers rewritten with a different symbol — bullets (`* `, `• ` → `- `; on Telegram
200 /// `- ` → `• `) and numbers (`1)` → `1.`).
201 pub rewritten_bullet: usize,
202 /// Tables rewritten into a different shape from the source (delimiter row and cell padding
203 /// normalized, or lowered to fixed width).
204 pub rewritten_table: usize,
205 /// Emphasis markers and links rewritten in a different notation — `_기울임_` → `*기울임*`,
206 /// `__굵게__` → `**굵게**`, `<url|텍스트>` → `[텍스트](url)`. Counted only on channels that
207 /// emit Markdown.
208 pub converted_marker: usize,
209}
210
211impl Repairs {
212 pub(crate) fn add(&mut self, other: Repairs) {
213 self.closed_emphasis += other.closed_emphasis;
214 self.closed_fence += other.closed_fence;
215 self.reverted_code_span += other.reverted_code_span;
216 self.dropped_marker += other.dropped_marker;
217 self.escaped_char += other.escaped_char;
218 self.tag_emphasis += other.tag_emphasis;
219 self.stripped_html += other.stripped_html;
220 self.rewritten_bullet += other.rewritten_bullet;
221 self.rewritten_table += other.rewritten_table;
222 self.converted_marker += other.converted_marker;
223 }
224
225 /// Whether normalization **repaired** anything — the first four (closed emphasis and fences,
226 /// reverted backticks, dropped markers). It asks whether the model broke formatting. Changes
227 /// made to fit the channel (escapes, bullets, tables …) are not counted — that is
228 /// [`Repairs::changed`].
229 pub fn any(&self) -> bool {
230 self.closed_emphasis + self.closed_fence + self.reverted_code_span + self.dropped_marker > 0
231 }
232
233 /// Whether **anything at all** was done, repair or channel change. It asks whether the output
234 /// may differ from the source (tidying such as collapsing blank lines is not counted —
235 /// `SPEC.md` 5.1).
236 pub fn changed(&self) -> bool {
237 *self != Repairs::default()
238 }
239}
240
241/// The result of [`render_with`] — the parts and the repairs.
242#[derive(Debug, Clone, PartialEq, Eq)]
243pub struct Rendered {
244 pub parts: Vec<String>,
245 pub repairs: Repairs,
246}
247
248/// Streaming converter.
249///
250/// Feed it chunks and it returns **only as much as is safe to emit right now**.
251/// Markup caught on a boundary (cut off at `**굵`) stays inside until the next chunk arrives.
252/// This is the core of the library — converting a finished document is a problem others have
253/// already solved, while the boundary problem shows up on every channel as long as you stream.
254///
255/// # Example
256///
257/// ```
258/// use mdwire::{Channel, Streamer};
259///
260/// let mut s = Streamer::new(Channel::TelegramHtml);
261/// let mut out = String::new();
262/// // A chunk boundary inside `**` never sends half a marker.
263/// out.push_str(s.push("앞말 **굵"));
264/// out.push_str(s.push("게** 뒷말"));
265/// out.push_str(s.finish());
266/// assert_eq!(out, "앞말 <b>굵게</b> 뒷말");
267/// ```
268pub struct Streamer {
269 engine: Engine,
270 /// [`Streamer::push`] 가 빌려주는 버퍼. 재사용하므로 조각마다 할당하지 않는다.
271 buf: String,
272 /// 줄바꿈이 아닌 글자를 하나라도 내보냈는가.
273 ///
274 /// **앞머리 빈 줄은 내보내지 않는다.** 문서가 주석이나 `<br>` 로 시작하면 첫 블록이
275 /// 비고 그 뒤의 줄바꿈만 남는데, 완성본은 조각 앞머리의 줄바꿈을 털고 시작한다
276 /// (`sink::PartsSink`). 스트리밍도 같아야 한다 — 그래야 둘이 같은 답을 낸다.
277 started: bool,
278 /// 마지막 [`Streamer::preview`] 의 꼬리. 재사용 버퍼 — 열린 블록만큼이지 문서 전체가 아니다.
279 tail: String,
280 /// `tail` 이 지금 상태를 반영하지 않는다 — 미리보기를 안 했거나 그 뒤에 조각이 더 왔다.
281 dirty: bool,
282 /// [`Streamer::revised`] 의 답. `finish` 가 정한다.
283 revised: bool,
284}
285
286impl Streamer {
287 pub fn new(channel: Channel) -> Self {
288 Self::with_options(channel, Options::default())
289 }
290
291 /// Builds one with options.
292 pub fn with_options(channel: Channel, options: Options) -> Self {
293 Self {
294 engine: Engine::new(channel, &options),
295 buf: String::new(),
296 started: false,
297 tail: String::new(),
298 dirty: true,
299 revised: true,
300 }
301 }
302
303 /// What normalization has repaired so far. After `finish`, it covers the whole document.
304 pub fn repairs(&self) -> Repairs {
305 self.engine.repairs()
306 }
307
308 /// `from` 뒤에 새로 붙은 출력에서 앞머리 줄바꿈을 턴다. 첫 글자가 나올 때까지만이다.
309 fn trim_leading(&mut self, out: &mut String, from: usize) {
310 if self.started {
311 return;
312 }
313 let fresh = &out[from..];
314 let keep = fresh.len() - fresh.trim_start_matches('\n').len();
315 if keep > 0 {
316 out.drain(from..from + keep);
317 }
318 if out.len() > from {
319 self.started = true;
320 }
321 }
322
323 /// Pushes a chunk and returns the output that can be emitted now.
324 ///
325 /// The returned slice is valid **only until the next call**. To avoid allocation entirely,
326 /// use [`Streamer::push_into`] — both call the same code (`SPEC.md` section 5).
327 pub fn push(&mut self, chunk: &str) -> &str {
328 let mut buf = std::mem::take(&mut self.buf);
329 buf.clear();
330 self.push_into(chunk, &mut buf);
331 self.buf = buf;
332 &self.buf
333 }
334
335 /// Writes directly into the caller's buffer. The canonical signature — zero allocations per
336 /// chunk.
337 pub fn push_into(&mut self, chunk: &str, out: &mut String) {
338 self.dirty |= !chunk.is_empty();
339 let from = out.len();
340 let mut sink = StringSink(out);
341 self.engine.feed(chunk, &mut sink);
342 self.trim_leading(out, from);
343 }
344
345 /// Signals the end of input. Emits everything left (open markup is closed).
346 pub fn finish(&mut self) -> &str {
347 let mut buf = std::mem::take(&mut self.buf);
348 buf.clear();
349 self.finish_into(&mut buf);
350 self.buf = buf;
351 &self.buf
352 }
353
354 /// The allocation-free version of [`Streamer::finish`].
355 pub fn finish_into(&mut self, out: &mut String) {
356 let from = out.len();
357 let mut sink = StringSink(out);
358 self.engine.finish(&mut sink);
359 self.trim_leading(out, from);
360 self.revised = self.dirty || self.tail != out[from..];
361 // 끝난 엔진에 더 그릴 꼬리는 없다. 비워 두지 않으면 뒤이은 `preview` 가 끝난 엔진을
362 // 복제해 finish 꼬리를 한 번 더 낸다.
363 self.tail.clear();
364 self.dirty = false;
365 }
366
367 /// **The tail that would follow the final output if input ended now.** Appending it to the
368 /// accumulated output gives a shape that can be sent on the spot — it goes where
369 /// [`Streamer::close_open`] goes, but also renders what is being held back: open emphasis
370 /// closed (`**굵` → `<b>굵</b>`), tables with the rows received so far, code spans closed.
371 /// It is the default for callers that redraw the whole accumulated output (React, Telegram
372 /// `editMessageText`, Slack `chat.update`).
373 ///
374 /// The tail takes the same `finish` path as batch rendering, so its syntax is always valid. But
375 /// it is a **guess** — later chunks can change the shape, for example a code span that never
376 /// closes turns back into text. After finishing, [`Streamer::revised`] tells you whether the
377 /// result differs from the last preview. Do not put it into the accumulated output itself.
378 ///
379 /// The cost is proportional to the size of the currently open block (the engine is cloned).
380 /// Call it when drawing the screen, not on every chunk.
381 ///
382 /// ```
383 /// use mdwire::{Channel, Streamer};
384 ///
385 /// let mut s = Streamer::new(Channel::TelegramHtml);
386 /// let mut acc = String::new();
387 /// s.push_into("앞말 **굵", &mut acc);
388 /// assert_eq!(acc, "앞말 "); // final output so far
389 /// assert_eq!(format!("{acc}{}", s.preview()), "앞말 <b>굵</b>");
390 ///
391 /// s.push_into("게** 끝", &mut acc);
392 /// let last = format!("{acc}{}", s.preview());
393 /// s.finish_into(&mut acc);
394 /// assert_eq!(acc, last);
395 /// assert!(!s.revised()); // the last frame is already the result
396 /// ```
397 pub fn preview(&mut self) -> &str {
398 // 그 뒤로 조각이 안 왔으면 같은 답이다 — 다시 그리는 쪽은 조각과 무관하게도 자주 부른다.
399 if !self.dirty {
400 return &self.tail;
401 }
402 let mut tail = std::mem::take(&mut self.tail);
403 tail.clear();
404 self.engine.preview(&mut StringSink(&mut tail));
405 if !self.started {
406 let keep = tail.len() - tail.trim_start_matches('\n').len();
407 tail.drain(..keep);
408 }
409 self.tail = tail;
410 self.dirty = false;
411 &self.tail
412 }
413
414 /// Appends [`Streamer::preview`] to the caller's buffer.
415 pub fn preview_into(&mut self, out: &mut String) {
416 out.push_str(self.preview());
417 }
418
419 /// **Whether the final output differs from the last preview** — check it after `finish`. If
420 /// false, the last screen drawn (accumulated output + `preview`) already is the final output, so no
421 /// redraw is needed. Telegram returns 400 ("message is not modified") for an edit with the same
422 /// content, so use this to skip the last edit. It is true if there was no preview or more
423 /// chunks arrived after it — **true means "may differ"**. If you throttle edits so the last
424 /// preview comes before the last chunk, it is true even when the final output is the same;
425 /// in that case compare against the last string you sent.
426 pub fn revised(&self) -> bool {
427 self.revised
428 }
429
430 /// **Makes what has been received so far safe to send as is.** It does not touch the state,
431 /// so streaming continues after appending it.
432 ///
433 /// Emphasis is held inside until its pair arrives, so it is already balanced, but a block's
434 /// opening markup (`<blockquote>`, `<pre>`, a heading's `<b>`) goes out before the block
435 /// ends — holding output until a code block ends would not be streaming. Callers that send
436 /// the accumulated output to a channel midway (editing a message as tokens arrive) append this
437 /// right before sending. **Do not put it into the accumulated output itself** — the next chunk
438 /// continues from there.
439 ///
440 /// ```
441 /// use mdwire::{Channel, Streamer};
442 ///
443 /// let mut s = Streamer::new(Channel::TelegramHtml);
444 /// let mut acc = String::new();
445 /// s.push_into("> 인용이 시작되고", &mut acc);
446 ///
447 /// let mut snapshot = acc.clone();
448 /// s.close_open(&mut snapshot); // safe to send now
449 /// assert_eq!(snapshot, "<blockquote>인용이 시작되고</blockquote>");
450 ///
451 /// s.push_into("\n> 이어진다\n", &mut acc); // the accumulated output just continues
452 /// s.finish_into(&mut acc);
453 /// assert_eq!(acc, "<blockquote>인용이 시작되고\n이어진다</blockquote>");
454 /// ```
455 pub fn close_open(&self, out: &mut String) {
456 self.engine.close_open(out);
457 }
458}
459
460/// Converts a finished document in one go. Splits at safe points when it exceeds the limit.
461///
462/// Split points are chosen **from the structure, not the rendered output** — only points where a
463/// block has ended and no markup is open become boundaries. Cutting by character count after
464/// conversion leaves `<code>` open across the cut, and the channel returns 400 (`DESIGN.md`).
465///
466/// # Example
467///
468/// ```
469/// use mdwire::{render, Channel};
470///
471/// let parts = render("## 제목\n\n**굵게** 있는 문단", Channel::TelegramHtml);
472/// assert_eq!(parts, vec!["<b>제목</b>\n\n<b>굵게</b> 있는 문단"]);
473/// ```
474pub fn render(input: &str, channel: Channel) -> Vec<String> {
475 render_with(input, channel, Options::default()).parts
476}
477
478/// [`render`] with options; also returns what normalization repaired.
479///
480/// ```
481/// use mdwire::{render_with, Channel, Options};
482///
483/// let options = Options { limit: Some(4096), ..Default::default() };
484/// let out = render_with("**굵게** 는 **영향 범위", Channel::SlackMarkdown, options);
485/// assert_eq!(out.parts, vec!["**굵게** 는 **영향 범위**"]);
486/// assert_eq!(out.repairs.closed_emphasis, 1);
487/// ```
488pub fn render_with(input: &str, channel: Channel, options: Options) -> Rendered {
489 let mut engine = Engine::new(channel, &options);
490 let mut sink = PartsSink::new(Vocab::from_options(channel, &options));
491 engine.feed(input, &mut sink);
492 engine.finish(&mut sink);
493 Rendered { repairs: engine.repairs(), parts: sink.into_parts() }
494}
495
496#[cfg(test)]
497mod tests {
498 use super::*;
499
500 #[test]
501 fn channel_limits_are_channel_specific() {
502 assert_eq!(Channel::TelegramHtml.limit(), 4096);
503 assert_eq!(Channel::SlackMarkdown.limit(), 12_000);
504 assert_eq!(Channel::GithubMarkdown.limit(), 65_536);
505 }
506
507 /// **조각 하나가 곧 메시지 하나다.** 태그 한가운데서 끊으면 조각이 `… <a ` 로
508 /// 끝나고, 채널은 그 메시지를 통째로 거절한다. 실제 문서(링크가 달린 긴 목록)에서
509 /// 나온 고장이다.
510 #[test]
511 fn parts_never_end_inside_a_tag() {
512 let mut input = String::new();
513 for i in 0..40 {
514 input.push_str(&format!(
515 "- `method{i}(options?: SomeLongOptionsType{i}): string` — 주소의 뒤집힌 꼴을 \
516 돌려준다 [src](https://example.com/owner/repo/blob/master/src/mod{i}.ts#L{i}28)\n"
517 ));
518 }
519 let parts = render(&input, Channel::TelegramHtml);
520 assert!(parts.len() > 1, "한도를 넘겨서 나뉘어야 하는 입력이다");
521 for (i, p) in parts.iter().enumerate() {
522 assert_eq!(
523 p.matches('<').count(),
524 p.matches('>').count(),
525 "조각 {i} 가 태그 한가운데서 끊겼다: …{}",
526 &p[p.len().saturating_sub(40)..]
527 );
528 assert!(p.chars().count() <= Channel::TelegramHtml.limit(), "조각 {i} 가 한도를 넘었다");
529 }
530 }
531}