Skip to main content

mdwire/
lib.rs

1//! mdwire — safely deliver agent-generated Markdown to chat channels.
2//!
3//! It does three things in one pipeline. The order is the design.
4//!
5//! 1. **Normalize** — LLM output is not valid CommonMark. Unbalanced emphasis,
6//!    emphasis that spans lines, and unclosed code fences are routine. Repair them first.
7//! 2. **Render for the channel** — translate into the syntax the target accepts. Syntax the
8//!    target cannot accept is not worth parsing in the first place (see the `Channel` docs below).
9//! 3. **Split safely** — at channel limits and streaming boundaries, never cut through markup.
10//!
11//! # Why no dependencies
12//!
13//! Existing parsers are all **batch** parsers — they take the whole document, build an AST,
14//! then render. That does not fit a problem where output must go out as tokens stream in.
15//! And the CJK-adjacent emphasis policy is baked into the parser, so with someone else's
16//! parser you cannot change it. That is exactly the problem this library sets out to fix.
17
18#![forbid(unsafe_code)]
19
20mod block;
21mod inline;
22mod sink;
23mod vocab;
24pub mod width;
25
26use block::Engine;
27use sink::{PartsSink, StringSink};
28use vocab::Vocab;
29
30/// Target channel. Each channel accepts a different syntax, and **the narrower output decides the
31/// parsing scope**.
32#[derive(Debug, Clone, Copy, PartialEq, Eq)]
33pub enum Channel {
34    /// Telegram `parse_mode=HTML`. Nine allowed tags:
35    /// `b i u s code pre a blockquote tg-spoiler`. No tables or headings. 4096 characters.
36    TelegramHtml,
37    /// Slack `markdown_text`. Slack converts standard Markdown itself. 12,000 characters.
38    /// Almost no conversion is needed; what is left is normalization and splitting.
39    SlackMarkdown,
40    /// GitHub comments and PR bodies (GFM). Renders tables, headings, and strikethrough.
41    /// 65,536 characters. Emits the same Markdown as Slack, but escapes the two characters GFM
42    /// reads as syntax (`~` `<`).
43    GithubMarkdown,
44    /// Notion page body (Notion-flavored Markdown — the API `markdown` field and connectors).
45    /// Four heading levels. Emits the same Markdown as GitHub, but strips inline HTML Notion
46    /// cannot render (it would show as text), and writes autolinks `<url>` as `[url](url)` (the
47    /// angle brackets would remain as text). Emphasis before a Korean particle is rendered by
48    /// Notion as is, so it is not turned into `<strong>` — doing so would make the tag show as text.
49    NotionMarkdown,
50    /// Strips all markup. The fallback path.
51    Plain,
52    /// An HTML fragment for the browser. Renders headings, lists, tables, and code blocks as tags.
53    /// No limit.
54    ///
55    /// Assumes the output goes straight into `innerHTML` — all text is escaped, HTML from the
56    /// source survives only as inline tags with their attributes dropped, and only `http(s)` and
57    /// `mailto` links become `<a>`. Appending [`Streamer::close_open`] to the streaming
58    /// accumulated output gives a shape that can be inserted as is.
59    Html,
60}
61
62impl Channel {
63    /// The name used for corpus directories and CLI arguments. The source of truth wherever a
64    /// channel is handled as a string.
65    pub fn name(self) -> &'static str {
66        match self {
67            Channel::TelegramHtml => "telegram-html",
68            Channel::SlackMarkdown => "slack-markdown",
69            Channel::GithubMarkdown => "github-markdown",
70            Channel::NotionMarkdown => "notion-markdown",
71            Channel::Plain => "plain",
72            Channel::Html => "html",
73        }
74    }
75
76    /// Every supported channel. The corpus and the harness iterate over this list.
77    pub fn all() -> [Channel; 6] {
78        [
79            Channel::TelegramHtml,
80            Channel::SlackMarkdown,
81            Channel::GithubMarkdown,
82            Channel::NotionMarkdown,
83            Channel::Plain,
84            Channel::Html,
85        ]
86    }
87
88    /// Looks up a channel by name.
89    pub fn parse(name: &str) -> Option<Channel> {
90        Self::all().into_iter().find(|c| c.name() == name)
91    }
92
93    /// This channel's message length limit, in characters. Splitting is based on it.
94    pub fn limit(self) -> usize {
95        match self {
96            Channel::TelegramHtml => 4096,
97            Channel::SlackMarkdown | Channel::Plain => 12_000,
98            // 코멘트 본문의 한도다. 넘기면 API 가 422 로 거절한다("Body is too long").
99            Channel::GithubMarkdown => 65_536,
100            // **재지 않았다.** 페이지 하나에 들어갈 본문이라 GitHub 과 같은 값을 둔다. 보내는 쪽
101            // 한도가 따로 있으면 [`Options::limit`] 으로 준다.
102            Channel::NotionMarkdown => 65_536,
103            // 브라우저에는 메시지 한도가 없다. 나누지 않는다.
104            Channel::Html => usize::MAX,
105        }
106    }
107}
108
109/// The smallest part limit a caller can set.
110///
111/// **Each part needs room to close and reopen its markup.** If the limit is smaller than a tag,
112/// the splitter cuts between the tag's characters — splitting Telegram `**x**` with a limit of 1
113/// produced `<` · `b` · `></b>` (found in review). This value fits several nested opening tags
114/// (quote, bold, code, `<pre><code class="language-…">`) and still leaves room for content. It is
115/// far from real use (subtracting a header's share from Telegram's 4096).
116pub const MIN_LIMIT: usize = 256;
117
118/// Conversion options — the part limit and the browser channel's policy.
119///
120/// **There is no input-syntax choice.** Accepting LLM output even when it strays from the standard
121/// is the job of the default reader.
122///
123/// Fields may be added, so build it as `Options { limit, ..Default::default() }`.
124#[derive(Debug, Clone, PartialEq, Eq, Default)]
125pub struct Options {
126    /// The limit for one part (characters of rendered output). `None` means [`Channel::limit`].
127    ///
128    /// **The sender decides the limit.** Sometimes the channel cannot — plain is a fallback with
129    /// no known destination, so sending it to Telegram needs 4096 (a 7,153-character part split at
130    /// 12,000 got a 400), and a sender that prepends a title must use that much less.
131    /// Streaming ([`Streamer`]) does not split, so it ignores this value — except for Notion
132    /// tables. A table over the limit comes out as several tables that repeat the header row; that
133    /// is the table's shape, so streaming emits it the same way. The browser channel
134    /// ([`Channel::Html`]) does not split either — the splitter does not close and reopen block
135    /// tags, so it would cut through the middle of a tag. Values below [`MIN_LIMIT`] are raised
136    /// to it.
137    pub limit: Option<usize>,
138    /// The browser channel's ([`Channel::Html`]) policy. Other channels ignore it.
139    pub html: HtmlOptions,
140}
141
142/// The browser channel's policy. The defaults are the most conservative — `<br>` line breaks,
143/// images as links only, and only `http`, `https`, and `mailto` links.
144#[derive(Debug, Clone, PartialEq, Eq, Default)]
145pub struct HtmlOptions {
146    pub line_breaks: LineBreaks,
147    pub images: Images,
148    /// Schemes accepted for link and image URLs (without the colon, like `"https"`). `None`
149    /// means `http`, `https`, and `mailto`. A list accepts **only** those — it does not add to the
150    /// defaults.
151    pub schemes: Option<Vec<String>>,
152}
153
154/// How to emit line breaks inside a block.
155#[derive(Debug, Clone, Copy, PartialEq, Eq, Default)]
156pub enum LineBreaks {
157    /// `<br>` — for text where the author's line breaks carry meaning, like chat or notes. Every
158    /// other channel keeps line breaks.
159    #[default]
160    Br,
161    /// The newline character only — the browser folds it into a space. For reading a document
162    /// wrapped at 80 columns as paragraphs.
163    Space,
164}
165
166/// How to emit an image `![alt](url)`.
167#[derive(Debug, Clone, Copy, PartialEq, Eq, Default)]
168pub enum Images {
169    /// `<a href>alt</a>` — nothing loads until it is clicked (no tracking pixels).
170    #[default]
171    Link,
172    /// `<img src alt>` — only when the URL has an allowed scheme. Otherwise emitted like `Link`.
173    Load,
174}
175
176/// Counts of what normalization repaired and what was changed to fit the channel. The first four
177/// (repairs) measure **how often the model breaks formatting**; the last six (changes) measure
178/// **what a channel changes, before adopting it** — to ask about each separately, use
179/// [`Repairs::any`] and [`Repairs::changed`].
180#[derive(Debug, Clone, Copy, PartialEq, Eq, Default)]
181pub struct Repairs {
182    /// Emphasis left unclosed at the end of a block and closed for it (like `**영향 범위`).
183    pub closed_emphasis: usize,
184    /// Code fences left unclosed at the end of the document and closed for it.
185    pub closed_fence: usize,
186    /// Unmatched backtick runs turned back into text instead of code.
187    pub reverted_code_span: usize,
188    /// Orphaned `**` dropped (preceded by text, like `꼬리**`).
189    pub dropped_marker: usize,
190    /// Characters escaped because the channel would read them as syntax (GitHub's `\~`, `\<`,
191    /// `\*`).
192    pub escaped_char: usize,
193    /// Emphasis emitted differently because the channel cannot read the markers in that position
194    /// (`**「설정」**가`) — `<strong>` on GitHub, U+2060 inserted inside the markers on Slack.
195    pub tag_emphasis: usize,
196    /// Source HTML stripped — tags the channel cannot render, comments, and `<br>` turned into
197    /// line breaks.
198    pub stripped_html: usize,
199    /// List markers rewritten with a different symbol — bullets (`* `, `• ` → `- `; on Telegram
200    /// `- ` → `• `) and numbers (`1)` → `1.`).
201    pub rewritten_bullet: usize,
202    /// Tables rewritten into a different shape from the source (delimiter row and cell padding
203    /// normalized, or lowered to fixed width).
204    pub rewritten_table: usize,
205    /// Emphasis markers and links rewritten in a different notation — `_기울임_` → `*기울임*`,
206    /// `__굵게__` → `**굵게**`, `<url|텍스트>` → `[텍스트](url)`. Counted only on channels that
207    /// emit Markdown.
208    pub converted_marker: usize,
209}
210
211impl Repairs {
212    pub(crate) fn add(&mut self, other: Repairs) {
213        self.closed_emphasis += other.closed_emphasis;
214        self.closed_fence += other.closed_fence;
215        self.reverted_code_span += other.reverted_code_span;
216        self.dropped_marker += other.dropped_marker;
217        self.escaped_char += other.escaped_char;
218        self.tag_emphasis += other.tag_emphasis;
219        self.stripped_html += other.stripped_html;
220        self.rewritten_bullet += other.rewritten_bullet;
221        self.rewritten_table += other.rewritten_table;
222        self.converted_marker += other.converted_marker;
223    }
224
225    /// Whether normalization **repaired** anything — the first four (closed emphasis and fences,
226    /// reverted backticks, dropped markers). It asks whether the model broke formatting. Changes
227    /// made to fit the channel (escapes, bullets, tables …) are not counted — that is
228    /// [`Repairs::changed`].
229    pub fn any(&self) -> bool {
230        self.closed_emphasis + self.closed_fence + self.reverted_code_span + self.dropped_marker > 0
231    }
232
233    /// Whether **anything at all** was done, repair or channel change. It asks whether the output
234    /// may differ from the source (tidying such as collapsing blank lines is not counted —
235    /// `SPEC.md` 5.1).
236    pub fn changed(&self) -> bool {
237        *self != Repairs::default()
238    }
239}
240
241/// The result of [`render_with`] — the parts and the repairs.
242#[derive(Debug, Clone, PartialEq, Eq)]
243pub struct Rendered {
244    pub parts: Vec<String>,
245    pub repairs: Repairs,
246}
247
248/// Streaming converter.
249///
250/// Feed it chunks and it returns **only as much as is safe to emit right now**.
251/// Markup caught on a boundary (cut off at `**굵`) stays inside until the next chunk arrives.
252/// This is the core of the library — converting a finished document is a problem others have
253/// already solved, while the boundary problem shows up on every channel as long as you stream.
254///
255/// # Example
256///
257/// ```
258/// use mdwire::{Channel, Streamer};
259///
260/// let mut s = Streamer::new(Channel::TelegramHtml);
261/// let mut out = String::new();
262/// // A chunk boundary inside `**` never sends half a marker.
263/// out.push_str(s.push("앞말 **굵"));
264/// out.push_str(s.push("게** 뒷말"));
265/// out.push_str(s.finish());
266/// assert_eq!(out, "앞말 <b>굵게</b> 뒷말");
267/// ```
268pub struct Streamer {
269    engine: Engine,
270    /// [`Streamer::push`] 가 빌려주는 버퍼. 재사용하므로 조각마다 할당하지 않는다.
271    buf: String,
272    /// 줄바꿈이 아닌 글자를 하나라도 내보냈는가.
273    ///
274    /// **앞머리 빈 줄은 내보내지 않는다.** 문서가 주석이나 `<br>` 로 시작하면 첫 블록이
275    /// 비고 그 뒤의 줄바꿈만 남는데, 완성본은 조각 앞머리의 줄바꿈을 털고 시작한다
276    /// (`sink::PartsSink`). 스트리밍도 같아야 한다 — 그래야 둘이 같은 답을 낸다.
277    started: bool,
278    /// 마지막 [`Streamer::preview`] 의 꼬리. 재사용 버퍼 — 열린 블록만큼이지 문서 전체가 아니다.
279    tail: String,
280    /// `tail` 이 지금 상태를 반영하지 않는다 — 미리보기를 안 했거나 그 뒤에 조각이 더 왔다.
281    dirty: bool,
282    /// [`Streamer::revised`] 의 답. `finish` 가 정한다.
283    revised: bool,
284}
285
286impl Streamer {
287    pub fn new(channel: Channel) -> Self {
288        Self::with_options(channel, Options::default())
289    }
290
291    /// Builds one with options.
292    pub fn with_options(channel: Channel, options: Options) -> Self {
293        Self {
294            engine: Engine::new(channel, &options),
295            buf: String::new(),
296            started: false,
297            tail: String::new(),
298            dirty: true,
299            revised: true,
300        }
301    }
302
303    /// What normalization has repaired so far. After `finish`, it covers the whole document.
304    pub fn repairs(&self) -> Repairs {
305        self.engine.repairs()
306    }
307
308    /// `from` 뒤에 새로 붙은 출력에서 앞머리 줄바꿈을 턴다. 첫 글자가 나올 때까지만이다.
309    fn trim_leading(&mut self, out: &mut String, from: usize) {
310        if self.started {
311            return;
312        }
313        let fresh = &out[from..];
314        let keep = fresh.len() - fresh.trim_start_matches('\n').len();
315        if keep > 0 {
316            out.drain(from..from + keep);
317        }
318        if out.len() > from {
319            self.started = true;
320        }
321    }
322
323    /// Pushes a chunk and returns the output that can be emitted now.
324    ///
325    /// The returned slice is valid **only until the next call**. To avoid allocation entirely,
326    /// use [`Streamer::push_into`] — both call the same code (`SPEC.md` section 5).
327    pub fn push(&mut self, chunk: &str) -> &str {
328        let mut buf = std::mem::take(&mut self.buf);
329        buf.clear();
330        self.push_into(chunk, &mut buf);
331        self.buf = buf;
332        &self.buf
333    }
334
335    /// Writes directly into the caller's buffer. The canonical signature — zero allocations per
336    /// chunk.
337    pub fn push_into(&mut self, chunk: &str, out: &mut String) {
338        self.dirty |= !chunk.is_empty();
339        let from = out.len();
340        let mut sink = StringSink(out);
341        self.engine.feed(chunk, &mut sink);
342        self.trim_leading(out, from);
343    }
344
345    /// Signals the end of input. Emits everything left (open markup is closed).
346    pub fn finish(&mut self) -> &str {
347        let mut buf = std::mem::take(&mut self.buf);
348        buf.clear();
349        self.finish_into(&mut buf);
350        self.buf = buf;
351        &self.buf
352    }
353
354    /// The allocation-free version of [`Streamer::finish`].
355    pub fn finish_into(&mut self, out: &mut String) {
356        let from = out.len();
357        let mut sink = StringSink(out);
358        self.engine.finish(&mut sink);
359        self.trim_leading(out, from);
360        self.revised = self.dirty || self.tail != out[from..];
361        // 끝난 엔진에 더 그릴 꼬리는 없다. 비워 두지 않으면 뒤이은 `preview` 가 끝난 엔진을
362        // 복제해 finish 꼬리를 한 번 더 낸다.
363        self.tail.clear();
364        self.dirty = false;
365    }
366
367    /// **The tail that would follow the final output if input ended now.** Appending it to the
368    /// accumulated output gives a shape that can be sent on the spot — it goes where
369    /// [`Streamer::close_open`] goes, but also renders what is being held back: open emphasis
370    /// closed (`**굵` → `<b>굵</b>`), tables with the rows received so far, code spans closed.
371    /// It is the default for callers that redraw the whole accumulated output (React, Telegram
372    /// `editMessageText`, Slack `chat.update`).
373    ///
374    /// The tail takes the same `finish` path as batch rendering, so its syntax is always valid. But
375    /// it is a **guess** — later chunks can change the shape, for example a code span that never
376    /// closes turns back into text. After finishing, [`Streamer::revised`] tells you whether the
377    /// result differs from the last preview. Do not put it into the accumulated output itself.
378    ///
379    /// The cost is proportional to the size of the currently open block (the engine is cloned).
380    /// Call it when drawing the screen, not on every chunk.
381    ///
382    /// ```
383    /// use mdwire::{Channel, Streamer};
384    ///
385    /// let mut s = Streamer::new(Channel::TelegramHtml);
386    /// let mut acc = String::new();
387    /// s.push_into("앞말 **굵", &mut acc);
388    /// assert_eq!(acc, "앞말 ");                       // final output so far
389    /// assert_eq!(format!("{acc}{}", s.preview()), "앞말 <b>굵</b>");
390    ///
391    /// s.push_into("게** 끝", &mut acc);
392    /// let last = format!("{acc}{}", s.preview());
393    /// s.finish_into(&mut acc);
394    /// assert_eq!(acc, last);
395    /// assert!(!s.revised());                          // the last frame is already the result
396    /// ```
397    pub fn preview(&mut self) -> &str {
398        // 그 뒤로 조각이 안 왔으면 같은 답이다 — 다시 그리는 쪽은 조각과 무관하게도 자주 부른다.
399        if !self.dirty {
400            return &self.tail;
401        }
402        let mut tail = std::mem::take(&mut self.tail);
403        tail.clear();
404        self.engine.preview(&mut StringSink(&mut tail));
405        if !self.started {
406            let keep = tail.len() - tail.trim_start_matches('\n').len();
407            tail.drain(..keep);
408        }
409        self.tail = tail;
410        self.dirty = false;
411        &self.tail
412    }
413
414    /// Appends [`Streamer::preview`] to the caller's buffer.
415    pub fn preview_into(&mut self, out: &mut String) {
416        out.push_str(self.preview());
417    }
418
419    /// **Whether the final output differs from the last preview** — check it after `finish`. If
420    /// false, the last screen drawn (accumulated output + `preview`) already is the final output, so no
421    /// redraw is needed. Telegram returns 400 ("message is not modified") for an edit with the same
422    /// content, so use this to skip the last edit. It is true if there was no preview or more
423    /// chunks arrived after it — **true means "may differ"**. If you throttle edits so the last
424    /// preview comes before the last chunk, it is true even when the final output is the same;
425    /// in that case compare against the last string you sent.
426    pub fn revised(&self) -> bool {
427        self.revised
428    }
429
430    /// **Makes what has been received so far safe to send as is.** It does not touch the state,
431    /// so streaming continues after appending it.
432    ///
433    /// Emphasis is held inside until its pair arrives, so it is already balanced, but a block's
434    /// opening markup (`<blockquote>`, `<pre>`, a heading's `<b>`) goes out before the block
435    /// ends — holding output until a code block ends would not be streaming. Callers that send
436    /// the accumulated output to a channel midway (editing a message as tokens arrive) append this
437    /// right before sending. **Do not put it into the accumulated output itself** — the next chunk
438    /// continues from there.
439    ///
440    /// ```
441    /// use mdwire::{Channel, Streamer};
442    ///
443    /// let mut s = Streamer::new(Channel::TelegramHtml);
444    /// let mut acc = String::new();
445    /// s.push_into("> 인용이 시작되고", &mut acc);
446    ///
447    /// let mut snapshot = acc.clone();
448    /// s.close_open(&mut snapshot);          // safe to send now
449    /// assert_eq!(snapshot, "<blockquote>인용이 시작되고</blockquote>");
450    ///
451    /// s.push_into("\n> 이어진다\n", &mut acc);  // the accumulated output just continues
452    /// s.finish_into(&mut acc);
453    /// assert_eq!(acc, "<blockquote>인용이 시작되고\n이어진다</blockquote>");
454    /// ```
455    pub fn close_open(&self, out: &mut String) {
456        self.engine.close_open(out);
457    }
458}
459
460/// Converts a finished document in one go. Splits at safe points when it exceeds the limit.
461///
462/// Split points are chosen **from the structure, not the rendered output** — only points where a
463/// block has ended and no markup is open become boundaries. Cutting by character count after
464/// conversion leaves `<code>` open across the cut, and the channel returns 400 (`DESIGN.md`).
465///
466/// # Example
467///
468/// ```
469/// use mdwire::{render, Channel};
470///
471/// let parts = render("## 제목\n\n**굵게** 있는 문단", Channel::TelegramHtml);
472/// assert_eq!(parts, vec!["<b>제목</b>\n\n<b>굵게</b> 있는 문단"]);
473/// ```
474pub fn render(input: &str, channel: Channel) -> Vec<String> {
475    render_with(input, channel, Options::default()).parts
476}
477
478/// [`render`] with options; also returns what normalization repaired.
479///
480/// ```
481/// use mdwire::{render_with, Channel, Options};
482///
483/// let options = Options { limit: Some(4096), ..Default::default() };
484/// let out = render_with("**굵게** 는 **영향 범위", Channel::SlackMarkdown, options);
485/// assert_eq!(out.parts, vec!["**굵게** 는 **영향 범위**"]);
486/// assert_eq!(out.repairs.closed_emphasis, 1);
487/// ```
488pub fn render_with(input: &str, channel: Channel, options: Options) -> Rendered {
489    let mut engine = Engine::new(channel, &options);
490    let mut sink = PartsSink::new(Vocab::from_options(channel, &options));
491    engine.feed(input, &mut sink);
492    engine.finish(&mut sink);
493    Rendered { repairs: engine.repairs(), parts: sink.into_parts() }
494}
495
496#[cfg(test)]
497mod tests {
498    use super::*;
499
500    #[test]
501    fn channel_limits_are_channel_specific() {
502        assert_eq!(Channel::TelegramHtml.limit(), 4096);
503        assert_eq!(Channel::SlackMarkdown.limit(), 12_000);
504        assert_eq!(Channel::GithubMarkdown.limit(), 65_536);
505    }
506
507    /// **조각 하나가 곧 메시지 하나다.** 태그 한가운데서 끊으면 조각이 `… <a ` 로
508    /// 끝나고, 채널은 그 메시지를 통째로 거절한다. 실제 문서(링크가 달린 긴 목록)에서
509    /// 나온 고장이다.
510    #[test]
511    fn parts_never_end_inside_a_tag() {
512        let mut input = String::new();
513        for i in 0..40 {
514            input.push_str(&format!(
515                "- `method{i}(options?: SomeLongOptionsType{i}): string` — 주소의 뒤집힌 꼴을 \
516                 돌려준다 [src](https://example.com/owner/repo/blob/master/src/mod{i}.ts#L{i}28)\n"
517            ));
518        }
519        let parts = render(&input, Channel::TelegramHtml);
520        assert!(parts.len() > 1, "한도를 넘겨서 나뉘어야 하는 입력이다");
521        for (i, p) in parts.iter().enumerate() {
522            assert_eq!(
523                p.matches('<').count(),
524                p.matches('>').count(),
525                "조각 {i} 가 태그 한가운데서 끊겼다: …{}",
526                &p[p.len().saturating_sub(40)..]
527            );
528            assert!(p.chars().count() <= Channel::TelegramHtml.limit(), "조각 {i} 가 한도를 넘었다");
529        }
530    }
531}