- 1
//! Telegram-safe HTML projection and standalone chunking. - 2
- 3
/// Convert a conservative GFM subset to Telegram-safe HTML. - 4
/// - 5
/// Supports: headings (as bold), bold/italic/strikethrough/spoiler, inline - 6
/// code and fenced code blocks (with language hint preserved), links, - 7
/// blockquotes (merged across consecutive lines), bulleted and numbered - 8
/// lists (including nesting by indentation), horizontal rules, and - 9
/// GFM tables (rendered as an aligned monospace block). - 10
pub fn markdown_to_html(markdown: &str) -> String { - 11
let mut out = String::with_capacity(markdown.len() + 64); - 12
let mut in_fence = false; - 13
let mut fence_lang = String::new(); - 14
let mut fence = Vec::new(); - 15
let mut table = Vec::new(); - 16
let mut quote = Vec::new(); - 17
let mut blank_pending = false; - 18
- 19
for line in markdown.lines() { - 20
let trimmed = line.trim_start(); - 21
- 22
if let Some(lang) = trimmed.strip_prefix("```") { - 23
flush_quote(&mut out, &mut quote); - 24
flush_table(&mut out, &mut table); - 25
if in_fence { - 26
out.push_str("<pre>"); - 27
if fence_lang.is_empty() { - 28
out.push_str("<code>"); - 29
} else { - 30
out.push_str("<code class=\"language-"); - 31
out.push_str(&escape_html(&fence_lang)); - 32
out.push_str("\">"); - 33
} - 34
out.push_str(&escape_html(&fence.join("\n"))); - 35
out.push_str("</code></pre>\n"); - 36
fence.clear(); - 37
fence_lang.clear(); - 38
} else { - 39
fence_lang = lang.trim().to_string(); - 40
} - 41
in_fence = !in_fence; - 42
blank_pending = false; - 43
continue; - 44
} - 45
if in_fence { - 46
fence.push(line.to_string()); - 47
continue; - 48
} - 49
- 50
if trimmed.is_empty() { - 51
flush_quote(&mut out, &mut quote); - 52
flush_table(&mut out, &mut table); - 53
blank_pending = !out.is_empty(); - 54
continue; - 55
} - 56
- 57
if is_spoiler(trimmed) { - 58
flush_quote(&mut out, &mut quote); - 59
flush_table(&mut out, &mut table); - 60
if blank_pending && !out.is_empty() { - 61
out.push('\n'); - 62
} - 63
blank_pending = false; - 64
out.push_str("<tg-spoiler>"); - 65
out.push_str(&inline_markdown(&escape_html(trimmed))); - 66
out.push_str("</tg-spoiler>\n"); - 67
continue; - 68
} - 69
- 70
if trimmed.starts_with('|') && trimmed.ends_with('|') && trimmed.len() > 1 { - 71
flush_quote(&mut out, &mut quote); - 72
if blank_pending { - 73
out.push('\n'); - 74
} - 75
blank_pending = false; - 76
table.push(trimmed.to_string()); - 77
continue; - 78
} - 79
flush_table(&mut out, &mut table); - 80
- 81
if is_horizontal_rule(trimmed) { - 82
flush_quote(&mut out, &mut quote); - 83
if blank_pending { - 84
out.push('\n'); - 85
} - 86
blank_pending = false; - 87
out.push_str("──────────\n"); - 88
continue; - 89
} - 90
- 91
if let Some(after) = trimmed.strip_prefix('#') - 92
&& (after.starts_with(' ') || after.starts_with('#')) - 93
{ - 94
let level = 1 + after.chars().take_while(|c| *c == '#').count(); - 95
let text = after.trim().trim_start_matches('#').trim(); - 96
if !text.is_empty() { - 97
flush_quote(&mut out, &mut quote); - 98
if blank_pending && !out.is_empty() { - 99
out.push('\n'); - 100
} - 101
blank_pending = false; - 102
let rendered = inline_markdown(&escape_html(text)); - 103
if level <= 2 { - 104
out.push_str("<b>"); - 105
out.push_str(&rendered.to_uppercase()); - 106
out.push_str("</b>\n\n"); - 107
} else { - 108
out.push_str("<b>"); - 109
out.push_str(&rendered); - 110
out.push_str("</b>\n\n"); - 111
} - 112
continue; - 113
} - 114
} - 115
- 116
if let Some(quote_line) = trimmed - 117
.strip_prefix("> ") - 118
.or_else(|| if trimmed == ">" { Some("") } else { None }) - 119
{ - 120
if blank_pending { - 121
flush_quote(&mut out, &mut quote); - 122
} - 123
blank_pending = false; - 124
quote.push(inline_markdown(&escape_html(quote_line))); - 125
continue; - 126
} - 127
flush_quote(&mut out, &mut quote); - 128
- 129
let indent = line.len() - trimmed.len(); - 130
let depth = indent / 2; - 131
- 132
if let Some(rest) = strip_task_item(trimmed) { - 133
blank_pending = false; - 134
out.push_str(&" ".repeat(depth)); - 135
out.push_str(bullet_for_depth(depth)); - 136
out.push(' '); - 137
out.push_str(&inline_markdown(&escape_html(&rest))); - 138
out.push('\n'); - 139
continue; - 140
} - 141
- 142
if let Some(item) = trimmed - 143
.strip_prefix("- ") - 144
.or_else(|| trimmed.strip_prefix("* ")) - 145
.or_else(|| trimmed.strip_prefix("+ ")) - 146
{ - 147
blank_pending = false; - 148
out.push_str(&" ".repeat(depth)); - 149
out.push_str(bullet_for_depth(depth)); - 150
out.push(' '); - 151
out.push_str(&inline_markdown(&escape_html(item))); - 152
out.push('\n'); - 153
continue; - 154
} - 155
- 156
if let Some((number, rest)) = split_ordered_item(trimmed) { - 157
blank_pending = false; - 158
out.push_str(&" ".repeat(depth)); - 159
out.push_str(&number); - 160
out.push_str(". "); - 161
out.push_str(&inline_markdown(&escape_html(rest))); - 162
out.push('\n'); - 163
continue; - 164
} - 165
- 166
if blank_pending && !out.is_empty() { - 167
out.push('\n'); - 168
} - 169
blank_pending = false; - 170
out.push_str(&inline_markdown(&escape_html(line))); - 171
out.push('\n'); - 172
} - 173
if in_fence { - 174
out.push_str("<pre>"); - 175
if fence_lang.is_empty() { - 176
out.push_str("<code>"); - 177
} else { - 178
out.push_str("<code class=\"language-"); - 179
out.push_str(&escape_html(&fence_lang)); - 180
out.push_str("\">"); - 181
} - 182
out.push_str(&escape_html(&fence.join("\n"))); - 183
out.push_str("</code></pre>\n"); - 184
} - 185
flush_quote(&mut out, &mut quote); - 186
flush_table(&mut out, &mut table); - 187
out.trim_end_matches('\n').to_string() - 188
} - 189
- 190
fn is_horizontal_rule(trimmed: &str) -> bool { - 191
let compact: String = trimmed.chars().filter(|c| !c.is_whitespace()).collect(); - 192
compact.len() >= 3 - 193
&& (compact.chars().all(|c| c == '-') - 194
|| compact.chars().all(|c| c == '*') - 195
|| compact.chars().all(|c| c == '_')) - 196
} - 197
- 198
fn is_spoiler(trimmed: &str) -> bool { - 199
trimmed.starts_with("||") && trimmed.ends_with("||") && trimmed.len() > 4 - 200
} - 201
- 202
fn split_ordered_item(trimmed: &str) -> Option<(String, &str)> { - 203
let digits_end = trimmed.find(|c: char| !c.is_ascii_digit())?; - 204
if digits_end == 0 { - 205
return None; - 206
} - 207
let (number, rest) = trimmed.split_at(digits_end); - 208
let rest = rest - 209
.strip_prefix(". ") - 210
.or_else(|| rest.strip_prefix(") "))?; - 211
Some((number.to_string(), rest)) - 212
} - 213
- 214
fn bullet_for_depth(depth: usize) -> &'static str { - 215
match depth % 3 { - 216
0 => "•", - 217
1 => "◦", - 218
_ => "▪", - 219
} - 220
} - 221
- 222
fn strip_task_item(trimmed: &str) -> Option<String> { - 223
trimmed - 224
.strip_prefix("- [ ] ") - 225
.map(|rest| format!("☐ {}", rest)) - 226
.or_else(|| { - 227
trimmed - 228
.strip_prefix("- [x] ") - 229
.map(|rest| format!("☑ {}", rest)) - 230
}) - 231
} - 232
- 233
fn flush_quote(out: &mut String, lines: &mut Vec<String>) { - 234
if lines.is_empty() { - 235
return; - 236
} - 237
out.push_str("<blockquote>"); - 238
out.push_str(&lines.join("\n")); - 239
out.push_str("</blockquote>\n"); - 240
lines.clear(); - 241
} - 242
- 243
fn flush_table(out: &mut String, rows: &mut Vec<String>) { - 244
if rows.is_empty() { - 245
return; - 246
} - 247
let parsed: Vec<Vec<String>> = rows - 248
.iter() - 249
.filter(|row| !is_separator_row(row)) - 250
.map(|row| { - 251
row.trim() - 252
.trim_matches('|') - 253
.split('|') - 254
.map(|cell| cell.trim().to_string()) - 255
.collect() - 256
}) - 257
.collect(); - 258
- 259
if parsed.is_empty() { - 260
rows.clear(); - 261
return; - 262
} - 263
- 264
let columns = parsed.iter().map(|row| row.len()).max().unwrap_or(0); - 265
let mut widths = vec![0usize; columns]; - 266
for row in &parsed { - 267
for (index, cell) in row.iter().enumerate() { - 268
widths[index] = widths[index].max(cell.chars().count()); - 269
} - 270
} - 271
- 272
let mut body = String::new(); - 273
for (row_index, row) in parsed.iter().enumerate() { - 274
for (index, width) in widths.iter().enumerate() { - 275
let cell = row.get(index).map(String::as_str).unwrap_or(""); - 276
body.push_str(cell); - 277
body.push_str(&" ".repeat(width.saturating_sub(cell.chars().count()))); - 278
if index + 1 < columns { - 279
body.push_str(" "); - 280
} - 281
} - 282
body.push('\n'); - 283
if row_index == 0 { - 284
let underline: usize = widths.iter().sum::<usize>() + (columns.saturating_sub(1) * 2); - 285
body.push_str(&"─".repeat(underline)); - 286
body.push('\n'); - 287
} - 288
} - 289
out.push_str("<pre>"); - 290
out.push_str(&escape_html(body.trim_end_matches('\n'))); - 291
out.push_str("</pre>\n"); - 292
rows.clear(); - 293
} - 294
- 295
fn is_separator_row(row: &str) -> bool { - 296
row.replace(['|', '-', ' ', ':'], "").is_empty() - 297
} - 298
- 299
fn escape_html(value: &str) -> String { - 300
value - 301
.replace('&', "&") - 302
.replace('<', "<") - 303
.replace('>', ">") - 304
} - 305
- 306
fn inline_markdown(value: &str) -> String { - 307
let mut codes = Vec::new(); - 308
let mut value = replace_code_spans(value, &mut codes); - 309
value = replace_links(&value); - 310
value = replace_pairs(&value, "~~", "<s>", "</s>"); - 311
value = replace_pairs(&value, "||", "<tg-spoiler>", "</tg-spoiler>"); - 312
value = replace_emphasis(&value, "**", "<b>", "</b>"); - 313
value = replace_emphasis(&value, "__", "<b>", "</b>"); - 314
value = replace_emphasis(&value, "*", "<i>", "</i>"); - 315
value = replace_emphasis(&value, "_", "<i>", "</i>"); - 316
for (index, code) in codes.iter().enumerate() { - 317
value = value.replace( - 318
&format!("\u{0}{index}\u{0}"), - 319
&format!("<code>{}</code>", escape_html(code)), - 320
); - 321
} - 322
value - 323
} - 324
- 325
fn replace_code_spans(value: &str, sink: &mut Vec<String>) -> String { - 326
let mut out = String::new(); - 327
let mut rest = value; - 328
while let Some((before, after)) = rest.split_once('`') { - 329
match after.split_once('`') { - 330
Some((content, tail)) => { - 331
sink.push(content.to_string()); - 332
out.push_str(before); - 333
out.push_str(&format!("\u{0}{}\u{0}", sink.len() - 1)); - 334
rest = tail; - 335
} - 336
None => { - 337
out.push_str(before); - 338
out.push('`'); - 339
rest = after; - 340
} - 341
} - 342
} - 343
out.push_str(rest); - 344
out - 345
} - 346
- 347
fn replace_links(value: &str) -> String { - 348
let mut out = String::new(); - 349
let mut rest = value; - 350
while let Some(open) = rest.find('[') { - 351
let Some(relative_text_end) = rest[open..].find("](") else { - 352
break; - 353
}; - 354
let text_end = open + relative_text_end; - 355
let Some(relative_url_end) = rest[text_end..].find(')') else { - 356
break; - 357
}; - 358
let url_end = text_end + relative_url_end; - 359
let url = &rest[text_end + 2..url_end]; - 360
if url.contains(['"', '<', '>', ' ']) { - 361
out.push_str(&rest[..open + 1]); - 362
rest = &rest[open + 1..]; - 363
continue; - 364
} - 365
out.push_str(&rest[..open]); - 366
out.push_str("<a href=\""); - 367
out.push_str(url); - 368
out.push_str("\">"); - 369
out.push_str(&rest[open + 1..text_end]); - 370
out.push_str("</a>"); - 371
rest = &rest[url_end + 1..]; - 372
} - 373
out.push_str(rest); - 374
out - 375
} - 376
- 377
fn replace_emphasis(value: &str, marker: &str, open: &str, close: &str) -> String { - 378
if marker.len() == 1 { - 379
replace_spaced_pairs(value, marker, open, close) - 380
} else { - 381
replace_pairs(value, marker, open, close) - 382
} - 383
} - 384
- 385
fn replace_pairs(value: &str, marker: &str, open: &str, close: &str) -> String { - 386
let mut out = String::new(); - 387
let mut rest = value; - 388
loop { - 389
match rest.split_once(marker) { - 390
Some((before, after)) => match after.split_once(marker) { - 391
Some((inner, tail)) if !inner.is_empty() => { - 392
out.push_str(before); - 393
out.push_str(open); - 394
out.push_str(inner); - 395
out.push_str(close); - 396
rest = tail; - 397
} - 398
_ => { - 399
out.push_str(before); - 400
out.push_str(marker); - 401
rest = after; - 402
} - 403
}, - 404
None => { - 405
out.push_str(rest); - 406
return out; - 407
} - 408
} - 409
} - 410
} - 411
- 412
fn replace_spaced_pairs(value: &str, marker: &str, open: &str, close: &str) -> String { - 413
let mut out = String::new(); - 414
let mut rest = value; - 415
loop { - 416
match rest.split_once(marker) { - 417
Some((before, after)) => { - 418
let opens = after.chars().next().is_some_and(|c| !c.is_whitespace()); - 419
let boundary = before.is_empty() - 420
|| before.chars().last().is_some_and(|c| !c.is_alphanumeric()); - 421
match after.split_once(marker) { - 422
Some((inner, tail)) if opens && boundary && !inner.trim().is_empty() => { - 423
out.push_str(before); - 424
out.push_str(open); - 425
out.push_str(inner.trim_end_matches(marker)); - 426
out.push_str(close); - 427
rest = tail; - 428
} - 429
_ => { - 430
out.push_str(before); - 431
out.push_str(marker); - 432
rest = after; - 433
} - 434
} - 435
} - 436
None => { - 437
out.push_str(rest); - 438
return out; - 439
} - 440
} - 441
} - 442
} - 443
- 444
#[derive(Clone)] - 445
struct OpenTag { - 446
name: String, - 447
opening: String, - 448
} - 449
- 450
/// Split HTML into independently parseable chunks. Open tags are closed at - 451
/// every boundary and reopened in the following chunk. - 452
pub fn split_html_chunks(html: &str, max_chars: Option<usize>) -> Vec<String> { - 453
let Some(cap) = max_chars else { - 454
return vec![html.to_string()]; - 455
}; - 456
if html.chars().count() <= cap { - 457
return vec![html.to_string()]; - 458
} - 459
- 460
if cap < 32 { - 461
return plain_chunks(html, cap); - 462
} - 463
let tokens = html_tokens(html); - 464
let mut chunks = Vec::new(); - 465
let mut current = String::new(); - 466
let mut open_tags: Vec<OpenTag> = Vec::new(); - 467
for token in tokens { - 468
if token.starts_with('<') && token.ends_with('>') { - 469
let mut prospective = open_tags.clone(); - 470
update_tag_stack(&token, &mut prospective); - 471
let projected = - 472
current.chars().count() + token.chars().count() + closing_tags_len(&prospective); - 473
if !current.is_empty() && projected > cap { - 474
flush_chunk(&mut current, &open_tags, &mut chunks); - 475
reopen_tags(&mut current, &open_tags); - 476
} - 477
update_tag_stack(&token, &mut open_tags); - 478
current.push_str(&token); - 479
continue; - 480
} - 481
for character in token.chars() { - 482
if current.chars().count() + 1 + closing_tags_len(&open_tags) > cap { - 483
flush_chunk(&mut current, &open_tags, &mut chunks); - 484
reopen_tags(&mut current, &open_tags); - 485
} - 486
current.push(character); - 487
} - 488
} - 489
if !current.trim().is_empty() { - 490
flush_chunk(&mut current, &open_tags, &mut chunks); - 491
} - 492
chunks - 493
} - 494
- 495
fn html_tokens(html: &str) -> Vec<String> { - 496
let mut tokens = Vec::new(); - 497
let mut rest = html; - 498
while let Some(start) = rest.find('<') { - 499
if start > 0 { - 500
tokens.push(rest[..start].to_string()); - 501
} - 502
let Some(end) = rest[start..].find('>') else { - 503
tokens.push(rest[start..].to_string()); - 504
return tokens; - 505
}; - 506
let end = start + end + 1; - 507
tokens.push(rest[start..end].to_string()); - 508
rest = &rest[end..]; - 509
} - 510
if !rest.is_empty() { - 511
tokens.push(rest.to_string()); - 512
} - 513
tokens - 514
} - 515
- 516
fn reopen_tags(current: &mut String, tags: &[OpenTag]) { - 517
for tag in tags { - 518
current.push_str(&tag.opening); - 519
} - 520
} - 521
- 522
fn plain_chunks(html: &str, cap: usize) -> Vec<String> { - 523
let plain = strip_html(html); - 524
let chars: Vec<char> = plain.chars().collect(); - 525
chars - 526
.chunks(cap) - 527
.map(|chunk| chunk.iter().collect()) - 528
.collect() - 529
} - 530
- 531
fn update_tag_stack(token: &str, stack: &mut Vec<OpenTag>) { - 532
if !token.starts_with('<') || !token.ends_with('>') { - 533
return; - 534
} - 535
let body = token[1..token.len() - 1].trim(); - 536
if let Some(name) = body.strip_prefix('/') { - 537
if let Some(index) = stack.iter().rposition(|tag| tag.name == name.trim()) { - 538
stack.remove(index); - 539
} - 540
return; - 541
} - 542
let name = body.split_whitespace().next().unwrap_or_default(); - 543
if !name.is_empty() && !body.ends_with('/') { - 544
stack.push(OpenTag { - 545
name: name.to_string(), - 546
opening: token.to_string(), - 547
}); - 548
} - 549
} - 550
- 551
fn closing_tags_len(tags: &[OpenTag]) -> usize { - 552
tags.iter().map(|tag| tag.name.chars().count() + 3).sum() - 553
} - 554
- 555
fn flush_chunk(current: &mut String, open_tags: &[OpenTag], chunks: &mut Vec<String>) { - 556
for tag in open_tags.iter().rev() { - 557
current.push_str("</"); - 558
current.push_str(&tag.name); - 559
current.push('>'); - 560
} - 561
let chunk = std::mem::take(current).trim().to_string(); - 562
if !chunk.is_empty() { - 563
chunks.push(chunk); - 564
} - 565
} - 566
- 567
/// Remove HTML tags for a last-resort plain Telegram retry. - 568
pub fn strip_html(html: &str) -> String { - 569
let mut out = String::new(); - 570
let mut in_tag = false; - 571
for character in html.chars() { - 572
match character { - 573
'<' => in_tag = true, - 574
'>' => in_tag = false, - 575
_ if !in_tag => out.push(character), - 576
_ => {} - 577
} - 578
} - 579
out - 580
} - 581
- 582
#[cfg(test)] - 583
mod tests { - 584
use super::*; - 585
- 586
#[test] - 587
fn converts_and_escapes_supported_markdown() { - 588
let html = markdown_to_html( - 589
"# Title\n\n**bold** *it* `code` [site](https://x.com/a)\n\n> quoted\n- item", - 590
); - 591
assert!(html.contains("<b>TITLE</b>")); - 592
assert!(html.contains("<b>bold</b>")); - 593
assert!(html.contains("<i>it</i>")); - 594
assert!(html.contains("<code>code</code>")); - 595
assert!(html.contains("<blockquote>quoted</blockquote>")); - 596
assert!(html.contains("• item")); - 597
assert!(!markdown_to_html("<script>x</script>").contains("<script>")); - 598
} - 599
- 600
#[test] - 601
fn renders_strikethrough_and_spoiler() { - 602
let html = markdown_to_html("~~gone~~ and ||hidden||"); - 603
assert!(html.contains("<s>gone</s>")); - 604
assert!(html.contains("<tg-spoiler>hidden</tg-spoiler>")); - 605
} - 606
- 607
#[test] - 608
fn renders_ordered_lists() { - 609
let html = markdown_to_html("1. first\n2. second"); - 610
assert!(html.contains("1. first")); - 611
assert!(html.contains("2. second")); - 612
} - 613
- 614
#[test] - 615
fn renders_nested_bullets_with_distinct_markers() { - 616
let html = markdown_to_html("- top\n - nested\n - deeper"); - 617
assert!(html.contains("• top")); - 618
assert!(html.contains("◦ nested")); - 619
assert!(html.contains("▪ deeper")); - 620
} - 621
- 622
#[test] - 623
fn merges_consecutive_blockquote_lines() { - 624
let html = markdown_to_html("> line one\n> line two"); - 625
assert_eq!( - 626
html, - 627
"<blockquote>line one\nline two</blockquote>".to_string() - 628
); - 629
} - 630
- 631
#[test] - 632
fn preserves_fence_language_as_code_class() { - 633
let html = markdown_to_html("```rust\nfn main() {}\n```"); - 634
assert!(html.contains("<pre><code class=\"language-rust\">")); - 635
assert!(html.contains("fn main() {}")); - 636
} - 637
- 638
#[test] - 639
fn renders_horizontal_rule() { - 640
let html = markdown_to_html("above\n\n---\n\nbelow"); - 641
assert!(html.contains("──────────")); - 642
} - 643
- 644
#[test] - 645
fn renders_aligned_table() { - 646
let html = markdown_to_html("| A | B |\n| - | - |\n| 1 | 22 |"); - 647
assert!(html.contains("<pre>")); - 648
assert!(html.contains("A")); - 649
assert!(html.contains("22")); - 650
} - 651
- 652
#[test] - 653
fn chunker_closes_and_reopens_tags() { - 654
let html = format!("<b>{}</b>", "word ".repeat(200)); - 655
let chunks = split_html_chunks(&html, Some(180)); - 656
assert!(chunks.len() > 1); - 657
assert!( - 658
chunks - 659
.iter() - 660
.all(|chunk| chunk.starts_with("<b>") && chunk.ends_with("</b>")) - 661
); - 662
assert!(chunks.iter().all(|chunk| chunk.chars().count() <= 180)); - 663
} - 664
} - 665
Indexing the workspace…
Vakyartha documentation is discovering safe artifacts, anchors, and source references.