Team Ai
Apppublic

Yash030/claude-code-proxy

sourceHugging Faceupdated 5mo agoView on Hugging Face
2likes
telegram_markdown.py328 linesDownload Raw Back to rendering
1"""Telegram MarkdownV2 utilities.2 3Renders common Markdown into Telegram MarkdownV2 format.4Used by the message handler and Telegram platform adapter.5"""6 7from markdown_it import MarkdownIt8 9from .markdown_tables import normalize_gfm_tables10 11MDV2_SPECIAL_CHARS = set("\\_*[]()~`>#+-=|{}.!")12MDV2_LINK_ESCAPE = set("\\)")13 14_MD = MarkdownIt("commonmark", {"html": False, "breaks": False})15_MD.enable("strikethrough")16_MD.enable("table")17 18 19def escape_md_v2(text: str) -> str:20    """Escape text for Telegram MarkdownV2."""21    return "".join(f"\\{ch}" if ch in MDV2_SPECIAL_CHARS else ch for ch in text)22 23 24def escape_md_v2_code(text: str) -> str:25    """Escape text for Telegram MarkdownV2 code spans/blocks."""26    return text.replace("\\", "\\\\").replace("`", "\\`")27 28 29def escape_md_v2_link_url(text: str) -> str:30    """Escape URL for Telegram MarkdownV2 link destination."""31    return "".join(f"\\{ch}" if ch in MDV2_LINK_ESCAPE else ch for ch in text)32 33 34def mdv2_bold(text: str) -> str:35    """Format text as bold in MarkdownV2."""36    return f"*{escape_md_v2(text)}*"37 38 39def mdv2_code_inline(text: str) -> str:40    """Format text as inline code in MarkdownV2."""41    return f"`{escape_md_v2_code(text)}`"42 43 44def format_status(emoji: str, label: str, suffix: str | None = None) -> str:45    """Format a status message with emoji and optional suffix."""46    base = f"{emoji} {mdv2_bold(label)}"47    if suffix:48        return f"{base} {escape_md_v2(suffix)}"49    return base50 51 52def render_markdown_to_mdv2(text: str) -> str:53    """Render common Markdown into Telegram MarkdownV2."""54    if not text:55        return ""56 57    text = normalize_gfm_tables(text)58    tokens = _MD.parse(text)59 60    def render_inline_table_plain(children) -> str:61        out: list[str] = []62        for tok in children:63            if tok.type == "text" or tok.type == "code_inline":64                out.append(tok.content)65            elif tok.type in {"softbreak", "hardbreak"}:66                out.append(" ")67            elif tok.type == "image" and tok.content:68                out.append(tok.content)69        return "".join(out)70 71    def render_inline_plain(children) -> str:72        out: list[str] = []73        for tok in children:74            if tok.type == "text" or tok.type == "code_inline":75                out.append(escape_md_v2(tok.content))76            elif tok.type in {"softbreak", "hardbreak"}:77                out.append("\n")78        return "".join(out)79 80    def render_inline(children) -> str:81        out: list[str] = []82        i = 083        while i < len(children):84            tok = children[i]85            t = tok.type86            if t == "text":87                out.append(escape_md_v2(tok.content))88            elif t in {"softbreak", "hardbreak"}:89                out.append("\n")90            elif t == "em_open" or t == "em_close":91                out.append("_")92            elif t == "strong_open" or t == "strong_close":93                out.append("*")94            elif t == "s_open" or t == "s_close":95                out.append("~")96            elif t == "code_inline":97                out.append(f"`{escape_md_v2_code(tok.content)}`")98            elif t == "link_open":99                href = ""100                if tok.attrs:101                    if isinstance(tok.attrs, dict):102                        href = tok.attrs.get("href", "")103                    else:104                        for key, val in tok.attrs:105                            if key == "href":106                                href = val107                                break108                inner_tokens = []109                i += 1110                while i < len(children) and children[i].type != "link_close":111                    inner_tokens.append(children[i])112                    i += 1113                link_text = ""114                for child in inner_tokens:115                    if child.type == "text" or child.type == "code_inline":116                        link_text += child.content117                out.append(118                    f"[{escape_md_v2(link_text)}]({escape_md_v2_link_url(href)})"119                )120            elif t == "image":121                href = ""122                alt = tok.content or ""123                if tok.attrs:124                    if isinstance(tok.attrs, dict):125                        href = tok.attrs.get("src", "")126                    else:127                        for key, val in tok.attrs:128                            if key == "src":129                                href = val130                                break131                if alt:132                    out.append(f"{escape_md_v2(alt)} ({escape_md_v2_link_url(href)})")133                else:134                    out.append(escape_md_v2_link_url(href))135            else:136                out.append(escape_md_v2(tok.content or ""))137            i += 1138        return "".join(out)139 140    out: list[str] = []141    list_stack: list[dict] = []142    pending_prefix: str | None = None143    blockquote_level = 0144    in_heading = False145 146    def apply_blockquote(val: str) -> str:147        if blockquote_level <= 0:148            return val149        prefix = "> " * blockquote_level150        return prefix + val.replace("\n", "\n" + prefix)151 152    i = 0153    while i < len(tokens):154        tok = tokens[i]155        t = tok.type156        if t == "paragraph_open":157            pass158        elif t == "paragraph_close":159            out.append("\n")160        elif t == "heading_open":161            in_heading = True162        elif t == "heading_close":163            in_heading = False164            out.append("\n")165        elif t == "bullet_list_open":166            list_stack.append({"type": "bullet", "index": 1})167        elif t == "bullet_list_close":168            if list_stack:169                list_stack.pop()170            out.append("\n")171        elif t == "ordered_list_open":172            start = 1173            if tok.attrs:174                if isinstance(tok.attrs, dict):175                    val = tok.attrs.get("start")176                    if val is not None:177                        try:178                            start = int(val)179                        except TypeError, ValueError:180                            start = 1181                else:182                    for key, val in tok.attrs:183                        if key == "start":184                            try:185                                start = int(val)186                            except TypeError, ValueError:187                                start = 1188                            break189            list_stack.append({"type": "ordered", "index": start})190        elif t == "ordered_list_close":191            if list_stack:192                list_stack.pop()193            out.append("\n")194        elif t == "list_item_open":195            if list_stack:196                top = list_stack[-1]197                if top["type"] == "bullet":198                    pending_prefix = "\\- "199                else:200                    pending_prefix = f"{top['index']}\\."201                    top["index"] += 1202                    pending_prefix += " "203        elif t == "list_item_close":204            out.append("\n")205        elif t == "blockquote_open":206            blockquote_level += 1207        elif t == "blockquote_close":208            blockquote_level = max(0, blockquote_level - 1)209            out.append("\n")210        elif t == "table_open":211            if pending_prefix:212                out.append(apply_blockquote(pending_prefix.rstrip()))213                out.append("\n")214                pending_prefix = None215 216            rows: list[list[str]] = []217            row_is_header: list[bool] = []218 219            j = i + 1220            in_thead = False221            in_row = False222            current_row: list[str] = []223            current_row_header = False224 225            in_cell = False226            cell_parts: list[str] = []227 228            while j < len(tokens):229                tt = tokens[j].type230                if tt == "thead_open":231                    in_thead = True232                elif tt == "thead_close":233                    in_thead = False234                elif tt == "tr_open":235                    in_row = True236                    current_row = []237                    current_row_header = in_thead238                elif tt in {"th_open", "td_open"}:239                    in_cell = True240                    cell_parts = []241                elif tt == "inline" and in_cell:242                    cell_parts.append(243                        render_inline_table_plain(tokens[j].children or [])244                    )245                elif tt in {"th_close", "td_close"} and in_cell:246                    cell = " ".join(cell_parts).strip()247                    current_row.append(cell)248                    in_cell = False249                    cell_parts = []250                elif tt == "tr_close" and in_row:251                    rows.append(current_row)252                    row_is_header.append(bool(current_row_header))253                    in_row = False254                elif tt == "table_close":255                    break256                j += 1257 258            if rows:259                col_count = max((len(r) for r in rows), default=0)260                norm_rows: list[list[str]] = []261                for r in rows:262                    if len(r) < col_count:263                        r = r + [""] * (col_count - len(r))264                    norm_rows.append(r)265 266                widths: list[int] = []267                for c in range(col_count):268                    w = max((len(r[c]) for r in norm_rows), default=0)269                    widths.append(max(w, 3))270 271                def fmt_row(272                    r: list[str], _w: list[int] = widths, _c: int = col_count273                ) -> str:274                    cells = [r[c].ljust(_w[c]) for c in range(_c)]275                    return "| " + " | ".join(cells) + " |"276 277                def fmt_sep(_w: list[int] = widths, _c: int = col_count) -> str:278                    cells = ["-" * _w[c] for c in range(_c)]279                    return "| " + " | ".join(cells) + " |"280 281                last_header_idx = -1282                for idx, is_h in enumerate(row_is_header):283                    if is_h:284                        last_header_idx = idx285 286                lines: list[str] = []287                for idx, r in enumerate(norm_rows):288                    lines.append(fmt_row(r))289                    if idx == last_header_idx:290                        lines.append(fmt_sep())291 292                table_text = "\n".join(lines).rstrip()293                out.append(f"```\n{escape_md_v2_code(table_text)}\n```")294                out.append("\n")295 296            i = j + 1297            continue298        elif t in {"code_block", "fence"}:299            code = escape_md_v2_code(tok.content.rstrip("\n"))300            out.append(f"```\n{code}\n```")301            out.append("\n")302        elif t == "inline":303            rendered = render_inline(tok.children or [])304            if in_heading:305                rendered = f"*{render_inline_plain(tok.children or [])}*"306            if pending_prefix:307                rendered = pending_prefix + rendered308                pending_prefix = None309            rendered = apply_blockquote(rendered)310            out.append(rendered)311        else:312            if tok.content:313                out.append(escape_md_v2(tok.content))314        i += 1315 316    return "".join(out).rstrip()317 318 319__all__ = [320    "escape_md_v2",321    "escape_md_v2_code",322    "escape_md_v2_link_url",323    "format_status",324    "mdv2_bold",325    "mdv2_code_inline",326    "render_markdown_to_mdv2",327]328