Coverage for src/lilbee/app/session_export.py: 100%

156 statements  

« prev     ^ index     » next       coverage.py v7.15.2, created at 2026-09-28 17:20 +0000

1"""Render a saved chat session as a markdown document and write it to disk.""" 

2 

3from __future__ import annotations 

4 

5import os 

6import re 

7import string 

8from collections.abc import Iterator 

9from functools import cache 

10from pathlib import Path 

11 

12import yaml 

13from markdown_it import MarkdownIt 

14from markdown_it.rules_inline.html_inline import html_inline as _stock_html_inline 

15from markdown_it.rules_inline.state_inline import StateInline 

16from markdown_it.token import Token 

17 

18from lilbee.core.security import write_private_text 

19from lilbee.core.text import collapse_whitespace, make_slug 

20from lilbee.retrieval.query.formatting import ( 

21 FILE_LINK_RE, 

22 SOURCES_BLOCK_MARKER, 

23 close_open_fence, 

24 open_code_fence, 

25 source_label, 

26 with_sources_block, 

27) 

28from lilbee.sessions import MessageRole, Session, SessionMessage, SessionMeta 

29 

30_EXPORT_SUFFIX = ".md" 

31SLUG_MAX_LEN = 60 

32_FALLBACK_STEM = "chat" 

33_ID_PREFIX_LEN = 8 

34_FRONT_MATTER_FENCE = "---" 

35_ROLE_HEADINGS: dict[MessageRole, str] = { 

36 MessageRole.USER: "User", 

37 MessageRole.ASSISTANT: "Assistant", 

38} 

39_ATX_MARKER = "#" 

40_TURN_HEADING_LEVEL = 2 

41_MAX_HEADING_LEVEL = 6 

42_ESCAPE = "\\" 

43# The only characters a markdown blank line may hold. 

44_MARKDOWN_BLANK = " \t" 

45_LINE_BREAK_RE = re.compile(r"\r\n?") 

46_ASCII_PUNCTUATION = frozenset(string.punctuation) 

47# An HTML heading's ``<h``/``</h`` plus level; a boundary after the digit (or the 

48# text simply ending there) is what excludes ``<header>``, ``<hr>`` and ``<h2o>``. 

49_HTML_HEADING_RE = re.compile(r"<(?P<slash>/?)(?P<tag>[Hh])(?P<level>[1-6])(?=[\s/>]|$)") 

50# An ordered-list marker at the start of a name, as in ``2. notes.md``. 

51_LIST_NUMBER_RE = re.compile(r"^(\d+)([.)])(?=\s|$)") 

52 

53 

54def session_markdown(session: Session) -> str: 

55 """The session as markdown: front matter, a title heading, one section per turn.""" 

56 parts = [_front_matter(session.meta), f"# {collapse_whitespace(session.meta.title)}"] 

57 parts.extend(_message_section(message) for message in session.messages) 

58 return "\n\n".join(parts) + "\n" 

59 

60 

61def _front_matter(meta: SessionMeta) -> str: 

62 fields = { 

63 "title": meta.title, 

64 "session": meta.id, 

65 "model": meta.model_ref, 

66 "created": meta.created_at, 

67 "updated": meta.updated_at, 

68 } 

69 if meta.forked_from: 

70 fields["forked_from"] = meta.forked_from 

71 dumped = yaml.safe_dump(fields, sort_keys=False, allow_unicode=True) 

72 return f"{_FRONT_MATTER_FENCE}\n{dumped}{_FRONT_MATTER_FENCE}" 

73 

74 

75def _message_section(message: SessionMessage) -> str: 

76 """One turn: the message with its headings nested, then its Sources list as plain names.""" 

77 content = _LINE_BREAK_RE.sub("\n", message.content).rstrip() 

78 text, stored_sources = _split_sources_list(content) 

79 body = _contained(text.rstrip()) 

80 if stored_sources: 

81 body += _contained(FILE_LINK_RE.sub(_plain_link, stored_sources)) 

82 elif message.sources: 

83 body = with_sources_block(body, message.sources, render=_plain_source) 

84 return f"## {_ROLE_HEADINGS[message.role]}\n\n{body}" 

85 

86 

87def _split_sources_list(content: str) -> tuple[str, str]: 

88 """*content* split before its last Sources list, unless that list is quoted in a code block. 

89 

90 A list after a code block the answer never closed (a cut-off answer) is lilbee's; 

91 a list inside a code block that closes after it is pasted text. 

92 """ 

93 text, marker, stored_sources = content.rpartition(SOURCES_BLOCK_MARKER) 

94 if not marker: 

95 return content, "" 

96 fence = open_code_fence(text) 

97 if fence is None or _never_closed(fence, content): 

98 return text, marker + stored_sources 

99 return content, "" 

100 

101 

102def _never_closed(fence: Token, content: str) -> bool: 

103 """Whether *fence*, opened in a prefix of *content*, is still open at its end.""" 

104 end = open_code_fence(content) 

105 return ( 

106 end is not None 

107 and end.map is not None 

108 and fence.map is not None 

109 and end.map[0] == fence.map[0] 

110 ) 

111 

112 

113def _contained(text: str) -> str: 

114 """*text* with open fences closed and headings nested, so it stays inside its turn.""" 

115 return _nest_headings(close_open_fence(text)) 

116 

117 

118def _plain_source(source: str) -> str: 

119 return _plain_name(source_label(source)) 

120 

121 

122def _plain_link(link: re.Match[str]) -> str: 

123 return _plain_name(link["label"]) 

124 

125 

126def _plain_name(name: str) -> str: 

127 """*name* on one line, escaped so it cannot start a heading, list, quote or other block.""" 

128 name = collapse_whitespace(name) 

129 if name[:1] in _ASCII_PUNCTUATION: 

130 return _ESCAPE + name 

131 return _LIST_NUMBER_RE.sub(r"\1\\\2", name, count=1) 

132 

133 

134def _html_inline_with_span(state: StateInline, silent: bool) -> bool: 

135 """The stock ``html_inline`` rule, plus the ``state.src`` span it matched on ``token.meta``.""" 

136 start = state.pos 

137 matched = _stock_html_inline(state, silent) 

138 if matched and not silent and state.tokens and state.tokens[-1].type == "html_inline": 

139 state.tokens[-1].meta = {"start": start, "end": state.pos} 

140 return matched 

141 

142 

143@cache 

144def _heading_parser() -> MarkdownIt: 

145 """A CommonMark parser whose ``html_inline`` tokens carry the source span they matched.""" 

146 md = MarkdownIt("commonmark") 

147 md.inline.ruler.at("html_inline", _html_inline_with_span) 

148 return md 

149 

150 

151def _nest_headings(text: str) -> str: 

152 """*text* with no heading, markdown or HTML, at or above the turn level. 

153 

154 The parser finds the headings, so a ``#`` in code or a ``#tag`` stays as 

155 written, and an HTML heading tag inside a fence or an inline code span 

156 stays as written too. A ``#`` heading drops two levels, to at most six; 

157 an underlined heading cannot go below level two, so its underline is 

158 escaped to text, and a blank line ends that text where the heading 

159 ended. An HTML heading tag, open or close, drops the same two levels, 

160 independently of whether it is ever closed. 

161 """ 

162 lines = text.split("\n") 

163 paragraph_ends: list[tuple[int, str | None]] = [] 

164 for token in _heading_parser().parse(text): 

165 if token.map: 

166 start, end = token.map 

167 paragraph_ends.append((end - 1, _nest_block(lines, token, start, end))) 

168 for last_line, blank_line in paragraph_ends: 

169 if blank_line is not None: 

170 lines[last_line] += "\n" + blank_line 

171 return "\n".join(lines) 

172 

173 

174def _nest_block(lines: list[str], token: Token, start: int, end: int) -> str | None: 

175 """Rewrite *token*'s lines *start* to *end*; the blank line to add after them, if any.""" 

176 if token.type == "heading_open": 

177 return _nest_heading(lines, token.markup, start, end) 

178 if token.type == "html_block": 

179 _demote_html_block(lines, start, end) 

180 elif token.type == "inline": 

181 _demote_html_inline(lines, token.content, token.children, start, end) 

182 return None 

183 

184 

185def _nest_heading(lines: list[str], markup: str, start: int, end: int) -> str | None: 

186 """Rewrite the heading on lines *start* to *end*; the blank line to add after it, if any.""" 

187 if markup.startswith(_ATX_MARKER): 

188 level = min(len(markup) + _TURN_HEADING_LEVEL, _MAX_HEADING_LEVEL) 

189 lines[start] = lines[start].replace(markup, _ATX_MARKER * level, 1) 

190 return None 

191 return _escape_underline(lines, markup, end) 

192 

193 

194def _escape_underline(lines: list[str], markup: str, end: int) -> str | None: 

195 """Escape the underline before line *end*; the blank line that keeps the next block, if any.""" 

196 underline = end - 1 

197 container, _, rest = lines[underline].partition(markup) 

198 lines[underline] = container + _ESCAPE + markup + rest 

199 if end < len(lines) and lines[end].strip(_MARKDOWN_BLANK): 

200 return container.rstrip() 

201 return None 

202 

203 

204def _demote_html_tag(match: re.Match[str]) -> str: 

205 """*match*, an HTML heading tag's opening chars, with its level raised by the turn offset.""" 

206 level = min(int(match["level"]) + _TURN_HEADING_LEVEL, _MAX_HEADING_LEVEL) 

207 return f"<{match['slash']}{match['tag']}{level}" 

208 

209 

210def _demote_html_block(lines: list[str], start: int, end: int) -> None: 

211 """Rewrite any HTML heading tag's level in the raw HTML block spanning *start* to *end*.""" 

212 joined = "\n".join(lines[start:end]) 

213 lines[start:end] = _HTML_HEADING_RE.sub(_demote_html_tag, joined).split("\n") 

214 

215 

216def _demote_html_inline( 

217 lines: list[str], content: str, children: list[Token] | None, start: int, end: int 

218) -> None: 

219 """Rewrite each real HTML heading tag's level within the inline span *start* to *end*.""" 

220 spans = [child.meta for child in children or () if child.type == "html_inline" and child.meta] 

221 if not spans: 

222 return 

223 real = _inside_spans(_HTML_HEADING_RE.finditer(content), spans) 

224 if not any(real): 

225 return 

226 raw = "\n".join(lines[start:end]) 

227 pieces: list[str] = [] 

228 cursor = 0 

229 # The parser strips only markers and whitespace and turns NUL into U+FFFD: matches pair up. 

230 for match, is_real in zip(_HTML_HEADING_RE.finditer(raw), real, strict=True): 

231 if is_real: 

232 pieces += [raw[cursor : match.start()], _demote_html_tag(match)] 

233 cursor = match.end() 

234 pieces.append(raw[cursor:]) 

235 lines[start:end] = "".join(pieces).split("\n") 

236 

237 

238def _inside_spans(matches: Iterator[re.Match[str]], spans: list[dict[str, int]]) -> list[bool]: 

239 """Whether each of *matches*, in order, starts inside one of the ordered *spans*.""" 

240 inside = [] 

241 spans_left = iter(spans) 

242 span = next(spans_left, None) 

243 for match in matches: 

244 while span is not None and span["end"] <= match.start(): 

245 span = next(spans_left, None) 

246 inside.append(span is not None and span["start"] <= match.start()) 

247 return inside 

248 

249 

250def default_export_name(meta: SessionMeta) -> str: 

251 """``<title-slug>-<id prefix>.md``, or ``chat-<id prefix>.md`` when the title has no slug.""" 

252 slug = make_slug(meta.title)[:SLUG_MAX_LEN].strip("-") or _FALLBACK_STEM 

253 return f"{slug}-{meta.id[:_ID_PREFIX_LEN]}{_EXPORT_SUFFIX}" 

254 

255 

256def write_session_markdown(session: Session, destination: str) -> Path: 

257 """Write *session* owner-only to *destination*, or to its default name inside 

258 it when it is a directory or ends in a separator. Returns the absolute path.""" 

259 target = Path(destination).expanduser() 

260 if target.is_dir() or destination.endswith(("/", os.sep)): 

261 target = target / default_export_name(session.meta) 

262 target = target.resolve() 

263 write_private_text(target, session_markdown(session)) 

264 return target