lmjtfy.git / packages / tree / src / lib.rs
lib.rsannotatedlib.rssource457 lines · 16.3 KB · raw
1//! The code pages' data, with no I/O: what GitHub's contents API says is at
2//! a path, read into types; a path checked before anything is asked about
3//! it; and markdown as HTML, its relative links pointing back into the code
4//! pages.
5//!
6//! GitHub's answer carries more than is read here, and some of it must
7//! never reach a page: a file's `download_url` holds a temporary access
8//! token. Only names, paths, sizes, a submodule's commit and a file's
9//! content are kept.
10#![forbid(unsafe_code)]
11
12pub mod annotate;
13pub mod icons;
14
15use base64::Engine;
16use base64::engine::general_purpose::STANDARD;
17use pulldown_cmark::{CodeBlockKind, CowStr, Event, Options, Parser, Tag, TagEnd, html};
18use serde::Deserialize;
19
20/// What an entry in a directory is.
21#[derive(Clone, Copy, Debug, PartialEq, Eq)]
22pub enum Kind {
23    Dir,
24    File,
25    /// Another repository, pinned at a commit.
26    Submodule,
27    Symlink,
28}
29
30#[derive(Clone, Debug, PartialEq)]
31pub struct Entry {
32    pub name: String,
33    /// From the repository's root, no leading slash.
34    pub path: String,
35    pub kind: Kind,
36    pub size: u64,
37}
38
39#[derive(Clone, Debug, PartialEq)]
40pub enum Body {
41    Text(String),
42    /// Not UTF-8 text: the bytes, for serving as they are.
43    Binary(Vec<u8>),
44    /// Over the megabyte GitHub sends the content of.
45    TooLarge,
46}
47
48#[derive(Clone, Debug, PartialEq)]
49pub enum Contents {
50    /// Directories and submodules first, then files, each by name.
51    Dir(Vec<Entry>),
52    File { path: String, size: u64, body: Body },
53    /// A submodule's own path, asked for: the commit it is pinned at.
54    Submodule { path: String, sha: String },
55}
56
57#[derive(Deserialize)]
58struct Raw {
59    name: String,
60    path: String,
61    #[serde(rename = "type")]
62    kind: String,
63    #[serde(default)]
64    size: u64,
65    #[serde(default)]
66    sha: String,
67    /// Read only to tell a submodule in a listing from a file: GitHub lists
68    /// one as a `file` with no download address. Never kept: the address
69    /// holds a token.
70    #[serde(default)]
71    download_url: Option<String>,
72    #[serde(default)]
73    content: Option<String>,
74    #[serde(default)]
75    encoding: Option<String>,
76}
77
78impl Raw {
79    fn kind(&self) -> Kind {
80        match self.kind.as_str() {
81            "dir" => Kind::Dir,
82            "submodule" => Kind::Submodule,
83            "symlink" => Kind::Symlink,
84            _ if self.download_url.is_none() => Kind::Submodule,
85            _ => Kind::File,
86        }
87    }
88}
89
90/// GitHub's answer for a path: a list for a directory, one object for
91/// anything else. `None` if it is neither.
92pub fn parse(json: &str) -> Option<Contents> {
93    if json.trim_start().starts_with('[') {
94        let listed: Vec<Raw> = serde_json::from_str(json).ok()?;
95        let mut entries: Vec<Entry> = listed
96            .into_iter()
97            .map(|raw| Entry { kind: raw.kind(), name: raw.name, path: raw.path, size: raw.size })
98            .collect();
99        entries.sort_by(|a, b| {
100            let folder = |entry: &Entry| !matches!(entry.kind, Kind::Dir | Kind::Submodule);
101            (folder(a), a.name.to_lowercase()).cmp(&(folder(b), b.name.to_lowercase()))
102        });
103        return Some(Contents::Dir(entries));
104    }
105    let raw: Raw = serde_json::from_str(json).ok()?;
106    match raw.kind() {
107        Kind::Submodule => Some(Contents::Submodule { path: raw.path, sha: raw.sha }),
108        Kind::Dir => None,
109        Kind::File | Kind::Symlink => {
110            let body = match (raw.encoding.as_deref(), raw.content.as_deref()) {
111                (Some("base64"), Some(content)) => {
112                    let packed: String = content.split_whitespace().collect();
113                    match STANDARD.decode(packed) {
114                        Ok(bytes) => text(bytes),
115                        Err(_) => Body::TooLarge,
116                    }
117                }
118                _ => Body::TooLarge,
119            };
120            Some(Contents::File { path: raw.path, size: raw.size, body })
121        }
122    }
123}
124
125/// A whole repository's tree, from GitHub's git trees API with `recursive=1`:
126/// every folder and file, and each submodule with the commit it is pinned
127/// at. For the code pages' explorer. `None` if it is not that shape, or if
128/// GitHub cut it short (`truncated`), since a partial tree would show files
129/// as missing that are not.
130pub fn parse_tree(json: &str) -> Option<Tree> {
131    #[derive(Deserialize)]
132    struct Listed {
133        truncated: bool,
134        tree: Vec<Node>,
135    }
136    #[derive(Deserialize)]
137    struct Node {
138        path: String,
139        #[serde(rename = "type")]
140        kind: String,
141        #[serde(default)]
142        size: u64,
143        sha: String,
144    }
145    let listed: Listed = serde_json::from_str(json).ok()?;
146    if listed.truncated {
147        return None;
148    }
149    let mut tree = Tree::default();
150    for node in listed.tree {
151        let kind = match node.kind.as_str() {
152            "tree" => Kind::Dir,
153            "commit" => Kind::Submodule,
154            _ => Kind::File,
155        };
156        if kind == Kind::Submodule {
157            tree.pins.push((node.path.clone(), node.sha));
158        }
159        let name = node.path.rsplit('/').next().unwrap_or(&node.path).to_owned();
160        tree.entries.push(Entry { name, path: node.path, kind, size: node.size });
161    }
162    Some(tree)
163}
164
165/// A repository's every entry, and where its submodules are pinned.
166#[derive(Clone, Debug, Default, PartialEq)]
167pub struct Tree {
168    pub entries: Vec<Entry>,
169    /// `(mount, commit)` for each submodule.
170    pub pins: Vec<(String, String)>,
171}
172
173impl Tree {
174    /// `inner`'s entries added under `mount`, as a submodule's are served.
175    pub fn mount(&mut self, mount: &str, inner: Tree) {
176        for entry in inner.entries {
177            self.entries.push(Entry { path: format!("{mount}/{}", entry.path), ..entry });
178        }
179    }
180
181    /// The entries directly in `folder` (`""` for the root): folders and
182    /// submodules first, then files, each by name.
183    pub fn children(&self, folder: &str) -> Vec<&Entry> {
184        let mut children: Vec<&Entry> = self.entries.iter().filter(|entry| parent(&entry.path) == folder).collect();
185        children.sort_by(|a, b| {
186            let file = |entry: &Entry| !matches!(entry.kind, Kind::Dir | Kind::Submodule);
187            (file(a), a.name.to_lowercase()).cmp(&(file(b), b.name.to_lowercase()))
188        });
189        children
190    }
191}
192
193fn text(bytes: Vec<u8>) -> Body {
194    if bytes.contains(&0) {
195        return Body::Binary(bytes);
196    }
197    String::from_utf8(bytes).map(Body::Text).unwrap_or_else(|error| Body::Binary(error.into_bytes()))
198}
199
200/// The media type a file is served as raw, from its name. Text is served as
201/// plain text whatever it is, so nothing in the repository runs as a page.
202pub fn media_type(path: &str, body: &Body) -> &'static str {
203    let extension = path.rsplit_once('.').map(|(_, extension)| extension.to_ascii_lowercase());
204    match (body, extension.as_deref()) {
205        (Body::Binary(_), Some("png")) => "image/png",
206        (Body::Binary(_), Some("jpg" | "jpeg")) => "image/jpeg",
207        (Body::Binary(_), Some("gif")) => "image/gif",
208        (Body::Binary(_), Some("webp")) => "image/webp",
209        (Body::Text(_), Some("svg")) => "image/svg+xml",
210        (Body::Text(_), _) => "text/plain; charset=utf-8",
211        _ => "application/octet-stream",
212    }
213}
214
215/// A path's segments, or `None` if one of them climbs out (`..`) or is `.`.
216/// Empty segments, from doubled or trailing slashes, are dropped.
217pub fn segments(path: &str) -> Option<Vec<&str>> {
218    let segments: Vec<&str> = path.split('/').filter(|segment| !segment.is_empty()).collect();
219    if segments.iter().any(|segment| matches!(*segment, "." | "..")) {
220        return None;
221    }
222    Some(segments)
223}
224
225/// Segments joined for a URL, each percent-encoded but for unreserved
226/// characters.
227pub fn encoded(segments: &[&str]) -> String {
228    let mut out = String::new();
229    for (index, segment) in segments.iter().enumerate() {
230        if index > 0 {
231            out.push('/');
232        }
233        for byte in segment.bytes() {
234            match byte {
235                b'A'..=b'Z' | b'a'..=b'z' | b'0'..=b'9' | b'-' | b'_' | b'.' | b'~' => out.push(byte as char),
236                _ => out.push_str(&format!("%{byte:02X}")),
237            }
238        }
239    }
240    out
241}
242
243/// The directory a path is in: `packages/rules` for `packages/rules/README.md`.
244pub fn parent(path: &str) -> &str {
245    path.rsplit_once('/').map_or("", |(parent, _)| parent)
246}
247
248/// A heading, for a document's table of contents.
249#[derive(Clone, Debug, PartialEq)]
250pub struct Heading {
251    /// 1 for `#`, 2 for `##`, and so on.
252    pub level: u8,
253    pub text: String,
254    /// The anchor the heading carries, unique in the document.
255    pub id: String,
256}
257
258/// A document rendered.
259#[derive(Clone, Debug, PartialEq)]
260pub struct Rendered {
261    pub html: String,
262    pub headings: Vec<Heading>,
263    /// Whether it has a diagram for the page to draw: a code block marked
264    /// `mermaid`, written out as `<pre class="mermaid">` with its source as
265    /// text.
266    pub mermaid: bool,
267}
268
269/// Markdown as HTML. A relative link or image is taken from `dir` and
270/// pointed at `base` (`/lmjtfy.git`), so a README's links stay in the code
271/// pages. Headings get anchors. Raw HTML in the markdown is shown as text,
272/// not run.
273pub fn markdown(text: &str, dir: &str, base: &str) -> Rendered {
274    let options = Options::ENABLE_TABLES | Options::ENABLE_STRIKETHROUGH | Options::ENABLE_TASKLISTS;
275    let mut events: Vec<Event> = Parser::new_ext(text, options)
276        .map(|event| match event {
277            Event::Start(Tag::Link { link_type, dest_url, title, id }) => {
278                Event::Start(Tag::Link { link_type, dest_url: relinked(&dest_url, dir, base), title, id })
279            }
280            Event::Start(Tag::Image { link_type, dest_url, title, id }) => {
281                Event::Start(Tag::Image { link_type, dest_url: CowStr::from(image(&dest_url, dir, base)), title, id })
282            }
283            Event::Html(raw) | Event::InlineHtml(raw) => Event::Text(raw),
284            event => event,
285        })
286        .collect();
287
288    // Anchors: each heading's text made a slug, numbered if it repeats.
289    let mut headings = Vec::new();
290    let mut index = 0;
291    while index < events.len() {
292        if let Event::Start(Tag::Heading { level, .. }) = &events[index] {
293            let level = *level as u8;
294            let mut text = String::new();
295            for event in &events[index + 1..] {
296                match event {
297                    Event::End(TagEnd::Heading(_)) => break,
298                    Event::Text(words) | Event::Code(words) => text.push_str(words),
299                    _ => {}
300                }
301            }
302            let mut id = slug(&text);
303            let taken = |id: &str| headings.iter().any(|heading: &Heading| heading.id == id);
304            if taken(&id) {
305                let mut n = 1;
306                while taken(&format!("{id}-{n}")) {
307                    n += 1;
308                }
309                id = format!("{id}-{n}");
310            }
311            if let Event::Start(Tag::Heading { id: anchor, .. }) = &mut events[index] {
312                *anchor = Some(CowStr::from(id.clone()));
313            }
314            headings.push(Heading { level, text, id });
315        }
316        index += 1;
317    }
318
319    // Diagrams: a `mermaid` block becomes a `<pre class="mermaid">` the page
320    // draws. Its source stays text, escaped like any other.
321    let mut mermaid = false;
322    let mut out = Vec::with_capacity(events.len());
323    let mut in_diagram = false;
324    for event in events {
325        match event {
326            Event::Start(Tag::CodeBlock(CodeBlockKind::Fenced(lang))) if lang.as_ref() == "mermaid" => {
327                in_diagram = true;
328                mermaid = true;
329                out.push(Event::Html(CowStr::from("<pre class=\"mermaid\">")));
330            }
331            Event::End(TagEnd::CodeBlock) if in_diagram => {
332                in_diagram = false;
333                out.push(Event::Html(CowStr::from("</pre>")));
334            }
335            event => out.push(event),
336        }
337    }
338    let mut html = String::new();
339    html::push_html(&mut html, out.into_iter());
340    Rendered { html, headings, mermaid }
341}
342
343/// A heading's anchor: lower case, letters and digits, words joined by `-`.
344pub fn slug(text: &str) -> String {
345    let mut slug = String::new();
346    for ch in text.chars().flat_map(char::to_lowercase) {
347        if ch.is_alphanumeric() {
348            slug.push(ch);
349        } else if !slug.is_empty() && !slug.ends_with('-') {
350            slug.push('-');
351        }
352    }
353    let slug = slug.trim_end_matches('-').to_owned();
354    if slug.is_empty() { "section".to_owned() } else { slug }
355}
356
357/// A CLAUDE.md as shown: its first line imports the README (`@README.md`),
358/// which the page shows as a link instead. The rest is what an agent is told
359/// beyond the README.
360pub fn agents(text: &str) -> (Option<&str>, &str) {
361    let mut lines = text.splitn(2, '\n');
362    let first = lines.next().unwrap_or_default().trim();
363    match first.strip_prefix('@') {
364        Some(imported) if !imported.is_empty() && !imported.contains(' ') => (Some(imported), lines.next().unwrap_or_default()),
365        _ => (None, text),
366    }
367}
368
369fn relinked<'a>(dest: &CowStr<'a>, dir: &str, base: &str) -> CowStr<'a> {
370    CowStr::from(link(dest, dir, base))
371}
372
373/// Where a chapter says it sits in its guide: its last line,
374/// `← Previous: [..](..) · Up: [..](..) · Next: [..](..) →`. Each is the
375/// link's text and where it goes, as a code page address.
376#[derive(Clone, Debug, Default, PartialEq)]
377pub struct Guide {
378    /// Whether the chapter has a guide line at all. One that has says where
379    /// it leads, even if that is nowhere: the last chapter has no Next.
380    pub line: bool,
381    pub previous: Option<(String, String)>,
382    pub next: Option<(String, String)>,
383}
384
385/// The guide line of a chapter in `dir`: the last line that names a
386/// Previous or a Next. A part with no link (a chapter whose previous is in
387/// another repository) is not there.
388pub fn guide(text: &str, dir: &str, base: &str) -> Guide {
389    let Some(line) = text.lines().rev().map(str::trim).find(|line| line.contains("Next:") || line.starts_with('←')) else {
390        return Guide::default();
391    };
392    let mut found = Guide { line: true, ..Guide::default() };
393    for part in line.split(" · ") {
394        let part = part.trim_matches(|c: char| c == '←' || c == '→' || c.is_whitespace());
395        let slot = if let Some(rest) = part.strip_prefix("Previous:") {
396            Some((&mut found.previous, rest))
397        } else {
398            part.strip_prefix("Next:").map(|rest| (&mut found.next, rest))
399        };
400        if let Some((slot, rest)) = slot
401            && let Some((label, dest)) = markdown_link(rest.trim())
402        {
403            *slot = Some((label.to_owned(), link(dest, dir, base)));
404        }
405    }
406    found
407}
408
409/// `[label](dest)` at the start of `text`.
410fn markdown_link(text: &str) -> Option<(&str, &str)> {
411    let rest = text.strip_prefix('[')?;
412    let (label, rest) = rest.split_once("](")?;
413    let (dest, _) = rest.split_once(')')?;
414    Some((label, dest))
415}
416
417/// Where an image in a file in `dir` is drawn from: the file itself
418/// (`?raw`), not its code page, which is HTML.
419pub fn image(dest: &str, dir: &str, base: &str) -> String {
420    let local = !(dest.contains("://") || dest.starts_with('#') || dest.starts_with("data:") || dest.is_empty());
421    let linked = link(dest, dir, base);
422    if local { format!("{}?raw", linked.split('#').next().unwrap_or_default()) } else { linked }
423}
424
425/// Where a link in a file in `dir` goes, as a code page address.
426pub fn link(dest: &str, dir: &str, base: &str) -> String {
427    let external = dest.contains("://") || dest.starts_with('#') || dest.starts_with("mailto:");
428    if external || dest.is_empty() {
429        return dest.to_owned();
430    }
431    let (path, fragment) = match dest.split_once('#') {
432        Some((path, fragment)) => (path, Some(fragment)),
433        None => (dest, None),
434    };
435    let mut joined: Vec<&str> = if path.starts_with('/') { Vec::new() } else { dir.split('/').filter(|s| !s.is_empty()).collect() };
436    for segment in path.split('/') {
437        match segment {
438            "" | "." => {}
439            ".." => {
440                joined.pop();
441            }
442            segment => joined.push(segment),
443        }
444    }
445    let mut out = format!("{base}/{}", joined.join("/"));
446    if path.ends_with('/') && !joined.is_empty() {
447        out.push('/');
448    }
449    if let Some(fragment) = fragment {
450        out.push('#');
451        out.push_str(fragment);
452    }
453    out
454}
455
456#[cfg(test)]
457mod tests;