Files
Elternbeirat/Elternbeirat.Web/Shared/Markdown.cs
T

266 lines
12 KiB
C#
Raw Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
using System.Text;
using System.Text.RegularExpressions;
using Markdig;
using Markdig.Renderers.Html;
using Markdig.Syntax;
using Markdig.Syntax.Inlines;
namespace Elternbeirat.Web.Shared;
/// <summary>
/// Renders the editors' Markdown (the <c>body</c>, <c>intro</c> and <c>answer</c>
/// fields from PocketBase) to the HTML the pages show.
/// </summary>
/// <remarks>
/// The pipeline is CommonMark plus exactly two extensions, each there for a reason
/// an editor can see:
/// <list type="bullet">
/// <item>
/// <description>
/// <b>Custom containers</b> (<c>::: name</c> … <c>:::</c>) turn a block
/// into <c>&lt;div class="name"&gt;</c>. That is how editors use the
/// design blocks (<c>kennzahlen</c>, <c>aufruf</c>, <c>kacheln</c>,
/// <c>hinweis</c>) without writing HTML. An unknown name just yields a
/// div without styling, so a typo never breaks a page.
/// </description>
/// </item>
/// <item>
/// <description>
/// <b>Pipe tables</b>: plain CommonMark has no tables at all, so without
/// this the table styles in <c>app.css</c> could never apply.
/// </description>
/// </item>
/// </list>
/// <para>
/// Raw HTML in the Markdown is <b>escaped</b>, not passed through
/// (<see cref="MarkdownPipelineBuilder"/>'s <c>DisableHtml</c>). The result is
/// rendered as a <c>MarkupString</c>, so a passed-through <c>&lt;script&gt;</c>
/// would run in the visitor's browser, and the privacy text promises that no
/// program code runs there. Editors style pages with the blocks instead.
/// </para>
/// <para>
/// After parsing, a paragraph (or list item) that consists of nothing but one
/// link gets the class <see cref="LoneLinkClass"/>. Pure CSS cannot tell
/// "a link alone on its line" from "a link inside a sentence" (selectors ignore
/// the text around an element), but the stylesheet needs exactly that to show a
/// lone <c>mailto:</c> link as a button and a lone PDF link as a file card.
/// </para>
/// </remarks>
/// <seealso cref="IconSet"/>
public static partial class Markdown
{
/// <summary>
/// The class marking a paragraph or list item whose only content is one link.
/// <c>app.css</c> keys the mail button and the PDF card off it.
/// </summary>
public const string LoneLinkClass = "lone-link";
private static readonly MarkdownPipeline Pipeline =
new MarkdownPipelineBuilder()
.UseCustomContainers()
.UsePipeTables()
.DisableHtml()
.Build();
/// <summary>
/// Converts a Markdown string to HTML.
/// </summary>
/// <param name="markdown">The editor's Markdown; may be <see langword="null"/>.</param>
/// <returns>
/// The rendered HTML, or an empty string for <see langword="null"/> or empty
/// input, so callers can bind the result directly.
/// </returns>
/// <example>
/// <code>
/// Markdown.ToHtml("::: hinweis\nBitte vormerken.\n:::");
/// // &lt;div class="hinweis"&gt;&lt;p&gt;Bitte vormerken.&lt;/p&gt;&lt;/div&gt;
/// </code>
/// </example>
public static string ToHtml(string? markdown) =>
string.IsNullOrEmpty(markdown) ? "" : Render(Parse(markdown));
/// <summary>
/// Extracts the first sentence of a Markdown text as plain text, e.g. as the
/// short description on a home page tile.
/// </summary>
/// <param name="markdown">The editor's Markdown; may be <see langword="null"/>.</param>
/// <param name="maxLength">
/// The longest sentence kept as is; a longer one is cut at the last word
/// boundary before the limit and ends in <c>…</c>.
/// </param>
/// <returns>
/// The first sentence of the first top-level paragraph, with all Markdown
/// removed and whitespace collapsed, or an empty string when there is no such
/// paragraph (empty body, or only headings, lists and blocks).
/// </returns>
/// <exception cref="ArgumentOutOfRangeException">
/// <paramref name="maxLength"/> is zero or negative.
/// </exception>
/// <remarks>
/// Only top-level paragraphs count: a heading repeats the title, and the text in
/// a list or a <c>:::</c> block is rarely a sentence that describes the page.
/// <para>
/// A sentence ends at <c>.</c>, <c>!</c> or <c>?</c> followed by a space or
/// the end. A period after a single letter or a number does not count, so
/// <c>z. B.</c> and <c>13. November</c> do not cut the sentence short. A
/// longer abbreviation such as <c>bzw.</c> still does; that costs the rest
/// of a teaser, never the page.
/// </para>
/// </remarks>
/// <example>
/// <code>
/// Markdown.FirstSentence("# Vorstand\n\nWir sind **sieben** Eltern. Mehr unten.");
/// // "Wir sind sieben Eltern."
/// </code>
/// </example>
public static string FirstSentence(string? markdown, int maxLength = 140) =>
maxLength <= 0
? throw new ArgumentOutOfRangeException(nameof(maxLength), maxLength, "Must be positive.")
: string.IsNullOrWhiteSpace(markdown)
? ""
: Parse(markdown).OfType<ParagraphBlock>().FirstOrDefault()?.Inline is { } inline
? Shorten(UpToSentenceEnd(PlainText(inline)), maxLength)
: "";
/// <summary>
/// Parses Markdown with the site's pipeline, for callers that take the document
/// apart before rendering it (see <see cref="Render"/>).
/// </summary>
/// <param name="markdown">The editor's Markdown.</param>
/// <returns>The parsed document.</returns>
internal static MarkdownDocument Parse(string markdown) =>
Markdig.Markdown.Parse(markdown, Pipeline);
/// <summary>
/// Renders a document from <see cref="Parse"/> to HTML, marking lone links on
/// the way, exactly as <see cref="ToHtml"/> does.
/// </summary>
/// <param name="document">The parsed document; changed in place.</param>
/// <returns>The rendered HTML.</returns>
internal static string Render(MarkdownDocument document) =>
Markdig.Markdown.ToHtml(MarkLoneLinks(document), Pipeline);
/// <summary>
/// Flattens inline Markdown to the text a reader sees: emphasis and link
/// markup dropped, link text kept, images left out, whitespace collapsed.
/// </summary>
/// <param name="inline">The inline content, e.g. of a paragraph or heading.</param>
/// <returns>The visible text, trimmed.</returns>
internal static string PlainText(ContainerInline inline) =>
Whitespace().Replace(AppendText(new StringBuilder(), inline).ToString(), " ").Trim();
/// <summary>
/// Whether an inline carries no visible content (whitespace or a line break).
/// </summary>
/// <param name="inline">The inline to inspect.</param>
/// <returns>
/// <see langword="true"/> if it can be ignored when looking for links that stand
/// alone in a paragraph.
/// </returns>
internal static bool IsBlank(Inline inline) =>
inline switch
{
LineBreakInline => true,
LiteralInline literal => literal.Content.IsEmptyOrWhitespace(),
_ => false,
};
/// <summary>
/// Appends the visible text of an inline and its children.
/// </summary>
/// <param name="text">The builder to append to.</param>
/// <param name="inline">The inline to flatten.</param>
/// <returns>The same <paramref name="text"/>, for chaining.</returns>
private static StringBuilder AppendText(StringBuilder text, Inline inline) =>
inline switch
{
LiteralInline literal => text.Append(literal.Content.ToString()),
CodeInline code => text.Append(code.Content),
HtmlEntityInline entity => text.Append(entity.Transcoded.ToString()),
AutolinkInline autolink => text.Append(autolink.Url),
LineBreakInline => text.Append(' '),
LinkInline { IsImage: true } => text,
ContainerInline container => container.Aggregate(text, AppendText),
_ => text,
};
/// <summary>
/// Cuts a text after its first sentence.
/// </summary>
/// <param name="text">Plain text.</param>
/// <returns>
/// The first sentence including its end mark, or the whole text if it has none.
/// </returns>
private static string UpToSentenceEnd(string text) =>
SentenceEnd().Match(text) is { Success: true } end ? text[..(end.Index + 1)] : text;
/// <summary>
/// Shortens a text to at most <paramref name="maxLength"/> characters plus an
/// ellipsis, cutting between words.
/// </summary>
/// <param name="text">Plain text.</param>
/// <param name="maxLength">The longest text kept as is.</param>
/// <returns>The text, shortened if needed.</returns>
/// <remarks>
/// The search for a space starts at <paramref name="maxLength"/> itself, so a
/// word that ends exactly at the limit is kept. A single word longer than the
/// limit is cut hard.
/// </remarks>
private static string Shorten(string text, int maxLength) =>
text.Length <= maxLength
? text
: text.LastIndexOf(' ', maxLength) is > 0 and var space
? $"{text[..space].TrimEnd(',', ';', ':', '-', '–')}…"
: $"{text[..maxLength]}…";
/// <summary>
/// Matches the end mark of a sentence: <c>.</c>, <c>!</c> or <c>?</c> before a
/// space or the end, unless it follows a lone letter (<c>z.</c>) or a number
/// (<c>13.</c>).
/// </summary>
[GeneratedRegex(@"(?<!\b\p{L}|\b\d+)[.!?](?=\s|$)")]
private static partial Regex SentenceEnd();
/// <summary>
/// Matches a run of whitespace, collapsed to one space in plain text.
/// </summary>
[GeneratedRegex(@"\s+")]
private static partial Regex Whitespace();
/// <summary>
/// Adds <see cref="LoneLinkClass"/> to every paragraph that holds nothing but
/// one link.
/// </summary>
/// <param name="document">The parsed document; changed in place.</param>
/// <returns>The same <paramref name="document"/>, for chaining.</returns>
/// <remarks>
/// Inside a list item the class goes on the item, not the paragraph: in a tight
/// list Markdig writes no <c>&lt;p&gt;</c> at all, so a class on the paragraph
/// would be silently dropped.
/// </remarks>
private static MarkdownDocument MarkLoneLinks(MarkdownDocument document)
{
foreach (var paragraph in document.Descendants<ParagraphBlock>().Where(IsLoneLink))
{
MarkdownObject target = paragraph.Parent is ListItemBlock { Count: 1 } item ? item : paragraph;
target.GetAttributes().AddClass(LoneLinkClass);
}
return document;
}
/// <summary>
/// Whether a paragraph's content is exactly one link, ignoring surrounding
/// whitespace and line breaks.
/// </summary>
/// <param name="paragraph">The paragraph to inspect.</param>
/// <returns>
/// <see langword="true"/> for one link (inline <c>[text](url)</c> or an
/// autolink <c>&lt;url&gt;</c>, but not an image) and nothing else.
/// </returns>
private static bool IsLoneLink(ParagraphBlock paragraph) =>
paragraph.Inline?
.Where(inline => !IsBlank(inline))
.ToList() is [LinkInline { IsImage: false } or AutolinkInline];
}