Files

356 lines
16 KiB
C#
Raw Permalink Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
using System.Text;
using System.Text.RegularExpressions;
using Markdig;
using Markdig.Extensions.CustomContainers;
using Markdig.Renderers.Html;
using Markdig.Syntax;
using Markdig.Syntax.Inlines;
namespace Elternbeirat.Web.Shared;
/// <summary>
/// Renders the editors' Markdown (the <c>body</c>, <c>intro</c> and <c>answer</c>
/// fields from PocketBase) to the HTML the pages show.
/// </summary>
/// <remarks>
/// The pipeline is CommonMark plus exactly two extensions, each there for a reason
/// an editor can see:
/// <list type="bullet">
/// <item>
/// <description>
/// <b>Custom containers</b> (<c>::: name</c> … <c>:::</c>) turn a block
/// into <c>&lt;div class="name"&gt;</c>. That is how editors use the
/// design blocks (<c>kennzahlen</c>, <c>aufruf</c>, <c>kacheln</c>,
/// <c>hinweis</c>, <c>team</c>) without writing HTML. An unknown name
/// just yields a div without styling, so a typo never breaks a page.
/// </description>
/// </item>
/// <item>
/// <description>
/// <b>Pipe tables</b>: plain CommonMark has no tables at all, so without
/// this the table styles in <c>app.css</c> could never apply.
/// </description>
/// </item>
/// </list>
/// <para>
/// Raw HTML in the Markdown is <b>escaped</b>, not passed through
/// (<see cref="MarkdownPipelineBuilder"/>'s <c>DisableHtml</c>). The result is
/// rendered as a <c>MarkupString</c>, so a passed-through <c>&lt;script&gt;</c>
/// would run in the visitor's browser, and the privacy text promises that no
/// program code runs there. Editors style pages with the blocks instead.
/// </para>
/// <para>
/// After parsing, a paragraph (or list item) that consists of nothing but one
/// link gets the class <see cref="LoneLinkClass"/>. Pure CSS cannot tell
/// "a link alone on its line" from "a link inside a sentence" (selectors ignore
/// the text around an element), but the stylesheet needs exactly that to show a
/// lone <c>mailto:</c> link as a button and a lone PDF link as a file card.
/// </para>
/// <para>
/// The list items of a <c>::: team</c> block become person cards
/// (<see cref="TeamMember"/>): the initials avatar and the split into name,
/// role and duties need the text itself, which CSS cannot read.
/// </para>
/// </remarks>
/// <seealso cref="IconSet"/>
/// <seealso cref="TeamMember"/>
public static partial class Markdown
{
/// <summary>
/// The class marking a paragraph or list item whose only content is one link.
/// <c>app.css</c> keys the mail button and the PDF card off it.
/// </summary>
public const string LoneLinkClass = "lone-link";
/// <summary>
/// The default teaser length: about two lines on a card, enough to say what a
/// post is about without turning the list into a wall of text.
/// </summary>
public const int TeaserLength = 160;
private static readonly MarkdownPipeline Pipeline =
new MarkdownPipelineBuilder()
.UseCustomContainers()
.UsePipeTables()
.DisableHtml()
.Build();
/// <summary>
/// Converts a Markdown string to HTML.
/// </summary>
/// <param name="markdown">The editor's Markdown; may be <see langword="null"/>.</param>
/// <returns>
/// The rendered HTML, or an empty string for <see langword="null"/> or empty
/// input, so callers can bind the result directly.
/// </returns>
/// <example>
/// <code>
/// Markdown.ToHtml("::: hinweis\nBitte vormerken.\n:::");
/// // &lt;div class="hinweis"&gt;&lt;p&gt;Bitte vormerken.&lt;/p&gt;&lt;/div&gt;
/// </code>
/// </example>
public static string ToHtml(string? markdown) =>
string.IsNullOrEmpty(markdown) ? "" : Render(Parse(markdown));
/// <summary>
/// Extracts the first sentence of a Markdown text as plain text, e.g. as the
/// short description on a home page tile.
/// </summary>
/// <param name="markdown">The editor's Markdown; may be <see langword="null"/>.</param>
/// <returns>
/// The first sentence of the first top-level paragraph, with all Markdown
/// removed, whitespace collapsed and a trailing colon dropped, or an empty
/// string when there is no such paragraph (empty body, or only headings, lists
/// and blocks).
/// </returns>
/// <remarks>
/// Only top-level paragraphs count: a heading repeats the title, and the text in
/// a list or a <c>:::</c> block is rarely a sentence that describes the page.
/// <para>
/// A sentence ends at <c>.</c>, <c>!</c> or <c>?</c> followed by a space or
/// the end. A period after a single letter or a number does not count, so
/// <c>z. B.</c> and <c>13. November</c> do not cut the sentence short. A
/// longer abbreviation such as <c>bzw.</c> still does; that costs the rest
/// of a teaser, never the page.
/// </para>
/// <para>
/// A paragraph that leads into a list often ends in <c>:</c> without a
/// period; as a teaser the colon would point at a list that is not there.
/// The sentence is returned whole: a long one is cut by the tile's CSS,
/// which knows the space it has, not by a character count.
/// </para>
/// </remarks>
/// <example>
/// <code>
/// Markdown.FirstSentence("# Vorstand\n\nWir sind **sieben** Eltern. Mehr unten.");
/// // "Wir sind sieben Eltern."
/// </code>
/// </example>
public static string FirstSentence(string? markdown) =>
string.IsNullOrWhiteSpace(markdown)
? ""
: Parse(markdown).OfType<ParagraphBlock>().FirstOrDefault()?.Inline is { } inline
? UpToSentenceEnd(PlainText(inline)).TrimEnd(':', ' ')
: "";
/// <summary>
/// Builds a short plain-text teaser from a Markdown text, e.g. for a post card.
/// </summary>
/// <param name="markdown">The editor's Markdown; may be <see langword="null"/>.</param>
/// <param name="maxLength">
/// The longest teaser, in characters, before the ellipsis. Defaults to
/// <see cref="TeaserLength"/>.
/// </param>
/// <returns>
/// The text of all paragraphs (including those in lists, quotes and <c>:::</c>
/// blocks), with all Markdown removed and whitespace collapsed. A longer text is
/// cut at the last word boundary within <paramref name="maxLength"/> and ends
/// in <c>…</c>. An empty string when there is no paragraph text at all.
/// </returns>
/// <exception cref="ArgumentOutOfRangeException">
/// <paramref name="maxLength"/> is zero or negative.
/// </exception>
/// <remarks>
/// Headings are skipped: on a card the title already heads the teaser. Unlike
/// <see cref="FirstSentence"/> the teaser does not stop at the first sentence,
/// because a post often opens with a short line ("Liebe Eltern,") that says
/// nothing on its own.
/// <para>
/// Punctuation left dangling at the cut (<c>,</c>, <c>.</c>, <c>:</c> …) is
/// dropped, so the teaser never ends in <c>,…</c>.
/// </para>
/// </remarks>
/// <example>
/// <code>
/// Markdown.Teaser("## Rückblick\n\nDer **Basar** war ein voller Erfolg.", 20);
/// // "Der Basar war ein…"
/// </code>
/// </example>
public static string Teaser(string? markdown, int maxLength = TeaserLength) =>
maxLength <= 0
? throw new ArgumentOutOfRangeException(nameof(maxLength), maxLength, "The teaser needs room for at least one character.")
: string.IsNullOrWhiteSpace(markdown)
? ""
: Shorten(
string.Join(
' ',
Parse(markdown)
.Descendants<ParagraphBlock>()
.Select(paragraph => paragraph.Inline is { } inline ? PlainText(inline) : "")
.Where(text => text.Length > 0)),
maxLength);
/// <summary>
/// Parses Markdown with the site's pipeline, for callers that take the document
/// apart before rendering it (see <see cref="Render"/>).
/// </summary>
/// <param name="markdown">The editor's Markdown.</param>
/// <returns>The parsed document.</returns>
internal static MarkdownDocument Parse(string markdown) =>
Markdig.Markdown.Parse(markdown, Pipeline);
/// <summary>
/// Renders a document from <see cref="Parse"/> to HTML, building the team cards
/// and marking lone links on the way, exactly as <see cref="ToHtml"/> does.
/// </summary>
/// <param name="document">The parsed document; changed in place.</param>
/// <returns>The rendered HTML.</returns>
internal static string Render(MarkdownDocument document) =>
Markdig.Markdown.ToHtml(MarkLoneLinks(BuildTeamCards(document)), Pipeline);
/// <summary>
/// Flattens inline Markdown to the text a reader sees: emphasis and link
/// markup dropped, link text kept, images left out, whitespace collapsed.
/// </summary>
/// <param name="inlines">
/// The inline content, e.g. of a paragraph or heading, or a part of it.
/// </param>
/// <returns>The visible text, trimmed.</returns>
internal static string PlainText(IEnumerable<Inline> inlines) =>
Whitespace().Replace(inlines.Aggregate(new StringBuilder(), AppendText).ToString(), " ").Trim();
/// <summary>
/// Whether an inline carries no visible content (whitespace or a line break).
/// </summary>
/// <param name="inline">The inline to inspect.</param>
/// <returns>
/// <see langword="true"/> if it can be ignored when looking for links that stand
/// alone in a paragraph.
/// </returns>
internal static bool IsBlank(Inline inline) =>
inline switch
{
LineBreakInline => true,
LiteralInline literal => literal.Content.IsEmptyOrWhitespace(),
_ => false,
};
/// <summary>
/// Appends the visible text of an inline and its children.
/// </summary>
/// <param name="text">The builder to append to.</param>
/// <param name="inline">The inline to flatten.</param>
/// <returns>The same <paramref name="text"/>, for chaining.</returns>
private static StringBuilder AppendText(StringBuilder text, Inline inline) =>
inline switch
{
LiteralInline literal => text.Append(literal.Content.ToString()),
CodeInline code => text.Append(code.Content),
HtmlEntityInline entity => text.Append(entity.Transcoded.ToString()),
AutolinkInline autolink => text.Append(autolink.Url),
LineBreakInline => text.Append(' '),
LinkInline { IsImage: true } => text,
ContainerInline container => container.Aggregate(text, AppendText),
_ => text,
};
/// <summary>
/// Cuts a text after its first sentence.
/// </summary>
/// <param name="text">Plain text.</param>
/// <returns>
/// The first sentence including its end mark, or the whole text if it has none.
/// </returns>
private static string UpToSentenceEnd(string text) =>
SentenceEnd().Match(text) is { Success: true } end ? text[..(end.Index + 1)] : text;
/// <summary>
/// Cuts a text to a word boundary and marks the cut.
/// </summary>
/// <param name="text">Plain text.</param>
/// <param name="maxLength">The longest result before the ellipsis.</param>
/// <returns>
/// The text unchanged if it fits; otherwise everything up to the last space
/// within <paramref name="maxLength"/>, trailing punctuation removed, plus
/// <c>…</c>. A single word longer than <paramref name="maxLength"/> is cut hard.
/// </returns>
/// <remarks>
/// The space is searched one character past the limit: if the text breaks
/// exactly at <paramref name="maxLength"/>, the last whole word still fits.
/// </remarks>
private static string Shorten(string text, int maxLength) =>
text.Length <= maxLength
? text
: (text[..(maxLength + 1)].LastIndexOf(' ') is var space and > 0 ? text[..space] : text[..maxLength])
.TrimEnd(' ', ',', ';', ':', '.', '-', '–') + "…";
/// <summary>
/// Matches the end mark of a sentence: <c>.</c>, <c>!</c> or <c>?</c> before a
/// space or the end, unless it follows a lone letter (<c>z.</c>) or a number
/// (<c>13.</c>).
/// </summary>
[GeneratedRegex(@"(?<!\b\p{L}|\b\d+)[.!?](?=\s|$)")]
private static partial Regex SentenceEnd();
/// <summary>
/// Matches a run of whitespace, collapsed to one space in plain text.
/// </summary>
[GeneratedRegex(@"\s+")]
private static partial Regex Whitespace();
/// <summary>
/// Turns the list items of every <c>::: team</c> block into person cards.
/// </summary>
/// <param name="document">The parsed document; changed in place.</param>
/// <returns>The same <paramref name="document"/>, for chaining.</returns>
/// <remarks>
/// Only lists directly in the block count; a nested list inside a card stays a
/// list. <c>::: team gross</c> adds the class
/// <see cref="TeamMember.LargeModifier"/> to the block, because Markdig only
/// turns the block name into a class, not the words after it.
/// </remarks>
private static MarkdownDocument BuildTeamCards(MarkdownDocument document)
{
foreach (var team in document.Descendants<CustomContainer>().Where(TeamMember.IsTeam).ToList())
{
if (TeamMember.IsLarge(team))
{
team.GetAttributes().AddClass(TeamMember.LargeModifier);
}
foreach (var item in team.OfType<ListBlock>().SelectMany(list => list.OfType<ListItemBlock>()))
{
TeamMember.ToCard(item);
}
}
return document;
}
/// <summary>
/// Adds <see cref="LoneLinkClass"/> to every paragraph that holds nothing but
/// one link.
/// </summary>
/// <param name="document">The parsed document; changed in place.</param>
/// <returns>The same <paramref name="document"/>, for chaining.</returns>
/// <remarks>
/// Inside a list item the class goes on the item, not the paragraph: in a tight
/// list Markdig writes no <c>&lt;p&gt;</c> at all, so a class on the paragraph
/// would be silently dropped.
/// </remarks>
private static MarkdownDocument MarkLoneLinks(MarkdownDocument document)
{
foreach (var paragraph in document.Descendants<ParagraphBlock>().Where(IsLoneLink))
{
MarkdownObject target = paragraph.Parent is ListItemBlock { Count: 1 } item ? item : paragraph;
target.GetAttributes().AddClass(LoneLinkClass);
}
return document;
}
/// <summary>
/// Whether a paragraph's content is exactly one link, ignoring surrounding
/// whitespace and line breaks.
/// </summary>
/// <param name="paragraph">The paragraph to inspect.</param>
/// <returns>
/// <see langword="true"/> for one link (inline <c>[text](url)</c> or an
/// autolink <c>&lt;url&gt;</c>, but not an image) and nothing else.
/// </returns>
private static bool IsLoneLink(ParagraphBlock paragraph) =>
paragraph.Inline?
.Where(inline => !IsBlank(inline))
.ToList() is [LinkInline { IsImage: false } or AutolinkInline];
}