356 lines
16 KiB
C#
356 lines
16 KiB
C#
using System.Text;
|
||
using System.Text.RegularExpressions;
|
||
using Markdig;
|
||
using Markdig.Extensions.CustomContainers;
|
||
using Markdig.Renderers.Html;
|
||
using Markdig.Syntax;
|
||
using Markdig.Syntax.Inlines;
|
||
|
||
namespace Elternbeirat.Web.Shared;
|
||
|
||
/// <summary>
|
||
/// Renders the editors' Markdown (the <c>body</c>, <c>intro</c> and <c>answer</c>
|
||
/// fields from PocketBase) to the HTML the pages show.
|
||
/// </summary>
|
||
/// <remarks>
|
||
/// The pipeline is CommonMark plus exactly two extensions, each there for a reason
|
||
/// an editor can see:
|
||
/// <list type="bullet">
|
||
/// <item>
|
||
/// <description>
|
||
/// <b>Custom containers</b> (<c>::: name</c> … <c>:::</c>) turn a block
|
||
/// into <c><div class="name"></c>. That is how editors use the
|
||
/// design blocks (<c>kennzahlen</c>, <c>aufruf</c>, <c>kacheln</c>,
|
||
/// <c>hinweis</c>, <c>team</c>) without writing HTML. An unknown name
|
||
/// just yields a div without styling, so a typo never breaks a page.
|
||
/// </description>
|
||
/// </item>
|
||
/// <item>
|
||
/// <description>
|
||
/// <b>Pipe tables</b>: plain CommonMark has no tables at all, so without
|
||
/// this the table styles in <c>app.css</c> could never apply.
|
||
/// </description>
|
||
/// </item>
|
||
/// </list>
|
||
/// <para>
|
||
/// Raw HTML in the Markdown is <b>escaped</b>, not passed through
|
||
/// (<see cref="MarkdownPipelineBuilder"/>'s <c>DisableHtml</c>). The result is
|
||
/// rendered as a <c>MarkupString</c>, so a passed-through <c><script></c>
|
||
/// would run in the visitor's browser, and the privacy text promises that no
|
||
/// program code runs there. Editors style pages with the blocks instead.
|
||
/// </para>
|
||
/// <para>
|
||
/// After parsing, a paragraph (or list item) that consists of nothing but one
|
||
/// link gets the class <see cref="LoneLinkClass"/>. Pure CSS cannot tell
|
||
/// "a link alone on its line" from "a link inside a sentence" (selectors ignore
|
||
/// the text around an element), but the stylesheet needs exactly that to show a
|
||
/// lone <c>mailto:</c> link as a button and a lone PDF link as a file card.
|
||
/// </para>
|
||
/// <para>
|
||
/// The list items of a <c>::: team</c> block become person cards
|
||
/// (<see cref="TeamMember"/>): the initials avatar and the split into name,
|
||
/// role and duties need the text itself, which CSS cannot read.
|
||
/// </para>
|
||
/// </remarks>
|
||
/// <seealso cref="IconSet"/>
|
||
/// <seealso cref="TeamMember"/>
|
||
public static partial class Markdown
|
||
{
|
||
/// <summary>
|
||
/// The class marking a paragraph or list item whose only content is one link.
|
||
/// <c>app.css</c> keys the mail button and the PDF card off it.
|
||
/// </summary>
|
||
public const string LoneLinkClass = "lone-link";
|
||
|
||
/// <summary>
|
||
/// The default teaser length: about two lines on a card, enough to say what a
|
||
/// post is about without turning the list into a wall of text.
|
||
/// </summary>
|
||
public const int TeaserLength = 160;
|
||
|
||
private static readonly MarkdownPipeline Pipeline =
|
||
new MarkdownPipelineBuilder()
|
||
.UseCustomContainers()
|
||
.UsePipeTables()
|
||
.DisableHtml()
|
||
.Build();
|
||
|
||
/// <summary>
|
||
/// Converts a Markdown string to HTML.
|
||
/// </summary>
|
||
/// <param name="markdown">The editor's Markdown; may be <see langword="null"/>.</param>
|
||
/// <returns>
|
||
/// The rendered HTML, or an empty string for <see langword="null"/> or empty
|
||
/// input, so callers can bind the result directly.
|
||
/// </returns>
|
||
/// <example>
|
||
/// <code>
|
||
/// Markdown.ToHtml("::: hinweis\nBitte vormerken.\n:::");
|
||
/// // <div class="hinweis"><p>Bitte vormerken.</p></div>
|
||
/// </code>
|
||
/// </example>
|
||
public static string ToHtml(string? markdown) =>
|
||
string.IsNullOrEmpty(markdown) ? "" : Render(Parse(markdown));
|
||
|
||
/// <summary>
|
||
/// Extracts the first sentence of a Markdown text as plain text, e.g. as the
|
||
/// short description on a home page tile.
|
||
/// </summary>
|
||
/// <param name="markdown">The editor's Markdown; may be <see langword="null"/>.</param>
|
||
/// <returns>
|
||
/// The first sentence of the first top-level paragraph, with all Markdown
|
||
/// removed, whitespace collapsed and a trailing colon dropped, or an empty
|
||
/// string when there is no such paragraph (empty body, or only headings, lists
|
||
/// and blocks).
|
||
/// </returns>
|
||
/// <remarks>
|
||
/// Only top-level paragraphs count: a heading repeats the title, and the text in
|
||
/// a list or a <c>:::</c> block is rarely a sentence that describes the page.
|
||
/// <para>
|
||
/// A sentence ends at <c>.</c>, <c>!</c> or <c>?</c> followed by a space or
|
||
/// the end. A period after a single letter or a number does not count, so
|
||
/// <c>z. B.</c> and <c>13. November</c> do not cut the sentence short. A
|
||
/// longer abbreviation such as <c>bzw.</c> still does; that costs the rest
|
||
/// of a teaser, never the page.
|
||
/// </para>
|
||
/// <para>
|
||
/// A paragraph that leads into a list often ends in <c>:</c> without a
|
||
/// period; as a teaser the colon would point at a list that is not there.
|
||
/// The sentence is returned whole: a long one is cut by the tile's CSS,
|
||
/// which knows the space it has, not by a character count.
|
||
/// </para>
|
||
/// </remarks>
|
||
/// <example>
|
||
/// <code>
|
||
/// Markdown.FirstSentence("# Vorstand\n\nWir sind **sieben** Eltern. Mehr unten.");
|
||
/// // "Wir sind sieben Eltern."
|
||
/// </code>
|
||
/// </example>
|
||
public static string FirstSentence(string? markdown) =>
|
||
string.IsNullOrWhiteSpace(markdown)
|
||
? ""
|
||
: Parse(markdown).OfType<ParagraphBlock>().FirstOrDefault()?.Inline is { } inline
|
||
? UpToSentenceEnd(PlainText(inline)).TrimEnd(':', ' ')
|
||
: "";
|
||
|
||
/// <summary>
|
||
/// Builds a short plain-text teaser from a Markdown text, e.g. for a post card.
|
||
/// </summary>
|
||
/// <param name="markdown">The editor's Markdown; may be <see langword="null"/>.</param>
|
||
/// <param name="maxLength">
|
||
/// The longest teaser, in characters, before the ellipsis. Defaults to
|
||
/// <see cref="TeaserLength"/>.
|
||
/// </param>
|
||
/// <returns>
|
||
/// The text of all paragraphs (including those in lists, quotes and <c>:::</c>
|
||
/// blocks), with all Markdown removed and whitespace collapsed. A longer text is
|
||
/// cut at the last word boundary within <paramref name="maxLength"/> and ends
|
||
/// in <c>…</c>. An empty string when there is no paragraph text at all.
|
||
/// </returns>
|
||
/// <exception cref="ArgumentOutOfRangeException">
|
||
/// <paramref name="maxLength"/> is zero or negative.
|
||
/// </exception>
|
||
/// <remarks>
|
||
/// Headings are skipped: on a card the title already heads the teaser. Unlike
|
||
/// <see cref="FirstSentence"/> the teaser does not stop at the first sentence,
|
||
/// because a post often opens with a short line ("Liebe Eltern,") that says
|
||
/// nothing on its own.
|
||
/// <para>
|
||
/// Punctuation left dangling at the cut (<c>,</c>, <c>.</c>, <c>:</c> …) is
|
||
/// dropped, so the teaser never ends in <c>,…</c>.
|
||
/// </para>
|
||
/// </remarks>
|
||
/// <example>
|
||
/// <code>
|
||
/// Markdown.Teaser("## Rückblick\n\nDer **Basar** war ein voller Erfolg.", 20);
|
||
/// // "Der Basar war ein…"
|
||
/// </code>
|
||
/// </example>
|
||
public static string Teaser(string? markdown, int maxLength = TeaserLength) =>
|
||
maxLength <= 0
|
||
? throw new ArgumentOutOfRangeException(nameof(maxLength), maxLength, "The teaser needs room for at least one character.")
|
||
: string.IsNullOrWhiteSpace(markdown)
|
||
? ""
|
||
: Shorten(
|
||
string.Join(
|
||
' ',
|
||
Parse(markdown)
|
||
.Descendants<ParagraphBlock>()
|
||
.Select(paragraph => paragraph.Inline is { } inline ? PlainText(inline) : "")
|
||
.Where(text => text.Length > 0)),
|
||
maxLength);
|
||
|
||
/// <summary>
|
||
/// Parses Markdown with the site's pipeline, for callers that take the document
|
||
/// apart before rendering it (see <see cref="Render"/>).
|
||
/// </summary>
|
||
/// <param name="markdown">The editor's Markdown.</param>
|
||
/// <returns>The parsed document.</returns>
|
||
internal static MarkdownDocument Parse(string markdown) =>
|
||
Markdig.Markdown.Parse(markdown, Pipeline);
|
||
|
||
/// <summary>
|
||
/// Renders a document from <see cref="Parse"/> to HTML, building the team cards
|
||
/// and marking lone links on the way, exactly as <see cref="ToHtml"/> does.
|
||
/// </summary>
|
||
/// <param name="document">The parsed document; changed in place.</param>
|
||
/// <returns>The rendered HTML.</returns>
|
||
internal static string Render(MarkdownDocument document) =>
|
||
Markdig.Markdown.ToHtml(MarkLoneLinks(BuildTeamCards(document)), Pipeline);
|
||
|
||
/// <summary>
|
||
/// Flattens inline Markdown to the text a reader sees: emphasis and link
|
||
/// markup dropped, link text kept, images left out, whitespace collapsed.
|
||
/// </summary>
|
||
/// <param name="inlines">
|
||
/// The inline content, e.g. of a paragraph or heading, or a part of it.
|
||
/// </param>
|
||
/// <returns>The visible text, trimmed.</returns>
|
||
internal static string PlainText(IEnumerable<Inline> inlines) =>
|
||
Whitespace().Replace(inlines.Aggregate(new StringBuilder(), AppendText).ToString(), " ").Trim();
|
||
|
||
/// <summary>
|
||
/// Whether an inline carries no visible content (whitespace or a line break).
|
||
/// </summary>
|
||
/// <param name="inline">The inline to inspect.</param>
|
||
/// <returns>
|
||
/// <see langword="true"/> if it can be ignored when looking for links that stand
|
||
/// alone in a paragraph.
|
||
/// </returns>
|
||
internal static bool IsBlank(Inline inline) =>
|
||
inline switch
|
||
{
|
||
LineBreakInline => true,
|
||
LiteralInline literal => literal.Content.IsEmptyOrWhitespace(),
|
||
_ => false,
|
||
};
|
||
|
||
/// <summary>
|
||
/// Appends the visible text of an inline and its children.
|
||
/// </summary>
|
||
/// <param name="text">The builder to append to.</param>
|
||
/// <param name="inline">The inline to flatten.</param>
|
||
/// <returns>The same <paramref name="text"/>, for chaining.</returns>
|
||
private static StringBuilder AppendText(StringBuilder text, Inline inline) =>
|
||
inline switch
|
||
{
|
||
LiteralInline literal => text.Append(literal.Content.ToString()),
|
||
CodeInline code => text.Append(code.Content),
|
||
HtmlEntityInline entity => text.Append(entity.Transcoded.ToString()),
|
||
AutolinkInline autolink => text.Append(autolink.Url),
|
||
LineBreakInline => text.Append(' '),
|
||
LinkInline { IsImage: true } => text,
|
||
ContainerInline container => container.Aggregate(text, AppendText),
|
||
_ => text,
|
||
};
|
||
|
||
/// <summary>
|
||
/// Cuts a text after its first sentence.
|
||
/// </summary>
|
||
/// <param name="text">Plain text.</param>
|
||
/// <returns>
|
||
/// The first sentence including its end mark, or the whole text if it has none.
|
||
/// </returns>
|
||
private static string UpToSentenceEnd(string text) =>
|
||
SentenceEnd().Match(text) is { Success: true } end ? text[..(end.Index + 1)] : text;
|
||
|
||
/// <summary>
|
||
/// Cuts a text to a word boundary and marks the cut.
|
||
/// </summary>
|
||
/// <param name="text">Plain text.</param>
|
||
/// <param name="maxLength">The longest result before the ellipsis.</param>
|
||
/// <returns>
|
||
/// The text unchanged if it fits; otherwise everything up to the last space
|
||
/// within <paramref name="maxLength"/>, trailing punctuation removed, plus
|
||
/// <c>…</c>. A single word longer than <paramref name="maxLength"/> is cut hard.
|
||
/// </returns>
|
||
/// <remarks>
|
||
/// The space is searched one character past the limit: if the text breaks
|
||
/// exactly at <paramref name="maxLength"/>, the last whole word still fits.
|
||
/// </remarks>
|
||
private static string Shorten(string text, int maxLength) =>
|
||
text.Length <= maxLength
|
||
? text
|
||
: (text[..(maxLength + 1)].LastIndexOf(' ') is var space and > 0 ? text[..space] : text[..maxLength])
|
||
.TrimEnd(' ', ',', ';', ':', '.', '-', '–') + "…";
|
||
|
||
/// <summary>
|
||
/// Matches the end mark of a sentence: <c>.</c>, <c>!</c> or <c>?</c> before a
|
||
/// space or the end, unless it follows a lone letter (<c>z.</c>) or a number
|
||
/// (<c>13.</c>).
|
||
/// </summary>
|
||
[GeneratedRegex(@"(?<!\b\p{L}|\b\d+)[.!?](?=\s|$)")]
|
||
private static partial Regex SentenceEnd();
|
||
|
||
/// <summary>
|
||
/// Matches a run of whitespace, collapsed to one space in plain text.
|
||
/// </summary>
|
||
[GeneratedRegex(@"\s+")]
|
||
private static partial Regex Whitespace();
|
||
|
||
/// <summary>
|
||
/// Turns the list items of every <c>::: team</c> block into person cards.
|
||
/// </summary>
|
||
/// <param name="document">The parsed document; changed in place.</param>
|
||
/// <returns>The same <paramref name="document"/>, for chaining.</returns>
|
||
/// <remarks>
|
||
/// Only lists directly in the block count; a nested list inside a card stays a
|
||
/// list. <c>::: team gross</c> adds the class
|
||
/// <see cref="TeamMember.LargeModifier"/> to the block, because Markdig only
|
||
/// turns the block name into a class, not the words after it.
|
||
/// </remarks>
|
||
private static MarkdownDocument BuildTeamCards(MarkdownDocument document)
|
||
{
|
||
foreach (var team in document.Descendants<CustomContainer>().Where(TeamMember.IsTeam).ToList())
|
||
{
|
||
if (TeamMember.IsLarge(team))
|
||
{
|
||
team.GetAttributes().AddClass(TeamMember.LargeModifier);
|
||
}
|
||
|
||
foreach (var item in team.OfType<ListBlock>().SelectMany(list => list.OfType<ListItemBlock>()))
|
||
{
|
||
TeamMember.ToCard(item);
|
||
}
|
||
}
|
||
|
||
return document;
|
||
}
|
||
|
||
/// <summary>
|
||
/// Adds <see cref="LoneLinkClass"/> to every paragraph that holds nothing but
|
||
/// one link.
|
||
/// </summary>
|
||
/// <param name="document">The parsed document; changed in place.</param>
|
||
/// <returns>The same <paramref name="document"/>, for chaining.</returns>
|
||
/// <remarks>
|
||
/// Inside a list item the class goes on the item, not the paragraph: in a tight
|
||
/// list Markdig writes no <c><p></c> at all, so a class on the paragraph
|
||
/// would be silently dropped.
|
||
/// </remarks>
|
||
private static MarkdownDocument MarkLoneLinks(MarkdownDocument document)
|
||
{
|
||
foreach (var paragraph in document.Descendants<ParagraphBlock>().Where(IsLoneLink))
|
||
{
|
||
MarkdownObject target = paragraph.Parent is ListItemBlock { Count: 1 } item ? item : paragraph;
|
||
target.GetAttributes().AddClass(LoneLinkClass);
|
||
}
|
||
|
||
return document;
|
||
}
|
||
|
||
/// <summary>
|
||
/// Whether a paragraph's content is exactly one link, ignoring surrounding
|
||
/// whitespace and line breaks.
|
||
/// </summary>
|
||
/// <param name="paragraph">The paragraph to inspect.</param>
|
||
/// <returns>
|
||
/// <see langword="true"/> for one link (inline <c>[text](url)</c> or an
|
||
/// autolink <c><url></c>, but not an image) and nothing else.
|
||
/// </returns>
|
||
private static bool IsLoneLink(ParagraphBlock paragraph) =>
|
||
paragraph.Inline?
|
||
.Where(inline => !IsBlank(inline))
|
||
.ToList() is [LinkInline { IsImage: false } or AutolinkInline];
|
||
}
|