245 lines
11 KiB
C#
245 lines
11 KiB
C#
using System.Text;
|
|
using System.Text.RegularExpressions;
|
|
using Markdig;
|
|
using Markdig.Renderers.Html;
|
|
using Markdig.Syntax;
|
|
using Markdig.Syntax.Inlines;
|
|
|
|
namespace Elternbeirat.Web.Shared;
|
|
|
|
/// <summary>
|
|
/// Renders the editors' Markdown (the <c>body</c>, <c>intro</c> and <c>answer</c>
|
|
/// fields from PocketBase) to the HTML the pages show.
|
|
/// </summary>
|
|
/// <remarks>
|
|
/// The pipeline is CommonMark plus exactly two extensions, each there for a reason
|
|
/// an editor can see:
|
|
/// <list type="bullet">
|
|
/// <item>
|
|
/// <description>
|
|
/// <b>Custom containers</b> (<c>::: name</c> … <c>:::</c>) turn a block
|
|
/// into <c><div class="name"></c>. That is how editors use the
|
|
/// design blocks (<c>kennzahlen</c>, <c>aufruf</c>, <c>kacheln</c>,
|
|
/// <c>hinweis</c>) without writing HTML. An unknown name just yields a
|
|
/// div without styling, so a typo never breaks a page.
|
|
/// </description>
|
|
/// </item>
|
|
/// <item>
|
|
/// <description>
|
|
/// <b>Pipe tables</b>: plain CommonMark has no tables at all, so without
|
|
/// this the table styles in <c>app.css</c> could never apply.
|
|
/// </description>
|
|
/// </item>
|
|
/// </list>
|
|
/// <para>
|
|
/// Raw HTML in the Markdown is <b>escaped</b>, not passed through
|
|
/// (<see cref="MarkdownPipelineBuilder"/>'s <c>DisableHtml</c>). The result is
|
|
/// rendered as a <c>MarkupString</c>, so a passed-through <c><script></c>
|
|
/// would run in the visitor's browser, and the privacy text promises that no
|
|
/// program code runs there. Editors style pages with the blocks instead.
|
|
/// </para>
|
|
/// <para>
|
|
/// After parsing, a paragraph (or list item) that consists of nothing but one
|
|
/// link gets the class <see cref="LoneLinkClass"/>. Pure CSS cannot tell
|
|
/// "a link alone on its line" from "a link inside a sentence" (selectors ignore
|
|
/// the text around an element), but the stylesheet needs exactly that to show a
|
|
/// lone <c>mailto:</c> link as a button and a lone PDF link as a file card.
|
|
/// </para>
|
|
/// </remarks>
|
|
/// <seealso cref="IconSet"/>
|
|
public static partial class Markdown
|
|
{
|
|
/// <summary>
|
|
/// The class marking a paragraph or list item whose only content is one link.
|
|
/// <c>app.css</c> keys the mail button and the PDF card off it.
|
|
/// </summary>
|
|
public const string LoneLinkClass = "lone-link";
|
|
|
|
private static readonly MarkdownPipeline Pipeline =
|
|
new MarkdownPipelineBuilder()
|
|
.UseCustomContainers()
|
|
.UsePipeTables()
|
|
.DisableHtml()
|
|
.Build();
|
|
|
|
/// <summary>
|
|
/// Converts a Markdown string to HTML.
|
|
/// </summary>
|
|
/// <param name="markdown">The editor's Markdown; may be <see langword="null"/>.</param>
|
|
/// <returns>
|
|
/// The rendered HTML, or an empty string for <see langword="null"/> or empty
|
|
/// input, so callers can bind the result directly.
|
|
/// </returns>
|
|
/// <example>
|
|
/// <code>
|
|
/// Markdown.ToHtml("::: hinweis\nBitte vormerken.\n:::");
|
|
/// // <div class="hinweis"><p>Bitte vormerken.</p></div>
|
|
/// </code>
|
|
/// </example>
|
|
public static string ToHtml(string? markdown) =>
|
|
string.IsNullOrEmpty(markdown) ? "" : Render(Parse(markdown));
|
|
|
|
/// <summary>
|
|
/// Extracts the first sentence of a Markdown text as plain text, e.g. as the
|
|
/// short description on a home page tile.
|
|
/// </summary>
|
|
/// <param name="markdown">The editor's Markdown; may be <see langword="null"/>.</param>
|
|
/// <returns>
|
|
/// The first sentence of the first top-level paragraph, with all Markdown
|
|
/// removed, whitespace collapsed and a trailing colon dropped, or an empty
|
|
/// string when there is no such paragraph (empty body, or only headings, lists
|
|
/// and blocks).
|
|
/// </returns>
|
|
/// <remarks>
|
|
/// Only top-level paragraphs count: a heading repeats the title, and the text in
|
|
/// a list or a <c>:::</c> block is rarely a sentence that describes the page.
|
|
/// <para>
|
|
/// A sentence ends at <c>.</c>, <c>!</c> or <c>?</c> followed by a space or
|
|
/// the end. A period after a single letter or a number does not count, so
|
|
/// <c>z. B.</c> and <c>13. November</c> do not cut the sentence short. A
|
|
/// longer abbreviation such as <c>bzw.</c> still does; that costs the rest
|
|
/// of a teaser, never the page.
|
|
/// </para>
|
|
/// <para>
|
|
/// A paragraph that leads into a list often ends in <c>:</c> without a
|
|
/// period; as a teaser the colon would point at a list that is not there.
|
|
/// The sentence is returned whole: a long one is cut by the tile's CSS,
|
|
/// which knows the space it has, not by a character count.
|
|
/// </para>
|
|
/// </remarks>
|
|
/// <example>
|
|
/// <code>
|
|
/// Markdown.FirstSentence("# Vorstand\n\nWir sind **sieben** Eltern. Mehr unten.");
|
|
/// // "Wir sind sieben Eltern."
|
|
/// </code>
|
|
/// </example>
|
|
public static string FirstSentence(string? markdown) =>
|
|
string.IsNullOrWhiteSpace(markdown)
|
|
? ""
|
|
: Parse(markdown).OfType<ParagraphBlock>().FirstOrDefault()?.Inline is { } inline
|
|
? UpToSentenceEnd(PlainText(inline)).TrimEnd(':', ' ')
|
|
: "";
|
|
|
|
/// <summary>
|
|
/// Parses Markdown with the site's pipeline, for callers that take the document
|
|
/// apart before rendering it (see <see cref="Render"/>).
|
|
/// </summary>
|
|
/// <param name="markdown">The editor's Markdown.</param>
|
|
/// <returns>The parsed document.</returns>
|
|
internal static MarkdownDocument Parse(string markdown) =>
|
|
Markdig.Markdown.Parse(markdown, Pipeline);
|
|
|
|
/// <summary>
|
|
/// Renders a document from <see cref="Parse"/> to HTML, marking lone links on
|
|
/// the way, exactly as <see cref="ToHtml"/> does.
|
|
/// </summary>
|
|
/// <param name="document">The parsed document; changed in place.</param>
|
|
/// <returns>The rendered HTML.</returns>
|
|
internal static string Render(MarkdownDocument document) =>
|
|
Markdig.Markdown.ToHtml(MarkLoneLinks(document), Pipeline);
|
|
|
|
/// <summary>
|
|
/// Flattens inline Markdown to the text a reader sees: emphasis and link
|
|
/// markup dropped, link text kept, images left out, whitespace collapsed.
|
|
/// </summary>
|
|
/// <param name="inline">The inline content, e.g. of a paragraph or heading.</param>
|
|
/// <returns>The visible text, trimmed.</returns>
|
|
internal static string PlainText(ContainerInline inline) =>
|
|
Whitespace().Replace(AppendText(new StringBuilder(), inline).ToString(), " ").Trim();
|
|
|
|
/// <summary>
|
|
/// Whether an inline carries no visible content (whitespace or a line break).
|
|
/// </summary>
|
|
/// <param name="inline">The inline to inspect.</param>
|
|
/// <returns>
|
|
/// <see langword="true"/> if it can be ignored when looking for links that stand
|
|
/// alone in a paragraph.
|
|
/// </returns>
|
|
internal static bool IsBlank(Inline inline) =>
|
|
inline switch
|
|
{
|
|
LineBreakInline => true,
|
|
LiteralInline literal => literal.Content.IsEmptyOrWhitespace(),
|
|
_ => false,
|
|
};
|
|
|
|
/// <summary>
|
|
/// Appends the visible text of an inline and its children.
|
|
/// </summary>
|
|
/// <param name="text">The builder to append to.</param>
|
|
/// <param name="inline">The inline to flatten.</param>
|
|
/// <returns>The same <paramref name="text"/>, for chaining.</returns>
|
|
private static StringBuilder AppendText(StringBuilder text, Inline inline) =>
|
|
inline switch
|
|
{
|
|
LiteralInline literal => text.Append(literal.Content.ToString()),
|
|
CodeInline code => text.Append(code.Content),
|
|
HtmlEntityInline entity => text.Append(entity.Transcoded.ToString()),
|
|
AutolinkInline autolink => text.Append(autolink.Url),
|
|
LineBreakInline => text.Append(' '),
|
|
LinkInline { IsImage: true } => text,
|
|
ContainerInline container => container.Aggregate(text, AppendText),
|
|
_ => text,
|
|
};
|
|
|
|
/// <summary>
|
|
/// Cuts a text after its first sentence.
|
|
/// </summary>
|
|
/// <param name="text">Plain text.</param>
|
|
/// <returns>
|
|
/// The first sentence including its end mark, or the whole text if it has none.
|
|
/// </returns>
|
|
private static string UpToSentenceEnd(string text) =>
|
|
SentenceEnd().Match(text) is { Success: true } end ? text[..(end.Index + 1)] : text;
|
|
|
|
/// <summary>
|
|
/// Matches the end mark of a sentence: <c>.</c>, <c>!</c> or <c>?</c> before a
|
|
/// space or the end, unless it follows a lone letter (<c>z.</c>) or a number
|
|
/// (<c>13.</c>).
|
|
/// </summary>
|
|
[GeneratedRegex(@"(?<!\b\p{L}|\b\d+)[.!?](?=\s|$)")]
|
|
private static partial Regex SentenceEnd();
|
|
|
|
/// <summary>
|
|
/// Matches a run of whitespace, collapsed to one space in plain text.
|
|
/// </summary>
|
|
[GeneratedRegex(@"\s+")]
|
|
private static partial Regex Whitespace();
|
|
|
|
/// <summary>
|
|
/// Adds <see cref="LoneLinkClass"/> to every paragraph that holds nothing but
|
|
/// one link.
|
|
/// </summary>
|
|
/// <param name="document">The parsed document; changed in place.</param>
|
|
/// <returns>The same <paramref name="document"/>, for chaining.</returns>
|
|
/// <remarks>
|
|
/// Inside a list item the class goes on the item, not the paragraph: in a tight
|
|
/// list Markdig writes no <c><p></c> at all, so a class on the paragraph
|
|
/// would be silently dropped.
|
|
/// </remarks>
|
|
private static MarkdownDocument MarkLoneLinks(MarkdownDocument document)
|
|
{
|
|
foreach (var paragraph in document.Descendants<ParagraphBlock>().Where(IsLoneLink))
|
|
{
|
|
MarkdownObject target = paragraph.Parent is ListItemBlock { Count: 1 } item ? item : paragraph;
|
|
target.GetAttributes().AddClass(LoneLinkClass);
|
|
}
|
|
|
|
return document;
|
|
}
|
|
|
|
/// <summary>
|
|
/// Whether a paragraph's content is exactly one link, ignoring surrounding
|
|
/// whitespace and line breaks.
|
|
/// </summary>
|
|
/// <param name="paragraph">The paragraph to inspect.</param>
|
|
/// <returns>
|
|
/// <see langword="true"/> for one link (inline <c>[text](url)</c> or an
|
|
/// autolink <c><url></c>, but not an image) and nothing else.
|
|
/// </returns>
|
|
private static bool IsLoneLink(ParagraphBlock paragraph) =>
|
|
paragraph.Inline?
|
|
.Where(inline => !IsBlank(inline))
|
|
.ToList() is [LinkInline { IsImage: false } or AutolinkInline];
|
|
}
|