Private/Generator/ConvertTo-OpenApiGenHelpText.ps1
|
function ConvertTo-OpenApiGenHelpText { <# .SYNOPSIS Turns a CommonMark/HTML description from an OpenAPI document into plain help text that also passes through PlatyPS into MDX (Docusaurus, Astro Starlight) unchanged. .DESCRIPTION Fenced code blocks and inline code spans are kept as they are. In the rest: - HTML comments are removed; <code>x</code> becomes `x`; <br> a line break; <p>, headings and lists paragraph breaks and '- ' items; <a href="u">t</a> 't (u)'; other tags are removed; < > " ' ' and & are decoded. - Markdown links and images become 'text (url)' (just the url when both are the same), <https://...> autolinks the url, **bold**, __bold__, *emphasis* and ~~strike~~ lose their markers, '#' heading markers and '>' quote markers are removed, '* ' and '+ ' list items become '- ', and backslash escapes are resolved. - A word that holds '<', '{' or '}' (after the above) is put in backticks, because MDX reads '<' as JSX and '{' as an expression, and a line that starts with 'import ' or 'export ' (read as ESM by MDX) gets a capital letter. Line endings become LF, trailing spaces are removed and more than one blank line becomes one. -SingleLine joins the lines with spaces (for a synopsis). #> [CmdletBinding()] [OutputType([string])] param( [Parameter()] [AllowNull()] [AllowEmptyString()] [string]$Text, [Parameter()] [switch]$SingleLine ) if ([string]::IsNullOrWhiteSpace($Text)) { return '' } $source = $Text.Replace("`r`n", "`n").Replace("`r", "`n").Replace("`t", ' ') # Fenced code blocks stay as they are $segments = New-Object -TypeName System.Collections.ArrayList $current = New-Object -TypeName System.Collections.ArrayList $fence = $null foreach ($line in $source.Split("`n")) { if ($null -eq $fence) { $open = [regex]::Match($line, '^\s{0,3}(```+|~~~+)') if ($open.Success) { if ($current.Count -gt 0) { [void]$segments.Add([pscustomobject]@{ Code = $false; Text = ($current -join "`n") }) $current.Clear() } $fence = $open.Groups[1].Value [void]$current.Add($line.Trim()) continue } [void]$current.Add($line) continue } [void]$current.Add($line) if ($line.Trim().StartsWith($fence)) { [void]$segments.Add([pscustomobject]@{ Code = $true; Text = ($current -join "`n") }) $current.Clear() $fence = $null } } if ($current.Count -gt 0) { if ($null -ne $fence) { # An unclosed fence: close it so the rest of the page is not read as code [void]$current.Add($fence) } [void]$segments.Add([pscustomobject]@{ Code = ($null -ne $fence); Text = ($current -join "`n") }) } $parts = foreach ($segment in $segments) { if ($segment.Code) { "`n" + $segment.Text + "`n" continue } $prose = [regex]::Replace($segment.Text, '<!--[\s\S]*?-->', '') $prose = [regex]::Replace($prose, '(?is)<code\b[^>]*>(.*?)</code>', { param($Match) '`' + $Match.Groups[1].Value.Replace('`', '') + '`' }) # Inline code spans stay as they are $pieces = [regex]::Split($prose, '(`+[^`]+?`+)') $converted = foreach ($piece in $pieces) { if ($piece -match '^`+[^`]+?`+$') { $piece continue } # Backslash escapes: hide the escaped character until the markers are handled $t = [regex]::Replace($piece, '\\([\\`*_{}\[\]()#+\-.!<>|~])', { param($Match) [string][char](0xE000 + [int][char]$Match.Groups[1].Value) }) $t = [regex]::Replace($t, '<((?:https?|mailto):[^<>\s]+)>', '$1') $t = [regex]::Replace($t, '(?is)<a\b[^>]*\bhref\s*=\s*["'']([^"'']*)["''][^>]*>(.*?)</a>', { param($Match) $label = $Match.Groups[2].Value.Trim() $url = $Match.Groups[1].Value.Trim() if ($label -eq '' -or $label -eq $url) { $url } else { "$label ($url)" } }) $t = [regex]::Replace($t, '(?i)<br\s*/?>', "`n") $t = [regex]::Replace($t, '(?i)</?(p|div|h[1-6]|table|tr)\b[^>]*>', "`n`n") $t = [regex]::Replace($t, '(?i)<li\b[^>]*>', "`n- ") $t = [regex]::Replace($t, '(?i)</?(ul|ol)\b[^>]*>', "`n") $t = [regex]::Replace($t, '(?i)</li\s*>', '') $t = [regex]::Replace($t, '(?i)</?(a|abbr|b|big|blockquote|caption|center|cite|dd|del|details|dfn|dl|dt|em|font|hr|i|img|ins|kbd|mark|pre|s|samp|small|span|strike|strong|sub|summary|sup|tbody|td|tfoot|th|thead|tt|u|var)\b(\s[^<>]*)?/?>', '') $t = $t.Replace('<', '<').Replace('>', '>').Replace('"', '"').Replace(''', "'").Replace(''', "'").Replace(' ', ' ').Replace('&', '&') $t = [regex]::Replace($t, '!\[([^\]]*)\]\(\s*([^)\s]+)[^)]*\)', { param($Match) if ($Match.Groups[1].Value.Trim() -eq '') { $Match.Groups[2].Value } else { "$($Match.Groups[1].Value) ($($Match.Groups[2].Value))" } }) $t = [regex]::Replace($t, '\[([^\]]+)\]\(\s*([^)\s]+)[^)]*\)', { param($Match) if ($Match.Groups[1].Value -eq $Match.Groups[2].Value) { $Match.Groups[2].Value } else { "$($Match.Groups[1].Value) ($($Match.Groups[2].Value))" } }) $t = [regex]::Replace($t, '(?m)^(\s*)[*+]\s+', '$1- ') $t = [regex]::Replace($t, '\*\*(.+?)\*\*', '$1') $t = [regex]::Replace($t, '(?<![\w_])__(.+?)__(?![\w_])', '$1') $t = [regex]::Replace($t, '(?<![\w*])\*(?![\s*])(.+?)(?<![\s*])\*(?![\w*])', '$1') $t = [regex]::Replace($t, '~~(.+?)~~', '$1') $t = [regex]::Replace($t, '(?m)^\s{0,3}#{1,6}[ \t]+(.*?)[ \t#]*$', "`n`$1`n") $t = [regex]::Replace($t, '(?m)^\s{0,3}>[ \t]?', '') $t = [regex]::Replace($t, '(?<=\S) {2,}(?=\S)', ' ') $t = [regex]::Replace($t, '[\uE000-\uE07F]', { param($Match) [string][char]([int][char]$Match.Value[0] - 0xE000) }) # MDX: '<' starts JSX and '{' an expression, so words that hold them become code $t = [regex]::Replace($t, '[^\s]*[<{}][^\s]*', { param($Match) $word = $Match.Value $lead = [regex]::Match($word, '^[(\["'']*').Value $rest = $word.Substring($lead.Length) $trail = [regex]::Match($rest, '[)\]"'',.;:!?]*$').Value $core = $rest.Substring(0, $rest.Length - $trail.Length) if ($core -eq '') { $core = $rest $trail = '' } $lead + '`' + $core.Replace('`', '') + '`' + $trail }) $t } -join $converted } $result = (-join $parts).Split("`n") | ForEach-Object -Process { $line = $_.TrimEnd() if ($line -cmatch '^(\s*)(import|export)(\s)') { $line = $Matches[1] + $Matches[2].Substring(0, 1).ToUpperInvariant() + $Matches[2].Substring(1) + $line.Substring($Matches[0].Length - 1) } $line } $result = [regex]::Replace(($result -join "`n"), '\n{3,}', "`n`n").Trim("`n", ' ') if ($SingleLine) { $result = [regex]::Replace($result, '\s*\n\s*', ' ') } return $result } |