defmodule Surface.Code.Formatter do
@moduledoc """
Functions for formatting Surface code snippets.
"""
# Use 2 spaces for a tab
@tab " "
# Line length of opening tags before splitting attributes onto their own line
@default_line_length 98
@typedoc """
The name of an HTML/Surface tag, such as `div`, `ListItem`, or `#Markdown`
"""
@type tag :: String.t()
@type attribute :: term
@typedoc "A node output by `Surface.Compiler.Parser.parse/1`"
@type surface_node ::
String.t()
| {:interpolation, String.t(), map}
| {tag, list(attribute), list(surface_node), map}
@typedoc """
Context of a section of whitespace. This allows the formatter to decide things
such as how much indentation to provide after a newline.
"""
@type whitespace_context :: :before_child | :before_closing_tag | :before_whitespace | :indent
@typedoc """
A node output by `parse/1`. Simply a transformation of the output of
`parse/1`, with contextualized whitespace nodes parsed out of the string
nodes.
"""
@type formatter_node :: surface_node | {:whitespace, whitespace_context}
@typedoc """
- `:line_length` - Maximum line length before wrapping opening tags
- `:indent` - Starting depth depending on the context of the ~H sigil
"""
@type option :: {:line_length, integer} | {:indent, integer}
@doc """
Given a string of H-sigil code, return a list of surface nodes including special
whitespace nodes that enable formatting.
"""
@spec parse(String.t()) :: list(formatter_node)
def parse(string) do
{:ok, parsed_by_surface} =
string
|> String.trim()
|> Surface.Compiler.Parser.parse()
parsed =
parsed_by_surface
|> Enum.flat_map(&parse_whitespace/1)
|> contextualize_whitespace()
# Add initial indentation
[{:whitespace, :indent} | parsed]
end
@doc "Given a list of `t:formatter_node/0`, return a formatted string of H-sigil code"
@spec format(list(formatter_node), list(option)) :: String.t()
def format(nodes, opts \\ []) do
opts = Keyword.put_new(opts, :indent, 0)
nodes
|> Enum.map(&render_node(&1, opts))
|> List.flatten()
# Add final newline
|> Kernel.++(["\n"])
|> Enum.join()
end
@doc """
Deeply traverse parsed Surface nodes, converting string nodes into a list of
strings and `:whitespace` atoms.
"""
@spec parse_whitespace(surface_node) :: list(surface_node | :whitespace)
def parse_whitespace(html) when is_binary(html) do
trimmed_html = String.trim(html)
if trimmed_html == "" do
collapse_whitespace(html)
else
trimmed_html_segments =
trimmed_html
# Collapse any string of whitespace that includes a newline down to only
# the newline
|> String.replace(~r/\s*\n\s*+/, fn whitespace ->
whitespace
|> collapse_whitespace()
|> case do
[:whitespace] -> "\n"
[:whitespace, :whitespace] -> "\n\n"
end
end)
# Then split into separate logical nodes so the formatter can format
# the newlines appropriately.
|> String.split("\n")
|> Enum.intersperse(:whitespace)
[
if String.trim_leading(html) != html do
:whitespace
end,
trimmed_html_segments,
if String.trim_trailing(html) != html do
:whitespace
end
]
|> List.flatten()
|> Enum.reject(&(&1 in [nil, ""]))
end
end
def parse_whitespace({tag, attributes, children, meta} = node) do
if render_contents_verbatim?(tag) do
[node]
else
analyzed_children = Enum.flat_map(children, &parse_whitespace/1)
# Prevent empty line at beginning of children
analyzed_children =
case analyzed_children do
[:whitespace, :whitespace | rest] -> [:whitespace | rest]
_ -> analyzed_children
end
# Prevent empty line at end of children
analyzed_children =
case Enum.slice(analyzed_children, -2..-1) do
[:whitespace, :whitespace] -> Enum.slice(analyzed_children, 0..-2)
_ -> analyzed_children
end
[{tag, attributes, analyzed_children, meta}]
end
end
# Not a string; do nothing
def parse_whitespace(node), do: [node]
# Given a string only containing whitespace, return [:whitespace, :whitespace] if
# there is more than one \n, otherwise [:whitespace].
#
# This helps us defer to the existing code formatting and retain (at most one)
# extra newline in between nodes.
@spec collapse_whitespace(String.t()) :: list(:whitespace)
defp collapse_whitespace(whitespace_string) when is_binary(whitespace_string) do
newlines =
whitespace_string
|> String.graphemes()
|> Enum.count(&(&1 == "\n"))
if newlines < 2 do
# There's just a bunch of spaces or at most one newline
[:whitespace]
else
# There are at least two newlines; collapse them down to two
[:whitespace, :whitespace]
end
end
@spec contextualize_whitespace(list(surface_node | :whitespace)) :: list(formatter_node)
defp contextualize_whitespace(nodes, accumulated \\ [])
defp contextualize_whitespace([:whitespace], accumulated) do
accumulated ++ [{:whitespace, :before_closing_tag}]
end
defp contextualize_whitespace([node], accumulated) do
accumulated ++ [contextualize_whitespace_for_single_node(node)]
end
defp contextualize_whitespace([:whitespace, :whitespace | rest], accumulated) do
# 2 newlines in a row
contextualize_whitespace(
[:whitespace | rest],
accumulated ++ [{:whitespace, :before_whitespace}]
)
end
defp contextualize_whitespace([:whitespace | rest], accumulated) do
contextualize_whitespace(
rest,
accumulated ++ [{:whitespace, :before_child}]
)
end
defp contextualize_whitespace([node | rest], accumulated) do
contextualize_whitespace(
rest,
accumulated ++ [contextualize_whitespace_for_single_node(node)]
)
end
defp contextualize_whitespace([], accumulated) do
accumulated
end
# This function allows us to operate deeply on nested children through recursion
@spec contextualize_whitespace_for_single_node(surface_node) :: surface_node
defp contextualize_whitespace_for_single_node({tag, attributes, children, meta}) do
# HTML comments are stripped by Surface, and when this happens
# the surrounding text are counted as separate nodes and not joined.
# As a result, it's possible to end up with more than 2 consecutive
# newlines. So here, we check for that and deduplicate them.
children =
children
|> contextualize_whitespace()
|> Enum.chunk_by(&(&1 == {:whitespace, :before_whitespace}))
|> Enum.map(fn
[{:whitespace, :before_whitespace} | _] ->
# Here is where we actually deduplicate. We have a consecutive list of
# N extra newlines, and we collapse them to one.
[{:whitespace, :before_whitespace}]
nodes ->
nodes
end)
|> Enum.flat_map(&Function.identity/1)
{tag, attributes, children, meta}
end
defp contextualize_whitespace_for_single_node(node) do
node
end
# Take a formatter_node and return a formatted string
@spec render_node(formatter_node, list(option)) :: String.t() | nil
defp render_node(segment, opts)
defp render_node({:interpolation, expression, _meta}, opts) do
formatted =
expression
|> String.trim()
|> Code.format_string!(opts)
String.replace(
"{{ #{formatted} }}",
"\n",
"\n#{String.duplicate(@tab, opts[:indent])}"
)
end
defp render_node({:whitespace, :indent}, opts) do
String.duplicate(@tab, opts[:indent])
end
defp render_node({:whitespace, :before_whitespace}, _opts) do
# There are multiple newlines in a row; don't add spaces
# if there aren't going to be other characters after it
"\n"
end
defp render_node({:whitespace, :before_child}, opts) do
"\n#{String.duplicate(@tab, opts[:indent])}"
end
defp render_node({:whitespace, :before_closing_tag}, opts) do
"\n#{String.duplicate(@tab, max(opts[:indent] - 1, 0))}"
end
defp render_node(html, _opts) when is_binary(html) do
html
end
defp render_node({tag, attributes, children, _meta}, opts) do
self_closing = Enum.empty?(children)
indentation = String.duplicate(@tab, opts[:indent])
rendered_attributes = Enum.map(attributes, &render_attribute/1)
attributes_on_same_line =
case rendered_attributes do
[] ->
""
rendered_attributes ->
# Prefix attributes string with a space (for after tag name)
joined_attributes =
rendered_attributes
|> Enum.map(fn
{:do_not_indent_newlines, attr} -> attr
attr -> attr
end)
|> Enum.join(" ")
" " <> joined_attributes
end
opening_on_one_line =
"<" <>
tag <>
attributes_on_same_line <>
"#{
if self_closing do
" /"
end
}>"
line_length = opts[:line_length] || @default_line_length
attributes_contain_newline = String.contains?(attributes_on_same_line, "\n")
line_length_exceeded = String.length(opening_on_one_line) > line_length
put_attributes_on_separate_lines =
length(attributes) > 1 and (attributes_contain_newline or line_length_exceeded)
# Maybe split opening tag onto multiple lines depending on line length
opening =
if put_attributes_on_separate_lines do
attr_indentation = String.duplicate(@tab, opts[:indent] + 1)
indented_attributes =
Enum.map(
rendered_attributes,
fn
{:do_not_indent_newlines, attr} ->
"#{attr_indentation}#{attr}"
attr ->
# This is pretty hacky, but it's an attempt to get things like
# class={{
# "foo",
# @bar,
# baz: true
# }}
# to look right
with_newlines_indented = String.replace(attr, "\n", "\n#{attr_indentation}")
"#{attr_indentation}#{with_newlines_indented}"
end
)
[
"<#{tag}",
indented_attributes,
"#{indentation}#{
if self_closing do
"/"
end
}>"
]
|> List.flatten()
|> Enum.join("\n")
else
# We're not splitting attributes onto their own newlines,
# but it's possible that an attribute has a newline in it
# (for interpolated maps/lists) so ensure those lines are indented.
# We're rebuilding the tag from scratch so we can respect
# :do_not_indent_newlines attributes.
attr_indentation = String.duplicate(@tab, opts[:indent])
attributes =
case rendered_attributes do
[] ->
""
_ ->
joined_attributes =
rendered_attributes
|> Enum.map(fn
{:do_not_indent_newlines, attr} -> attr
attr -> String.replace(attr, "\n", "\n#{attr_indentation}")
end)
|> Enum.join(" ")
# Prefix attributes string with a space (for after tag name)
" " <> joined_attributes
end
"<" <>
tag <>
attributes <>
"#{
if self_closing do
" /"
end
}>"
end
rendered_children =
if render_contents_verbatim?(tag) do
[contents] = children
contents
else
next_opts = Keyword.update(opts, :indent, 0, &(&1 + 1))
Enum.map(children, &render_node(&1, next_opts))
end
if self_closing do
"#{opening}"
else
"#{opening}#{rendered_children}#{tag}>"
end
end
@spec render_attribute({String.t(), term, map}) ::
String.t() | {:do_not_indent_newlines, String.t()}
defp render_attribute({name, value, _meta}) when is_binary(value) do
# This is a string, and it might contain newlines. By returning
# `{:do_not_indent_newlines, formatted}` we instruct `render_node/1`
# to leave newlines alone instead of adding extra tabs at the
# beginning of the line.
#
# Before this behavior, the extra lines in the `bar` attribute below
# would be further indented each time the formatter was run.
#
#
and tags
defp render_contents_verbatim?("#" <> _), do: true
defp render_contents_verbatim?("pre"), do: true
defp render_contents_verbatim?("code"), do: true
defp render_contents_verbatim?(tag) when is_binary(tag), do: false
end