defmodule Floki do alias Floki.Finder alias Floki.Parser @moduledoc """ A HTML parser and seeker. This is a simple HTML parser that enables searching using CSS like selectors. You can search elements by class, tag name and id. ## Example Assuming that you have the following HTML: ```html

Floki

Github page philss
``` You can perform the following queries: * Floki.find(html, "#content") : returns the section with all children; * Floki.find(html, ".headline") : returns a list with the `p` element; * Floki.find(html, "a") : returns a list with the `a` element; * Floki.find(html, "[data-model=user]") : returns a list with elements that match that data attribute; * Floki.find(html, "#content a") # returns all links inside content section; * Floki.find(html, ".headline, a") # returns the .headline elements and links. Each HTML node is represented by a tuple like: {tag_name, attributes, children_nodes} Example of node: {"p", [{"class", "headline"}], ["Floki"]} So even if the only child node is the element text, it is represented inside a list. You can write a simple HTML crawler (with support of [HTTPoison](https://github.com/edgurgel/httpoison)) with a few lines of code: html |> Floki.find(".pages a") |> Floki.attribute("href") |> Enum.map(fn(url) -> HTTPoison.get!(url) end) It is simple as that! """ @type html_tree :: tuple | list @doc """ Parses a HTML string. ## Examples iex> Floki.parse("
hello world
") {"div", [{"class", "js-action"}], ["hello world"]} iex> Floki.parse("
first
second
") [{"div", [], ["first"]}, {"div", [], ["second"]}] """ @spec parse(binary) :: html_tree def parse(html) do Parser.parse(html) end @self_closing_tags ["area", "base", "br", "col", "command", "embed", "hr", "img", "input", "keygen", "link", "mete", "param", "source", "track", "wbr"] @doc """ Converts HTML tree to raw HTML. Note that the resultant HTML may be different from the original one. Spaces after tags and doctypes are ignored. ## Examples iex> Floki.parse(~s(
my content
)) |> Floki.raw_html ~s(
my content
) """ def raw_html(html_tree), do: raw_html(html_tree, "") defp raw_html([], html), do: html defp raw_html(tuple, html) when is_tuple(tuple), do: raw_html([tuple], html) defp raw_html([string|tail], html) when is_binary(string), do: raw_html(tail, html <> string) defp raw_html([{:comment, comment}|tail], html), do: raw_html(tail, html <> "") defp raw_html([{type, attrs, children}|tail], html) do raw_html(tail, html <> tag_for(type, tag_attrs(attrs), children)) end defp tag_attrs(attr_list) do attr_list |> Enum.reduce("", fn({attr, value}, attrs) -> ~s(#{attrs} #{attr}="#{value}") end) |> String.strip end defp tag_for(type, attrs, _children) when type in @self_closing_tags do case attrs do "" -> "<#{type}/>" _ -> "<#{type} #{attrs}/>" end end defp tag_for(type, attrs, children) do case attrs do "" -> "<#{type}>#{raw_html(children)}" _ -> "<#{type} #{attrs}>#{raw_html(children)}" end end @doc """ Find elements inside a HTML tree or string. ## Examples iex> Floki.find("

hello

", ".hint") [{"span", [{"class", "hint"}], ["hello"]}] iex> Floki.find("
Content
", "#important") [{"div", [{"id", "important"}], [{"div", [], ["Content"]}]}] iex> Floki.find("

Google

", "a") [{"a", [{"href", "https://google.com"}], ["Google"]}] """ @spec find(binary | html_tree, binary) :: html_tree def find(html, selector) when is_binary(html) do parse(html) |> Finder.find(selector) end def find(html_tree, selector) do Finder.find(html_tree, selector) end @doc """ Returns the text nodes from a HTML tree. By default, it will perform a deep search through the HTML tree. You can disable deep search with the option `deep` assigned to false. ## Examples iex> Floki.text("
hello world
") "hello world" iex> Floki.text("
hello world
", deep: false) " world" """ @spec text(html_tree | binary) :: binary def text(html, opts \\ [deep: true]) do html_tree = case is_binary(html) do true -> parse(html) false -> html end search_strategy = case opts[:deep] do true -> Floki.DeepText false -> Floki.FlatText end search_strategy.get(html_tree) end @doc """ Returns a list with attribute values for a given selector. ## Examples iex> Floki.attribute("Google", "a", "href") ["https://google.com"] """ @spec attribute(binary | html_tree, binary, binary) :: list def attribute(html, selector, attribute_name) do html |> find(selector) |> attribute_values(attribute_name) end @doc """ Returns a list with attribute values from elements. ## Examples iex> Floki.attribute("Google", "href") ["https://google.com"] """ @spec attribute(binary | html_tree, binary) :: list def attribute(html_tree, attribute_name) when is_binary(html_tree) do html_tree |> parse |> attribute_values(attribute_name) end def attribute(elements, attribute_name) do attribute_values(elements, attribute_name) end defp attribute_values(element, attr_name) when is_tuple(element) do attribute_values([element], attr_name) end defp attribute_values(elements, attr_name) do values = Enum.reduce elements, [], fn({_, attributes, _}, acc) -> case attribute_match?(attributes, attr_name) do {_attr_name, value} -> [value|acc] _ -> acc end end Enum.reverse(values) end defp attribute_match?(attributes, attribute_name) do Enum.find attributes, fn({attr_name, _}) -> attr_name == attribute_name end end end