Fix compile-time link validation and add tag page links

- Restore HTML link extraction in LinkValidator (removed in a83634d
  under the false premise that post bodies are raw markdown; they
  are HTML rendered by NimblePublisher at compile time). The missing
  regex made extract_links/1 find zero links, silently disabling
  compile-time validation.
- Support /blog/{blog_id}/tag/{tag} links: validate blog ID,
  require non-empty tag (tags are user-defined, e.g. pi.dev).
- Fix invalid links in two posts: tag/Pi.dev -> tag/pi.dev,
  2026-07-13-synthetic-tdd.md -> synthetic-tdd.
- Fix test warnings: use Plug.Test deprecation, unused import,
  runtime-defined TestBlogValid module.
- Add regression tests for HTML extraction and tag page links.
This commit is contained in:
Firehose Bot
2026-10-08 08:43:51 +01:00
parent 127eb474d2
commit c5db0cbda8
7 changed files with 80 additions and 10 deletions
+41 -5
View File
@@ -21,6 +21,12 @@ defmodule Blogex.LinkValidator do
* Must not contain consecutive hyphens
* Query strings and anchor fragments are allowed after the slug
## Tag page links
* `/blog/{blog_id}/tag/{tag}` links point to tag pages
* The tag may not be empty; otherwise any non-empty tag is accepted
(tags are user-defined and may contain dots, uppercase letters, etc.)
## Usage
# Validate a single link
@@ -50,21 +56,34 @@ defmodule Blogex.LinkValidator do
`/blog/{engineering|releases}/{slug}`. External links and non-blog
internal links are ignored.
Handles markdown link syntax `[text](url)`.
Handles both markdown link syntax `[text](url)` and HTML `<a href="url">`.
(Post bodies are HTML, rendered by NimblePublisher at compile time.)
## Examples
iex> extract_links("[link](/blog/engineering/post)")
["/blog/engineering/post"]
iex> extract_links("<p><a href=\"/blog/engineering/post\">link</a></p>")
["/blog/engineering/post"]
iex> extract_links("See [GitHub](https://github.com)")
[]
"""
@spec extract_links(String.t()) :: [String.t()]
def extract_links(body) when is_binary(body) do
~r/\[([^\]]+)\]\(([^)]+)\)/
|> Regex.scan(body)
|> Enum.map(fn [_, _, path] -> path end)
markdown_links =
~r/\[([^\]]+)\]\(([^)]+)\)/
|> Regex.scan(body)
|> Enum.map(fn [_, _, path] -> path end)
html_links =
~r/<a\s+href=["']([^"']*)["']/i
|> Regex.scan(body)
|> Enum.map(fn [_, path] -> path end)
(markdown_links ++ html_links)
|> Enum.uniq()
|> Enum.filter(&internal_blog_link?/1)
end
@@ -90,6 +109,9 @@ defmodule Blogex.LinkValidator do
iex> validate_link("/blog/engineering/My-Post")
{:error, "slug must be lowercase alphanumeric with hyphens: My-Post"}
iex> validate_link("/blog/engineering/tag/pi.dev")
:ok
"""
@spec validate_link(String.t()) :: :ok | {:error, String.t()}
def validate_link(link) when is_binary(link) do
@@ -99,7 +121,13 @@ defmodule Blogex.LinkValidator do
{blog_id_str, slug_part} ->
case Map.fetch(@valid_blog_ids, blog_id_str) do
{:ok, _blog_atom} -> validate_slug(slug_part)
{:ok, _blog_atom} ->
if String.starts_with?(slug_part, "tag/") do
validate_tag(String.replace_prefix(slug_part, "tag/", ""))
else
validate_slug(slug_part)
end
:error -> {:error, "unknown blog ID: #{blog_id_str}"}
end
end
@@ -209,4 +237,12 @@ defmodule Blogex.LinkValidator do
{:error, "slug must be lowercase alphanumeric with hyphens: #{slug}"}
end
end
@doc false
@spec validate_tag(String.t()) :: :ok | {:error, String.t()}
defp validate_tag(tag) when tag == "" do
{:error, "empty tag in tag link"}
end
defp validate_tag(_tag), do: :ok
end