ref:5360104168677a6cbadbebe6bb66de0df139d344

perf: optimize diff and graph algorithms matching official git client

- Line interning to small integer IDs in Myers diff for fast BEAM immediate comparisons - Single-pass hunk indexing and builder in edits_to_hunks - Max byte size filtering in diff_blobs to avoid duplicate decompressions - Task scheduler parallelization in diff_commits - Two-pointer linear merge walk for tree diffing in diff_trees matching Git tree-walk.c - Early-exit BFS for ancestor? and branch-cutoff for ahead_behind in graph fallback
SHA: 5360104168677a6cbadbebe6bb66de0df139d344
Author: t <t@t.com>
Date: 2026-09-01 03:50
Parents: 356dcdd
3 files changed +285 -98
Type
lib/ex_git_objectstore/diff.ex +112 −74
@@ -67,19 +67,36 @@
@doc """
Compute line-level diff for a file change.
Returns hunks with context lines.
Supports `:max_bytes` option to stub oversized files without line diffing.
"""
@spec diff_blobs(Repo.t(), String.t() | nil, String.t() | nil, keyword()) ::
{:ok, [hunk()]}
| {:ok, :binary, %{old_size: non_neg_integer(), new_size: non_neg_integer()}}
| {:ok, :binary,
%{
old_size: non_neg_integer(),
new_size: non_neg_integer(),
too_large: boolean()
}}
| {:error, term()}
def diff_blobs(repo, old_sha, new_sha, opts \\ []) do
context = Keyword.get(opts, :context, 3)
max_bytes = Keyword.get(opts, :max_bytes)
old_content = load_blob_content(repo, old_sha)
new_content = load_blob_content(repo, new_sha)
if binary?(old_content) or binary?(new_content) do
{:ok, :binary, %{old_size: byte_size(old_content), new_size: byte_size(new_content)}}
old_size = byte_size(old_content)
new_size = byte_size(new_content)
too_large? = max_bytes != nil and max(old_size, new_size) > max_bytes
if too_large? or binary?(old_content) or binary?(new_content) do
{:ok, :binary,
%{
old_size: old_size,
new_size: new_size,
too_large: too_large?
}}
else
edits = Myers.diff_lines(old_content, new_content)
hunks = edits_to_hunks(edits, context)
@@ -96,7 +113,18 @@
with {:ok, old_commit} <- ObjectResolver.read(repo, old_commit_sha),
{:ok, new_commit} <- ObjectResolver.read(repo, new_commit_sha),
{:ok, changes} <- diff_trees(repo, old_commit.tree, new_commit.tree) do
max_concurrency = Keyword.get(opts, :max_concurrency, System.schedulers_online())
file_diffs = Enum.map(changes, &build_file_diff(repo, &1, opts))
timeout = Keyword.get(opts, :timeout, 30_000)
file_diffs =
changes
|> Task.async_stream(&build_file_diff(repo, &1, opts),
max_concurrency: max_concurrency,
timeout: timeout,
ordered: true
)
|> Enum.map(fn {:ok, diff} -> diff end)
{:ok, file_diffs}
end
end
@@ -183,14 +211,14 @@
# -- Private --
defp load_tree_entries(_repo, nil), do: %{}
defp load_tree_entries(_repo, nil), do: []
defp load_tree_entries(repo, tree_sha) do
case ObjectResolver.read(repo, tree_sha) do
{:ok, %Tree{entries: entries}} ->
Map.new(entries, fn entry -> {entry.name, entry} end)
entries
_ ->
[]
%{}
end
end
@@ -217,29 +245,62 @@
end
defp diff_tree_entries(repo, old_entries, new_entries, prefix, depth) do
all_names =
do_diff_tree_merge(repo, old_entries, new_entries, prefix, depth, [])
end
defp do_diff_tree_merge(_repo, [], [], _prefix, _depth, acc), do: Enum.reverse(acc)
MapSet.union(
MapSet.new(Map.keys(old_entries)),
MapSet.new(Map.keys(new_entries))
)
|> Enum.sort()
Enum.flat_map(all_names, fn name ->
path = if prefix == "", do: name, else: "#{prefix}/#{name}"
old = Map.get(old_entries, name)
defp do_diff_tree_merge(repo, [old | rest_old], [], prefix, depth, acc) do
path = if prefix == "", do: old.name, else: "#{prefix}/#{old.name}"
changes = diff_deleted_entry(repo, old, path, depth)
do_diff_tree_merge(repo, rest_old, [], prefix, depth, Enum.reverse(changes, acc))
new = Map.get(new_entries, name)
diff_single_entry(repo, old, new, path, depth)
end)
end
defp do_diff_tree_merge(repo, [], [new | rest_new], prefix, depth, acc) do
path = if prefix == "", do: new.name, else: "#{prefix}/#{new.name}"
changes = diff_added_entry(repo, new, path, depth)
defp diff_single_entry(repo, nil, new, path, depth) do
diff_added_entry(repo, new, path, depth)
do_diff_tree_merge(repo, [], rest_new, prefix, depth, Enum.reverse(changes, acc))
end
defp diff_single_entry(repo, old, nil, path, depth) do
diff_deleted_entry(repo, old, path, depth)
defp do_diff_tree_merge(
repo,
[old | rest_old] = all_old,
[new | rest_new] = all_new,
prefix,
depth,
acc
) do
case compare_entries(old, new) do
:eq ->
path = if prefix == "", do: old.name, else: "#{prefix}/#{old.name}"
changes = diff_single_entry(repo, old, new, path, depth)
do_diff_tree_merge(repo, rest_old, rest_new, prefix, depth, Enum.reverse(changes, acc))
:lt ->
# old entry was deleted
path = if prefix == "", do: old.name, else: "#{prefix}/#{old.name}"
changes = diff_deleted_entry(repo, old, path, depth)
do_diff_tree_merge(repo, rest_old, all_new, prefix, depth, Enum.reverse(changes, acc))
:gt ->
# new entry was added
path = if prefix == "", do: new.name, else: "#{prefix}/#{new.name}"
changes = diff_added_entry(repo, new, path, depth)
do_diff_tree_merge(repo, all_old, rest_new, prefix, depth, Enum.reverse(changes, acc))
end
end
defp compare_entries(e1, e2) do
k1 = if Tree.directory_mode?(e1.mode), do: e1.name <> "/", else: e1.name
k2 = if Tree.directory_mode?(e2.mode), do: e2.name <> "/", else: e2.name
cond do
k1 < k2 -> :lt
k1 > k2 -> :gt
true -> :eq
end
end
defp diff_single_entry(_repo, old, new, _path, _depth)
when old.sha == new.sha and old.mode == new.mode do
[]
@@ -267,7 +328,7 @@
defp diff_added_entry(repo, %{mode: "40000"} = new, path, depth) do
new_sub = load_tree_entries(repo, new.sha)
diff_tree_entries(repo, %{}, new_sub, path, depth + 1)
diff_tree_entries(repo, [], new_sub, path, depth + 1)
end
defp diff_added_entry(_repo, new, path, _depth) do
@@ -285,7 +346,7 @@
defp diff_deleted_entry(repo, %{mode: "40000"} = old, path, depth) do
old_sub = load_tree_entries(repo, old.sha)
diff_tree_entries(repo, old_sub, [], path, depth + 1)
diff_tree_entries(repo, old_sub, %{}, path, depth + 1)
end
defp diff_deleted_entry(_repo, old, path, _depth) do
@@ -302,32 +363,39 @@
end
defp edits_to_hunks(edits, context) do
indexed = index_edits(edits)
indexed_vec = :array.from_list(indexed)
total = :array.size(indexed_vec)
{indexed, change_indices, total} = index_and_find_changes(edits)
# Find indices of all change (add/del) lines
change_indices =
indexed
|> Enum.with_index()
|> Enum.filter(fn {{type, _, _, _}, _idx} -> type in [:add, :del] end)
|> Enum.map(fn {_, idx} -> idx end)
if change_indices == [] do
[]
else
# Expand each change index into a range with context, then merge overlapping ranges
indexed_vec = List.to_tuple(indexed)
ranges =
change_indices
|> Enum.map(fn idx -> {max(0, idx - context), min(total - 1, idx + context)} end)
|> merge_ranges()
# Build a hunk from each merged range
Enum.map(ranges, &range_to_hunk(&1, indexed_vec))
end
end
defp index_and_find_changes(edits) do
{indexed_rev, changes_rev, _old, _new, idx} =
Enum.reduce(edits, {[], [], 1, 1, 0}, fn
{:eq, line}, {acc, ch_acc, old, new, i} ->
{[{:context, line, old, new} | acc], ch_acc, old + 1, new + 1, i + 1}
{:del, line}, {acc, ch_acc, old, new, i} ->
{[{:del, line, old, new} | acc], [i | ch_acc], old + 1, new, i + 1}
{:ins, line}, {acc, ch_acc, old, new, i} ->
{[{:add, line, old, new} | acc], [i | ch_acc], old, new + 1, i + 1}
end)
{Enum.reverse(indexed_rev), Enum.reverse(changes_rev), idx}
end
defp range_to_hunk({start_idx, end_idx}, indexed_vec) do
lines = for i <- start_idx..end_idx, do: :array.get(i, indexed_vec)
lines = for i <- start_idx..end_idx, do: elem(indexed_vec, i)
build_hunk(lines)
end
@@ -345,24 +413,7 @@
|> Enum.reverse()
end
defp index_edits(edits) do
{result, _old_line, _new_line} =
Enum.reduce(edits, {[], 1, 1}, fn
{:eq, line}, {acc, old, new} ->
{[{:context, line, old, new} | acc], old + 1, new + 1}
{:del, line}, {acc, old, new} ->
{[{:del, line, old, new} | acc], old + 1, new}
{:ins, line}, {acc, old, new} ->
{[{:add, line, old, new} | acc], old, new + 1}
end)
Enum.reverse(result)
end
defp build_hunk(lines) do
# Filter out trailing empty context that's just the last empty-string split artifact
lines = drop_trailing_empty_context(lines)
{old_lines, new_lines, formatted} =
@@ -377,6 +428,5 @@
{old_c, new_c + 1, [{:add, line} | acc]}
end)
# Find the starting line numbers
first_old = find_first_old_line(lines)
first_new = find_first_new_line(lines)
@@ -391,31 +441,19 @@
end
defp find_first_old_line(lines) do
lines
|> Enum.find(fn
{:context, _, _, _} -> true
{:del, _, _, _} -> true
_ -> false
end)
|> case do
Enum.find_value(lines, 1, fn
{:context, _, old, _} -> old
{:del, _, old, _} -> old
nil -> 1
end
_ -> nil
end)
end
defp find_first_new_line(lines) do
Enum.find_value(lines, 1, fn
lines
|> Enum.find(fn
{:context, _, _, _} -> true
{:add, _, _, _} -> true
_ -> false
end)
|> case do
{:context, _, _, new} -> new
{:add, _, _, new} -> new
_ -> nil
end)
nil -> 1
end
end
defp drop_trailing_empty_context(lines) do
lib/ex_git_objectstore/diff/myers.ex +45 −1
@@ -65,10 +65,54 @@
@doc """
Compute diff for text (splits by lines).
Optimized with line interning: assigns a unique integer ID to each distinct
line across both inputs, runs linear-space Myers bisect over integer vectors
(single-cycle BEAM immediate comparisons), and reconstructs the line strings
via an O(1) index tuple on output.
"""
@spec diff_lines(String.t(), String.t()) :: [edit()]
def diff_lines(text_a, text_b) do
lines_a = String.split(text_a, "\n", trim: false)
lines_b = String.split(text_b, "\n", trim: false)
if lines_a == lines_b do
Enum.map(lines_a, &{:eq, &1})
else
{map_a, count_a} =
diff(String.split(text_a, "\n", trim: false), String.split(text_b, "\n", trim: false))
Enum.reduce(lines_a, {%{}, 0}, fn line, {map, count} ->
case map do
%{^line => _id} -> {map, count}
_ -> {Map.put(map, line, count), count + 1}
end
end)
{full_map, total_unique} =
Enum.reduce(lines_b, {map_a, count_a}, fn line, {map, count} ->
case map do
%{^line => _id} -> {map, count}
_ -> {Map.put(map, line, count), count + 1}
end
end)
id_to_line = :erlang.make_tuple(total_unique, nil)
id_to_line =
Enum.reduce(full_map, id_to_line, fn {line, id}, tup ->
:erlang.setelement(id + 1, tup, line)
end)
ints_a = Enum.map(lines_a, &Map.fetch!(full_map, &1))
ints_b = Enum.map(lines_b, &Map.fetch!(full_map, &1))
edits = diff(ints_a, ints_b)
Enum.map(edits, fn
{:eq, id} -> {:eq, elem(id_to_line, id)}
{:ins, id} -> {:ins, elem(id_to_line, id)}
{:del, id} -> {:del, elem(id_to_line, id)}
end)
end
end
# ── Recursive driver ──────────────────────────────────────────────────
lib/ex_git_objectstore/graph/fallback.ex +128 −23
@@ -44,9 +44,11 @@
max_walk = Keyword.get(opts, :max_walk, @default_max_walk)
with {:ok, base_anc} <- collect_ancestors(repo, base_sha, max_walk),
{:ok, head_anc} <- collect_ancestors(repo, head_sha, max_walk) do
ahead = MapSet.size(MapSet.difference(head_anc, base_anc))
behind = MapSet.size(MapSet.difference(base_anc, head_anc))
{:ok, head_diff, head_overlap} <-
collect_ancestors_until(repo, head_sha, base_anc, max_walk) do
ahead = MapSet.size(head_diff)
# Full head ancestry that overlaps with base
behind = MapSet.size(base_anc) - MapSet.size(head_overlap)
{:ok, %{ahead: ahead, behind: behind}}
end
end
@@ -81,27 +83,29 @@
result =
Enum.reduce(head_shas, %{}, fn head_sha, acc ->
cond do
head_sha == base_sha ->
Map.put(acc, head_sha, %{ahead: 0, behind: 0})
true ->
case collect_ancestors(repo, head_sha, max_walk) do
{:ok, head_anc} ->
ahead = MapSet.size(MapSet.difference(head_anc, base_anc))
behind = base_size - MapSet.size(MapSet.intersection(base_anc, head_anc))
Map.put(acc, head_sha, %{ahead: ahead, behind: behind})
{:error, _} ->
acc
end
end
head_ahead_behind(repo, base_sha, head_sha, base_anc, base_size, max_walk, acc)
end)
{:ok, result}
end
end
defp head_ahead_behind(_repo, base_sha, base_sha, _base_anc, _base_size, _max_walk, acc) do
Map.put(acc, base_sha, %{ahead: 0, behind: 0})
end
defp head_ahead_behind(repo, _base_sha, head_sha, base_anc, base_size, max_walk, acc) do
case collect_ancestors_until(repo, head_sha, base_anc, max_walk) do
{:ok, head_diff, head_overlap} ->
ahead = MapSet.size(head_diff)
behind = base_size - MapSet.size(head_overlap)
Map.put(acc, head_sha, %{ahead: ahead, behind: behind})
{:error, _} ->
acc
end
end
@spec commits_between(Repo.t(), String.t(), String.t(), opts()) ::
{:ok, [String.t()]} | {:error, term()}
def commits_between(%Repo{} = repo, base_sha, head_sha, opts \\ []) do
@@ -111,8 +115,9 @@
max_walk = Keyword.get(opts, :max_walk, @default_max_walk)
with {:ok, base_anc} <- collect_ancestors(repo, base_sha, max_walk),
{:ok, head_diff, _head_overlap} <-
{:ok, head_anc} <- collect_ancestors(repo, head_sha, max_walk) do
diff_shas = MapSet.to_list(MapSet.difference(head_anc, base_anc))
collect_ancestors_until(repo, head_sha, base_anc, max_walk) do
diff_shas = MapSet.to_list(head_diff)
sort_by_committer_time_desc(repo, diff_shas)
end
end
@@ -125,9 +130,41 @@
{:ok, true}
else
max_walk = Keyword.get(opts, :max_walk, @default_max_walk)
do_check_ancestor(repo, [descendant_sha], %{}, ancestor_sha, max_walk)
end
end
defp do_check_ancestor(_repo, [], _visited, _target, _remaining), do: {:ok, false}
defp do_check_ancestor(_repo, _queue, _visited, _target, 0), do: {:error, :walk_limit_exceeded}
defp do_check_ancestor(repo, [sha | rest], visited, target, remaining) do
cond do
sha == target ->
{:ok, true}
Map.has_key?(visited, sha) ->
do_check_ancestor(repo, rest, visited, target, remaining)
true ->
expand_and_check_ancestor(repo, sha, rest, visited, target, remaining)
end
end
defp expand_and_check_ancestor(repo, sha, rest, visited, target, remaining) do
visited = Map.put(visited, sha, true)
case ObjectResolver.read(repo, sha) do
{:ok, %Commit{parents: parents}} ->
if target in parents do
{:ok, true}
else
do_check_ancestor(repo, parents ++ rest, visited, target, remaining - 1)
end
{:error, :not_found} ->
with {:ok, desc_anc} <- collect_ancestors(repo, descendant_sha, max_walk) do
{:ok, MapSet.member?(desc_anc, ancestor_sha)}
end
{:error, {:missing_commit, sha}}
{:error, _} = err ->
err
end
end
@@ -146,5 +183,73 @@
case do_collect(repo, [sha], %{}, max_walk) do
{:ok, visited_map} -> {:ok, MapSet.new(Map.keys(visited_map))}
{:error, _} = err -> err
end
end
# Early-terminating walk: stops following any parent branch once it hits a commit in `base_anc`.
# Returns `{:ok, diff_set, overlap_ancestors_of_head}`.
defp collect_ancestors_until(repo, head_sha, base_anc, max_walk) do
case do_collect_until(repo, [head_sha], %{}, %{}, base_anc, max_walk) do
{:ok, diff_map, overlap_map} ->
full_overlap = expand_overlap_ancestors(repo, Map.keys(overlap_map), max_walk)
{:ok, MapSet.new(Map.keys(diff_map)), full_overlap}
{:error, _} = err ->
err
end
end
defp expand_overlap_ancestors(repo, overlap_shas, max_walk) do
Enum.reduce(overlap_shas, MapSet.new(), fn sha, acc ->
case collect_ancestors(repo, sha, max_walk) do
{:ok, anc} -> MapSet.union(acc, anc)
_ -> acc
end
end)
end
defp do_collect_until(_repo, [], diff_acc, overlap_acc, _base_anc, _remaining) do
{:ok, diff_acc, overlap_acc}
end
defp do_collect_until(_repo, _queue, _diff_acc, _overlap_acc, _base_anc, 0) do
{:error, :walk_limit_exceeded}
end
defp do_collect_until(repo, [sha | rest], diff_acc, overlap_acc, base_anc, remaining) do
cond do
Map.has_key?(diff_acc, sha) or Map.has_key?(overlap_acc, sha) ->
do_collect_until(repo, rest, diff_acc, overlap_acc, base_anc, remaining)
MapSet.member?(base_anc, sha) ->
do_collect_until(
repo,
rest,
diff_acc,
Map.put(overlap_acc, sha, true),
base_anc,
remaining
)
true ->
diff_acc = Map.put(diff_acc, sha, true)
case ObjectResolver.read(repo, sha) do
{:ok, %Commit{parents: parents}} ->
do_collect_until(
repo,
parents ++ rest,
diff_acc,
overlap_acc,
base_anc,
remaining - 1
)
{:error, :not_found} ->
{:error, {:missing_commit, sha}}
{:error, _} = err ->
err
end
end
end