git: Add word diff highlighting (#43269)

This PR adds word/character diff for expanded diff hunks that have both
a deleted and added section, as well as a setting `word_diff_enabled` to
enable/disable word diffs per language.

- `word_diff_enabled`: Defaults to true. Whether or not expanded diff
hunks will show word diff highlights when they're able to.

### Preview
<img width="1502" height="430" alt="image"
src="https://github.com/user-attachments/assets/1a8d5b71-449e-44cd-bc87-d6b65bfca545"
/>

### Architecture

I had three architecture goals I wanted to have when adding word diff
support:

- Caching: We should only calculate word diffs once and save the result.
This is because calculating word diffs can be expensive, and Zed should
always be responsive.
- Don't block the main thread: Word diffs should be computed in the
background to prevent hanging Zed.
- Lazy calculation: We should calculate word diffs for buffers that are
not visible to a user.

To accomplish the three goals, word diffs are computed as a part of
`BufferDiff` diff hunk processing because it happens on a background
thread, is cached until the file is edited, and is only refreshed for
open buffers.

My original implementation calculated word diffs every frame in the
Editor element. This had the benefit of lazy evaluation because it only
calculated visible frames, but it didn't have caching for the
calculations, and the code wasn't organized. Because the hunk
calculations would happen in two separate places instead of just
`BufferDiff`. Finally, it always happened on the main thread because it
was during the `EditorElement` layout phase.

I used Zed's
[`diff_internal`](https://github.com/zed-industries/zed/blob/02b2aa6c50c03d3005bec2effbc9f87161fbb1e8/crates/language/src/text_diff.rs#L230-L267)
as a starting place for word diff calculations because it uses
`Imara_diff` behind the scenes and already has language-specific
support.

#### Future Improvements

In the future, we could add `AST` based word diff highlights, e.g.
https://github.com/zed-industries/zed/pull/43691.

Release Notes:

- git: Show word diff highlight in expanded diff hunks with less than 5
lines.
- git: Add `word_diff_enabled` as a language setting that defaults to
true.

---------

Co-authored-by: David Kleingeld <davidsk@zed.dev>
Co-authored-by: Cole Miller <cole@zed.dev>
Co-authored-by: cameron <cameron.studdstreet@gmail.com>
Co-authored-by: Lukas Wirth <lukas@zed.dev>
This commit is contained in:
Anthony Eid
2025-12-01 22:36:30 -05:00
committed by GitHub
co-authored by David Kleingeld Cole Miller cameron Lukas Wirth
parent 2df5993eb0
commit 464c0be2b7
22 changed files with 547 additions and 48 deletions
+26
View File
@@ -152,6 +152,8 @@ pub struct MultiBufferDiffHunk {
pub diff_base_byte_range: Range<BufferOffset>,
/// Whether or not this hunk also appears in the 'secondary diff'.
pub secondary_status: DiffHunkSecondaryStatus,
/// The word diffs for this hunk.
pub word_diffs: Vec<Range<MultiBufferOffset>>,
}
impl MultiBufferDiffHunk {
@@ -561,6 +563,7 @@ pub struct MultiBufferSnapshot {
}
#[derive(Debug, Clone)]
/// A piece of text in the multi-buffer
enum DiffTransform {
Unmodified {
summary: MBTextSummary,
@@ -961,6 +964,8 @@ struct MultiBufferCursor<'a, MBD, BD> {
cached_region: Option<MultiBufferRegion<'a, MBD, BD>>,
}
/// Matches transformations to an item
/// This is essentially a more detailed version of DiffTransform
#[derive(Clone)]
struct MultiBufferRegion<'a, MBD, BD> {
buffer: &'a BufferSnapshot,
@@ -3870,11 +3875,31 @@ impl MultiBufferSnapshot {
} else {
range.end.row + 1
};
let word_diffs = (!hunk.base_word_diffs.is_empty()
|| !hunk.buffer_word_diffs.is_empty())
.then(|| {
let hunk_start_offset =
Anchor::in_buffer(excerpt.id, hunk.buffer_range.start).to_offset(self);
hunk.base_word_diffs
.iter()
.map(|diff| hunk_start_offset + diff.start..hunk_start_offset + diff.end)
.chain(
hunk.buffer_word_diffs
.into_iter()
.map(|diff| Anchor::range_in_buffer(excerpt.id, diff).to_offset(self)),
)
.collect()
})
.unwrap_or_default();
Some(MultiBufferDiffHunk {
row_range: MultiBufferRow(range.start.row)..MultiBufferRow(end_row),
buffer_id: excerpt.buffer_id,
excerpt_id: excerpt.id,
buffer_range: hunk.buffer_range.clone(),
word_diffs,
diff_base_byte_range: BufferOffset(hunk.diff_base_byte_range.start)
..BufferOffset(hunk.diff_base_byte_range.end),
secondary_status: hunk.secondary_status,
@@ -6834,6 +6859,7 @@ where
TextDimension::add_assign(&mut buffer_end, &buffer_range_len);
let start = self.diff_transforms.start().output_dimension.0;
let end = self.diff_transforms.end().output_dimension.0;
Some(MultiBufferRegion {
buffer,
excerpt,
+112 -10
View File
@@ -351,7 +351,7 @@ fn test_excerpt_boundaries_and_clipping(cx: &mut App) {
}
#[gpui::test]
fn test_diff_boundary_anchors(cx: &mut TestAppContext) {
async fn test_diff_boundary_anchors(cx: &mut TestAppContext) {
let base_text = "one\ntwo\nthree\n";
let text = "one\nthree\n";
let buffer = cx.new(|cx| Buffer::local(text, cx));
@@ -393,7 +393,7 @@ fn test_diff_boundary_anchors(cx: &mut TestAppContext) {
}
#[gpui::test]
fn test_diff_hunks_in_range(cx: &mut TestAppContext) {
async fn test_diff_hunks_in_range(cx: &mut TestAppContext) {
let base_text = "one\ntwo\nthree\nfour\nfive\nsix\nseven\neight\n";
let text = "one\nfour\nseven\n";
let buffer = cx.new(|cx| Buffer::local(text, cx));
@@ -473,7 +473,7 @@ fn test_diff_hunks_in_range(cx: &mut TestAppContext) {
}
#[gpui::test]
fn test_editing_text_in_diff_hunks(cx: &mut TestAppContext) {
async fn test_editing_text_in_diff_hunks(cx: &mut TestAppContext) {
let base_text = "one\ntwo\nfour\nfive\nsix\nseven\n";
let text = "one\ntwo\nTHREE\nfour\nfive\nseven\n";
let buffer = cx.new(|cx| Buffer::local(text, cx));
@@ -905,7 +905,7 @@ fn test_empty_multibuffer(cx: &mut App) {
}
#[gpui::test]
fn test_empty_diff_excerpt(cx: &mut TestAppContext) {
async fn test_empty_diff_excerpt(cx: &mut TestAppContext) {
let multibuffer = cx.new(|_| MultiBuffer::new(Capability::ReadWrite));
let buffer = cx.new(|cx| Buffer::local("", cx));
let base_text = "a\nb\nc";
@@ -1235,7 +1235,7 @@ fn test_resolving_anchors_after_replacing_their_excerpts(cx: &mut App) {
}
#[gpui::test]
fn test_basic_diff_hunks(cx: &mut TestAppContext) {
async fn test_basic_diff_hunks(cx: &mut TestAppContext) {
let text = indoc!(
"
ZERO
@@ -1480,7 +1480,7 @@ fn test_basic_diff_hunks(cx: &mut TestAppContext) {
}
#[gpui::test]
fn test_repeatedly_expand_a_diff_hunk(cx: &mut TestAppContext) {
async fn test_repeatedly_expand_a_diff_hunk(cx: &mut TestAppContext) {
let text = indoc!(
"
one
@@ -1994,7 +1994,7 @@ fn test_set_excerpts_for_buffer_rename(cx: &mut TestAppContext) {
}
#[gpui::test]
fn test_diff_hunks_with_multiple_excerpts(cx: &mut TestAppContext) {
async fn test_diff_hunks_with_multiple_excerpts(cx: &mut TestAppContext) {
let base_text_1 = indoc!(
"
one
@@ -3236,6 +3236,7 @@ fn check_multibuffer_edits(
fn test_history(cx: &mut App) {
let test_settings = SettingsStore::test(cx);
cx.set_global(test_settings);
let group_interval: Duration = Duration::from_millis(1);
let buffer_1 = cx.new(|cx| {
let mut buf = Buffer::local("1234", cx);
@@ -3476,7 +3477,7 @@ async fn test_enclosing_indent(cx: &mut TestAppContext) {
}
#[gpui::test]
fn test_summaries_for_anchors(cx: &mut TestAppContext) {
async fn test_summaries_for_anchors(cx: &mut TestAppContext) {
let base_text_1 = indoc!(
"
bar
@@ -3553,7 +3554,7 @@ fn test_summaries_for_anchors(cx: &mut TestAppContext) {
}
#[gpui::test]
fn test_trailing_deletion_without_newline(cx: &mut TestAppContext) {
async fn test_trailing_deletion_without_newline(cx: &mut TestAppContext) {
let base_text_1 = "one\ntwo".to_owned();
let text_1 = "one\n".to_owned();
@@ -4278,8 +4279,10 @@ fn test_random_chunk_bitmaps(cx: &mut App, mut rng: StdRng) {
}
}
#[gpui::test(iterations = 100)]
#[gpui::test(iterations = 10)]
fn test_random_chunk_bitmaps_with_diffs(cx: &mut App, mut rng: StdRng) {
let settings_store = SettingsStore::test(cx);
cx.set_global(settings_store);
use buffer_diff::BufferDiff;
use util::RandomCharIter;
@@ -4435,6 +4438,105 @@ fn test_random_chunk_bitmaps_with_diffs(cx: &mut App, mut rng: StdRng) {
}
}
fn collect_word_diffs(
base_text: &str,
modified_text: &str,
cx: &mut TestAppContext,
) -> Vec<String> {
let buffer = cx.new(|cx| Buffer::local(modified_text, cx));
let diff = cx.new(|cx| BufferDiff::new_with_base_text(base_text, &buffer, cx));
cx.run_until_parked();
let multibuffer = cx.new(|cx| {
let mut multibuffer = MultiBuffer::singleton(buffer.clone(), cx);
multibuffer.add_diff(diff.clone(), cx);
multibuffer
});
multibuffer.update(cx, |multibuffer, cx| {
multibuffer.expand_diff_hunks(vec![Anchor::min()..Anchor::max()], cx);
});
let snapshot = multibuffer.read_with(cx, |multibuffer, cx| multibuffer.snapshot(cx));
let text = snapshot.text();
snapshot
.diff_hunks()
.flat_map(|hunk| hunk.word_diffs)
.map(|range| text[range.start.0..range.end.0].to_string())
.collect()
}
#[gpui::test]
async fn test_word_diff_simple_replacement(cx: &mut TestAppContext) {
let settings_store = cx.update(|cx| SettingsStore::test(cx));
cx.set_global(settings_store);
let base_text = "hello world foo bar\n";
let modified_text = "hello WORLD foo BAR\n";
let word_diffs = collect_word_diffs(base_text, modified_text, cx);
assert_eq!(word_diffs, vec!["world", "bar", "WORLD", "BAR"]);
}
#[gpui::test]
async fn test_word_diff_consecutive_modified_lines(cx: &mut TestAppContext) {
let settings_store = cx.update(|cx| SettingsStore::test(cx));
cx.set_global(settings_store);
let base_text = "aaa bbb\nccc ddd\n";
let modified_text = "aaa BBB\nccc DDD\n";
let word_diffs = collect_word_diffs(base_text, modified_text, cx);
assert_eq!(
word_diffs,
vec!["bbb", "ddd", "BBB", "DDD"],
"consecutive modified lines should produce word diffs when line counts match"
);
}
#[gpui::test]
async fn test_word_diff_modified_lines_with_deletion_between(cx: &mut TestAppContext) {
let settings_store = cx.update(|cx| SettingsStore::test(cx));
cx.set_global(settings_store);
let base_text = "aaa bbb\ndeleted line\nccc ddd\n";
let modified_text = "aaa BBB\nccc DDD\n";
let word_diffs = collect_word_diffs(base_text, modified_text, cx);
assert_eq!(
word_diffs,
Vec::<String>::new(),
"modified lines with a deleted line between should not produce word diffs"
);
}
#[gpui::test]
async fn test_word_diff_disabled(cx: &mut TestAppContext) {
let settings_store = cx.update(|cx| {
let mut settings_store = SettingsStore::test(cx);
settings_store.update_user_settings(cx, |settings| {
settings.project.all_languages.defaults.word_diff_enabled = Some(false);
});
settings_store
});
cx.set_global(settings_store);
let base_text = "hello world\n";
let modified_text = "hello WORLD\n";
let word_diffs = collect_word_diffs(base_text, modified_text, cx);
assert_eq!(
word_diffs,
Vec::<String>::new(),
"word diffs should be empty when disabled"
);
}
/// Tests `excerpt_containing` and `excerpts_for_range` (functions mapping multi-buffer text-coordinates to excerpts)
#[gpui::test]
fn test_excerpts_containment_functions(cx: &mut App) {