diff --git a/Cargo.lock b/Cargo.lock index 50028977..7b8a62d9 100644 --- a/Cargo.lock +++ b/Cargo.lock @@ -1800,6 +1800,7 @@ dependencies = [ "mergify-core", "mergify-events", "mergify-freeze", + "mergify-json-merge", "mergify-queue", "mergify-stack", "mergify-tui", @@ -1885,6 +1886,15 @@ dependencies = [ "wiremock", ] +[[package]] +name = "mergify-json-merge" +version = "0.0.0" +dependencies = [ + "mergify-core", + "tempfile", + "tracing", +] + [[package]] name = "mergify-queue" version = "0.0.0" diff --git a/README.md b/README.md index c88b9c97..74a4422b 100644 --- a/README.md +++ b/README.md @@ -172,6 +172,14 @@ Every command group maps to a section of the maintenance. [Docs](https://docs.mergify.com/merge-protections/freeze/) - **`mergify config`** — Validate your configuration and simulate actions before you merge. [Docs](https://docs.mergify.com/configuration/file-format/#validating-with-the-cli) +- **`mergify merge-driver`** — Git merge drivers that merge by file + format instead of by line, falling back to git's line merge when they + cannot. `json` merges JSON structurally, keeping ours' formatting: + + ```shell + git config merge.json.driver "mergify merge-driver json --marker-size %L --path %P %A %O %B" + echo '*.json merge=json' >> .gitattributes + ``` - **`mergify self-update`** — Update the CLI to the latest release. - **`mergify completions `** — Print a shell completion script ([see below](#shell-completions)). diff --git a/crates/mergify-cli/Cargo.toml b/crates/mergify-cli/Cargo.toml index 3bbdf97d..76a54cf8 100644 --- a/crates/mergify-cli/Cargo.toml +++ b/crates/mergify-cli/Cargo.toml @@ -24,6 +24,7 @@ mergify-config = { path = "../mergify-config" } mergify-core = { path = "../mergify-core" } mergify-events = { path = "../mergify-events" } mergify-freeze = { path = "../mergify-freeze" } +mergify-json-merge = { path = "../mergify-json-merge" } mergify-queue = { path = "../mergify-queue" } mergify-stack = { path = "../mergify-stack" } # Shared TTY/color palette — the binary resolves `--color` once at diff --git a/crates/mergify-cli/src/main.rs b/crates/mergify-cli/src/main.rs index c8969d7a..e7846700 100644 --- a/crates/mergify-cli/src/main.rs +++ b/crates/mergify-cli/src/main.rs @@ -124,6 +124,7 @@ const NATIVE_COMMANDS: &[(&str, &str)] = &[ ("freeze", "create"), ("freeze", "update"), ("freeze", "delete"), + ("merge-driver", "json"), ("stack", "checkout"), ("stack", "drop"), ("stack", "edit"), @@ -294,6 +295,9 @@ enum NativeCommand { /// `mergify self-update [--force] [--check]` — replace the /// running binary with the latest release. SelfUpdate(self_update::Options), + /// `mergify merge-driver json ` — run by git + /// as a merge driver. + MergeDriverJson(MergeDriverJsonCli), } struct StackEditOpts { @@ -720,7 +724,8 @@ fn init_tracing(verbose: u8, debug: bool, color: mergify_tui::ColorChoice) { let directives = format!( "warn,mergify_cli={level},mergify_core={level},mergify_stack={level},\ mergify_ci={level},mergify_queue={level},mergify_freeze={level},\ - mergify_config={level},mergify_tui={level},mergify_auth={level}" + mergify_config={level},mergify_tui={level},mergify_auth={level},\ + mergify_json_merge={level}" ); let filter = EnvFilter::try_from_default_env().unwrap_or_else(|_| EnvFilter::new(directives)); let _ = tracing_subscriber::fmt() @@ -814,6 +819,9 @@ fn dispatch_from_parsed(parsed: CliRoot) -> Dispatch { StackSubcommand::Setup(cli) => Dispatch::Native(NativeCommand::StackSetup(cli.into())), }, Subcommands::SelfUpdate(cli) => Dispatch::Native(NativeCommand::SelfUpdate(cli.into())), + Subcommands::MergeDriver(MergeDriverArgs { + command: MergeDriverSubcommand::Json(cli), + }) => Dispatch::Native(NativeCommand::MergeDriverJson(cli)), Subcommands::Completions(cli) => Dispatch::Native(NativeCommand::Completions(cli.shell)), Subcommands::Internal(InternalArgs { command: @@ -2564,6 +2572,16 @@ fn run_native(cmd: NativeCommand) -> ExitCode { self_update::run(&opts).await?; Ok(mergify_core::ExitCode::Success) } + NativeCommand::MergeDriverJson(cli) => { + mergify_json_merge::run(&mergify_json_merge::DriverOptions { + ours: &cli.ours, + base: &cli.base, + theirs: &cli.theirs, + marker_size: cli.marker_size, + path: cli.path.as_deref(), + })?; + Ok(mergify_core::ExitCode::Success) + } NativeCommand::InternalRebaseTodoRewrite(opts) => { let action = match opts.action { InternalRebaseAction::Edit => { @@ -2911,6 +2929,16 @@ enum Subcommands { /// the merge queue from merging — for release windows, incidents, /// or code freezes. Freeze(FreezeArgs), + /// Git merge drivers that resolve conflicts by file format. + /// + /// Commands git runs in place of its line merge for the paths a + /// `merge` attribute assigns them, configured with + /// `git config merge..driver "mergify merge-driver + /// %A %O %B"`. Each one merges by structure where it can and falls + /// back to git's own line merge, conflict markers included, where + /// it cannot. + #[command(name = "merge-driver")] + MergeDriver(MergeDriverArgs), /// Create and maintain stacked pull requests. /// /// Manage a stack of dependent branches and their pull requests: @@ -4592,6 +4620,51 @@ enum AuthSubcommand { Status, } +#[derive(clap::Args)] +struct MergeDriverArgs { + #[command(subcommand)] + command: MergeDriverSubcommand, +} + +#[derive(Subcommand)] +enum MergeDriverSubcommand { + /// Merge a JSON file by structure instead of by line. + /// + /// Two edits to different keys, or to different elements of an + /// array, merge even when they sit on neighbouring lines. The + /// result keeps ours' formatting and key order, with theirs' + /// changes spliced in as theirs wrote them. It declines — and falls + /// back to `git merge-file`, leaving conflict markers — when both + /// sides changed the same value differently, when one side changed + /// a key the other deleted, and when both inserted into the same + /// place in an array. Configure it with + /// `git config merge.json.driver "mergify merge-driver json + /// --marker-size %L --path %P %A %O %B"` and `*.json merge=json` + /// in `.gitattributes`. + Json(MergeDriverJsonCli), +} + +#[derive(clap::Args)] +struct MergeDriverJsonCli { + /// Conflict-marker length for the line-merge fallback (git's `%L`). + #[arg(long, value_name = "N")] + marker_size: Option, + + /// Path being merged (git's `%P`), named in messages. + #[arg(long, value_name = "PATH")] + path: Option, + + /// Our version (git's `%A`), overwritten with the result. + ours: PathBuf, + + /// The merge base (git's `%O`); empty when both sides added the + /// file. + base: PathBuf, + + /// Their version (git's `%B`). + theirs: PathBuf, +} + #[derive(clap::Args)] struct FreezeArgs { /// Mergify token. Falls back to ``MERGIFY_TOKEN``, then the @@ -4769,6 +4842,7 @@ mod tests { "queue", "events", "freeze", + "merge-driver", "stack", "self-update", "completions" diff --git a/crates/mergify-cli/src/snapshots/mergify__tests__cli_schema_golden.snap b/crates/mergify-cli/src/snapshots/mergify__tests__cli_schema_golden.snap index 732ffe9e..e8b6a319 100644 --- a/crates/mergify-cli/src/snapshots/mergify__tests__cli_schema_golden.snap +++ b/crates/mergify-cli/src/snapshots/mergify__tests__cli_schema_golden.snap @@ -2447,6 +2447,129 @@ expression: schema "subcommandRequired": true, "usage": "mergify freeze [OPTIONS] " }, + { + "about": "Git merge drivers that resolve conflicts by file format", + "aliases": [], + "args": [], + "commands": [ + { + "about": "Merge a JSON file by structure instead of by line", + "aliases": [], + "args": [ + { + "default": null, + "env": null, + "global": false, + "help": "Conflict-marker length for the line-merge fallback (git's `%L`)", + "id": "marker_size", + "kind": "option", + "long": "marker-size", + "longHelp": "Conflict-marker length for the line-merge fallback (git's `%L`)", + "numArgs": "1", + "possibleValues": [], + "required": false, + "short": null, + "valueHint": null, + "valueNames": [ + "N" + ] + }, + { + "default": null, + "env": null, + "global": false, + "help": "Path being merged (git's `%P`), named in messages", + "id": "path", + "kind": "option", + "long": "path", + "longHelp": "Path being merged (git's `%P`), named in messages", + "numArgs": "1", + "possibleValues": [], + "required": false, + "short": null, + "valueHint": null, + "valueNames": [ + "PATH" + ] + }, + { + "default": null, + "env": null, + "global": false, + "help": "Our version (git's `%A`), overwritten with the result", + "id": "ours", + "kind": "positional", + "long": null, + "longHelp": "Our version (git's `%A`), overwritten with the result", + "numArgs": "1", + "possibleValues": [], + "required": true, + "short": null, + "valueHint": "anyPath", + "valueNames": [ + "OURS" + ] + }, + { + "default": null, + "env": null, + "global": false, + "help": "The merge base (git's `%O`); empty when both sides added the file", + "id": "base", + "kind": "positional", + "long": null, + "longHelp": "The merge base (git's `%O`); empty when both sides added the file", + "numArgs": "1", + "possibleValues": [], + "required": true, + "short": null, + "valueHint": "anyPath", + "valueNames": [ + "BASE" + ] + }, + { + "default": null, + "env": null, + "global": false, + "help": "Their version (git's `%B`)", + "id": "theirs", + "kind": "positional", + "long": null, + "longHelp": "Their version (git's `%B`)", + "numArgs": "1", + "possibleValues": [], + "required": true, + "short": null, + "valueHint": "anyPath", + "valueNames": [ + "THEIRS" + ] + } + ], + "commands": [], + "longAbout": "Merge a JSON file by structure instead of by line.\n\nTwo edits to different keys, or to different elements of an array, merge even when they sit on neighbouring lines. The result keeps ours' formatting and key order, with theirs' changes spliced in as theirs wrote them. It declines — and falls back to `git merge-file`, leaving conflict markers — when both sides changed the same value differently, when one side changed a key the other deleted, and when both inserted into the same place in an array. Configure it with `git config merge.json.driver \"mergify merge-driver json --marker-size %L --path %P %A %O %B\"` and `*.json merge=json` in `.gitattributes`.", + "name": "json", + "path": [ + "mergify", + "merge-driver", + "json" + ], + "source": "native", + "subcommandRequired": false, + "usage": "mergify merge-driver json [OPTIONS] " + } + ], + "longAbout": "Git merge drivers that resolve conflicts by file format.\n\nCommands git runs in place of its line merge for the paths a `merge` attribute assigns them, configured with `git config merge..driver \"mergify merge-driver %A %O %B\"`. Each one merges by structure where it can and falls back to git's own line merge, conflict markers included, where it cannot.", + "name": "merge-driver", + "path": [ + "mergify", + "merge-driver" + ], + "source": "native", + "subcommandRequired": true, + "usage": "mergify merge-driver [OPTIONS] " + }, { "about": "Create and maintain stacked pull requests", "aliases": [], diff --git a/crates/mergify-cli/tests/merge_driver_json.rs b/crates/mergify-cli/tests/merge_driver_json.rs new file mode 100644 index 00000000..4a987d42 --- /dev/null +++ b/crates/mergify-cli/tests/merge_driver_json.rs @@ -0,0 +1,142 @@ +//! End-to-end tests for `mergify merge-driver json`: the freshly built +//! binary, run by a real git as a configured merge driver — once from a +//! working tree (`git merge`, a laptop) and once from a bare repository +//! (`git merge-tree` with `info/attributes`, how Mergify's merge queue +//! composes a batch). + +use std::path::Path; +use std::process::Command; +use std::process::Output; + +const BASE: &str = r#"{ + "name": "dashboard", + "devDependencies": { + "@types/lodash": "4.17.24", + "@types/luxon": "3.7.1", + "typescript": "5.9.2" + } +} +"#; + +fn git(dir: &Path, args: &[&str]) -> Output { + let driver = format!( + "'{}' merge-driver json --marker-size %L --path %P %A %O %B", + env!("CARGO_BIN_EXE_mergify") + ); + Command::new("git") + .env("GIT_CONFIG_GLOBAL", "/dev/null") + .env("GIT_CONFIG_NOSYSTEM", "1") + .arg("-C") + .arg(dir) + .args(["-c", "user.email=t@e.com", "-c", "user.name=T"]) + .args(["-c", &format!("merge.json.driver={driver}")]) + .args(args) + .output() + .unwrap_or_else(|e| panic!("spawn git {args:?}: {e}")) +} + +#[track_caller] +fn ok(dir: &Path, args: &[&str]) -> String { + let out = git(dir, args); + assert!( + out.status.success(), + "git {args:?}: {}{}", + String::from_utf8_lossy(&out.stdout), + String::from_utf8_lossy(&out.stderr) + ); + String::from_utf8_lossy(&out.stdout).trim().to_owned() +} + +/// A repository whose `main` holds `BASE`, with `ours` and `theirs` +/// branches each committing their own version of it. +fn repo(ours: &str, theirs: &str) -> tempfile::TempDir { + let dir = tempfile::tempdir().unwrap(); + let d = dir.path(); + ok(d, &["init", "-q", "-b", "main"]); + std::fs::write(d.join(".gitattributes"), "*.json merge=json\n").unwrap(); + std::fs::write(d.join("package.json"), BASE).unwrap(); + ok(d, &["add", "."]); + ok(d, &["commit", "-q", "-m", "base"]); + for (branch, content) in [("ours", ours), ("theirs", theirs)] { + ok(d, &["checkout", "-q", "-b", branch, "main"]); + std::fs::write(d.join("package.json"), content).unwrap(); + ok(d, &["commit", "-q", "-am", branch]); + } + ok(d, &["checkout", "-q", "ours"]); + dir +} + +#[test] +fn neighbouring_bumps_merge_in_a_working_tree() { + let dir = repo( + &BASE.replace("4.17.24", "4.17.25"), + &BASE.replace("3.7.1", "3.7.4"), + ); + ok(dir.path(), &["merge", "-q", "--no-edit", "theirs"]); + assert_eq!( + std::fs::read_to_string(dir.path().join("package.json")).unwrap(), + BASE.replace("4.17.24", "4.17.25").replace("3.7.1", "3.7.4") + ); +} + +#[test] +fn the_same_dependency_bumped_twice_conflicts_with_markers() { + let dir = repo( + &BASE.replace("3.7.1", "3.7.4"), + &BASE.replace("3.7.1", "3.8.0"), + ); + let out = git(dir.path(), &["merge", "--no-edit", "theirs"]); + assert!(!out.status.success()); + let stderr = String::from_utf8_lossy(&out.stderr); + assert!( + stderr.contains( + "mergify: package.json: both sides changed this value differently \ + (at `/devDependencies/@types~1luxon`); left conflict markers" + ), + "{stderr}" + ); + let content = std::fs::read_to_string(dir.path().join("package.json")).unwrap(); + assert!( + content.contains("<<<<<<< ours\n \"@types/luxon\": \"3.7.4\",\n=======\n"), + "{content}" + ); +} + +#[test] +fn a_bare_merge_tree_resolves_through_info_attributes() { + // The merge queue's shape: no working tree, so the attributes come + // from `$GIT_DIR/info/attributes` and the merge from `merge-tree`. + let work = repo( + &BASE.replace("4.17.24", "4.17.25"), + &BASE.replace("3.7.1", "3.7.4"), + ); + let bare = tempfile::tempdir().unwrap(); + let src = work.path().to_str().unwrap(); + ok(bare.path(), &["clone", "-q", "--bare", src, "."]); + std::fs::write( + bare.path().join("info/attributes"), + "package.json merge=json\n", + ) + .unwrap(); + let tree = ok( + bare.path(), + &["merge-tree", "--write-tree", "ours", "theirs"], + ); + let merged = ok( + bare.path(), + &["cat-file", "-p", &format!("{tree}:package.json")], + ); + assert_eq!( + format!("{merged}\n"), + BASE.replace("4.17.24", "4.17.25").replace("3.7.1", "3.7.4") + ); + + // Without the attribute, the same merge is the conflict the driver + // exists to remove. + std::fs::write(bare.path().join("info/attributes"), "").unwrap(); + let out = git( + bare.path(), + &["merge-tree", "--write-tree", "ours", "theirs"], + ); + assert_eq!(out.status.code(), Some(1)); +} diff --git a/crates/mergify-json-merge/Cargo.toml b/crates/mergify-json-merge/Cargo.toml new file mode 100644 index 00000000..118a9a83 --- /dev/null +++ b/crates/mergify-json-merge/Cargo.toml @@ -0,0 +1,20 @@ +[package] +name = "mergify-json-merge" +version = "0.0.0" +edition.workspace = true +rust-version.workspace = true +license.workspace = true +repository.workspace = true +authors.workspace = true +description = "Structural three-way merge of JSON files, run by `mergify merge-driver json` as a git merge driver." +publish = false + +[dependencies] +mergify-core = { path = "../mergify-core" } +tracing = { workspace = true } + +[dev-dependencies] +tempfile = { workspace = true } + +[lints] +workspace = true diff --git a/crates/mergify-json-merge/src/driver.rs b/crates/mergify-json-merge/src/driver.rs new file mode 100644 index 00000000..3d692dbb --- /dev/null +++ b/crates/mergify-json-merge/src/driver.rs @@ -0,0 +1,138 @@ +//! The git side: `mergify merge-driver json %A %O %B`. +//! +//! git hands the driver three temporary files — ours (`%A`), the merge +//! base (`%O`) and theirs (`%B`) — and reads the result back from `%A`. +//! Exit 0 means merged; anything from 1 to 127 means conflicted, with +//! `%A` as the conflicted content. (128 and above is a dead driver and +//! aborts the whole merge, which is why nothing here may crash.) +//! +//! When the structural merge declines, the driver does what git would +//! have done without it: a line merge, `git merge-file`, which writes +//! conflict markers into `%A` for a human to resolve. So enabling the +//! driver never turns a merge git completes cleanly into a conflict, +//! with one exception that is the point: a clean line merge of three +//! valid JSON files whose result is NOT valid JSON (two edits around +//! one trailing comma) is reported as a conflict instead of landing. + +use std::borrow::Cow; +use std::fs; +use std::path::Path; +use std::process::Command; + +use mergify_core::CliError; + +use crate::merge::Decline; +use crate::merge::Reason; +use crate::merge::Side; +use crate::merge::merge; +use crate::value; + +pub struct DriverOptions<'a> { + /// `%A`: ours, overwritten with the result. + pub ours: &'a Path, + /// `%O`: the merge base; empty when both sides added the file. + pub base: &'a Path, + /// `%B`: theirs. + pub theirs: &'a Path, + /// `%L`: the conflict-marker length git would use for this path. + pub marker_size: Option, + /// `%P`: the path being merged, for messages only. + pub path: Option<&'a str>, +} + +/// Merge, writing the result over `opts.ours`. `Ok` is a clean merge; +/// [`CliError::Conflict`] is one left for a human, markers and all. +pub fn run(opts: &DriverOptions<'_>) -> Result<(), CliError> { + let read = |path: &Path, side: Side| { + fs::read(path).map_err(|e| CliError::wrap(format!("read {side} ({})", path.display()), e)) + }; + let ours = read(opts.ours, Side::Ours)?; + let base = read(opts.base, Side::Base)?; + let theirs = read(opts.theirs, Side::Theirs)?; + + // git's own trivial cases, settled on bytes before parsing anything. + if ours == theirs || base == theirs { + return Ok(()); + } + if base == ours { + return write(opts.ours, &theirs); + } + + let decline = match utf8(&base, &ours, &theirs) { + Ok((b, o, t)) => match merge(b, o, t) { + Ok(Cow::Borrowed(_)) => return Ok(()), + Ok(Cow::Owned(text)) => return write(opts.ours, text.as_bytes()), + Err(decline) => decline, + }, + Err(decline) => decline, + }; + line_merge(opts, &decline) +} + +fn utf8<'a>( + base: &'a [u8], + ours: &'a [u8], + theirs: &'a [u8], +) -> Result<(&'a str, &'a str, &'a str), Decline> { + let check = |bytes: &'a [u8], side| { + std::str::from_utf8(bytes).map_err(|e| Decline { + pointer: String::new(), + reason: Reason::NotJson { + side, + error: format!("not UTF-8 ({e})"), + }, + }) + }; + Ok(( + check(base, Side::Base)?, + check(ours, Side::Ours)?, + check(theirs, Side::Theirs)?, + )) +} + +fn write(path: &Path, bytes: &[u8]) -> Result<(), CliError> { + fs::write(path, bytes) + .map_err(|e| CliError::wrap(format!("write the merge result ({})", path.display()), e)) +} + +fn line_merge(opts: &DriverOptions<'_>, decline: &Decline) -> Result<(), CliError> { + let at = opts.path.map(|p| format!("{p}: ")).unwrap_or_default(); + let mut git = Command::new("git"); + git.arg("merge-file"); + if let Some(size) = opts.marker_size { + git.arg(format!("--marker-size={size}")); + } + git.args(["-L", "ours", "-L", "base", "-L", "theirs"]) + .arg(opts.ours) + .arg(opts.base) + .arg(opts.theirs); + let status = git + .status() + .map_err(|e| CliError::wrap(format!("{at}run git merge-file"), e))?; + match status.code() { + Some(0) => { + // All three inputs parsed if the structural merge got far + // enough to decline on something else, so the line merge's + // result has to parse too. + if !matches!(decline.reason, Reason::NotJson { .. }) { + let merged = fs::read(opts.ours) + .map_err(|e| CliError::wrap(format!("{at}read the line merge"), e))?; + let valid = + std::str::from_utf8(&merged).is_ok_and(|text| value::parse(text).is_ok()); + if !valid { + return Err(CliError::Conflict(format!( + "{at}{decline}; the line merge git falls back to produced invalid JSON" + ))); + } + } + tracing::info!("{at}{decline}; merged line by line instead"); + Ok(()) + } + Some(1..=127) => Err(CliError::Conflict(format!( + "{at}{decline}; left conflict markers" + ))), + _ => Err(CliError::Generic(format!( + "{at}git merge-file failed ({status})" + ))), + } +} diff --git a/crates/mergify-json-merge/src/lib.rs b/crates/mergify-json-merge/src/lib.rs new file mode 100644 index 00000000..b86118d3 --- /dev/null +++ b/crates/mergify-json-merge/src/lib.rs @@ -0,0 +1,65 @@ +//! Structural three-way merge of JSON files, as a git merge driver. +//! +//! git's line merge reports a conflict whenever two edits touch the same +//! or adjacent lines. In a JSON file that is mostly noise: two pull +//! requests bumping neighbouring dependencies in a `package.json`, or +//! adding different endpoints to a generated schema, conflict for no +//! reason a person would recognise. This crate merges by structure +//! instead, and is run by git as `mergify merge-driver json %A %O %B`. +//! +//! The bar is **correct whenever it claims success, declining otherwise**. +//! A decline falls back to git's own line merge (see [`driver`]), so it +//! costs exactly what having no driver costs; a wrong merge would ship a +//! file nobody reviewed. +//! +//! # What it merges, and what it declines +//! +//! - **Objects** merge key by key. A key one side changed takes that +//! side's value; a key both changed is merged recursively, and two +//! different scalars (the same dependency bumped to two versions) +//! decline. A key one side deleted is deleted, unless the other side +//! changed it, which declines. An object both sides added is merged +//! the same way with every key new. +//! - **Key order** is ours'. A key only theirs added goes in its sorted +//! place when both sides keep the object sorted by key, and otherwise +//! right after the key that precedes it in theirs. A reorder theirs +//! made to keys ours also has is not carried over: the values all are, +//! the order is ours'. +//! - **Arrays** merge element by element (diff3 over elements). A +//! stretch only one side changed takes that side's version. A stretch +//! both changed merges only when both edited the same elements in +//! place — same count, nothing moved — and each element is then merged +//! recursively; this is what lets two edits to two fields of the same +//! `OpenAPI` parameter merge. Everything else declines, **including two +//! insertions at the same point**: for a JSON-Schema `required` list +//! either order is right, for an ordered pipeline neither might be, +//! and nothing in the file says which kind of array it is. The same +//! insertion on both sides is taken once. +//! +//! # Formatting +//! +//! The result is ours' text, edited, never a re-serialisation (see +//! [`merge`](crate::merge::merge)): unchanged regions are copied byte +//! for byte and theirs' changes are spliced in as theirs wrote them, +//! shifted to ours' indentation. An item theirs adds takes the separator +//! ours uses in that container. For a file both sides wrote with the same +//! formatter, the result is that formatter's output. +//! +//! What it cannot promise is that the result equals what a *generator* +//! would produce from the merged sources. A schema generated in +//! declaration order puts two new fields of one model where the merged +//! source declares them, which no merge of the outputs can know. Where CI +//! regenerates the file and compares, such a difference turns the batch +//! red rather than landing silently: a safe failure, but a failure. + +mod driver; +mod merge; +mod sequence; +mod value; + +pub use driver::DriverOptions; +pub use driver::run; +pub use merge::Decline; +pub use merge::Reason; +pub use merge::Side; +pub use merge::merge; diff --git a/crates/mergify-json-merge/src/merge.rs b/crates/mergify-json-merge/src/merge.rs new file mode 100644 index 00000000..bb5bcd95 --- /dev/null +++ b/crates/mergify-json-merge/src/merge.rs @@ -0,0 +1,959 @@ +//! The structural three-way merge, and the text it produces. +//! +//! The output is OURS' TEXT, edited. Whatever ours already holds is +//! copied byte for byte, and each change theirs made is spliced in as +//! theirs wrote it. Nothing is re-serialised, so key order, indentation, +//! blank lines and number spellings survive everywhere the merge did not +//! have to touch. +//! +//! Before claiming success, the merged text is parsed again and compared +//! with the value the merge decided on. A splice that came out wrong (a +//! stray comma, a lost member) is caught there and becomes a decline, so +//! a mistake in the text assembly can cost a conflict but never a wrong +//! merge. + +use std::borrow::Cow; +use std::collections::HashMap; +use std::collections::HashSet; +use std::fmt; + +use crate::sequence; +use crate::sequence::Chunk; +use crate::value; +use crate::value::Kind; +use crate::value::Member; +use crate::value::Value; +use crate::value::same; + +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +pub enum Side { + Base, + Ours, + Theirs, +} + +impl fmt::Display for Side { + fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result { + f.write_str(match self { + Self::Base => "base", + Self::Ours => "ours", + Self::Theirs => "theirs", + }) + } +} + +/// Why the merge refused. Every variant is a case where more than one +/// result is defensible, or none is. +#[derive(Debug, Clone, PartialEq, Eq)] +pub enum Reason { + NotJson { + side: Side, + error: String, + }, + /// Both sides changed the same scalar, or changed its type, to + /// different values. + BothChanged, + /// Both sides added the same key (or the same file) with values + /// that are not both objects and differ. + BothAdded, + DeletedAndChanged { + deleted_by: Side, + }, + /// Both sides changed the same stretch of an array, and not by + /// editing the same elements in place. + ArrayEditsOverlap, + ArrayTooLarge, + /// The merged text did not parse back as the merged value. + RenderMismatch, +} + +impl fmt::Display for Reason { + fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result { + match self { + Self::NotJson { side, error } => write!(f, "{side} is not valid JSON: {error}"), + Self::BothChanged => f.write_str("both sides changed this value differently"), + Self::BothAdded => f.write_str("both sides added this value, differently"), + Self::DeletedAndChanged { deleted_by } => { + let other = if *deleted_by == Side::Ours { + Side::Theirs + } else { + Side::Ours + }; + write!(f, "{deleted_by} deleted this key and {other} changed it") + } + Self::ArrayEditsOverlap => { + f.write_str("both sides changed the same part of this array") + } + Self::ArrayTooLarge => f.write_str("this array changed too much to align"), + Self::RenderMismatch => { + f.write_str("the merged text did not read back as the merged value") + } + } + } +} + +#[derive(Debug, Clone, PartialEq, Eq)] +pub struct Decline { + /// JSON Pointer (RFC 6901) to the value the merge stopped at; empty + /// for the document root. + pub pointer: String, + pub reason: Reason, +} + +impl fmt::Display for Decline { + fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result { + match &self.reason { + Reason::NotJson { .. } | Reason::RenderMismatch => write!(f, "{}", self.reason), + reason if self.pointer.is_empty() => write!(f, "{reason} (at the document root)"), + reason => write!(f, "{reason} (at `{}`)", self.pointer), + } + } +} + +impl std::error::Error for Decline {} + +/// Merge `theirs` into `ours`, both descended from `base`. Returns the +/// merged document, borrowing `ours` when the merge leaves it as is. +/// +/// An empty (or blank) `base` means both sides added the file: git +/// hands an add/add to the driver that way. +pub fn merge<'a>(base: &str, ours: &'a str, theirs: &str) -> Result, Decline> { + let parse = |text, side| { + value::parse(text).map_err(|e| Decline { + pointer: String::new(), + reason: Reason::NotJson { + side, + error: e.to_string(), + }, + }) + }; + let o = parse(ours, Side::Ours)?; + let t = parse(theirs, Side::Theirs)?; + let b = if base.trim_matches([' ', '\t', '\n', '\r']).is_empty() { + None + } else { + Some(parse(base, Side::Base)?) + }; + let mut merger = Merger { + ours, + theirs, + path: Vec::new(), + }; + let out = merger.merge_value(b.as_ref(), &o, &t)?; + if matches!(out.expect, Expect::Ours(_)) { + return Ok(Cow::Borrowed(ours)); + } + let text = format!("{}{}{}", &ours[..o.start], out.text, &ours[o.end..]); + if !value::parse(&text).is_ok_and(|merged| matches(&merged, &out.expect)) { + return Err(Decline { + pointer: String::new(), + reason: Reason::RenderMismatch, + }); + } + Ok(Cow::Owned(text)) +} + +/// What a merged value must read back as, kept beside its text for the +/// final check. +enum Expect<'a> { + /// Ours' value, and its text is ours' span verbatim. + Ours(&'a Value), + Theirs(&'a Value), + Object(Vec<(&'a str, Expect<'a>)>), + Array(Vec>), +} + +fn matches(v: &Value, expect: &Expect<'_>) -> bool { + match (expect, &v.kind) { + (Expect::Ours(w) | Expect::Theirs(w), _) => same(v, w), + (Expect::Object(want), Kind::Object(got)) => { + got.len() == want.len() + && got + .iter() + .zip(want) + .all(|(m, (key, e))| m.key == *key && matches(&m.value, e)) + } + (Expect::Array(want), Kind::Array(got)) => { + got.len() == want.len() && got.iter().zip(want).all(|(v, e)| matches(v, e)) + } + _ => false, + } +} + +struct Out<'a> { + text: Cow<'a, str>, + expect: Expect<'a>, +} + +/// One item of a container being rebuilt. `ours_index` is its position +/// in ours' container when it came from there: two items that were +/// neighbours in ours keep the separator that stood between them. +struct Item<'a> { + ours_index: Option, + text: Cow<'a, str>, +} + +struct Merger<'a> { + ours: &'a str, + theirs: &'a str, + path: Vec, +} + +impl<'a> Merger<'a> { + fn decline(&self, reason: Reason) -> Decline { + let mut pointer = String::new(); + for segment in &self.path { + pointer.push('/'); + pointer.push_str(&segment.replace('~', "~0").replace('/', "~1")); + } + Decline { pointer, reason } + } + + fn keep(&self, o: &'a Value) -> Out<'a> { + Out { + text: Cow::Borrowed(&self.ours[o.start..o.end]), + expect: Expect::Ours(o), + } + } + + /// Theirs' value, written where ours' stands. + fn take_theirs(&self, o: &'a Value, t: &'a Value) -> Out<'a> { + Out { + text: reindent( + &self.theirs[t.start..t.end], + line_indent(self.theirs, t.start), + line_indent(self.ours, o.start), + ), + expect: Expect::Theirs(t), + } + } + + fn merge_value( + &mut self, + b: Option<&'a Value>, + o: &'a Value, + t: &'a Value, + ) -> Result, Decline> { + if same(o, t) || b.is_some_and(|b| same(b, t)) { + return Ok(self.keep(o)); + } + if b.is_some_and(|b| same(b, o)) { + return Ok(self.take_theirs(o, t)); + } + match (b.map(|b| &b.kind), &o.kind, &t.kind) { + (Some(Kind::Object(bm)), Kind::Object(om), Kind::Object(tm)) => { + self.merge_object(Some(bm), o, om, t, tm) + } + (None, Kind::Object(om), Kind::Object(tm)) => self.merge_object(None, o, om, t, tm), + (Some(Kind::Array(bi)), Kind::Array(oi), Kind::Array(ti)) => { + self.merge_array(bi, o, oi, t, ti) + } + (None, _, _) => Err(self.decline(Reason::BothAdded)), + _ => Err(self.decline(Reason::BothChanged)), + } + } + + /// Key by key. `base` is `None` when both sides added the object, so + /// every key is new to both. + fn merge_object( + &mut self, + base: Option<&'a [Member]>, + o: &'a Value, + om: &'a [Member], + t: &'a Value, + tm: &'a [Member], + ) -> Result, Decline> { + let base = base.unwrap_or_default(); + let index = |members: &'a [Member]| -> HashMap<&'a str, &'a Member> { + members.iter().map(|m| (m.key.as_str(), m)).collect() + }; + let in_base = index(base); + let in_theirs = index(tm); + let in_ours: HashSet<&str> = om.iter().map(|m| m.key.as_str()).collect(); + + for bm in base { + if !in_ours.contains(bm.key.as_str()) + && let Some(tv) = in_theirs.get(bm.key.as_str()) + && !same(&bm.value, &tv.value) + { + self.path.push(bm.key.clone()); + return Err(self.decline(Reason::DeletedAndChanged { + deleted_by: Side::Ours, + })); + } + } + + let Additions { + in_order, + at_front, + after, + } = additions(om, tm, &in_ours, &in_base); + let (from, to) = ( + line_indent(self.theirs, t.start), + line_indent(self.ours, o.start), + ); + let theirs_member = |tv: &'a Member| Item { + ours_index: None, + text: reindent(&self.theirs[tv.start..tv.value.end], from, to), + }; + let mut changed = !in_order.is_empty() || !at_front.is_empty() || !after.is_empty(); + let mut items = Vec::new(); + let mut expect = Vec::new(); + for tv in &at_front { + items.push(theirs_member(tv)); + expect.push((tv.key.as_str(), Expect::Theirs(&tv.value))); + } + for (j, ov) in om.iter().enumerate() { + let key = ov.key.as_str(); + self.path.push(ov.key.clone()); + let merged = match (in_base.get(key), in_theirs.get(key)) { + (Some(bv), Some(tv)) => { + Some(self.merge_value(Some(&bv.value), &ov.value, &tv.value)?) + } + (None, Some(tv)) => Some(self.merge_value(None, &ov.value, &tv.value)?), + (Some(bv), None) if same(&bv.value, &ov.value) => None, + (Some(_), None) => { + return Err(self.decline(Reason::DeletedAndChanged { + deleted_by: Side::Theirs, + })); + } + (None, None) => Some(self.keep(&ov.value)), + }; + self.path.pop(); + match merged { + None => changed = true, + Some(out) => { + let text = if let Expect::Ours(_) = out.expect { + Cow::Borrowed(&self.ours[ov.start..ov.value.end]) + } else { + changed = true; + Cow::Owned(format!( + "{}{}", + &self.ours[ov.start..ov.value.start], + out.text + )) + }; + items.push(Item { + ours_index: Some(j), + text, + }); + expect.push((key, out.expect)); + } + } + for tv in after.get(key).into_iter().flatten() { + items.push(theirs_member(tv)); + expect.push((tv.key.as_str(), Expect::Theirs(&tv.value))); + } + } + for tv in in_order { + let at = expect.partition_point(|(key, _)| *key < tv.key.as_str()); + items.insert(at, theirs_member(tv)); + expect.insert(at, (tv.key.as_str(), Expect::Theirs(&tv.value))); + } + if !changed { + return Ok(self.keep(o)); + } + Ok(Out { + text: Cow::Owned(self.render_container(o, t, &items)), + expect: Expect::Object(expect), + }) + } + + /// Element-wise diff3. Where only one side changed a stretch, that + /// side's version is taken. Where both did, the merge goes on only + /// if they changed the same elements in place (same count, nothing + /// moved), merging each element; anything else declines — two + /// insertions at the same point included, since which goes first is + /// a question only the array's meaning answers. + fn merge_array( + &mut self, + bi: &'a [Value], + o: &'a Value, + oi: &'a [Value], + t: &'a Value, + ti: &'a [Value], + ) -> Result, Decline> { + let chunks = sequence::diff3(bi, oi, ti) + .map_err(|sequence::TooLarge| self.decline(Reason::ArrayTooLarge))?; + let (from, to) = ( + line_indent(self.theirs, t.start), + line_indent(self.ours, o.start), + ); + let mut changed = false; + let mut items = Vec::new(); + let mut expect = Vec::new(); + let keep = |items: &mut Vec>, expect: &mut Vec>, j: usize| { + items.push(Item { + ours_index: Some(j), + text: Cow::Borrowed(&self.ours[oi[j].start..oi[j].end]), + }); + expect.push(Expect::Ours(&oi[j])); + }; + for chunk in chunks { + let (rb, ro, rt) = match chunk { + Chunk::Stable { o: j } => { + keep(&mut items, &mut expect, j); + continue; + } + Chunk::Unstable { b, o, t } => (b, o, t), + }; + let (bs, os, ts) = (&bi[rb], &oi[ro.clone()], &ti[rt]); + if seq_same(bs, ts) || seq_same(os, ts) { + ro.for_each(|j| keep(&mut items, &mut expect, j)); + } else if seq_same(bs, os) { + changed = true; + for tv in ts { + items.push(Item { + ours_index: None, + text: reindent(&self.theirs[tv.start..tv.end], from, to), + }); + expect.push(Expect::Theirs(tv)); + } + } else if bs.len() == os.len() + && bs.len() == ts.len() + && !moved(bs, os) + && !moved(bs, ts) + { + for (n, j) in ro.enumerate() { + self.path.push(j.to_string()); + let out = self.merge_value(Some(&bs[n]), &os[n], &ts[n])?; + self.path.pop(); + changed |= !matches!(out.expect, Expect::Ours(_)); + items.push(Item { + ours_index: Some(j), + text: out.text, + }); + expect.push(out.expect); + } + } else { + return Err(self.decline(Reason::ArrayEditsOverlap)); + } + } + if !changed { + return Ok(self.keep(o)); + } + Ok(Out { + text: Cow::Owned(self.render_container(o, t, &items)), + expect: Expect::Array(expect), + }) + } + + /// Rebuild a container from `items`, in ours' formatting: ours' + /// brackets and the whitespace inside them, and between two items + /// that were neighbours in ours, the separator that stood there. + /// Anywhere else the separator is one of ours' (or, when ours has + /// fewer than two items to take one from, theirs'). + fn render_container(&self, o: &Value, t: &Value, items: &[Item<'_>]) -> String { + let (ours, theirs) = (self.ours, self.theirs); + let os = o.item_spans(); + let ts = t.item_spans(); + let (from, to) = (line_indent(theirs, t.start), line_indent(ours, o.start)); + let (open, close) = (&ours[o.start..=o.start], &ours[o.end - 1..o.end]); + if items.is_empty() { + return format!("{open}{close}"); + } + let (leading, trailing) = match (os.first(), os.last(), ts.first(), ts.last()) { + (Some(f), Some(l), _, _) => ( + Cow::Borrowed(&ours[o.start + 1..f.0]), + Cow::Borrowed(&ours[l.1..o.end - 1]), + ), + (_, _, Some(f), Some(l)) => ( + reindent(&theirs[t.start + 1..f.0], from, to), + reindent(&theirs[l.1..t.end - 1], from, to), + ), + _ => (Cow::Borrowed(""), Cow::Borrowed("")), + }; + let separator = if os.len() >= 2 { + Cow::Borrowed(&ours[os[0].1..os[1].0]) + } else if ts.len() >= 2 { + reindent(&theirs[ts[0].1..ts[1].0], from, to) + } else if !leading.is_empty() { + Cow::Owned(format!(",{leading}")) + } else if ours.contains(": ") { + // A one-line container with nothing to copy a separator + // from: follow the document's own spacing. + Cow::Borrowed(", ") + } else { + Cow::Borrowed(",") + }; + let mut text = String::from(open); + text.push_str(&leading); + for (n, item) in items.iter().enumerate() { + if n > 0 { + match (items[n - 1].ours_index, item.ours_index) { + (Some(a), Some(b)) if b == a + 1 => text.push_str(&ours[os[a].1..os[b].0]), + _ => text.push_str(&separator), + } + } + text.push_str(&item.text); + } + text.push_str(&trailing); + text.push_str(close); + text + } +} + +/// Where the keys only theirs added go, by one of two rules. +struct Additions<'a> { + /// Sorted rule: each goes to its sorted place among the result's keys. + in_order: Vec<&'a Member>, + /// Anchor rule: before ours' first key ... + at_front: Vec<&'a Member>, + /// ... or right after the ours key named here. + after: HashMap<&'a str, Vec<&'a Member>>, +} + +/// A key only theirs added goes where theirs put it. When both sides +/// keep the object sorted — a package.json's dependencies, a generated +/// schema's components — that is its sorted place among ours' keys, even +/// beside a key ours added at the same spot. Otherwise it goes right +/// after the key that precedes it in theirs and that ours also has. +fn additions<'a>( + om: &[Member], + tm: &'a [Member], + in_ours: &HashSet<&str>, + in_base: &HashMap<&str, &Member>, +) -> Additions<'a> { + let sorted = is_sorted(om) && is_sorted(tm); + let mut found = Additions { + in_order: Vec::new(), + at_front: Vec::new(), + after: HashMap::new(), + }; + let mut anchor = None; + for tv in tm { + let key = tv.key.as_str(); + if in_ours.contains(key) { + anchor = Some(key); + } else if in_base.contains_key(key) { + // Ours deleted it and theirs left it as it was. + } else if sorted { + found.in_order.push(tv); + } else { + match anchor { + Some(a) => found.after.entry(a).or_default().push(tv), + None => found.at_front.push(tv), + } + } + } + found +} + +/// Keys in strictly ascending byte order. Vacuously true for fewer than +/// two keys, which costs nothing: with one key or none, theirs' order is +/// the only evidence and sorted insertion follows it. +fn is_sorted(members: &[Member]) -> bool { + members.windows(2).all(|pair| pair[0].key < pair[1].key) +} + +fn seq_same(a: &[Value], b: &[Value]) -> bool { + a.len() == b.len() && a.iter().zip(b).all(|(x, y)| same(x, y)) +} + +/// Whether `side` holds, at some position, an element base holds at a +/// different one — a move, which an in-place merge would misread as two +/// unrelated edits. +fn moved(base: &[Value], side: &[Value]) -> bool { + side.iter().enumerate().any(|(i, v)| { + !same(v, &base[i]) && base.iter().enumerate().any(|(j, w)| j != i && same(v, w)) + }) +} + +/// The run of spaces and tabs that starts the line holding `pos`. +fn line_indent(text: &str, pos: usize) -> &str { + let line = text[..pos].rfind('\n').map_or(0, |n| n + 1); + let rest = &text[line..]; + let width = rest.len() - rest.trim_start_matches([' ', '\t']).len(); + &rest[..width] +} + +/// Move a block of text taken from a line indented `from` to a line +/// indented `to`: every line after the first that starts with `from` +/// starts with `to` instead. JSON only allows newlines between tokens, +/// so this never touches the inside of a string. +fn reindent<'t>(text: &'t str, from: &str, to: &str) -> Cow<'t, str> { + if from == to || !text.contains('\n') { + return Cow::Borrowed(text); + } + let mut out = String::with_capacity(text.len()); + for (n, line) in text.split('\n').enumerate() { + if n > 0 { + out.push('\n'); + if let Some(rest) = line.strip_prefix(from) { + out.push_str(to); + out.push_str(rest); + continue; + } + } + out.push_str(line); + } + Cow::Owned(out) +} + +#[cfg(test)] +mod tests { + use super::*; + + #[track_caller] + fn merged(base: &str, ours: &str, theirs: &str) -> String { + match merge(base, ours, theirs) { + Ok(text) => text.into_owned(), + Err(e) => panic!("declined: {e}"), + } + } + + #[track_caller] + fn declined(base: &str, ours: &str, theirs: &str) -> Decline { + match merge(base, ours, theirs) { + Ok(text) => panic!("merged:\n{text}"), + Err(e) => e, + } + } + + const PACKAGE: &str = r#"{ + "name": "dashboard", + "devDependencies": { + "@types/lodash": "4.17.24", + "@types/luxon": "3.7.1", + "typescript": "5.9.2" + } +} +"#; + + #[test] + fn adjacent_bumps_both_land() { + // The census's shape: two renovate bumps on neighbouring lines, + // which git's line merge reports as a conflict. + let ours = PACKAGE.replace("4.17.24", "4.17.25"); + let theirs = PACKAGE.replace("3.7.1", "3.7.4"); + assert_eq!( + merged(PACKAGE, &ours, &theirs), + PACKAGE + .replace("4.17.24", "4.17.25") + .replace("3.7.1", "3.7.4") + ); + } + + #[test] + fn a_key_theirs_added_lands_where_theirs_put_it() { + let ours = PACKAGE.replace("4.17.24", "4.17.25"); + let theirs = PACKAGE.replace( + " \"typescript\"", + " \"@types/node\": \"24.0.0\",\n \"typescript\"", + ); + let want = PACKAGE.replace("4.17.24", "4.17.25").replace( + " \"typescript\"", + " \"@types/node\": \"24.0.0\",\n \"typescript\"", + ); + assert_eq!(merged(PACKAGE, &ours, &theirs), want); + } + + #[test] + fn keys_added_at_the_front_and_the_end_keep_theirs_order() { + let base = "{\n \"b\": 1\n}\n"; + let ours = "{\n \"b\": 2\n}\n"; + let theirs = "{\n \"a\": 0,\n \"a2\": 0,\n \"b\": 1,\n \"c\": 3,\n \"d\": 4\n}\n"; + assert_eq!( + merged(base, ours, theirs), + "{\n \"a\": 0,\n \"a2\": 0,\n \"b\": 2,\n \"c\": 3,\n \"d\": 4\n}\n" + ); + } + + #[test] + fn keys_both_sides_added_at_one_spot_stay_sorted() { + // Two new dependencies between the same two neighbours: the line + // merge conflicts, and placing theirs right after its theirs-side + // predecessor would put `b` after ours' `c`. + let base = "{\n \"a\": 1,\n \"d\": 4\n}\n"; + let ours = "{\n \"a\": 1,\n \"c\": 3,\n \"d\": 4\n}\n"; + let theirs = "{\n \"a\": 1,\n \"b\": 2,\n \"d\": 4\n}\n"; + let want = "{\n \"a\": 1,\n \"b\": 2,\n \"c\": 3,\n \"d\": 4\n}\n"; + assert_eq!(merged(base, ours, theirs), want); + assert_eq!(merged(base, theirs, ours), want); + } + + #[test] + fn in_an_unsorted_object_theirs_keys_follow_their_predecessor() { + let base = r#"{"z": 1, "a": 1, "q": 1}"#; + let ours = r#"{"z": 1, "a": 1, "y": 0, "q": 1}"#; + let theirs = r#"{"z": 1, "a": 1, "m": 2, "q": 1}"#; + assert_eq!( + merged(base, ours, theirs), + r#"{"z": 1, "a": 1, "m": 2, "y": 0, "q": 1}"# + ); + } + + #[test] + fn deletions_take_their_separator_with_them() { + let base = "{\n \"a\": 1,\n \"b\": 2,\n \"c\": 3\n}"; + let bump = "{\n \"a\": 1,\n \"b\": 2,\n \"c\": 4\n}"; + assert_eq!( + merged(base, bump, "{\n \"b\": 2,\n \"c\": 3\n}"), + "{\n \"b\": 2,\n \"c\": 4\n}" + ); + let bump = "{\n \"a\": 0,\n \"b\": 2,\n \"c\": 3\n}"; + assert_eq!( + merged(base, bump, "{\n \"a\": 1,\n \"b\": 2\n}"), + "{\n \"a\": 0,\n \"b\": 2\n}" + ); + let base = "{\"a\": {\"x\": 1}, \"b\": 1}"; + assert_eq!( + merged( + base, + "{\"a\": {}, \"b\": 1}", + "{\"a\": {\"x\": 1}, \"b\": 2}" + ), + "{\"a\": {}, \"b\": 2}" + ); + } + + #[test] + fn emptying_an_object_and_filling_an_empty_one() { + let base = "{\"a\": {\"x\": 1}, \"b\": {}, \"c\": 0}"; + let ours = "{\"a\": {}, \"b\": {}, \"c\": 1}"; + let theirs = + "{\n \"a\": {\"x\": 1},\n \"b\": {\n \"y\": 2,\n \"z\": 3\n },\n \"c\": 0\n}"; + assert_eq!( + merged(base, ours, theirs), + "{\"a\": {}, \"b\": {\n \"y\": 2,\n \"z\": 3\n}, \"c\": 1}" + ); + // Ours emptied it and theirs added to it: theirs' layout fills + // the brackets. + let theirs = "{\n \"k\": 1,\n \"n\": 2\n}"; + assert_eq!(merged("{\"k\": 1}", "{}", theirs), "{\n \"n\": 2\n}"); + } + + #[test] + fn the_same_change_on_both_sides_is_taken_once() { + let ours = PACKAGE.replace("3.7.1", "3.7.4"); + assert_eq!(merged(PACKAGE, &ours, &ours.clone()), ours); + } + + #[test] + fn the_same_dependency_bumped_twice_declines() { + let ours = PACKAGE.replace("3.7.1", "3.7.4"); + let theirs = PACKAGE.replace("3.7.1", "3.8.0"); + assert_eq!( + declined(PACKAGE, &ours, &theirs), + Decline { + pointer: "/devDependencies/@types~1luxon".into(), + reason: Reason::BothChanged, + } + ); + } + + #[test] + fn a_change_to_what_the_other_side_deleted_declines() { + let base = r#"{"a": {"x": 1}, "b": 1}"#; + assert_eq!( + declined(base, r#"{"b": 1}"#, r#"{"a": {"x": 2}, "b": 1}"#).reason, + Reason::DeletedAndChanged { + deleted_by: Side::Ours + } + ); + assert_eq!( + declined(base, r#"{"a": {"x": 2}, "b": 1}"#, r#"{"b": 1}"#), + Decline { + pointer: "/a".into(), + reason: Reason::DeletedAndChanged { + deleted_by: Side::Theirs + }, + } + ); + // Deleted on both sides, or deleted on one and untouched on the + // other, is not a conflict. + assert_eq!(merged(base, r#"{"b": 2}"#, r#"{"b": 1}"#), r#"{"b": 2}"#); + assert_eq!(merged(base, r#"{"a": {"x": 1}}"#, r#"{"b": 1}"#), "{}"); + } + + #[test] + fn objects_both_sides_added_are_merged_key_by_key() { + let base = r#"{"a": 1}"#; + let ours = r#"{"a": 1, "new": {"x": 1, "both": true}}"#; + let theirs = r#"{"a": 1, "new": {"both": true, "y": 2}}"#; + assert_eq!( + merged(base, ours, theirs), + r#"{"a": 1, "new": {"x": 1, "both": true, "y": 2}}"# + ); + let theirs = r#"{"a": 1, "new": {"x": 2}}"#; + assert_eq!(declined(base, ours, theirs).pointer, "/new/x"); + assert_eq!( + declined(base, r#"{"a": 1, "n": 1}"#, r#"{"a": 1, "n": 2}"#).reason, + Reason::BothAdded + ); + } + + #[test] + fn an_empty_base_is_an_add_add() { + assert_eq!( + merged("", r#"{"b": 1}"#, r#"{"a": 2}"#), + r#"{"a": 2, "b": 1}"# + ); + assert_eq!(merged("\n", "[1]\n", "[1]\n"), "[1]\n"); + assert_eq!(declined("", "[1]", "[2]").reason, Reason::BothAdded); + } + + #[test] + fn edits_to_different_parts_of_an_array_merge() { + let base = "[\n \"a\",\n \"b\",\n \"c\"\n]"; + let ours = "[\n \"A\",\n \"b\",\n \"c\"\n]"; + let theirs = "[\n \"a\",\n \"b\",\n \"c\",\n \"d\"\n]"; + assert_eq!( + merged(base, ours, theirs), + "[\n \"A\",\n \"b\",\n \"c\",\n \"d\"\n]" + ); + // Neighbouring elements edited in place, one per side. + assert_eq!(merged("[1, 2, 3]", "[9, 2, 3]", "[1, 8, 3]"), "[9, 8, 3]"); + // One side removes, the other edits elsewhere. + assert_eq!( + merged("[1, 2, 3, 4]", "[1, 3, 4]", "[1, 2, 3, 5]"), + "[1, 3, 5]" + ); + } + + #[test] + fn two_insertions_at_the_same_point_decline() { + // A `required` list would want both; an ordered pipeline might + // want either order or neither. Not ours to guess. + assert_eq!( + declined( + r#"{"required": ["a"]}"#, + r#"{"required": ["a", "b"]}"#, + r#"{"required": ["a", "c"]}"# + ), + Decline { + pointer: "/required".into(), + reason: Reason::ArrayEditsOverlap, + } + ); + // The same insertion on both sides is one insertion. + assert_eq!( + merged(r#"["a"]"#, r#"["a", "b"]"#, r#"["a", "b"]"#), + r#"["a", "b"]"# + ); + } + + #[test] + fn elements_edited_in_place_merge_recursively() { + let base = r#"[{"name": "a", "in": "query"}, {"name": "b"}]"#; + let ours = r#"[{"name": "a", "in": "path"}, {"name": "b"}]"#; + let theirs = r#"[{"name": "a", "in": "query", "required": true}, {"name": "b"}]"#; + assert_eq!( + merged(base, ours, theirs), + r#"[{"name": "a", "in": "path", "required": true}, {"name": "b"}]"# + ); + } + + #[test] + fn a_moved_element_is_not_merged_in_place() { + let base = r#"[{"k": 1}, {"k": 2}]"#; + let ours = r#"[{"k": 2}, {"k": 1, "x": 0}]"#; + let theirs = r#"[{"k": 1}, {"k": 2, "y": 0}]"#; + assert_eq!( + declined(base, ours, theirs).reason, + Reason::ArrayEditsOverlap + ); + } + + #[test] + fn a_type_change_against_an_edit_declines() { + assert_eq!( + declined(r#"{"a": [1]}"#, r#"{"a": {"x": 1}}"#, r#"{"a": [1, 2]}"#).reason, + Reason::BothChanged + ); + } + + #[test] + fn invalid_json_declines() { + let e = declined("{}", "{,}", "{}"); + assert!( + matches!( + e.reason, + Reason::NotJson { + side: Side::Ours, + .. + } + ), + "{e:?}" + ); + assert_eq!( + e.to_string(), + "ours is not valid JSON: expected a string at byte 1" + ); + let e = declined(r#"{"a":1,"a":2}"#, "{}", r#"{"b": 1}"#); + assert!( + matches!( + e.reason, + Reason::NotJson { + side: Side::Base, + .. + } + ), + "{e:?}" + ); + } + + #[test] + fn untouched_regions_are_copied_byte_for_byte() { + // Odd spacing, a blank line, number spellings and a key order + // nothing sorts: all of it survives a change elsewhere. + let base = "{ \"z\" :1.50 ,\n\n \"a\":[ 1,2 ],\"m\": {\"q\": true}}\r\n"; + let ours = base.replace("1.50", "1.5e0"); + let theirs = base.replace("true", "false"); + assert_eq!( + merged(base, &ours, &theirs), + "{ \"z\" :1.5e0 ,\n\n \"a\":[ 1,2 ],\"m\": {\"q\": false}}\r\n" + ); + } + + #[test] + fn a_reformatted_side_does_not_hide_the_other_sides_change() { + let base = r#"{"a": 1, "b": 2}"#; + let ours = "{\n \"a\": 1,\n \"b\": 2\n}"; + let theirs = r#"{"a": 1, "b": 3}"#; + assert_eq!(merged(base, ours, theirs), r#"{"a": 1, "b": 3}"#); + } + + #[test] + fn theirs_text_is_shifted_to_ours_depth() { + let base = "{\n \"a\": {\n \"x\": 1\n }\n}"; + let ours = "{\n \"a\": {\n \"x\": 1\n },\n \"b\": 0\n}"; + // Theirs indents by four where ours indents by two. Theirs' block + // is shifted to start at ours' depth; its inner steps stay + // theirs'. + let theirs = "{\n \"a\": {\n \"x\": 1,\n \"y\": {\n \"z\": 2\n }\n }\n}"; + assert_eq!( + merged(base, ours, theirs), + "{\n \"a\": {\n \"x\": 1,\n \"y\": {\n \"z\": 2\n }\n },\n \"b\": 0\n}" + ); + } + + #[test] + fn the_read_back_check_compares_structure_not_text() { + let v = value::parse(r#"{"a": [1, {"b": 2}]}"#).unwrap_or_else(|e| panic!("{e}")); + let Kind::Object(m) = &v.kind else { panic!() }; + let Kind::Array(items) = &m[0].value.kind else { + panic!() + }; + let good = Expect::Object(vec![( + "a", + Expect::Array(vec![Expect::Ours(&items[0]), Expect::Theirs(&items[1])]), + )]); + assert!(matches(&v, &good)); + let wrong_key = Expect::Object(vec![("b", Expect::Ours(&m[0].value))]); + assert!(!matches(&v, &wrong_key)); + let short = Expect::Object(vec![("a", Expect::Array(vec![Expect::Ours(&items[0])]))]); + assert!(!matches(&v, &short)); + } + + #[test] + fn indentation_helpers() { + assert_eq!(line_indent("a\n \t\"b\": 1", 6), " \t"); + assert_eq!(line_indent("x", 0), ""); + assert_eq!( + reindent("{\n \"a\": 1\n }", " ", "\t"), + "{\n\t \"a\": 1\n\t}" + ); + assert!(matches!(reindent("[1, 2]", " ", ""), Cow::Borrowed(_))); + } +} diff --git a/crates/mergify-json-merge/src/sequence.rs b/crates/mergify-json-merge/src/sequence.rs new file mode 100644 index 00000000..2cf6cafd --- /dev/null +++ b/crates/mergify-json-merge/src/sequence.rs @@ -0,0 +1,194 @@ +//! Element-wise diff3 over three JSON arrays. +//! +//! Aligns ours and theirs against base with a longest common +//! subsequence each, then cuts the three arrays into chunks: a *stable* +//! chunk is one base element both sides kept, and an *unstable* chunk is +//! the stretch between two stable ones, where at least one side changed +//! something. This is diff3 with elements in place of lines. + +use std::ops::Range; + +use crate::value::Value; +use crate::value::same; + +/// Above this many cells the alignment table (4 bytes a cell) is not +/// built and the merge declines. The common prefix and suffix are +/// trimmed first, so only arrays changed at both ends by thousands of +/// elements get here. +const MAX_TABLE_CELLS: usize = 4_000_000; + +#[derive(Debug, PartialEq, Eq)] +pub(crate) enum Chunk { + /// A base element both sides kept; ours holds it at `o`. + Stable { o: usize }, + Unstable { + b: Range, + o: Range, + t: Range, + }, +} + +pub(crate) struct TooLarge; + +pub(crate) fn diff3( + base: &[Value], + ours: &[Value], + theirs: &[Value], +) -> Result, TooLarge> { + let to_ours = align(base, ours)?; + let to_theirs = align(base, theirs)?; + // Where the next chunk starts, in base, ours and theirs. + let (mut at_b, mut at_o, mut at_t) = (0, 0, 0); + let mut chunks = Vec::new(); + loop { + let stable = (at_b..base.len()).find_map(|j| match (to_ours[j], to_theirs[j]) { + (Some(in_ours), Some(in_theirs)) => Some((j, in_ours, in_theirs)), + _ => None, + }); + let (nb, no, nt) = stable.unwrap_or((base.len(), ours.len(), theirs.len())); + if nb > at_b || no > at_o || nt > at_t { + chunks.push(Chunk::Unstable { + b: at_b..nb, + o: at_o..no, + t: at_t..nt, + }); + } + if stable.is_none() { + return Ok(chunks); + } + chunks.push(Chunk::Stable { o: no }); + (at_b, at_o, at_t) = (nb + 1, no + 1, nt + 1); + } +} + +/// For each element of `a`, its index in `b` under one longest common +/// subsequence, or `None`. Matched indices increase monotonically. +fn align(a: &[Value], b: &[Value]) -> Result>, TooLarge> { + let mut out = vec![None; a.len()]; + let mut prefix = 0; + while prefix < a.len() && prefix < b.len() && same(&a[prefix], &b[prefix]) { + out[prefix] = Some(prefix); + prefix += 1; + } + let mut suffix = 0; + while suffix < a.len() - prefix + && suffix < b.len() - prefix + && same(&a[a.len() - 1 - suffix], &b[b.len() - 1 - suffix]) + { + out[a.len() - 1 - suffix] = Some(b.len() - 1 - suffix); + suffix += 1; + } + let (am, bm) = (&a[prefix..a.len() - suffix], &b[prefix..b.len() - suffix]); + if am.is_empty() || bm.is_empty() { + return Ok(out); + } + let width = bm.len() + 1; + if (am.len() + 1).saturating_mul(width) > MAX_TABLE_CELLS { + return Err(TooLarge); + } + // `len[i * width + j]` is the LCS length of `am[i..]` and `bm[j..]`. + let mut len = vec![0u32; (am.len() + 1) * width]; + for i in (0..am.len()).rev() { + for j in (0..bm.len()).rev() { + len[i * width + j] = if same(&am[i], &bm[j]) { + len[(i + 1) * width + j + 1] + 1 + } else { + len[(i + 1) * width + j].max(len[i * width + j + 1]) + }; + } + } + let (mut i, mut j) = (0, 0); + while i < am.len() && j < bm.len() { + if same(&am[i], &bm[j]) { + out[prefix + i] = Some(prefix + j); + i += 1; + j += 1; + } else if len[(i + 1) * width + j] >= len[i * width + j + 1] { + i += 1; + } else { + j += 1; + } + } + Ok(out) +} + +#[cfg(test)] +mod tests { + use super::*; + use crate::value::Kind; + use crate::value::parse; + + fn items(text: &str) -> Vec { + match parse(text).map(|v| v.kind) { + Ok(Kind::Array(items)) => items, + _ => panic!("not an array: {text}"), + } + } + + fn chunks(b: &str, o: &str, t: &str) -> Vec { + diff3(&items(b), &items(o), &items(t)).unwrap_or_else(|_| panic!("too large")) + } + + #[test] + fn alignment_is_a_longest_common_subsequence() { + let got = + align(&items("[1,2,3,4,5]"), &items("[0,2,3,9,5,6]")).unwrap_or_else(|_| panic!()); + assert_eq!(got, vec![None, Some(1), Some(2), None, Some(4)]); + } + + #[test] + fn separate_edits_land_in_separate_chunks() { + assert_eq!( + chunks("[1,2,3]", "[0,1,2,3]", "[1,2,3,4]"), + vec![ + Chunk::Unstable { + b: 0..0, + o: 0..1, + t: 0..0 + }, + Chunk::Stable { o: 1 }, + Chunk::Stable { o: 2 }, + Chunk::Stable { o: 3 }, + Chunk::Unstable { + b: 3..3, + o: 4..4, + t: 3..4 + }, + ] + ); + } + + #[test] + fn edits_with_no_kept_element_between_them_share_a_chunk() { + assert_eq!( + chunks("[1,2,3]", "[9,2,3]", "[1,8,3]"), + vec![ + Chunk::Unstable { + b: 0..2, + o: 0..2, + t: 0..2 + }, + Chunk::Stable { o: 2 }, + ] + ); + } + + #[test] + fn a_huge_rewrite_is_refused() { + let a = format!( + "[{}]", + (0..2100) + .map(|i| i.to_string()) + .collect::>() + .join(",") + ); + let b = format!( + "[{}]", + (5000..7100) + .map(|i| i.to_string()) + .collect::>() + .join(",") + ); + assert!(align(&items(&a), &items(&b)).is_err()); + } +} diff --git a/crates/mergify-json-merge/src/value.rs b/crates/mergify-json-merge/src/value.rs new file mode 100644 index 00000000..dd9d44ac --- /dev/null +++ b/crates/mergify-json-merge/src/value.rs @@ -0,0 +1,528 @@ +//! A lossless JSON reader. +//! +//! Every value keeps the byte span it was read from, so the merge can +//! copy what did not change verbatim instead of re-serialising the +//! document. `serde_json` cannot do that: its `Value` keeps neither +//! spans nor (without the `preserve_order` feature, which would change +//! key order for every other crate in the binary) key order. +//! +//! The reader is strict RFC 8259 — no comments, no trailing commas — and +//! also refuses what JSON permits but a merge cannot reason about: +//! duplicate keys (which one wins is up to the reader) and lone UTF-16 +//! surrogates (not representable as text). + +use std::fmt; + +/// Deeper input is refused rather than recursed into: a stack overflow +/// kills the process with a signal, and git aborts the WHOLE merge on a +/// driver that dies that way instead of recording one conflicted path. +pub(crate) const MAX_DEPTH: usize = 256; + +#[derive(Debug)] +pub(crate) struct Value { + /// Byte offset of the value's first character. + pub start: usize, + /// Byte offset just past the value's last character. + pub end: usize, + /// Structural hash: equal values have equal hashes. Lets the array + /// alignment compare elements in constant time in the common case. + pub hash: u64, + pub kind: Kind, +} + +#[derive(Debug)] +pub(crate) enum Kind { + Null, + Bool(bool), + /// As written. `1` and `1.0` compare unequal, which can only make + /// the merge decline more often, never merge wrongly. + Number(String), + /// Decoded, so `"A"` and `"A"` compare equal. + String(String), + Array(Vec), + Object(Vec), +} + +#[derive(Debug)] +pub(crate) struct Member { + /// Byte offset of the key's opening quote. + pub start: usize, + pub key: String, + pub value: Value, +} + +impl Value { + /// The `(start, end)` span of each item of a container: the whole + /// `"key": value` for an object member, the value for an array + /// element. Empty for a scalar. + pub(crate) fn item_spans(&self) -> Vec<(usize, usize)> { + match &self.kind { + Kind::Array(items) => items.iter().map(|v| (v.start, v.end)).collect(), + Kind::Object(members) => members.iter().map(|m| (m.start, m.value.end)).collect(), + _ => Vec::new(), + } + } +} + +/// Structural equality: same type, same content, and — for objects — +/// the same keys in the same order. Key order counts so that a side +/// whose only change was reordering keys is not mistaken for unchanged. +pub(crate) fn same(a: &Value, b: &Value) -> bool { + if a.hash != b.hash { + return false; + } + match (&a.kind, &b.kind) { + (Kind::Null, Kind::Null) => true, + (Kind::Bool(x), Kind::Bool(y)) => x == y, + (Kind::Number(x), Kind::Number(y)) | (Kind::String(x), Kind::String(y)) => x == y, + (Kind::Array(x), Kind::Array(y)) => { + x.len() == y.len() && x.iter().zip(y).all(|(p, q)| same(p, q)) + } + (Kind::Object(x), Kind::Object(y)) => { + x.len() == y.len() + && x.iter() + .zip(y) + .all(|(p, q)| p.key == q.key && same(&p.value, &q.value)) + } + _ => false, + } +} + +#[derive(Debug)] +pub(crate) struct ParseError { + pub offset: usize, + pub message: &'static str, +} + +impl fmt::Display for ParseError { + fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result { + write!(f, "{} at byte {}", self.message, self.offset) + } +} + +/// Parse a whole document. A leading byte-order mark is skipped (spans +/// stay relative to the full text, so the merge keeps ours' BOM). +pub(crate) fn parse(text: &str) -> Result { + let mut parser = Parser { + bytes: text.as_bytes(), + text, + pos: 0, + depth: 0, + }; + if text.starts_with('\u{feff}') { + parser.pos = '\u{feff}'.len_utf8(); + } + parser.skip_ws(); + let value = parser.value()?; + parser.skip_ws(); + if parser.pos != parser.bytes.len() { + return Err(parser.error("trailing characters after the JSON value")); + } + Ok(value) +} + +struct Parser<'a> { + bytes: &'a [u8], + text: &'a str, + pos: usize, + depth: usize, +} + +impl Parser<'_> { + fn error(&self, message: &'static str) -> ParseError { + ParseError { + offset: self.pos, + message, + } + } + + fn peek(&self) -> Option { + self.bytes.get(self.pos).copied() + } + + fn skip_ws(&mut self) { + while let Some(b' ' | b'\t' | b'\n' | b'\r') = self.peek() { + self.pos += 1; + } + } + + fn expect(&mut self, byte: u8, message: &'static str) -> Result<(), ParseError> { + if self.peek() == Some(byte) { + self.pos += 1; + Ok(()) + } else { + Err(self.error(message)) + } + } + + fn value(&mut self) -> Result { + let start = self.pos; + let kind = match self.peek() { + Some(b'{') => self.object()?, + Some(b'[') => self.array()?, + Some(b'"') => Kind::String(self.string()?), + Some(b't') => self.literal("true", Kind::Bool(true))?, + Some(b'f') => self.literal("false", Kind::Bool(false))?, + Some(b'n') => self.literal("null", Kind::Null)?, + Some(b'-' | b'0'..=b'9') => self.number()?, + Some(_) => return Err(self.error("unexpected character")), + None => return Err(self.error("unexpected end of input")), + }; + Ok(Value { + start, + end: self.pos, + hash: hash_of(&kind), + kind, + }) + } + + fn literal(&mut self, word: &'static str, kind: Kind) -> Result { + if self.bytes[self.pos..].starts_with(word.as_bytes()) { + self.pos += word.len(); + Ok(kind) + } else { + Err(self.error("invalid literal")) + } + } + + fn digits(&mut self) -> usize { + let from = self.pos; + while let Some(b'0'..=b'9') = self.peek() { + self.pos += 1; + } + self.pos - from + } + + fn number(&mut self) -> Result { + let start = self.pos; + if self.peek() == Some(b'-') { + self.pos += 1; + } + match self.peek() { + Some(b'0') => self.pos += 1, + Some(b'1'..=b'9') => { + self.digits(); + } + _ => return Err(self.error("invalid number")), + } + if self.peek() == Some(b'.') { + self.pos += 1; + if self.digits() == 0 { + return Err(self.error("invalid number")); + } + } + if let Some(b'e' | b'E') = self.peek() { + self.pos += 1; + if let Some(b'+' | b'-') = self.peek() { + self.pos += 1; + } + if self.digits() == 0 { + return Err(self.error("invalid number")); + } + } + Ok(Kind::Number(self.text[start..self.pos].to_owned())) + } + + fn hex4(&mut self) -> Result { + let mut code = 0; + for _ in 0..4 { + let digit = match self.peek() { + Some(b @ b'0'..=b'9') => b - b'0', + Some(b @ b'a'..=b'f') => b - b'a' + 10, + Some(b @ b'A'..=b'F') => b - b'A' + 10, + _ => return Err(self.error("invalid \\u escape")), + }; + code = code * 16 + u32::from(digit); + self.pos += 1; + } + Ok(code) + } + + fn string(&mut self) -> Result { + self.expect(b'"', "expected a string")?; + let mut out = String::new(); + loop { + // Copy the run up to the next quote, backslash or control + // character in one go. Those are all ASCII, so the slice + // boundaries always fall on character boundaries. + let run = self.pos; + while let Some(b) = self.peek() { + if b == b'"' || b == b'\\' || b < 0x20 { + break; + } + self.pos += 1; + } + out.push_str(&self.text[run..self.pos]); + match self.peek() { + Some(b'"') => { + self.pos += 1; + return Ok(out); + } + Some(b'\\') => { + self.pos += 1; + let escaped = match self.peek() { + Some(b'"') => '"', + Some(b'\\') => '\\', + Some(b'/') => '/', + Some(b'b') => '\u{8}', + Some(b'f') => '\u{c}', + Some(b'n') => '\n', + Some(b'r') => '\r', + Some(b't') => '\t', + Some(b'u') => { + self.pos += 1; + out.push(self.unicode_escape()?); + continue; + } + _ => return Err(self.error("invalid escape")), + }; + self.pos += 1; + out.push(escaped); + } + Some(_) => return Err(self.error("control character in string")), + None => return Err(self.error("unterminated string")), + } + } + } + + /// Decode what follows a `\u`, pairing surrogates. + fn unicode_escape(&mut self) -> Result { + let high = self.hex4()?; + let code = if (0xD800..0xDC00).contains(&high) { + if !self.bytes[self.pos..].starts_with(b"\\u") { + return Err(self.error("lone surrogate in \\u escape")); + } + self.pos += 2; + let low = self.hex4()?; + if !(0xDC00..0xE000).contains(&low) { + return Err(self.error("lone surrogate in \\u escape")); + } + 0x10000 + ((high - 0xD800) << 10) + (low - 0xDC00) + } else { + high + }; + char::from_u32(code).ok_or_else(|| self.error("lone surrogate in \\u escape")) + } + + fn enter(&mut self) -> Result<(), ParseError> { + self.depth += 1; + if self.depth > MAX_DEPTH { + return Err(self.error("nesting too deep")); + } + self.pos += 1; + self.skip_ws(); + Ok(()) + } + + fn array(&mut self) -> Result { + self.enter()?; + let mut items = Vec::new(); + if self.peek() == Some(b']') { + self.pos += 1; + self.depth -= 1; + return Ok(Kind::Array(items)); + } + loop { + items.push(self.value()?); + self.skip_ws(); + match self.peek() { + Some(b',') => { + self.pos += 1; + self.skip_ws(); + } + Some(b']') => { + self.pos += 1; + self.depth -= 1; + return Ok(Kind::Array(items)); + } + _ => return Err(self.error("expected `,` or `]`")), + } + } + } + + fn object(&mut self) -> Result { + self.enter()?; + let mut members: Vec = Vec::new(); + if self.peek() == Some(b'}') { + self.pos += 1; + self.depth -= 1; + return Ok(Kind::Object(members)); + } + loop { + let start = self.pos; + let key = self.string()?; + self.skip_ws(); + self.expect(b':', "expected `:`")?; + self.skip_ws(); + let value = self.value()?; + members.push(Member { start, key, value }); + self.skip_ws(); + match self.peek() { + Some(b',') => { + self.pos += 1; + self.skip_ws(); + } + Some(b'}') => { + self.pos += 1; + self.depth -= 1; + reject_duplicate_keys(&members)?; + return Ok(Kind::Object(members)); + } + _ => return Err(self.error("expected `,` or `}`")), + } + } + } +} + +/// Sorting borrowed keys rather than filling a set of owned ones: this +/// runs for every object of a two-megabyte schema, three times a merge. +fn reject_duplicate_keys(members: &[Member]) -> Result<(), ParseError> { + let mut keys: Vec<(&str, usize)> = members.iter().map(|m| (m.key.as_str(), m.start)).collect(); + keys.sort_unstable(); + match keys.windows(2).find(|pair| pair[0].0 == pair[1].0) { + Some(pair) => Err(ParseError { + offset: pair[0].1.max(pair[1].1), + message: "duplicate key", + }), + None => Ok(()), + } +} + +/// 64-bit FNV-1a, written out because the std hasher is randomly seeded +/// per process and nothing here needs more than a fast pre-filter. +struct Fnv(u64); + +impl Fnv { + fn new(tag: u8) -> Self { + let mut h = Self(0xcbf2_9ce4_8422_2325); + h.write(&[tag]); + h + } + + fn write(&mut self, bytes: &[u8]) { + for &b in bytes { + self.0 ^= u64::from(b); + self.0 = self.0.wrapping_mul(0x0000_0100_0000_01b3); + } + } + + fn write_str(&mut self, s: &str) { + // Length-prefixed, so `["ab", "c"]` and `["a", "bc"]` differ. + self.write(&s.len().to_le_bytes()); + self.write(s.as_bytes()); + } +} + +fn hash_of(kind: &Kind) -> u64 { + let mut h; + match kind { + Kind::Null => h = Fnv::new(0), + Kind::Bool(b) => { + h = Fnv::new(1); + h.write(&[u8::from(*b)]); + } + Kind::Number(n) => { + h = Fnv::new(2); + h.write_str(n); + } + Kind::String(s) => { + h = Fnv::new(3); + h.write_str(s); + } + Kind::Array(items) => { + h = Fnv::new(4); + for item in items { + h.write(&item.hash.to_le_bytes()); + } + } + Kind::Object(members) => { + h = Fnv::new(5); + for member in members { + h.write_str(&member.key); + h.write(&member.value.hash.to_le_bytes()); + } + } + } + h.0 +} + +#[cfg(test)] +mod tests { + use super::*; + + fn ok(text: &str) -> Value { + parse(text).unwrap_or_else(|e| panic!("{text:?}: {e}")) + } + + fn err(text: &str) -> &'static str { + parse(text).expect_err(text).message + } + + #[test] + fn spans_point_at_the_source_text() { + let text = "{\n \"a\": [1, true],\n \"b\": \"x\"\n}\n"; + let v = ok(text); + let Kind::Object(members) = &v.kind else { + panic!("not an object") + }; + assert_eq!( + &text[v.start..v.end], + "{\n \"a\": [1, true],\n \"b\": \"x\"\n}" + ); + assert_eq!( + &text[members[0].start..members[0].value.end], + "\"a\": [1, true]" + ); + assert_eq!(&text[members[1].value.start..members[1].value.end], "\"x\""); + } + + #[test] + fn strings_are_decoded() { + let v = ok(r#""aA\n😀\/""#); + assert!(matches!(&v.kind, Kind::String(s) if s == "aA\n\u{1f600}/")); + assert!(same(&ok(r#""A""#), &ok(r#""A""#))); + } + + #[test] + fn numbers_compare_as_written() { + assert!(!same(&ok("1"), &ok("1.0"))); + assert!(same(&ok("-0.5e+3"), &ok("-0.5e+3"))); + } + + #[test] + fn key_order_is_part_of_equality() { + assert!(!same(&ok(r#"{"a":1,"b":2}"#), &ok(r#"{"b":2,"a":1}"#))); + assert!(same(&ok(r#"{"a":1, "b":2}"#), &ok("{\"a\": 1,\n\"b\": 2}"))); + } + + #[test] + fn a_bom_is_skipped() { + let v = ok("\u{feff}{}"); + assert_eq!(v.start, 3); + } + + #[test] + fn rejects_what_a_merge_cannot_reason_about() { + assert_eq!(err(r#"{"a":1,"a":2}"#), "duplicate key"); + assert_eq!(err(r#""\ud800""#), "lone surrogate in \\u escape"); + assert_eq!(err(r#""\udc00""#), "lone surrogate in \\u escape"); + assert_eq!(err("[1,]"), "unexpected character"); + assert_eq!( + err("{} // comment"), + "trailing characters after the JSON value" + ); + assert_eq!(err("01"), "trailing characters after the JSON value"); + assert_eq!(err("1."), "invalid number"); + assert_eq!(err("\"a\tb\""), "control character in string"); + assert_eq!(err(""), "unexpected end of input"); + assert_eq!(err("truth"), "invalid literal"); + } + + #[test] + fn nesting_is_bounded() { + let deep = "[".repeat(MAX_DEPTH + 1) + &"]".repeat(MAX_DEPTH + 1); + assert_eq!(err(&deep), "nesting too deep"); + let fine = "[".repeat(MAX_DEPTH) + &"]".repeat(MAX_DEPTH); + ok(&fine); + } +} diff --git a/crates/mergify-json-merge/tests/driver.rs b/crates/mergify-json-merge/tests/driver.rs new file mode 100644 index 00000000..78b8bcc4 --- /dev/null +++ b/crates/mergify-json-merge/tests/driver.rs @@ -0,0 +1,119 @@ +//! `run`: the file handling around the merge, and the line-merge +//! fallback a decline hands over to. + +use std::path::PathBuf; + +use mergify_core::CliError; +use mergify_json_merge::DriverOptions; +use mergify_json_merge::run; + +struct Files { + _dir: tempfile::TempDir, + ours: PathBuf, + base: PathBuf, + theirs: PathBuf, +} + +fn files(base: &str, ours: &str, theirs: &str) -> Files { + let dir = tempfile::tempdir().unwrap(); + let path = |name: &str, content: &str| { + let p = dir.path().join(name); + std::fs::write(&p, content).unwrap(); + p + }; + Files { + ours: path("ours", ours), + base: path("base", base), + theirs: path("theirs", theirs), + _dir: dir, + } +} + +fn merge(f: &Files) -> (Result<(), CliError>, String) { + let result = run(&DriverOptions { + ours: &f.ours, + base: &f.base, + theirs: &f.theirs, + marker_size: None, + path: Some("x.json"), + }); + (result, std::fs::read_to_string(&f.ours).unwrap()) +} + +#[test] +fn a_structural_merge_is_written_over_ours() { + let f = files("[1, 2, 3]\n", "[0, 2, 3]\n", "[1, 2, 4]\n"); + let (result, ours) = merge(&f); + assert!(result.is_ok(), "{result:?}"); + assert_eq!(ours, "[0, 2, 4]\n"); +} + +#[test] +fn the_trivial_cases_need_no_parse() { + // Not JSON at all, and still settled: one side unchanged. + let f = files("a\n", "b\n", "a\n"); + assert!(merge(&f).0.is_ok()); + assert_eq!(merge(&f).1, "b\n"); + let f = files("a\n", "a\n", "c\n"); + assert_eq!(merge(&f).1, "c\n"); +} + +#[test] +fn a_decline_leaves_git_conflict_markers() { + let f = files("{\"a\": 1}\n", "{\"a\": 2}\n", "{\"a\": 3}\n"); + let (result, ours) = merge(&f); + let Err(CliError::Conflict(msg)) = result else { + panic!("{result:?}") + }; + assert_eq!( + msg, + "x.json: both sides changed this value differently (at `/a`); left conflict markers" + ); + assert_eq!( + ours, + "<<<<<<< ours\n{\"a\": 2}\n=======\n{\"a\": 3}\n>>>>>>> theirs\n" + ); +} + +#[test] +fn not_json_falls_back_to_the_line_merge() { + // A comment makes it JSONC; git's line merge is what it gets, as it + // would without the driver. + let base = "// x\n{\n \"a\": 1,\n \"b\": 2,\n \"c\": 3\n}\n"; + let f = files( + base, + &base.replace("\"a\": 1", "\"a\": 0"), + &base.replace("\"c\": 3", "\"c\": 4"), + ); + let (result, ours) = merge(&f); + assert!(result.is_ok(), "{result:?}"); + assert_eq!( + ours, + base.replace("\"a\": 1", "\"a\": 0") + .replace("\"c\": 3", "\"c\": 4") + ); +} + +#[test] +fn a_clean_line_merge_into_invalid_json_is_a_conflict() { + // Both sides add `n`, with different values, at opposite ends of + // the object: a structural decline, and lines far enough apart for + // the line merge to call it clean — with `n` in it twice. + let base = "{\n \"a\": 1,\n \"b\": 2,\n \"c\": 3\n}\n"; + let ours = base.replace("{\n", "{\n \"n\": 1,\n"); + let theirs = base.replace("\"c\": 3\n", "\"c\": 3,\n \"n\": 2\n"); + let f = files(base, &ours, &theirs); + let (result, merged) = merge(&f); + assert!( + merged.contains("\"n\": 1") && merged.contains("\"n\": 2"), + "{merged}" + ); + let Err(CliError::Conflict(msg)) = result else { + panic!("{result:?}") + }; + assert_eq!( + msg, + "x.json: both sides added this value, differently (at `/n`); \ + the line merge git falls back to produced invalid JSON" + ); +}