From fdc2c9b1515d6d2216e7d25cf85f7c2c58fa9236 Mon Sep 17 00:00:00 2001
From: Tauan BF <11513929+tauanbinato@users.noreply.github.com>
Date: Sun, 27 Sep 2026 19:12:11 -0300
Subject: [PATCH 1/5] Split clones.rs into the matching core, the copies kept
apart, the walk frame and its tests
---
src/analysis/clones.rs | 1909 ----------------------------------
src/analysis/clones/apart.rs | 170 +++
src/analysis/clones/frame.rs | 261 +++++
src/analysis/clones/mod.rs | 838 +++++++++++++++
src/analysis/clones/tests.rs | 650 ++++++++++++
5 files changed, 1919 insertions(+), 1909 deletions(-)
delete mode 100644 src/analysis/clones.rs
create mode 100644 src/analysis/clones/apart.rs
create mode 100644 src/analysis/clones/frame.rs
create mode 100644 src/analysis/clones/mod.rs
create mode 100644 src/analysis/clones/tests.rs
diff --git a/src/analysis/clones.rs b/src/analysis/clones.rs
deleted file mode 100644
index 98e269e..0000000
--- a/src/analysis/clones.rs
+++ /dev/null
@@ -1,1909 +0,0 @@
-//! Type-2 clone candidates across the selected files and explicit context.
-//! Identifiers and literals are normalized; windows start and end on whole
-//! statements inside function bodies; identifiers must be renamed consistently.
-use super::{
- fast_hash, is_comment, line_of, text,
- units::{Kind, Unit},
-};
-use std::{
- collections::{BTreeMap, BTreeSet},
- ops::Range,
- path::{Path, PathBuf},
-};
-use tree_sitter::Node;
-
-pub const MIN_BYTES: usize = 120;
-/// Consecutive matching statements that seed a candidate window.
-pub const MIN_STATEMENTS: usize = 2;
-/// Statements a reported copy needs.
-pub const MIN_CLONE_STATEMENTS: usize = 3;
-pub const RUN_CAP: usize = 64;
-pub const FILE_CAP: usize = 8;
-const DIFFERENCES: usize = 12;
-/// A statement pair repeated more often than this is an idiom; its extra pairs are not compared.
-const SEED_OCCURRENCES: usize = 48;
-
-pub struct SourceFile<'a> {
- pub path: &'a Path,
- pub source: &'a str,
- /// False for explicit context: a pair needs at least one selected site.
- pub selected: bool,
- pub units: &'a [Unit],
- /// Lines excluded from comparison, such as test code when tests are not judged.
- pub excluded: Vec>,
- /// The package the file belongs to; explicit context has none.
- pub package: Option<&'a crate::packages::Package>,
-}
-
-#[derive(Clone, Debug, PartialEq, Eq)]
-pub struct Site {
- pub file: usize,
- pub path: PathBuf,
- pub span: Range,
- pub start_line: usize,
- pub end_line: usize,
- /// Enclosing function or method, when there is one.
- pub function: Option,
- pub function_source: Option,
- pub quote: String,
-}
-
-#[derive(Clone, Debug, PartialEq, Eq)]
-pub struct Difference {
- pub a: String,
- pub b: String,
-}
-
-#[derive(Clone, Debug)]
-pub struct Pair {
- pub a: Site,
- pub b: Site,
- pub differences: Vec,
- /// Non-whitespace bytes of the shorter site.
- pub size: usize,
- /// Distinct sites in this pair's clone group, including `a` and `b`.
- pub occurrences: usize,
- /// The group's other copies, reported with the judged pair.
- pub copies: Vec,
- /// Hash of the normalized statements; stable across renames and moves.
- pub normalized: String,
-}
-
-impl Pair {
- pub fn rank(&self) -> usize {
- self.size * self.occurrences
- }
-}
-
-#[derive(Default)]
-pub struct Candidates {
- pub pairs: Vec,
- /// Pairs dropped by the per-run or per-file caps, by owning path.
- pub omitted: BTreeMap,
-}
-
-#[derive(Clone, Copy, PartialEq, Eq)]
-enum TokenKind {
- Identifier,
- Literal,
- Other,
-}
-
-struct Token<'a> {
- kind: TokenKind,
- text: &'a str,
- start: usize,
- /// A call of the function the token sits in: `bound(child, …)` inside
- /// `bound`. Recursion names the function itself, so two walks calling
- /// themselves differ by no renamed name.
- own: bool,
-}
-
-/// Placeholders for renamed identifiers and literals. The control character
-/// keeps them apart from any real token text.
-const IDENTIFIER_TOKEN: &str = "\u{1}id";
-const LITERAL_TOKEN: &str = "\u{1}lit";
-
-impl Token<'_> {
- fn normal(&self) -> &str {
- match self.kind {
- TokenKind::Identifier => IDENTIFIER_TOKEN,
- TokenKind::Literal => LITERAL_TOKEN,
- TokenKind::Other => self.text,
- }
- }
-}
-
-#[derive(Clone)]
-struct Statement {
- span: Range,
- tokens: Range,
- hash: u64,
- /// Part of the frame of a walk rather than its work.
- frame: Option,
- /// Go's error check or deferred cleanup, which every call site repeats.
- idiom: bool,
-}
-
-/// The frame of a tree walk: the statements every walk has, whatever it
-/// does at each node.
-#[derive(Clone, Copy, PartialEq, Eq)]
-enum Frame {
- /// A branch that only leaves, as `if node.kind() == "call" { return; }`.
- Exit,
- /// A call of the function itself and nothing else, or a loop or branch
- /// that only does that or leaves, as
- /// `for child in node.named_children(&mut cursor) { bound(child, names); }`.
- Recursion,
-}
-
-struct Block {
- file: usize,
- statements: Vec,
- /// The statements are the whole body of a function.
- whole: bool,
-}
-
-struct Parsed<'a> {
- tokens: Vec>,
-}
-
-pub fn find(files: &[SourceFile<'_>]) -> Candidates {
- let (parsed, blocks) = statement_blocks(files);
- let local: BTreeSet = files
- .iter()
- .filter_map(|f| f.package?.name.clone())
- .collect();
- let mut pairs: Vec = matching_windows(&blocks)
- .into_iter()
- .filter(|&((bx, _), (by, _), _)| {
- let (a, b) = (&files[blocks[bx].file], &files[blocks[by].file]);
- crate::packages::linked(a.package, b.package, &local)
- && !separate_examples(a.path, b.path)
- && !separate_tests(a, b)
- })
- .filter_map(|window| pair(files, &parsed, &blocks, window))
- .filter(|p| !deprecated(files, &p.a) && !deprecated(files, &p.b))
- .filter(|p| !retired(&p.a.path) && !retired(&p.b.path))
- .collect();
- drop_nested(&mut pairs);
- pairs.sort_by(by_rank);
- let mut pairs = representatives(pairs);
- // Groups rank by size times their number of copies.
- pairs.sort_by(by_rank);
- capped(one_per_function_pair(pairs))
-}
-
-/// Whether a copy lies in a function or type marked deprecated: it goes
-/// with the next major version, so sharing its code with its replacement
-/// is not worth doing. flysystem's deprecated phpseclib 2 adapter was
-/// paired with its phpseclib 3 successor in 7 reviews.
-fn deprecated(files: &[SourceFile<'_>], site: &Site) -> bool {
- let file = &files[site.file];
- let Ok(Some(tree)) = crate::syntax::parse(file.path, file.source) else {
- return false;
- };
- let mut node = tree
- .root_node()
- .descendant_for_byte_range(site.span.start, site.span.start);
- while let Some(current) = node {
- let kind = current.kind();
- let declaration = kind.ends_with("_declaration")
- || kind.ends_with("_definition")
- || kind.ends_with("_item")
- || matches!(kind, "method" | "class" | "module" | "function");
- if declaration && super::units::deprecated(current, file.source) {
- return true;
- }
- node = current.parent();
- }
- false
-}
-
-/// Whether a file lies in a directory of retired code, such as
-/// `deprecated`, `archive` or a proof of concept: like code marked
-/// deprecated, it is not worth sharing code with. A Unity project's
-/// `Assets/ProofOfConcept` builders, kept as a reference with no menu entry,
-/// were paired with the live scene builders in six wrong reviews. `legacy`
-/// is left out, since legacy code is often still served.
-fn retired(path: &Path) -> bool {
- path.parent().is_some_and(|dir| {
- dir.iter().any(|part| {
- let part = part
- .to_string_lossy()
- .to_ascii_lowercase()
- .replace(['-', '_'], "");
- [
- "deprecated",
- "archive",
- "archived",
- "attic",
- "graveyard",
- "retired",
- "obsolete",
- "proofofconcept",
- "poc",
- "pocs",
- ]
- .contains(&part.as_str())
- })
- })
-}
-
-/// A directory of example code: `examples`, `demo`, `tutorial`, or a name
-/// such as `blog_examples`.
-fn example_directory(part: &str) -> bool {
- let part = part.to_ascii_lowercase();
- [
- "example",
- "examples",
- "demo",
- "demos",
- "tutorial",
- "tutorials",
- "docs_src",
- ]
- .contains(&part.as_str())
- || part.ends_with("_examples")
- || part.ends_with("-examples")
-}
-
-/// Two Bend 2 tests: each is a whole program pinned to the output its run
-/// prints, so their copies are the point of each test. Of 4 shared-logic
-/// findings between such tests on thirteen Bend 2 projects, all were wrong.
-fn separate_tests(a: &SourceFile<'_>, b: &SourceFile<'_>) -> bool {
- let test = |f: &SourceFile<'_>| {
- crate::analysis::bend::file(f.path)
- && crate::analysis::bend::expected_output(f.source).is_some()
- };
- a.path != b.path && test(a) && test(b)
-}
-
-/// Whether a file sits in a benchmark directory.
-pub(crate) fn benchmark_code(path: &Path) -> bool {
- path.parent().is_some_and(|dir| {
- dir.iter()
- .any(|part| benchmark_directory(&part.to_string_lossy()))
- })
-}
-
-fn benchmark_directory(part: &str) -> bool {
- matches!(
- part.to_ascii_lowercase().as_str(),
- "bench" | "benches" | "benchmark" | "benchmarks"
- )
-}
-
-/// Whether a file is example code, written to be read beside other examples.
-/// Also a top-level `samples` or `sample` directory (a Java package named
-/// `samples` is source), a .NET project named like `MediatR.Examples.Autofac`,
-/// and Go's `example_*_test.go` files, which show how to call a package.
-/// Directories below a JVM source root (`src/main/java`) are packages, not
-/// examples: Spring Initializr names a new project's package
-/// `com.example.demo`, which made every finding of such a project a note.
-pub(crate) fn example_code(path: &Path) -> bool {
- let name = path
- .file_name()
- .map(|n| n.to_string_lossy().to_ascii_lowercase())
- .unwrap_or_default();
- let top = path
- .iter()
- .next()
- .map(|p| p.to_string_lossy().to_ascii_lowercase())
- .filter(|_| path.iter().count() > 1)
- .unwrap_or_default();
- let directories: Vec = path
- .parent()
- .map(|dir| {
- dir.iter()
- .map(|p| p.to_string_lossy().into_owned())
- .collect()
- })
- .unwrap_or_default();
- let packages = jvm_source_root(&directories).unwrap_or(directories.len());
- (name.starts_with("example_") && name.ends_with("_test.go"))
- || matches!(top.as_str(), "samples" | "sample")
- || directories[..packages]
- .iter()
- .any(|part| example_directory(part) || part.to_ascii_lowercase().contains(".examples"))
-}
-
-/// Where the package directories of a JVM source root begin: after
-/// `src//java` (or `kotlin`, `scala`, `groovy`).
-fn jvm_source_root(directories: &[String]) -> Option {
- directories
- .windows(3)
- .position(|w| {
- w[0] == "src" && matches!(w[2].as_str(), "java" | "kotlin" | "scala" | "groovy")
- })
- .map(|at| at + 3)
-}
-
-/// Whether two files are separate variants of one example, kept side by
-/// side on purpose: under the same `examples` (or `demo`, `tutorial`)
-/// directory, in different directories below it. django-styleguide shows a
-/// Google login flow written by hand in `blog_examples/…/raw` and with the
-/// SDK in `…/sdk`; their copies are the point. Benchmarks kept so are
-/// separate programs too: each of bendlang/bend's `bench/runtime/*` is a
-/// standalone program measured beside its C, TypeScript and Lean twins,
-/// and the 6 shared-logic findings across them were labeled wrong.
-fn separate_examples(a: &Path, b: &Path) -> bool {
- let example = |part: &str| example_directory(part) || benchmark_directory(part);
- let dirs = |p: &Path| -> Vec {
- p.parent()
- .map(|d| d.iter().map(|c| c.to_string_lossy().into_owned()).collect())
- .unwrap_or_default()
- };
- let (a, b) = (dirs(a), dirs(b));
- let Some(root) = a.iter().zip(&b).position(|(x, y)| x == y && example(x)) else {
- return false;
- };
- a[..=root] == b[..=root] && a[root + 1..] != b[root + 1..]
-}
-
-/// Two copied windows of the same two functions, split by one differing
-/// statement, are one repetition: keep the higher-ranked pair only.
-fn one_per_function_pair(pairs: Vec) -> Vec {
- let mut seen = BTreeSet::new();
- pairs
- .into_iter()
- .filter(|pair| {
- let (Some(a), Some(b)) = (&pair.a.function, &pair.b.function) else {
- return true;
- };
- let mut key = [(&pair.a.path, a), (&pair.b.path, b)];
- key.sort();
- seen.insert(key.map(|(path, name)| (path.clone(), name.clone())))
- })
- .collect()
-}
-
-/// Tokens of every file and the statement blocks inside unit bodies.
-fn statement_blocks<'a>(files: &[SourceFile<'a>]) -> (Vec>, Vec) {
- let mut parsed = Vec::new();
- let mut blocks = Vec::new();
- for (index, file) in files.iter().enumerate() {
- let Ok(Some(tree)) = crate::syntax::parse(file.path, file.source) else {
- parsed.push(Parsed { tokens: Vec::new() });
- continue;
- };
- let mut tokens = Vec::new();
- leaves(tree.root_node(), file.source, &mut tokens);
- mark_recursion(&mut tokens, file.units, file.path);
- let bodies: Vec> = file
- .units
- .iter()
- .filter(|u| !u.equality)
- .filter_map(|u| u.body.clone())
- .collect();
- collect_blocks(tree.root_node(), file, index, &bodies, &tokens, &mut blocks);
- parsed.push(Parsed { tokens });
- }
- (parsed, blocks)
-}
-
-/// A maximal run of matching statement hashes: (block, start) twice and its length.
-type Window = ((usize, usize), (usize, usize), usize);
-
-/// Seed on consecutive statement pairs, then extend each diagonal as far as the
-/// hashes keep matching; a diagonal already covered is not reported again.
-fn matching_windows(blocks: &[Block]) -> Vec {
- let mut covered = BTreeSet::new();
- let mut found = Vec::new();
- for ((bx, kx), (by, ky)) in seed_pairs(blocks) {
- let diagonal = (bx, by, kx as isize - ky as isize);
- if covered.contains(&(diagonal, kx)) {
- continue;
- }
- let n = extend(blocks, (bx, kx), (by, ky));
- covered.extend((0..n).map(|t| (diagonal, kx + t)));
- if bx == by && one_run(&blocks[bx].statements[kx.min(ky)..kx.max(ky) + n]) {
- continue;
- }
- found.push(((bx, kx), (by, ky), n));
- }
- found
-}
-
-/// Every two places one seed occurs, among its first `SEED_OCCURRENCES`;
-/// two places in one block must be far enough apart not to overlap.
-fn seed_pairs(blocks: &[Block]) -> impl Iterator- {
- seeds(blocks).into_values().flat_map(|mut places| {
- places.truncate(SEED_OCCURRENCES);
- let pairs: Vec<_> = places
- .iter()
- .enumerate()
- .flat_map(|(x, &a)| places[x + 1..].iter().map(move |&b| (a, b)))
- .filter(|&((bx, kx), (by, ky))| bx != by || ky >= kx + MIN_STATEMENTS)
- .collect();
- pairs
- })
-}
-
-/// Statements that all read alike, such as sqlite-utils' nine
-/// `x = self.value_or_default("x", x)` lines or a list of lazy imports: a
-/// list of one kind of statement, which matches itself shifted by one.
-fn one_run(statements: &[Statement]) -> bool {
- statements
- .windows(2)
- .all(|pair| pair[0].hash == pair[1].hash)
-}
-
-/// Every place a pair of consecutive statement hashes occurs.
-fn seeds(blocks: &[Block]) -> BTreeMap<(u64, u64), Vec<(usize, usize)>> {
- let mut seeds = BTreeMap::<(u64, u64), Vec<(usize, usize)>>::new();
- for (b, block) in blocks.iter().enumerate() {
- for k in 0..block.statements.len().saturating_sub(1) {
- let key = (block.statements[k].hash, block.statements[k + 1].hash);
- seeds.entry(key).or_default().push((b, k));
- }
- }
- seeds
-}
-
-/// How many statements match from two seeds; a window never overlaps itself.
-fn extend(blocks: &[Block], (bx, kx): (usize, usize), (by, ky): (usize, usize)) -> usize {
- let (sx, sy) = (&blocks[bx].statements, &blocks[by].statements);
- let mut n = MIN_STATEMENTS;
- while kx + n < sx.len()
- && ky + n < sy.len()
- && sx[kx + n].hash == sy[ky + n].hash
- && (bx != by || kx + n < ky)
- {
- n += 1;
- }
- n
-}
-
-/// A candidate pair from one window, when its tokens align with consistent
-/// renaming and it is large enough to report. The owner is a selected site.
-fn pair(
- files: &[SourceFile<'_>],
- parsed: &[Parsed<'_>],
- blocks: &[Block],
- window: Window,
-) -> Option {
- let ((bx, kx), (by, ky), n) = window;
- let (fx, fy) = (blocks[bx].file, blocks[by].file);
- if !files[fx].selected && !files[fy].selected {
- return None;
- }
- let x = &blocks[bx].statements[kx..kx + n];
- let y = &blocks[by].statements[ky..ky + n];
- let tx = &parsed[fx].tokens[x[0].tokens.start..x[n - 1].tokens.end];
- let ty = &parsed[fy].tokens[y[0].tokens.start..y[n - 1].tokens.end];
- let differences = align(tx, ty)?;
- let span_x = x[0].span.start..x[n - 1].span.end;
- let span_y = y[0].span.start..y[n - 1].span.end;
- let size =
- compact(&files[fx].source[span_x.clone()]).min(compact(&files[fy].source[span_y.clone()]));
- // A repeated pair of statements is usually an idiom, such as a call and
- // its check.
- if n < MIN_CLONE_STATEMENTS
- || size < MIN_BYTES
- || only_frame(files, blocks, window)
- || mostly_guards(files, blocks, window)
- {
- return None;
- }
- let normalized = crate::schema::hash(
- tx.iter()
- .map(Token::normal)
- .collect::>()
- .join(crate::schema::HASH_SEPARATOR)
- .as_bytes(),
- );
- let a = site(files, fx, span_x);
- let b = site(files, fy, span_y);
- // Ties keep path and line order.
- let swap = !files[fx].selected
- || (files[fy].selected && (&b.path, b.start_line) < (&a.path, a.start_line));
- let (a, b, differences) = if swap {
- let flipped = differences
- .into_iter()
- .map(|d| Difference { a: d.b, b: d.a })
- .collect();
- (b, a, flipped)
- } else {
- (a, b, differences)
- };
- Some(Pair {
- a,
- b,
- differences,
- size,
- occurrences: 2,
- copies: Vec::new(),
- normalized,
- })
-}
-
-/// Two parts of recursive functions that share little beyond the frame of a
-/// walk: early exits and recursion into the function itself. Two walks that
-/// stop at a different kind and recurse into their children share that
-/// frame whatever they do at each node, so a copy of part of them needs two
-/// statements beyond it, or one of half the size a copy needs. A copy of
-/// the whole of both functions is a copy, however small its work.
-fn only_frame(files: &[SourceFile<'_>], blocks: &[Block], window: Window) -> bool {
- let ((bx, kx), (by, ky), n) = window;
- let x = &blocks[bx].statements[kx..kx + n];
- let y = &blocks[by].statements[ky..ky + n];
- let recurses = |s: &[Statement]| s.iter().any(|s| s.frame == Some(Frame::Recursion));
- let whole = |b: usize, k: usize| blocks[b].whole && k == 0 && n == blocks[b].statements.len();
- if !recurses(x) || !recurses(y) || whole(bx, kx) && whole(by, ky) {
- return false;
- }
- let work: Vec<(&Statement, &Statement)> = x
- .iter()
- .zip(y)
- .filter(|(a, b)| a.frame.is_none() || b.frame.is_none())
- .collect();
- let bytes = |file: usize, s: &Statement| compact(&files[file].source[s.span.clone()]);
- let size = work
- .iter()
- .map(|(a, _)| bytes(blocks[bx].file, a))
- .sum::()
- .min(work.iter().map(|(_, b)| bytes(blocks[by].file, b)).sum());
- work.len() < MIN_STATEMENTS && size < MIN_BYTES / 2
-}
-
-/// Part of two Go functions that is mostly error checks and deferred
-/// cleanups: `if err != nil { return err }` after each call and
-/// `defer tx.Rollback()`. wtf's per-entity store functions shared a
-/// transaction's begin, rollback and error checks around calls to their own
-/// type's functions, which read as copies. A copy of part of them needs as
-/// much other work as a copy of a walk's frame does; a copy of the whole of
-/// both functions is still one.
-fn mostly_guards(files: &[SourceFile<'_>], blocks: &[Block], window: Window) -> bool {
- let ((bx, kx), (by, ky), n) = window;
- let x = &blocks[bx].statements[kx..kx + n];
- let y = &blocks[by].statements[ky..ky + n];
- let whole = |b: usize, k: usize| blocks[b].whole && k == 0 && n == blocks[b].statements.len();
- if !x.iter().any(|s| s.idiom) || whole(bx, kx) && whole(by, ky) {
- return false;
- }
- let work: Vec<(&Statement, &Statement)> = x
- .iter()
- .zip(y)
- .filter(|(a, b)| !a.idiom || !b.idiom)
- .collect();
- let bytes = |file: usize, s: &Statement| compact(&files[file].source[s.span.clone()]);
- let size = work
- .iter()
- .map(|(a, _)| bytes(blocks[bx].file, a))
- .sum::()
- .min(work.iter().map(|(_, b)| bytes(blocks[by].file, b)).sum());
- work.len() < MIN_CLONE_STATEMENTS && size < MIN_BYTES
-}
-
-/// Go's `if err != nil { return …, err }` or a `defer` statement.
-fn go_idiom(statement: Node<'_>, source: &str) -> bool {
- match statement.kind() {
- "defer_statement" => true,
- "if_statement" => {
- let checks_err = statement
- .child_by_field_name("condition")
- .is_some_and(|c| compact_text(&source[c.byte_range()]) == "err!=nil");
- let returns = statement
- .child_by_field_name("consequence")
- .is_some_and(|block| {
- // Newer Go grammars wrap a block's statements in a list.
- let list = block
- .named_child(0)
- .filter(|c| c.kind() == "statement_list")
- .unwrap_or(block);
- let mut cursor = list.walk();
- let body: Vec> = list.named_children(&mut cursor).collect();
- !body.is_empty() && body.iter().all(|s| s.kind() == "return_statement")
- });
- checks_err && returns && statement.child_by_field_name("alternative").is_none()
- }
- _ => false,
- }
-}
-
-fn compact_text(text: &str) -> String {
- text.chars().filter(|c| !c.is_whitespace()).collect()
-}
-
-/// Drop pairs whose sites both lie inside a larger pair's sites.
-fn drop_nested(pairs: &mut Vec) {
- let snapshot = pairs.clone();
- pairs.retain(|p| {
- !snapshot.iter().any(|q| {
- let larger = q.a.span.len() + q.b.span.len() > p.a.span.len() + p.b.span.len();
- larger
- && ((contains(&q.a, &p.a) && contains(&q.b, &p.b))
- || (contains(&q.a, &p.b) && contains(&q.b, &p.a)))
- })
- });
-}
-
-/// Keep ranked groups within the per-run and per-file caps; count the rest.
-fn capped(pairs: Vec) -> Candidates {
- let mut omitted = BTreeMap::::new();
- let mut per_file = BTreeMap::::new();
- let mut kept = Vec::new();
- for pair in pairs {
- let count = per_file.entry(pair.a.path.clone()).or_default();
- if kept.len() < RUN_CAP && *count < FILE_CAP {
- *count += 1;
- kept.push(pair);
- } else {
- *omitted.entry(pair.a.path.clone()).or_default() += 1;
- }
- }
- Candidates {
- pairs: kept,
- omitted,
- }
-}
-
-fn by_rank(p: &Pair, q: &Pair) -> std::cmp::Ordering {
- q.rank()
- .cmp(&p.rank())
- .then_with(|| (&p.a.path, p.a.start_line).cmp(&(&q.a.path, q.a.start_line)))
- .then_with(|| (&p.b.path, p.b.start_line).cmp(&(&q.b.path, q.b.start_line)))
-}
-
-fn overlaps(x: &Site, y: &Site) -> bool {
- x.path == y.path && x.span.start < y.span.end && y.span.start < x.span.end
-}
-
-/// Sites that cover at least half of each other describe the same code.
-fn same_code(x: &Site, y: &Site) -> bool {
- if x.path != y.path {
- return false;
- }
- let shared = x
- .span
- .end
- .min(y.span.end)
- .saturating_sub(x.span.start.max(y.span.start));
- 2 * shared >= x.span.len() && 2 * shared >= y.span.len()
-}
-
-/// One judged pair per clone group. Pairs whose sites repeat the same code
-/// (mutual half overlap) are linked; the first pair of each group in rank
-/// order represents it and carries the other copies. Linking on plain overlap
-/// let short idioms inside a larger copy chain unrelated code together.
-fn representatives(pairs: Vec) -> Vec {
- let groups = same_code_groups(&pairs);
- let mut sites = BTreeMap::>::new();
- for (i, pair) in pairs.iter().enumerate() {
- let group = sites.entry(groups[i]).or_default();
- for site in [&pair.a, &pair.b] {
- if !group.iter().any(|known| overlaps(known, site)) {
- group.push(site.clone());
- }
- }
- }
- let mut kept = Vec::new();
- for (i, mut pair) in pairs.into_iter().enumerate() {
- if groups[i] != i {
- continue;
- }
- pair.copies = sites[&i]
- .iter()
- .filter(|s| !overlaps(s, &pair.a) && !overlaps(s, &pair.b))
- .cloned()
- .collect();
- pair.copies
- .sort_by(|x, y| (&x.path, x.start_line).cmp(&(&y.path, y.start_line)));
- pair.occurrences = 2 + pair.copies.len();
- kept.push(pair);
- }
- kept
-}
-
-/// For each pair, the first pair of its group: pairs whose sites repeat the
-/// same code are linked, transitively (union-find).
-fn same_code_groups(pairs: &[Pair]) -> Vec {
- fn root(parent: &mut [usize], mut i: usize) -> usize {
- while parent[i] != i {
- parent[i] = parent[parent[i]];
- i = parent[i];
- }
- i
- }
- let mut parent: Vec = (0..pairs.len()).collect();
- for i in 0..pairs.len() {
- for j in i + 1..pairs.len() {
- let (p, q) = (&pairs[i], &pairs[j]);
- let linked = [&p.a, &p.b]
- .iter()
- .any(|x| same_code(x, &q.a) || same_code(x, &q.b));
- if linked {
- let (ri, rj) = (root(&mut parent, i), root(&mut parent, j));
- parent[ri.max(rj)] = ri.min(rj);
- }
- }
- }
- (0..pairs.len()).map(|i| root(&mut parent, i)).collect()
-}
-
-fn contains(outer: &Site, inner: &Site) -> bool {
- outer.path == inner.path
- && outer.span.start <= inner.span.start
- && inner.span.end <= outer.span.end
-}
-
-fn compact(text: &str) -> usize {
- text.bytes().filter(|b| !b.is_ascii_whitespace()).count()
-}
-
-fn site(files: &[SourceFile<'_>], index: usize, span: Range) -> Site {
- let file = &files[index];
- let unit = file
- .units
- .iter()
- .filter(|u| u.callable() && u.span.start <= span.start && span.end <= u.span.end)
- .min_by_key(|u| u.span.len());
- Site {
- file: index,
- path: file.path.to_path_buf(),
- start_line: line_of(file.source, span.start),
- end_line: line_of(file.source, span.end.saturating_sub(1)),
- function: unit.map(|u| u.name.clone()),
- function_source: unit.map(|u| u.source(file.source).to_string()),
- quote: file.source[span.clone()].to_string(),
- span,
- }
-}
-
-/// Aligned tokens must match after normalization, and each identifier must map
-/// to exactly one identifier on the other side. Returns renamed names and values.
-fn align(x: &[Token<'_>], y: &[Token<'_>]) -> Option> {
- if x.len() != y.len() {
- return None;
- }
- let mut forward = BTreeMap::new();
- let mut backward = BTreeMap::new();
- let mut differences = Vec::new();
- for (a, b) in x.iter().zip(y) {
- if a.normal() != b.normal() {
- return None;
- }
- // Each side calling itself is the same step, not a rename.
- if a.own && b.own {
- continue;
- }
- if a.kind == TokenKind::Identifier
- && (*forward.entry(a.text).or_insert(b.text) != b.text
- || *backward.entry(b.text).or_insert(a.text) != a.text)
- {
- return None;
- }
- if a.kind != TokenKind::Other && a.text != b.text {
- let difference = Difference {
- a: a.text.to_string(),
- b: b.text.to_string(),
- };
- if !differences.contains(&difference) && differences.len() < DIFFERENCES {
- differences.push(difference);
- }
- }
- }
- Some(differences)
-}
-
-fn leaves<'a>(node: Node<'_>, source: &'a str, tokens: &mut Vec>) {
- if is_comment(node) {
- return;
- }
- let kind = node.kind();
- let literal = matches!(
- kind,
- "string_content"
- | "string_fragment"
- | "integer_literal"
- | "float_literal"
- | "char_literal"
- | "decimal_integer_literal"
- | "hex_integer_literal"
- | "octal_integer_literal"
- | "binary_integer_literal"
- | "decimal_floating_point_literal"
- | "hex_floating_point_literal"
- | "character_literal"
- | "number"
- | "integer"
- | "float"
- | "string_literal_content"
- | "raw_string_content"
- | "verbatim_string_literal"
- | "real_literal"
- );
- if node.child_count() == 0 || literal {
- let text = text(node, source);
- if text.trim().is_empty() {
- return;
- }
- let kind = if literal {
- TokenKind::Literal
- } else if kind.ends_with("identifier")
- || matches!(
- kind,
- "identifier" | "constant" | "instance_variable" | "name"
- )
- {
- TokenKind::Identifier
- } else {
- TokenKind::Other
- };
- tokens.push(Token {
- kind,
- text,
- start: node.start_byte(),
- own: false,
- });
- return;
- }
- let mut cursor = node.walk();
- for child in node.children(&mut cursor) {
- leaves(child, source, tokens);
- }
-}
-
-fn collect_blocks(
- node: Node<'_>,
- file: &SourceFile<'_>,
- index: usize,
- bodies: &[Range],
- tokens: &[Token<'_>],
- blocks: &mut Vec,
-) {
- if holds_statements(node)
- && bodies
- .iter()
- .any(|b| b.start <= node.start_byte() && node.end_byte() <= b.end)
- {
- let all = block_statements(node, file, tokens);
- // A Go body holds its statements in a `statement_list` inside the block.
- let body = node
- .parent()
- .filter(|p| node.kind() == "statement_list" && p.kind() == "block")
- .unwrap_or(node);
- let whole = all.iter().all(Option::is_some)
- && file
- .units
- .iter()
- .any(|u| u.callable() && u.body == Some(body.byte_range()));
- // Excluded statements break a window, so split the block there.
- for statements in all.split(Option::is_none) {
- let statements: Vec = statements.iter().flatten().cloned().collect();
- if !statements.is_empty() {
- blocks.push(Block {
- file: index,
- statements,
- whole,
- });
- }
- }
- }
- let mut cursor = node.walk();
- for child in node.named_children(&mut cursor) {
- collect_blocks(child, file, index, bodies, tokens, blocks);
- }
-}
-
-/// A node whose named children are statements. Ruby holds statements in a
-/// `body_statement` or `block_body`, and in the `then`, `else` and `do` of a
-/// branch or loop; its `block` is a `{ … }` argument around a `block_body`.
-/// PHP holds them in a `compound_statement`, and a Java constructor's
-/// statements are in a `constructor_body`.
-fn holds_statements(node: Node<'_>) -> bool {
- let ruby_block = node.kind() == "block" && node.parent().is_some_and(|p| p.kind() == "call");
- matches!(
- node.kind(),
- "block"
- | "statement_block"
- | "statement_list"
- | "compound_statement"
- | "body_statement"
- | "block_body"
- | "then"
- | "else"
- | "do"
- | "constructor_body"
- ) && !ruby_block
-}
-
-/// Objects a method calls itself on, as in `self.walk(`, `this.walk(`,
-/// `Self::walk(`, `cls.walk(` or PHP's `$this->walk(` and `static::walk(`.
-const RECEIVERS: [&str; 5] = ["self", "Self", "this", "cls", "static"];
-
-/// Mark each call of the function it sits in: its name followed by its
-/// arguments inside the body of the innermost callable of that name. A
-/// method calls itself on the object itself (`self.walk(`, `this.walk(`,
-/// `Self::walk(`); a bare `walk(` inside it calls a free or imported
-/// function, except in Java, C# and Ruby, where a bare call reaches the
-/// method through its object. A call through another path, as
-/// `native::get()` inside `get`, names a different function.
-fn mark_recursion(tokens: &mut [Token<'_>], units: &[Unit], path: &Path) {
- let implicit = matches!(
- path.extension().and_then(|x| x.to_str()),
- Some("java" | "cs" | "rb")
- );
- let callables: Vec<&Unit> = units.iter().filter(|u| u.callable()).collect();
- let names: BTreeSet<&str> = callables.iter().map(|u| u.short_name.as_str()).collect();
- for i in 1..tokens.len() {
- let token = &tokens[i - 1];
- if token.kind != TokenKind::Identifier
- || tokens[i].text != "("
- || !names.contains(token.text)
- {
- continue;
- }
- let Some(unit) = callables
- .iter()
- .filter(|u| u.body.as_ref().is_some_and(|b| b.contains(&token.start)))
- .min_by_key(|u| u.span.len())
- .filter(|u| u.short_name == token.text)
- else {
- continue;
- };
- let qualifier = i
- .checked_sub(2)
- .filter(|&q| matches!(tokens[q].text, "." | "::" | "->" | "?."));
- tokens[i - 1].own = match qualifier {
- None => implicit || unit.kind == Kind::Function,
- Some(q) => q
- .checked_sub(1)
- .is_some_and(|o| RECEIVERS.contains(&tokens[o].text)),
- };
- }
-}
-
-/// The tokens of one node.
-fn tokens_of<'t, 'a>(tokens: &'t [Token<'a>], node: Node<'_>) -> &'t [Token<'a>] {
- let start = tokens.partition_point(|t| t.start < node.start_byte());
- let end = tokens.partition_point(|t| t.start < node.end_byte());
- &tokens[start..end]
-}
-
-/// Part of a walk's frame rather than its work, if it is.
-fn frame(statement: Node<'_>, tokens: &[Token<'_>]) -> Option {
- if exit_guard(statement, tokens) {
- Some(Frame::Exit)
- } else if recursion(statement, tokens) {
- Some(Frame::Recursion)
- } else {
- None
- }
-}
-
-/// A call of the function itself and nothing else, as `bound(child, names);`
-/// or `return self.walk(node.parent)`, or a loop or branch whose statements
-/// only do that or leave early, with at least one call. Work anywhere else
-/// in its body, as in a match arm, a conditional expression or a call that
-/// wraps the recursion, makes it more than the frame.
-fn recursion(statement: Node<'_>, tokens: &[Token<'_>]) -> bool {
- if own_call(statement, tokens) {
- return true;
- }
- let node = expression(statement);
- let ruby_block = node.kind() == "call" && node.child_by_field_name("block").is_some();
- if !ruby_block
- && !matches!(
- node.kind(),
- "for_expression"
- | "for_statement"
- | "for_in_statement"
- | "enhanced_for_statement"
- | "foreach_statement"
- | "for"
- | "while_expression"
- | "while_statement"
- | "while"
- | "until"
- | "loop_expression"
- | "do_statement"
- | "if_expression"
- | "if_statement"
- | "if"
- | "unless"
- )
- {
- return false;
- }
- let mut body = Vec::new();
- branch_statements(node, &mut body);
- body.iter().any(|&s| recursion(s, tokens))
- && body
- .iter()
- .all(|&s| exits(s, tokens) || exit_guard(s, tokens) || recursion(s, tokens))
-}
-
-/// A statement that is only a call of the function it sits in, as
-/// `walk(child);`, `return self.walk(node.parent)`, `await this.walk(child);`
-/// or `walk(child)?;`.
-fn own_call(statement: Node<'_>, tokens: &[Token<'_>]) -> bool {
- let mut words = tokens_of(tokens, statement);
- if let [first, rest @ ..] = words
- && matches!(first.text, "return" | "await")
- {
- words = rest;
- }
- while let [rest @ .., last] = words
- && matches!(last.text, ";" | "?")
- {
- words = rest;
- }
- if let [receiver, separator, rest @ ..] = words
- && RECEIVERS.contains(&receiver.text)
- && matches!(separator.text, "." | "::" | "->" | "?.")
- {
- words = rest;
- }
- let [name, arguments @ ..] = words else {
- return false;
- };
- if !name.own || arguments.first().is_none_or(|t| t.text != "(") {
- return false;
- }
- // The call's closing parenthesis ends the statement.
- let mut depth = 0usize;
- for (i, token) in arguments.iter().enumerate() {
- match token.text {
- "(" => depth += 1,
- ")" => {
- depth -= 1;
- if depth == 0 {
- return i + 1 == arguments.len();
- }
- }
- _ => {}
- }
- }
- false
-}
-
-/// A branch that only leaves, as `if node.kind() == "call" { return; }`,
-/// `if (done) return;` or `return if done`.
-fn exit_guard(statement: Node<'_>, tokens: &[Token<'_>]) -> bool {
- let node = expression(statement);
- if !matches!(
- node.kind(),
- "if_expression" | "if_statement" | "if" | "unless" | "if_modifier" | "unless_modifier"
- ) {
- return false;
- }
- let mut body = Vec::new();
- branch_statements(node, &mut body);
- !body.is_empty()
- && body
- .into_iter()
- .all(|s| exits(s, tokens) || exit_guard(s, tokens))
-}
-
-/// The expression a Rust or JavaScript statement wraps, as the `for` loop
-/// of a Rust `expression_statement`.
-fn expression(statement: Node<'_>) -> Node<'_> {
- statement
- .named_child(0)
- .filter(|_| {
- statement.kind() == "expression_statement" && statement.named_child_count() == 1
- })
- .unwrap_or(statement)
-}
-
-/// The statements a loop or branch runs, in all its branches: the
-/// statements of its blocks, or the one statement of a branch without
-/// braces. Its header, as a condition or the collection a loop walks, is
-/// not a statement.
-fn branch_statements<'t>(node: Node<'t>, statements: &mut Vec>) {
- let before = statements.len();
- let mut cursor = node.walk();
- for field in ["body", "consequence", "alternative", "block"] {
- for part in node.children_by_field_name(field, &mut cursor) {
- statements_in(part, statements);
- }
- }
- // A JavaScript or Rust `else` holds its statement or block without a field.
- if statements.len() == before && node.kind() == "else_clause" {
- let mut cursor = node.walk();
- for part in node.named_children(&mut cursor) {
- statements_in(part, statements);
- }
- }
-}
-
-/// The statements of one part of a loop or branch.
-fn statements_in<'t>(part: Node<'t>, statements: &mut Vec>) {
- if is_comment(part) {
- return;
- }
- if holds_statements(part) {
- let mut cursor = part.walk();
- for child in part.named_children(&mut cursor) {
- statements_in_block(child, statements);
- }
- } else if matches!(
- part.kind(),
- "else_clause" | "elif_clause" | "else_if_clause" | "elsif" | "block" | "do_block"
- ) {
- // Further branches, and the `{ … }` or `do … end` a Ruby call runs.
- branch_statements(part, statements);
- } else {
- statements.push(part);
- }
-}
-
-/// One statement of a block; Go holds a block's statements in a list.
-fn statements_in_block<'t>(child: Node<'t>, statements: &mut Vec>) {
- if is_comment(child) {
- return;
- }
- if holds_statements(child) {
- statements_in(child, statements);
- } else {
- statements.push(child);
- }
-}
-
-/// `return`, `break`, `continue` or `next` with no value, or with nothing.
-fn exits(statement: Node<'_>, tokens: &[Token<'_>]) -> bool {
- match tokens_of(tokens, statement) {
- [first, rest @ ..] => {
- matches!(first.text, "return" | "break" | "continue" | "next")
- && rest
- .iter()
- .all(|t| matches!(t.text, ";" | "None" | "nil" | "null"))
- }
- [] => false,
- }
-}
-
-/// A block's statements with normalized-token hashes; `None` for excluded lines.
-fn block_statements(
- node: Node<'_>,
- file: &SourceFile<'_>,
- tokens: &[Token<'_>],
-) -> Vec
';\n}\n?>\n\n";
-
-#[test]
-fn a_php_page_script_is_one_unit_judged_by_every_security_rule() {
- let (project, options) = project_with(&[("page.php", PAGE)], &catalog::SECURITY);
- let (_, plan) = planned(&project, &options);
- let request = &plan.requests[0].request;
- let functions = request["state"]["functions"].as_array().unwrap();
- let names: Vec<&str> = functions
- .iter()
- .filter_map(|f| f["name"].as_str())
- .collect();
- assert_eq!(names, ["escape_html", "top-level code"]);
- let script = functions[1]["source"].as_str().unwrap();
- assert!(script.starts_with("require_once 'lib.php';\nif (isset($_GET['id']))"));
- assert!(!script.contains("function escape_html") && !script.contains("