diff --git a/CHANGELOG.md b/CHANGELOG.md index d825142..e37dfce 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -7,7 +7,29 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0 ## [Unreleased] +## [0.2.0] - 2026-07-08 + +This release adds a static code-analysis toolkit on top of the tokenizer: +diff/PR analysis, complexity metrics, n-gram "naturalness", and clone/plagiarism +similarity — plus a comprehensive per-language test suite that hardened comment +handling across the board. + ### Added +- **Fingerprinting & similarity** (`PyReprism.fingerprints`): winnowing k-gram + fingerprints over the normalized token stream for clone/plagiarism detection — + `fingerprint()`, `similarity()`/`containment()` (rename-invariant by default), + and a `FingerprintIndex` for many-to-many detection over a corpus. CLI: + `pyreprism similarity a b` and `pyreprism clones DIR --threshold`. +- **N-gram analysis & code naturalness** (`PyReprism.ngrams`): token/type n-gram + extraction and frequency counts, plus an `NgramModel` (add-k smoothing, + save/load) that measures cross-entropy / perplexity against a trained corpus + ("naturalness of software"). CLI: `pyreprism ngrams` and + `pyreprism perplexity --train`. +- **Complexity metrics** computed from the token stream: `halstead()` + (volume/difficulty/effort/bugs), `cyclomatic_complexity()` (approximate McCabe), + `maintainability_index()` (0–100), `max_nesting_depth()`, and `code_metrics()` + which bundles them with the line/token stats. Exposed on the CLI via + `pyreprism stats --full`. - **Diff processing** (`PyReprism.diffs`): parse unified/`git` diffs and analyze the changed code per file in its own language. Includes churn metrics (`diff_stats`: added/removed split into code, comment and blank), cosmetic @@ -17,6 +39,23 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0 `old_source` enables accurate full-file classification. New CLI command `pyreprism diff` (`--json` / `--csv` / `--per-file` / `--cosmetic`). +### Changed +- Adopted the standard `src/` layout with tests at the top level, consolidated + all dependencies into `pyproject.toml` extras, and rewrote the landing page. +- Single-sourced the version from the package and automated releases: pushing a + `vX.Y.Z` tag now publishes to PyPI (Trusted Publishing) and creates a GitHub + Release. Added Python 3.13 to the CI matrix. +- Added a comprehensive, data-driven per-language comment test suite that + requires every registered language to be covered. + +### Fixed +- Fixed comment stripping in 8 more languages surfaced by the new test suite: + `smalltalk` (kept the comment and deleted the code) and + `eiffel`/`bro`/`coffeescript`/`io`/`nix`/`gherkin`/`gedcom` (trailing comments + not removed and newlines dropped). +- Fixed a CLI argument-ordering incompatibility on Python ≤ 3.11 (options must + follow positionals). + ## [0.1.0] - 2026-07-07 This release turns PyReprism from a comment-removal helper into a full @@ -73,6 +112,7 @@ ML-oriented normalization, an optional accurate backend, and batch processing. - Early beta releases: comment removal for an initial set of languages and the `Normalizer` whitespace helper. -[Unreleased]: https://github.com/unlv-evol/PyReprism/compare/v0.1.0...HEAD +[Unreleased]: https://github.com/unlv-evol/PyReprism/compare/v0.2.0...HEAD +[0.2.0]: https://github.com/unlv-evol/PyReprism/compare/v0.1.0...v0.2.0 [0.1.0]: https://github.com/unlv-evol/PyReprism/releases/tag/v0.1.0 [0.0.4]: https://github.com/unlv-evol/PyReprism/releases/tag/v0.0.4 diff --git a/README.md b/README.md index ee861f8..526a01a 100644 --- a/README.md +++ b/README.md @@ -87,6 +87,56 @@ s.comment_to_code_ratio, s.comment_density s.as_dict() # includes per-token-type counts, ready for JSON / dataframes ``` +Complexity metrics (computed from the token stream): + +```python +pr.halstead(source, lang="python").volume # Halstead volume/difficulty/effort/bugs +pr.cyclomatic_complexity(source, lang="python") # approximate McCabe complexity +pr.maintainability_index(source, lang="python") # 0–100 (higher is better) +pr.code_metrics(source, lang="python") # everything above in one dict +``` + +On the CLI: `pyreprism stats --full file.py` (add `--json` for machine output). + +### Similarity & clone / plagiarism detection + +Winnowing k-gram fingerprints over the *normalized* token stream, so matches +survive variable renaming and literal changes (Type-2 clones): + +```python +from PyReprism import fingerprints as fp + +fp.similarity(code_a, code_b, "python") # 0.0–1.0 (Jaccard of fingerprints) +fp.containment(code_a, code_b, "python") # how much of A appears in B + +index = fp.FingerprintIndex() # many-to-many clone detection +index.add_paths("submissions/") +index.similar_pairs(threshold=0.7) # -> [(file_a, file_b, score), ...] +``` + +On the CLI: `pyreprism similarity a.py b.py` and +`pyreprism clones submissions/ --threshold 0.7`. + +### N-grams & code "naturalness" + +Token n-grams (over token text or, structurally, over token *types*) and an +n-gram language model that measures how predictable/"natural" code is +(Hindle et al.): + +```python +from PyReprism import ngrams + +ngrams.ngram_counts(source, "python", n=3).most_common(10) +ngrams.ngrams(source, "python", n=2, types=True) # structural n-grams + +model = ngrams.train("corpus/", n=3) # train on a code corpus +model.perplexity(ngrams.token_sequence(source, "python")) # lower = more natural +model.save("model.json") +``` + +On the CLI: `pyreprism ngrams file.py -n 3 --top 20` and +`pyreprism perplexity --train corpus/ file.py`. + ### Normalization for ML / clone detection Canonicalize code so that only its structure remains — rename identifiers to diff --git a/docs/fingerprints.rst b/docs/fingerprints.rst new file mode 100644 index 0000000..bb122b2 --- /dev/null +++ b/docs/fingerprints.rst @@ -0,0 +1,27 @@ +.. _fingerprints_toplevel: + +========================================= +Fingerprinting & similarity +========================================= + +The :mod:`PyReprism.fingerprints` module detects clones and plagiarism using +**winnowing** k-gram fingerprints (Schleimer, Wilkerson & Aiken) over the +*normalized* token stream, so matches are robust to variable renaming and +literal changes (Type-2 clones):: + + from PyReprism import fingerprints as fp + + fp.similarity(code_a, code_b, "python") # Jaccard of fingerprints, 0-1 + fp.containment(code_a, code_b, "python") # fraction of A found in B + + index = fp.FingerprintIndex() + index.add_paths("submissions/") + index.similar_pairs(threshold=0.7) + +Fingerprints are deterministic (CRC32-hashed) and comparable across runs. + +.. automodule:: PyReprism.fingerprints + :members: + :undoc-members: + :show-inheritance: + :exclude-members: __dict__, __weakref__ diff --git a/docs/index.rst b/docs/index.rst index caf7b99..453d6e0 100644 --- a/docs/index.rst +++ b/docs/index.rst @@ -11,6 +11,8 @@ PyReprism documentation! cli tokens metrics + ngrams + fingerprints engines batch diffs diff --git a/docs/intro.rst b/docs/intro.rst index c055ea4..189fff2 100644 --- a/docs/intro.rst +++ b/docs/intro.rst @@ -1,53 +1,100 @@ .. _intro_toplevel: -================== -Overview / Install -================== - -PyReprism is a Python framework that helps researchers and developers the task of source code preprocessing. With PyReprism, you can easily match, extract, count, and remove comments, whitespaces, operators, numbers and other language specific constructs from over 150 programming languages and file extensions. - +======== +Overview +======== + +**PyReprism** is a Python framework for source-code preprocessing. It lets you +**match, extract, count, and remove** comments, strings, numbers, operators, +keywords and other language-specific constructs across **145+ programming +languages and file formats** — through a small high-level API, a command-line +tool, code metrics, ML-oriented normalization, and diff/pull-request analysis. + +Highlights +========== + +* One consistent API for every language: ``remove_*`` / ``extract_*`` / + ``count_*`` / ``match_*`` for each construct, plus a lossless ``tokenize()``. +* Code metrics (``stats``) and ML canonicalization (``normalize``) for + clone/plagiarism detection and code-embedding pipelines. +* Optional high-accuracy `Pygments`_ backend — with **zero required + dependencies** by default. +* Batch/corpus analysis and unified/``git`` diff processing. +* A ``pyreprism`` command-line tool (also ``python -m PyReprism``). Requirements ============ -* `Python`_ 3.6 or newer -* `Git`_ +* `Python`_ 3.8 or newer -.. _Python: https://www.python.org -.. _Git: https://git-scm.com/ +Install +======= -Installing PyReprism -==================== +.. code-block:: shell -Installing PyReprism is easily done using `pip`_. Assuming it is installed, just run the following from the command-line: + pip install PyReprism -.. _pip: https://pip.pypa.io/en/latest/installing.html +For the higher-accuracy tokenizer backend: -.. sourcecode:: none +.. code-block:: shell - $ pip install PyReprism + pip install "PyReprism[accurate]" -Source Code +Quick start =========== -PyReprism's git repo is available on GitHub, which can be browsed at: +.. code-block:: python + + import PyReprism as pr + + source = """ + # a comment + x = 5 + 6 + print(x) # inline + """ + + pr.remove_comments(source, lang="python") # code without comments + pr.extract_comments(source, lang="python") # ['# a comment', '# inline'] + pr.count_comments(source, lang="python") # 2 + +``lang`` accepts a language name (``"python"``), a file extension (``".py"``), +or a language class. Canonicalize code for machine learning, or measure it: + +.. code-block:: python - * https://github.com/unlv-evol/PyReprism.git + pr.normalize("total = price * 42", lang="python") # 'VAR1 = VAR2 * 0' + pr.stats(source, lang="python").comment_to_code_ratio -and cloned using:: +Command line +============ + +.. code-block:: shell + + pyreprism remove comments file.py + pyreprism scan myproject/ --csv # aggregate code metrics over a tree + git diff | pyreprism diff --cosmetic # flag comment/whitespace-only changes + +Documentation +============= - $ git clone https://github.com/unlv-evol/PyReprism.git - $ cd PyReprism +Full documentation, including the API reference and the list of supported +languages, is available at https://pyreprism.readthedocs.io. -Optionally (but suggested), make use of virtual environment. Therefore, before installing the requirements run:: - - $ python3 -m venv venv - $ source venv/bin/activate +Source code +=========== + +PyReprism is developed on GitHub at +https://github.com/unlv-evol/PyReprism. To work on it locally: -Install the requirements:: - - $ pip install -r requirements.txt +.. code-block:: shell -and run the tests using pytest:: + git clone https://github.com/unlv-evol/PyReprism.git + cd PyReprism + python3 -m venv venv && source venv/bin/activate + pip install -e ".[dev]" + pytest - $ pytest +See ``CONTRIBUTING.md`` for contribution guidelines. + +.. _Python: https://www.python.org +.. _Pygments: https://pygments.org diff --git a/docs/ngrams.rst b/docs/ngrams.rst new file mode 100644 index 0000000..20f1970 --- /dev/null +++ b/docs/ngrams.rst @@ -0,0 +1,28 @@ +.. _ngrams_toplevel: + +============================== +N-grams & code naturalness +============================== + +The :mod:`PyReprism.ngrams` module extracts token n-grams and provides an +n-gram language model for measuring code *naturalness* (cross-entropy / +perplexity), following Hindle et al., "On the Naturalness of Software". + +n-grams can be taken over token **text** or over token **types** (structural, +AST-free), which is useful for clone detection and style analysis:: + + from PyReprism import ngrams + + ngrams.ngram_counts(source, "python", n=3).most_common(10) + ngrams.ngrams(source, "python", n=2, types=True) + + model = ngrams.train("corpus/", n=3) + seq = ngrams.token_sequence(source, "python") + model.perplexity(seq) # lower = more predictable / "natural" + model.save("model.json") + +.. automodule:: PyReprism.ngrams + :members: + :undoc-members: + :show-inheritance: + :exclude-members: __dict__, __weakref__ diff --git a/src/PyReprism/__init__.py b/src/PyReprism/__init__.py index 07c8260..ef1b4bc 100644 --- a/src/PyReprism/__init__.py +++ b/src/PyReprism/__init__.py @@ -17,11 +17,11 @@ import os from typing import List, Optional, Sequence, Type, Union -from .metrics import CodeStats +from .metrics import CodeStats, Halstead from .tokens import Token, TokenType from .utils.normalizer import Normalizer -__version__ = "0.1.0" +__version__ = "0.2.0" LanguageLike = Union[str, "type"] @@ -274,6 +274,46 @@ def stats(source: str, lang: LanguageLike, engine: str = 'regex') -> CodeStats: return _tokenops.stats(_engine_tokens(source, lang, engine), source) +def halstead(source: str, lang: LanguageLike, engine: str = 'regex') -> Halstead: + """Return the :class:`Halstead` complexity measures for ``source``.""" + if _is_regex(engine): + return _resolve(lang).halstead(source) + from . import _tokenops + return _tokenops.halstead(_engine_tokens(source, lang, engine)) + + +def cyclomatic_complexity(source: str, lang: LanguageLike, engine: str = 'regex') -> int: + """Approximate McCabe cyclomatic complexity (token-based).""" + if _is_regex(engine): + return _resolve(lang).cyclomatic_complexity(source) + from . import _tokenops + return _tokenops.cyclomatic(_engine_tokens(source, lang, engine)) + + +def maintainability_index(source: str, lang: LanguageLike, engine: str = 'regex') -> float: + """SEI-normalized Maintainability Index in ``[0, 100]`` (higher is better).""" + if _is_regex(engine): + return _resolve(lang).maintainability_index(source) + from . import _tokenops + tokens = _engine_tokens(source, lang, engine) + return _tokenops.maintainability_index(tokens, _tokenops.stats(tokens, source).code_lines) + + +def code_metrics(source: str, lang: LanguageLike, engine: str = 'regex') -> dict: + """Return a combined metrics dict (line stats + Halstead + complexity + MI).""" + if _is_regex(engine): + return _resolve(lang).code_metrics(source) + from . import _tokenops + tokens = _engine_tokens(source, lang, engine) + stats_obj = _tokenops.stats(tokens, source) + data = stats_obj.as_dict() + data['halstead'] = _tokenops.halstead(tokens).as_dict() + data['cyclomatic_complexity'] = _tokenops.cyclomatic(tokens) + data['max_nesting_depth'] = _tokenops.max_nesting_depth(tokens) + data['maintainability_index'] = _tokenops.maintainability_index(tokens, stats_obj.code_lines) + return data + + def normalize(source: str, lang: LanguageLike, engine: str = 'regex', **options) -> str: """Return a canonicalized form of ``source`` for ML / clone detection. @@ -329,7 +369,7 @@ def preprocess(source: str, lang: LanguageLike, steps: Sequence[str] = ('comment __all__ = [ '__version__', - 'Token', 'TokenType', 'CodeStats', 'Normalizer', + 'Token', 'TokenType', 'CodeStats', 'Halstead', 'Normalizer', 'get_language', 'detect_language', 'remove_comments', 'extract_comments', 'count_comments', 'match_comments', 'remove_keywords', 'extract_keywords', 'count_keywords', @@ -338,5 +378,6 @@ def preprocess(source: str, lang: LanguageLike, steps: Sequence[str] = ('comment 'remove_strings', 'extract_strings', 'extract_identifiers', 'remove_whitespaces', 'blank_comments', 'stats', 'normalize', + 'halstead', 'cyclomatic_complexity', 'maintainability_index', 'code_metrics', 'tokenize', 'preprocess', ] diff --git a/src/PyReprism/_tokenops.py b/src/PyReprism/_tokenops.py index ccee1d5..d4c0a54 100644 --- a/src/PyReprism/_tokenops.py +++ b/src/PyReprism/_tokenops.py @@ -4,11 +4,24 @@ regex and pygments backends share one implementation of remove/extract/count, normalization and metrics. """ -from typing import List +import math +from collections import Counter +from typing import List, Set -from .metrics import CodeStats +from .metrics import CodeStats, Halstead from .tokens import Token, TokenType +# Language-agnostic decision points used for the approximate cyclomatic complexity. +DECISION_KEYWORDS = frozenset({ + 'if', 'elif', 'elseif', 'for', 'foreach', 'while', 'case', 'when', 'catch', + 'except', 'and', 'or', 'unless', +}) +# Substrings within operator tokens that add a decision (&&, ||, ternary ?). +DECISION_OPERATORS = ('&&', '||', '?') + +_OPERAND_TYPES = (TokenType.IDENTIFIER, TokenType.NUMBER, TokenType.STRING) +_OPEN_BRACKETS, _CLOSE_BRACKETS = set('([{'), set(')]}') + def remove(tokens: List[Token], ttype: TokenType) -> str: return ''.join(t.value for t in tokens if t.type is not ttype) @@ -76,3 +89,65 @@ def stats(tokens: List[Token], source: str) -> CodeStats: identifier_tokens=counts[TokenType.IDENTIFIER], operator_tokens=counts[TokenType.OPERATOR], ) + + +def halstead(tokens: List[Token]) -> Halstead: + """Compute :class:`~PyReprism.metrics.Halstead` measures from ``tokens``.""" + operators: Counter = Counter() + operands: Counter = Counter() + for tok in tokens: + if tok.type in (TokenType.OPERATOR, TokenType.KEYWORD): + operators[tok.value] += 1 + elif tok.type is TokenType.OTHER and tok.value.strip(): + operators[tok.value] += 1 + elif tok.type in _OPERAND_TYPES: + operands[tok.value] += 1 + return Halstead( + distinct_operators=len(operators), distinct_operands=len(operands), + total_operators=sum(operators.values()), total_operands=sum(operands.values()), + ) + + +def cyclomatic(tokens: List[Token], decision_keywords: Set[str] = DECISION_KEYWORDS) -> int: + """Approximate McCabe cyclomatic complexity: ``1 + decision points``. + + Counts branch keywords plus ``&&``/``||``/``?`` operators. This is a + token-level approximation; an exact value needs a control-flow graph. + """ + complexity = 1 + for tok in tokens: + # Branch words are usually KEYWORD, but word-operators (and/or) can land + # as IDENTIFIER in the generic lexer, so check both. + if (tok.type in (TokenType.KEYWORD, TokenType.IDENTIFIER) + and tok.value.lower() in decision_keywords): + complexity += 1 + elif tok.type is TokenType.OPERATOR: + for op in DECISION_OPERATORS: + complexity += tok.value.count(op) + return complexity + + +def max_nesting_depth(tokens: List[Token]) -> int: + """Maximum bracket nesting depth (``()``/``[]``/``{}``), ignoring strings/comments.""" + depth = deepest = 0 + for tok in tokens: + if tok.type in (TokenType.STRING, TokenType.COMMENT): + continue + for ch in tok.value: + if ch in _OPEN_BRACKETS: + depth += 1 + deepest = max(deepest, depth) + elif ch in _CLOSE_BRACKETS: + depth = max(0, depth - 1) + return deepest + + +def maintainability_index(tokens: List[Token], code_lines: int) -> float: + """SEI-normalized Maintainability Index in ``[0, 100]`` (higher is better).""" + volume = halstead(tokens).volume + complexity = cyclomatic(tokens) + raw = (171 + - 5.2 * math.log(max(volume, 1)) + - 0.23 * complexity + - 16.2 * math.log(max(code_lines, 1))) + return round(max(0.0, min(100.0, raw * 100 / 171)), 2) diff --git a/src/PyReprism/cli.py b/src/PyReprism/cli.py index 60ea14e..155b9eb 100644 --- a/src/PyReprism/cli.py +++ b/src/PyReprism/cli.py @@ -153,10 +153,13 @@ def _cmd_tokenize(args) -> int: def _cmd_stats(args) -> int: + from . import code_metrics multiple = _multiple(args.paths) for label, text, filename, base in _iter_inputs(args.paths): cls = _resolve(args.lang, filename, text) - if args.engine in (None, 'regex'): + if args.full: + data = code_metrics(text, lang=cls, engine=args.engine) + elif args.engine in (None, 'regex'): data = cls.stats(text).as_dict() else: data = _tokenops.stats(get_engine(args.engine).tokenize(text, cls), text).as_dict() @@ -165,12 +168,25 @@ def _cmd_stats(args) -> int: else: if multiple: print(f"==> {label} <==") - width = max(len(k) for k in data) - for key, value in data.items(): + flat = _flatten(data) + width = max(len(k) for k in flat) + for key, value in flat.items(): print(f"{key.ljust(width)} {value}") return 0 +def _flatten(data: dict, prefix: str = '') -> dict: + """Flatten one level of nested dicts for text output (e.g. halstead.volume).""" + out = {} + for key, value in data.items(): + if isinstance(value, dict): + for sub, subval in value.items(): + out[f"{prefix}{key}.{sub}"] = subval + else: + out[f"{prefix}{key}"] = value + return out + + def _cmd_normalize(args) -> int: options = dict( drop_comments=not args.keep_comments, @@ -264,6 +280,58 @@ def _cmd_diff(args) -> int: return 0 +def _cmd_ngrams(args) -> int: + from .ngrams import ngram_counts + multiple = _multiple(args.paths) + for label, text, filename, base in _iter_inputs(args.paths): + cls = _resolve(args.lang, filename, text) + counts = ngram_counts(text, cls, n=args.n, types=args.types) + if multiple: + print(f"==> {label} <==") + if args.json: + print(json.dumps([[list(g), c] for g, c in counts.most_common(args.top)])) + else: + for gram, count in counts.most_common(args.top): + print(f"{count}\t{' '.join(gram)}") + return 0 + + +def _cmd_perplexity(args) -> int: + from .ngrams import token_sequence, train + model = train(args.train, n=args.n, types=args.types) + for label, text, filename, base in _iter_inputs(args.paths): + cls = _resolve(args.lang, filename, text) + seq = token_sequence(text, cls, types=args.types) + print(f"{model.perplexity(seq):.4f}\t{label}") + return 0 + + +def _cmd_similarity(args) -> int: + from .fingerprints import similarity + with open(args.a, 'r', encoding='utf-8', errors='replace') as handle: + text_a = handle.read() + with open(args.b, 'r', encoding='utf-8', errors='replace') as handle: + text_b = handle.read() + lang = _resolve(args.lang, args.a, text_a) + score = similarity(text_a, text_b, lang, k=args.k, w=args.w, + normalize=not args.no_normalize, types=args.types) + print(f"{score:.4f}") + return 0 + + +def _cmd_clones(args) -> int: + from .fingerprints import FingerprintIndex + index = FingerprintIndex(k=args.k, w=args.w, normalize=not args.no_normalize, + types=args.types) + index.add_paths(args.paths) + pairs = index.similar_pairs(args.threshold) + for first, second, score in pairs: + print(f"{score:.4f}\t{first}\t{second}") + if not pairs: + print(f"no file pairs at or above similarity {args.threshold}", file=sys.stderr) + return 0 + + def _cmd_languages(args) -> int: from .languages import _load_all_languages from .languages.registry import LanguageRegistry @@ -365,6 +433,9 @@ def add_output(p): p_stats = sub.add_parser('stats', help='report line/token metrics') add_common(p_stats) p_stats.add_argument('--json', action='store_true', help='emit JSON') + p_stats.add_argument('--full', action='store_true', + help='include Halstead, cyclomatic complexity, nesting and ' + 'maintainability index') p_stats.set_defaults(func=_cmd_stats) p_norm = sub.add_parser('normalize', @@ -407,6 +478,49 @@ def add_output(p): help='list only files whose change is comment/whitespace-only') p_diff.set_defaults(func=_cmd_diff) + p_ngrams = sub.add_parser('ngrams', help='list the most common token n-grams') + add_common(p_ngrams) + p_ngrams.add_argument('-n', type=int, default=3, help='n-gram size (default: 3)') + p_ngrams.add_argument('--top', type=int, default=20, help='show the top K (default: 20)') + p_ngrams.add_argument('--types', action='store_true', + help='n-grams over token types (structural), not text') + p_ngrams.add_argument('--json', action='store_true', help='emit JSON') + p_ngrams.set_defaults(func=_cmd_ngrams) + + p_pp = sub.add_parser('perplexity', + help='score code "naturalness" against a trained corpus') + p_pp.add_argument('--train', required=True, metavar='PATH', + help='corpus directory/files to train the n-gram model on') + add_common(p_pp) + p_pp.add_argument('-n', type=int, default=3, help='n-gram size (default: 3)') + p_pp.add_argument('--types', action='store_true', + help='model token types (structural) instead of text') + p_pp.set_defaults(func=_cmd_perplexity) + + def add_fingerprint_opts(p): + p.add_argument('-k', type=int, default=5, help='k-gram size (default: 5)') + p.add_argument('-w', type=int, default=4, help='winnowing window (default: 4)') + p.add_argument('--no-normalize', action='store_true', + help='fingerprint raw tokens (not rename-invariant)') + p.add_argument('--types', action='store_true', + help='fingerprint token types only (fully structural)') + + p_sim = sub.add_parser('similarity', + help='fingerprint similarity (0-1) between two files') + p_sim.add_argument('a') + p_sim.add_argument('b') + p_sim.add_argument('-l', '--lang', help='language (auto-detected from the first file)') + add_fingerprint_opts(p_sim) + p_sim.set_defaults(func=_cmd_similarity) + + p_clones = sub.add_parser('clones', + help='find similar/duplicate files in a tree (clone/plagiarism)') + p_clones.add_argument('paths', nargs='+', help='directories or files') + p_clones.add_argument('--threshold', type=float, default=0.6, + help='minimum similarity to report (default: 0.6)') + add_fingerprint_opts(p_clones) + p_clones.set_defaults(func=_cmd_clones) + p_langs = sub.add_parser('languages', help='list supported languages and extensions') p_langs.set_defaults(func=_cmd_languages) diff --git a/src/PyReprism/fingerprints.py b/src/PyReprism/fingerprints.py new file mode 100644 index 0000000..851cf78 --- /dev/null +++ b/src/PyReprism/fingerprints.py @@ -0,0 +1,145 @@ +"""Code fingerprinting and similarity for clone / plagiarism detection. + +Uses **winnowing** (Schleimer, Wilkerson & Aiken) over k-grams of a *normalized* +token stream, so matches are robust to renaming and literal changes (Type-2 +clones). Hashes are deterministic (CRC32) so fingerprints are stable and +persistable across runs. +""" +import zlib +from collections import defaultdict +from typing import Dict, List, Sequence, Set, Tuple + +from .ngrams import ngrams_of +from .tokens import TokenType + +_SKIP = {TokenType.WHITESPACE, TokenType.COMMENT} + + +def normalized_tokens(source: str, lang, *, normalize: bool = True, + types: bool = False) -> List[str]: + """Reduce ``source`` to a token sequence for fingerprinting. + + With ``types`` the token kind is emitted (fully structural). With + ``normalize`` (default) identifiers/numbers/strings collapse to ``ID``/ + ``NUM``/``STR`` while keywords, operators and punctuation stay literal — so + renaming variables or changing literals does not change the fingerprint. + """ + from . import get_language + cls = lang if isinstance(lang, type) else get_language(lang) + placeholders = {TokenType.IDENTIFIER: 'ID', TokenType.NUMBER: 'NUM', TokenType.STRING: 'STR'} + out = [] + for tok in cls.tokenize(source): + if tok.type in _SKIP: + continue + if types: + out.append(tok.type.value) + elif normalize and tok.type in placeholders: + out.append(placeholders[tok.type]) + else: + out.append(tok.value) + return out + + +def _winnow(hashes: Sequence[int], w: int) -> Set[int]: + """Return the winnowing fingerprint set (selected minimum hashes).""" + if not hashes: + return set() + if len(hashes) < w: + return {min(hashes)} + fingerprints: Set[int] = set() + last = -1 + for i in range(len(hashes) - w + 1): + window = hashes[i:i + w] + local_min = min(window) + # rightmost occurrence of the minimum in the window (per the paper) + j = i + max(idx for idx, v in enumerate(window) if v == local_min) + if j != last: + fingerprints.add(hashes[j]) + last = j + return fingerprints + + +def fingerprint(source: str, lang, *, k: int = 5, w: int = 4, + normalize: bool = True, types: bool = False) -> Set[int]: + """Return the winnowing fingerprint (set of hashes) of ``source``. + + ``k`` is the k-gram (shingle) size, ``w`` the winnowing window. + """ + seq = normalized_tokens(source, lang, normalize=normalize, types=types) + if not seq: + return set() + grams = ngrams_of(seq, k) if len(seq) >= k else [tuple(seq)] + hashes = [zlib.crc32(' '.join(g).encode('utf-8')) for g in grams] + return _winnow(hashes, w) + + +def jaccard(a: Set[int], b: Set[int]) -> float: + """Jaccard similarity of two fingerprint sets, in ``[0, 1]``.""" + if not a and not b: + return 1.0 + if not a or not b: + return 0.0 + return len(a & b) / len(a | b) + + +def similarity(a: str, b: str, lang, **options) -> float: + """Similarity (Jaccard of fingerprints) between two sources, in ``[0, 1]``.""" + return jaccard(fingerprint(a, lang, **options), fingerprint(b, lang, **options)) + + +def containment(a: str, b: str, lang, **options) -> float: + """Fraction of ``a``'s fingerprints also present in ``b`` (asymmetric).""" + fa = fingerprint(a, lang, **options) + if not fa: + return 0.0 + return len(fa & fingerprint(b, lang, **options)) / len(fa) + + +class FingerprintIndex: + """An inverted index of fingerprints for many-to-many clone detection.""" + + def __init__(self, *, k: int = 5, w: int = 4, normalize: bool = True, types: bool = False): + self.options = dict(k=k, w=w, normalize=normalize, types=types) + self.prints: Dict[str, Set[int]] = {} + self._inverted: Dict[int, Set[str]] = defaultdict(set) + + def add(self, name: str, source: str, lang) -> Set[int]: + fp = fingerprint(source, lang, **self.options) + self.prints[name] = fp + for h in fp: + self._inverted[h].add(name) + return fp + + def add_paths(self, paths, *, recursive: bool = True, include=None, exclude=None) -> None: + """Fingerprint every supported file under ``paths``.""" + from .batch import iter_source_files + for _base, path, cls in iter_source_files(paths, recursive=recursive, + include=include, exclude=exclude): + try: + self.add(str(path), path.read_text(encoding='utf-8', errors='replace'), cls) + except Exception: # pragma: no cover - defensive I/O guard + continue + + def matches(self, name: str, threshold: float = 0.0) -> List[Tuple[str, float]]: + """Return ``(other, score)`` for entries similar to ``name``, best first.""" + target = self.prints[name] + candidates: Set[str] = set() + for h in target: + candidates |= self._inverted[h] + candidates.discard(name) + scored = [(other, jaccard(target, self.prints[other])) for other in candidates] + scored = [pair for pair in scored if pair[1] >= threshold] + return sorted(scored, key=lambda pair: -pair[1]) + + def similar_pairs(self, threshold: float = 0.6) -> List[Tuple[str, str, float]]: + """Return unique ``(a, b, score)`` pairs at or above ``threshold``.""" + seen: Set[Tuple[str, str]] = set() + pairs: List[Tuple[str, str, float]] = [] + for name in self.prints: + for other, score in self.matches(name, threshold): + key = tuple(sorted((name, other))) + if key in seen: + continue + seen.add(key) + pairs.append((key[0], key[1], score)) + return sorted(pairs, key=lambda item: -item[2]) diff --git a/src/PyReprism/languages/base.py b/src/PyReprism/languages/base.py index 3bdd046..96e5f44 100644 --- a/src/PyReprism/languages/base.py +++ b/src/PyReprism/languages/base.py @@ -306,6 +306,44 @@ def stats(cls, source: str) -> CodeStats: from .. import _tokenops return _tokenops.stats(cls.tokenize(source), source) + @classmethod + def halstead(cls, source: str): + """Return the :class:`~PyReprism.metrics.Halstead` measures for ``source``.""" + from .. import _tokenops + return _tokenops.halstead(cls.tokenize(source)) + + @classmethod + def cyclomatic_complexity(cls, source: str) -> int: + """Approximate McCabe cyclomatic complexity (token-based).""" + from .. import _tokenops + return _tokenops.cyclomatic(cls.tokenize(source)) + + @classmethod + def max_nesting_depth(cls, source: str) -> int: + """Maximum bracket nesting depth.""" + from .. import _tokenops + return _tokenops.max_nesting_depth(cls.tokenize(source)) + + @classmethod + def maintainability_index(cls, source: str) -> float: + """SEI-normalized Maintainability Index in ``[0, 100]`` (higher is better).""" + from .. import _tokenops + tokens = cls.tokenize(source) + return _tokenops.maintainability_index(tokens, _tokenops.stats(tokens, source).code_lines) + + @classmethod + def code_metrics(cls, source: str) -> dict: + """Return a combined metrics dict (line stats + Halstead + complexity + MI).""" + from .. import _tokenops + tokens = cls.tokenize(source) + stats = _tokenops.stats(tokens, source) + data = stats.as_dict() + data['halstead'] = _tokenops.halstead(tokens).as_dict() + data['cyclomatic_complexity'] = _tokenops.cyclomatic(tokens) + data['max_nesting_depth'] = _tokenops.max_nesting_depth(tokens) + data['maintainability_index'] = _tokenops.maintainability_index(tokens, stats.code_lines) + return data + # ------------------------------------------------------------------ normalize @classmethod def normalize(cls, source: str, **options) -> str: diff --git a/src/PyReprism/languages/bro.py b/src/PyReprism/languages/bro.py index de9690c..ae87d43 100644 --- a/src/PyReprism/languages/bro.py +++ b/src/PyReprism/languages/bro.py @@ -21,7 +21,7 @@ def keywords(cls) -> list: @classmethod def comment_regex(cls): - return re.compile(r'(?P#.*?$)|(?P[^#\n].*?$)', re.MULTILINE) + return re.compile(r'(?P#.*?$)|(?P.[^#]*)', re.DOTALL | re.MULTILINE) @classmethod def number_regex(cls): diff --git a/src/PyReprism/languages/coffeescript.py b/src/PyReprism/languages/coffeescript.py index f66ac35..c1d08be 100644 --- a/src/PyReprism/languages/coffeescript.py +++ b/src/PyReprism/languages/coffeescript.py @@ -17,7 +17,7 @@ def keywords(cls) -> list: @classmethod def comment_regex(cls): - return re.compile(r'(?P#.*?$|###.*?###)|(?P[^#\n].*?$)', re.DOTALL | re.MULTILINE) + return re.compile(r'(?P###[\s\S]*?###|#.*?$)|(?P.[^#]*)', re.DOTALL | re.MULTILINE) @classmethod def number_regex(cls): diff --git a/src/PyReprism/languages/eiffel.py b/src/PyReprism/languages/eiffel.py index de4e58a..f02d59d 100644 --- a/src/PyReprism/languages/eiffel.py +++ b/src/PyReprism/languages/eiffel.py @@ -17,8 +17,9 @@ def keywords(cls) -> list: @classmethod def comment_regex(cls): - # Eiffel uses -- for line comments and { ... } block comments; conservative pattern - pattern = re.compile(r'(?P--.*?$|\{[\s\S]*?\})|(?P[^\n]+)', re.DOTALL | re.MULTILINE) + # Eiffel uses -- for line comments; strings are double-quoted. + pattern = re.compile(r'(?P--.*?$)|(?P"[^"\n]*"|.[^"-]*)', + re.DOTALL | re.MULTILINE) return pattern @classmethod diff --git a/src/PyReprism/languages/gedcom.py b/src/PyReprism/languages/gedcom.py index b714ba5..5b7edd0 100644 --- a/src/PyReprism/languages/gedcom.py +++ b/src/PyReprism/languages/gedcom.py @@ -25,7 +25,7 @@ def keywords(cls) -> list: @classmethod def comment_regex(cls): # GEDCOM has numeric-level lines; no formal comment token, but accept lines starting with "0" or "1" as data; treat lines starting with "#" as comment for safety - return re.compile(r'(?P^#.*?$)|(?P^[^#\n].*?$)', re.MULTILINE) + return re.compile(r'(?P^#.*?$)|(?P.[^\n]*|\n)', re.MULTILINE) @classmethod def remove_comments(cls, source_code: str, isList: bool = False): diff --git a/src/PyReprism/languages/gherkin.py b/src/PyReprism/languages/gherkin.py index fec9c49..cdf1fa8 100644 --- a/src/PyReprism/languages/gherkin.py +++ b/src/PyReprism/languages/gherkin.py @@ -23,7 +23,7 @@ def keywords(cls) -> list: @classmethod def comment_regex(cls): - return re.compile(r'(?P^#.*?$)|(?P^[^#\n].*?$)', re.MULTILINE) + return re.compile(r'(?P^\s*#.*?$)|(?P.[^\n]*|\n)', re.MULTILINE) @classmethod def remove_comments(cls, source_code: str, isList: bool = False): diff --git a/src/PyReprism/languages/io.py b/src/PyReprism/languages/io.py index 2462e7a..55ff909 100644 --- a/src/PyReprism/languages/io.py +++ b/src/PyReprism/languages/io.py @@ -20,7 +20,7 @@ def keywords(cls) -> list: @classmethod def comment_regex(cls): - return re.compile(r'(?P^#.*?$)|(?P^[^#\n].*?$)', re.MULTILINE) + return re.compile(r'(?P#.*?$|//.*?$|/\*[\s\S]*?\*/)|(?P"[^"\n]*"|.[^#/"]*)', re.DOTALL | re.MULTILINE) @classmethod def number_regex(cls): diff --git a/src/PyReprism/languages/nix.py b/src/PyReprism/languages/nix.py index 5d403f0..2e49b11 100644 --- a/src/PyReprism/languages/nix.py +++ b/src/PyReprism/languages/nix.py @@ -20,7 +20,7 @@ def keywords(cls) -> list: @classmethod def comment_regex(cls): - return re.compile(r'(?P^#.*?$)|(?P^[^#\n].*?$)', re.MULTILINE) + return re.compile(r'(?P#.*?$|/\*[\s\S]*?\*/)|(?P"[^"\n]*"|.[^#/"]*)', re.DOTALL | re.MULTILINE) @classmethod def number_regex(cls): diff --git a/src/PyReprism/languages/smalltalk.py b/src/PyReprism/languages/smalltalk.py index 2063054..3dc8165 100644 --- a/src/PyReprism/languages/smalltalk.py +++ b/src/PyReprism/languages/smalltalk.py @@ -20,7 +20,13 @@ def keywords(cls) -> list: @classmethod def comment_regex(cls) -> re.Pattern: - return re.compile(r'(?P".*?"|".*?$|^.*?")|(?P[^\"]*)', re.DOTALL | re.MULTILINE) + # In Smalltalk, double quotes delimit comments and single quotes delimit + # strings. Keep strings (and other code) in the noncomment group. + return re.compile( + r'(?P"(?:[^"]|"")*")|' + r"(?P'(?:[^']|'')*'|.[^\"']*)", + re.DOTALL | re.MULTILINE, + ) @classmethod def number_regex(cls) -> re.Pattern: diff --git a/src/PyReprism/metrics.py b/src/PyReprism/metrics.py index 3e6fc13..4786eb3 100644 --- a/src/PyReprism/metrics.py +++ b/src/PyReprism/metrics.py @@ -1,7 +1,67 @@ -"""Code metrics produced by :meth:`BaseLanguage.stats`.""" +"""Code metrics produced by :meth:`BaseLanguage.stats` and friends.""" +import math from dataclasses import asdict, dataclass +@dataclass(frozen=True) +class Halstead: + """`Halstead complexity measures `_. + + Derived from token counts: operators are operator/keyword/punctuation tokens, + operands are identifier/number/string tokens. + """ + + distinct_operators: int # n1 + distinct_operands: int # n2 + total_operators: int # N1 + total_operands: int # N2 + + @property + def vocabulary(self) -> int: + return self.distinct_operators + self.distinct_operands + + @property + def length(self) -> int: + return self.total_operators + self.total_operands + + @property + def volume(self) -> float: + return self.length * math.log2(self.vocabulary) if self.vocabulary else 0.0 + + @property + def difficulty(self) -> float: + if not self.distinct_operands: + return 0.0 + return (self.distinct_operators / 2) * (self.total_operands / self.distinct_operands) + + @property + def effort(self) -> float: + return self.difficulty * self.volume + + @property + def time_seconds(self) -> float: + """Estimated implementation time (Halstead's ``E / 18``).""" + return self.effort / 18 + + @property + def bugs(self) -> float: + """Estimated delivered bugs (``V / 3000``).""" + return self.volume / 3000 + + def as_dict(self) -> dict: + data = asdict(self) + data.update( + vocabulary=self.vocabulary, + length=self.length, + volume=round(self.volume, 2), + difficulty=round(self.difficulty, 2), + effort=round(self.effort, 2), + time_seconds=round(self.time_seconds, 2), + bugs=round(self.bugs, 4), + ) + return data + + @dataclass(frozen=True) class CodeStats: """Line- and token-level metrics for a source string. diff --git a/src/PyReprism/ngrams.py b/src/PyReprism/ngrams.py new file mode 100644 index 0000000..39b23e7 --- /dev/null +++ b/src/PyReprism/ngrams.py @@ -0,0 +1,169 @@ +"""Token n-gram analysis and an n-gram language model for code "naturalness". + +Built on :meth:`~PyReprism.languages.base.BaseLanguage.tokenize`. Supports value +n-grams (over token text) and *type* n-grams (over token kinds — structural and +AST-free). :class:`NgramModel` estimates a smoothed n-gram distribution over a +corpus so a file's cross-entropy / perplexity can be measured, following Hindle +et al., "On the Naturalness of Software". +""" +import json +import math +from collections import Counter +from typing import Iterable, List, Optional, Sequence, Tuple + +from .tokens import TokenType + +START, END = '', '' +_SKIP = {TokenType.WHITESPACE} + + +def token_sequence(source: str, lang, *, types: bool = False, + include_comments: bool = False) -> List[str]: + """Reduce ``source`` to a flat list of token strings. + + With ``types`` the token *kind* is emitted instead of its text (structural + n-grams). Whitespace is always skipped; comments are skipped unless + ``include_comments``. + """ + from . import get_language + cls = get_language(lang) if not isinstance(lang, type) else lang + skip = set(_SKIP) + if not include_comments: + skip.add(TokenType.COMMENT) + out = [] + for tok in cls.tokenize(source): + if tok.type in skip: + continue + out.append(tok.type.value if types else tok.value) + return out + + +def ngrams(source: str, lang, n: int = 3, *, types: bool = False, + include_comments: bool = False, pad: bool = False) -> List[Tuple[str, ...]]: + """Return the list of token ``n``-grams in ``source``.""" + seq = token_sequence(source, lang, types=types, include_comments=include_comments) + return ngrams_of(seq, n, pad=pad) + + +def ngrams_of(seq: Sequence[str], n: int, *, pad: bool = False) -> List[Tuple[str, ...]]: + """Return the ``n``-grams of a token sequence.""" + if n < 1: + raise ValueError("n must be >= 1") + if pad: + seq = [START] * (n - 1) + list(seq) + [END] + return [tuple(seq[i:i + n]) for i in range(len(seq) - n + 1)] + + +def ngram_counts(source: str, lang, n: int = 3, **kwargs) -> Counter: + """Return a :class:`collections.Counter` of ``n``-grams in ``source``.""" + return Counter(ngrams(source, lang, n, **kwargs)) + + +class NgramModel: + """A smoothed n-gram language model over token sequences. + + Train with :meth:`fit`/:meth:`update` (sequences of token strings), then score + a sequence's :meth:`cross_entropy` or :meth:`perplexity`. Uses add-k + (Laplace) smoothing with an implicit unknown-token slot. + """ + + def __init__(self, n: int = 3, k: float = 1.0, types: bool = False): + if n < 1: + raise ValueError("n must be >= 1") + self.n = n + self.k = k + self.types = types + self.ngram_counts: Counter = Counter() + self.context_counts: Counter = Counter() + self.vocab: set = set() + + # ---------------------------------------------------------------- training + def update(self, seq: Sequence[str]) -> "NgramModel": + padded = [START] * (self.n - 1) + list(seq) + [END] + for i in range(self.n - 1, len(padded)): + context = tuple(padded[i - (self.n - 1):i]) + token = padded[i] + self.ngram_counts[context + (token,)] += 1 + self.context_counts[context] += 1 + self.vocab.add(token) + return self + + def fit(self, sequences: Iterable[Sequence[str]]) -> "NgramModel": + for seq in sequences: + self.update(seq) + return self + + # ----------------------------------------------------------------- scoring + @property + def vocab_size(self) -> int: + return len(self.vocab) + 1 # + 1 for unseen () tokens + + def _logprob(self, context: Tuple[str, ...], token: str) -> float: + num = self.ngram_counts.get(context + (token,), 0) + self.k + den = self.context_counts.get(context, 0) + self.k * self.vocab_size + return math.log2(num / den) + + def logprob(self, seq: Sequence[str]) -> float: + """Total log2-probability of ``seq`` under the model.""" + padded = [START] * (self.n - 1) + list(seq) + [END] + total = 0.0 + for i in range(self.n - 1, len(padded)): + context = tuple(padded[i - (self.n - 1):i]) + total += self._logprob(context, padded[i]) + return total + + def cross_entropy(self, seq: Sequence[str]) -> float: + """Average bits per token (lower = more predictable / "natural").""" + count = len(seq) + 1 # predicted tokens include the END marker + if count == 0: + return 0.0 + return -self.logprob(seq) / count + + def perplexity(self, seq: Sequence[str]) -> float: + """``2 ** cross_entropy`` — lower means more natural.""" + return 2 ** self.cross_entropy(seq) + + # ------------------------------------------------------------- persistence + def to_dict(self) -> dict: + return { + 'n': self.n, 'k': self.k, 'types': self.types, + 'vocab': sorted(self.vocab), + 'ngrams': [[list(g), c] for g, c in self.ngram_counts.items()], + 'contexts': [[list(g), c] for g, c in self.context_counts.items()], + } + + def save(self, path: str) -> None: + with open(path, 'w', encoding='utf-8') as handle: + json.dump(self.to_dict(), handle) + + @classmethod + def from_dict(cls, data: dict) -> "NgramModel": + model = cls(n=data['n'], k=data.get('k', 1.0), types=data.get('types', False)) + model.vocab = set(data.get('vocab', [])) + model.ngram_counts = Counter({tuple(g): c for g, c in data.get('ngrams', [])}) + model.context_counts = Counter({tuple(g): c for g, c in data.get('contexts', [])}) + return model + + @classmethod + def load(cls, path: str) -> "NgramModel": + with open(path, 'r', encoding='utf-8') as handle: + return cls.from_dict(json.load(handle)) + + +def train(paths, *, n: int = 3, types: bool = False, include_comments: bool = False, + recursive: bool = True, include=None, exclude=None, + lang: Optional[str] = None) -> NgramModel: + """Train an :class:`NgramModel` over a corpus (files or directories).""" + from .batch import iter_source_files + + model = NgramModel(n=n, types=types) + for _base, path, cls in iter_source_files(paths, recursive=recursive, + include=include, exclude=exclude): + language = lang or cls + try: + text = path.read_text(encoding='utf-8', errors='replace') + model.update(token_sequence(text, language, types=types, + include_comments=include_comments)) + except Exception: # pragma: no cover - defensive I/O guard + continue + return model diff --git a/tests/languages/test_abap.py b/tests/languages/test_abap.py deleted file mode 100644 index 3546ea9..0000000 --- a/tests/languages/test_abap.py +++ /dev/null @@ -1,37 +0,0 @@ -from PyReprism.languages.abap import Abap -import pytest - -class TestAbap: - @staticmethod - def test_instance_creation(): - try: - instance = Abap() - except Exception as e: - pytest.fail(f"Instance creation failed with exception: {e}") - assert isinstance(instance, Abap) - - @staticmethod - def test_extension(): - ext = Abap.file_extension() - assert ext == ".abap" - - @staticmethod - def test_remove_comments(): - source_code = ''' -* This is a full line comment -DATA: lv_value TYPE i. -" This is an inline comment -lv_value = 42. " Inline comment at the end of a line -WRITE: / 'This is not a comment'. -(* This is a multi-line comment - that spans multiple lines -*) -WRITE: / 'This is also not a comment'. -''' - expected_output = ''' -WRITE: / 'This is also not a comment'. -''' - output = Abap.remove_comments(source_code) - - assert output == expected_output.strip() - print("Test remove_comments passed!") diff --git a/tests/languages/test_actionscript.py b/tests/languages/test_actionscript.py deleted file mode 100644 index fccdb77..0000000 --- a/tests/languages/test_actionscript.py +++ /dev/null @@ -1,40 +0,0 @@ -from PyReprism.languages.actionscript import ActionScript -import pytest - -class TestActionScript: - @staticmethod - def test_instance_creation(): - try: - instance = ActionScript() - except Exception as e: - pytest.fail(f"Instance creation failed with exception: {e}") - assert isinstance(instance, ActionScript) - - @staticmethod - def test_extension(): - ext = ActionScript.file_extension() - assert ext == ".as" - - @staticmethod - def test_remove_comments(): - source_code = ''' -private function sayHello():void { -// This function prints "Hello, World!" to the console -trace("Hello, World!"); // Output: Hello, World! -/* This is -a multiline comment -*/ -} -''' - expected_output = ''' -private function sayHello():void { - -trace("Hello, World!"); - -} - -''' - output = ActionScript.remove_comments(source_code) - - assert output == expected_output.strip() - print("Test remove_comments passed!") diff --git a/tests/languages/test_all_languages.py b/tests/languages/test_all_languages.py new file mode 100644 index 0000000..e0ccc6b --- /dev/null +++ b/tests/languages/test_all_languages.py @@ -0,0 +1,256 @@ +"""Comprehensive, data-driven comment-handling tests for every language. + +Each language has one sample containing a comment written in that language's +own syntax. The tests assert that ``remove_comments`` strips the comment while +keeping the surrounding code, and that ``extract_comments`` returns it. + +``test_every_registered_language_is_covered`` fails if a new language is added +without a sample here, so coverage cannot silently regress. +""" +import pytest + +from PyReprism.languages import _load_all_languages +from PyReprism.languages.registry import LanguageRegistry + +_load_all_languages() + +# Markers embedded in every sample. +CODE_A, CODE_B, SECRET = "keepAA", "keepBB", "zzsecretzz" + +# name -> source string containing ``SECRET`` inside a comment (language syntax). +SAMPLES = { + 'Abap': 'keepAA "zzsecretzz\nkeepBB\n', + 'ActionScript': 'keepAA //zzsecretzz\nkeepBB\n', + 'Ada': 'keepAA --zzsecretzz\nkeepBB\n', + 'ApacheConf': 'keepAA #zzsecretzz\nkeepBB\n', + 'Apl': 'keepAA ⍝zzsecretzz\nkeepBB\n', + 'AppleScript': 'keepAA #zzsecretzz\nkeepBB\n', + 'Arduino': 'keepAA //zzsecretzz\nkeepBB\n', + 'Arff': 'keepAA %zzsecretzz\nkeepBB\n', + 'Asciidoc': 'keepAA\n//zzsecretzz\nkeepBB\n', + 'Asm6502': 'keepAA ;zzsecretzz\nkeepBB\n', + 'Aspnet': 'keepAA //zzsecretzz\nkeepBB\n', + 'AutoHotKey': 'keepAA ;zzsecretzz\nkeepBB\n', + 'Autoit': 'keepAA ;zzsecretzz\nkeepBB\n', + 'Bash': 'keepAA #zzsecretzz\nkeepBB\n', + 'Basic': "keepAA 'zzsecretzz\nkeepBB\n", + 'Batch': 'keepAA ::zzsecretzz\nkeepBB\n', + 'Bison': 'keepAA\n//zzsecretzz\nkeepBB\n', + 'Bro': 'keepAA #zzsecretzz\nkeepBB\n', + 'C': 'keepAA //zzsecretzz\nkeepBB\n', + 'CPP': 'keepAA //zzsecretzz\nkeepBB\n', + 'CSS': 'keepAA\n/*zzsecretzz*/\nkeepBB\n', + 'CSharp': 'keepAA //zzsecretzz\nkeepBB\n', + 'Clike': 'keepAA //zzsecretzz\nkeepBB\n', + 'Clojure': 'keepAA\n;zzsecretzz\nkeepBB\n', + 'CoffeeScript': 'keepAA #zzsecretzz\nkeepBB\n', + 'Crystal': 'keepAA #zzsecretzz\nkeepBB\n', + 'CssExtras': 'keepAA\n/*zzsecretzz*/\nkeepBB\n', + 'D': 'keepAA //zzsecretzz\nkeepBB\n', + 'Dart': 'keepAA //zzsecretzz\nkeepBB\n', + 'Diff': 'keepAA\n#zzsecretzz\nkeepBB\n', + 'Django': 'keepAA #zzsecretzz\nkeepBB\n', + 'Docker': 'keepAA #zzsecretzz\nkeepBB\n', + 'Eiffel': 'keepAA --zzsecretzz\nkeepBB\n', + 'Elixir': 'keepAA #zzsecretzz\nkeepBB\n', + 'ErLang': 'keepAA %zzsecretzz\nkeepBB\n', + 'Erb': 'keepAA #zzsecretzz\nkeepBB\n', + 'FSharp': 'keepAA //zzsecretzz\nkeepBB\n', + 'Flow': 'keepAA //zzsecretzz\nkeepBB\n', + 'ForTran': 'keepAA !zzsecretzz\nkeepBB\n', + 'Gedcom': 'keepAA\n#zzsecretzz\nkeepBB\n', + 'Gherkin': 'keepAA\n#zzsecretzz\nkeepBB\n', + 'Git': 'keepAA\n#zzsecretzz\nkeepBB\n', + 'Glsl': 'keepAA //zzsecretzz\nkeepBB\n', + 'Go': 'keepAA //zzsecretzz\nkeepBB\n', + 'GraphSql': 'keepAA #zzsecretzz\nkeepBB\n', + 'Groovy': 'keepAA //zzsecretzz\nkeepBB\n', + 'HTML': 'keepAA\n\nkeepBB\n', + 'Haml': 'keepAA\n-#zzsecretzz\nkeepBB\n', + 'Handlebars': 'keepAA\n{{!zzsecretzz}}\nkeepBB\n', + 'Haskell': 'keepAA --zzsecretzz\nkeepBB\n', + 'Haxe': 'keepAA //zzsecretzz\nkeepBB\n', + 'Hpkp': 'keepAA #zzsecretzz\nkeepBB\n', + 'Hsts': 'keepAA #zzsecretzz\nkeepBB\n', + 'INI': 'keepAA\n;zzsecretzz\nkeepBB\n', + 'IO': 'keepAA //zzsecretzz\nkeepBB\n', + 'IchigoJam': 'keepAA #zzsecretzz\nkeepBB\n', + 'Icon': 'keepAA #zzsecretzz\nkeepBB\n', + 'Inform7': 'keepAA //zzsecretzz\nkeepBB\n', + 'J': 'keepAA NB.zzsecretzz\nkeepBB\n', + 'Java': 'keepAA //zzsecretzz\nkeepBB\n', + 'JavaScript': 'keepAA //zzsecretzz\nkeepBB\n', + 'Jolie': 'keepAA //zzsecretzz\nkeepBB\n', + 'Json': 'keepAA //zzsecretzz\nkeepBB\n', + 'Jsx': 'keepAA //zzsecretzz\nkeepBB\n', + 'Julia': 'keepAA #zzsecretzz\nkeepBB\n', + 'Keyman': 'keepAA //zzsecretzz\nkeepBB\n', + 'Kotlin': 'keepAA //zzsecretzz\nkeepBB\n', + 'LOLCODE': 'keepAA BTW zzsecretzz\nkeepBB\n', + 'LUA': 'keepAA --zzsecretzz\nkeepBB\n', + 'Latex': 'keepAA %zzsecretzz\nkeepBB\n', + 'Less': 'keepAA //zzsecretzz\nkeepBB\n', + 'Liquid': 'keepAA\n{% comment %}zzsecretzz{% endcomment %}\nkeepBB\n', + 'LiveScript': 'keepAA #zzsecretzz\nkeepBB\n', + 'MEL': 'keepAA //zzsecretzz\nkeepBB\n', + 'MakeFile': 'keepAA #zzsecretzz\nkeepBB\n', + 'MarkDown': 'keepAA\n\nkeepBB\n', + 'MarkUp': 'keepAA\n\nkeepBB\n', + 'MarkupTemplating': 'keepAA\n\nkeepBB\n', + 'MatLab': 'keepAA %zzsecretzz\nkeepBB\n', + 'Mizar': 'keepAA ::zzsecretzz\nkeepBB\n', + 'Monkey': "keepAA 'zzsecretzz\nkeepBB\n", + 'N4js': 'keepAA //zzsecretzz\nkeepBB\n', + 'NASM': 'keepAA\n;zzsecretzz\nkeepBB\n', + 'NIM': 'keepAA #zzsecretzz\nkeepBB\n', + 'NIX': 'keepAA #zzsecretzz\nkeepBB\n', + 'NSIS': 'keepAA #zzsecretzz\nkeepBB\n', + 'Nginx': 'keepAA #zzsecretzz\nkeepBB\n', + 'ObjectiveC': 'keepAA //zzsecretzz\nkeepBB\n', + 'Ocaml': 'keepAA\n(*zzsecretzz*)\nkeepBB\n', + 'OpenCL': 'keepAA //zzsecretzz\nkeepBB\n', + 'Oz': 'keepAA %zzsecretzz\nkeepBB\n', + 'PHP': 'keepAA //zzsecretzz\nkeepBB\n', + 'PHPExtras': 'keepAA //zzsecretzz\nkeepBB\n', + 'PLSQL': 'keepAA --zzsecretzz\nkeepBB\n', + 'PariGP': 'keepAA \\\\zzsecretzz\nkeepBB\n', + 'Parser': 'keepAA #zzsecretzz\nkeepBB\n', + 'Pascal': 'keepAA //zzsecretzz\nkeepBB\n', + 'Perl': 'keepAA #zzsecretzz\nkeepBB\n', + 'PowerShell': 'keepAA #zzsecretzz\nkeepBB\n', + 'Processing': 'keepAA //zzsecretzz\nkeepBB\n', + 'Prolog': 'keepAA %zzsecretzz\nkeepBB\n', + 'Properties': 'keepAA #zzsecretzz\nkeepBB\n', + 'Protobuf': 'keepAA //zzsecretzz\nkeepBB\n', + 'Pug': 'keepAA //zzsecretzz\nkeepBB\n', + 'Puppet': 'keepAA #zzsecretzz\nkeepBB\n', + 'Pure': 'keepAA //zzsecretzz\nkeepBB\n', + 'Python': 'keepAA #zzsecretzz\nkeepBB\n', + 'Q': 'keepAA //zzsecretzz\nkeepBB\n', + 'Qore': 'keepAA //zzsecretzz\nkeepBB\n', + 'R': 'keepAA #zzsecretzz\nkeepBB\n', + 'Reason': 'keepAA //zzsecretzz\nkeepBB\n', + 'RenPy': 'keepAA #zzsecretzz\nkeepBB\n', + 'Rest': 'keepAA\n.. zzsecretzz\nkeepBB\n', + 'Rip': 'keepAA #zzsecretzz\nkeepBB\n', + 'Roboconf': 'keepAA #zzsecretzz\nkeepBB\n', + 'Ruby': 'keepAA #zzsecretzz\nkeepBB\n', + 'Rust': 'keepAA //zzsecretzz\nkeepBB\n', + 'SAS': 'keepAA\n/*zzsecretzz*/\nkeepBB\n', + 'SQL': 'keepAA //zzsecretzz\nkeepBB\n', + 'Sass': 'keepAA //zzsecretzz\nkeepBB\n', + 'Scala': 'keepAA //zzsecretzz\nkeepBB\n', + 'Scheme': 'keepAA ;zzsecretzz\nkeepBB\n', + 'Scss': 'keepAA //zzsecretzz\nkeepBB\n', + 'SmallTalk': 'keepAA\n"zzsecretzz"\nkeepBB\n', + 'Smarty': 'keepAA\n{*zzsecretzz*}\nkeepBB\n', + 'Soy': 'keepAA //zzsecretzz\nkeepBB\n', + 'Stylus': 'keepAA\n//zzsecretzz\nkeepBB\n', + 'Swift': 'keepAA //zzsecretzz\nkeepBB\n', + 'Tcl': 'keepAA #zzsecretzz\nkeepBB\n', + 'Textile': 'keepAA #zzsecretzz\nkeepBB\n', + 'Tsx': 'keepAA //zzsecretzz\nkeepBB\n', + 'Twig': 'keepAA\n{#zzsecretzz#}\nkeepBB\n', + 'TypeScript': 'keepAA //zzsecretzz\nkeepBB\n', + 'Vbnet': "keepAA 'zzsecretzz\nkeepBB\n", + 'Velocity': 'keepAA ##zzsecretzz\nkeepBB\n', + 'Verilog': 'keepAA //zzsecretzz\nkeepBB\n', + 'Vhdl': 'keepAA --zzsecretzz\nkeepBB\n', + 'Vim': 'keepAA "zzsecretzz\nkeepBB\n', + 'VisualBasic': "keepAA 'zzsecretzz\nkeepBB\n", + 'Wasm': 'keepAA ;zzsecretzz\nkeepBB\n', + 'Xeora': 'keepAA //zzsecretzz\nkeepBB\n', + 'Xojo': 'keepAA //zzsecretzz\nkeepBB\n', + 'Yaml': 'keepAA #zzsecretzz\nkeepBB\n', +} + +# Languages whose comment model doesn't fit the shared sample; tested separately. +SPECIAL = {'CSP', 'BrainFuck'} + + +@pytest.mark.parametrize('name', sorted(SAMPLES)) +def test_remove_comments_strips_comment_and_keeps_code(name): + cls = LanguageRegistry.get(name) + assert cls is not None, f"{name} is not registered" + out = cls.remove_comments(SAMPLES[name]) + assert CODE_A in out and CODE_B in out, f"{name}: code was removed -> {out!r}" + assert SECRET not in out, f"{name}: comment was not stripped -> {out!r}" + + +@pytest.mark.parametrize('name', sorted(SAMPLES)) +def test_extract_comments_returns_the_comment(name): + cls = LanguageRegistry.get(name) + comments = cls.extract_comments(SAMPLES[name]) + assert any(SECRET in c for c in comments), f"{name}: comment not extracted -> {comments!r}" + + +@pytest.mark.parametrize('name', sorted(SAMPLES)) +def test_remove_comments_is_idempotent(name): + cls = LanguageRegistry.get(name) + once = cls.remove_comments(SAMPLES[name]) + assert cls.remove_comments(once) == once, f"{name}: remove_comments not idempotent" + + +def test_csp_has_no_comment_syntax(): + # CSP is header text with no comments; nothing is treated as a comment + # (not even the "//" in a URL). + csp = LanguageRegistry.get('CSP') + out = csp.remove_comments("default-src 'self'; script-src https://example.com") + assert "https://example.com" in out + assert "default-src 'self'" in out + + +def test_brainfuck_keeps_commands_strips_prose(): + # In Brainfuck every non-command character is a comment. + bf = LanguageRegistry.get('BrainFuck') + out = bf.remove_comments('+++ hello , world [.]') + assert '+++' in out and ',' in out and '[' in out and '.' in out + assert 'hello' not in out and 'world' not in out + + +# Block comments that span several lines (the SAMPLES above are mostly one line). +MULTILINE_BLOCKS = { + 'C': 'keepAA\n/*\nzzsecretzz\n*/\nkeepBB\n', + 'CPP': 'keepAA\n/*\nzzsecretzz\n*/\nkeepBB\n', + 'Java': 'keepAA\n/*\nzzsecretzz\n*/\nkeepBB\n', + 'JavaScript': 'keepAA\n/*\nzzsecretzz\n*/\nkeepBB\n', + 'Go': 'keepAA\n/*\nzzsecretzz\n*/\nkeepBB\n', + 'Rust': 'keepAA\n/*\nzzsecretzz\n*/\nkeepBB\n', + 'CSS': 'keepAA\n/*\nzzsecretzz\n*/\nkeepBB\n', + 'PHP': 'keepAA\n/*\nzzsecretzz\n*/\nkeepBB\n', + 'Ocaml': 'keepAA\n(*\nzzsecretzz\n*)\nkeepBB\n', + 'Pascal': 'keepAA\n(*\nzzsecretzz\n*)\nkeepBB\n', + 'HTML': 'keepAA\n\nkeepBB\n', + 'MarkDown': 'keepAA\n\nkeepBB\n', +} + + +@pytest.mark.parametrize('name', sorted(MULTILINE_BLOCKS)) +def test_multiline_block_comments(name): + cls = LanguageRegistry.get(name) + out = cls.remove_comments(MULTILINE_BLOCKS[name]) + assert CODE_A in out and CODE_B in out, f"{name}: code removed -> {out!r}" + assert SECRET not in out, f"{name}: block comment not stripped -> {out!r}" + + +@pytest.mark.parametrize('quote', ['"""', "'''"]) +@pytest.mark.parametrize('name', ['Python', 'Django']) +def test_triple_quoted_block_comments(name, quote): + # Python (and Django, which processes .py files) treats triple-quoted + # string blocks as removable docstring "comments". + cls = LanguageRegistry.get(name) + out = cls.remove_comments(f"x = 1\n{quote}\n{SECRET}\n{quote}\ny = 2\n") + assert 'x = 1' in out and 'y = 2' in out + assert SECRET not in out + + +def test_every_registered_language_is_covered(): + registered = set(LanguageRegistry.all()) + covered = set(SAMPLES) | SPECIAL + missing = registered - covered + assert not missing, ( + f"these registered languages have no comment test: {sorted(missing)}. " + f"Add a sample to SAMPLES in this file." + ) + stale = covered - registered - SPECIAL + assert not stale, f"these samples reference unregistered languages: {sorted(stale)}" diff --git a/tests/languages/test_c.py b/tests/languages/test_c.py deleted file mode 100644 index 7e3a6a2..0000000 --- a/tests/languages/test_c.py +++ /dev/null @@ -1,46 +0,0 @@ -from PyReprism.languages.c import C - -class TestJava: - @staticmethod - def test_extension(): - ext = C.file_extension() - assert ext == ".c" - - @staticmethod - def test_remove_comments(): - source_code = ''' -int main() { -// This is a single-line comment -int a = 5; -int b = 6; -/* -This is a multi-line -comment -*/ -} -''' - - expected_output = ''' -int main() { - -int a = 5; -int b = 6; - -} -''' - - output = C.remove_comments(source_code) - assert output == expected_output.strip() - print("Test remove_comments passed!") - - # @staticmethod - # def test_keywords(): - # expected_keywords = [ - # 'abstract', 'continue', 'for', 'new', 'switch', 'assert', 'default', 'goto', 'package', 'synchronized', - # 'boolean', 'do', 'if', 'private', 'this', 'break', 'double', 'implements', 'protected', 'throw', 'byte', - # 'else', 'import', 'public', 'throws', 'case', 'enum', 'instanceof', 'return', 'transient', 'catch', - # 'extends', 'int', 'short', 'try', 'char', 'final', 'interface', 'static', 'void', 'class', 'finally', - # 'long', 'strictfp', 'volatile', 'const', 'float', 'native', 'super', 'while' - # ] - # result = Java.keywords() - # assert result == expected_keywords \ No newline at end of file diff --git a/tests/languages/test_cpp.py b/tests/languages/test_cpp.py deleted file mode 100644 index d76d05b..0000000 --- a/tests/languages/test_cpp.py +++ /dev/null @@ -1,42 +0,0 @@ -from PyReprism.languages.cpp import CPP -import pytest - -class TestJava: - @staticmethod - def test_instance_creation(): - try: - instance = CPP() - except Exception as e: - pytest.fail(f"Instance creation failed with exception: {e}") - assert isinstance(instance, CPP) - @staticmethod - def test_extension(): - ext = CPP.file_extension() - assert ext == ".cpp" - - @staticmethod - def test_remove_comments(): - source_code = ''' -int main() { -// This is a single-line comment -int a = 5; -int b = 6; -/* -This is a multi-line -comment -*/ -} -''' - - expected_output = ''' -int main() { - -int a = 5; -int b = 6; - -} -''' - - output = CPP.remove_comments(source_code) - assert output == expected_output.strip() - print("Test remove_comments passed!") \ No newline at end of file diff --git a/tests/languages/test_dart.py b/tests/languages/test_dart.py deleted file mode 100644 index e69de29..0000000 diff --git a/tests/languages/test_django.py b/tests/languages/test_django.py deleted file mode 100644 index c9d49a6..0000000 --- a/tests/languages/test_django.py +++ /dev/null @@ -1,63 +0,0 @@ -from PyReprism.languages.django import Django -import pytest - - -class TestDjango: - - @staticmethod - def test_instance_creation(): - try: - instance = Django() - except Exception as e: - pytest.fail(f"Instance creation failed with exception: {e}") - assert isinstance(instance, Django) - - @staticmethod - def test_extension(): - ext = Django.file_extension() - assert ext == ".py" - - @staticmethod - def test_remove_comments(): - source_code = ''' - # This is a single-line comment - def greet(): - """ This is a docstring - that spans multiple lines """ - print("Hello, World!") # Inline comment - ''' - - expected_output = ''' - def greet(): - - print("Hello, World!") - ''' - - output = Django.remove_comments(source_code) - - assert output == expected_output.strip() - print("Test remove_comments passed!") - - @staticmethod - def test_remove_comments_two(): - source_code = """ - # single line comment - x = 5 + 6 - ''' - multiline - comment - ''' - print(x) - """ - - expected_output = ''' - x = 5 + 6 - - print(x) - ''' - - output = Django.remove_comments(source_code) - - assert output == expected_output.strip() - print("Test remove_comments_two passed!") - \ No newline at end of file diff --git a/tests/languages/test_go.py b/tests/languages/test_go.py deleted file mode 100644 index e69de29..0000000 diff --git a/tests/languages/test_html.py b/tests/languages/test_html.py deleted file mode 100644 index e7b7122..0000000 --- a/tests/languages/test_html.py +++ /dev/null @@ -1,37 +0,0 @@ -from PyReprism.languages.html import HTML - -class TestHTML: - @staticmethod - def test_extension(): - ext = HTML.file_extension() - assert ext == ".html" - - @staticmethod - def test_remove_comments(): - source_code = ''' - -

This is a paragraph

-} -''' - - expected_output = ''' - -

This is a paragraph

-} -''' - - output = HTML.remove_comments(source_code) - assert output == expected_output.strip() - print("Test remove_comments passed!") - - # @staticmethod - # def test_keywords(): - # expected_keywords = [ - # 'abstract', 'continue', 'for', 'new', 'switch', 'assert', 'default', 'goto', 'package', 'synchronized', - # 'boolean', 'do', 'if', 'private', 'this', 'break', 'double', 'implements', 'protected', 'throw', 'byte', - # 'else', 'import', 'public', 'throws', 'case', 'enum', 'instanceof', 'return', 'transient', 'catch', - # 'extends', 'int', 'short', 'try', 'char', 'final', 'interface', 'static', 'void', 'class', 'finally', - # 'long', 'strictfp', 'volatile', 'const', 'float', 'native', 'super', 'while' - # ] - # result = Java.keywords() - # assert result == expected_keywords \ No newline at end of file diff --git a/tests/languages/test_java.py b/tests/languages/test_java.py deleted file mode 100644 index 984a1e6..0000000 --- a/tests/languages/test_java.py +++ /dev/null @@ -1,58 +0,0 @@ -from PyReprism.languages.java import Java -import pytest - -class TestJava: - @staticmethod - def test_instance_creation(): - try: - instance = Java() - except Exception as e: - pytest.fail(f"Instance creation failed with exception: {e}") - assert isinstance(instance, Java) - - @staticmethod - def test_extension(): - ext = Java.file_extension() - assert ext == ".java" - - @staticmethod - def test_remove_comments(): - source_code = ''' -public class { -// This is a single-line comment -int a = 5; -int b = 6; -/* -This is a multi-line -comment -*/ -} -''' - - expected_output = ''' -public class { - -int a = 5; -int b = 6; - -} -''' - - output = Java.remove_comments(source_code) - - assert output == expected_output.strip() - print("Test remove_comments passed!") - - @staticmethod - def test_keywords(): - expected_keywords = [ - 'abstract', 'continue', 'for', 'new', 'switch', 'assert', 'default', 'goto', 'package', 'synchronized', - 'boolean', 'do', 'if', 'private', 'this', 'break', 'double', 'implements', 'protected', 'throw', 'byte', - 'else', 'import', 'public', 'throws', 'case', 'enum', 'instanceof', 'return', 'transient', 'catch', - 'extends', 'int', 'short', 'try', 'char', 'final', 'interface', 'static', 'void', 'class', 'finally', - 'long', 'strictfp', 'volatile', 'const', 'float', 'native', 'super', 'while' - ] - result = Java.keywords() - assert result == expected_keywords - - \ No newline at end of file diff --git a/tests/languages/test_javascript.py b/tests/languages/test_javascript.py deleted file mode 100644 index e69de29..0000000 diff --git a/tests/languages/test_kotlin.py b/tests/languages/test_kotlin.py deleted file mode 100644 index e69de29..0000000 diff --git a/tests/languages/test_python.py b/tests/languages/test_python.py deleted file mode 100644 index 9542bcf..0000000 --- a/tests/languages/test_python.py +++ /dev/null @@ -1,60 +0,0 @@ -from PyReprism.languages.python import Python -import pytest - -class TestPython: - @staticmethod - def test_instance_creation(): - try: - instance = Python() - except Exception as e: - pytest.fail(f"Instance creation failed with exception: {e}") - assert isinstance(instance, Python) - @staticmethod - def test_extension(): - ext = Python.file_extension() - assert ext == ".py" - - @staticmethod - def test_remove_comments(): - source_code = ''' - # This is a single-line comment - def greet(): - """ This is a docstring - that spans multiple lines """ - print("Hello, World!") # Inline comment - ''' - - expected_output = ''' - def greet(): - - print("Hello, World!") - ''' - - output = Python.remove_comments(source_code) - - assert output == expected_output.strip() - print("Test remove_comments passed!") - - @staticmethod - def test_remove_comments_two(): - source_code = """ - # single line comment - x = 5 + 6 - ''' - multiline - comment - ''' - print(x) - """ - - expected_output = ''' - x = 5 + 6 - - print(x) - ''' - - output = Python.remove_comments(source_code) - - assert output == expected_output.strip() - print("Test remove_comments_two passed!") - \ No newline at end of file diff --git a/tests/languages/test_rust.py b/tests/languages/test_rust.py deleted file mode 100644 index e69de29..0000000 diff --git a/tests/test_fingerprints.py b/tests/test_fingerprints.py new file mode 100644 index 0000000..d867aa5 --- /dev/null +++ b/tests/test_fingerprints.py @@ -0,0 +1,103 @@ +"""Tests for fingerprinting and clone/plagiarism similarity.""" +import pytest + +from PyReprism import fingerprints as fp +from PyReprism.languages import _load_all_languages +from PyReprism.languages.registry import LanguageRegistry +from PyReprism.cli import main + +_load_all_languages() + +ORIGINAL = "def total(items):\n s = 0\n for x in items:\n s = s + x\n return s\n" +# same structure, renamed identifiers + changed literal -> Type-2 clone +RENAMED = "def sum_all(values):\n acc = 1\n for v in values:\n acc = acc + v\n return acc\n" +UNRELATED = "class Foo:\n def bar(self):\n print('hello there world')\n" + + +# --------------------------------------------------------------- fingerprint +def test_fingerprint_is_deterministic(): + assert fp.fingerprint(ORIGINAL, 'python') == fp.fingerprint(ORIGINAL, 'python') + assert all(isinstance(h, int) for h in fp.fingerprint(ORIGINAL, 'python')) + + +def test_empty_source_has_empty_fingerprint(): + assert fp.fingerprint('', 'python') == set() + + +# ---------------------------------------------------------------- similarity +def test_identical_similarity_is_one(): + assert fp.similarity(ORIGINAL, ORIGINAL, 'python') == 1.0 + + +def test_renamed_clone_is_detected_with_normalization(): + # default normalize=True makes it rename/literal invariant + assert fp.similarity(ORIGINAL, RENAMED, 'python') > 0.9 + + +def test_unrelated_code_is_dissimilar(): + assert fp.similarity(ORIGINAL, UNRELATED, 'python') < 0.3 + + +def test_no_normalize_is_sensitive_to_renaming(): + with_norm = fp.similarity(ORIGINAL, RENAMED, 'python', normalize=True) + without = fp.similarity(ORIGINAL, RENAMED, 'python', normalize=False) + assert without < with_norm + + +def test_containment_is_asymmetric_fraction(): + assert fp.containment(ORIGINAL, RENAMED, 'python') > 0.9 + assert fp.containment('', ORIGINAL, 'python') == 0.0 + + +def test_jaccard_edge_cases(): + assert fp.jaccard(set(), set()) == 1.0 + assert fp.jaccard({1}, set()) == 0.0 + assert fp.jaccard({1, 2}, {2, 3}) == pytest.approx(1 / 3) + + +# --------------------------------------------------------------------- index +def test_index_similar_pairs(): + idx = fp.FingerprintIndex(k=4, w=3) + idx.add('a', ORIGINAL, 'python') + idx.add('b', RENAMED, 'python') + idx.add('c', UNRELATED, 'python') + pairs = idx.similar_pairs(threshold=0.5) + assert [(a, b) for a, b, _ in pairs] == [('a', 'b')] + assert idx.matches('a', threshold=0.5)[0][0] == 'b' + + +def test_index_add_paths(tmp_path): + (tmp_path / 'x.py').write_text(ORIGINAL) + (tmp_path / 'y.py').write_text(RENAMED) + (tmp_path / 'z.py').write_text(UNRELATED) + idx = fp.FingerprintIndex(k=4, w=3) + idx.add_paths(tmp_path) + assert len(idx.prints) == 3 + pairs = idx.similar_pairs(threshold=0.5) + assert len(pairs) == 1 and pairs[0][2] > 0.9 + + +# ---------------------------------------------------- works for all languages +def test_fingerprint_runs_for_all_languages(): + sample = 'a = f(x) + g(y)\nb = f(x) + g(y)\n' + for name, cls in LanguageRegistry.all().items(): + assert isinstance(fp.fingerprint(sample, cls), set) + + +# --------------------------------------------------------------------------- CLI +def test_cli_similarity(tmp_path, capsys): + a = tmp_path / 'a.py' + b = tmp_path / 'b.py' + a.write_text(ORIGINAL) + b.write_text(RENAMED) + assert main(['similarity', str(a), str(b)]) == 0 + assert float(capsys.readouterr().out.strip()) > 0.9 + + +def test_cli_clones(tmp_path, capsys): + (tmp_path / 'a.py').write_text(ORIGINAL) + (tmp_path / 'b.py').write_text(RENAMED) + (tmp_path / 'c.py').write_text(UNRELATED) + assert main(['clones', str(tmp_path), '--threshold', '0.5']) == 0 + out = capsys.readouterr().out + assert 'a.py' in out and 'b.py' in out and 'c.py' not in out diff --git a/tests/test_metrics.py b/tests/test_metrics.py index f87c111..3a392e1 100644 --- a/tests/test_metrics.py +++ b/tests/test_metrics.py @@ -1,11 +1,12 @@ -"""Tests for stats(), normalize() and blank_comments().""" +"""Tests for stats(), normalize(), blank_comments() and complexity metrics.""" import json import pytest import PyReprism as pr from PyReprism.languages import _load_all_languages -from PyReprism.metrics import CodeStats +from PyReprism.languages.registry import LanguageRegistry +from PyReprism.metrics import CodeStats, Halstead from PyReprism.cli import main _load_all_languages() @@ -125,3 +126,78 @@ def test_cli_normalize_keep_flags(monkeypatch, capsys): monkeypatch.setattr('sys.stdin', io.StringIO('total = 42\n')) assert main(['normalize', '--lang', 'python', '--keep-names', '--keep-numbers']) == 0 assert capsys.readouterr().out.strip() == 'total = 42' + + +# ------------------------------------------------------------------- halstead +def test_halstead_counts_and_derived(): + h = pr.halstead('x = a + b * a', lang='python') + assert isinstance(h, Halstead) + assert h.total_operators >= 1 and h.total_operands >= 1 + # 'a' appears twice -> counted once as a distinct operand + assert h.distinct_operands < h.total_operands + assert h.volume > 0 and h.difficulty > 0 + assert set(h.as_dict()) >= {'volume', 'difficulty', 'effort', 'bugs', 'vocabulary'} + + +def test_halstead_empty_source_is_zero(): + h = pr.halstead('', lang='python') + assert h.volume == 0.0 and h.difficulty == 0.0 + + +# ---------------------------------------------------------------- cyclomatic +def test_cyclomatic_straightline_is_one(): + assert pr.cyclomatic_complexity('x = 1\ny = 2\n', lang='python') == 1 + + +def test_cyclomatic_counts_branches_and_boolean_ops(): + # base 1 + if + and + elif + for = 5 + src = 'def f(n):\n if n > 0 and n < 9:\n return 1\n elif n:\n for i in n:\n pass\n' + assert pr.cyclomatic_complexity(src, lang='python') == 5 + + +def test_cyclomatic_counts_c_style_operators(): + # base 1 + if + && + ternary ? + assert pr.cyclomatic_complexity('if (a && b) return x ? y : z;', lang='c') == 4 + + +# ---------------------------------------------------------------- nesting / MI +def test_max_nesting_depth(): + py = LanguageRegistry.get('Python') + assert py.max_nesting_depth('f(g(h(x)))') == 3 + assert py.max_nesting_depth('a + b') == 0 + + +def test_maintainability_index_range_and_ordering(): + simple = pr.maintainability_index('x = 1\n', lang='python') + complex_src = ('def f(a, b, c, d):\n' + ' if a and b or c and d:\n' + ' return a * b + c - d / a\n' * 3) + hard = pr.maintainability_index(complex_src, lang='python') + assert 0 <= hard <= simple <= 100 + + +def test_code_metrics_bundles_everything(): + m = pr.code_metrics('def f():\n return 1 # c\n', lang='python') + assert 'code_lines' in m and 'halstead' in m + assert 'cyclomatic_complexity' in m and 'maintainability_index' in m + assert 'max_nesting_depth' in m + + +# ------------------------------------------------------- works for every language +def test_complexity_metrics_run_for_all_languages(): + sample = 'a = 1\nif a and b:\n f(g(x))\n' + for name, cls in LanguageRegistry.all().items(): + assert cls.cyclomatic_complexity(sample) >= 1 + assert cls.max_nesting_depth(sample) >= 0 + assert 0 <= cls.maintainability_index(sample) <= 100 + assert cls.halstead(sample).length >= 0 + + +# --------------------------------------------------------------------------- CLI +def test_cli_stats_full(monkeypatch, capsys): + import io + monkeypatch.setattr('sys.stdin', io.StringIO('if (a && b) { return 1; }\n')) + assert main(['stats', '--lang', 'c', '--full', '--json']) == 0 + data = json.loads(capsys.readouterr().out) + assert data['cyclomatic_complexity'] >= 2 + assert 'volume' in data['halstead'] + assert 'maintainability_index' in data diff --git a/tests/test_ngrams.py b/tests/test_ngrams.py new file mode 100644 index 0000000..02332c0 --- /dev/null +++ b/tests/test_ngrams.py @@ -0,0 +1,126 @@ +"""Tests for token n-gram analysis and the naturalness model.""" +import io +import json + +import pytest + +from PyReprism import ngrams as ng +from PyReprism.ngrams import NgramModel, train +from PyReprism.languages import _load_all_languages +from PyReprism.languages.registry import LanguageRegistry +from PyReprism.cli import main + +_load_all_languages() + + +# ---------------------------------------------------------------- extraction +def test_value_ngrams(): + grams = ng.ngrams('a = b + c', 'python', n=2) + assert ('a', '=') in grams and ('+', 'c') in grams + + +def test_type_ngrams_are_structural(): + grams = ng.ngrams('a = 1', 'python', n=2, types=True) + assert ('identifier', 'operator') in grams + assert all(all(isinstance(t, str) for t in g) for g in grams) + + +def test_ngram_counts_and_padding(): + counts = ng.ngram_counts('a a a', 'python', n=2) + assert counts[('a', 'a')] == 2 + padded = ng.ngrams('a', 'python', n=2, pad=True) + assert ('', 'a') in padded and ('a', '') in padded + + +def test_comments_skipped_by_default(): + seq = ng.token_sequence('x = 1 # note', 'python') + assert 'note' not in ' '.join(seq) + seq2 = ng.token_sequence('x = 1 # note', 'python', include_comments=True) + assert any('note' in t for t in seq2) + + +def test_ngrams_of_validates_n(): + with pytest.raises(ValueError): + ng.ngrams_of(['a', 'b'], 0) + + +# --------------------------------------------------------------------- model +CORPUS = [ + 'def f(x):\n return x + 1\n', + 'def g(y):\n return y * 2\n', + 'def h(z):\n return z - 3\n', +] + + +def _model(): + return NgramModel(n=2).fit(ng.token_sequence(s, 'python') for s in CORPUS) + + +def test_natural_code_scores_lower_perplexity(): + m = _model() + natural = ng.token_sequence('def k(w):\n return w + 4\n', 'python') + weird = ng.token_sequence('@ ~ ^ & | ? !', 'python') + assert m.perplexity(natural) < m.perplexity(weird) + + +def test_perplexity_is_two_pow_cross_entropy(): + m = _model() + seq = ng.token_sequence('def q(a):\n return a + 7\n', 'python') + assert m.perplexity(seq) == pytest.approx(2 ** m.cross_entropy(seq)) + + +def test_model_persistence_round_trip(): + m = _model() + restored = NgramModel.from_dict(json.loads(json.dumps(m.to_dict()))) + seq = ng.token_sequence(CORPUS[0], 'python') + assert restored.perplexity(seq) == pytest.approx(m.perplexity(seq)) + + +def test_model_rejects_bad_n(): + with pytest.raises(ValueError): + NgramModel(n=0) + + +# ---------------------------------------------------------------------- train +def test_train_over_corpus(tmp_path): + (tmp_path / 'a.py').write_text(CORPUS[0]) + (tmp_path / 'b.py').write_text(CORPUS[1]) + model = train(tmp_path, n=2) + assert model.vocab + assert model.perplexity(ng.token_sequence(CORPUS[0], 'python')) > 0 + + +# ------------------------------------------------------ works for all languages +def test_token_sequence_for_all_languages(): + sample = 'a = f(x) + 1\n' + for name, cls in LanguageRegistry.all().items(): + seq = ng.token_sequence(sample, cls) + assert isinstance(seq, list) + + +# --------------------------------------------------------------------------- CLI +def test_cli_ngrams(monkeypatch, capsys): + monkeypatch.setattr('sys.stdin', io.StringIO('a = b + c\na = b + d\n')) + assert main(['ngrams', '--lang', 'python', '-n', '2', '--top', '3']) == 0 + out = capsys.readouterr().out + assert 'a =' in out + + +def test_cli_ngrams_json(monkeypatch, capsys): + monkeypatch.setattr('sys.stdin', io.StringIO('a = 1\n')) + assert main(['ngrams', '--lang', 'python', '-n', '2', '--json']) == 0 + data = json.loads(capsys.readouterr().out) + assert isinstance(data, list) and isinstance(data[0][0], list) + + +def test_cli_perplexity(tmp_path, capsys): + corpus = tmp_path / 'corpus' + corpus.mkdir() + for i, src in enumerate(CORPUS): + (corpus / f'{i}.py').write_text(src) + target = tmp_path / 'k.py' + target.write_text('def k(w):\n return w + 4\n') + assert main(['perplexity', '--train', str(corpus), str(target)]) == 0 + out = capsys.readouterr().out + assert str(target) in out + assert float(out.split('\t')[0]) > 0