diff --git a/.dockerignore b/.dockerignore new file mode 100644 index 0000000..e5392e6 --- /dev/null +++ b/.dockerignore @@ -0,0 +1,11 @@ +# The image reads only the Rust workspace and the two licences, so deny +# everything and re-admit exactly those. A narrow context also keeps +# `rust/target` — gigabytes on a developer machine — out of the build. +* + +!rust +rust/target +rust/vendor/*/target + +!LICENSE-APACHE +!LICENSE-MIT diff --git a/.github/dependabot.yml b/.github/dependabot.yml index 9a05fa8..a66bb6f 100644 --- a/.github/dependabot.yml +++ b/.github/dependabot.yml @@ -51,3 +51,55 @@ updates: update-types: - minor - patch + - package-ecosystem: npm + directory: /editors/vscode + schedule: + interval: weekly + day: monday + time: "09:45" + timezone: Asia/Tokyo + cooldown: + default-days: 7 + open-pull-requests-limit: 10 + labels: + - dependencies + - "area: editors" + commit-message: + prefix: "chore(deps)" + ignore: + # @types/vscode has to stay on the version engines.vscode names, or the + # extension compiles against API the editors it claims to support do not + # have. + - dependency-name: "@types/vscode" + - dependency-name: "*" + update-types: + - version-update:semver-major + groups: + npm-non-major: + patterns: + - "*" + update-types: + - minor + - patch + - package-ecosystem: docker + directory: / + schedule: + interval: weekly + day: monday + time: "10:00" + timezone: Asia/Tokyo + cooldown: + default-days: 7 + open-pull-requests-limit: 10 + labels: + - dependencies + - "area: ci" + commit-message: + prefix: "chore(deps)" + ignore: + # The builder stage is pinned to the MSRV toolchain on purpose; a major + # or minor Rust bump is a deliberate change, not a dependency update. + - dependency-name: rust + update-types: + - version-update:semver-major + - version-update:semver-minor diff --git a/.github/rulesets/main.json b/.github/rulesets/main.json index affa17d..fc1d475 100644 --- a/.github/rulesets/main.json +++ b/.github/rulesets/main.json @@ -33,10 +33,18 @@ {"context": "rust"}, {"context": "msrv"}, {"context": "reference"}, + {"context": "docker"}, + {"context": "dogfood"}, + {"context": "vscode"}, + {"context": "docs"}, {"context": "host-smoke (ubuntu-latest)"}, {"context": "host-smoke (macos-15)"}, {"context": "host-smoke (windows-2025)"}, + {"context": "action-smoke (ubuntu-latest)"}, + {"context": "action-smoke (macos-15)"}, + {"context": "action-smoke (windows-2025)"}, {"context": "Analyze (actions)"}, + {"context": "Analyze (javascript-typescript)"}, {"context": "Analyze (python)"}, {"context": "Analyze (rust)"} ], diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml index a64e44f..156b5d4 100644 --- a/.github/workflows/ci.yml +++ b/.github/workflows/ci.yml @@ -30,9 +30,36 @@ jobs: - run: cargo fmt --all --manifest-path rust/Cargo.toml -- --check - run: cargo clippy --manifest-path rust/Cargo.toml --workspace --all-targets --locked -- -D warnings - run: cargo test --manifest-path rust/Cargo.toml --workspace --all-targets --locked + # `--all-targets` above builds every target but silently drops the + # doctests, so the examples in the library rustdoc are only ever + # compiled and run by this step. + - run: cargo test --manifest-path rust/Cargo.toml --doc --workspace --locked + # docs/library.md is hand-written prose and the step above never reads + # it: `--doc` compiles what is in the crate sources and nothing else. The + # page says every example on it is compiled and run, so it is handed to + # `rustdoc` as its own doctest file, linked against the library it + # documents. + - name: The examples on the library page still compile + run: | + cargo build --manifest-path rust/Cargo.toml --locked -p ocomment-core + rustdoc --test docs/library.md --edition 2024 \ + --extern ocomment_core=rust/target/debug/libocomment_core.rlib \ + -L rust/target/debug/deps + # The binary crate is in here for its links alone: nothing publishes its + # rustdoc, but its modules document each other, and a link that names a + # function somebody has since renamed is a wrong sentence wherever it is + # written. `missing_docs` stays off for it — a `clap` derive has no + # documentation to give. + - name: The documentation builds with no broken links + env: + RUSTDOCFLAGS: -D warnings + run: cargo doc --manifest-path rust/Cargo.toml --no-deps -p ocomment-core -p ocomment-plugin-sdk -p ocomment --locked - run: python3 tools/check_embedded_specs.py + - run: python3 tools/check_hooks.py - run: python3 -m pip install --disable-pip-version-check jsonschema==4.25.1 - run: cargo build --manifest-path rust/Cargo.toml --locked -p ocomment + - run: python3 tools/check_directives.py + - run: python3 tools/gen_docs.py --check - run: python3 tools/validate_schemas.py - name: Check publishable file sets run: ./tools/package-list.sh @@ -63,6 +90,83 @@ jobs: - run: opam install ./ocaml/ocomment-ref.opam --deps-only --with-test - run: opam exec -- dune runtest --root ocaml - run: opam exec -- ./tools/differential.sh + - run: cargo build --manifest-path rust/Cargo.toml --locked -p ocomment + - name: The reference still builds with every comment stripped out of it + run: | + set -euo pipefail + rm -rf "${RUNNER_TEMP}/strip-ocaml" + mkdir -p "${RUNNER_TEMP}/strip-ocaml" + git ls-files -z ocaml | xargs -0 cp --parents -t "${RUNNER_TEMP}/strip-ocaml" + # Only the strip moves into the copy, and it does so in a subshell: + # it has to start there so it inherits none of this repository's + # `.ocomment.toml`, while `opam exec` resolves the local switch + # setup-ocaml made in the workspace and stops finding it from + # anywhere else. So dune is pointed at the copy instead of moved to + # it, and the step never leaves GITHUB_WORKSPACE. + ( + cd "${RUNNER_TEMP}/strip-ocaml" + "${GITHUB_WORKSPACE}/rust/target/debug/ocomment" \ + fix --policy all --force-protected 2>&1 | tee "${RUNNER_TEMP}/strip-ocaml.log" + ) + grep -qE 'Removed [0-9]+ comments? in [0-9]+ files?' "${RUNNER_TEMP}/strip-ocaml.log" + opam exec -- dune build --root "${RUNNER_TEMP}/strip-ocaml/ocaml" + + dogfood: + runs-on: ubuntu-latest + timeout-minutes: 30 + steps: + - uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7 + with: + persist-credentials: false + - uses: dtolnay/rust-toolchain@4360b52568e2003a75bf9bc1d59f33a8e3fc893c # stable + - run: cargo build --manifest-path rust/Cargo.toml --locked -p ocomment + - run: python3 -m pip install --disable-pip-version-check pyyaml==6.0.2 + # The one invariant no byte-level fixture can state: a YAML block scalar + # reads the lines below it, so the hole a removal leaves on a comment line + # can be read back as part of a value. This strips thousands of generated + # documents under every layout and every policy and asks a real YAML + # parser whether they still mean the same thing. + # + # The corpus and both enumerated sweeps run in full here -- they are where + # the hazard lives and they are the same documents on every run. Only the + # pseudo-random set is cut, because its cost is linear and its value is + # not: `python3 tools/yaml_roundtrip.py` runs the whole 2400 on demand, + # and `--seed` moves it. Every pass is one `fsync` per rewritten file, so + # the tool overlaps them rather than waiting on them in turn. + - name: Removing YAML comments never changes what the document parses to + run: python3 tools/yaml_roundtrip.py --cases 200 + - name: Report the environment and the configuration OComment resolved + run: | + set -euo pipefail + ./rust/target/debug/ocomment doctor + ./rust/target/debug/ocomment config explain + # A bare run is the gate: it walks the repository under the ordinary + # hidden-file and size limits and under `.ocomment.toml`, so a new + # explanatory comment that carries no tag fails the build. + - name: OComment checks its own repository + run: ./rust/target/debug/ocomment --format github + - name: Strip every comment out of a copy of the workspace + run: | + set -euo pipefail + rm -rf "${RUNNER_TEMP}/strip" + mkdir -p "${RUNNER_TEMP}/strip" + git ls-files -z rust spec | xargs -0 cp --parents -t "${RUNNER_TEMP}/strip" + # The copy inherits no configuration of its own, and the patched + # crate under rust/vendor is not ours to rewrite. The spec corpus + # travels with the copy because the crate's spec-fixture tests read + # it from `../../spec`; it is the tests' input, so it is excluded + # from the strip rather than rewritten by it. + printf 'version = 1\n\n[files]\nexclude = ["rust/vendor/**", "spec/fixtures/**"]\n' \ + >"${RUNNER_TEMP}/strip/.ocomment.toml" + cd "${RUNNER_TEMP}/strip" + "${GITHUB_WORKSPACE}/rust/target/debug/ocomment" \ + fix --policy all --force-protected 2>&1 | tee "${RUNNER_TEMP}/strip.log" + grep -qE 'Removed [0-9]+ comments? in [0-9]+ files?' "${RUNNER_TEMP}/strip.log" + - name: The stripped workspace still builds and still passes the core tests + run: | + set -euo pipefail + cargo build --manifest-path "${RUNNER_TEMP}/strip/rust/Cargo.toml" --workspace --locked + cargo test --manifest-path "${RUNNER_TEMP}/strip/rust/Cargo.toml" -p ocomment-core --locked host-smoke: strategy: @@ -84,3 +188,148 @@ jobs: if: runner.os == 'Windows' shell: pwsh run: '& rust/target/release/ocomment.exe --version' + + action-smoke: + strategy: + fail-fast: false + matrix: + os: [ubuntu-latest, macos-15, windows-2025] + runs-on: ${{ matrix.os }} + timeout-minutes: 30 + steps: + - uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7 + with: + persist-credentials: false + - uses: dtolnay/rust-toolchain@4360b52568e2003a75bf9bc1d59f33a8e3fc893c # stable + - run: cargo build --manifest-path rust/Cargo.toml --locked -p ocomment + - name: Create the action fixture + shell: bash + run: | + set -euo pipefail + mkdir -p action-fixture + printf 'fn main() {\n let value = 1; // removable\n}\n' >action-fixture/sample.rs + printf 'def sample():\n return 1 # removable\n' >action-fixture/sample.py + - name: Run the composite action against the fixture + id: smoke + uses: ./ + with: + command: check + paths: action-fixture + format: sarif + sarif-file: ocomment.sarif + upload-sarif: "false" + fail-on-findings: "false" + verify-attestation: "false" + binary-path: rust/target/debug/ocomment + - name: Validate the SARIF the action produced + shell: bash + env: + SMOKE_EXIT_CODE: ${{ steps.smoke.outputs.exit-code }} + SMOKE_SARIF_FILE: ${{ steps.smoke.outputs.sarif-file }} + SMOKE_VERSION: ${{ steps.smoke.outputs.version }} + run: | + set -euo pipefail + if [ "${SMOKE_EXIT_CODE}" != "1" ]; then + echo "::error::the fixture has removable comments, so the action should report exit code 1, not ${SMOKE_EXIT_CODE}" + exit 1 + fi + if [ -z "${SMOKE_VERSION}" ]; then + echo "::error::the action reported no version" + exit 1 + fi + python=python3 + command -v python3 >/dev/null 2>&1 || python=python + "${python}" tools/validate_schemas.py --sarif "${SMOKE_SARIF_FILE}" | tee sarif-report.txt + grep -q 'with 2 ocomment results' sarif-report.txt + + vscode: + runs-on: ubuntu-latest + timeout-minutes: 30 + defaults: + run: + working-directory: editors/vscode + steps: + - uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7 + with: + persist-credentials: false + - uses: actions/setup-node@820762786026740c76f36085b0efc47a31fe5020 # v7 + with: + node-version: 22 + cache: npm + cache-dependency-path: editors/vscode/package-lock.json + - uses: dtolnay/rust-toolchain@4360b52568e2003a75bf9bc1d59f33a8e3fc893c # stable + - run: npm ci + - run: npm run lint + - run: npm run compile + # The manifest suite is what pins the extension version to the crate + # version, so it runs before anything is built from either of them. + - run: npm run unit + - name: Build the ocomment the extension launches + working-directory: ${{ github.workspace }} + run: cargo build --manifest-path rust/Cargo.toml --locked -p ocomment + - name: Put that ocomment first on PATH + working-directory: ${{ github.workspace }} + run: echo "${GITHUB_WORKSPACE}/rust/target/debug" >>"$GITHUB_PATH" + # `npm test` downloads a real VS Code and drives it, so it needs a + # display; the runner has no X server of its own. + - run: xvfb-run -a npm test + - name: Package the extension the release would publish + run: npx --no @vscode/vsce package --out ocomment.vsix + - uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7 + with: + name: ocomment-vsix + path: editors/vscode/ocomment.vsix + if-no-files-found: error + + docker: + runs-on: ubuntu-latest + timeout-minutes: 30 + steps: + - uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7 + with: + persist-credentials: false + # A source build on one platform, which is the path a release never + # takes, so the Dockerfile's own builder stage cannot rot between + # releases. The step after the smoke test takes the release path over the + # same file. + - name: Build the image from source + shell: bash + run: docker build -t ocomment:ci . + - name: Smoke test the image + shell: bash + run: | + set -euo pipefail + docker run --rm ocomment:ci --version + status=0 + docker run --rm -v "$PWD/spec:/src" ocomment:ci check --format json \ + >container-report.json || status=$? + if [ "$status" -gt 1 ]; then + echo "::error::the image failed to scan the mounted directory (exit ${status})" + exit 1 + fi + python3 -c 'import json, sys; json.load(open(sys.argv[1]))' container-report.json + # The release image is not compiled: the workflow replaces the `builder` + # stage with a buildx named context holding the musl binaries the release + # matrix already built. Handing the image its own binary back through + # that context exercises the second path over the same Dockerfile, so a + # release build is never the first to find the layout broken. The hosted + # runner's default buildx builder supplies `--build-context`; this step + # uses that same builder. + - name: Build the image again through the release path + shell: bash + run: | + set -euo pipefail + mkdir -p release/binaries/out/amd64 + container=$(docker create ocomment:ci) + docker cp "${container}:/ocomment" release/binaries/out/amd64/ocomment + docker rm "${container}" + # The Dockerfile's own builder stage hands the final stage a binary + # `install -m 0555` made read-only and executable, and a release + # archive carries the same mode. A named context that handed over an + # 0755 copy would be the one path where the image was built from a + # writable binary, so this step reproduces the mode the release + # really ships. + chmod 0555 release/binaries/out/amd64/ocomment + docker buildx build --build-context builder=release/binaries \ + --load -t ocomment:ci-prebuilt . + docker run --rm ocomment:ci-prebuilt --version diff --git a/.github/workflows/codeql.yml b/.github/workflows/codeql.yml index 1af359b..c4bfaad 100644 --- a/.github/workflows/codeql.yml +++ b/.github/workflows/codeql.yml @@ -32,6 +32,8 @@ jobs: include: - language: actions build-mode: none + - language: javascript-typescript + build-mode: none - language: python build-mode: none - language: rust diff --git a/.github/workflows/docs.yml b/.github/workflows/docs.yml new file mode 100644 index 0000000..7aa5789 --- /dev/null +++ b/.github/workflows/docs.yml @@ -0,0 +1,91 @@ +name: Docs + +on: + push: + branches: [main] + # The generated pages under docs/ are what a CLI change moves, and the + # `rust` job of CI fails until they are regenerated in the same commit, so + # a change that alters `--help` reaches this filter as a docs/ change. + paths: + - docs/** + - spec/** + - tools/gen_docs.py + - .github/workflows/docs.yml + # No path filter here: `docs` is a required status check, so it has to run on + # every pull request rather than only on the ones that touch the book. + pull_request: + workflow_dispatch: + +permissions: + contents: read + +concurrency: + group: docs-${{ github.event.pull_request.number || github.ref }} + cancel-in-progress: ${{ github.event_name == 'pull_request' }} + +env: + CARGO_TERM_COLOR: always + +jobs: + docs: + runs-on: ubuntu-latest + timeout-minutes: 20 + steps: + - uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7 + with: + persist-credentials: false + - uses: dtolnay/rust-toolchain@4360b52568e2003a75bf9bc1d59f33a8e3fc893c # stable + # Pinned: mdBook decides the rendered HTML, so an unpinned tool would + # let the published site change under a commit that touched nothing. + # The archive is fetched by hand because the repository's action policy + # does not allow third-party actions outside its allowlist. + - name: Install mdBook 0.5.4 + shell: bash + run: | + set -euo pipefail + curl -fsSL https://github.com/rust-lang/mdBook/releases/download/v0.5.4/mdbook-v0.5.4-x86_64-unknown-linux-gnu.tar.gz \ + | tar -xz -C /usr/local/bin + mdbook --version + - run: cargo build --manifest-path rust/Cargo.toml --locked -p ocomment + # The site may not restate anything the binary or spec/ no longer says. + - run: python3 tools/gen_docs.py --check + - run: mdbook build docs + # `create-missing = false` in docs/book.toml makes the build above fail on + # a SUMMARY entry with no file behind it, so this only has to catch the + # opposite: a chapter that was written and never linked from SUMMARY.md. + - name: Every page under docs/ is in the book + run: | + set -euo pipefail + status=0 + for page in docs/*.md; do + name=$(basename "$page") + if [ "$name" = "SUMMARY.md" ]; then continue; fi + if ! grep -q "($name)" docs/SUMMARY.md; then + echo "::error file=docs/SUMMARY.md::${name} is not listed in docs/SUMMARY.md" + status=1 + fi + done + exit "$status" + - uses: actions/upload-pages-artifact@fc324d3547104276b827a68afc52ff2a11cc49c9 # v5 + with: + path: target/book + + deploy-pages: + # Pages serves one site, so a deploy is never cancelled halfway and never + # races another: this group is deliberately separate from the workflow's. + concurrency: + group: pages + cancel-in-progress: false + if: github.event_name == 'push' && github.ref == 'refs/heads/main' + needs: docs + runs-on: ubuntu-latest + timeout-minutes: 10 + environment: + name: github-pages + url: ${{ steps.deployment.outputs.page_url }} + permissions: + pages: write # Publish the built book as this repository's Pages site. + id-token: write # Prove to the Pages API which workflow run is deploying. + steps: + - id: deployment + uses: actions/deploy-pages@cd2ce8fcbc39b97be8ca5fce6e763baed58fa128 # v5 diff --git a/.github/workflows/release.yml b/.github/workflows/release.yml index feaa114..06ca51a 100644 --- a/.github/workflows/release.yml +++ b/.github/workflows/release.yml @@ -186,6 +186,159 @@ jobs: GH_TOKEN: ${{ github.token }} run: gh release create "$GITHUB_REF_NAME" release/* --generate-notes --verify-tag + publish-container: + needs: publish-release + runs-on: ubuntu-latest + timeout-minutes: 30 + permissions: + actions: read # Download the musl archives the build matrix produced. + attestations: write # Publish build provenance for the pushed image. + contents: read # Read the Dockerfile and the licences out of the tag. + id-token: write # Obtain keyless Sigstore identities for signing. + packages: write # Push the image into this repository's GHCR namespace. + steps: + - uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7 + with: + persist-credentials: false + - uses: actions/download-artifact@3e5f45b2cfb9172054b4087a40e8e0b5a5461e7c # v8 + with: + pattern: ocomment-*-unknown-linux-musl + path: musl + merge-multiple: true + # The image ships the binaries this release already built, smoke tested, + # signed, and published as archives rather than a second compilation of + # the same tag. `builder` below is the buildx named context the Dockerfile + # copies from, so this layout is the whole contract between them. + - name: Lay the released musl binaries out as the `builder` context + shell: bash + run: | + set -euo pipefail + extract() { + mkdir -p "release/binaries/out/$2" + tar -xzf "musl/ocomment-$1.tar.gz" -C "release/binaries/out/$2" \ + --strip-components=1 "ocomment-$1/ocomment" + chmod 0755 "release/binaries/out/$2/ocomment" + } + extract x86_64-unknown-linux-musl amd64 + extract aarch64-unknown-linux-musl arm64 + # GHCR rejects an uppercase path, and this repository has one. + - name: Resolve the GHCR image name + id: image + shell: bash + run: echo "name=ghcr.io/${GITHUB_REPOSITORY,,}" >>"$GITHUB_OUTPUT" + - uses: docker/setup-qemu-action@96fe6ef7f33517b61c61be40b68a1882f3264fb8 # v4 + - uses: docker/setup-buildx-action@37fe631027851001ddb9b187196cc803df7f5f0e # v4 + - uses: docker/login-action@dbcb813823bdd20940b903addbd779551569679f # v4 + with: + registry: ghcr.io + username: ${{ github.actor }} + password: ${{ secrets.GITHUB_TOKEN }} + - id: meta + uses: docker/metadata-action@dc802804100637a589fabce1cb79ff13a1411302 # v6 + with: + images: ${{ steps.image.outputs.name }} + tags: | + type=semver,pattern={{version}} + type=semver,pattern={{major}}.{{minor}} + type=raw,value=latest + labels: | + org.opencontainers.image.title=ocomment + org.opencontainers.image.description=Fast, byte-preserving comment checker and remover + org.opencontainers.image.licenses=MIT OR Apache-2.0 + org.opencontainers.image.documentation=https://github.com/${{ github.repository }}/blob/${{ github.ref_name }}/docs/docker.md + - id: build + uses: docker/build-push-action@53b7df96c91f9c12dcc8a07bcb9ccacbed38856a # v7 + with: + context: . + build-contexts: builder=release/binaries + platforms: linux/amd64,linux/arm64 + push: true + tags: ${{ steps.meta.outputs.tags }} + labels: ${{ steps.meta.outputs.labels }} + annotations: ${{ steps.meta.outputs.annotations }} + provenance: mode=max + sbom: true + - uses: sigstore/cosign-installer@6f9f17788090df1f26f669e9d70d6ae9567deba6 # v4.1.2 + - name: Sign the pushed image + shell: bash + env: + DIGEST: ${{ steps.build.outputs.digest }} + IMAGE: ${{ steps.image.outputs.name }} + run: cosign sign --yes "${IMAGE}@${DIGEST}" + - name: Attest build provenance + uses: actions/attest-build-provenance@4d101475d8b20a2381f78447822ac1eab6504dd8 # v4 + with: + subject-name: ${{ steps.image.outputs.name }} + subject-digest: ${{ steps.build.outputs.digest }} + push-to-registry: true + + publish-vscode: + needs: publish-release + runs-on: ubuntu-latest + environment: vscode-marketplace + timeout-minutes: 30 + permissions: + contents: write # Attach the signed .vsix to the release for this tag. + id-token: write # Obtain keyless Sigstore identities for signing. + defaults: + run: + working-directory: editors/vscode + steps: + - uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7 + with: + persist-credentials: false + - uses: actions/setup-node@820762786026740c76f36085b0efc47a31fe5020 # v7 + with: + node-version: 22 + cache: npm + cache-dependency-path: editors/vscode/package-lock.json + # A Marketplace version cannot be republished, so a version that is not + # the tag has to stop the job before anything is uploaded rather than + # after. `npm run unit` below checks the same file against the crate + # version; this checks it against the tag being released. + - name: The Marketplace version is the tag + run: | + set -euo pipefail + manifest=$(node -p "require('./package.json').version") + if [ "v${manifest}" != "${GITHUB_REF_NAME}" ]; then + echo "::error::editors/vscode/package.json is ${manifest}, but the tag is ${GITHUB_REF_NAME}" + exit 1 + fi + - run: npm ci + - run: npm run lint + - run: npm run compile + - run: npm run unit + - name: Package the extension + run: npx --no @vscode/vsce package --out "ocomment-${GITHUB_REF_NAME}.vsix" + - uses: sigstore/cosign-installer@6f9f17788090df1f26f669e9d70d6ae9567deba6 # v4.1.2 + - name: Sign the extension + run: | + set -euo pipefail + cosign sign-blob --yes \ + --bundle "ocomment-${GITHUB_REF_NAME}.vsix.sigstore.json" \ + "ocomment-${GITHUB_REF_NAME}.vsix" + # The Marketplace serves its own copy, so the release keeps the exact + # bytes that were signed and published, for anyone who wants to verify + # an installed extension against this tag. + - name: Attach the extension and its signature to the release + env: + GH_TOKEN: ${{ github.token }} + run: | + set -euo pipefail + gh release upload "$GITHUB_REF_NAME" \ + "ocomment-${GITHUB_REF_NAME}.vsix" \ + "ocomment-${GITHUB_REF_NAME}.vsix.sigstore.json" + # Both publishers read their token out of the environment, so neither is + # passed on a command line where the runner would log it. + - name: Publish to the Visual Studio Marketplace + env: + VSCE_PAT: ${{ secrets.VSCE_PAT }} + run: npx --no @vscode/vsce publish --packagePath "ocomment-${GITHUB_REF_NAME}.vsix" + - name: Publish to Open VSX + env: + OVSX_PAT: ${{ secrets.OVSX_PAT }} + run: npx --no ovsx publish "ocomment-${GITHUB_REF_NAME}.vsix" + publish-crates: needs: publish-release runs-on: ubuntu-latest diff --git a/.gitignore b/.gitignore index 81f7e5c..f2179ad 100644 --- a/.gitignore +++ b/.gitignore @@ -1,6 +1,7 @@ /rust/target/ /ocaml/_build/ target/ +/target/book/ /.ocomment.lock.tmp __pycache__/ *.py[cod] @@ -13,3 +14,9 @@ __pycache__/ .vscode/ *.swp *.tmp +/editors/vscode/node_modules/ +/editors/vscode/dist/ +/editors/vscode/out/ +/editors/vscode/.vscode-test/ +*.vsix +/rust/target*/ diff --git a/.ocomment.toml b/.ocomment.toml new file mode 100644 index 0000000..7d91225 --- /dev/null +++ b/.ocomment.toml @@ -0,0 +1,35 @@ +# NOTE: OComment checks its own repository. `ocomment` from the root is the +# NOTE: gate the `dogfood` CI job runs, and Lefthook runs `ocomment check +# NOTE: --staged` before every commit; see CONTRIBUTING.md for the tag +# NOTE: convention this configuration enforces. TOML is a built-in language, so +# NOTE: this file is now one of the files that convention applies to. +version = 1 + +[files] +exclude = [ + # NOTE: Vendored crates, fixture bytes, and packaging or benchmark scratch are + # NOTE: not ours to rewrite: fixture comments are the test input itself. + "rust/vendor/**", + "spec/fixtures/**", + "editors/vscode/test-fixtures/**", + "release-extras/**", + "benchmarks/**", + # NOTE: A lock file is written by its resolver rather than by hand, and the + # NOTE: header Cargo puts at the top of this one comes back on the next + # NOTE: `cargo update`. + "rust/Cargo.lock", +] + +[policy] +mode = "legal" +layout = "lines" +keep_kind = ["doc-line", "doc-block"] +keep_regex = [ + '^(//|\(\*|/\*|#)\s*(NOTE|SAFETY|INVARIANT|PERF|TODO|FIXME|HACK)\b', + # NOTE: The version beside a SHA-pinned action, now that YAML is scanned. + # NOTE: CONTRIBUTING.md requires every `uses:` to carry one and Dependabot + # NOTE: rewrites it when it moves the pin, so it is read by a machine rather + # NOTE: than by a reader and has no rationale to tag. The pattern is the whole + # NOTE: comment, so prose that merely opens with a version is still prose. + '^#\s*v[0-9]+(\.[0-9]+)*$', +] diff --git a/.pre-commit-hooks.yaml b/.pre-commit-hooks.yaml new file mode 100644 index 0000000..abf2176 --- /dev/null +++ b/.pre-commit-hooks.yaml @@ -0,0 +1,22 @@ +# Hook definitions consumed by pre-commit when this repository is used as a +# `repo:` entry. The `files:` patterns are generated from spec/languages.toml +# and are enforced by tools/check_hooks.py, which CI runs on every change. +# +# `language: system` requires `ocomment` to already be on PATH: pre-commit's +# `language: rust` runs `cargo install --path .` at the checkout root, and this +# repository's manifest lives in rust/, so it cannot build these hooks. +- id: ocomment-check + name: ocomment check + description: 'Report removable comments in the staged source files; exit 1 blocks the commit, exit 2 signals an invalid source, configuration, plugin, or I/O failure.' + entry: ocomment check + language: system + types: [text] + files: '(?i)\.(bash|c|cc|cjs|cpp|cs|css|csx|cts|cu|cuh|cxx|dart|gemspec|go|h|hh|hpp|htm|html|hxx|java|jbuilder|js|json5|jsonc|jsx|kt|kts|lua|m|markdown|md|mjs|ml|mli|mlt|mm|mts|php|phpt|phtml|pl|pm|podspec|py|pyi|pyw|r|rake|rb|rbi|rbw|rmd|rockspec|rs|ru|sass|sc|scala|scss|sh|shtml|sql|svelte|swift|t|thor|toml|ts|tsx|vue|xhtml|yaml|yml|zig|zon|zsh)$' +- id: ocomment-fix + name: ocomment fix + description: 'Remove comments from the staged source files in place; exit 1 from the following ocomment-check run, or a file pre-commit sees modified, blocks the commit until the result is reviewed and staged.' + entry: ocomment fix + language: system + types: [text] + require_serial: true + files: '(?i)\.(bash|c|cc|cjs|cpp|cs|css|csx|cts|cu|cuh|cxx|dart|gemspec|go|h|hh|hpp|htm|html|hxx|java|jbuilder|js|json5|jsonc|jsx|kt|kts|lua|m|markdown|md|mjs|ml|mli|mlt|mm|mts|php|phpt|phtml|pl|pm|podspec|py|pyi|pyw|r|rake|rb|rbi|rbw|rmd|rockspec|rs|ru|sass|sc|scala|scss|sh|shtml|sql|svelte|swift|t|thor|toml|ts|tsx|vue|xhtml|yaml|yml|zig|zon|zsh)$' diff --git a/CHANGELOG.md b/CHANGELOG.md index e487d1b..c954914 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -7,10 +7,854 @@ All notable changes to OComment will be documented here. The project follows ### Added -- Byte-oriented scanners and transformations for 15 built-in languages and the +- Byte-oriented scanners and transformations for 30 built-in languages and the documented dialects. - CLI, staged Git fixes, LSP 3.18 server, declarative profiles, and sandboxed WASM component plugins. - Independent OCaml reference implementation and byte-for-byte differential fixtures. - Cross-platform CI, packaging definitions, and release verification gates. +- Full `--help` for every command and every possible value, an exit-status, + files, and examples epilogue, and an `ocomment man` subcommand that renders + the manual page. +- `-q`/`--quiet`, `-v`/`--verbose`, a `--progress` live scanning counter, and a + one-line preview of the reported comment that `--no-preview` turns off. +- `-` as a target: `check`, `diff`, and `scan` read standard input under the + `` pseudo-path. +- `fix --dry-run`, which prints the patch `fix` would apply and writes nothing. + Skipped paths are reported on standard error, so its standard output stays a + patch that `git apply` accepts. +- `fix -i`/`--interactive`, which asks about each removable comment in turn — + showing it with three lines of context either side and the line the removal + would leave behind, capped at its first and last three lines so a tall comment + cannot push the question off the screen — and writes only the accepted ones, + through the same rollback-backed transaction a plain `fix` uses. `y`, `n`, + `a` (the rest of this file), `d` (keep the rest of this file), `q` (stop + asking and apply), `x` (abort and write nothing) and `?`. It needs a terminal + on standard input and standard output, and refuses `--staged`, `--dry-run`, + `-q`, and the machine formats rather than quietly ignoring one of the two + flags. +- `ocomment doctor` probes the optional tools OComment shells out to — `curl`, + `gh`, `oras`, and `cosign`, alongside `git` — and reports the environment it + resolved: the working directory, the root, the configuration files it merged, + and whether its output is a terminal. A missing tool is a row in the report + naming what needs it, never a failing run. +- `init --force` and `init --stdout`; `init` otherwise refuses to overwrite an + existing file and notes a configuration that already applies to the directory. +- `--explain`, which lists every comment a human `check` or `scan` met, kept + ones included, and names the rule that decided each one together with the + setting behind it: the `[policy]` table of a named file, a + `[languages.]` table, the `[[overrides]]` entry whose globs matched, the + command-line flag, or the built-in default. A comment a built-in rule decided + is left with the flag that would overrule it. The machine formats refuse the + flag rather than ignoring it, and so does every command that writes no report + of comments for it to annotate. +- The repository checks itself. `.ocomment.toml` runs the `legal` policy with + `doc-line` and `doc-block` kept and protects any comment headed `NOTE`, + `SAFETY`, `INVARIANT`, `PERF`, `TODO`, `FIXME`, or `HACK`, so an explanatory + comment that says why it is there survives and one that only restates the + line below it does not. Every such comment in `rust/` and `ocaml/` carries + its tag, as does every one in the Python and shell tooling and in the + `Dockerfile`; the only paths left out of the gate are vendored crates, + fixture bytes, and packaging and benchmark scratch. `SAFETY` is reserved for + its Rust-wide meaning — justifying an `unsafe` block — and a rationale about + bytes or spoofing is an `INVARIANT`. `lefthook.yml` runs `ocomment check + --staged` before each commit, and the `dogfood` CI job runs a bare `ocomment` + over the tree, reports the environment through `doctor` and `config explain`, + and then strips every comment out of a copy of the sources with `fix --policy + all --force-protected` and rebuilds it: the Rust workspace builds and + `ocomment-core` still passes its tests, and the `reference` job does the same + for the OCaml reference. `CONTRIBUTING.md` documents the tags. +- An official VS Code extension, `P4suta.ocomment`, under `editors/vscode`. It + is a client only: it launches the separately installed `ocomment lsp`, + attaches it to the thirty-five language identifiers OComment scans, and + exposes the server's quick fixes, `source.fixAll.ocomment`, code lens, and + pull diagnostics, plus `OComment: Remove comments in file`, `... in + workspace`, `OComment: Restart server`, `OComment: Show output`, and a status + bar count. `ocomment.path` resolves a relative path against the workspace and + expands a leading `~`; a missing binary is a notification pointing at the + install instructions rather than a silent failure. The extension is disabled + in untrusted workspaces, because that setting names an executable it + launches. The extension version is the crate version, checked by the + extension's own suite on every pull request and against the tag before + `publish-vscode` can upload anything. A `vscode` CI job lints, compiles, + builds the binary the extension launches, and drives a real VS Code under + `xvfb-run`; `publish-vscode` signs the `.vsix` with cosign, attaches it to + the release, and publishes to the Marketplace and Open VSX. The extension + holds one file system watcher for its lifetime rather than one per start, and + every start, stop, and restart is queued behind the last, so a settings + change during a restart cannot leave a second server running with nothing + holding it. +- Item-by-item documentation for `ocomment-core` and `ocomment-plugin-sdk`, + and a gate that keeps it. `missing_docs` is denied through + `[workspace.lints]` for both library crates, so a public type, field, + variant, or method added without a doc comment fails `cargo clippy`. The + crate documentation states what byte-preserving means, that spans are + half-open and edits sorted and non-overlapping, and gives the policy + against comment-kind table; `scan`, `transform`, `transform_spans`, + `apply_edits`, `detect_language`, `explain_disposition`, + `DeclarativeProfile`, `SourceMap` and `IncrementalDocument` each carry a + runnable example, and `# Errors` and `# Panics` sections say what a call + refuses and what it asserts. CI now runs `cargo test --doc` — which + `--all-targets` silently skips — and `cargo doc` with `-D warnings`, so an + example that stops compiling or a broken intra-doc link fails the build. + `IncrementalError`, `LineDelimiter`, `BlockDelimiter`, `StringDelimiter` + and `ProtectedPattern` are exported from the crate root: each appeared in a + public signature that no downstream caller could name. Four runnable + examples under `rust/ocomment-core/examples` — `strip`, `external_spans`, + `incremental`, `profile` — and both library crates carry + `[package.metadata.docs.rs]`. +- `ocomment languages` is generated from `spec/languages.toml`, which the + binary now embeds, and `--format json` writes that table as an array of + objects: `name`, `extensions`, `dialects`, and, where a row has them, + `extension_dialects`, `reserved_names`, `shebangs`, and `notes`. The shared + table now records the dialect an extension selects — `.m` is Objective-C, + `.mm` Objective-C++, `.cu` CUDA — the whole file names that carry no + extension at all (`Dockerfile`, `Containerfile`, `Makefile`, `GNUmakefile`, + `.profile`, `.bashrc`, `.zshrc`, `tsconfig.json`, `jsconfig.json`), and the + interpreter names a `#!` line is read for, and `docs/languages.md` is + generated with them. `tools/check_embedded_specs.py` holds the embedded copy + to the canonical file, and `rust/ocomment/tests/spec_languages.rs` checks + every claim the table makes against the code that has to honour it: each + extension, reserved name, and shebang against `detect_language`, each row of + dialects against the list the binary prints when it refuses one, the schema + enumerations against the same vocabulary, and both listings against the table + itself. +- TOML is a built-in language, scanned by a lexer of its own rather than by the + profile engine. `#` opens the only comment form there is, and every string + form hides one: basic and literal strings, the multi-line forms of both — + where the closing delimiter is the last three of a run of up to five quotes — + and the quoted keys written in either. `.toml` selects it, as do the lock + files written in TOML that carry no extension of their own (`Cargo.lock`, + `Pipfile`, `poetry.lock`, `uv.lock`, `pdm.lock`; `Pipfile.lock` is JSON and + is not among them). Taplo's `#:schema` and `# taplo:` lines are directives a + removal keeps. +- Lua is a built-in language, scanned by a lexer of its own. `--` opens a short + comment and a long bracket after it — `--[[`, `--[==[` — a long one, which + ends only at the closing bracket of its own level; the same brackets without + the `--` are long strings, and `a[b[1]]` is neither, because a long bracket + needs its second `[`. Short strings carry `\z`, which swallows the whitespace + and newlines after it, and a backslash before a line ending, which carries it + into the string. `---` is the documentation comment of LDoc and the Lua + language server, a fourth dash makes an ordinary divider, and `---@diagnostic` + is a directive where the other annotations are documentation, alongside the + `-- luacheck:`, `-- selene:`, `-- stylua:` and `-- luacov:` lines. `.lua` and + `.rockspec` select it, as does a `lua` or `luajit` `#!` line — which, like any + first line that opens with `#`, the loader skips. +- YAML is a built-in language, scanned by a lexer of its own. `#` opens the + only comment form there is, and only where white space separates it from the + token in front of it, so the `#` of a URL fragment and the one inside a plain + `a#b` are content. Both quoted styles hide it, over a line break included, + and so does a block scalar: `|` and `>` with their indentation and chomping + indicators take a comment on the header line and then swallow every following + line more indented than the node they hang off, empty lines and document + markers decided in column zero. `.yml` and `.yaml` select it, as do the + configuration files written in YAML that carry no extension of their own + (`.clang-format`, `.clang-tidy`, `.yamllint`). The `# yaml-language-server:`, + `# yamllint`, `# renovate:`, `# checkov:skip`, `# trivy:ignore`, `# nosec`, + `# kics-scan` and `# @schema` lines are directives a removal keeps. +- PHP is a built-in language, scanned by a lexer of its own. A PHP file is two + languages at once: it opens in inline HTML, where every byte is output + verbatim and nothing is a comment, and `` leaves again — carrying + one line break away with it. A bare `` comment in a PHP file is not reported, which + `docs/languages.md` says out loud. It is also the only place an editor may + restart a scan from, because which mode a byte sits in is decided by + everything above it, so a file that is all PHP is rescanned from the top. +- Ruby is a built-in language, scanned by a lexer of its own. Four of Ruby's + tokens are spelled with a byte that is also an operator, and only where the + token stands decides which, so the scanner keeps the four states Ruby's own + lexer answers those questions from: `/` is a regular expression where a value + is expected and division after an operand, `%` opens `%q %Q %w %W %i %I %s %r + %x` and the bare `%(...)` in the first place and is modulo in the second, `?` + is a one-character string or the ternary operator, and `<<` opens a here + document or appends — with a bare word between the two, where white space in + front and none behind makes `puts /x/` a pattern and `a <` is kept because Roslyn's own + `GeneratedCodeUtilities.BeginsWithAutoGeneratedComment` searches the comments + in front of a file's first token for it and exempts a file that carries one + from every analyzer that opts out of generated code; `// ReSharper disable` and + `// ReSharper restore` are kept as the bounds of the region an inspection is + turned off over; and `// csharpier-ignore`, `-start` and `-end` are kept as the + whole comments CSharpier compares against. `.cs` and `.csx` select the + language, as does a `dotnet-script` `#!` line, and `#!` at the very first byte + is the script preamble; it has no dialects and no reserved file names. Ground + truth is the Roslyn lexer the .NET SDK 10.0.400 ships, with `csharpier` 1.3.0 + for the formatter marker: over the 70,630 C# files of dotnet/runtime, + dotnet/roslyn, dotnet/aspnetcore, dotnet/efcore, Newtonsoft.Json, Serilog and + ImageSharp, all 1,946,012 comments it reports come back with the same byte + span and the same kind. Four of those files are called invalid: two are not + C# at all — one is Visual Basic under a `.cs` name and one a deliberate + parser-error fixture, and Roslyn raises 236 and 4 errors against them — and + two are the conditional-section limitation above, an apostrophe in a block of + prose written under an `#if false`. 23,510 of those files were additionally + stripped of every comment and handed back to Roslyn: all 650,360 removals + left a file that still parses with no error, so no code byte was read as + comment. +- The C++ raw-string delimiter search is bounded by the d-char class instead of + searching the document for a `(` that may never come, so a stray `R"` in C++ + code no longer costs every line under it its restart point. The four give-up + paths whose bytes the scan then consumes — an unterminated OCaml quoted + string, C++ raw string, PostgreSQL dollar quote and Oracle q-quote — record + nothing, which leaves only the reads the scan really does rewind behind + recording one: the here-document delimiter parse and Swift's two searches for + the end of a regular expression literal. `ocaml_quoted_string` no longer + records the byte the scan is standing on, which `scan_ocaml` asks of every + byte of a document. Full-scan results are unchanged; what changes is how many + checkpoints an incremental rescan may start from. +- Scala is a built-in language. Its block comment nests as Dart's does, and the + documentation comment is the one the Scala 3 compiler's comment reader + answers to: a comment is documentation exactly when its text starts with + `/**` (`Comment.isDocComment`), so `/**/` and `/***/` are documentation + comments and `///` — which scaladoc does not read — is an ordinary line + comment, as is `//!`. +- A Scala string is interpolated exactly when an identifier stands directly + before its quote: the compiler's lexer turns that identifier into + `INTERPOLATIONID`, so `s"..."`, `raw"..."` and a custom interpolator such as + `xml"..."` interpolate, while a keyword — its own token — and a number leave + the quote to a plain string whose `$` is content. Inside an interpolated + string `$$` and `$"` write a literal `$` and `"` — the quote after a `$` + never closes the string — and `${ ... }` opens an expression that is code, + so a comment written there is a comment and may carry a line break a + single-line string's text could not. A triple-quoted string closes on the + first three quotes of a run and makes any further quotes of the run part of + its value, so `"""a""""` is the string `a"`, which is where Scala parts + company with Kotlin's existing scanner; a backquoted identifier may hold + `//` without it being a comment; and a character literal or symbol holds a + single character or an identifier, so it can never hide one. +- The XML literal is the one Scala construct whose text is not code: the + compiler's lexer emits an `XMLSTART` token at the `<` and the parser + re-reads the literal with an XML scanner, so this scanner follows the + parser — a `//` in element text is protected rather than removed. Element + text, CDATA and processing instructions are opaque, `{ ... }` in text or an + attribute is code, `` is an XML comment, and the literal ends at + the close tag matching its root or at a self-closing `/>`. A literal begins + exactly where the lexer says one does: a `<` preceded by space, tab, line + feed, `{`, `(` or `>` and followed by an XML name start, `!` or `?`, so + `x` stays a comparison and `x ` opens a literal. +- Vue and Svelte are built-in languages. A component's `\n\n\n"; + let report = scan(source, Language::Vue, ScanOptions::default()); + assert!(report.valid, "diagnostics: {:?}", report.diagnostics); + assert_eq!( + report + .comments + .iter() + .map(|comment| (comment.span.start, comment.span.end, comment.kind)) + .collect::>(), + vec![ + (25, 30, CommentKind::Line), + (49, 58, CommentKind::Block), + (99, 106, CommentKind::Line), + ] + ); +} + +/// A `lang` this scanner has no rules for makes the block opaque: a +/// `\n\n\n"; + let report = scan(source, Language::Vue, ScanOptions::default()); + assert!(report.valid, "diagnostics: {:?}", report.diagnostics); + assert!(report.comments.is_empty(), "{:?}", report.comments); +} + +/// The `v-pre` directive makes an element's content raw text, so the mustache +/// it holds is not code and the `//` in it is not a comment. +/// +/// Ground truth, `@vue/compiler-sfc` 3.5: `
{{ x // c }}
` +/// parses with the whole content as one text node. +#[test] +fn vue_v_pre_elements_are_opaque() { + let source = b"
{{ x // not }}
\n\n"; + let report = scan(source, Language::Vue, ScanOptions::default()); + assert!(report.valid, "diagnostics: {:?}", report.diagnostics); + assert_eq!(report.comments.len(), 1, "{:?}", report.comments); + assert_eq!(report.comments[0].span, ByteSpan::new(43, 56)); + assert_eq!(report.comments[0].kind, CommentKind::HtmlComment); +} + +/// Vue is detected from the `.vue` of a single-file component. +#[test] +fn vue_is_detected_from_its_extension() { + let found = detect_language(Some(Path::new("App.vue")), b"\n") + .expect("detected by extension"); + assert_eq!(found.language, Language::Vue); + assert_eq!(found.reason, "extension"); +} + +/// A Svelte component's template is HTML with code in its braces: every +/// `{ ... }` opens an expression whose comments are comments — a line one runs +/// to the end of its line — and `` is an HTML comment. +/// +/// Ground truth, `svelte/compiler` 5.56: the source below parses with the +/// `/* c */` and `// d` as comments of their expressions and the HTML comment +/// as a comment node. +#[test] +fn svelte_expressions_and_comments_in_the_template() { + let source = b"

{x /* c */}

\n\n

{y // d\n}

\n"; + let report = scan(source, Language::Svelte, ScanOptions::default()); + assert!(report.valid, "diagnostics: {:?}", report.diagnostics); + assert_eq!( + report + .comments + .iter() + .map(|comment| (comment.span.start, comment.span.end, comment.kind)) + .collect::>(), + vec![ + (6, 13, CommentKind::Block), + (19, 32, CommentKind::HtmlComment), + (39, 43, CommentKind::Line), + ] + ); +} + +/// A Svelte component's `\n\n"; + let report = scan(source, Language::Svelte, ScanOptions::default()); + assert!(report.valid, "diagnostics: {:?}", report.diagnostics); + assert_eq!( + report + .comments + .iter() + .map(|comment| (comment.span.start, comment.span.end, comment.kind)) + .collect::>(), + vec![(19, 24, CommentKind::Line), (55, 62, CommentKind::Line),] + ); +} + +/// Svelte is detected from the `.svelte` of a component. +#[test] +fn svelte_is_detected_from_its_extension() { + let found = detect_language(Some(Path::new("App.svelte")), b"

x

\n") + .expect("detected by extension"); + assert_eq!(found.language, Language::Svelte); + assert_eq!(found.reason, "extension"); +} + +/// The three layouts leave a Vue file a line, columns, or nothing. +#[test] +fn vue_layouts_leave_a_line_columns_or_nothing() { + let source = b"\n"; + let options = TransformOptions { + scan: ScanOptions { + policy: Policy::All, + ..Default::default() + }, + ..Default::default() + }; + let lines = transform(source, Language::Vue, options); + assert_eq!(lines.output, b"\n"); + let compact = transform( + source, + Language::Vue, + TransformOptions { + layout: Layout::Compact, + scan: ScanOptions { + policy: Policy::All, + ..Default::default() + }, + }, + ); + assert_eq!(compact.output, b"\n"); +} + +/// An HTML comment in Markdown is an HTML block per CommonMark 4.6, so it is +/// a comment that `safe` keeps as DOM-observable; it may span lines. +/// +/// Ground truth, `commonmark` 0.31: the source below parses with the comment +/// as one `html_block` node. +#[test] +fn markdown_html_comments_are_comments() { + let source = b"text\n\nmore\n\n"; + let report = scan(source, Language::Markdown, ScanOptions::default()); + assert!(report.valid, "diagnostics: {:?}", report.diagnostics); + assert_eq!( + report + .comments + .iter() + .map(|comment| (comment.span.start, comment.span.end, comment.kind)) + .collect::>(), + vec![ + (5, 18, CommentKind::HtmlComment), + (24, 39, CommentKind::HtmlComment), + ] + ); +} + +/// A fenced code block is scanned as the language its info string names: the +/// `// c` inside a `rust` fence and the `# c` inside a `ruby` fence are +/// comments of those languages. +/// +/// Ground truth, `commonmark` 0.31: the fences parse as `code_block` nodes +/// with the info strings `rust` and `ruby`. +#[test] +fn markdown_fenced_code_blocks_are_scanned_per_language() { + let source = b"```rust\n// c\n```\n~~~ruby\n# c\n~~~\ntext\n"; + let report = scan(source, Language::Markdown, ScanOptions::default()); + assert!(report.valid, "diagnostics: {:?}", report.diagnostics); + assert_eq!( + report + .comments + .iter() + .map(|comment| (comment.span.start, comment.span.end, comment.kind)) + .collect::>(), + vec![(8, 12, CommentKind::Line), (25, 28, CommentKind::Line)] + ); +} + +/// A fence whose info string names no language — or names none at all — is +/// opaque: its body holds no comment this scanner may take. +#[test] +fn markdown_unknown_fences_are_opaque() { + let source = b"```nope\n// not\n```\n```\n/* not */\n```\n"; + let report = scan(source, Language::Markdown, ScanOptions::default()); + assert!(report.valid, "diagnostics: {:?}", report.diagnostics); + assert!(report.comments.is_empty(), "{:?}", report.comments); +} + +/// Inline code spans and indented code blocks are opaque: the `//` and +/// `/* */` inside them are code text, not comments, while a comment outside +/// them is a comment. +/// +/// Ground truth, `commonmark` 0.31: the spans parse as `code` and +/// `code_block` nodes, and `//` in the prose is text — Markdown has no line +/// comment of its own. +#[test] +fn markdown_inline_and_indented_code_are_opaque() { + let source = b"`// not`\n /* not */\n more\n\ntext // real\n"; + let report = scan(source, Language::Markdown, ScanOptions::default()); + assert!(report.valid, "diagnostics: {:?}", report.diagnostics); + assert!(report.comments.is_empty(), "{:?}", report.comments); +} + +/// A fence closes only at a run of its own marker at least as long as its +/// own: the `// c` under a four-backtick closer is a comment, and a line of +/// three backticks inside a three-backtick block closes it, leaving the +/// `// not` below it text. +/// +/// Ground truth, `commonmark` 0.31: the first source's code block carries the +/// `// c` line and the second's ends at the first closer. +#[test] +fn markdown_fences_close_only_at_their_own_marker() { + let source = b"```rust\n// c\n````\n"; + let report = scan(source, Language::Markdown, ScanOptions::default()); + assert!(report.valid, "diagnostics: {:?}", report.diagnostics); + assert_eq!(report.comments.len(), 1, "{:?}", report.comments); + assert_eq!(report.comments[0].span, ByteSpan::new(8, 12)); + + let source = b"```rust\n```\n// not\n"; + let report = scan(source, Language::Markdown, ScanOptions::default()); + assert!(report.valid, "diagnostics: {:?}", report.diagnostics); + assert!(report.comments.is_empty(), "{:?}", report.comments); +} + +/// Markdown is detected from `.md`, `.markdown` and the `.Rmd` of an R +/// Markdown document, whose `{r}` chunk headers name R. +#[test] +fn markdown_is_detected_from_its_extensions() { + for path in [ + Path::new("README.md"), + Path::new("index.markdown"), + Path::new("report.Rmd"), + ] { + let found = detect_language(Some(path), b"text\n").expect("detected by extension"); + assert_eq!(found.language, Language::Markdown); + assert_eq!(found.reason, "extension"); + } +} + +/// A Perl `#` runs to the end of its line, and a POD block is opaque: the +/// `#` inside `=head1 ... =cut` is documentation text, not a comment. +/// +/// Ground truth, perl 5.38: the source below passes `perl -c`. +#[test] +fn perl_comments_and_pod_blocks() { + let source = b"=head1 NAME\n# not a comment\n=cut\nmy $x = 1; # comment\n"; + let report = scan(source, Language::Perl, ScanOptions::default()); + assert!(report.valid, "diagnostics: {:?}", report.diagnostics); + assert_eq!(report.comments.len(), 1, "{:?}", report.comments); + assert_eq!(report.comments[0].span, ByteSpan::new(44, 53)); +} + +/// Every Perl string and quote-word form hides a `#` written inside it: the +/// single and double quotes and backticks, the `q`, `qq`, `qw` and `qx` +/// forms, the `m`, `s`, `tr` and `y` operators with their own delimiters. +/// +/// Ground truth, perl 5.38: the source below passes `perl -c`. +#[test] +fn perl_strings_and_quote_words_hide_comment_openers() { + let source = b"my $a = '# not';\nmy $b = \"# not\";\nmy $c = `# not`;\nmy $d = q{# not};\nmy $e = qq{# not};\nmy $f = qw(a # b);\nmy $g = qx{# not};\nmy $h = m{# not};\nmy $i = s{# not}{x};\nmy $j = tr{a#}{b#};\n# remove\n"; + let report = scan(source, Language::Perl, ScanOptions::default()); + assert!(report.valid, "diagnostics: {:?}", report.diagnostics); + assert_eq!(report.comments.len(), 1, "{:?}", report.comments); + assert_eq!(report.comments[0].span, ByteSpan::new(185, 193)); +} + +/// A here-document's body is opaque until the line that names the terminator, +/// however the terminator was written: plain, quoted, or with the indented +/// `<<~` form. +/// +/// Ground truth, perl 5.38: the sources below pass `perl -c`. +#[test] +fn perl_heredocs_are_opaque() { + let source = b"my $a = <<'EOF';\n# not a comment\nEOF\nmy $b = <<\"EOF\";\n# not a comment\nEOF\nmy $c = < Vec { + let result = transform( + source, + language, + TransformOptions { + scan: ScanOptions { + policy, + ..ScanOptions::default() + }, + layout, + }, + ); + let mut cursor = 0; + for edit in &result.edits { + assert!( + edit.span.start <= edit.span.end, + "inverted edit {:?}", + edit.span + ); + assert!( + edit.span.start >= cursor, + "edit {:?} overlaps its predecessor, which ended at {cursor}", + edit.span + ); + assert!( + edit.span.end <= source.len(), + "edit {:?} reaches past the {}-byte source", + edit.span, + source.len() + ); + cursor = edit.span.end; + } + assert_eq!( + apply_edits(source, &result.edits), + result.output, + "the edits do not reproduce the output" + ); + assert_eq!( + result + .report + .comments + .iter() + .filter(|comment| comment.disposition.is_remove()) + .count(), + result.edits.len(), + "one edit per removed comment" + ); + assert_eq!( + result.source_map.original_to_output(source.len()), + Some(result.output.len()), + "the end of the source does not map to the end of the output" + ); + assert_eq!( + result.source_map.output_to_original(result.output.len()), + Some(source.len()), + "the end of the output does not map back to the end of the source" + ); + assert_eq!( + result.source_map.original_to_output(0), + Some(0), + "the start of the source does not map to the start of the output" + ); + result.output +} + +/// `compact` over Rust under the default `safe` policy. +fn compact(source: &str) -> String { + String::from_utf8(transformed( + source.as_bytes(), + Language::Rust, + Policy::Safe, + Layout::Compact, + )) + .expect("compact never splits a character") +} + +/// `lines` over the same source, for the side-by-side assertions. +fn lines(source: &str) -> String { + String::from_utf8(transformed( + source.as_bytes(), + Language::Rust, + Policy::Safe, + Layout::Lines, + )) + .expect("lines never splits a character") +} + +#[test] +fn a_comment_alone_on_its_line_takes_the_line_with_it() { + let source = "fn main() {}\n// one\n// two\nlet x = 1;\n"; + assert_eq!(compact(source), "fn main() {}\nlet x = 1;\n"); + assert_eq!(lines(source), "fn main() {}\n\n\nlet x = 1;\n"); +} + +#[test] +fn indentation_of_a_removed_line_goes_with_it() { + let source = "fn main() {\n // note\n let x = 1;\n}\n"; + assert_eq!(compact(source), "fn main() {\n let x = 1;\n}\n"); + assert_eq!(lines(source), "fn main() {\n \n let x = 1;\n}\n"); +} + +#[test] +fn crlf_lines_keep_their_endings() { + let source = "let x = 1;\r\n// note\r\nlet y = 2;\r\n"; + assert_eq!(compact(source), "let x = 1;\r\nlet y = 2;\r\n"); + let shared = "let x = 1; /* one\r\ntwo */ let y = 2;\r\n"; + assert_eq!(compact(shared), "let x = 1;\r\n let y = 2;\r\n"); + assert_eq!(lines(shared), "let x = 1; \r\n let y = 2;\r\n"); +} + +#[test] +fn the_first_line_of_a_file_goes_like_any_other() { + let source = "// header\nfn main() {}\n"; + assert_eq!(compact(source), "fn main() {}\n"); + assert_eq!(lines(source), "\nfn main() {}\n"); + assert_eq!(compact("// only\n"), ""); + assert_eq!(compact("// only"), ""); +} + +#[test] +fn a_surviving_line_keeps_the_ending_it_had_or_its_absence() { + assert_eq!(compact("let x = 1; // note"), "let x = 1;"); + assert_eq!(lines("let x = 1; // note"), "let x = 1; "); + // NOTE: The last line held nothing else, so it goes; the line before it + // NOTE: keeps the terminator it always had. + assert_eq!(compact("let x = 1;\n// note"), "let x = 1;\n"); + assert_eq!(compact("let x = 1;\n// one\n// two"), "let x = 1;\n"); + // NOTE: Here the terminator that ended the surviving line was inside the + // NOTE: comment, so it comes back even though the file ended without one. + assert_eq!(compact("let x = 1; /* one\ntwo */"), "let x = 1;\n"); + assert_eq!(lines("let x = 1; /* one\ntwo */"), "let x = 1; \n"); +} + +#[test] +fn whitespace_left_before_an_end_of_line_comment_is_trimmed() { + assert_eq!(compact("let x = 1; \t // note\n"), "let x = 1;\n"); + assert_eq!(lines("let x = 1; \t // note\n"), "let x = 1; \t \n"); + assert_eq!(compact("let x = 1; /* note */ \n"), "let x = 1;\n"); + assert_eq!(compact("let x = 1; /* note */ "), "let x = 1;"); +} + +#[test] +fn a_block_comment_that_shares_a_line_with_code_keeps_that_line() { + let before = "let a = 1; /* one\ntwo\nthree */\nlet b = 2;\n"; + assert_eq!(compact(before), "let a = 1;\nlet b = 2;\n"); + assert_eq!(lines(before), "let a = 1; \n\n\nlet b = 2;\n"); + let after = "let a = 1;\n/* one\ntwo\nthree */ let b = 2;\n"; + assert_eq!(compact(after), "let a = 1;\n let b = 2;\n"); + let both = "let a = 1; /* one\ntwo\nthree */ let b = 2;\n"; + assert_eq!(compact(both), "let a = 1;\n let b = 2;\n"); + assert_eq!(lines(both), "let a = 1; \n\n let b = 2;\n"); +} + +#[test] +fn a_block_comment_alone_on_its_lines_takes_all_of_them() { + let source = "let a = 1;\n/* one\ntwo\nthree */\nlet b = 2;\n"; + assert_eq!(compact(source), "let a = 1;\nlet b = 2;\n"); + assert_eq!(lines(source), "let a = 1;\n\n\n\nlet b = 2;\n"); +} + +#[test] +fn a_comment_between_two_tokens_is_left_exactly_as_lines_leaves_it() { + for source in [ + "let x = a/* widen */+ b;\n", + "let x = a /* widen */ + b;\n", + "let x = a/* widen */ + b;\n", + "let x = a/* one */b/* two */c;\n", + ] { + assert_eq!( + compact(source), + lines(source), + "compact and lines disagree about {source:?}" + ); + } +} + +#[test] +fn lines_and_columns_are_not_touched_by_the_compact_rules() { + let source = "fn main() {\n // note\n let x = a /* widen */ + b; // trailing\n}\n"; + assert_eq!( + String::from_utf8(transformed( + source.as_bytes(), + Language::Rust, + Policy::Safe, + Layout::Lines, + )) + .unwrap(), + "fn main() {\n \n let x = a + b; \n}\n" + ); + assert_eq!( + String::from_utf8(transformed( + source.as_bytes(), + Language::Rust, + Policy::Safe, + Layout::Columns, + )) + .unwrap(), + "fn main() {\n \n let x = a + b; \n}\n" + ); + assert_eq!(compact(source), "fn main() {\n let x = a + b;\n}\n"); +} + +#[test] +fn an_html_comment_still_closes_up_completely() { + let whole_line = "

a

\n\n

b

\n"; + assert_eq!( + String::from_utf8(transformed( + whole_line.as_bytes(), + Language::Html, + Policy::All, + Layout::Compact, + )) + .unwrap(), + "

a

\n

b

\n" + ); + let inline = "ab"; + assert_eq!( + String::from_utf8(transformed( + inline.as_bytes(), + Language::Html, + Policy::All, + Layout::Compact, + )) + .unwrap(), + "ab" + ); + let trailing = "

a

\n

b

\n"; + assert_eq!( + String::from_utf8(transformed( + trailing.as_bytes(), + Language::Html, + Policy::All, + Layout::Compact, + )) + .unwrap(), + "

a

\n

b

\n" + ); +} + +#[test] +fn a_unicode_line_terminator_ends_a_line_like_any_other() { + // NOTE: ECMA-262 12.3: U+2028 LINE SEPARATOR is a LineTerminator, so it + // NOTE: ends the comment, and the line it ended goes with the comment. + let source = "let a = 1;\u{2028}// note\u{2028}let b = 2;\n"; + assert_eq!( + String::from_utf8(transformed( + source.as_bytes(), + Language::JavaScript, + Policy::Safe, + Layout::Compact, + )) + .unwrap(), + "let a = 1;\u{2028}let b = 2;\n" + ); + // NOTE: `lines` keeps the emptied line and the space that kept the two + // NOTE: terminators from meeting. + assert_eq!( + String::from_utf8(transformed( + source.as_bytes(), + Language::JavaScript, + Policy::Safe, + Layout::Lines, + )) + .unwrap(), + "let a = 1;\u{2028} \u{2028}let b = 2;\n" + ); +} + +#[test] +fn a_kept_comment_holds_its_line_open() { + let source = "// rustfmt::skip\n// note\nfn main() {}\n"; + assert_eq!(compact(source), "// rustfmt::skip\nfn main() {}\n"); + let shared = "let x = 1; // rustfmt::skip\n/* note */\nlet y = 2;\n"; + assert_eq!(compact(shared), "let x = 1; // rustfmt::skip\nlet y = 2;\n"); +} + +#[test] +fn external_spans_with_blanks_between_them_stay_non_overlapping() { + // NOTE: A plugin or a declarative profile may report any spans the + // NOTE: validator accepts, including two with nothing but blanks between + // NOTE: them, and the edits still have to be sorted and non-overlapping. + let source = b"x\na \nb"; + let result = transform_spans( + source, + Language::Unknown, + &[ + (ByteSpan::new(2, 3), CommentKind::Line), + (ByteSpan::new(4, 5), CommentKind::Line), + ], + TransformOptions { + layout: Layout::Compact, + ..TransformOptions::default() + }, + ) + .expect("the spans are sorted, non-empty and inside the source"); + let mut cursor = 0; + for edit in &result.edits { + assert!( + edit.span.start >= cursor, + "edit {:?} overlaps its predecessor, which ended at {cursor}", + edit.span + ); + cursor = edit.span.end; + } + assert_eq!(apply_edits(source, &result.edits), result.output); + assert_eq!(result.output, b"x\n\nb"); +} + +/// The half-open span of `needle`, which must occur exactly once in `source`. +fn only_span(source: &[u8], needle: &[u8]) -> ByteSpan { + let mut found = source + .windows(needle.len()) + .enumerate() + .filter(|(_, window)| *window == needle) + .map(|(start, _)| start); + let start = found + .next() + .unwrap_or_else(|| panic!("`{}` is not in the source", String::from_utf8_lossy(needle))); + assert_eq!( + found.next(), + None, + "`{}` occurs more than once", + String::from_utf8_lossy(needle) + ); + ByteSpan::new(start, start + needle.len()) +} + +/// The hand-off gets the positional keep a built-in scan gets. +/// +/// A YAML block scalar reads the lines below it, so the comment that ends one +/// is not commentary: take its line and the kept directive under it is handed +/// back to the body. The bytes of that comment say nothing about this, so an +/// external scanner cannot classify it — `transform_spans` has to apply the +/// rule itself. It has to under every layout, because the least any of them can +/// leave in place of that line is a blank one, and a blank line is content of +/// the body above it whatever its indentation. +#[test] +fn external_spans_keep_the_comment_a_yaml_block_scalar_leans_on() { + let source = + b"a: 1 # trailing note\nk: |\n body\n# ends the block\n # yamllint disable\nz: 1\n"; + let trailing = only_span(source, b"# trailing note"); + let ends_block = only_span(source, b"# ends the block"); + let directive = only_span(source, b"# yamllint disable"); + // NOTE: All three layouts agree about the structural comment and differ + // NOTE: only over the trailing note: `columns` pads its width back, `lines` + // NOTE: leaves the space in front of it, `compact` trims that space away. + let expected: [(Layout, &[u8]); 3] = [ + ( + Layout::Lines, + b"a: 1 \nk: |\n body\n# ends the block\n # yamllint disable\nz: 1\n", + ), + ( + Layout::Columns, + b"a: 1 \nk: |\n body\n# ends the block\n # yamllint disable\nz: 1\n", + ), + ( + Layout::Compact, + b"a: 1\nk: |\n body\n# ends the block\n # yamllint disable\nz: 1\n", + ), + ]; + for (layout, output) in expected { + let result = transform_spans( + source, + Language::Yaml, + &[ + (trailing, CommentKind::Line), + (ends_block, CommentKind::Line), + (directive, CommentKind::Directive), + ], + TransformOptions { + layout, + ..TransformOptions::default() + }, + ) + .expect("the spans are sorted, non-empty and inside the source"); + assert_eq!( + result.report.comments[1].disposition, + Disposition::Keep { + reason: "structural in a YAML block scalar trail".into() + }, + "{layout:?} let the hand-off remove the comment the block scalar ends at" + ); + assert!( + result.edits.iter().all(|edit| edit.span != ends_block), + "{layout:?} emitted an edit for it anyway: {:?}", + result.edits + ); + // NOTE: The trailing note leans on nothing, so it goes: the pass is the + // NOTE: one keep the shape asks for, not a blanket amnesty for YAML. + assert_eq!( + result.edits.len(), + 1, + "{layout:?} edits: {:?}", + result.edits + ); + assert_eq!(result.edits[0].span.end, trailing.end, "{layout:?}"); + assert_eq!(result.output, output, "{layout:?}"); + assert_eq!( + apply_edits(source, &result.edits), + result.output, + "{layout:?}" + ); + } +} + +proptest! { + /// With every removed comment between two tokens on one line, `compact` + /// has no line to drop and no trailing whitespace to trim, so it must + /// leave exactly the bytes `lines` leaves. + #[test] + fn compact_equals_lines_when_no_comment_ends_its_line( + left in "[a-z]{1,8}", body in "[a-z ]{0,20}", right in "[a-z]{1,8}", tail in "[a-z]{1,8}") + { + let source = format!("{left}/*{body}*/{right}\n{tail}\n"); + prop_assert_eq!(compact(&source), lines(&source)); + } + + /// A comment alone on its line is the one case the two layouts differ + /// over, and they differ by exactly that line. + #[test] + fn compact_drops_the_line_that_lines_leaves_blank( + indent in " {0,6}", body in "[a-z ]{0,20}", head in "[a-z]{1,8}", tail in "[a-z]{1,8}") + { + let source = format!("{head}\n{indent}// note {body}\n{tail}\n"); + let compacted = compact(&source); + prop_assert_eq!(&compacted, &format!("{head}\n{tail}\n")); + prop_assert_eq!(lines(&source), format!("{head}\n{indent}\n{tail}\n")); + prop_assert!(compacted.len() < lines(&source).len()); + } +} diff --git a/rust/ocomment-core/tests/names.rs b/rust/ocomment-core/tests/names.rs new file mode 100644 index 0000000..6b61473 --- /dev/null +++ b/rust/ocomment-core/tests/names.rs @@ -0,0 +1,485 @@ +//! Stable-name contract for the public enums. +//! +//! `as_str` is the single source of truth for every user-visible spelling: it +//! must equal the serde name byte-for-byte, round-trip through `FromStr`, and +//! agree with `Display`. Every historical alias is pinned here so a refactor +//! cannot silently drop one. + +use ocomment_core::{ + CommentKind, Dialect, Disposition, Language, Layout, Policy, ScanOptions, Severity, scan, +}; +use std::{ + collections::{BTreeSet, HashSet}, + str::FromStr, +}; + +fn serde_name(value: &T) -> String { + serde_json::to_value(value) + .expect("enum serializes") + .as_str() + .expect("enum serializes as a string") + .to_owned() +} + +/// Every variant of `$type` agrees with serde, `FromStr`, and `Display`, and no +/// spelling is claimed by two variants. +macro_rules! check_stable_names { + ($type:ident) => {{ + let mut seen = BTreeSet::new(); + for value in $type::ALL { + assert_eq!( + serde_name(&value), + value.as_str(), + "{}::{value:?} serde name differs from as_str", + stringify!($type) + ); + assert_eq!( + $type::from_str(value.as_str()), + Ok(value), + "{}::{value:?} canonical name does not round-trip", + stringify!($type) + ); + assert_eq!( + value.to_string(), + value.as_str(), + "{}::{value:?} Display differs from as_str", + stringify!($type) + ); + assert!( + seen.insert(value.as_str()), + "{}::{value:?} name `{}` is claimed twice", + stringify!($type), + value.as_str() + ); + for alias in value.aliases() { + assert_eq!( + $type::from_str(alias), + Ok(value), + "{}::{value:?} alias `{alias}` does not parse", + stringify!($type) + ); + assert!( + seen.insert(alias), + "{}::{value:?} alias `{alias}` is claimed twice", + stringify!($type) + ); + } + } + seen + }}; +} + +#[test] +fn language_names_are_stable() { + let seen = check_stable_names!(Language); + assert_eq!(Language::ALL.len(), 30); + assert!( + !seen.contains("unknown"), + "Unknown must stay out of the parseable set" + ); + assert_eq!(serde_name(&Language::Unknown), "unknown"); + assert_eq!(Language::Unknown.as_str(), "unknown"); + assert_eq!(Language::Unknown.to_string(), "unknown"); + assert!(!Language::ALL.contains(&Language::Unknown)); +} + +#[test] +fn dialect_names_are_stable() { + check_stable_names!(Dialect); + assert_eq!(Dialect::ALL.len(), 17); +} + +#[test] +fn comment_kind_names_are_stable() { + check_stable_names!(CommentKind); + assert_eq!(CommentKind::ALL.len(), 11); +} + +#[test] +fn policy_names_are_stable() { + check_stable_names!(Policy); + assert_eq!(Policy::ALL.len(), 3); +} + +#[test] +fn layout_names_are_stable() { + check_stable_names!(Layout); + assert_eq!(Layout::ALL.len(), 3); +} + +#[test] +fn severity_names_are_stable() { + check_stable_names!(Severity); + assert_eq!(Severity::ALL.len(), 4); +} + +/// Every spelling [`Language::from_str`] accepts, written out rather than +/// generated, so that a rename shows up here as a changed line. +/// +/// The table is also checked *against* [`Language::aliases`] below: a language +/// added without a row, or an alias added to that function and nowhere else, +/// fails here instead of shipping unpinned. +#[test] +fn language_aliases_are_pinned() { + let cases = [ + ("rust", Language::Rust), + ("rs", Language::Rust), + ("ocaml", Language::Ocaml), + ("ml", Language::Ocaml), + ("c", Language::C), + ("cpp", Language::Cpp), + ("c++", Language::Cpp), + ("cxx", Language::Cpp), + ("go", Language::Go), + ("golang", Language::Go), + ("java", Language::Java), + ("javascript", Language::JavaScript), + ("js", Language::JavaScript), + ("jsx", Language::JavaScript), + ("ecmascript", Language::JavaScript), + ("typescript", Language::TypeScript), + ("ts", Language::TypeScript), + ("tsx", Language::TypeScript), + ("python", Language::Python), + ("py", Language::Python), + ("shell", Language::Shell), + ("sh", Language::Shell), + ("bash", Language::Shell), + ("zsh", Language::Shell), + ("html", Language::Html), + ("htm", Language::Html), + ("css", Language::Css), + ("jsonc", Language::Jsonc), + ("json5", Language::Jsonc), + ("sql", Language::Sql), + ("kotlin", Language::Kotlin), + ("kt", Language::Kotlin), + ("kts", Language::Kotlin), + ("toml", Language::Toml), + ("lua", Language::Lua), + ("yaml", Language::Yaml), + ("yml", Language::Yaml), + ("php", Language::Php), + ("ruby", Language::Ruby), + ("rb", Language::Ruby), + ("zig", Language::Zig), + ("r", Language::R), + ("rscript", Language::R), + ("dart", Language::Dart), + ("swift", Language::Swift), + ("csharp", Language::CSharp), + ("cs", Language::CSharp), + ("c#", Language::CSharp), + ("scala", Language::Scala), + ("vue", Language::Vue), + ("svelte", Language::Svelte), + ("markdown", Language::Markdown), + ("perl", Language::Perl), + ]; + for (text, expected) in cases { + assert_eq!(Language::from_str(text), Ok(expected), "`{text}`"); + } + // NOTE: The other direction. `cases` is what pins the spellings, so every + // NOTE: canonical name and every alias the crate publishes has to be one of + // NOTE: its rows -- otherwise a new language, or a new alias for an old + // NOTE: one, would be accepted by `from_str` with nothing holding it there. + let pinned: HashSet<(&str, Language)> = cases.into_iter().collect(); + let mut missing = Vec::new(); + for language in Language::ALL { + for spelling in std::iter::once(language.as_str()).chain(language.aliases().iter().copied()) + { + if !pinned.contains(&(spelling, language)) { + missing.push(format!("(\"{spelling}\", Language::{language:?})")); + } + } + } + assert!( + missing.is_empty(), + "these spellings are accepted by `Language::from_str` but pinned by no row \ + of this table: {missing:?}" + ); +} + +#[test] +fn language_parsing_ignores_case_dashes_and_underscores() { + for text in ["RUST", "Rust", "-r-u-s-t-", "r_u_s_t"] { + assert_eq!(Language::from_str(text), Ok(Language::Rust), "`{text}`"); + } + assert_eq!( + Language::from_str("Java_Script"), + Ok(Language::JavaScript), + "underscores are stripped" + ); + assert_eq!(Language::from_str("C++"), Ok(Language::Cpp)); +} + +#[test] +fn dialect_aliases_are_pinned() { + let cases = [ + ("standard", Dialect::Standard), + ("jsx", Dialect::Jsx), + ("tsx", Dialect::Tsx), + ("objective-c", Dialect::ObjectiveC), + ("objc", Dialect::ObjectiveC), + ("objective-cpp", Dialect::ObjectiveCpp), + ("objective-c++", Dialect::ObjectiveCpp), + ("objcpp", Dialect::ObjectiveCpp), + ("gnu-c", Dialect::GnuC), + ("gnuc", Dialect::GnuC), + ("gnu-cpp", Dialect::GnuCpp), + ("gnu-c++", Dialect::GnuCpp), + ("gnucpp", Dialect::GnuCpp), + ("cuda", Dialect::Cuda), + ("posix-sh", Dialect::PosixSh), + ("posix", Dialect::PosixSh), + ("sh", Dialect::PosixSh), + ("bash53", Dialect::Bash53), + ("bash-5.3", Dialect::Bash53), + ("bash", Dialect::Bash53), + ("zsh", Dialect::Zsh), + ("postgresql", Dialect::PostgreSql), + ("postgres", Dialect::PostgreSql), + ("pgsql", Dialect::PostgreSql), + ("mysql", Dialect::MySql), + ("sqlite", Dialect::Sqlite), + ("t-sql", Dialect::TSql), + ("tsql", Dialect::TSql), + ("oracle", Dialect::Oracle), + ]; + for (text, expected) in cases { + assert_eq!(Dialect::from_str(text), Ok(expected), "`{text}`"); + } +} + +#[test] +fn dialect_parsing_folds_case_and_underscores() { + assert_eq!(Dialect::from_str("Objective_C"), Ok(Dialect::ObjectiveC)); + assert_eq!(Dialect::from_str("GNU-CPP"), Ok(Dialect::GnuCpp)); + assert_eq!(Dialect::from_str("bash_5.3"), Ok(Dialect::Bash53)); + assert_eq!(Dialect::from_str("T_SQL"), Ok(Dialect::TSql)); +} + +#[test] +fn comment_kind_aliases_are_pinned() { + let cases = [ + ("line", CommentKind::Line), + ("block", CommentKind::Block), + ("doc-line", CommentKind::DocLine), + ("doc", CommentKind::DocLine), + ("doc-block", CommentKind::DocBlock), + ("directive", CommentKind::Directive), + ("pragma", CommentKind::Directive), + ("license", CommentKind::License), + ("legal", CommentKind::License), + ("html", CommentKind::HtmlComment), + ("html-comment", CommentKind::HtmlComment), + ("shebang", CommentKind::Shebang), + ("encoding", CommentKind::Encoding), + ("optimizer-hint", CommentKind::OptimizerHint), + ("version-comment", CommentKind::VersionComment), + ]; + for (text, expected) in cases { + assert_eq!(CommentKind::from_str(text), Ok(expected), "`{text}`"); + } +} + +#[test] +fn comment_kind_parsing_folds_case_and_underscores() { + assert_eq!( + CommentKind::from_str("DOC_BLOCK"), + Ok(CommentKind::DocBlock) + ); + assert_eq!( + CommentKind::from_str("Optimizer_Hint"), + Ok(CommentKind::OptimizerHint) + ); + assert_eq!( + CommentKind::from_str("HTML_COMMENT"), + Ok(CommentKind::HtmlComment) + ); +} + +#[test] +fn policy_and_layout_aliases_are_pinned() { + assert_eq!(Policy::from_str("safe"), Ok(Policy::Safe)); + assert_eq!(Policy::from_str("legal"), Ok(Policy::Legal)); + assert_eq!(Policy::from_str("all"), Ok(Policy::All)); + assert_eq!(Policy::from_str("SAFE"), Ok(Policy::Safe)); + assert_eq!(Layout::from_str("lines"), Ok(Layout::Lines)); + assert_eq!(Layout::from_str("columns"), Ok(Layout::Columns)); + assert_eq!(Layout::from_str("compact"), Ok(Layout::Compact)); + assert_eq!(Layout::from_str("Compact"), Ok(Layout::Compact)); + assert!(Policy::ALL.iter().all(|value| value.aliases().is_empty())); + assert!(Layout::ALL.iter().all(|value| value.aliases().is_empty())); +} + +#[test] +fn rejection_messages_are_unchanged() { + assert_eq!( + Language::from_str("unknown"), + Err("unsupported language `unknown`".to_owned()) + ); + assert_eq!( + Language::from_str("Klingon"), + Err("unsupported language `Klingon`".to_owned()) + ); + assert_eq!( + Dialect::from_str("mariadb"), + Err("unknown dialect `mariadb`".to_owned()) + ); + assert_eq!( + CommentKind::from_str("footnote"), + Err("unknown comment kind `footnote`".to_owned()) + ); + assert_eq!( + Policy::from_str("paranoid"), + Err("unknown policy `paranoid`".to_owned()) + ); + assert_eq!( + Layout::from_str("grid"), + Err("unknown layout `grid`".to_owned()) + ); +} + +#[test] +fn disposition_display_is_human_readable() { + assert_eq!(Disposition::Remove.to_string(), "remove"); + assert_eq!( + Disposition::Keep { + reason: "legal policy".to_owned() + } + .to_string(), + "keep (legal policy)" + ); +} + +#[test] +fn disposition_serde_shape_is_frozen() { + assert_eq!( + serde_json::to_value(Disposition::Remove).unwrap(), + serde_json::json!({"action": "remove"}) + ); + assert_eq!( + serde_json::to_value(Disposition::Keep { + reason: "legal policy".to_owned() + }) + .unwrap(), + serde_json::json!({"action": "keep", "reason": "legal policy"}) + ); +} + +/// The differential protocol freezes these six strings; the OCaml reference +/// compares them byte-for-byte. +const KEEP_REASONS: [&str; 6] = [ + "kept by kind or regex override", + "required source preamble", + "HTML comments are DOM-observable", + "tool or language directive", + "legal policy", + "structural in a YAML block scalar trail", +]; + +/// One fixture for `keep_reasons_are_observable_through_scan`: a source, how it +/// is scanned, how many comments it holds, and which of them carries the frozen +/// reason under test. The count is pinned per fixture so a scanner that started +/// finding a comment more or fewer fails here rather than sliding the index. +struct ReasonFixture { + source: &'static [u8], + language: Language, + options: ScanOptions, + comments: usize, + index: usize, + reason: &'static str, +} + +#[test] +fn keep_reasons_are_observable_through_scan() { + let cases = [ + ReasonFixture { + source: b"// keep me\n", + language: Language::Rust, + options: ScanOptions { + keep_kinds: vec![CommentKind::Line], + ..Default::default() + }, + comments: 1, + index: 0, + reason: "kept by kind or regex override", + }, + ReasonFixture { + source: b"#!/bin/sh\n", + language: Language::Shell, + options: ScanOptions::default(), + comments: 1, + index: 0, + reason: "required source preamble", + }, + ReasonFixture { + source: b"\n", + language: Language::Html, + options: ScanOptions::default(), + comments: 1, + index: 0, + reason: "HTML comments are DOM-observable", + }, + ReasonFixture { + source: b"// rustfmt::skip\n", + language: Language::Rust, + options: ScanOptions::default(), + comments: 1, + index: 0, + reason: "tool or language directive", + }, + ReasonFixture { + source: b"// Copyright 2026 Example\n", + language: Language::Rust, + options: ScanOptions { + policy: Policy::Legal, + ..Default::default() + }, + comments: 1, + index: 0, + reason: "legal policy", + }, + /* NOTE: The one reason that needs a second comment to exist at all: the + * block scalar leans on the first comment only because the directive + * below it survives and is indented into the body. */ + ReasonFixture { + source: b"k: |\n a\n# ends the block\n # yamllint disable\nz: 1\n", + language: Language::Yaml, + options: ScanOptions::default(), + comments: 2, + index: 0, + reason: "structural in a YAML block scalar trail", + }, + ]; + let mut observed = BTreeSet::new(); + for case in cases { + observed.insert(case.reason); + let report = scan(case.source, case.language, case.options); + assert_eq!( + report.comments.len(), + case.comments, + "`{}` fixture found {:?}", + case.reason, + report.comments + ); + assert_eq!( + report.comments[case.index].disposition, + Disposition::Keep { + reason: case.reason.to_owned() + }, + "`{}` fixture", + case.reason + ); + assert_eq!( + report.comments[case.index].disposition.to_string(), + format!("keep ({})", case.reason) + ); + } + assert_eq!( + observed, + BTreeSet::from(KEEP_REASONS), + "the fixtures no longer exercise every frozen keep reason" + ); +} diff --git a/rust/ocomment-core/tests/properties.proptest-regressions b/rust/ocomment-core/tests/properties.proptest-regressions index db46a2a..357568a 100644 --- a/rust/ocomment-core/tests/properties.proptest-regressions +++ b/rust/ocomment-core/tests/properties.proptest-regressions @@ -5,3 +5,4 @@ # It is recommended to check this file in to source control so that # everyone who runs the test benefits from these saved cases. cc 82a94048b80c617d96de131fb906d6ccbd0ad11ce86ba8a5020e6a241aad96a6 # shrinks to source = [0, 0, 0, 0, 0, 42, 0, 0, 0, 0, 39, 128, 10, 34, 39, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0], replacement = [], first = 7458001845501105093, second = 8922367292645105975 +cc 139e96148e557b3279c30b38950e0c1adfdea4d3e05d82059df8362c86852ad0 # shrinks to source = [0, 0, 35, 0, 39, 128, 34, 39, 10, 39, 0, 35, 35, 0, 0, 35, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0], replacement = [128], first = 14929730563465487330, second = 5900713899202105475 diff --git a/rust/ocomment-core/tests/properties.rs b/rust/ocomment-core/tests/properties.rs index 997df5b..5db7c2b 100644 --- a/rust/ocomment-core/tests/properties.rs +++ b/rust/ocomment-core/tests/properties.rs @@ -1,8 +1,53 @@ +//! Randomised properties the engine holds for every input. +//! +//! The generators favour the bytes that open and close lexical states, so +//! the cases are unlikely rather than merely random. + use ocomment_core::{ ByteSpan, DocumentChange, IncrementalDocument, Language, Layout, ScanOptions, TransformOptions, - scan, transform, + lexical_pool, scan, transform, }; -use proptest::prelude::*; +use proptest::{prelude::*, sample::select}; + +/// A pool length as a `prop_oneof!` weight, so that drawing uniformly from a +/// pool of `n` gives each of its members the weight one arm would have. +fn weight(length: usize) -> u32 { + u32::try_from(length).expect("the pool is far smaller than a weight") +} + +/// One byte of the shared pool, or a uniformly random one. +/// +/// The pool is `ocomment_core::lexical_pool::BYTES`, and the checkpoint +/// properties in `src/incremental.rs` draw from the same one: a fragment worth +/// generating against the whole-file scanner is worth generating against the +/// incremental one. The extra `\n` arm doubles that byte's weight, because a +/// line boundary is where most of the interesting lexical states begin and end. +fn lexical_byte() -> impl Strategy { + prop_oneof![ + 4 => any::(), + 1 => Just(b'\n'), + weight(lexical_pool::BYTES.len()) => select(lexical_pool::BYTES), + ] +} + +/// A fragment: one byte of the pool, or one whole token from it. +/// +/// The tokens are `ocomment_core::lexical_pool::TOKENS` — multi-byte openers a +/// single-byte alphabet can never synthesise, and the reason each of them is +/// there is written out beside the list. Each is drawn as often as one byte is, +/// which is what the eight-to-one weight in front of the byte arm keeps in +/// proportion. +fn lexical_fragment() -> impl Strategy> { + prop_oneof![ + 8 => lexical_byte().prop_map(|byte| vec![byte]), + weight(lexical_pool::TOKENS.len()) => select(lexical_pool::TOKENS).prop_map(<[u8]>::to_vec), + ] +} + +/// A source built from at most `fragments` raw bytes and literal tokens. +fn lexical_source(fragments: std::ops::Range) -> impl Strategy> { + prop::collection::vec(lexical_fragment(), fragments).prop_map(|fragments| fragments.concat()) +} fn newlines(bytes: &[u8]) -> Vec { bytes @@ -13,7 +58,7 @@ fn newlines(bytes: &[u8]) -> Vec { } proptest! { - #![proptest_config(ProptestConfig::with_cases(256))] + #![proptest_config(ProptestConfig::default())] #[test] fn lines_layout_keeps_the_exact_newline_sequence(body in "[A-Za-z0-9 \\t\\r\\n]{0,200}") { @@ -68,20 +113,8 @@ proptest! { #[test] fn arbitrary_incremental_edits_match_full_scans_for_every_builtin( - source in prop::collection::vec(prop_oneof![ - 4 => any::(), - 2 => Just(b'\n'), - 1 => Just(b'\r'), - 1 => Just(b'/'), - 1 => Just(b'*'), - 1 => Just(b'\''), - 1 => Just(b'"'), - 1 => Just(b'#'), - 1 => Just(b'`'), - 1 => Just(b'{'), - 1 => Just(b'}'), - ], 0..96), - replacement in prop::collection::vec(any::(), 0..32), + source in lexical_source(0..48), + replacement in lexical_source(0..8), first in any::(), second in any::(), ) { @@ -89,7 +122,7 @@ proptest! { let left = first % modulus; let right = second % modulus; let span = ByteSpan::new(left.min(right), left.max(right)); - for language in Language::BUILT_INS { + for language in Language::ALL { let mut document = IncrementalDocument::new( source.clone(), language, @@ -117,3 +150,50 @@ proptest! { prop_assert_eq!(newlines(&source), newlines(&result.output)); } } + +/// The two counterexamples recorded in `properties.proptest-regressions` were +/// drawn from the single-byte alphabet this file's generator no longer uses on +/// its own, so proptest can no longer replay them from their seeds. They are +/// kept here verbatim instead: both are unterminated Rust character literals +/// whose six-byte lookahead straddles the rescan window. +#[test] +fn recorded_counterexamples_still_match_a_full_scan_for_every_builtin() { + let cases: [(&[u8], ByteSpan, &[u8]); 2] = [ + ( + &[ + 0, 0, 0, 0, 0, 42, 0, 0, 0, 0, 39, 128, 10, 34, 39, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, + 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, + ], + ByteSpan::new(13, 15), + &[], + ), + ( + &[ + 0, 0, 35, 0, 39, 128, 34, 39, 10, 39, 0, 35, 35, 0, 0, 35, 0, 0, 0, 0, 0, 0, 0, 0, + 0, 0, + ], + ByteSpan::new(5, 8), + &[128], + ), + ]; + for (source, span, replacement) in cases { + for language in Language::ALL { + let mut document = + IncrementalDocument::new(source.to_vec(), language, ScanOptions::default(), 1); + document + .apply_changes( + &[DocumentChange { + span, + replacement: replacement.to_vec(), + }], + 2, + ) + .unwrap(); + assert_eq!( + document.report(), + &scan(document.source(), language, ScanOptions::default()), + "incremental mismatch for {language} at {span:?}", + ); + } + } +} diff --git a/rust/ocomment-core/tests/source_guards.rs b/rust/ocomment-core/tests/source_guards.rs new file mode 100644 index 0000000..d88679b --- /dev/null +++ b/rust/ocomment-core/tests/source_guards.rs @@ -0,0 +1,364 @@ +//! Guards that read this crate's own source text. +//! +//! A property test can only find what its generator can draw. These read the +//! generators instead, so an alphabet that is shared today cannot be forked +//! quietly by a line added tomorrow. + +use std::collections::BTreeSet; + +/// The two files that build sources for a property test, embedded at compile +/// time so the scan does not depend on the directory the test runs in. +/// +/// They are the pair `crate::lexical_pool` exists for: the checkpoint and +/// incremental properties inside the crate, and the whole-file properties +/// outside it. Both draw from the same alphabet on purpose, because a fragment +/// worth generating against the whole-file scanner is worth generating against +/// the incremental one. +const GENERATOR_SOURCES: [(&str, &str); 2] = [ + ("src/incremental.rs", include_str!("../src/incremental.rs")), + ("tests/properties.rs", include_str!("properties.rs")), +]; + +/// The generators whose arms have to come out of the shared pool, in both +/// files. `lexical_byte` draws one byte and `lexical_fragment` one byte or one +/// whole token, and between them they are the entire alphabet either suite +/// generates from. +const SOURCE_GENERATORS: [&str; 2] = ["lexical_byte", "lexical_fragment"]; + +/// The one other function allowed to hold a `prop_oneof!`. +/// +/// `edit_endpoint` draws an offset into a document rather than a byte of one, +/// so its `Just(0usize)` and `Just(usize::MAX)` arms name positions and not +/// source text; they cannot put a delimiter in front of a scanner and are +/// therefore no part of the alphabet. It is exempt by name so that a *third* +/// generator of source bytes cannot appear beside the two without failing here. +const OFFSET_GENERATORS: [&str; 1] = ["edit_endpoint"]; + +/// The paths the pool is reachable under. `src/incremental.rs` is inside the +/// crate and `tests/properties.rs` outside it, so the same two constants are +/// spelled differently in the two files and neither spelling may be the only +/// one accepted. +const POOL_PATHS: [&str; 2] = ["crate::lexical_pool::", "lexical_pool::"]; + +/// The pool constants themselves, which are also the only names an arm may +/// draw from. +const POOLS: [&str; 2] = ["BYTES", "TOKENS"]; + +/// The single arm allowed to name a byte of its own, and the reason it is: +/// `\n` is already in [`ocomment_core::lexical_pool::BYTES`], and repeating it +/// as an arm doubles its weight rather than adding a byte the pool lacks. Every +/// other literal would be an alphabet one suite has and the other does not. +const NEWLINE_ARM: &str = "Just(b'\\n')"; + +/// The macro this guard reads. +/// +/// It is matched with the bracket that opens its arms rather than on the name +/// alone, so that the prose which merely names it is not read as an +/// invocation. All three of `proptest`'s spellings count: the macro is +/// `macro_rules!`, so `prop_oneof![...]`, `prop_oneof!(...)` and +/// `prop_oneof!{...}` are one and the same invocation and a guard that read +/// only the first would be blind to a generator written with either other. +const MACRO: &str = "prop_oneof!"; + +/// The brackets an invocation of [`MACRO`] may be written with, paired with +/// what closes each. +const MACRO_BRACKETS: [(char, char); 3] = [('[', ']'), ('(', ')'), ('{', '}')]; + +/// One arm of a `prop_oneof!`, with the weight and the strategy separated and +/// the line wrapping taken out. +#[derive(Debug)] +struct Arm { + /// The file the arm was read from. + file: &'static str, + /// The generator it belongs to. + function: String, + /// The strategy expression, whitespace collapsed to single spaces. + strategy: String, +} + +/// Every arm of the two source generators draws from +/// [`ocomment_core::lexical_pool`], from the uniform-random byte beside it, or +/// is the one `\n` arm that reweights a byte the pool already holds. +/// +/// The invariant is textual, and deliberately so: nothing at run time can ask a +/// `Strategy` what alphabet it came from. What a fork would look like is a +/// `Just(b'%')` or a `select(&[...])` added to one file when a language needs a +/// new opener — which is exactly the edit that must go into `lexical_pool` +/// instead, where the other suite gets it too and where the reason for it is +/// written down next to it. +#[test] +fn every_generated_source_byte_comes_from_the_shared_pool() { + let mut offenders = Vec::new(); + let mut newline_arms = 0; + let mut pool_arms = 0; + for arm in source_generator_arms() { + if arm.strategy == NEWLINE_ARM { + newline_arms += 1; + continue; + } + if arm.strategy.contains('\'') || arm.strategy.contains('"') { + offenders.push(format!( + "{}: `{}` in `{}` spells a literal of its own", + arm.file, arm.strategy, arm.function + )); + continue; + } + if draws_from_pool(&arm.strategy) { + pool_arms += 1; + continue; + } + if arm.strategy.starts_with("any::<") || arm.strategy.starts_with("lexical_byte()") { + continue; + } + offenders.push(format!( + "{}: `{}` in `{}` draws from neither the pool nor `any`", + arm.file, arm.strategy, arm.function + )); + } + assert!( + offenders.is_empty(), + "these `prop_oneof!` arms build source bytes outside \ + `ocomment_core::lexical_pool`, so the two property suites no longer \ + generate from one alphabet: {offenders:?}" + ); + // NOTE: A reader that matched nothing would pass the loop above forever. + assert_eq!( + newline_arms, + GENERATOR_SOURCES.len(), + "each file reweights `\\n` exactly once, and the guard found {newline_arms} such arm(s)" + ); + assert_eq!( + pool_arms, + GENERATOR_SOURCES.len() * POOLS.len(), + "each file draws from both pools exactly once, and the guard found {pool_arms} such arm(s)" + ); +} + +/// Both pool constants are drawn in both files, so neither suite can quietly +/// stop generating whole tokens — the multi-byte openers a single-byte alphabet +/// can never synthesise — while still passing the arm check above. +#[test] +fn both_files_draw_from_both_pools() { + for (file, source) in GENERATOR_SOURCES { + for pool in POOLS { + assert!( + POOL_PATHS + .iter() + .any(|path| source.contains(&format!("{path}{pool}"))), + "{file} never names `lexical_pool::{pool}`" + ); + } + } +} + +/// Every `prop_oneof!` in the two files sits in a generator this guard knows. +/// A new one somewhere else would be an alphabet the arm check never reads. +#[test] +fn the_guard_reads_every_prop_oneof() { + let allowed: BTreeSet<&str> = SOURCE_GENERATORS + .into_iter() + .chain(OFFSET_GENERATORS) + .collect(); + let mut found = 0; + for (file, source) in GENERATOR_SOURCES { + for at in macro_sites(source) { + let function = enclosing_function(source, at); + assert!( + allowed.contains(function.as_str()), + "{file}: `prop_oneof!` in `{function}`, which this guard does not read" + ); + found += 1; + } + } + assert!( + found >= GENERATOR_SOURCES.len() * SOURCE_GENERATORS.len(), + "only {found} `prop_oneof!` site(s) were found, fewer than the two generators \ + each of the two files declares" + ); +} + +/// Every offset in `source` where [`MACRO`] is invoked, whichever of +/// [`MACRO_BRACKETS`] the invocation is written with. A mention with no bracket +/// behind it is prose and no site. +fn macro_sites(source: &str) -> Vec { + source + .match_indices(MACRO) + .filter(|(at, _)| { + source[at + MACRO.len()..] + .trim_start() + .starts_with(is_macro_opener) + }) + .map(|(at, _)| at) + .collect() +} + +/// Whether `character` opens the arms of an invocation. +fn is_macro_opener(character: char) -> bool { + MACRO_BRACKETS + .iter() + .any(|(opener, _)| *opener == character) +} + +/// Whether `character` is either half of one of [`MACRO_BRACKETS`], which is +/// what the arm reader counts depth over. +fn bracket_depth(character: char) -> i32 { + if is_macro_opener(character) { + 1 + } else if MACRO_BRACKETS + .iter() + .any(|(_, closer)| *closer == character) + { + -1 + } else { + 0 + } +} + +/// `proptest` accepts `prop_oneof!` written with any of the three bracket +/// pairs, and this guard has to see all three: a generator added as +/// `prop_oneof!( ... )` that the reader never matched would be an alphabet the +/// arm check never reads *and* would slip past +/// [`the_guard_reads_every_prop_oneof`], which can only complain about the +/// sites it finds. +#[test] +fn the_guard_reads_every_bracket_a_prop_oneof_may_be_written_with() { + for (opener, closer) in [('[', ']'), ('(', ')'), ('{', '}')] { + let sample = format!( + "fn forked() -> impl Strategy {{\n \ + prop_oneof!{opener}\n 1 => Just(b'%'),\n \ + 2 => any::(),\n {closer}\n}}\n" + ); + let sites = macro_sites(&sample); + assert_eq!( + sites, + vec![sample.find(MACRO).expect("the sample invokes the macro")], + "`prop_oneof!{opener}` was not read as an invocation" + ); + assert_eq!(enclosing_function(&sample, sites[0]), "forked"); + assert_eq!( + arm_strategies(&sample, sites[0]), + vec!["Just(b'%')", "any::()"], + "`prop_oneof!{opener}` arms were not read" + ); + } + // NOTE: Prose that names the macro without invoking it is not a site, which + // NOTE: is what lets the doc comments in both files go on naming it. + assert!(macro_sites("/// A pool length as a `prop_oneof!` weight.\n").is_empty()); +} + +/// Every arm of every source generator, across both files. +fn source_generator_arms() -> Vec { + let mut arms = Vec::new(); + for (file, source) in GENERATOR_SOURCES { + for at in macro_sites(source) { + let function = enclosing_function(source, at); + if !SOURCE_GENERATORS.contains(&function.as_str()) { + continue; + } + for strategy in arm_strategies(source, at) { + arms.push(Arm { + file, + function: function.clone(), + strategy, + }); + } + } + } + assert!(!arms.is_empty(), "no `prop_oneof!` arm was read at all"); + arms +} + +/// Whether a strategy expression selects one of the shared pools and nothing +/// else. `select` is the only way either generator reaches a pool, so an arm +/// that names a pool without selecting from it is not one of these. +fn draws_from_pool(strategy: &str) -> bool { + POOL_PATHS.iter().any(|path| { + POOLS + .iter() + .any(|pool| strategy.starts_with(&format!("select({path}{pool})"))) + }) +} + +/// The name of the function byte `at` falls inside: the last line at or before +/// it that opens a `fn`. Both files declare every generator at the head of its +/// own line, which is what makes this exact rather than a guess. +fn enclosing_function(source: &str, at: usize) -> String { + let mut name = String::new(); + let mut offset = 0; + for line in source.split_inclusive('\n') { + if offset > at { + break; + } + if let Some(rest) = line.trim_start().strip_prefix("fn ") { + name = rest + .split(['(', '<']) + .next() + .unwrap_or_default() + .trim() + .to_owned(); + } + offset += line.len(); + } + name +} + +/// The strategy expression of each arm of the `prop_oneof!` beginning at `at`, +/// with its line wrapping collapsed to single spaces. +/// +/// The macro's arms are `weight => strategy`, separated by commas that a +/// generic argument or a nested call may also contain, so the split is made at +/// bracket depth zero and nowhere else. Depth counts all three of +/// [`MACRO_BRACKETS`], because the outermost pair is whichever one the +/// invocation was written with. +fn arm_strategies(source: &str, at: usize) -> Vec { + let open = at + + source[at..] + .find(is_macro_opener) + .expect("`prop_oneof!` is written with a bracket"); + let mut depth = 0; + let mut end = open; + for (index, character) in source[open..].char_indices() { + depth += bracket_depth(character); + if depth == 0 && bracket_depth(character) < 0 { + end = open + index; + break; + } + } + assert!(end > open, "the `prop_oneof!` at {at} is never closed"); + let mut strategies = Vec::new(); + for arm in split_top_level(&source[open + 1..end]) { + let arm = arm.trim(); + if arm.is_empty() { + continue; + } + let strategy = arm + .split_once("=>") + .unwrap_or_else(|| panic!("`{arm}` is not a `weight => strategy` arm")) + .1; + strategies.push(collapsed(strategy)); + } + assert!(!strategies.is_empty(), "the `prop_oneof!` at {at} is empty"); + strategies +} + +/// `body` split on the commas that sit outside every bracket pair. +fn split_top_level(body: &str) -> Vec<&str> { + let mut parts = Vec::new(); + let mut depth = 0; + let mut start = 0; + for (index, character) in body.char_indices() { + depth += bracket_depth(character); + if character == ',' && depth == 0 { + parts.push(&body[start..index]); + start = index + 1; + } + } + parts.push(&body[start..]); + parts +} + +/// One expression with every run of whitespace collapsed to a single space, so +/// an arm that wraps across lines reads as the one expression it is. +fn collapsed(text: &str) -> String { + text.split_whitespace().collect::>().join(" ") +} diff --git a/rust/ocomment-core/tests/spec_fixtures.rs b/rust/ocomment-core/tests/spec_fixtures.rs new file mode 100644 index 0000000..94a34c9 --- /dev/null +++ b/rust/ocomment-core/tests/spec_fixtures.rs @@ -0,0 +1,526 @@ +//! The shared fixture corpus in `spec/fixtures/v1`, run against this crate. +//! +//! `tools/differential.py` feeds the same corpus to this implementation and to +//! the OCaml reference and compares the two. That says the pair agree; it does +//! not say what they agree on, and it needs a built OCaml tree to say anything +//! at all. This test is the other half: it runs every case here, checks the +//! ones carrying a recorded `expect` block against it, and holds the whole +//! corpus to the engine's structural promises — a scan never panics, and the +//! edits of a transformation are sorted, non-overlapping, and reproduce the +//! output when applied in one pass. +//! +//! `spec/fixtures/README.md` documents the case schema and how an `expect` +//! block is recorded. + +use ocomment_core::{ + ByteSpan, CommentKind, DeclarativeProfile, Disposition, Edit, Language, Layout, ScanReport, + TransformOptions, TransformResult, apply_edits, scan, scan_profile, transform, + transform_profile, transform_spans, +}; +use serde_json::{Value, json}; +use std::{collections::BTreeSet, fs, path::PathBuf, str::FromStr}; + +// INVARIANT: The floors live in `spec/fixtures/v1/floor.txt`, which +// INVARIANT: `tools/differential.py` reads too, so a case deleted from the +// INVARIANT: corpus fails the Rust test suite as well — on a machine with no +// INVARIANT: OCaml toolchain — and neither runner can be raised or lowered on +// INVARIANT: its own. `cases` is the least number of cases the corpus may hold; +// INVARIANT: `expectations` is the least number of those that must carry a +// INVARIANT: recorded `expect` block. A case with none is still held to the +// INVARIANT: structural promises below, so the second floor is what stops the +// INVARIANT: corpus from quietly degrading into that weaker check. Deleting a +// INVARIANT: block to re-record it is the documented way to change a recorded +// INVARIANT: behaviour, and `differential.py --record` puts it back before this +// INVARIANT: test is meant to run again. +const FLOOR_FILE: &str = "floor.txt"; + +/// One floor from `floor.txt`, which holds a `#` comment or a name and a +/// decimal count per line. A missing name is a failure rather than a zero: the +/// file is the only place either runner reads these numbers from. +fn floor(name: &str) -> usize { + let path = corpus_directory().join(FLOOR_FILE); + let text = std::fs::read_to_string(&path) + .unwrap_or_else(|error| panic!("read {}: {error}", path.display())); + for (number, line) in text.lines().enumerate() { + let trimmed = line.trim(); + if trimmed.is_empty() || trimmed.starts_with('#') { + continue; + } + let mut fields = trimmed.split_whitespace(); + let (Some(key), Some(count), None) = (fields.next(), fields.next(), fields.next()) else { + panic!( + "{FLOOR_FILE}:{}: expected `name count`, got {line:?}", + number + 1 + ); + }; + let count: usize = count.parse().unwrap_or_else(|error| { + panic!( + "{FLOOR_FILE}:{}: {count:?} is not a count: {error}", + number + 1 + ) + }); + if key == name { + return count; + } + } + panic!("{FLOOR_FILE}: no `{name}` floor"); +} + +/// One comment as an `expect` block records it, and as the checks compare it. +#[derive(Debug, Eq, PartialEq)] +struct ExpectedComment { + start: usize, + end: usize, + kind: String, + action: String, +} + +/// One diagnostic as an `expect` block records it. +#[derive(Debug, Eq, PartialEq)] +struct ExpectedDiagnostic { + code: String, + start: usize, + end: usize, +} + +/// Where the corpus and its `floor.txt` live, relative to this crate. +fn corpus_directory() -> PathBuf { + PathBuf::from(env!("CARGO_MANIFEST_DIR")).join("../../spec/fixtures/v1") +} + +/// Every corpus document, in file-name order, with the file each case came from. +fn corpus() -> Vec<(String, Value)> { + let directory = corpus_directory(); + let mut paths: Vec<_> = fs::read_dir(&directory) + .unwrap_or_else(|error| panic!("read {}: {error}", directory.display())) + .map(|entry| entry.expect("corpus directory entry").path()) + .filter(|path| path.extension().is_some_and(|value| value == "json")) + .collect(); + paths.sort(); + assert!( + !paths.is_empty(), + "no corpus documents in {}", + directory.display() + ); + paths + .into_iter() + .map(|path| { + let name = path + .file_name() + .expect("corpus file name") + .to_string_lossy() + .into_owned(); + let text = + fs::read_to_string(&path).unwrap_or_else(|error| panic!("read {name}: {error}")); + let document: Value = + serde_json::from_str(&text).unwrap_or_else(|error| panic!("parse {name}: {error}")); + assert_eq!( + document["version"], + json!(1), + "{name}: unsupported corpus version" + ); + (name, document) + }) + .collect() +} + +/// Every case in the corpus, paired with the document it came from. +fn cases() -> Vec<(String, Value)> { + corpus() + .into_iter() + .flat_map(|(name, document)| { + let list = document["cases"] + .as_array() + .unwrap_or_else(|| panic!("{name}: `cases` is not an array")); + list.clone() + .into_iter() + .map(move |case| (name.clone(), case)) + }) + .collect() +} + +/// The `id` of a case, which every message names. +fn id(case: &Value) -> &str { + case["id"].as_str().expect("case `id` is a string") +} + +/// The case source, from whichever of the two encodings it carries. +fn source_bytes(case: &Value) -> Vec { + match (case.get("source_utf8"), case.get("source_base64")) { + (Some(text), None) => text + .as_str() + .unwrap_or_else(|| panic!("{}: `source_utf8` is not a string", id(case))) + .as_bytes() + .to_vec(), + (None, Some(encoded)) => decode_base64( + encoded + .as_str() + .unwrap_or_else(|| panic!("{}: `source_base64` is not a string", id(case))), + ) + .unwrap_or_else(|error| panic!("{}: `source_base64` {error}", id(case))), + _ => panic!( + "{}: exactly one of `source_utf8` and `source_base64`", + id(case) + ), + } +} + +/// The scan options a case asks for; `dialect` sits beside `language`, not in +/// `options`, and `layout` belongs to the transformation rather than the scan. +fn options(case: &Value) -> TransformOptions { + let mut value = case.get("options").cloned().unwrap_or_else(|| json!({})); + let object = value + .as_object_mut() + .unwrap_or_else(|| panic!("{}: `options` is not an object", id(case))); + let layout = object + .remove("layout") + .map_or(Ok(Layout::Lines), serde_json::from_value); + if let Some(dialect) = case.get("dialect") { + object.insert("dialect".into(), dialect.clone()); + } + TransformOptions { + scan: serde_json::from_value(value) + .unwrap_or_else(|error| panic!("{}: `options` {error}", id(case))), + layout: layout.unwrap_or_else(|error| panic!("{}: `layout` {error}", id(case))), + } +} + +/// The language a case names. +fn language(case: &Value) -> Language { + Language::from_str( + case["language"] + .as_str() + .unwrap_or_else(|| panic!("{}: `language` is not a string", id(case))), + ) + .unwrap_or_else(|error| panic!("{}: {error}", id(case))) +} + +/// The external comment spans of a `transform-spans` case. +fn spans(case: &Value) -> Vec<(ByteSpan, CommentKind)> { + case["spans"] + .as_array() + .unwrap_or_else(|| panic!("{}: `spans` is not an array", id(case))) + .iter() + .map(|span| { + let start = usize::try_from(span["start"].as_u64().expect("span start")) + .expect("span start fits"); + let end = + usize::try_from(span["end"].as_u64().expect("span end")).expect("span end fits"); + let kind: CommentKind = + serde_json::from_value(span["kind"].clone()).expect("span kind"); + (ByteSpan::new(start, end), kind) + }) + .collect() +} + +/// The caller-supplied edits of an `apply_edits` case. +fn edits(case: &Value) -> Vec { + case["edits"] + .as_array() + .unwrap_or_else(|| panic!("{}: `edits` is not an array", id(case))) + .iter() + .map(|edit| { + let span = &edit["span"]; + let start = usize::try_from(span["start"].as_u64().expect("edit start")) + .expect("edit start fits"); + let end = + usize::try_from(span["end"].as_u64().expect("edit end")).expect("edit end fits"); + let replacement = decode_base64( + edit["replacement_base64"] + .as_str() + .expect("edit replacement_base64"), + ) + .expect("edit replacement_base64"); + Edit { + span: ByteSpan::new(start, end), + replacement, + } + }) + .collect() +} + +/// The declarative profile of a `*-profile` case. +fn profile(case: &Value) -> DeclarativeProfile { + serde_json::from_value(case["profile"].clone()) + .unwrap_or_else(|error| panic!("{}: `profile` {error}", id(case))) +} + +/// What running one case produced: a report, and the bytes when it made any. +struct Outcome { + report: Option, + output: Option>, +} + +impl Outcome { + /// A transformation, whose edits must reproduce its own output. + fn transformed(case: &Value, source: &[u8], result: TransformResult) -> Self { + assert!( + result + .edits + .windows(2) + .all(|pair| pair[0].span.end <= pair[1].span.start), + "{}: edits are not sorted and non-overlapping: {:?}", + id(case), + result.edits + ); + assert!( + result + .edits + .iter() + .all(|edit| edit.span.start <= edit.span.end && edit.span.end <= source.len()), + "{}: an edit falls outside the {}-byte source: {:?}", + id(case), + source.len(), + result.edits + ); + assert_eq!( + apply_edits(source, &result.edits), + result.output, + "{}: applying the edits does not reproduce the output", + id(case) + ); + Self { + report: Some(result.report), + output: Some(result.output), + } + } +} + +/// Run one case through the operation it names. +fn run(case: &Value, source: &[u8]) -> Outcome { + let operation = case + .get("operation") + .and_then(Value::as_str) + .unwrap_or("transform"); + let options = options(case); + match operation { + "scan" => Outcome { + report: Some(scan(source, language(case), options.scan)), + output: None, + }, + "transform" => { + Outcome::transformed(case, source, transform(source, language(case), options)) + } + "transform-spans" => Outcome::transformed( + case, + source, + transform_spans(source, language(case), &spans(case), options) + .unwrap_or_else(|error| panic!("{}: {error}", id(case))), + ), + "scan-profile" => Outcome { + report: Some( + scan_profile(source, &profile(case), options.scan) + .unwrap_or_else(|error| panic!("{}: {error}", id(case))), + ), + output: None, + }, + "transform-profile" => Outcome::transformed( + case, + source, + transform_profile(source, &profile(case), options) + .unwrap_or_else(|error| panic!("{}: {error}", id(case))), + ), + "apply_edits" => Outcome { + report: None, + output: Some(apply_edits(source, &edits(case))), + }, + other => panic!("{}: unsupported operation `{other}`", id(case)), + } +} + +/// Check one case against its recorded `expect` block. +fn check_expectation(case: &Value, expect: &Value, outcome: &Outcome) { + let case_id = id(case); + if let Some(valid) = expect.get("valid") { + let report = outcome + .report + .as_ref() + .unwrap_or_else(|| panic!("{case_id}: `expect.valid` needs an operation that reports")); + assert_eq!(json!(report.valid), *valid, "{case_id}: `valid`"); + } + if let Some(comments) = expect.get("comments") { + let report = outcome.report.as_ref().unwrap_or_else(|| { + panic!("{case_id}: `expect.comments` needs an operation that reports") + }); + let observed: Vec<_> = report + .comments + .iter() + .map(|comment| ExpectedComment { + start: comment.span.start, + end: comment.span.end, + kind: comment.kind.as_str().to_owned(), + action: match comment.disposition { + Disposition::Remove => "remove".to_owned(), + Disposition::Keep { .. } => "keep".to_owned(), + }, + }) + .collect(); + let recorded: Vec<_> = comments + .as_array() + .unwrap_or_else(|| panic!("{case_id}: `expect.comments` is not an array")) + .iter() + .map(|comment| ExpectedComment { + start: usize::try_from(comment["start"].as_u64().expect("comment start")) + .expect("fits"), + end: usize::try_from(comment["end"].as_u64().expect("comment end")).expect("fits"), + kind: comment["kind"].as_str().expect("comment kind").to_owned(), + action: comment["action"] + .as_str() + .expect("comment action") + .to_owned(), + }) + .collect(); + assert_eq!(observed, recorded, "{case_id}: `comments`"); + } + if let Some(diagnostics) = expect.get("diagnostics") { + let report = outcome.report.as_ref().unwrap_or_else(|| { + panic!("{case_id}: `expect.diagnostics` needs an operation that reports") + }); + let observed: Vec<_> = report + .diagnostics + .iter() + .map(|item| ExpectedDiagnostic { + code: item.code.clone(), + start: item.span.start, + end: item.span.end, + }) + .collect(); + let recorded: Vec<_> = diagnostics + .as_array() + .unwrap_or_else(|| panic!("{case_id}: `expect.diagnostics` is not an array")) + .iter() + .map(|item| ExpectedDiagnostic { + code: item["code"].as_str().expect("diagnostic code").to_owned(), + start: usize::try_from(item["start"].as_u64().expect("diagnostic start")) + .expect("fits"), + end: usize::try_from(item["end"].as_u64().expect("diagnostic end")).expect("fits"), + }) + .collect(); + assert_eq!(observed, recorded, "{case_id}: `diagnostics`"); + } + if let Some(wanted) = expected_output(case_id, expect) { + let output = outcome.output.as_ref().unwrap_or_else(|| { + panic!("{case_id}: `expect.output_*` needs an operation that writes bytes") + }); + assert_eq!( + String::from_utf8_lossy(output), + String::from_utf8_lossy(&wanted), + "{case_id}: `output` (lossy rendering; the bytes are what is compared)" + ); + assert_eq!(*output, wanted, "{case_id}: `output` bytes"); + } +} + +/// The output bytes an `expect` block pins, if it pins any. +fn expected_output(case_id: &str, expect: &Value) -> Option> { + match (expect.get("output_utf8"), expect.get("output_base64")) { + (Some(text), None) => Some( + text.as_str() + .unwrap_or_else(|| panic!("{case_id}: `output_utf8` is not a string")) + .as_bytes() + .to_vec(), + ), + (None, Some(encoded)) => Some( + decode_base64( + encoded + .as_str() + .unwrap_or_else(|| panic!("{case_id}: `output_base64` is not a string")), + ) + .unwrap_or_else(|error| panic!("{case_id}: `output_base64` {error}")), + ), + (None, None) => None, + (Some(_), Some(_)) => panic!("{case_id}: at most one of `output_utf8` and `output_base64`"), + } +} + +/// Standard base64 with padding; the corpus carries binary sources this way. +fn decode_base64(text: &str) -> Result, String> { + const TABLE: &[u8; 64] = b"ABCDEFGHIJKLMNOPQRSTUVWXYZabcdefghijklmnopqrstuvwxyz0123456789+/"; + let cleaned: Vec<_> = text + .bytes() + .filter(|byte| !byte.is_ascii_whitespace()) + .collect(); + if cleaned.len() % 4 != 0 { + return Err("has a length that is not a multiple of four".into()); + } + let value = |byte: u8| { + TABLE + .iter() + .position(|candidate| *candidate == byte) + .map(|index| u8::try_from(index).expect("base64 index fits")) + .ok_or_else(|| format!("has the invalid byte {byte:#04x}")) + }; + let mut output = Vec::with_capacity(cleaned.len() / 4 * 3); + for chunk in cleaned.chunks(4) { + let a = value(chunk[0])?; + let b = value(chunk[1])?; + output.push((a << 2) | (b >> 4)); + if chunk[2] != b'=' { + let c = value(chunk[2])?; + output.push((b << 4) | (c >> 2)); + if chunk[3] != b'=' { + output.push((c << 6) | value(chunk[3])?); + } + } + } + Ok(output) +} + +#[test] +fn the_corpus_is_well_formed() { + let cases = cases(); + let minimum = floor("cases"); + assert!( + cases.len() >= minimum, + "the corpus holds {} case(s), fewer than the {minimum} required by \ + spec/fixtures/v1/{FLOOR_FILE}", + cases.len() + ); + let mut seen = BTreeSet::new(); + for (file, case) in &cases { + let case_id = id(case); + assert!( + seen.insert(case_id.to_owned()), + "duplicate fixture id `{case_id}` in {file}" + ); + assert!(!case_id.is_empty(), "{file}: a case has an empty id"); + assert!( + case.get("note") + .and_then(Value::as_str) + .is_some_and(|note| !note.is_empty()), + "{case_id}: every case carries a `note` naming the specification it comes from" + ); + let _ = source_bytes(case); + let _ = language(case); + let _ = options(case); + } +} + +#[test] +fn every_case_runs_and_keeps_the_engine_promises() { + for (_, case) in cases() { + let source = source_bytes(&case); + let _ = run(&case, &source); + } +} + +#[test] +fn every_recorded_expectation_still_holds() { + let mut recorded = 0; + for (_, case) in cases() { + let Some(expect) = case.get("expect") else { + continue; + }; + recorded += 1; + let source = source_bytes(&case); + let outcome = run(&case, &source); + check_expectation(&case, expect, &outcome); + } + let minimum = floor("expectations"); + assert!( + recorded >= minimum, + "{recorded} case(s) carry a recorded `expect` block, fewer than the \ + {minimum} required by spec/fixtures/v1/{FLOOR_FILE}; record the missing \ + ones with `python3 tools/differential.py --record`" + ); +} diff --git a/rust/ocomment-plugin-sdk/Cargo.toml b/rust/ocomment-plugin-sdk/Cargo.toml index 53cb989..2bda6df 100644 --- a/rust/ocomment-plugin-sdk/Cargo.toml +++ b/rust/ocomment-plugin-sdk/Cargo.toml @@ -13,7 +13,14 @@ keywords = ["comments", "plugin", "wasm", "component-model"] categories = ["development-tools", "wasm"] include = ["src/**", "README.md"] +[lints] +workspace = true + [dependencies] ocomment-core = { version = "0.1.0", path = "../ocomment-core" } serde.workspace = true thiserror.workspace = true + +[package.metadata.docs.rs] +all-features = true +rustdoc-args = ["--generate-link-to-definition"] diff --git a/rust/ocomment-plugin-sdk/src/lib.rs b/rust/ocomment-plugin-sdk/src/lib.rs index 30559c4..46415ab 100644 --- a/rust/ocomment-plugin-sdk/src/lib.rs +++ b/rust/ocomment-plugin-sdk/src/lib.rs @@ -1,31 +1,134 @@ //! Versioned scanner-plugin boundary. +//! +//! A scanner plugin finds comments in a syntax `ocomment-core` has no scanner +//! for and hands their spans back to the host, which puts them through the +//! ordinary policy with +//! [`transform_spans`](ocomment_core::transform_spans). This crate is the +//! contract between the two: the [`PluginComment`] a guest returns, the +//! [`API_VERSION`] it was built against, and the [`validate_comments`] check +//! the host runs before it trusts any of it. +//! +//! A plugin is untrusted code, so nothing it returns is taken on faith. The +//! host validates first and refuses the whole batch on the first fault; it +//! never removes bytes on the strength of a span it has not checked. +//! +//! ``` +//! use ocomment_core::{ByteSpan, CommentKind}; +//! use ocomment_plugin_sdk::{API_VERSION, PluginComment, ValidationError, validate_comments}; +//! +//! let source = b"a ;; note\n"; +//! let found = [PluginComment { +//! span: ByteSpan::new(2, 9), +//! kind: CommentKind::Line, +//! }]; +//! assert!(validate_comments(source.len(), API_VERSION, &found).is_ok()); +//! +//! // A guest built against another revision of the contract is refused +//! // before its spans are even read. +//! assert!(matches!( +//! validate_comments(source.len(), API_VERSION + 1, &found), +//! Err(ValidationError::ApiVersion { .. }), +//! )); +//! ``` use ocomment_core::{ByteSpan, CommentKind}; use serde::{Deserialize, Serialize}; use thiserror::Error; +/// The revision of this contract that host and guest must agree on. +/// +/// It is bumped whenever the shape of a [`PluginComment`] or the rules in +/// [`validate_comments`] change. A guest reports the version it was built +/// against and the host refuses anything else, so a plugin compiled against +/// an older SDK fails loudly instead of being misread. pub const API_VERSION: u32 = 1; +/// One comment a plugin found. +/// +/// A plugin reports where a comment is and what it is; it never decides +/// whether the comment is removed. That stays with the host's policy, so one +/// configuration governs built-in and plugin-scanned files alike. #[derive(Clone, Debug, Eq, PartialEq, Serialize, Deserialize)] pub struct PluginComment { + /// Where the comment's bytes are, delimiters included. pub span: ByteSpan, + /// What the comment is, which is what the host's policy then judges. pub kind: CommentKind, } +/// Why a plugin's answer cannot be trusted. +/// +/// Each variant is a way a guest could otherwise make the host remove bytes +/// it should not, or spend unbounded work trying. #[derive(Clone, Debug, Error, Eq, PartialEq)] pub enum ValidationError { + /// The guest was built against a different revision of this contract. #[error("plugin API version {received} is unsupported; host supports {supported}")] - ApiVersion { received: u32, supported: u32 }, + ApiVersion { + /// The version the guest reported. + received: u32, + /// The only version this host accepts, [`API_VERSION`]. + supported: u32, + }, + /// A span is inverted or reaches past the end of the source. #[error("plugin comment span is outside the {source_len}-byte source")] - OutOfBounds { source_len: usize }, + OutOfBounds { + /// The length of the source the spans had to fit in. + source_len: usize, + }, + /// A span starts before its predecessor ends, which no single-pass edit + /// could apply. #[error("plugin spans are not strictly sorted and non-overlapping")] Overlap, + /// A span covers no bytes, so it names no comment. #[error("plugin comment spans must not be empty")] EmptySpan, + /// More spans than the source could hold comments, which is a guest + /// spending the host's memory rather than reporting anything. #[error("plugin returned more than the allowed {limit} spans")] - SpanLimit { limit: usize }, + SpanLimit { + /// The most spans this source could have justified. + limit: usize, + }, } +/// Check everything a plugin returned before the host acts on any of it. +/// +/// The version is checked first, so a guest built against another revision is +/// refused before its spans are read at all. The spans must then each be +/// non-empty, inside the source, and start no earlier than the previous one +/// ended — the same contract +/// [`transform_spans`](ocomment_core::transform_spans) enforces, checked here +/// so a host can refuse a plugin's whole answer rather than a single span of +/// it. The count is capped as well: no source can hold more comments than it +/// has bytes, plus one. +/// +/// # Errors +/// +/// Returns the [`ValidationError`] for the first fault found. On any error +/// the batch is refused whole; there is no partial acceptance. +/// +/// # Examples +/// +/// ``` +/// use ocomment_core::{ByteSpan, CommentKind}; +/// use ocomment_plugin_sdk::{API_VERSION, PluginComment, ValidationError, validate_comments}; +/// +/// let comment = |start, end| PluginComment { +/// span: ByteSpan::new(start, end), +/// kind: CommentKind::Line, +/// }; +/// +/// assert!(validate_comments(10, API_VERSION, &[comment(0, 2), comment(2, 10)]).is_ok()); +/// assert_eq!( +/// validate_comments(10, API_VERSION, &[comment(4, 7), comment(6, 8)]), +/// Err(ValidationError::Overlap), +/// ); +/// assert_eq!( +/// validate_comments(10, API_VERSION, &[comment(9, 11)]), +/// Err(ValidationError::OutOfBounds { source_len: 10 }), +/// ); +/// ``` pub fn validate_comments( source_len: usize, api_version: u32, diff --git a/rust/ocomment/Cargo.toml b/rust/ocomment/Cargo.toml index 95bcb40..63b4c63 100644 --- a/rust/ocomment/Cargo.toml +++ b/rust/ocomment/Cargo.toml @@ -7,7 +7,7 @@ rust-version.workspace = true license.workspace = true repository.workspace = true homepage = "https://github.com/P4suta/OComment" -documentation = "https://docs.rs/ocomment" +documentation = "https://p4suta.github.io/OComment/" readme = "README.md" keywords = ["comments", "linter", "source-code", "lsp", "wasm"] categories = ["command-line-utilities", "development-tools"] @@ -30,6 +30,7 @@ path = "src/main.rs" anyhow.workspace = true clap.workspace = true clap_complete.workspace = true +clap_mangen.workspace = true globset.workspace = true ignore.workspace = true ocomment-core = { version = "0.1.0", path = "../ocomment-core" } @@ -45,6 +46,7 @@ thiserror.workspace = true toml.workspace = true tokio.workspace = true tower-lsp.workspace = true +unicode-width.workspace = true wasm_component_layer.workspace = true wasmi.workspace = true wasmi_runtime_layer.workspace = true diff --git a/rust/ocomment/assets/config.schema.json b/rust/ocomment/assets/config.schema.json index dae4bf1..7e64936 100644 --- a/rust/ocomment/assets/config.schema.json +++ b/rust/ocomment/assets/config.schema.json @@ -70,7 +70,7 @@ "$defs": { "strings": { "type": "array", "items": { "type": "string" }, "default": [] }, "policy": { "enum": ["safe", "legal", "all"], "default": "safe" }, - "language": { "enum": ["rust", "ocaml", "c", "cpp", "go", "java", "javascript", "typescript", "python", "shell", "html", "css", "jsonc", "sql", "kotlin"] }, + "language": { "enum": ["rust", "ocaml", "c", "cpp", "go", "java", "javascript", "typescript", "python", "shell", "html", "css", "jsonc", "sql", "kotlin", "toml", "lua", "yaml", "php", "ruby", "zig", "r", "dart", "swift", "csharp", "scala", "vue", "svelte", "markdown", "perl"] }, "kind": { "enum": ["line", "block", "doc-line", "doc-block", "directive", "license", "html-comment", "shebang", "encoding", "optimizer-hint", "version-comment"] }, "kinds": { "type": "array", "items": { "$ref": "#/$defs/kind" }, "default": [] }, "languageConfig": { @@ -97,7 +97,7 @@ } }, "dialect": { - "enum": ["standard", "jsx", "tsx", "objective-c", "objective-cpp", "gnu-c", "gnu-cpp", "cuda", "posix-sh", "bash53", "zsh", "postgresql", "mysql", "sqlite", "t-sql", "oracle"] + "enum": ["standard", "jsx", "tsx", "objective-c", "objective-cpp", "gnu-c", "gnu-cpp", "cuda", "posix-sh", "bash53", "zsh", "postgresql", "mysql", "sqlite", "t-sql", "oracle", "scss"] }, "nonEmptyStrings": { "type": "array", "minItems": 1, "items": { "type": "string", "minLength": 1 } diff --git a/rust/ocomment/assets/languages.toml b/rust/ocomment/assets/languages.toml new file mode 100644 index 0000000..7331623 --- /dev/null +++ b/rust/ocomment/assets/languages.toml @@ -0,0 +1,254 @@ +version = 1 + +# NOTE: The canonical list of the languages OComment has a built-in scanner +# NOTE: for. +# NOTE: +# NOTE: `extensions` holds every suffix that selects the language, without the +# NOTE: dot, and `dialects` every dialect it accepts, in the order the binary +# NOTE: lists them when it refuses one. An extension that selects a dialect +# NOTE: other than `standard` says so in `extension_dialects`; `reserved_names` +# NOTE: and `shebangs` are the whole file names and interpreter names that +# NOTE: select the language when the extension does not decide it. `notes` is a +# NOTE: remark for the human listing. +# NOTE: +# NOTE: The `files:` pattern of the published pre-commit hooks is generated +# NOTE: from the extensions here (tools/check_hooks.py), docs/languages.md is +# NOTE: generated from the whole table (tools/gen_docs.py), and `ocomment +# NOTE: languages` prints the copy the binary embeds. +# NOTE: rust/ocomment/tests/spec_languages.rs checks every claim below against +# NOTE: the detector and the binary. + +[[languages]] +name = "rust" +extensions = ["rs"] +dialects = ["standard"] + +[[languages]] +name = "ocaml" +extensions = ["ml", "mli", "mlt"] +dialects = ["standard"] +notes = "OCaml 5.5 lexical forms" + +[[languages]] +name = "c" +extensions = ["c", "h", "m"] +dialects = ["standard", "objective-c", "gnu-c"] +extension_dialects = { m = "objective-c" } + +[[languages]] +name = "cpp" +extensions = ["cc", "cpp", "cxx", "hh", "hpp", "hxx", "mm", "cu", "cuh"] +dialects = ["standard", "objective-cpp", "gnu-cpp", "cuda"] +extension_dialects = { mm = "objective-cpp", cu = "cuda", cuh = "cuda" } + +[[languages]] +name = "go" +extensions = ["go"] +dialects = ["standard"] + +[[languages]] +name = "java" +extensions = ["java"] +dialects = ["standard"] +notes = "Unicode escape translation" + +[[languages]] +name = "javascript" +extensions = ["js", "mjs", "cjs", "jsx"] +dialects = ["standard", "jsx"] +extension_dialects = { jsx = "jsx" } +shebangs = ["node", "deno"] + +[[languages]] +name = "typescript" +extensions = ["ts", "mts", "cts", "tsx"] +dialects = ["standard", "tsx"] +extension_dialects = { tsx = "tsx" } + +[[languages]] +name = "python" +extensions = ["py", "pyw", "pyi"] +dialects = ["standard"] +shebangs = ["python"] + +[[languages]] +name = "shell" +extensions = ["sh", "bash", "zsh"] +dialects = ["standard", "posix-sh", "bash53", "zsh"] +extension_dialects = { sh = "posix-sh", bash = "bash53", zsh = "zsh" } +reserved_names = [ + "Dockerfile", + "Containerfile", + "Makefile", + "GNUmakefile", + ".profile", + ".bashrc", + ".zshrc", +] +shebangs = ["sh", "bash", "zsh"] + +[[languages]] +name = "html" +extensions = ["html", "htm", "xhtml", "shtml"] +dialects = ["standard"] +notes = "script and style bodies are scanned as JavaScript and CSS" + +[[languages]] +name = "css" +extensions = ["css", "scss", "sass"] +dialects = ["standard", "scss"] +extension_dialects = { scss = "scss", sass = "scss" } + +[[languages]] +name = "jsonc" +extensions = ["jsonc", "json5"] +dialects = ["standard"] +reserved_names = ["tsconfig.json", "jsconfig.json"] + +[[languages]] +name = "sql" +extensions = ["sql"] +dialects = ["standard", "postgresql", "mysql", "sqlite", "t-sql", "oracle"] + +[[languages]] +name = "kotlin" +extensions = ["kt", "kts"] +dialects = ["standard"] + +[[languages]] +name = "toml" +extensions = ["toml"] +dialects = ["standard"] +reserved_names = [ + "Cargo.lock", + "Pipfile", + "poetry.lock", + "uv.lock", + "pdm.lock", +] +notes = "TOML v1.0.0 lexical forms" + +[[languages]] +name = "lua" +extensions = ["lua", "rockspec"] +dialects = ["standard"] +shebangs = ["lua", "luajit"] +notes = "Lua 5.4 lexical forms" + +[[languages]] +name = "yaml" +extensions = ["yml", "yaml"] +dialects = ["standard"] +reserved_names = [ + ".clang-format", + ".clang-tidy", + ".yamllint", +] +notes = "YAML 1.2.2 lexical forms" + +[[languages]] +name = "php" +extensions = ["php", "phtml", "phpt"] +dialects = ["standard"] +shebangs = ["php"] +notes = "PHP 8 lexical forms; the inline HTML around the tags is opaque" + +[[languages]] +name = "ruby" +extensions = [ + "rb", + "rbw", + "rake", + "gemspec", + "ru", + "podspec", + "jbuilder", + "thor", + "rbi", +] +dialects = ["standard"] +reserved_names = [ + "Gemfile", + "Rakefile", + "Guardfile", + "Capfile", + "Vagrantfile", + "Brewfile", + "Podfile", + "Fastfile", + "Appfile", + "Berksfile", + "Thorfile", + "Dangerfile", + ".irbrc", + ".pryrc", +] +shebangs = ["truffleruby", "jruby", "ruby"] +notes = "Ruby 3.3 lexical forms" + +[[languages]] +name = "zig" +extensions = ["zig", "zon"] +dialects = ["standard"] +notes = "Zig 0.16 lexical forms; no block comment, so `/*` is two operators" + +[[languages]] +name = "r" +extensions = ["r"] +dialects = ["standard"] +reserved_names = [".Rprofile"] +shebangs = ["rscript", "r"] +notes = "R 4.3 lexical forms; `%op%` is opaque and `r\"(...)\"` is a raw string" + +[[languages]] +name = "dart" +extensions = ["dart"] +dialects = ["standard"] +shebangs = ["dart"] +notes = "Dart 3.13 lexical forms; block comments nest and `${...}` is code" + +[[languages]] +name = "swift" +extensions = ["swift"] +dialects = ["standard"] +shebangs = ["swift"] +notes = "Swift 6 lexical forms; block comments nest and `#/.../#` is a regular expression literal" + +[[languages]] +name = "csharp" +extensions = ["cs", "csx"] +dialects = ["standard"] +shebangs = ["dotnet-script"] +notes = "C# 13 lexical forms; a line opening with `#` is a preprocessor directive and carries at most a `//` comment" + +[[languages]] +name = "scala" +extensions = ["scala", "sc"] +dialects = ["standard"] +shebangs = ["scala-cli", "scala"] +notes = "Scala 3.8 lexical forms; block comments nest, an identifier before a quote makes the string interpolate, and an XML literal is opaque text with `{...}` code" + +[[languages]] +name = "vue" +extensions = ["vue"] +dialects = ["standard"] +notes = "Vue single-file components; the template is HTML with `{{ ... }}` code and the script and style bodies are their own languages, the `lang` attribute choosing which" + +[[languages]] +name = "svelte" +extensions = ["svelte"] +dialects = ["standard"] +notes = "Svelte components; the template is HTML with `{ ... }` code and the script and style bodies are their own languages, the `lang` attribute choosing which" + +[[languages]] +name = "markdown" +extensions = ["md", "markdown", "rmd"] +dialects = ["standard"] +notes = "CommonMark documents; HTML comments are comments, fenced code blocks are scanned as the language their info string names, and inline and indented code are opaque" + +[[languages]] +name = "perl" +extensions = ["pl", "pm", "t"] +dialects = ["standard"] +shebangs = ["perl"] +notes = "Perl 5.38 lexical forms; quote words and regexes hide a `#`, POD blocks are opaque, and a `/` the parse context alone settles is reported as lexically ambiguous" diff --git a/rust/ocomment/src/atomic.rs b/rust/ocomment/src/atomic.rs index 6cda7a8..6fb44d8 100644 --- a/rust/ocomment/src/atomic.rs +++ b/rust/ocomment/src/atomic.rs @@ -54,7 +54,14 @@ pub fn apply_transaction(plans: Vec) -> Result<()> { std::process::id() )); if backup.exists() { - bail!("rollback path {} already exists", backup.display()); + /* INVARIANT: The journal holds the file as it was before the interrupted + * run, so deleting it unread can be the loss the rollback existed + * to prevent. */ + bail!( + "rollback path {} already exists; a previous ocomment run may have been \ + interrupted — inspect and delete it before retrying", + backup.display() + ); } prepared.push(Prepared { plan, @@ -66,8 +73,8 @@ pub fn apply_transaction(plans: Vec) -> Result<()> { for index in 0..prepared.len() { if let Err(error) = commit_one(&mut prepared[index]) { let failed_path = prepared[index].plan.path.clone(); - // Include the failing item: a rename may have created its backup - // before installing or syncing the replacement failed. + /* INVARIANT: Include the failing item: a rename may have created its backup + * before installing or syncing the replacement failed. */ let rollback_error = rollback(&prepared[..=index]); return Err(match rollback_error { Ok(()) => anyhow!( @@ -166,4 +173,39 @@ mod tests { permissions.readonly() ); } + + /// A journal left over from an interrupted run is the only thing standing + /// between the caller and a retry, and it holds the pre-run contents of a + /// file. The refusal has to say both: what the file is, and that reading + /// it before deleting it is the point. + /// + /// The name carries this process's own id, so the test can plant exactly + /// the journal the transaction is about to reach for. + #[test] + fn an_existing_rollback_journal_says_what_to_do_about_it() { + let directory = tempfile::tempdir().unwrap(); + let path = directory.path().join("x.rs"); + fs::write(&path, b"old").unwrap(); + let journal = directory + .path() + .join(format!(".x.rs.ocomment-rollback-{}-0", std::process::id())); + fs::write(&journal, b"interrupted").unwrap(); + + let error = apply_transaction(vec![WritePlan { + path: path.clone(), + original: b"old".to_vec(), + replacement: b"new".to_vec(), + }]) + .unwrap_err(); + + assert_eq!( + error.to_string(), + format!( + "rollback path {} already exists; a previous ocomment run may have been \ + interrupted — inspect and delete it before retrying", + journal.display() + ) + ); + assert_eq!(fs::read(&path).unwrap(), b"old", "the file was rewritten"); + } } diff --git a/rust/ocomment/src/cli.rs b/rust/ocomment/src/cli.rs index d45e772..99bda3f 100644 --- a/rust/ocomment/src/cli.rs +++ b/rust/ocomment/src/cli.rs @@ -1,67 +1,252 @@ use crate::{ atomic::{WritePlan, apply_transaction}, - config, files, git, lsp, - output::{self, Operation, OutputFormat, Presentation, ProcessedFile}, + config, files, git, interactive, lsp, + output::{ + self, Explanations, FileExplanation, Operation, OutputFormat, Presentation, ProcessedFile, + RenderOptions, Verbosity, + }, plugin, + values::{CommentKindArg, DialectArg, LanguageArg, LayoutArg, PolicyArg}, }; -use anyhow::{Context, Result}; +use anyhow::{Context, Result, bail, ensure}; use clap::{Args, CommandFactory, Parser, Subcommand, ValueEnum}; use clap_complete::{Shell, generate}; -use ocomment_core::{CommentKind, Dialect, Language, Layout, Policy, transform}; +use ocomment_core::{CommentKind, Dialect, Language, transform}; use rayon::prelude::*; +use serde::{Deserialize, Serialize}; use std::{ + collections::BTreeMap, fs, io::{self, IsTerminal, Read, Write}, path::PathBuf, + sync::atomic::{AtomicBool, AtomicUsize, Ordering}, }; +const LONG_ABOUT: &str = "\ +OComment scans source bytes without requiring UTF-8 and reports or removes \ +comment tokens. The default policy protects source preambles and tool or \ +language directives. Rewrites are prepared and committed as one \ +rollback-backed transaction."; + +const AFTER_LONG_HELP: &str = "\ +EXIT STATUS + 0 Nothing removable was found and every requested change was applied. + 1 Removable comments were reported, or a diff was printed. + 2 Invalid source, configuration, plugin, or I/O failure. + +FILES + .ocomment.toml Project configuration, merged over the user file. + .ocommentignore Extra ignore patterns honoured by repository walks. + .ocomment.lock Pinned digests of the installed WASM scanner plugins. + $XDG_CONFIG_HOME/ocomment/config.toml + User configuration, merged over the built-in defaults. + +EXAMPLES + ocomment + Check the current directory and report removable comments. + ocomment fix --policy all --layout compact src + Remove every comment under src and close the gaps it leaves. + ocomment strip --language rust < before.rs > after.rs + Strip one file from standard input to standard output. + +SEE ALSO + The complete schemas and guides are available in the OComment repository."; + +/// The roff sections `clap_mangen` cannot derive, carrying the same content as +/// the `--help` epilogue above. A line that would start with `.` is escaped +/// with `\&` so roff reads a file name as text rather than as a macro. +const MAN_SECTIONS: &str = r#".SH EXIT STATUS +.TP +.B 0 +Nothing removable was found and every requested change was applied. +.TP +.B 1 +Removable comments were reported, or a diff was printed. +.TP +.B 2 +Invalid source, configuration, plugin, or I/O failure. +.SH FILES +.TP +.B \&.ocomment.toml +Project configuration, merged over the user file. +.TP +.B \&.ocommentignore +Extra ignore patterns honoured by repository walks. +.TP +.B \&.ocomment.lock +Pinned digests of the installed WASM scanner plugins. +.TP +.B $XDG_CONFIG_HOME/ocomment/config.toml +User configuration, merged over the built\-in defaults. +.SH EXAMPLES +.TP +.B ocomment +Check the current directory and report removable comments. +.TP +.B ocomment fix \-\-policy all \-\-layout compact src +Remove every comment under src and close the gaps it leaves. +.TP +.B ocomment strip \-\-language rust < before.rs > after.rs +Strip one file from standard input to standard output. +.SH SEE ALSO +The complete schemas and guides are available in the OComment repository. +"#; + #[derive(Parser)] #[command( name = "ocomment", version, about = "Check and remove source-code comments safely" )] +#[command(long_about = LONG_ABOUT)] #[command(args_conflicts_with_subcommands = true)] +#[command(after_long_help = AFTER_LONG_HELP)] struct Cli { - #[command(flatten)] - common: CommonArgs, - #[command(subcommand)] - command: Option, - /// Paths for the implicit `check` command. + /// Files or directories to check; `-` reads standard input (default: current directory). #[arg(value_name = "PATH")] paths: Vec, + /// The command to run; `check` runs when none is given. + #[command(subcommand)] + command: Option, + /// Configuration, policy, and output options shared by every command. + #[command(flatten)] + common: CommonArgs, } #[derive(Clone, Debug, Args)] struct CommonArgs { - /// Explicit configuration file. - #[arg(long, global = true)] + /// Read this configuration file instead of discovering `.ocomment.toml`. + #[arg(long, global = true, value_name = "FILE")] config: Option, - /// Output encoding. - #[arg(long, global = true, value_enum, default_value_t)] - format: OutputFormat, - #[arg(long, global = true)] - policy: Option, - #[arg(long, global = true)] - layout: Option, - #[arg(long, global = true)] - language: Option, - #[arg(long, global = true)] - dialect: Option, - #[arg(long = "keep-kind", global = true, value_delimiter = ',')] - keep_kind: Vec, - #[arg(long = "remove-kind", global = true, value_delimiter = ',')] - remove_kind: Vec, + /// What may be removed and how the source is interpreted. + #[command(flatten)] + policy: PolicyArgs, + /// How results are encoded and decorated. + #[command(flatten)] + output: OutputArgs, +} + +#[derive(Clone, Debug, Args)] +#[command(next_help_heading = "Policy")] +struct PolicyArgs { + /// Which classes of comment the run is allowed to remove. + #[arg( + long, + global = true, + value_enum, + ignore_case = true, + value_name = "POLICY" + )] + policy: Option, + /// How the bytes left behind by a removed comment are laid out. + #[arg( + long, + global = true, + value_enum, + ignore_case = true, + value_name = "LAYOUT" + )] + layout: Option, + /// Force this language instead of detecting it from path and contents. + #[arg( + long, + global = true, + value_enum, + ignore_case = true, + value_name = "LANGUAGE" + )] + language: Option, + /// Force this dialect of the selected language. + #[arg( + long, + global = true, + value_enum, + ignore_case = true, + value_name = "DIALECT" + )] + dialect: Option, + /// Comma-separated comment kinds to protect on top of the policy. + #[arg( + long = "keep-kind", + global = true, + value_enum, + ignore_case = true, + value_delimiter = ',', + value_name = "KIND" + )] + keep_kind: Vec, + /// Comma-separated comment kinds to remove regardless of the policy. + #[arg( + long = "remove-kind", + global = true, + value_enum, + ignore_case = true, + value_delimiter = ',', + value_name = "KIND" + )] + remove_kind: Vec, + /// Apply the edits that are still provably safe when the source fails to scan. #[arg(long, global = true)] force_invalid: bool, + /// Remove protected comments such as shebang and encoding preambles. #[arg(long, global = true)] force_protected: bool, - #[arg(long, global = true, value_enum, default_value_t)] +} + +#[derive(Clone, Debug, Args)] +#[command(next_help_heading = "Output")] +struct OutputArgs { + /// Output encoding. + #[arg( + long, + global = true, + value_enum, + default_value_t, + value_name = "FORMAT" + )] + format: OutputFormat, + /// When to colour terminal output. + #[arg(long, global = true, value_enum, default_value_t, value_name = "WHEN")] color: ColorChoice, - #[arg(long, global = true, value_enum, default_value_t)] + /// When to emit terminal hyperlinks for reported paths. + #[arg(long, global = true, value_enum, default_value_t, value_name = "WHEN")] hyperlinks: AutoChoice, - #[arg(long, global = true, value_enum, default_value_t)] + /// Omit the one-line comment text from human `check` and `scan` lines. + #[arg(long, global = true)] + no_preview: bool, + /// List every comment human `check` and `scan` met and name the rule and setting behind each one. + #[arg(long, global = true)] + explain: bool, + /// When to draw the live scanning counter on standard error. + #[arg(long, global = true, value_enum, default_value_t, value_name = "WHEN")] progress: AutoChoice, + /// Drop the run summary and notes; the command's product (findings, patch, listing) is still written. + #[arg(short, long, global = true, conflicts_with = "verbose")] + quiet: bool, + /// Trace what is scanned and summarize every comment kind and skipped file. + #[arg(short, long, global = true)] + verbose: bool, +} + +impl CommonArgs { + /// The language forced on the command line, if any. + fn language(&self) -> Option { + self.policy.language.map(Language::from) + } + + /// The dialect forced on the command line, if any. + fn dialect(&self) -> Option { + self.policy.dialect.map(Dialect::from) + } + + /// How much of the human report this run may write. + fn verbosity(&self) -> Verbosity { + match (self.output.quiet, self.output.verbose) { + (true, _) => Verbosity::Quiet, + (_, true) => Verbosity::Verbose, + _ => Verbosity::Normal, + } + } } #[derive(Clone, Copy, Debug, Default, ValueEnum)] @@ -82,24 +267,50 @@ enum AutoChoice { #[derive(Subcommand)] enum Command { + /// Report removable comments (default command) Check(TargetArgs), - Fix(TargetArgs), + /// Remove comments in place through an atomic, rollback-backed transaction + Fix(FixArgs), + /// Print a unified diff of the changes fix would make Diff(TargetArgs), + /// List every comment with its kind, disposition and byte span Scan(TargetArgs), + /// Read source on stdin and write the stripped result to stdout Strip, + /// Run the LSP 3.18 server over stdio Lsp, + /// Write a starter .ocomment.toml or Lefthook configuration Init(InitArgs), + /// Show, locate, explain, or export the resolved configuration Config(ConfigArgs), + /// List built-in languages, extensions, and dialects Languages, + /// Manage sandboxed WASM scanner plugins Plugin(PluginArgs), - Completions { shell: Shell }, + /// Generate shell completions + Completions { + /// Shell whose completion script is written to stdout. + shell: Shell, + }, + /// Diagnose the environment (config, git, plugins, tools) Doctor, + /// Render the roff manual page to stdout + Man, } #[derive(Clone, Debug, Default, Args)] struct TargetArgs { + /// Files or directories to process; `-` reads standard input (default: current directory). #[arg(value_name = "PATH")] paths: Vec, + /// Whether the Git index, rather than the working tree, is the source. + #[command(flatten)] + git: GitArgs, +} + +/// The `--staged` pair, shared by every command that can read the Git index. +#[derive(Clone, Debug, Default, Args)] +struct GitArgs { /// Read and update Git index blobs rather than treating the working tree as the source. #[arg(long)] staged: bool, @@ -108,12 +319,53 @@ struct TargetArgs { index_only: bool, } +#[derive(Args)] +struct FixArgs { + /* NOTE: `fix` rewrites files in place and refuses the `-` that stands for + * standard input, so its PATH list is not the one every other command + * takes and does not borrow that command's help line. */ + /// Files or directories to rewrite (default: current directory). + #[arg(value_name = "PATH")] + paths: Vec, + /// Whether the Git index, rather than the working tree, is rewritten. + #[command(flatten)] + git: GitArgs, + /// Print the patch `fix` would apply and write nothing. + #[arg(long)] + dry_run: bool, + /// Ask about each comment in turn and remove only the accepted ones. + /// + /// The index has no working-tree line to show a hunk from, `--dry-run` + /// writes nothing whatever the answers were, and `-q` asks for a run with + /// no commentary at all. None of the three can also be a conversation. + #[arg(short = 'i', long, conflicts_with_all = ["staged", "dry_run", "quiet"])] + interactive: bool, +} + +impl FixArgs { + /// The same targets in the shape every other command hands to the run. + fn target(self) -> TargetArgs { + TargetArgs { + paths: self.paths, + git: self.git, + } + } +} + #[derive(Args)] struct InitArgs { + /// Which starter file to write. #[arg(value_enum, default_value_t)] kind: InitKind, + /// For the Lefthook hook, run `fix` instead of `check`. #[arg(long)] fix: bool, + /// Replace the file if it already exists. + #[arg(long, conflicts_with = "stdout")] + force: bool, + /// Print the template to standard output and write no file. + #[arg(long)] + stdout: bool, } #[derive(Clone, Copy, Debug, Default, ValueEnum)] @@ -125,6 +377,7 @@ enum InitKind { #[derive(Args)] struct ConfigArgs { + /// Which view of the resolved configuration to print. #[arg(value_enum, default_value_t)] action: ConfigAction, } @@ -140,39 +393,100 @@ enum ConfigAction { #[derive(Args)] struct PluginArgs { + /// The plugin operation to run. #[command(subcommand)] command: PluginCommand, } #[derive(Subcommand)] enum PluginCommand { + /// Install a plugin and pin its digest in .ocomment.lock Add { + /// Path or URL of the WASM component to install. source: String, - #[arg(long)] + /// Name to register the plugin under (default: the file stem). + #[arg(long, value_name = "NAME")] name: Option, - #[arg(long)] + /// Expected SHA-256 digest of the component, verified before install. + #[arg(long, value_name = "HEX")] sha256: Option, - #[arg(long)] + /// Publisher identity recorded alongside the pinned digest. + #[arg(long, value_name = "IDENTITY")] identity: Option, }, + /// Uninstall a plugin and drop its lock entry Remove { + /// Name of the plugin to remove. name: String, }, + /// List the installed plugins and their pinned digests List, + /// Re-fetch plugins and refresh their pinned digests Update { + /// Name of the plugin to update (default: all of them). name: Option, }, + /// Check installed plugins against their pinned digests Verify { + /// Name of the plugin to verify (default: all of them). name: Option, }, + /// Scaffold a new plugin crate from the scanner WIT world New { + /// Directory to create the plugin crate in. path: PathBuf, }, } +/// The `fix` variants that change what a run does with what it found. Every +/// other command runs with neither. +#[derive(Clone, Copy, Default)] +struct RunFlags { + /// The run produces the patch `fix` would apply and writes nothing. + dry_run: bool, + /// The run asks about each comment before removing it. + interactive: bool, +} + +impl RunFlags { + const NONE: Self = Self { + dry_run: false, + interactive: false, + }; + const DRY_RUN: Self = Self { + dry_run: true, + interactive: false, + }; + const INTERACTIVE: Self = Self { + dry_run: false, + interactive: true, + }; +} + pub fn run() -> Result { let cli = Cli::parse(); let common = cli.common; + /* NOTE: The machine formats are schemas rather than prose, and none of them has + * a place to put an explanation; the JSON one is closed to extension by + * design. So the combination is refused instead of quietly doing nothing. */ + if common.output.explain && common.output.format != OutputFormat::Human { + bail!("--explain is only available with --format human"); + } + /* NOTE: The flag annotates a report of comments, and only `check`, `scan` and + * the implicit command write one: `fix` reports the files it rewrote, + * `diff` writes a patch, `strip` writes the stripped source, and the rest + * of the commands answer a question that is not about comments at all. + * `--explain` is global, so it is named as an allow-list — a command added + * later has to opt in — and everything else is refused rather than + * quietly doing nothing. */ + if common.output.explain + && !matches!( + cli.command, + None | Some(Command::Check(_) | Command::Scan(_)) + ) + { + bail!("--explain is only available with `check` and `scan`"); + } match cli.command { None => run_target( Operation::Check, @@ -181,61 +495,142 @@ pub fn run() -> Result { ..Default::default() }, &common, + RunFlags::NONE, ), - Some(Command::Check(args)) => run_target(Operation::Check, args, &common), - Some(Command::Fix(args)) => run_target(Operation::Fix, args, &common), - Some(Command::Diff(args)) => run_target(Operation::Diff, args, &common), - Some(Command::Scan(args)) => run_target(Operation::Scan, args, &common), + Some(Command::Check(args)) => run_target(Operation::Check, args, &common, RunFlags::NONE), + /* NOTE: `--dry-run` runs the diff and reports it in fix vocabulary: the two + * commands must agree on the patch, so only the wording differs. */ + Some(Command::Fix(args)) if args.dry_run => { + run_target(Operation::Diff, args.target(), &common, RunFlags::DRY_RUN) + } + Some(Command::Fix(args)) if args.interactive => { + /* NOTE: The prompt is prose on a terminal and the answers come back the + * same way; a machine format has nowhere to put either, so the + * combination is refused rather than one of the two flags being + * quietly dropped. It is refused before the terminal is looked at, + * because the pair is wrong however the run was started. */ + if common.output.format != OutputFormat::Human { + bail!("--interactive is only available with --format human"); + } + /* NOTE: Without somebody there to answer, the questions would be read out + * of whatever the pipe happened to carry and files would be + * rewritten from it. Nothing is scanned, let alone written. */ + if !io::stdin().is_terminal() || !io::stdout().is_terminal() { + bail!("--interactive needs a terminal; run without -i or use `ocomment diff`"); + } + run_target( + Operation::Fix, + args.target(), + &common, + RunFlags::INTERACTIVE, + ) + } + Some(Command::Fix(args)) => { + run_target(Operation::Fix, args.target(), &common, RunFlags::NONE) + } + Some(Command::Diff(args)) => run_target(Operation::Diff, args, &common, RunFlags::NONE), + Some(Command::Scan(args)) => run_target(Operation::Scan, args, &common, RunFlags::NONE), Some(Command::Strip) => run_strip(&common), Some(Command::Lsp) => lsp::run(common.config.as_deref()), Some(Command::Init(args)) => run_init(args), Some(Command::Config(args)) => run_config(args, &common), - Some(Command::Languages) => { - print_languages(); - Ok(0) - } + Some(Command::Languages) => print_languages(&common), Some(Command::Plugin(args)) => run_plugin(args, &common), - Some(Command::Completions { shell }) => { - generate(shell, &mut Cli::command(), "ocomment", &mut io::stdout()); - Ok(0) - } + Some(Command::Completions { shell }) => run_completions(shell), Some(Command::Doctor) => run_doctor(&common), + Some(Command::Man) => run_man(), } } -fn run_target(operation: Operation, args: TargetArgs, common: &CommonArgs) -> Result { +fn run_target( + operation: Operation, + args: TargetArgs, + common: &CommonArgs, + flags: RunFlags, +) -> Result { let mut resolved = config::load(common.config.as_deref())?; - apply_cli_overrides(&mut resolved.config, common); + apply_cli_overrides(&mut resolved, common); let plugin_host = plugin::PluginHost::load(&resolved.root, &resolved.config.plugins)?; let presentation = presentation(common); - let staged = args.staged || resolved.config.git.staged; + let verbosity = common.verbosity(); + /* NOTE: The trace is part of the human report; a machine format keeps standard + * error empty however loud the run was asked to be. */ + if verbosity == Verbosity::Verbose && common.output.format == OutputFormat::Human { + trace_run(&resolved, &args.paths)?; + } + let progress = progress_enabled(common); + let staged = args.git.staged || resolved.config.git.staged; + if operation == Operation::Fix && !staged && args.paths.is_empty() { + note_fix_scope(&resolved, common)?; + } + /* NOTE: `git` names a staged path relative to the repository root rather than to + * the working directory, so a staged run measures its paths against the + * root from there. Every other run measures them from where it was typed. */ + if staged && let Some(repository) = config::locate_repository(&resolved.cwd) { + resolved.cwd = repository; + } + /* NOTE: `fix --dry-run` writes nothing, but it is still the command whose job is + * to rewrite files in place, and standard input cannot be rewritten. */ + let rewrites = operation == Operation::Fix || flags.dry_run; + let (paths, stdin) = target_paths(&args.paths, rewrites, staged)?; if staged { + /* NOTE: A staged run reports index blobs through a path that carries no + * policy trace, so it says so rather than printing a listing with + * every explanation quietly missing. */ + if common.output.explain { + bail!( + "--explain is not available with --staged; explain the working tree with \ + `ocomment check --explain`" + ); + } return git::run_staged(git::StagedRequest { operation, - paths: &args.paths, + paths: &paths, resolved: &resolved, - format: common.format, - index_only: args.index_only || resolved.config.git.index_only, + format: common.output.format, + index_only: args.git.index_only || resolved.config.git.index_only, plugin_host: &plugin_host, - forced_language: common.language, - forced_dialect: common.dialect, + forced_language: common.language(), + forced_dialect: common.dialect(), presentation, + verbosity, + preview: !common.output.no_preview, + dry_run: flags.dry_run, }); } - let discovery = files::discover(&args.paths, &resolved, common.language, common.dialect)?; - let files: Vec<_> = discovery + let discovery = read_targets(&paths, stdin, &resolved, common)?; + let total = discovery.files.len(); + let counter = Progress::default(); + let explain = common.output.explain; + let processed = discovery .files .par_iter() .map(|file| { - let (mut language, mut options) = - resolved.for_path(&file.path, file.language, file.dialect); - if let Some(value) = common.language { + /* NOTE: Only an explaining run pays for the trace; every other one takes + * the hot path it always took. */ + let (mut language, mut options, trace) = if explain { + let (language, options, trace) = + resolved.for_path_traced(&file.path, file.language, file.dialect); + (language, options, Some(trace)) + } else { + let (language, options) = + resolved.for_path(&file.path, file.language, file.dialect); + (language, options, None) + }; + if let Some(value) = common.language() { language = value; } - if let Some(value) = common.dialect { + if let Some(value) = common.dialect() { config::validate_dialect(language, value)?; options.scan.dialect = value; } + /* NOTE: Recorded as the scan is about to run with them, `--language` and + * `--dialect` included, so an explanation accounts for the run that + * actually happened. */ + let material = trace.map(|trace| FileExplanation { + options: options.scan.clone(), + trace, + }); let result = if let Some(name) = &file.plugin { let language_name = file .path @@ -250,20 +645,49 @@ fn run_target(operation: Operation, args: TargetArgs, common: &CommonArgs) -> Re } else { transform(&file.source, language, options) }; - Ok::<_, anyhow::Error>(ProcessedFile { - path: file.path.clone(), - source: file.source.clone(), - language, - result, - }) + if progress { + counter.report(total); + } + Ok::<_, anyhow::Error>(( + ProcessedFile { + path: file.path.clone(), + source: file.source.clone(), + language, + result, + }, + material, + )) }) - .collect::>>()?; + .collect::>>(); + if progress { + counter.clear(); + } + /* NOTE: The explanations travel beside the files rather than inside them: a + * staged run reports the same `ProcessedFile` and has no trace to put in + * one, and the path is what the renderer looks each file up by anyway. */ + let processed = processed?; + let mut explanations = Explanations::new(); + let mut files = Vec::with_capacity(processed.len()); + for (file, material) in processed { + if let Some(material) = material { + explanations.insert(file.path.clone(), material); + } + files.push(file); + } let report_invalid = output::invalid(&files); let io_invalid = discovery.skipped.iter().any(|item| item.error); let invalid = report_invalid || io_invalid; let may_fix = !io_invalid && (!report_invalid || resolved.config.policy.force_invalid); - if operation == Operation::Fix && may_fix { + /* NOTE: An interactive run replaces the whole `fix` report: what it wrote is the + * answers it was given, and the ordinary summary counts what the run + * *could* have removed. A run the invalid-file gate has already stopped + * falls through instead, so that report says why nothing was written. */ + if flags.interactive && may_fix { + return run_interactive(&files, &discovery.skipped, invalid, presentation, verbosity); + } + let applied = operation == Operation::Fix && may_fix; + if applied { let plans = files .iter() .filter(|file| file.source != file.result.output) @@ -275,12 +699,22 @@ fn run_target(operation: Operation, args: TargetArgs, common: &CommonArgs) -> Re .collect(); apply_transaction(plans)?; } - output::render( + output::render_explained( &files, &discovery.skipped, - common.format, - operation, - presentation, + &RenderOptions { + format: common.output.format, + operation, + presentation, + verbosity, + preview: !common.output.no_preview, + explain, + dry_run: flags.dry_run, + force_invalid: resolved.config.policy.force_invalid, + applied, + policy: resolved.config.policy.mode, + }, + &explanations, )?; if invalid { return Ok(2); @@ -291,168 +725,633 @@ fn run_target(operation: Operation, args: TargetArgs, common: &CommonArgs) -> Re } } +/// Ask about each comment this run would remove, write the accepted removals +/// through the same transaction a plain `fix` uses, and report what the answers +/// came to. +/// +/// A clean abort is not a failure of the run: `x` is the answer for a fix that +/// should never have started, and it exits 0 having touched nothing. +fn run_interactive( + files: &[ProcessedFile], + skipped: &[files::SkippedFile], + invalid: bool, + presentation: Presentation, + verbosity: Verbosity, +) -> Result { + let offered: usize = files.iter().map(|file| file.result.edits.len()).sum(); + let selection = { + let stdin = io::stdin(); + let mut answers = stdin.lock(); + let mut questions = output::stdout(); + let selection = interactive::select(files, &mut answers, &mut questions, &presentation)?; + /* NOTE: The conversation is on standard output and the verdict that follows + * is on standard error; a terminal sees both, so the buffer is emptied + * first to keep them in the order they were written. */ + output::finish(&mut questions)?; + selection + }; + let outcome = output::InteractiveOutcome { + removed: selection.accepted, + reviewed: selection.accepted + selection.declined, + offered, + changed: selection.plans.len(), + scanned: files.len(), + }; + let aborted = selection.aborted; + if !aborted { + apply_transaction(selection.plans)?; + } + let stderr = io::stderr(); + let mut report = stderr.lock(); + if aborted { + output::note(&mut report, "Aborted; nothing was written.")?; + return Ok(0); + } + /* NOTE: A skipped path can be the whole answer to a run that was never asked a + * question, so the one command that writes no report of its own still says + * why it passed a file over. */ + for line in output::skip_lines(skipped, presentation, verbosity) { + output::note(&mut report, &line)?; + } + output::note(&mut report, &output::interactive_summary(outcome))?; + if invalid { Ok(2) } else { Ok(0) } +} + +/// How the PATH list names standard input. +const STDIN_ARGUMENT: &str = "-"; + +/// Split the requested targets into ordinary paths and the `-` that stands for +/// standard input, refusing the combinations that cannot be honoured. +fn target_paths(paths: &[PathBuf], rewrites: bool, staged: bool) -> Result<(Vec, bool)> { + let is_stdin = |path: &PathBuf| path.as_os_str() == STDIN_ARGUMENT; + match paths.iter().filter(|path| is_stdin(path)).count() { + 0 => return Ok((paths.to_vec(), false)), + 1 => {} + /* NOTE: A pipe is consumed once; a second `-` would silently report the same + * bytes twice or nothing at all. */ + _ => bail!("cannot read standard input twice; `-` may appear only once"), + } + if rewrites { + bail!("cannot rewrite standard input in place; use `ocomment strip`"); + } + if staged { + bail!("cannot read standard input with --staged; the index is the source"); + } + Ok(( + paths + .iter() + .filter(|path| !is_stdin(path)) + .cloned() + .collect(), + true, + )) +} + +/// Discover the named paths and, when `-` was among them, fold the bytes read +/// from standard input in as one more file so a piped run takes exactly the +/// same reporting path as a walked one. +fn read_targets( + paths: &[PathBuf], + stdin: bool, + resolved: &config::ResolvedConfig, + common: &CommonArgs, +) -> Result { + if !stdin { + return files::discover(paths, resolved, common.language(), common.dialect()); + } + /* NOTE: An empty list means "the whole repository" only when no target was named + * at all; `-` on its own is a target, and walking would ignore it. */ + let mut discovery = if paths.is_empty() { + files::Discovery::default() + } else { + files::discover(paths, resolved, common.language(), common.dialect())? + }; + let mut bytes = Vec::new(); + io::stdin() + .lock() + .read_to_end(&mut bytes) + .context("cannot read standard input")?; + match files::stdin_source(bytes, resolved, common.language(), common.dialect()) { + Ok(file) => discovery.files.push(file), + /* NOTE: A skip that cannot be reported per file — nothing was named to skip + * — is a usage error the run must not swallow. */ + Err(skipped) if skipped.error => { + let reason = skipped.reason; + bail!("{reason}") + } + Err(skipped) => discovery.skipped.push(skipped), + } + discovery + .files + .sort_by(|left, right| left.path.cmp(&right.path)); + discovery + .skipped + .sort_by(|left, right| left.path.cmp(&right.path)); + Ok(discovery) +} + +/// Strip one file from standard input to standard output. +/// +/// The product is the stripped source itself — the bytes of the file, not a +/// report about it — so there is no report for a machine format to encode. +/// Writing the source under `--format sarif` would answer with something that +/// is not SARIF, and wrapping it in one of the schemas would answer with +/// something that is not the file, so the flag is refused the way +/// `ocomment languages` refuses the formats that carry no language table. fn run_strip(common: &CommonArgs) -> Result { + ensure!( + common.output.format == OutputFormat::Human, + "`ocomment strip` is only available with --format human" + ); let mut source = Vec::new(); io::stdin() .lock() .read_to_end(&mut source) .context("cannot read standard input")?; let mut resolved = config::load(common.config.as_deref())?; - apply_cli_overrides(&mut resolved.config, common); + apply_cli_overrides(&mut resolved, common); let detection = common - .language - .map(|language| (language, common.dialect.unwrap_or(Dialect::Standard))) + .language() + .map(|language| (language, common.dialect().unwrap_or(Dialect::Standard))) .or_else(|| { ocomment_core::detect_language(None, &source) .map(|value| (value.language, value.dialect)) }) - .context("cannot detect stdin language; pass --language")?; - let (language, mut options) = - resolved.for_path(std::path::Path::new(""), detection.0, detection.1); - if let Some(value) = common.dialect { + .context(files::STDIN_LANGUAGE_HELP)?; + let (language, mut options) = resolved.for_path( + std::path::Path::new(files::STDIN_PATH), + detection.0, + detection.1, + ); + if let Some(value) = common.dialect() { config::validate_dialect(language, value)?; options.scan.dialect = value; } let result = transform(&source, language, options); + let stderr = io::stderr(); + let mut report = stderr.lock(); for diagnostic in &result.report.diagnostics { - eprintln!( - "stdin:{}..{}: {}: {}", - diagnostic.span.start, diagnostic.span.end, diagnostic.code, diagnostic.message - ); + output::note( + &mut report, + &format!( + "stdin:{}..{}: {}: {}", + diagnostic.span.start, diagnostic.span.end, diagnostic.code, diagnostic.message + ), + )?; } if !result.report.valid && !resolved.config.policy.force_invalid { return Ok(2); } - io::stdout() - .lock() - .write_all(&result.output) - .context("cannot write standard output")?; + let mut stdout = output::stdout(); + output::wrote(stdout.write_all(&result.output))?; + output::finish(&mut stdout)?; Ok(if result.report.valid { 0 } else { 2 }) } -fn apply_cli_overrides(config: &mut config::Config, common: &CommonArgs) { - if let Some(value) = common.policy { - config.policy.mode = value; +/// Layer the command line over the merged configuration, noting what it +/// overrode so `--explain` can name the flag rather than a file that never +/// mentioned the setting. +fn apply_cli_overrides(resolved: &mut config::ResolvedConfig, common: &CommonArgs) { + let policy = &common.policy; + let config = &mut resolved.config; + let overrides = &mut resolved.cli_overrides; + if let Some(value) = policy.policy { + config.policy.mode = *value; + overrides.policy = true; } - if let Some(value) = common.layout { - config.policy.layout = value; + if let Some(value) = policy.layout { + config.policy.layout = *value; + overrides.layout = true; } - if !common.keep_kind.is_empty() { + if !policy.keep_kind.is_empty() { + /* NOTE: The flag adds to the configured list rather than replacing it, so + * the boundary is what tells the two apart afterwards. */ + overrides.keep_kind_from = Some(config.policy.keep_kind.len()); config .policy .keep_kind - .extend(common.keep_kind.iter().copied()); + .extend(policy.keep_kind.iter().copied().map(CommentKind::from)); } - if !common.remove_kind.is_empty() { + if !policy.remove_kind.is_empty() { + overrides.remove_kind_from = Some(config.policy.remove_kind.len()); config .policy .remove_kind - .extend(common.remove_kind.iter().copied()); + .extend(policy.remove_kind.iter().copied().map(CommentKind::from)); } - if common.force_invalid { + if policy.force_invalid { config.policy.force_invalid = true; } - if common.force_protected { + if policy.force_protected { config.policy.force_protected = true; } } +/// The apostrophe definition `roff` writes at the top of every fragment it +/// renders. A page needs it once, so it is stripped from every fragment after +/// the first. +const ROFF_PREAMBLE: &str = concat!(r".ie \n(.g .ds Aq \(aq", "\n", r".el .ds Aq '", "\n"); + +/// Append one rendered `roff` fragment to the page under construction. +fn append_fragment(page: &mut String, fragment: &[u8]) -> Result<()> { + let text = std::str::from_utf8(fragment).context("the manual page is not valid UTF-8")?; + page.push_str(text.strip_prefix(ROFF_PREAMBLE).unwrap_or(text)); + Ok(()) +} + +/// Render the arguments that belong to one command alone, as `.SS` subsections. +/// +/// `clap_mangen` renders a single page for the root command, so an argument +/// declared on a subcommand — `fix --dry-run`, `init --force`, `plugin add +/// --sha256` — would never reach the manual at all. Every command is walked +/// and the arguments it does not inherit are written under its own heading. +fn command_options(command: &clap::Command, path: &str, page: &mut String) -> Result<()> { + for subcommand in command.get_subcommands() { + if subcommand.is_hide_set() || subcommand.get_name() == "help" { + continue; + } + let name = format!("{path} {}", subcommand.get_name()); + /* NOTE: The global arguments already have one entry each under OPTIONS, + * POLICY, and OUTPUT, and `--help` is on every command by definition. + * Repeating them here would bury the few arguments this section is + * for. Hiding is how `clap_mangen` is told to skip an argument. */ + let mut own = subcommand.clone(); + let inherited: Vec = own + .get_arguments() + .filter(|argument| { + argument.is_global_set() + || argument.get_id() == "help" + || argument.get_id() == "version" + }) + .map(|argument| argument.get_id().clone()) + .collect(); + for id in inherited { + own = own.mut_arg(id, |argument| argument.hide(true)); + } + let mut fragment = Vec::new(); + clap_mangen::Man::new(own) + .render_options_section(&mut fragment) + .context("cannot render the manual page")?; + let mut rendered = String::new(); + append_fragment(&mut rendered, &fragment)?; + /* NOTE: A command with nothing of its own renders an empty fragment, and an + * empty heading would claim otherwise. */ + if let Some(body) = rendered.strip_prefix(".SH OPTIONS\n") + && !body.is_empty() + { + page.push_str(&format!(".SS {}\n{body}", name.replace('-', "\\-"))); + } + command_options(subcommand, &name, page)?; + } + Ok(()) +} + +/// Render the roff manual page from the parser definition itself. +fn run_man() -> Result { + /* NOTE: `clap_mangen` renders `after_long_help` as one opaque `.SH EXTRA` body, + * so the page is built without it and the same content is appended below + * as real roff sections. It is assembled section by section rather than + * through `render`, because the per-command options belong next to the + * command list and `render` puts VERSION after it. + * + * The `.TH` date is left blank on purpose: stamping the build date would + * make two reproducible builds of the same source disagree. */ + let man = clap_mangen::Man::new(Cli::command().after_long_help(None)) + .title("OCOMMENT") + .manual("User Commands"); + let mut page = String::new(); + let mut fragment = Vec::new(); + // NOTE: The title fragment keeps the apostrophe definition the whole page needs. + man.render_title(&mut fragment) + .context("cannot render the manual page")?; + page.push_str(std::str::from_utf8(&fragment).context("the manual page is not valid UTF-8")?); + type Section = fn(&clap_mangen::Man, &mut dyn Write) -> io::Result<()>; + for section in [ + clap_mangen::Man::render_name_section as Section, + clap_mangen::Man::render_synopsis_section, + clap_mangen::Man::render_description_section, + clap_mangen::Man::render_options_section, + clap_mangen::Man::render_subcommands_section, + ] { + fragment.clear(); + section(&man, &mut fragment).context("cannot render the manual page")?; + append_fragment(&mut page, &fragment)?; + } + let mut per_command = String::new(); + let mut root = Cli::command(); + /* NOTE: Building propagates the global arguments into every subcommand, which is + * what makes them recognizable as inherited below. */ + root.build(); + command_options(&root, "ocomment", &mut per_command)?; + if !per_command.is_empty() { + page.push_str(".SH COMMAND OPTIONS\n"); + page.push_str(&per_command); + } + fragment.clear(); + man.render_version_section(&mut fragment) + .context("cannot render the manual page")?; + append_fragment(&mut page, &fragment)?; + if !page.ends_with('\n') { + page.push('\n'); + } + page.push_str(MAN_SECTIONS); + let mut stdout = output::stdout(); + output::wrote(stdout.write_all(page.as_bytes()))?; + output::finish(&mut stdout)?; + Ok(0) +} + +/// Write the shell completion script. +/// +/// `clap_complete` writes straight into the handle it is given and panics if +/// that write fails, so it is given a buffer in memory and the one write that +/// can fail is made here. +fn run_completions(shell: Shell) -> Result { + let mut script = Vec::new(); + generate(shell, &mut Cli::command(), "ocomment", &mut script); + let mut stdout = output::stdout(); + output::wrote(stdout.write_all(&script))?; + output::finish(&mut stdout)?; + Ok(0) +} + fn run_init(args: InitArgs) -> Result { - match args.kind { - InitKind::Config => create_new( + /* NOTE: Writing the file is only the first half of the task, so each template + * carries the step that finishes it. */ + let (path, contents, next_step) = match args.kind { + InitKind::Config => ( config::CONFIG_FILE, - include_str!("../assets/default-config.toml"), - )?, + include_str!("../assets/default-config.toml").to_owned(), + "edit [policy] and run `ocomment check`", + ), InitKind::Lefthook => { let command = if args.fix { "ocomment fix --staged" } else { "ocomment check --staged" }; - create_new( + ( "lefthook.yml", - &format!("pre-commit:\n commands:\n ocomment:\n run: {command}\n"), - )?; + format!("pre-commit:\n commands:\n ocomment:\n run: {command}\n"), + "run `lefthook install` to activate the hook", + ) } + }; + let mut stdout = output::stdout(); + if args.stdout { + /* NOTE: Nothing is created, so nothing is said about creating it: the + * template alone is on standard output, ready to be redirected. */ + output::wrote(write!(stdout, "{contents}"))?; + output::finish(&mut stdout)?; + return Ok(0); } + write_template(&mut stdout, path, &contents, args.force, next_step)?; + /* NOTE: The note is advice about the file that now exists, so it follows the + * line that reports it — and a refused `init` never reaches it, because + * there is no new file for an inherited configuration to layer under. + * Standard output is flushed first so a terminal reading both streams sees + * the creation before the note about it. */ + output::finish(&mut stdout)?; + note_inherited_config()?; Ok(0) } -fn create_new(path: &str, contents: &str) -> Result<()> { - let mut file = fs::OpenOptions::new() - .write(true) - .create_new(true) - .open(path) - .with_context(|| format!("refusing to overwrite {path}"))?; - file.write_all(contents.as_bytes())?; - println!("created {path}"); +/// Say so when a project configuration from a parent directory already governs +/// this directory. +/// +/// The starter file layers over it rather than starting from nothing, and the +/// hook a `lefthook` run installs will read it — either way the reader is +/// better off knowing before they start editing. It is a note and not a +/// refusal: a nested per-crate configuration is a normal thing to want. +/// +/// The search starts at the parent so that the file this very run is about to +/// write — or the one `--force` is replacing — is never reported as inherited. +fn note_inherited_config() -> Result<()> { + let Ok(directory) = std::env::current_dir() else { + return Ok(()); + }; + let Some(inherited) = directory.parent().and_then(config::locate_project) else { + return Ok(()); + }; + let stderr = io::stderr(); + let mut report = stderr.lock(); + output::note( + &mut report, + &format!( + "note: {} already applies to this directory", + inherited.display() + ), + ) +} + +/// Write one starter file, refusing an existing one unless `force` says +/// otherwise. +/// +/// The refusal is `create_new` rather than a prior `exists()` test: between +/// such a test and the open the file could appear, and never writing over +/// someone's edited configuration is the whole point of the check. +fn write_template( + output: &mut impl Write, + path: &str, + contents: &str, + force: bool, + next_step: &str, +) -> Result<()> { + let mut options = fs::OpenOptions::new(); + options.write(true); + if force { + options.create(true).truncate(true); + } else { + options.create_new(true); + } + let mut file = options.open(path).map_err(|error| { + if error.kind() == io::ErrorKind::AlreadyExists { + anyhow::anyhow!( + "{path} already exists; use --force to overwrite or --stdout to print the template" + ) + } else { + anyhow::Error::new(error).context(format!("cannot write {path}")) + } + })?; + file.write_all(contents.as_bytes()) + .with_context(|| format!("cannot write {path}"))?; + output::wrote(writeln!(output, "created {path} — {next_step}"))?; Ok(()) } +/// Answer one question about the configuration. +/// +/// Every answer here is about settings rather than about comments: the merged +/// file as TOML, where the files were found, how they were layered, and the +/// schema they are checked against. None of the report schemas has a place to +/// put any of that — `--format json` would name the report format, not the +/// TOML `show` writes or the JSON Schema `schema` writes — so the flag is +/// refused rather than accepted and ignored. fn run_config(args: ConfigArgs, common: &CommonArgs) -> Result { + ensure!( + common.output.format == OutputFormat::Human, + "`ocomment config` is only available with --format human" + ); + let mut stdout = output::stdout(); match args.action { - ConfigAction::Schema => print!("{}", include_str!("../assets/config.schema.json")), + ConfigAction::Schema => { + output::wrote(write!( + stdout, + "{}", + include_str!("../assets/config.schema.json") + ))?; + } action => { let mut resolved = config::load(common.config.as_deref())?; - apply_cli_overrides(&mut resolved.config, common); + apply_cli_overrides(&mut resolved, common); match action { ConfigAction::Show => { resolved.config.version = Some(1); - print!("{}", toml::to_string_pretty(&resolved.config)?); + output::wrote(write!( + stdout, + "{}", + toml::to_string_pretty(&resolved.config)? + ))?; } ConfigAction::Locate => { if let Some(path) = &resolved.trace.user { - println!("user\t{}", path.display()); + output::wrote(writeln!(stdout, "user\t{}", path.display()))?; } if let Some(path) = &resolved.trace.project { - println!("project\t{}", path.display()); + output::wrote(writeln!(stdout, "project\t{}", path.display()))?; } if let Some(path) = &resolved.trace.explicit { - println!("explicit\t{}", path.display()); + output::wrote(writeln!(stdout, "explicit\t{}", path.display()))?; } if resolved.trace.user.is_none() && resolved.trace.project.is_none() && resolved.trace.explicit.is_none() { - println!("built-in defaults"); + output::wrote(writeln!(stdout, "built-in defaults"))?; } } ConfigAction::Explain => { - println!("precedence: built-in < XDG user < project < path override < CLI"); - println!("root: {}", resolved.root.display()); - println!( - "policy: {:?}; layout: {:?}", + output::wrote(writeln!( + stdout, + "precedence: built-in < XDG user < project < path override < CLI" + ))?; + output::wrote(writeln!(stdout, "root: {}", root_row(&resolved)))?; + output::wrote(writeln!( + stdout, + "policy: {}; layout: {}", resolved.config.policy.mode, resolved.config.policy.layout - ); + ))?; } ConfigAction::Schema => unreachable!(), } } } + output::finish(&mut stdout)?; Ok(0) } -fn print_languages() { - println!("language\textensions / guaranteed dialects"); - println!("rust\trs"); - println!("ocaml\tml,mli (OCaml 5.5 lexical forms)"); - println!("c\tc,h / standard, GNU, Objective-C"); - println!("cpp\tcc,cpp,cxx,hpp / standard, GNU, Objective-C++, CUDA"); - println!("go\tgo"); - println!("java\tjava (Unicode escape translation)"); - println!("javascript\tjs,mjs,cjs,jsx / ECMAScript, JSX"); - println!("typescript\tts,mts,cts,tsx / TypeScript, TSX"); - println!("python\tpy,pyw,pyi"); - println!("shell\tsh,bash,zsh / POSIX sh, Bash 5.3, zsh"); - println!("html\thtml,htm / recursive script and style"); - println!("css\tcss"); - println!("jsonc\tjsonc,json5"); - println!("sql\tsql / PostgreSQL, MySQL, SQLite, T-SQL, Oracle"); - println!("kotlin\tkt,kts"); +/// The shared language table, embedded from `spec/languages.toml` at build +/// time so a released binary carries the same list the repository publishes. +/// `tools/check_embedded_specs.py` and `spec_languages.rs` both fail when the +/// copy under `assets/` stops being the canonical file. +const LANGUAGE_TABLE: &str = include_str!("../assets/languages.toml"); + +/// One language of the shared table. +/// +/// The field names are the keys of `spec/languages.toml` and the members of the +/// objects `--format json` writes; the three that can be empty are left out of +/// the JSON rather than written as an empty collection, so a reader can tell +/// "no reserved names" from "reserved names not described". +#[derive(Debug, Deserialize, Serialize)] +#[serde(deny_unknown_fields)] +struct LanguageRow { + /// The canonical language name, which is also what `--language` takes. + name: String, + /// Every file extension that selects the language, without the dot. + extensions: Vec, + /// Every dialect the language accepts, in the order `--dialect` names them + /// when it refuses one. + dialects: Vec, + /// The extensions that select a dialect other than `standard`. + #[serde(default, skip_serializing_if = "BTreeMap::is_empty")] + extension_dialects: BTreeMap, + /// Whole file names that select the language when the extension does not. + #[serde(default, skip_serializing_if = "Vec::is_empty")] + reserved_names: Vec, + /// Interpreter names that select the language from a `#!` line. + #[serde(default, skip_serializing_if = "Vec::is_empty")] + shebangs: Vec, + /// A short remark for the last column of the human listing. + #[serde(default, skip_serializing_if = "Option::is_none")] + notes: Option, +} + +/// The shared table as a whole. +#[derive(Debug, Deserialize)] +#[serde(deny_unknown_fields)] +struct LanguageTable { + /// The schema version of the file; only `1` exists. + version: u32, + /// One row per built-in language, in the order they are listed. + languages: Vec, +} + +/// Read the embedded table. +/// +/// A failure here is a broken build rather than a broken run — the bytes are +/// compiled in — so the error says which file is at fault instead of blaming +/// the command line. +fn language_table() -> Result> { + let table: LanguageTable = toml::from_str(LANGUAGE_TABLE) + .context("the embedded spec/languages.toml is not a language table")?; + ensure!( + table.version == 1, + "the embedded spec/languages.toml is version {}, which this build does not know", + table.version + ); + Ok(table.languages) +} + +/// Print the shared language table. +/// +/// The human listing is one tab-separated row per language — name, extensions, +/// dialects, and the spec's remark where it has one — and `--format json` +/// writes the same rows as an array of objects. The other formats are report +/// schemas with nowhere to put a language table, so they are refused rather +/// than quietly answered with the human one. +fn print_languages(common: &CommonArgs) -> Result { + let rows = language_table()?; + let mut stdout = output::stdout(); + match common.output.format { + OutputFormat::Human => { + output::wrote(writeln!(stdout, "language\textensions\tdialects\tnotes"))?; + for row in &rows { + let extensions = row.extensions.join(","); + let dialects = row.dialects.join(","); + let line = match &row.notes { + Some(notes) => format!("{}\t{extensions}\t{dialects}\t{notes}", row.name), + None => format!("{}\t{extensions}\t{dialects}", row.name), + }; + output::wrote(writeln!(stdout, "{line}"))?; + } + } + /* NOTE: Rendered into a string rather than straight into the writer, so the + * one write is raised through `output::wrote` and a reader that closed + * the pipe still ends the run quietly. */ + OutputFormat::Json => { + let json = serde_json::to_string_pretty(&rows) + .context("cannot render the language table as JSON")?; + output::wrote(writeln!(stdout, "{json}"))?; + } + _ => bail!("`ocomment languages` is only available with --format human or --format json"), + } + output::finish(&mut stdout)?; + Ok(0) } fn run_plugin(args: PluginArgs, common: &CommonArgs) -> Result { let resolved = config::load(common.config.as_deref())?; + let mut stdout = output::stdout(); match args.command { PluginCommand::Add { source, @@ -460,63 +1359,322 @@ fn run_plugin(args: PluginArgs, common: &CommonArgs) -> Result { sha256, identity, } => plugin::add( + &mut stdout, &resolved.root, &source, name.as_deref(), sha256.as_deref(), identity.as_deref(), )?, - PluginCommand::Remove { name } => plugin::remove(&resolved.root, &name)?, - PluginCommand::List => plugin::list(&resolved.root)?, - PluginCommand::Update { name } => plugin::update(&resolved.root, name.as_deref())?, - PluginCommand::Verify { name } => plugin::verify(&resolved.root, name.as_deref())?, - PluginCommand::New { path } => plugin::new_plugin(&path)?, + PluginCommand::Remove { name } => plugin::remove(&mut stdout, &resolved.root, &name)?, + PluginCommand::List => plugin::list(&mut stdout, &resolved.root)?, + PluginCommand::Update { name } => { + plugin::update(&mut stdout, &resolved.root, name.as_deref())?; + } + PluginCommand::Verify { name } => { + plugin::verify(&mut stdout, &resolved.root, name.as_deref())?; + } + PluginCommand::New { path } => plugin::new_plugin(&mut stdout, &path)?, } + output::finish(&mut stdout)?; Ok(0) } +/// What `git` is needed for. The four plugin purposes are declared beside the +/// spawn sites that name them in a failure; this one belongs to the flag it +/// serves, and is worded the same way so the rows read alike. +const STAGED_READS: &str = "--staged"; + +/// The optional external tools OComment shells out to, in the order `doctor` +/// reports them: the binary, the arguments that make it identify itself, and +/// the part of a run that stops working without it. Not one of them is needed +/// to check or fix a file, so a missing tool is a row in the report and never +/// a failing run. +const PROBED_TOOLS: [(&str, &[&str], &str); 5] = [ + ("git", &["--version"], STAGED_READS), + ("curl", &["--version"], plugin::HTTPS_SOURCES), + ("gh", &["--version"], plugin::GH_SOURCES), + ("oras", &["version"], plugin::OCI_SOURCES), + ("cosign", &["version"], plugin::SIGNATURE_VERIFICATION), +]; + +/// What one external tool answered when `doctor` asked it to identify itself. +enum Probe { + /// It ran and said this about itself. + Found(String), + /// Nothing by that name is on `PATH`. + Missing, + /// It is there but could not be started, or answered with a failure. + Failed(String), +} + +/// Ask one external tool for its version. +/// +/// The answer is read from standard output, or from standard error for the +/// tools that put their banner there, and it is the line the tool chose: +/// `doctor` reports what a tool says about itself rather than parsing it into +/// fields that the next release would rename. +fn probe(tool: &str, args: &[&str]) -> Probe { + let output = match std::process::Command::new(tool).args(args).output() { + Ok(output) => output, + Err(error) if error.kind() == io::ErrorKind::NotFound => return Probe::Missing, + Err(error) => return Probe::Failed(error.to_string()), + }; + let version = version_line(&output.stdout).or_else(|| version_line(&output.stderr)); + match (output.status.success(), version) { + (true, Some(line)) => Probe::Found(line), + (true, None) => Probe::Failed("ran but said nothing about itself".to_owned()), + (false, Some(line)) => Probe::Failed(line), + (false, None) => Probe::Failed(output.status.to_string()), + } +} + +/// The line a tool identifies itself by, out of everything it printed. +/// +/// Usually that is the first line carrying anything, but `cosign version` +/// draws six lines of ASCII art before it mentions a version, and a row +/// showing the top of that banner would tell the reader nothing. A version has +/// a number in it, so the first line with a digit wins and the first non-empty +/// line is the fallback for a tool that names no number at all. +/// +/// A banner that is not UTF-8 is still worth showing, so the bytes are read +/// lossily rather than dropped, and one line of it is kept: a row of the +/// report stays one line whatever the tool decided to print. +/// +/// The tool chose those bytes, so the line it identifies itself by is +/// untrusted input on its way to a terminal, and it is sanitised exactly like +/// a comment preview before it becomes a row. +fn version_line(bytes: &[u8]) -> Option { + let text = String::from_utf8_lossy(bytes); + let mut fallback = None; + for line in text.lines().map(str::trim).filter(|line| !line.is_empty()) { + if line.chars().any(|character| character.is_ascii_digit()) { + return Some(output::sanitize_line(line)); + } + fallback.get_or_insert_with(|| output::sanitize_line(line)); + } + fallback +} + fn run_doctor(common: &CommonArgs) -> Result { - println!("ocomment {}", env!("CARGO_PKG_VERSION")); + /* NOTE: Asked before standard output is locked for the report, so the answer is + * about the same handle the report is written to. */ + let stdout_tty = io::stdout().is_terminal(); + let mut stdout = output::stdout(); + output::wrote(writeln!(stdout, "ocomment {}", env!("CARGO_PKG_VERSION")))?; + match std::env::current_dir() { + Ok(directory) => output::wrote(writeln!( + stdout, + "cwd: {}", + output::sanitize_path(&directory.to_string_lossy()) + ))?, + Err(error) => output::wrote(writeln!(stdout, "cwd: unavailable ({error})"))?, + } let resolved = config::load(common.config.as_deref())?; - println!("configuration: ok (root {})", resolved.root.display()); - println!("languages: {} built in", Language::BUILT_INS.len()); - if std::process::Command::new("git") - .arg("--version") - .output() - .is_ok() - { - println!("git: available"); - } else { - println!("git: unavailable (only --staged is affected)"); + output::wrote(writeln!(stdout, "root: {}", root_row(&resolved)))?; + for source in config_trace(&resolved.trace) { + output::wrote(writeln!(stdout, "config: {source}"))?; + } + output::wrote(writeln!(stdout, "configuration: ok"))?; + output::wrote(writeln!( + stdout, + "languages: {} built in", + Language::ALL.len() + ))?; + /* NOTE: Whether the report is decorated is the first thing a reader piping it + * somewhere wants explained, and both halves of that answer are here. */ + output::wrote(writeln!( + stdout, + "stdout: {}", + if stdout_tty { + "a terminal" + } else { + "not a terminal" + } + ))?; + output::wrote(writeln!( + stdout, + "NO_COLOR: {}", + if std::env::var_os("NO_COLOR").is_some() { + "set" + } else { + "unset" + } + ))?; + for (tool, arguments, purpose) in PROBED_TOOLS { + let row = match probe(tool, arguments) { + Probe::Found(version) => format!("{tool}: {version}"), + Probe::Missing => format!("{tool}: not found (needed for {purpose})"), + Probe::Failed(reason) => format!("{tool}: failed (needed for {purpose}): {reason}"), + }; + output::wrote(writeln!(stdout, "{row}"))?; } - plugin::verify(&resolved.root, None)?; - println!( + plugin::verify(&mut stdout, &resolved.root, None)?; + output::wrote(writeln!( + stdout, "LSP: stdio server available; on-save is opt-in ({})", resolved.config.lsp.on_save - ); + ))?; + output::finish(&mut stdout)?; Ok(0) } +/// How many files may be processed between two redraws of the counter. +const PROGRESS_STEP: usize = 50; + +/// Whether this run draws the live scanning counter. The counter is terminal +/// decoration: it never belongs in a machine format, and `-q` silences it. +fn progress_enabled(common: &CommonArgs) -> bool { + common.output.format == OutputFormat::Human + && common.verbosity() != Verbosity::Quiet + && match common.output.progress { + AutoChoice::Auto => io::stderr().is_terminal(), + AutoChoice::Always => true, + AutoChoice::Never => false, + } +} + +/// The live scanning counter: how many files it has seen, and whether it ever +/// put a line on the screen. +#[derive(Default)] +struct Progress { + scanned: AtomicUsize, + drawn: AtomicBool, +} + +impl Progress { + /// Advance the live `n/total` counter, rewriting one line on standard + /// error rather than scrolling a line for every file. + fn report(&self, total: usize) { + let seen = self.scanned.fetch_add(1, Ordering::Relaxed) + 1; + if !seen.is_multiple_of(PROGRESS_STEP) && seen != total { + return; + } + let mut stderr = io::stderr().lock(); + let _ = write!(stderr, "\rocomment: scanning {seen}/{total} files"); + let _ = stderr.flush(); + self.drawn.store(true, Ordering::Relaxed); + } + + /// Erase the counter so the report that follows starts on a clean line. + /// + /// A run with nothing to scan draws no counter, and erasing a line it + /// never wrote would put an escape sequence on a standard error whose + /// reader was promised only the summary. + fn clear(&self) { + if !self.drawn.load(Ordering::Relaxed) { + return; + } + let mut stderr = io::stderr().lock(); + let _ = write!(stderr, "\r\x1b[2K"); + let _ = stderr.flush(); + } +} + fn presentation(common: &CommonArgs) -> Presentation { let stdout_tty = io::stdout().is_terminal(); - let stderr_tty = io::stderr().is_terminal(); let no_color = std::env::var_os("NO_COLOR").is_some(); Presentation { color: !no_color - && match common.color { + && match common.output.color { ColorChoice::Auto => stdout_tty, ColorChoice::Always => true, ColorChoice::Never => false, }, - hyperlinks: match common.hyperlinks { + hyperlinks: match common.output.hyperlinks { AutoChoice::Auto => stdout_tty, AutoChoice::Always => true, AutoChoice::Never => false, }, - progress: match common.progress { - AutoChoice::Auto => stderr_tty, - AutoChoice::Always => true, - AutoChoice::Never => false, - }, } } + +/// The project root, as a report names it. +/// +/// A directory name is chosen by whoever made the directory, not by OComment, +/// so a row carrying one is untrusted text on its way to a terminal for the +/// same reason a probed tool's version line is — and, unlike one, it must not +/// be cut short: a path that ends in an ellipsis names no directory at all. +fn root_row(resolved: &config::ResolvedConfig) -> String { + output::sanitize_path(&resolved.root.to_string_lossy()) +} + +/// What the run was pointed at, in the words the caller used, or the implicit +/// target that stands in when they named nothing. +fn target_label(paths: &[PathBuf]) -> String { + if paths.is_empty() { + return files::DEFAULT_TARGET.to_owned(); + } + paths + .iter() + .map(|path| path.display().to_string()) + .collect::>() + .join(" ") +} + +/// Say where a bare `fix` is pointed when that is not where the project +/// starts. +/// +/// A reader who has only ever run `ocomment fix` from the top of a repository +/// can read the bare command as "fix the project", and it is the one command +/// that writes. So the run that was told nothing about where to write names +/// both the target it chose and the root the configuration came from, once, +/// before it starts. A caller who named a path has already said what they +/// meant, and from the root itself the two are the same directory: either way +/// the line would be noise. +fn note_fix_scope(resolved: &config::ResolvedConfig, common: &CommonArgs) -> Result<()> { + if resolved.cwd == resolved.root + || common.output.format != OutputFormat::Human + || common.verbosity() == Verbosity::Quiet + { + return Ok(()); + } + let stderr = io::stderr(); + let mut report = stderr.lock(); + output::note( + &mut report, + &format!( + "note: fixing files under {} (project root: {})", + files::DEFAULT_TARGET, + root_row(resolved) + ), + ) +} + +/// The `--verbose` header: where the run is rooted, what it was pointed at, +/// and which configuration files it merged. +fn trace_run(resolved: &config::ResolvedConfig, paths: &[PathBuf]) -> Result<()> { + let stderr = io::stderr(); + let mut report = stderr.lock(); + output::note(&mut report, &format!("root: {}", root_row(resolved)))?; + output::note(&mut report, &format!("target: {}", target_label(paths)))?; + for source in config_trace(&resolved.trace) { + output::note(&mut report, &format!("config: {source}"))?; + } + Ok(()) +} + +/// Which configuration files a run merged, one line each in the order they +/// were layered, or the single line that says there were none. +/// +/// `doctor` and the `-v` trace both report this, and a reader comparing the +/// two is entitled to read the same answer twice, so they read it from here. +fn config_trace(trace: &config::ConfigTrace) -> Vec { + let sources: Vec = [ + ("user", &trace.user), + ("project", &trace.project), + ("explicit", &trace.explicit), + ] + .into_iter() + .filter_map(|(label, path)| { + /* INVARIANT: The row carries a directory name OComment did not choose, so it is + * sanitised for the same reason a `root` row is. */ + path.as_ref() + .map(|path| format!("{label} {}", output::sanitize_path(&path.to_string_lossy()))) + }) + .collect(); + if sources.is_empty() { + return vec!["built-in defaults".to_owned()]; + } + sources +} diff --git a/rust/ocomment/src/config.rs b/rust/ocomment/src/config.rs index d75e262..3f9b4a8 100644 --- a/rust/ocomment/src/config.rs +++ b/rust/ocomment/src/config.rs @@ -1,14 +1,14 @@ use anyhow::{Context, Result, anyhow, bail, ensure}; use globset::{Glob, GlobMatcher}; use ocomment_core::{ - CommentKind, DeclarativeProfile, Dialect, Language, Layout, Policy, ScanOptions, - TransformOptions, validate_profile, + CommentKind, DeclarativeProfile, Dialect, DispositionExplanation, Language, Layout, Policy, + ScanOptions, TransformOptions, validate_profile, }; use serde::{Deserialize, Serialize}; use std::{ collections::BTreeMap, env, fs, - path::{Path, PathBuf}, + path::{Component, Path, PathBuf}, }; pub const CONFIG_FILE: &str = ".ocomment.toml"; @@ -142,6 +142,199 @@ pub struct PathOverride { pub remove_regex: Vec, } +/// Where one effective setting came from. +/// +/// The layers are the ones [`ResolvedConfig::for_path`] merges, and a source +/// names the layer a value arrived on rather than the value itself, so +/// `--explain` can send a reader to the table they have to edit. +#[derive(Clone, Debug, Default, Eq, PartialEq)] +pub enum Source { + /// The `[policy]` table, or the built-in default when no file set it. + #[default] + Global, + /// A `[languages.]` table. + Language(String), + /// The `[[overrides]]` table at `index`, whose globs matched the path. + Override { index: usize, paths: Vec }, + /// A flag on the command line. + Cli { flag: &'static str }, +} + +impl Source { + /// How an explanation names this source, given the file a `Global` value + /// was written in. + /// + /// `#N` counts an `[[overrides]]` table from zero, the way the regex + /// indices printed beside it count the patterns they address. + fn describe(&self, origin: Option<&Path>) -> String { + match self { + Self::Global => match origin { + Some(path) => format!("[policy] in {}", path.display()), + None => "built-in defaults".to_owned(), + }, + Self::Language(name) => format!("[languages.{name}]"), + Self::Override { index, paths } => { + format!("[[overrides]] #{index}, paths = {paths:?}") + } + Self::Cli { flag } => format!("{flag} on the command line"), + } + } +} + +/// The `[policy]` keys a trace can attribute to a file, spelled as the file +/// spells them. +const POLICY_KEYS: [&str; 6] = [ + "mode", + "layout", + "keep_kind", + "remove_kind", + "keep_regex", + "remove_regex", +]; + +/// Which configuration file last set each `[policy]` key. A key no file sets +/// keeps no entry, and an explanation calls it a built-in default rather than +/// sending the reader to a file that never mentions it. +type PolicyOrigins = BTreeMap<&'static str, PathBuf>; + +/// What the command line overrode, recorded while it was applied so a trace +/// can say the command line rather than the file the value would otherwise +/// have been written in. +#[derive(Clone, Copy, Debug, Default)] +pub struct CliOverrides { + pub policy: bool, + pub layout: bool, + /// Where the `--keep-kind` values start in `policy.keep_kind`: the command + /// line appends to the configured list instead of replacing it, so only + /// the tail of that list belongs to the command line. + pub keep_kind_from: Option, + /// The same boundary for `--remove-kind` in `policy.remove_kind`. + pub remove_kind_from: Option, +} + +/// Where every effective setting for one path came from. +/// +/// The `*_kind` and `*_regex` vectors run parallel to the vectors in the +/// [`ScanOptions`] that [`ResolvedConfig::for_path_traced`] returned beside +/// this: entry `i` says which layer contributed entry `i` of that list. +#[derive(Clone, Debug, Default)] +pub struct PolicyTrace { + pub policy: Source, + /// Where the layout came from. No disposition depends on the layout, so no + /// explanation names it yet; it is recorded because the trace answers the + /// question for the whole `[policy]` block and `--explain` will not be its + /// only caller. + #[allow(dead_code)] + pub layout: Source, + pub keep_kind: Vec, + pub remove_kind: Vec, + pub keep_regex: Vec, + pub remove_regex: Vec, + origins: PolicyOrigins, +} + +impl PolicyTrace { + /// Where the setting that decided `explanation` came from, worded the way + /// `--explain` prints it, or `None` when a built-in rule decided it and + /// there is no table to point at. + /// + /// `options` is the one the explanation was produced from: a regex + /// explanation carries its index into those lists, and a kind explanation + /// is found by the kind it names. + pub fn origin_of( + &self, + explanation: &DispositionExplanation, + options: &ScanOptions, + ) -> Option { + let position = |kinds: &[CommentKind], kind: &CommentKind| { + kinds.iter().position(|value| value == kind) + }; + let (source, key) = match explanation { + DispositionExplanation::KeptByKind(kind) => ( + self.keep_kind.get(position(&options.keep_kinds, kind)?)?, + "keep_kind", + ), + DispositionExplanation::RemovedByKind(kind) => ( + self.remove_kind + .get(position(&options.remove_kinds, kind)?)?, + "remove_kind", + ), + DispositionExplanation::KeptByRegex { index, .. } => { + (self.keep_regex.get(*index)?, "keep_regex") + } + DispositionExplanation::RemovedByRegex { index, .. } => { + (self.remove_regex.get(*index)?, "remove_regex") + } + /* NOTE: Every one of these is the policy having the last word, whether it + * took the comment out or protected it. */ + DispositionExplanation::RemovedByPolicy(_) + | DispositionExplanation::RemovedByDefault(_) + | DispositionExplanation::KeptLicense { .. } => (&self.policy, "mode"), + // NOTE: A built-in rule, decided by no setting at all. + DispositionExplanation::ProtectedPreamble + | DispositionExplanation::KeptHtml + | DispositionExplanation::KeptDirective { .. } + | DispositionExplanation::KeptStructural { .. } => return None, + }; + Some(source.describe(self.origins.get(key).map(PathBuf::as_path))) + } +} + +/// One layer of the policy merge, as the trace replays it. +struct TracedLayer<'a> { + source: Source, + policy: Option, + layout: Option, + keep_kind: &'a [CommentKind], + remove_kind: &'a [CommentKind], + keep_regex: &'a [String], + remove_regex: &'a [String], +} + +/// Attribute every entry of one merged list to the layer that introduced it. +/// +/// [`ResolvedConfig::for_path`] starts from the global list verbatim and then +/// appends whatever a later layer adds that is not there yet, so replaying that +/// walk reproduces the merged list position for position. The tail of the +/// global list from `cli_from` on is what `flag` appended to it. +fn attribute( + global: &[T], + cli_from: Option, + flag: &'static str, + layers: &[(&Source, &[T])], +) -> Vec { + let cli_from = cli_from.unwrap_or(usize::MAX); + let mut sources: Vec = (0..global.len()) + .map(|index| { + if index >= cli_from { + Source::Cli { flag } + } else { + Source::Global + } + }) + .collect(); + let mut seen = global.to_vec(); + for (source, values) in layers { + for value in *values { + if !seen.contains(value) { + seen.push(value.clone()); + sources.push((*source).clone()); + } + } + } + sources +} + +/// Where a setting that holds a single value came from, before any language or +/// path layer has had its say. +fn scalar_source(overridden: bool, flag: &'static str) -> Source { + if overridden { + Source::Cli { flag } + } else { + Source::Global + } +} + struct CompiledOverride { matchers: Vec, value: PathOverride, @@ -157,19 +350,55 @@ pub struct ConfigTrace { pub struct ResolvedConfig { pub config: Config, pub trace: ConfigTrace, + /// Where the project starts: the directory `.ocomment.toml` was found in, + /// the repository above the working directory, or the working directory + /// itself. It decides where configuration is discovered, what the file and + /// override globs are written relative to, and where the plugin lock + /// lives — no longer what a command with no path walks. pub root: PathBuf, + /// The directory the command was run from, which is what a path typed on + /// the command line is relative to. + pub cwd: PathBuf, + /// What the command line overrode, filled in after the files were merged. + pub cli_overrides: CliOverrides, overrides: Vec, + origins: PolicyOrigins, } impl ResolvedConfig { + /// Where `path` sits under the project root, spelled the way a + /// configuration glob is written. + /// + /// `files.include`, `files.exclude`, and every `[[overrides]].paths` + /// pattern is relative to the root, while a path named on the command line + /// is relative to the working directory. The two agree only when the + /// command is run from the root, so the path is resolved against the + /// directory it was typed in before it is measured against the root, and + /// the separators come out as forward slashes so one glob reads the same + /// on every platform. + /// + /// A path outside the root — an explicit target above it, say — has no + /// root-relative spelling at all, so it keeps its absolute one and only an + /// absolute glob can match it. + pub fn relative_to_root(&self, path: &Path) -> String { + /* NOTE: Standard input has no place on disk. The pseudo-path is what the + * renderers print, so it is also what the globs are shown. */ + if path.as_os_str() == crate::files::STDIN_PATH { + return crate::files::STDIN_PATH.to_owned(); + } + let joined = self.cwd.join(path); + let absolute = lexical(&std::path::absolute(&joined).unwrap_or(joined)); + let relative = absolute.strip_prefix(&self.root).unwrap_or(&absolute); + relative.to_string_lossy().replace('\\', "/") + } + pub fn for_path( &self, path: &Path, language: Language, dialect: Dialect, ) -> (Language, TransformOptions) { - let relative = path.strip_prefix(&self.root).unwrap_or(path); - let normalized = relative.to_string_lossy().replace('\\', "/"); + let normalized = self.relative_to_root(path); let mut chosen_language = language; let mut chosen_dialect = dialect; let mut policy = self.config.policy.mode; @@ -230,6 +459,109 @@ impl ResolvedConfig { }; (chosen_language, TransformOptions { scan, layout }) } + + /// The same answer as [`Self::for_path`], with a record of where each + /// setting came from. + /// + /// The values are [`Self::for_path`]'s own, so what a run does and what + /// `--explain` says about it cannot disagree; only the attribution is + /// computed here, by replaying the same merge with the layer names + /// attached. `--explain` is the only caller, which is why the hot path is + /// left as it was. + pub fn for_path_traced( + &self, + path: &Path, + language: Language, + dialect: Dialect, + ) -> (Language, TransformOptions, PolicyTrace) { + let (chosen_language, options) = self.for_path(path, language, dialect); + let normalized = self.relative_to_root(path); + let mut layers = Vec::new(); + /* NOTE: `for_path` looks the language table up under the language it was + * handed, not under the one an override may have changed it to. */ + if let Some(config) = self.config.languages.get(language.as_str()) { + layers.push(TracedLayer { + source: Source::Language(language.as_str().to_owned()), + policy: config.policy, + layout: config.layout, + keep_kind: &config.keep_kind, + remove_kind: &config.remove_kind, + keep_regex: &config.keep_regex, + remove_regex: &config.remove_regex, + }); + } + for (index, compiled) in self.overrides.iter().enumerate() { + if !compiled + .matchers + .iter() + .any(|matcher| matcher.is_match(&normalized)) + { + continue; + } + let value = &compiled.value; + layers.push(TracedLayer { + source: Source::Override { + index, + paths: value.paths.clone(), + }, + policy: value.policy, + layout: value.layout, + keep_kind: &value.keep_kind, + remove_kind: &value.remove_kind, + keep_regex: &value.keep_regex, + remove_regex: &value.remove_regex, + }); + } + let cli = self.cli_overrides; + let keep_kinds: Vec<_> = layers + .iter() + .map(|layer| (&layer.source, layer.keep_kind)) + .collect(); + let remove_kinds: Vec<_> = layers + .iter() + .map(|layer| (&layer.source, layer.remove_kind)) + .collect(); + let keep_patterns: Vec<_> = layers + .iter() + .map(|layer| (&layer.source, layer.keep_regex)) + .collect(); + let remove_patterns: Vec<_> = layers + .iter() + .map(|layer| (&layer.source, layer.remove_regex)) + .collect(); + let mut trace = PolicyTrace { + policy: scalar_source(cli.policy, "--policy"), + layout: scalar_source(cli.layout, "--layout"), + keep_kind: attribute( + &self.config.policy.keep_kind, + cli.keep_kind_from, + "--keep-kind", + &keep_kinds, + ), + remove_kind: attribute( + &self.config.policy.remove_kind, + cli.remove_kind_from, + "--remove-kind", + &remove_kinds, + ), + /* NOTE: No flag supplies a pattern, so no entry of either list can have + * come from the command line. */ + keep_regex: attribute(&self.config.policy.keep_regex, None, "", &keep_patterns), + remove_regex: attribute(&self.config.policy.remove_regex, None, "", &remove_patterns), + origins: self.origins.clone(), + }; + /* NOTE: A single-valued setting is not merged but replaced, so the last layer + * that names it is the one that decided it. */ + for layer in &layers { + if layer.policy.is_some() { + trace.policy = layer.source.clone(); + } + if layer.layout.is_some() { + trace.layout = layer.source.clone(); + } + } + (chosen_language, options, trace) + } } pub fn load(explicit: Option<&Path>) -> Result { @@ -244,16 +576,35 @@ pub fn load(explicit: Option<&Path>) -> Result { let user_path = user_config_path().filter(|path| path.is_file()); let mut merged: toml::Value = toml::from_str(&toml::to_string(&Config::default())?)?; let mut trace = ConfigTrace::default(); + let mut origins = PolicyOrigins::new(); if let Some(path) = &user_path { - merge_value(&mut merged, parse_layer(path, false)?); + merge_layer( + &mut merged, + &mut origins, + parse_layer(path, false)?, + path, + &cwd, + ); trace.user = Some(path.clone()); } if let Some(path) = &project_path { - merge_value(&mut merged, parse_layer(path, true)?); + merge_layer( + &mut merged, + &mut origins, + parse_layer(path, true)?, + path, + &cwd, + ); trace.project = Some(path.clone()); } if let Some(path) = explicit { - merge_value(&mut merged, parse_layer(path, true)?); + merge_layer( + &mut merged, + &mut origins, + parse_layer(path, true)?, + path, + &cwd, + ); trace.explicit = Some(path.to_path_buf()); } let mut config: Config = merged @@ -272,15 +623,52 @@ pub fn load(explicit: Option<&Path>) -> Result { config, trace, root, + cwd, + cli_overrides: CliOverrides::default(), overrides, + origins, }) } +/// Layer one configuration file over the merged document, noting every +/// `[policy]` key it sets on the way. +/// +/// A later layer overwrites an earlier one exactly as `merge_value` does, so +/// what is left is the file whose value survived the merge — the one an +/// explanation is worth sending a reader to. +fn merge_layer( + merged: &mut toml::Value, + origins: &mut PolicyOrigins, + layer: toml::Value, + path: &Path, + cwd: &Path, +) { + if let Some(policy) = layer.get("policy") { + for key in POLICY_KEYS { + if policy.get(key).is_some() { + origins.insert(key, origin_label(path, cwd)); + } + } + } + merge_value(merged, layer); +} + +/// How an explanation names a configuration file: relative to the directory +/// the command was run from when it sits there, and absolute otherwise. +/// +/// The label is repeated on every explained line, so the short spelling is +/// worth having — but only where it still names the file the reader would open. +/// A file further up the tree, or the user file under `$HOME`, keeps its +/// absolute path. +fn origin_label(path: &Path, cwd: &Path) -> PathBuf { + path.strip_prefix(cwd).unwrap_or(path).to_path_buf() +} + fn validate_languages(config: &Config) -> Result<()> { for (name, language_config) in &config.languages { - let language: Language = name - .parse() - .map_err(|_| anyhow!("unknown language configuration key `{name}`"))?; + let language: Language = name.parse().map_err(|_| { + anyhow!("unknown language configuration key `{name}`; see `ocomment languages`") + })?; if let Some(dialect) = language_config.dialect { validate_dialect(language, dialect) .with_context(|| format!("invalid dialect for [languages.{name}]"))?; @@ -295,36 +683,47 @@ fn validate_languages(config: &Config) -> Result<()> { Ok(()) } +/// Every dialect the scanner accepts for `language`, in canonical order. +pub fn supported_dialects(language: Language) -> &'static [Dialect] { + match language { + Language::JavaScript => &[Dialect::Standard, Dialect::Jsx], + Language::TypeScript => &[Dialect::Standard, Dialect::Tsx], + Language::C => &[Dialect::Standard, Dialect::ObjectiveC, Dialect::GnuC], + Language::Cpp => &[ + Dialect::Standard, + Dialect::ObjectiveCpp, + Dialect::GnuCpp, + Dialect::Cuda, + ], + Language::Css => &[Dialect::Standard, Dialect::Scss], + Language::Shell => &[ + Dialect::Standard, + Dialect::PosixSh, + Dialect::Bash53, + Dialect::Zsh, + ], + Language::Sql => &[ + Dialect::Standard, + Dialect::PostgreSql, + Dialect::MySql, + Dialect::Sqlite, + Dialect::TSql, + Dialect::Oracle, + ], + _ => &[Dialect::Standard], + } +} + pub fn validate_dialect(language: Language, dialect: Dialect) -> Result<()> { - let compatible = match language { - Language::JavaScript => matches!(dialect, Dialect::Standard | Dialect::Jsx), - Language::TypeScript => matches!(dialect, Dialect::Standard | Dialect::Tsx), - Language::C => matches!( - dialect, - Dialect::Standard | Dialect::GnuC | Dialect::ObjectiveC - ), - Language::Cpp => matches!( - dialect, - Dialect::Standard | Dialect::GnuCpp | Dialect::ObjectiveCpp | Dialect::Cuda - ), - Language::Shell => matches!( - dialect, - Dialect::Standard | Dialect::PosixSh | Dialect::Bash53 | Dialect::Zsh - ), - Language::Sql => matches!( - dialect, - Dialect::Standard - | Dialect::PostgreSql - | Dialect::MySql - | Dialect::Sqlite - | Dialect::TSql - | Dialect::Oracle - ), - _ => dialect == Dialect::Standard, - }; + let supported = supported_dialects(language); ensure!( - compatible, - "dialect `{dialect:?}` is not supported for language `{language}`" + supported.contains(&dialect), + "unsupported dialect `{dialect}` for {language}; supported: {}", + supported + .iter() + .map(|value| value.as_str()) + .collect::>() + .join(", ") ); Ok(()) } @@ -348,8 +747,21 @@ fn validate_policy_regexes(config: &Config) -> Result<()> { .flat_map(|item| item.keep_regex.iter().chain(&item.remove_regex)), ); for pattern in patterns { - regex::bytes::Regex::new(pattern) - .with_context(|| format!("invalid comment policy regex `{pattern}`"))?; + regex::bytes::Regex::new(pattern).map_err(|error| { + /* INVARIANT: Both halves of this line came out of a file in the project: the + * pattern the caller wrote, and a parse error that quotes that + * same pattern back with a caret under it. Neither may reach a + * terminal verbatim, and the line stays one line. The pattern + * keeps the spacing it was written with, because a reader who is + * shown something else cannot find it in the file; the parse + * error, which `regex` spreads over four lines, is folded onto + * this one and kept whole. */ + anyhow!( + "invalid comment policy regex `{}`: {}", + crate::output::sanitize_path(pattern), + crate::output::sanitize_message(&error.to_string()) + ) + })?; } Ok(()) } @@ -359,15 +771,30 @@ fn parse_layer(path: &Path, require_version: bool) -> Result { fs::read_to_string(path).with_context(|| format!("cannot read {}", path.display()))?; let config: Config = toml::from_str(&text).map_err(|error| { let message = error.to_string(); + /* INVARIANT: `toml` quotes the line it stopped on, with a caret under the byte + * that is wrong with it, so the whole of that line — bytes a project + * file chose, an escape sequence among them — is on its way to a + * terminal over four lines of diagram. It is folded onto the one line + * an error is and kept whole, the way an invalid `[policy]` regex is: + * the caret means nothing once the lines are joined, and the sentence + * after it is the entire answer. The hint reads the unfolded message + * because it quotes nothing back — only a key it found in the schema. */ anyhow!( "invalid configuration {}: {}{}", - path.display(), - message, + /* NOTE: The path is the project's too — a directory it named — so it is + * held to what every other path in the report is held to: printed + * as it was spelled, with nothing in it a terminal would act on. */ + crate::output::sanitize_path(&path.display().to_string()), + crate::output::sanitize_message(&message), unknown_key_hint(&message) ) })?; if require_version && config.version != Some(1) { - bail!("{} must contain `version = 1`", path.display()); + /* NOTE: The path is repeated deliberately: the first half is the verdict on + * a file the reader may not have opened, the second is the edit that + * settles it, and an editor is opened on the second one. */ + let path = path.display(); + bail!("{path} must contain `version = 1` (add `version = 1` at the top of {path})"); } toml::from_str(&text).with_context(|| format!("cannot parse {}", path.display())) } @@ -422,6 +849,34 @@ fn compile_overrides(overrides: &[PathOverride]) -> Result .collect() } +/// Resolve `.` and `..` without asking the file system. +/// +/// A configuration glob is matched against text, so the text has to be the one +/// the reader would have written: `../sibling/main.rs`, named from `nested/`, +/// is `sibling/main.rs` under the root, and leaving the `..` in place would +/// let it match a `nested/**` override it is not under. The resolution is +/// lexical because the path need not exist and because `canonicalize` would +/// also resolve the symbolic links the root itself may be reached through, +/// which would leave the two ends of the comparison in different spellings. +pub(crate) fn lexical(path: &Path) -> PathBuf { + let mut resolved = PathBuf::new(); + for component in path.components() { + match component { + Component::CurDir => {} + Component::ParentDir + if matches!( + resolved.components().next_back(), + Some(Component::Normal(_)) + ) => + { + resolved.pop(); + } + component => resolved.push(component), + } + } + resolved +} + pub fn locate_project(start: &Path) -> Option { let mut directory = Some(start); while let Some(current) = directory { @@ -517,6 +972,74 @@ mod tests { assert!(unknown_key_hint(message).contains("layout")); } + /// `for_path_traced` must not become a second copy of the merge that can + /// drift from it: the values it returns are `for_path`'s own, and the trace + /// beside them lines up with those values position for position. + #[test] + fn a_traced_lookup_returns_the_untraced_answer_and_lines_up_with_it() { + let directory = tempfile::tempdir().unwrap(); + let root = directory.path().to_path_buf(); + let mut config = Config::default(); + config.policy.keep_regex = vec!["global".to_owned()]; + config.policy.keep_kind = vec![CommentKind::Line]; + config.languages.insert( + "rust".to_owned(), + LanguageConfig { + keep_regex: vec!["language".to_owned()], + ..LanguageConfig::default() + }, + ); + config.overrides = vec![PathOverride { + paths: vec!["nested/**".to_owned()], + policy: Some(Policy::All), + /* NOTE: The duplicate is dropped by the merge, so the trace must not + * record a source for it either. */ + keep_regex: vec!["override".to_owned(), "global".to_owned()], + ..PathOverride::default() + }]; + let overrides = compile_overrides(&config.overrides).unwrap(); + let resolved = ResolvedConfig { + config, + trace: ConfigTrace::default(), + root: root.clone(), + cwd: root.clone(), + cli_overrides: CliOverrides { + keep_kind_from: Some(0), + ..CliOverrides::default() + }, + overrides, + origins: PolicyOrigins::new(), + }; + let path = root.join("nested/a.rs"); + + let (language, options) = resolved.for_path(&path, Language::Rust, Dialect::Standard); + let (traced_language, traced_options, trace) = + resolved.for_path_traced(&path, Language::Rust, Dialect::Standard); + assert_eq!(traced_language, language); + assert_eq!(traced_options, options); + + let override_source = Source::Override { + index: 0, + paths: vec!["nested/**".to_owned()], + }; + assert_eq!(options.scan.keep_regex, ["global", "language", "override"]); + assert_eq!( + trace.keep_regex, + [ + Source::Global, + Source::Language("rust".to_owned()), + override_source.clone(), + ] + ); + assert_eq!(trace.policy, override_source); + assert_eq!( + trace.keep_kind, + [Source::Cli { + flag: "--keep-kind" + }] + ); + } + #[test] fn repository_root_accepts_directory_and_worktree_markers() { for marker_is_directory in [true, false] { diff --git a/rust/ocomment/src/files.rs b/rust/ocomment/src/files.rs index 8425f09..8fefd11 100644 --- a/rust/ocomment/src/files.rs +++ b/rust/ocomment/src/files.rs @@ -1,10 +1,10 @@ use crate::config::ResolvedConfig; -use anyhow::{Context, Result}; +use anyhow::{Context, Result, anyhow}; use globset::{Glob, GlobSet, GlobSetBuilder}; use ignore::WalkBuilder; use ocomment_core::{DeclarativeProfile, Detection, Dialect, Language, detect_language}; use std::{ - fs, + env, fs, path::{Path, PathBuf}, }; @@ -23,6 +23,10 @@ pub struct SkippedFile { pub path: PathBuf, pub reason: String, pub error: bool, + /// The path itself was named on the command line. Such a skip is always + /// reported on its own line; a skip found while walking a directory is + /// folded into the end-of-run summary instead. + pub explicit: bool, } #[derive(Default)] @@ -31,13 +35,137 @@ pub struct Discovery { pub skipped: Vec, } +/// The path standard input is reported under. It is not a real file name: the +/// renderers print it, and the configuration override matcher sees it, exactly +/// as it reads here. +pub const STDIN_PATH: &str = ""; + +/// What both `strip` and a `-` target say when the bytes carry no signature to +/// detect a language from. Standard input has no name to fall back on, so the +/// only way forward is for the caller to name the language. +pub const STDIN_LANGUAGE_HELP: &str = "cannot detect the language of standard input; \ +pass --language (see `ocomment languages`)"; + +/// Why a file OComment has no scanner for is passed over, and the two ways out +/// of it: consult the list of what is built in, or name a language anyway. +/// +/// The end-of-run summary must not repeat this sentence once per file, so it +/// folds the reason onto a short key of its own; `output::skip_label` is what +/// ties the two together. +pub const NO_LANGUAGE: &str = + "no built-in language for this file (see `ocomment languages`; use --language to force)"; + +/// Why a named path was not found. A relative path is resolved against the +/// working directory, which is exactly what a caller who typed it from the +/// wrong place cannot see, so the directory that was searched is named. +fn missing_path_reason() -> String { + env::current_dir().map_or_else( + |_| "path does not exist".to_owned(), + |cwd| { + format!( + "path does not exist (checked relative to {})", + cwd.display() + ) + }, + ) +} + +/// Turn the bytes read from standard input into a source file the ordinary +/// pipeline can process, or the skip that says why it cannot. Detection has no +/// path to work with, so it is driven by `--language` or by the contents. +/// +/// Declarative profiles and plugins route on a file extension, which standard +/// input does not have; a pipe is therefore always handled by a built-in +/// language or not at all. +pub fn stdin_source( + bytes: Vec, + resolved: &ResolvedConfig, + forced_language: Option, + forced_dialect: Option, +) -> Result { + let skipped = |reason: &str, error: bool| SkippedFile { + path: PathBuf::from(STDIN_PATH), + reason: reason.to_owned(), + error, + /* NOTE: Standard input was named on the command line, so its skip is always + * reported on its own line rather than folded into the summary. */ + explicit: true, + }; + if bytes.iter().take(8192).any(|byte| *byte == 0) { + return Err(skipped("binary file (NUL byte)", false)); + } + let detection = forced_language + .map(|language| Detection { + language, + dialect: forced_dialect.unwrap_or(Dialect::Standard), + reason: "command-line", + }) + .or_else(|| detect_language(None, &bytes)); + let Some(Detection { + language, dialect, .. + }) = detection + else { + return Err(skipped(STDIN_LANGUAGE_HELP, true)); + }; + if forced_language.is_none() + && resolved + .config + .languages + .get(language.as_str()) + .and_then(|item| item.enabled) + == Some(false) + { + return Err(skipped("language disabled by configuration", false)); + } + Ok(SourceFile { + path: PathBuf::from(STDIN_PATH), + source: bytes, + language, + dialect: forced_dialect.unwrap_or(dialect), + profile: None, + plugin: None, + }) +} + +/// What a command with no PATH walks. +/// +/// The project root is where the configuration was found, not what the caller +/// is looking at: a command run from a subdirectory checks that subdirectory, +/// the way every other file-walking developer tool does. Reaching back up to +/// the root would put files the caller cannot see — and, with `fix`, files +/// they did not mean to rewrite — into the run. +pub const DEFAULT_TARGET: &str = "."; + +/// The one name a walk never offers, whatever else was asked for. +/// +/// `.git` is git's own storage rather than source, and `git` itself never +/// treats it as a candidate for anything. Neither may a tool that rewrites +/// files in place: `ocomment fix .` in a fresh repository would otherwise +/// rewrite every sample hook git had just written into `.git/hooks`. Naming +/// a directory lifts the hidden-file rule and so does `files.hidden`, so the +/// exclusion cannot hang off either of them. +/// +/// A submodule or a linked worktree keeps its `.git` as a *file* pointing at +/// the storage instead of holding it, which is why the name is matched rather +/// than the file type. +const GIT_DIRECTORY: &str = ".git"; + pub fn discover( paths: &[PathBuf], resolved: &ResolvedConfig, forced_language: Option, forced_dialect: Option, ) -> Result { - discover_with_scope(paths, resolved, forced_language, forced_dialect, true) + let implicit = [PathBuf::from(DEFAULT_TARGET)]; + /* NOTE: The substituted target stands in for an argument nobody typed, so it is + * walked with the ordinary limits: only a path the caller actually named + * is a request to look past the hidden-file and size rules. */ + let (paths, explicit) = if paths.is_empty() { + (&implicit[..], false) + } else { + (paths, true) + }; + discover_with_scope(paths, resolved, forced_language, forced_dialect, explicit) } /// Discover workspace roots with normal traversal limits. Unlike explicit CLI @@ -56,6 +184,8 @@ fn discover_with_scope( let include = compile_globs(&resolved.config.files.include)?; let exclude = compile_globs(&resolved.config.files.exclude)?; let mut discovery = Discovery::default(); + /* NOTE: Only an editor asking for its workspace arrives here without a target; + * `discover` gives a command line the current directory instead. */ let targets: Vec<_> = if paths.is_empty() { vec![(resolved.root.clone(), false)] } else { @@ -74,6 +204,7 @@ fn discover_with_scope( load_one( &path, explicit_scope, + explicit_scope, resolved, forced_language, forced_dialect, @@ -87,8 +218,8 @@ fn discover_with_scope( builder .follow_links(resolved.config.files.follow_symlinks) .standard_filters(ignore) - // `standard_filters` also resets the hidden-file flag, so this - // must come afterwards for explicitly named directories. + /* NOTE: `standard_filters` also resets the hidden-file flag, so this + * must come afterwards for explicitly named directories. */ .hidden(!explicit_scope && !resolved.config.files.hidden) .git_ignore(ignore) .git_global(ignore) @@ -98,12 +229,17 @@ fn discover_with_scope( if ignore { builder.add_custom_ignore_filename(".ocommentignore"); } + /* NOTE: The filter is never asked about the walk root, so a caller who + * names a path inside `.git` — or `.git` itself — is still + * answered; only what a walk *wanders* into is excluded. */ + builder.filter_entry(|entry| entry.file_name() != GIT_DIRECTORY); for entry in builder.build() { match entry { Ok(entry) if entry.file_type().is_some_and(|kind| kind.is_file()) => { load_one( entry.path(), explicit_scope, + false, resolved, forced_language, forced_dialect, @@ -117,14 +253,16 @@ fn discover_with_scope( path: path.clone(), reason: error.to_string(), error: true, + explicit: explicit_scope, }), } } } else { discovery.skipped.push(SkippedFile { path, - reason: "path does not exist".into(), + reason: missing_path_reason(), error: true, + explicit: explicit_scope, }); } } @@ -134,16 +272,45 @@ fn discover_with_scope( discovery .files .dedup_by(|left, right| left.path == right.path); + /* INVARIANT: A path is reached twice whenever it is named beside a directory holding + * it, and it is one file either way: `files` says so with the sort and the + * dedup above, and a skip is one file just as much — a report that + * annotates the same path twice reads as two problems with it. Which of + * the two entries survives is not arbitrary. An error decides the exit + * code, and a path the caller actually typed is answered on a line of its + * own rather than folded into the summary, so the entry that says the most + * is sorted to the front of its path and is the one the dedup keeps. */ + discovery.skipped.sort_by(|left, right| { + left.path + .cmp(&right.path) + .then(right.error.cmp(&left.error)) + .then(right.explicit.cmp(&left.explicit)) + }); discovery .skipped - .sort_by(|left, right| left.path.cmp(&right.path)); + .dedup_by(|left, right| left.path == right.path); Ok(discovery) } +/// The name a walked file is reported under. +/// +/// The implicit target is `.`, so a walk rooted there hands back every entry +/// as `./name`. `ocomment` and `ocomment check name` report one file, and a +/// reader — or a `git apply` reading the patch — is owed one spelling of it, +/// so the prefix the walk root contributed is dropped. The target itself is +/// left alone: `.` names a directory, and `` names nothing. +fn reported_path(path: &Path) -> PathBuf { + match path.strip_prefix(DEFAULT_TARGET) { + Ok(stripped) if !stripped.as_os_str().is_empty() => stripped.to_path_buf(), + _ => path.to_path_buf(), + } +} + #[allow(clippy::too_many_arguments)] fn load_one( path: &Path, explicit_scope: bool, + explicit_path: bool, resolved: &ResolvedConfig, forced_language: Option, forced_dialect: Option, @@ -151,14 +318,18 @@ fn load_one( exclude: &GlobSet, discovery: &mut Discovery, ) { - let relative = path.strip_prefix(&resolved.root).unwrap_or(path); - if (!include.is_empty() && !include.is_match(relative)) || exclude.is_match(relative) { + let path = &reported_path(path); + /* NOTE: The globs are written relative to the root; the path was typed — or + * walked — relative to the working directory, so it is measured against + * the root before either set is asked about it. */ + let relative = resolved.relative_to_root(path); + if (!include.is_empty() && !include.is_match(&relative)) || exclude.is_match(&relative) { return; } let link_metadata = match path.symlink_metadata() { Ok(value) => value, Err(error) => { - discovery.skipped.push(skip(path, error)); + discovery.skipped.push(skip(path, explicit_path, error)); return; } }; @@ -168,6 +339,7 @@ fn load_one( path: path.to_path_buf(), reason: "symbolic link".into(), error: false, + explicit: explicit_path, }); return; } @@ -175,26 +347,28 @@ fn load_one( Ok(metadata) if metadata.is_file() => metadata, Ok(_) => return, Err(error) => { - discovery.skipped.push(skip(path, error)); + discovery.skipped.push(skip(path, explicit_path, error)); return; } } } else { link_metadata }; - // Every path under an explicitly named directory is explicit for hidden and size handling. + /* NOTE: Every path under an explicitly named directory is explicit for hidden and + * size handling. */ if !explicit_scope && metadata.len() > resolved.config.files.max_size { discovery.skipped.push(SkippedFile { path: path.to_path_buf(), reason: format!("larger than {} bytes", resolved.config.files.max_size), error: false, + explicit: explicit_path, }); return; } let source = match fs::read(path) { Ok(value) => value, Err(error) => { - discovery.skipped.push(skip(path, error)); + discovery.skipped.push(skip(path, explicit_path, error)); return; } }; @@ -203,6 +377,7 @@ fn load_one( path: path.to_path_buf(), reason: "binary file (NUL byte)".into(), error: false, + explicit: explicit_path, }); return; } @@ -227,6 +402,7 @@ fn load_one( path: path.to_path_buf(), reason: "language disabled by configuration".into(), error: false, + explicit: explicit_path, }); return; } @@ -250,8 +426,9 @@ fn load_one( if language == Language::Unknown && profile.is_none() && plugin.is_none() { discovery.skipped.push(SkippedFile { path: path.to_path_buf(), - reason: "unknown language".into(), + reason: NO_LANGUAGE.into(), error: false, + explicit: explicit_path, }); return; } @@ -291,18 +468,39 @@ pub fn profile_for_path(path: &Path, resolved: &ResolvedConfig) -> Option Result { +/// Compile one of the `[files]` glob lists. +/// +/// A walk asks for these in `discover_with_scope` and a staged run asks for +/// them in `git::run_staged`; both measure a path against the project root +/// first, so both get the same answer for the same path. +pub(crate) fn compile_globs(patterns: &[String]) -> Result { let mut builder = GlobSetBuilder::new(); for pattern in patterns { - builder.add(Glob::new(pattern).with_context(|| format!("invalid file glob `{pattern}`"))?); + let glob = Glob::new(pattern).map_err(|error| { + /* INVARIANT: Both halves of this line came out of a file in the project: the + * pattern the caller wrote, and a `globset` parse error that + * quotes that same pattern straight back. Neither may reach a + * terminal verbatim, and the line stays one line. The pattern + * keeps the spacing it was written with, because a reader who is + * shown something else cannot find it in the file. It is the same + * treatment `config::validate_regexes` gives the other pattern a + * project file carries. */ + anyhow!( + "invalid file glob `{}`: {}", + crate::output::sanitize_path(pattern), + crate::output::sanitize_message(&error.to_string()) + ) + })?; + builder.add(glob); } builder.build().context("cannot compile file globs") } -fn skip(path: &Path, error: impl std::fmt::Display) -> SkippedFile { +fn skip(path: &Path, explicit: bool, error: impl std::fmt::Display) -> SkippedFile { SkippedFile { path: path.to_path_buf(), reason: error.to_string(), error: true, + explicit, } } diff --git a/rust/ocomment/src/git.rs b/rust/ocomment/src/git.rs index ec980e2..a4772cf 100644 --- a/rust/ocomment/src/git.rs +++ b/rust/ocomment/src/git.rs @@ -1,7 +1,10 @@ use crate::{ atomic::{WritePlan, apply_transaction}, config::ResolvedConfig, - output::{self, Operation, OutputFormat, Presentation, ProcessedFile}, + files::SkippedFile, + output::{ + self, Operation, OutputFormat, Presentation, ProcessedFile, RenderOptions, Verbosity, + }, plugin::PluginHost, }; use anyhow::{Context, Result, anyhow, bail}; @@ -10,10 +13,11 @@ use ocomment_core::{ detect_language, transform, }; use std::{ + collections::BTreeSet, ffi::{OsStr, OsString}, fs, io::Write, - path::{Path, PathBuf}, + path::{Component, Path, PathBuf}, process::{Command, Stdio}, }; use tempfile::NamedTempFile; @@ -35,6 +39,11 @@ pub struct StagedRequest<'a> { pub forced_language: Option, pub forced_dialect: Option, pub presentation: Presentation, + pub verbosity: Verbosity, + pub preview: bool, + /// The run only previews the patch; `fix --dry-run` writes nothing to + /// the index and reports what a real run would remove. + pub dry_run: bool, } pub fn run_staged(request: StagedRequest<'_>) -> Result { @@ -48,13 +57,25 @@ pub fn run_staged(request: StagedRequest<'_>) -> Result { forced_language, forced_dialect, presentation, + verbosity, + preview, + dry_run, } = request; let root = repository_root()?; - let names = staged_paths(&root, paths)?; + let (blobs, mut skipped) = configured_paths(&root, staged_paths(&root, paths)?, resolved)?; let mut entries = Vec::new(); - for path in names { + for StagedBlob { path, named } in blobs { let source = index_blob(&root, &path)?; if source.iter().take(8192).any(|byte| *byte == 0) { + /* NOTE: A walk says why it passed a file over, and so does this: a hook + * that stages a PNG beside its source has to read as one file + * scanned and one passed over, not as two files with nothing + * to say about them. */ + skipped.push(skipped_blob( + path, + "binary file (NUL byte)".to_owned(), + named, + )); continue; } let detection = forced_language @@ -72,6 +93,11 @@ pub fn run_staged(request: StagedRequest<'_>) -> Result { .then(|| crate::files::plugin_for_path(&path, resolved)) .flatten(); if detection.is_none() && profile.is_none() && routed_plugin.is_none() { + skipped.push(skipped_blob( + path, + crate::files::NO_LANGUAGE.to_owned(), + named, + )); continue; } let language = detection @@ -161,6 +187,10 @@ pub fn run_staged(request: StagedRequest<'_>) -> Result { }); } entries.sort_by(|left, right| left.path.cmp(&right.path)); + /* NOTE: The size skips were found before the blobs were read and the rest while + * reading them, so the two arrive interleaved by nothing at all; a + * machine format publishes this list, which owes its reader one order. */ + skipped.sort_by(|left, right| left.path.cmp(&right.path)); let files: Vec<_> = entries .iter() .map(|entry| entry.processed.clone()) @@ -173,12 +203,27 @@ pub fn run_staged(request: StagedRequest<'_>) -> Result { .iter() .any(|diagnostic| diagnostic.code == "staged-existing-block-comment") }); - if operation == Operation::Fix - && (!invalid || (resolved.config.policy.force_invalid && !staged_conflict)) - { + let applied = operation == Operation::Fix + && (!invalid || (resolved.config.policy.force_invalid && !staged_conflict)); + if applied { fix_index(&root, &entries, index_only)?; } - output::render(&files, &[], format, operation, presentation)?; + output::render( + &files, + &skipped, + &RenderOptions { + format, + operation, + presentation, + verbosity, + preview, + explain: false, + dry_run, + force_invalid: resolved.config.policy.force_invalid, + applied, + policy: resolved.config.policy.mode, + }, + )?; if invalid { return Ok(2); } @@ -205,14 +250,17 @@ fn fix_index(root: &Path, entries: &[IndexEntry], index_only: bool) -> Result<() let original_index = fs::read(&index_path) .with_context(|| format!("cannot read Git index {}", index_path.display()))?; if index_path.with_file_name("index.lock").exists() { - bail!("Git index is locked; no files were modified"); + bail!( + "Git index is locked; no files were modified; another Git process may be \ + running, or remove a stale .git/index.lock" + ); } let mut temporary_index = NamedTempFile::new_in(index_path.parent().unwrap_or(root))?; temporary_index.write_all(&original_index)?; temporary_index.flush()?; temporary_index.as_file_mut().sync_all()?; - // Close the file before Git replaces it through `.lock`; retaining an - // open NamedTempFile handle makes this update fail on Windows. + /* NOTE: Close the file before Git replaces it through `.lock`; retaining an + * open NamedTempFile handle makes this update fail on Windows. */ let temporary_path = temporary_index.into_temp_path(); for entry in &changed { @@ -254,8 +302,8 @@ fn fix_index(root: &Path, entries: &[IndexEntry], index_only: bool) -> Result<() }); } } - // Treat the index itself as the last journaled file. The shared transaction - // rolls working-tree files and index back together on any rename failure. + /* INVARIANT: Treat the index itself as the last journaled file. The shared transaction + * rolls working-tree files and index back together on any rename failure. */ plans.push(WritePlan { path: index_path, original: original_index, @@ -300,15 +348,71 @@ fn map_edits_uniquely(index: &[u8], working: &[u8], edits: &[Edit]) -> Result Result { let mut output = command_output( Command::new("git").args(["rev-parse", "--show-toplevel"]), - "not inside a Git repository", + /* NOTE: Git's own words follow: they name the directory it searched from, + * which is the difference between "wrong directory" and "no repository". */ + "--staged needs a Git repository", )?; trim_line_ending(&mut output); Ok(bytes_to_path(&output)) } -fn staged_paths(root: &Path, filters: &[PathBuf]) -> Result> { +/// Every staged path a run has to consider, and which of them the caller +/// named. +struct StagedPaths { + /// What `git diff --cached` answered: root-relative, sorted, each path + /// once however many pathspecs covered it. + paths: Vec, + /// The paths a pathspec picked out, which is what lifts the project's own + /// limits from them. + named: BTreeSet, +} + +/// A staged path `[files]` let through, and whether the caller named it. +struct StagedBlob { + path: PathBuf, + named: bool, +} + +/// Ask `git` what is staged, one question for each pathspec. +/// +/// A pathspec is `git`'s to interpret and nobody else's. `.hidden/*.rs` is a +/// wildcard it expands, an absolute path is one it makes root-relative, `.` is +/// a directory it resolves against the directory the command was typed in, and +/// the answer to all of them is a path relative to the repository root. So +/// each pathspec is put to `git` on its own and the answers are unioned, which +/// leaves the run with both of the things it needs — the paths to scan, and +/// the paths a caller asked about — without restating a word of pathspec +/// syntax here, and so without the two readings drifting apart. +/// +/// It costs one `git` invocation for each pathspec the caller typed, where the +/// single combined question it replaces cost one for all of them. A hook that +/// passes its staged file names in one by one pays that per name, next to the +/// four this run already spends on every path it keeps, and it buys the only +/// reading of a pathspec that `git` itself would agree with. +fn staged_paths(root: &Path, filters: &[PathBuf]) -> Result { + let base = pathspec_base(root); + let mut paths = Vec::new(); + let mut named = BTreeSet::new(); + if filters.is_empty() { + paths = list_staged(&base, None)?; + } else { + for pathspec in filters { + let listed = list_staged(&base, Some(pathspec))?; + if !names_whole_tree(pathspec, &base, root) { + named.extend(listed.iter().cloned()); + } + paths.extend(listed); + } + } + paths.sort(); + paths.dedup(); + Ok(StagedPaths { paths, named }) +} + +/// The staged paths one pathspec covers, or all of them when there is none. +fn list_staged(base: &Path, pathspec: Option<&Path>) -> Result> { let mut command = Command::new("git"); - command.current_dir(root).args([ + command.current_dir(base).args([ "diff", "--cached", "--name-only", @@ -316,27 +420,197 @@ fn staged_paths(root: &Path, filters: &[PathBuf]) -> Result> { "--diff-filter=ACMR", "--", ]); - command.args(filters); + if let Some(pathspec) = pathspec { + command.arg(pathspec); + } let output = command_output(&mut command, "cannot list staged paths")?; - let mut paths: Vec<_> = output + Ok(output .split(|byte| *byte == 0) .filter(|bytes| !bytes.is_empty()) .map(bytes_to_path) - .collect(); - paths.sort(); - paths.dedup(); - Ok(paths) + .collect()) } -fn index_blob(root: &Path, path: &Path) -> Result> { +/// The directory a relative pathspec is measured from. +/// +/// It is the directory the command was typed in, which is what `git` resolves +/// `.` or `../lib` against. `resolved.cwd` cannot answer this question: a +/// staged run points it at the repository root before it begins, because a +/// staged path arrives root-relative and a `[files]` glob is written +/// root-relative too. The root stands in where the working directory cannot be +/// read at all: it is inside the repository by construction, so the worst it +/// can do is read a relative pathspec as the top of the tree would. +fn pathspec_base(root: &Path) -> PathBuf { + std::env::current_dir().unwrap_or_else(|_| root.to_path_buf()) +} + +/// Whether a pathspec covers the repository, and so picks nothing out of it. +/// +/// `ocomment check --staged .` from the top of the repository asks for the run +/// that `ocomment check --staged` already is, and it has to get that run's +/// answer — every `[files]` limit included. Naming a path is what lifts those +/// limits, and the whole tree is not a path anybody picked out: a hook that +/// spells its run with a trailing `.` would otherwise put exactly the hidden +/// or oversized blob through a commit that a bare run passes over. The same +/// `.` typed in `src/` does pick a subtree out, which is why the pathspec is +/// resolved where it was written before it is compared. +/// +/// Only a pathspec that is a path is understood here. `git`'s `:(magic)` +/// spellings are read as picking something out, which is the reading that +/// answers about the paths the caller wrote rather than silently dropping +/// them. +fn names_whole_tree(pathspec: &Path, base: &Path, root: &Path) -> bool { + let joined = base.join(pathspec); + let absolute = std::path::absolute(&joined).unwrap_or(joined); + crate::config::lexical(&absolute) == crate::config::lexical(root) +} + +/// Drop the staged paths `[files]` puts out of bounds, and say which of them +/// were passed over. +/// +/// `git diff --cached` answers with every path the commit carries, which is a +/// different question from the one `[files]` answers: a vendored tree the +/// project excludes is still staged on the commit that updates it. A walk +/// applies `include` and `exclude` in `files::load_one`, so a staged run +/// applies them here, and to the same root-relative spelling — `git` names a +/// staged path relative to the repository root, and `run_target` has already +/// pointed `resolved.cwd` there for exactly this reason. +/// +/// A staged path nobody named is a walked path: it never carries the licence +/// an explicit argument does to look past the project's own limits. That is the +/// whole of `[files]` and not just its two glob lists — `hidden` decides +/// whether a dot-directory is looked into at all and `max_size` decides how +/// much of a file is worth reading, and a hook that applied neither would put +/// through a commit exactly what a walk would never have reached. +/// +/// A path the caller *did* name is the other case, and +/// [`StagedPaths::named`] is what tells the two apart. +/// `ocomment check --staged .hidden/x.rs` is a request about that file, so +/// answering "0 files" because the project does not walk into dot-directories +/// reads as a clean file rather than as a path out of bounds — which is why a +/// walk lifts both limits for an explicit argument, and why this lifts them +/// for the same argument spelled as a pathspec. +/// +/// The two limits answer differently when they do apply, because they mean +/// differently. A hidden path was never a candidate, so it leaves no trace; an +/// oversized blob is a file the run *met* and declined, so it comes back as the +/// same folded "too large" skip a walk reports, counted in the summary rather +/// than annotated once per file. +fn configured_paths( + root: &Path, + staged: StagedPaths, + resolved: &ResolvedConfig, +) -> Result<(Vec, Vec)> { + let include = crate::files::compile_globs(&resolved.config.files.include)?; + let exclude = crate::files::compile_globs(&resolved.config.files.exclude)?; + let max_size = resolved.config.files.max_size; + let StagedPaths { paths, named } = staged; + let mut kept = Vec::new(); + let mut skipped = Vec::new(); + for path in paths { + let relative = resolved.relative_to_root(&path); + /* NOTE: The glob lists bound a named path too — a walk asks them about every + * candidate before it asks anything else, and `load_one` asks them of + * an explicit argument exactly as it asks them of a walked one. */ + if (!include.is_empty() && !include.is_match(&relative)) || exclude.is_match(&relative) { + continue; + } + let explicit = named.contains(&path); + if !explicit && !resolved.config.files.hidden && has_hidden_component(&path) { + continue; + } + if !explicit && index_blob_size(root, &path)? > max_size { + skipped.push(SkippedFile { + path, + reason: format!("larger than {max_size} bytes"), + error: false, + /* NOTE: Nobody typed this path, so its skip is folded into the summary + * exactly as a walked one is. */ + explicit: false, + }); + continue; + } + kept.push(StagedBlob { + path, + named: explicit, + }); + } + Ok((kept, skipped)) +} + +/// A staged blob that was met and declined. +/// +/// Neither reason depends on what the caller typed — a PNG is not text and a +/// `.md` file has no scanner however it got into the commit — but who typed +/// the path decides where the skip is reported. One nobody named is counted in +/// the end-of-run summary under the short label [`crate::output::skip_label`] +/// gives it, and listed per file only when `-v` asks for the list; one the +/// caller named is answered on a line of its own, because +/// `ocomment check --staged notes.md` that says only "nothing to check" reads +/// as a clean file rather than as a file nothing could read. +fn skipped_blob(path: PathBuf, reason: String, named: bool) -> SkippedFile { + SkippedFile { + path, + reason, + error: false, + explicit: named, + } +} + +/// Whether any component of a staged path is a hidden name. +/// +/// `git` names a staged path relative to the repository root, so every +/// component of it is a real directory or file name — there is no walk root in +/// front to leave out, the way `ignore` leaves one out. A leading `.` is the +/// only byte that decides it, so a name that is not UTF-8 is judged on the +/// bytes it actually has rather than on a lossy reading of them. +fn has_hidden_component(path: &Path) -> bool { + path.components().any(|component| { + matches!(component, Component::Normal(name) if name.as_encoded_bytes().starts_with(b".")) + }) +} + +/// How `git` is asked for the staged version of a path: `:` names the index. +fn index_specification(path: &Path) -> OsString { let mut specification = OsString::from(":"); specification.push(path_for_git(path)); + specification +} + +/// How large the staged blob is, without reading it. +/// +/// The size is asked of the index rather than of the working tree, because +/// `--staged` judges the bytes the commit will carry: a file can be a line +/// long on disk and a megabyte in the index, or the other way round. +/// +/// Asking costs one `git` invocation for each path that got past the globs, +/// next to the three the run already spends on every path it keeps. It buys +/// the thing `max_size` exists for, which is that an oversized blob is never +/// brought into memory at all — measuring it from `index_blob`'s answer would +/// have read it first. +fn index_blob_size(root: &Path, path: &Path) -> Result { + let mut output = command_output( + Command::new("git") + .current_dir(root) + .arg("cat-file") + .arg("-s") + .arg(index_specification(path)), + &format!("cannot measure staged blob {}", path.display()), + )?; + trim_line_ending(&mut output); + std::str::from_utf8(&output) + .ok() + .and_then(|text| text.trim().parse().ok()) + .with_context(|| format!("staged blob {} has no size", path.display())) +} + +fn index_blob(root: &Path, path: &Path) -> Result> { command_output( Command::new("git") .current_dir(root) .arg("cat-file") .arg("blob") - .arg(specification), + .arg(index_specification(path)), &format!("cannot read staged blob {}", path.display()), ) } @@ -404,7 +678,12 @@ fn hash_object(root: &Path, bytes: &[u8]) -> Result { .stdout(Stdio::piped()) .stderr(Stdio::piped()) .spawn()?; - child.stdin.take().expect("piped stdin").write_all(bytes)?; + child + .stdin + .take() + .expect("piped stdin") + .write_all(bytes) + .context("cannot write the rewritten blob to git hash-object")?; let output = child.wait_with_output()?; if !output.status.success() { bail!( diff --git a/rust/ocomment/src/interactive.rs b/rust/ocomment/src/interactive.rs new file mode 100644 index 0000000..a9f3d66 --- /dev/null +++ b/rust/ocomment/src/interactive.rs @@ -0,0 +1,758 @@ +//! The comment-by-comment prompt behind `fix --interactive`. +//! +//! The run has already transformed every file by the time this module is +//! reached, so what it asks about is a list of edits that were computed +//! together. Applying only some of them is safe because a replacement is +//! computed from the *source* alone: under `layout = "columns"` it is exactly +//! as wide as the comment it stands for, so a removal moves nothing that comes +//! after it, and under every other layout it depends only on the bytes either +//! side of its own span. `partial_column_edits_keep_the_replacement_the_transform_computed` +//! pins that. +//! +//! The prompt is line-based on purpose: no raw mode, no cursor addressing, no +//! terminal library. One question, one line of answer, and a transcript that a +//! test can read. + +use crate::{ + atomic::WritePlan, + output::{ + Presentation, ProcessedFile, color, line_column, sanitize_path, sanitize_source_line, wrote, + }, +}; +use anyhow::Result; +use ocomment_core::{Comment, Edit, apply_edits}; +use std::io::{BufRead, Write}; + +/// What the reader decided about the removable comments of one run. +/// +/// Deliberately not `Debug`: a plan carries the whole before-and-after text of +/// a source file, and the one thing this type must never do is put it on a +/// terminal by accident. +#[derive(Default)] +pub struct Selection { + /// One plan per file that keeps at least one accepted removal. + pub plans: Vec, + pub accepted: usize, + pub declined: usize, + /// The reader asked for the run to write nothing at all. + pub aborted: bool, +} + +/// The question, ending in a space rather than a newline so the answer is typed +/// on the same line. +const PROMPT: &str = "Remove? [y,n,a,d,q,x,?] "; + +/// What each answer does, in the order the prompt lists them. +const HELP: [&str; 7] = [ + "y - remove this comment", + "n - keep it", + "a - remove it and every remaining comment in this file", + "d - keep it and every remaining comment in this file", + "q - stop asking and apply the removals accepted so far", + "x - abort; write nothing", + "? - show this help", +]; + +/// What is said to an answer that is not one of them. A typo is never taken for +/// a decision about somebody's source file. +const UNKNOWN: &str = "unknown answer; press ? for help"; + +/// How many unchanged lines are shown either side of the change. +const CONTEXT: usize = 3; + +/// Ask about every comment this run would remove and collect the answers into +/// the writes they come to. +/// +/// `input` and `output` are the reader's terminal; they are parameters so the +/// whole conversation can be driven from a script in a test. +pub fn select( + files: &[ProcessedFile], + input: &mut dyn BufRead, + output: &mut dyn Write, + presentation: &Presentation, +) -> Result { + let total: usize = files.iter().map(|file| file.result.edits.len()).sum(); + let mut selection = Selection::default(); + let mut position = 0usize; + let mut stopped = false; + for file in files { + let items = offers(file); + if items.is_empty() { + continue; + } + let mut accepted: Vec = Vec::new(); + // NOTE: The answer `a` or `d` left standing for the rest of this file. + let mut standing: Option = None; + for (index, item) in items.iter().enumerate() { + let (comment, edit) = *item; + position += 1; + let remove = match standing { + Some(answer) => answer, + None => { + let place = Place { + index: index + 1, + of: items.len(), + position, + total, + }; + show(output, file, comment, edit, place, presentation)?; + match ask(input, output, presentation)? { + Answer::Yes => true, + Answer::No => false, + Answer::AllInFile => { + standing = Some(true); + true + } + Answer::NoneInFile => { + standing = Some(false); + false + } + Answer::Stop => { + stopped = true; + break; + } + /* NOTE: Everything accepted so far goes with it: `x` is the + * answer for a run that should never have started. */ + Answer::Abort => { + return Ok(Selection { + aborted: true, + ..Selection::default() + }); + } + Answer::Help => unreachable!("`ask` answers `?` itself"), + } + } + }; + if remove { + selection.accepted += 1; + accepted.push(edit.clone()); + } else { + selection.declined += 1; + } + } + if !accepted.is_empty() { + let replacement = apply_edits(&file.source, &accepted); + if replacement != file.source { + selection.plans.push(WritePlan { + path: file.path.clone(), + original: file.source.clone(), + replacement, + }); + } + } + if stopped { + break; + } + } + Ok(selection) +} + +/// The comments this run would remove, each with the edit that removes it. +/// +/// `transform` pushes exactly one edit per removable comment, in source order, +/// so the two lists line up pairwise. A file whose source failed to scan has no +/// edits at all, and nothing about it is offered — the same gate a +/// non-interactive `fix` applies before it writes. +fn offers(file: &ProcessedFile) -> Vec<(&Comment, &Edit)> { + file.result + .report + .comments + .iter() + .filter(|comment| comment.disposition.is_remove()) + .zip(file.result.edits.iter()) + .collect() +} + +/// Where one question sits, in its file and in the run. +struct Place { + index: usize, + of: usize, + position: usize, + total: usize, +} + +/// Write the question's heading and the hunk it is about. +fn show( + output: &mut dyn Write, + file: &ProcessedFile, + comment: &Comment, + edit: &Edit, + place: Place, + presentation: &Presentation, +) -> Result<()> { + let (line, column) = line_column(&file.source, comment.span.start); + wrote(writeln!( + output, + "{}:{line}:{column} {} comment ({} of {} in file, {} of {} total)", + sanitize_path(&file.path.display().to_string()), + comment.kind, + place.index, + place.of, + place.position, + place.total + ))?; + for row in hunk(&file.source, edit, presentation) { + wrote(writeln!(output, "{row}"))?; + } + Ok(()) +} + +/// The lines the reader is answering for: the ones the comment sits on as they +/// are, the same ones as this single edit would leave them, and `CONTEXT` lines +/// of unchanged source either side. +/// +/// The "after" text is produced by applying this one edit and nothing else, so +/// what is shown is what answering `y` to this question alone would do. +fn hunk(source: &[u8], edit: &Edit, presentation: &Presentation) -> Vec { + let length = source.len(); + let begin = edit.span.start.min(length); + let finish = edit.span.end.clamp(begin, length); + let start = line_start(source, begin); + /* NOTE: The last byte the span covers, so a span that ends exactly on a line + * break does not drag the following line into the hunk. */ + let inner = if finish > begin { finish - 1 } else { begin }; + let end = line_end(source, inner); + let after = apply_edits(source, std::slice::from_ref(edit)); + let shift = edit.replacement.len() as isize - (finish - begin) as isize; + let moved = (end as isize + shift).clamp(start as isize, after.len() as isize); + #[expect( + clippy::cast_sign_loss, + reason = "clamped to `start..=after.len()`, both of which are lengths" + )] + let moved = moved as usize; + + let mut rows = Vec::new(); + for line in preceding(source, start, CONTEXT) { + rows.push(rendered(' ', line, presentation)); + } + changed(&mut rows, '-', &rows_of(&source[start..end]), presentation); + changed( + &mut rows, + '+', + &collapse_blanks(rows_of(&after[start..moved])), + presentation, + ); + for line in following(source, end, CONTEXT) { + rows.push(rendered(' ', line, presentation)); + } + rows +} + +/// How many lines of one changed side are shown before the rest are folded +/// into a single marker: `CONTEXT` at each end, the same window the unchanged +/// context gets. +const BLOCK: usize = 2 * CONTEXT; + +/// One side of the change, capped so a comment taller than the screen cannot +/// push the question off it. +/// +/// A block comment can run to any length, and the reader is answering about +/// the comment, not reading it here: the first and last `CONTEXT` lines say +/// which comment it is and where it ends, and the marker between them says how +/// much was left out rather than pretending there was nothing. +fn changed(rows: &mut Vec, marker: char, lines: &[&[u8]], presentation: &Presentation) { + let show = |rows: &mut Vec, block: &[&[u8]]| { + rows.extend( + block + .iter() + .map(|line| rendered(marker, line, presentation)), + ); + }; + if lines.len() <= BLOCK { + show(rows, lines); + return; + } + show(rows, &lines[..CONTEXT]); + rows.push(elision(marker, lines.len() - BLOCK, presentation)); + show(rows, &lines[lines.len() - CONTEXT..]); +} + +/// What stands in for the lines a capped side folded away. It carries the +/// marker of the side it belongs to so the two columns stay aligned, and is +/// dimmed rather than tinted so it is never read as a line of the source. +fn elision(marker: char, hidden: usize, presentation: &Presentation) -> String { + format!( + "{}{marker}... {hidden} more line{} ...{}", + color("\x1b[2m", presentation.color), + if hidden == 1 { "" } else { "s" }, + color("\x1b[0m", presentation.color) + ) +} + +/// Runs of the same blank line folded to one. +/// +/// Under `layout = "lines"` a removed block comment leaves exactly as many +/// empty lines as it occupied, and the twenty-seventh of them tells the reader +/// nothing the first did not. +fn collapse_blanks(lines: Vec<&[u8]>) -> Vec<&[u8]> { + let mut kept: Vec<&[u8]> = Vec::with_capacity(lines.len()); + for line in lines { + let blank = line.iter().all(u8::is_ascii_whitespace); + if blank && kept.last() == Some(&line) { + continue; + } + kept.push(line); + } + kept +} + +/// One line of the hunk: its marker, its terminal-safe text, and the colour +/// that says which of the three it is. +fn rendered(marker: char, line: &[u8], presentation: &Presentation) -> String { + let tint = match marker { + '-' => "\x1b[31m", + '+' => "\x1b[32m", + _ => "\x1b[2m", + }; + format!( + "{}{marker}{}{}", + color(tint, presentation.color), + sanitize_source_line(&String::from_utf8_lossy(line)), + color("\x1b[0m", presentation.color) + ) +} + +/// One block of bytes as the lines it holds, with the carriage return of a +/// CRLF file left out of the text rather than shown as a control character. +fn rows_of(block: &[u8]) -> Vec<&[u8]> { + block + .split(|byte| *byte == b'\n') + .map(|line| line.strip_suffix(b"\r").unwrap_or(line)) + .collect() +} + +/// The start of the line byte `offset` falls on. +fn line_start(source: &[u8], offset: usize) -> usize { + source[..offset] + .iter() + .rposition(|byte| *byte == b'\n') + .map_or(0, |at| at + 1) +} + +/// The end of the line byte `offset` falls on, before its terminator. +fn line_end(source: &[u8], offset: usize) -> usize { + source[offset..] + .iter() + .position(|byte| *byte == b'\n') + .map_or(source.len(), |at| offset + at) +} + +/// Up to `count` whole lines ending just before `start`, in source order. +fn preceding(source: &[u8], start: usize, count: usize) -> Vec<&[u8]> { + let mut lines = Vec::new(); + let mut at = start; + while lines.len() < count && at > 0 { + /* NOTE: `at` is a line start, so the byte before it is the terminator of the + * line being collected. */ + let end = at - 1; + let begin = line_start(source, end); + lines.push(&source[begin..end]); + at = begin; + } + lines.reverse(); + lines +} + +/// Up to `count` whole lines starting just after `end`. +fn following(source: &[u8], end: usize, count: usize) -> Vec<&[u8]> { + let mut lines = Vec::new(); + let mut at = end; + while lines.len() < count && at < source.len() { + /* NOTE: Step over the terminator `end` stopped in front of. A file whose last + * line ends in one has nothing after it, and the loop ends here. */ + at += 1; + if at >= source.len() { + break; + } + let stop = line_end(source, at); + lines.push(&source[at..stop]); + at = stop; + } + lines +} + +/// One decision about one comment. +enum Answer { + Yes, + No, + AllInFile, + NoneInFile, + /// Stop asking and apply what was accepted. + Stop, + /// Throw the whole run away. + Abort, + Help, +} + +/// Put the question and read one answer, explaining itself and asking again +/// until the reader gives one. +/// +/// The answer is read as bytes rather than as a line of text: a terminal can +/// deliver anything, and a stray byte is a typo to ask about again, not an I/O +/// failure that ends a run somebody is in the middle of. +fn ask( + input: &mut dyn BufRead, + output: &mut dyn Write, + presentation: &Presentation, +) -> Result { + loop { + wrote(write!(output, "{PROMPT}"))?; + /* NOTE: The question ends without a newline, so it has to be pushed out by + * hand before the run blocks waiting for the answer to it. */ + wrote(output.flush())?; + let mut line = Vec::new(); + /* NOTE: Nothing left to read is a reader who is no longer there to answer, + * which is the one answer that must not be guessed at. */ + if input.read_until(b'\n', &mut line)? == 0 { + return Ok(Answer::Abort); + } + match parse(&String::from_utf8_lossy(&line)) { + Some(Answer::Help) => { + for entry in HELP { + wrote(writeln!(output, "{}", dimmed(entry, presentation)))?; + } + } + Some(answer) => return Ok(answer), + None => wrote(writeln!(output, "{}", dimmed(UNKNOWN, presentation)))?, + } + } +} + +/// Commentary beside the question, told apart from it by being dimmed. +fn dimmed(text: &str, presentation: &Presentation) -> String { + format!( + "{}{text}{}", + color("\x1b[2m", presentation.color), + color("\x1b[0m", presentation.color) + ) +} + +/// The answer one typed line stands for, or `None` for anything else. +fn parse(line: &str) -> Option { + match line.trim().to_ascii_lowercase().as_str() { + "y" | "yes" => Some(Answer::Yes), + "n" | "no" => Some(Answer::No), + "a" => Some(Answer::AllInFile), + "d" => Some(Answer::NoneInFile), + "q" => Some(Answer::Stop), + "x" => Some(Answer::Abort), + "?" | "h" | "help" => Some(Answer::Help), + _ => None, + } +} + +#[cfg(test)] +mod tests { + use super::*; + use ocomment_core::{Language, Layout, TransformOptions, apply_edits, transform}; + use std::{io::Cursor, path::PathBuf}; + + /// Two removable block comments, one line, one file. + const TWO: &str = "a/* one */b/* two */c\n"; + + fn file(name: &str, text: &str) -> ProcessedFile { + let source = text.as_bytes().to_vec(); + let result = transform(&source, Language::C, TransformOptions::default()); + ProcessedFile { + path: PathBuf::from(name), + source, + language: Language::C, + result, + } + } + + /// Drive `select` with a scripted answer per line and collect everything it + /// wrote to the terminal. + fn ask(files: &[ProcessedFile], script: &str) -> (Selection, String) { + let mut input = Cursor::new(script.as_bytes().to_vec()); + let mut written: Vec = Vec::new(); + let selection = select(files, &mut input, &mut written, &Presentation::default()).unwrap(); + (selection, String::from_utf8(written).unwrap()) + } + + fn replacement(selection: &Selection) -> String { + assert_eq!(selection.plans.len(), 1, "expected exactly one write plan"); + String::from_utf8(selection.plans[0].replacement.clone()).unwrap() + } + + /// The answers apply to one comment each: the accepted span is gone and the + /// declined one is still in the bytes that would be written. + #[test] + fn yes_and_no_apply_only_the_accepted_comment() { + let (selection, _) = ask(&[file("a.c", TWO)], "y\nn\n"); + assert_eq!((selection.accepted, selection.declined), (1, 1)); + assert!(!selection.aborted); + assert_eq!(replacement(&selection), "a b/* two */c\n"); + assert_eq!(selection.plans[0].path, PathBuf::from("a.c")); + assert_eq!(selection.plans[0].original, TWO.as_bytes()); + } + + /// The question says which comment it is about — where it starts, what kind + /// it is, and how far through the file and the run it sits — and shows the + /// line as it stands against the line the answer would leave behind. + #[test] + fn the_prompt_names_the_comment_and_shows_the_hunk() { + let (_, transcript) = ask(&[file("a.c", TWO)], "y\nn\n"); + assert!( + transcript.contains("a.c:1:2 block comment (1 of 2 in file, 1 of 2 total)\n"), + "the first question did not name its comment:\n{transcript}" + ); + assert!( + transcript.contains("a.c:1:12 block comment (2 of 2 in file, 2 of 2 total)\n"), + "the second question did not name its comment:\n{transcript}" + ); + assert!( + transcript.contains("-a/* one */b/* two */c\n+a b/* two */c\n"), + "the first question did not show the line it would rewrite:\n{transcript}" + ); + assert!( + transcript.contains("-a/* one */b/* two */c\n+a/* one */b c\n"), + "the second question did not show the line it would rewrite:\n{transcript}" + ); + assert_eq!( + transcript.matches("Remove? [y,n,a,d,q,x,?] ").count(), + 2, + "one question was asked per comment:\n{transcript}" + ); + } + + /// Three lines either side of the comment are shown unprefixed, so the + /// reader can tell what the line is doing before answering for it. + #[test] + fn the_hunk_carries_three_lines_of_context_on_each_side() { + let source = "1\n2\n3\n4\n5\nx/* c */y\n6\n7\n8\n9\n10\n"; + let (_, transcript) = ask(&[file("a.c", source)], "n\n"); + assert!( + transcript.contains(" 3\n 4\n 5\n-x/* c */y\n+x y\n 6\n 7\n 8\n"), + "the hunk is not three lines of context around the change:\n{transcript}" + ); + assert!( + !transcript.contains(" 2\n"), + "the hunk reached a fourth line above the change:\n{transcript}" + ); + assert!( + !transcript.contains(" 9\n"), + "the hunk reached a fourth line below the change:\n{transcript}" + ); + } + + /// A block comment 27 lines tall, with a line of source either side. + fn tall(lines: usize) -> String { + let mut text = String::from("before\n/* comment line 1\n"); + for line in 2..lines { + text.push_str(&format!(" * comment line {line}\n")); + } + text.push_str(&format!(" * comment line {lines} */\nafter\n")); + text + } + + /// A comment tall enough to fill the screen would push the question off it. + /// Both sides of the change are capped at `CONTEXT` lines each end, with one + /// marker standing for everything folded away, so the prompt stays in view. + #[test] + fn a_tall_hunk_is_capped_on_both_sides() { + let source = tall(27); + let (_, transcript) = ask(&[file("a.c", &source)], "n\n"); + let removed = transcript + .lines() + .filter(|line| line.starts_with('-')) + .count(); + let added = transcript + .lines() + .filter(|line| line.starts_with('+')) + .count(); + assert!( + removed <= 2 * CONTEXT + 1, + "the removed side printed {removed} lines:\n{transcript}" + ); + assert!( + added <= 2 * CONTEXT + 1, + "the added side printed {added} lines:\n{transcript}" + ); + assert!( + transcript.contains("-/* comment line 1\n"), + "the removed side lost the first line of the comment:\n{transcript}" + ); + assert!( + transcript.contains("- * comment line 27 */\n"), + "the removed side lost the last line of the comment:\n{transcript}" + ); + assert!( + transcript.contains("more line"), + "a capped hunk did not say how much it folded away:\n{transcript}" + ); + assert!( + transcript.contains("Remove? [y,n,a,d,q,x,?] "), + "the question never arrived:\n{transcript}" + ); + } + + /// A change that fits is shown whole: nothing is folded and nothing says it + /// was. + #[test] + fn a_short_hunk_is_shown_whole() { + let (_, transcript) = ask(&[file("a.c", TWO)], "n\nn\n"); + assert!( + !transcript.contains("more line"), + "a hunk that fits was capped anyway:\n{transcript}" + ); + assert!( + transcript.contains("-a/* one */b/* two */c\n+a b/* two */c\n"), + "the whole change was not shown:\n{transcript}" + ); + } + + /// `a` answers for the rest of the file at once and asks nothing more about + /// it; the next file starts asking again. + #[test] + fn a_removes_the_rest_of_the_file_without_asking() { + let files = [file("a.c", TWO), file("b.c", TWO)]; + let (selection, transcript) = ask(&files, "a\ny\nn\n"); + assert_eq!((selection.accepted, selection.declined), (3, 1)); + assert_eq!(selection.plans.len(), 2); + assert_eq!( + String::from_utf8(selection.plans[0].replacement.clone()).unwrap(), + "a b c\n" + ); + assert_eq!( + String::from_utf8(selection.plans[1].replacement.clone()).unwrap(), + "a b/* two */c\n" + ); + assert_eq!( + transcript.matches("Remove? [y,n,a,d,q,x,?] ").count(), + 3, + "`a` kept asking about the file it answered for:\n{transcript}" + ); + } + + /// `d` is the same for the other answer: nothing in the file is removed, so + /// the file has no plan at all. + #[test] + fn d_keeps_the_rest_of_the_file_without_asking() { + let files = [file("a.c", TWO), file("b.c", TWO)]; + let (selection, transcript) = ask(&files, "d\ny\nn\n"); + assert_eq!((selection.accepted, selection.declined), (1, 3)); + assert_eq!(selection.plans.len(), 1); + assert_eq!(selection.plans[0].path, PathBuf::from("b.c")); + assert_eq!( + transcript.matches("Remove? [y,n,a,d,q,x,?] ").count(), + 3, + "`d` kept asking about the file it answered for:\n{transcript}" + ); + } + + /// `q` stops the run where it stands and keeps what was already accepted. + #[test] + fn q_applies_what_was_accepted_and_stops_asking() { + let files = [file("a.c", TWO), file("b.c", TWO)]; + let (selection, transcript) = ask(&files, "y\nq\ny\n"); + assert_eq!(selection.accepted, 1); + assert!(!selection.aborted); + assert_eq!(replacement(&selection), "a b/* two */c\n"); + assert_eq!(selection.plans[0].path, PathBuf::from("a.c")); + assert_eq!( + transcript.matches("Remove? [y,n,a,d,q,x,?] ").count(), + 2, + "`q` asked another question:\n{transcript}" + ); + } + + /// `x` throws the run away, accepted answers included. + #[test] + fn x_writes_nothing() { + let files = [file("a.c", TWO), file("b.c", TWO)]; + let (selection, _) = ask(&files, "y\nx\ny\n"); + assert!(selection.aborted); + assert!( + selection.plans.is_empty(), + "an aborted run still produced something to write" + ); + } + + /// A closed input is a reader who is no longer there to answer, which is + /// the one answer that cannot be guessed at: it aborts. + #[test] + fn end_of_input_aborts_like_x() { + let (selection, _) = ask(&[file("a.c", TWO)], "y\n"); + assert!(selection.aborted, "the second question ran out of input"); + assert!(selection.plans.is_empty()); + } + + /// `?` is not an answer; it explains the answers and asks again. + #[test] + fn help_is_shown_and_the_question_repeated() { + let (selection, transcript) = ask(&[file("a.c", TWO)], "?\nn\nn\n"); + assert_eq!((selection.accepted, selection.declined), (0, 2)); + assert!( + transcript.contains("a - remove it and every remaining comment in this file"), + "`?` did not explain the answers:\n{transcript}" + ); + assert_eq!( + transcript.matches("Remove? [y,n,a,d,q,x,?] ").count(), + 3, + "`?` did not ask the first question again:\n{transcript}" + ); + } + + /// Anything else is a typo, not a decision, and is never taken for one. + #[test] + fn an_unknown_answer_re_prompts() { + let (selection, transcript) = ask(&[file("a.c", TWO)], "z\n\ny\nn\n"); + assert_eq!((selection.accepted, selection.declined), (1, 1)); + assert_eq!(replacement(&selection), "a b/* two */c\n"); + assert!( + transcript.contains("unknown answer"), + "an unknown answer went unremarked:\n{transcript}" + ); + assert_eq!( + transcript.matches("Remove? [y,n,a,d,q,x,?] ").count(), + 4, + "an unknown answer did not ask again:\n{transcript}" + ); + } + + /// Under `layout = "columns"` a removal is replaced by exactly as many + /// display columns as the comment occupied, and every such replacement is + /// measured from the *source*, not from whatever earlier removals left + /// behind. That is what lets this command apply a subset of the edits a + /// transform produced: a width-preserving replacement moves nothing, so + /// each remaining comment still begins at the display column its own + /// replacement was computed for. + /// + /// Pinned by transforming the partially edited bytes again and requiring + /// the replacement to come out byte-identical to the one the full transform + /// computed — the tab inside the second comment makes that replacement + /// depend on the column it starts at. + #[test] + fn partial_column_edits_keep_the_replacement_the_transform_computed() { + let source = b"x/* one */y/* a\tb */z\n"; + let options = TransformOptions { + layout: Layout::Columns, + ..TransformOptions::default() + }; + let full = transform(source, Language::C, options.clone()); + assert_eq!(full.edits.len(), 2, "the fixture lost a comment"); + + for taken in [0usize, 1] { + let kept = 1 - taken; + let partial = apply_edits(source, std::slice::from_ref(&full.edits[taken])); + let again = transform(&partial, Language::C, options.clone()); + assert_eq!( + again.edits.len(), + 1, + "the partially edited source lost the comment that was kept" + ); + assert_eq!( + again.edits[0].replacement, full.edits[kept].replacement, + "applying edit #{taken} on its own changed what edit #{kept} replaces" + ); + assert_eq!( + again.output, full.output, + "applying edit #{taken} and then the rest is not the whole transform" + ); + } + + let both: Vec = full.edits.clone(); + assert_eq!(apply_edits(source, &both), full.output); + } +} diff --git a/rust/ocomment/src/lsp.rs b/rust/ocomment/src/lsp.rs index d004c47..d677e4e 100644 --- a/rust/ocomment/src/lsp.rs +++ b/rust/ocomment/src/lsp.rs @@ -1,6 +1,7 @@ use crate::{ config::{self, ResolvedConfig}, files, + output::{kept_label, removable_label}, plugin::PluginHost, }; use anyhow::Result as AnyResult; @@ -155,7 +156,7 @@ impl Backend { code: Some(NumberOrString::String("removable-comment".into())), code_description: None, source: Some("ocomment".into()), - message: format!("removable {:?} comment", comment.kind), + message: removable_label(comment.kind), related_information: None, tags: Some(vec![DiagnosticTag::UNNECESSARY]), data: None, @@ -855,11 +856,10 @@ impl LanguageServer for Backend { return Ok(None); }; let text = match &comment.disposition { - Disposition::Remove => format!("OComment: removable {:?} comment", comment.kind), - Disposition::Keep { reason } => format!( - "OComment protects this {:?} comment: {reason}", - comment.kind - ), + Disposition::Remove => format!("OComment: {}", removable_label(comment.kind)), + Disposition::Keep { reason } => { + format!("OComment: {}", kept_label(comment.kind, reason)) + } }; Ok(Some(Hover { contents: HoverContents::Scalar(MarkedString::String(text)), @@ -1056,6 +1056,16 @@ fn language_from_lsp(id: &str, uri: &Url, source: &[u8]) -> (Language, Dialect) "typescriptreact" => (Language::TypeScript, Dialect::Tsx), "objective-c" => (Language::C, Dialect::ObjectiveC), "objective-cpp" => (Language::Cpp, Dialect::ObjectiveCpp), + "cuda-cpp" => (Language::Cpp, Dialect::Cuda), + /* NOTE: One editor id covers sh, Bash, and zsh alike, and the dialects + * differ — `$'...'` is an ANSI-C quoted string in the last two only. + * The id settles the language, so the dialect is taken from the path + * and the bytes whenever they agree it is a shell script at all, and + * falls back to the language default when a buffer offers neither. */ + "shellscript" => ( + Language::Shell, + detected_dialect(uri, source, Language::Shell).unwrap_or(Dialect::Standard), + ), id => id .parse() .map(|language| (language, Dialect::Standard)) @@ -1068,6 +1078,15 @@ fn language_from_lsp(id: &str, uri: &Url, source: &[u8]) -> (Language, Dialect) } } +/// The dialect the path and the bytes imply, when they agree with the language +/// the client named. +fn detected_dialect(uri: &Url, source: &[u8], language: Language) -> Option { + let path = uri.to_file_path().ok(); + detect_language(path.as_deref(), source) + .filter(|detection| detection.language == language) + .map(|detection| detection.dialect) +} + fn incremental_for_document( uri: &Url, document: &Document, @@ -1155,7 +1174,7 @@ fn progress_percentage(completed: usize, total: usize) -> u32 { } fn stable_text_hash(bytes: &[u8]) -> u64 { - // FNV-1a is sufficient for an opaque, session-local LSP result identifier. + // NOTE: FNV-1a is sufficient for an opaque, session-local LSP result identifier. bytes.iter().fold(0xcbf29ce484222325, |hash, byte| { (hash ^ u64::from(*byte)).wrapping_mul(0x100000001b3) }) @@ -1255,6 +1274,41 @@ fn next_line_start(source: &[u8], start: usize) -> Option { #[cfg(test)] mod tests { use super::*; + + /// Every language identifier the VS Code extension attaches the server to + /// has to reach a built-in language here. + /// + /// The two lists are written in different vocabularies: `Language` is named + /// after the language and an editor identifier after whatever the editor + /// calls it, and the two agree for most of them by luck rather than by + /// construction — `objective-c`, `cuda-cpp`, `javascriptreact` and + /// `shellscript` do not agree at all, which is what + /// [`language_from_lsp`]'s arms are for. Nothing else notices when they + /// stop agreeing: the extension still activates for the identifier, the + /// server still opens the document, and the scan comes back with the + /// `unknown-language` diagnostic and no comments. So the selector is read + /// here rather than trusted. + #[test] + fn every_editor_language_identifier_reaches_a_built_in_language() { + let manifest: serde_json::Value = + serde_json::from_str(include_str!("../../../editors/vscode/package.json")) + .expect("package.json parses"); + let identifiers = + manifest["contributes"]["configuration"]["properties"]["ocomment.languages"]["default"] + .as_array() + .expect("`ocomment.languages` has an array default"); + let uri = Url::parse("file:///buffer").expect("a well-formed URI"); + assert!(!identifiers.is_empty()); + for value in identifiers { + let id = value.as_str().expect("a language identifier is a string"); + let (language, _) = language_from_lsp(id, &uri, b""); + assert!( + Language::ALL.contains(&language), + "the editor identifier `{id}` reaches {language}, which has no built-in scanner" + ); + } + } + #[test] fn position_round_trip_all_encodings() { let source = "a😀b\rnext\r\n二\nlast".as_bytes(); diff --git a/rust/ocomment/src/main.rs b/rust/ocomment/src/main.rs index ad05a9b..6a44590 100644 --- a/rust/ocomment/src/main.rs +++ b/rust/ocomment/src/main.rs @@ -3,18 +3,90 @@ mod cli; mod config; mod files; mod git; +mod interactive; mod lsp; mod output; mod plugin; +mod values; -use std::process::ExitCode; +use std::{ + io::{self, Write}, + process::ExitCode, +}; + +/// Whether the reader of the program's own output is what ended the run. +/// +/// `ocomment … | head` closes the pipe as soon as the reader has what it came +/// for. That is the reader finishing, not the run failing, so — following the +/// convention `rg` and `fd` set — the process ends quietly with status 0 +/// rather than reporting an I/O error to a terminal that may itself be gone. +/// The failing write can be several layers down: the serializer wraps it, and +/// the caller adds context on top. +/// +/// Only the writers of *our* report may claim this, and they say so by tagging +/// the failure with [`output::OutputPipeClosed`]. A bare `BrokenPipe` from +/// anywhere else — the write that feeds a rewritten blob to `git hash-object`, +/// above all — is a real failure whose silent success would lose data. +fn output_pipe_closed(error: &anyhow::Error) -> bool { + error + .chain() + .any(|cause| cause.downcast_ref::().is_some()) +} fn main() -> ExitCode { match cli::run() { Ok(code) => ExitCode::from(code), + Err(error) if output_pipe_closed(&error) => ExitCode::SUCCESS, Err(error) => { - eprintln!("ocomment: {error:#}"); + /* NOTE: Nothing is left to try if even the report cannot be written, and + * `eprintln!` would panic there — an abort under the release + * profile — so the failure of the last write is dropped. */ + let _ = writeln!(io::stderr(), "ocomment: {error:#}"); ExitCode::from(2) } } } + +#[cfg(test)] +mod tests { + use super::{output::OutputPipeClosed, output_pipe_closed}; + use anyhow::{Context, Result}; + use std::io::{Error, ErrorKind}; + + /// The tag survives however many layers of context are added on top of it. + #[test] + fn a_tagged_output_pipe_is_recognized_through_its_context() { + let error = Result::<()>::Err(anyhow::Error::new(OutputPipeClosed)) + .context("cannot write standard output") + .context("check failed") + .unwrap_err(); + assert!(output_pipe_closed(&error)); + } + + /// `git hash-object` exiting before it reads the blob raises a bare + /// `BrokenPipe` that no output writer tagged. Ending quietly there would + /// report a `fix --staged` that never happened. + #[test] + fn an_untagged_broken_pipe_is_not_an_output_pipe_closure() { + let error = Result::<()>::Err(Error::from(ErrorKind::BrokenPipe).into()) + .context("cannot write the rewritten blob to git hash-object") + .context("fix failed") + .unwrap_err(); + assert!(!output_pipe_closed(&error)); + } + + #[test] + fn another_io_failure_is_not_an_output_pipe_closure() { + let error = Result::<()>::Err(Error::from(ErrorKind::StorageFull).into()) + .context("cannot write standard output") + .unwrap_err(); + assert!(!output_pipe_closed(&error)); + } + + #[test] + fn an_error_carrying_no_io_failure_is_not_an_output_pipe_closure() { + assert!(!output_pipe_closed(&anyhow::anyhow!( + "plugin `x` is not locked" + ))); + } +} diff --git a/rust/ocomment/src/output.rs b/rust/ocomment/src/output.rs index eea9c53..080b416 100644 --- a/rust/ocomment/src/output.rs +++ b/rust/ocomment/src/output.rs @@ -1,14 +1,22 @@ -use crate::files::SkippedFile; +use crate::{ + config::PolicyTrace, + files::{NO_LANGUAGE, STDIN_PATH, SkippedFile}, +}; use anyhow::Result; use clap::ValueEnum; -use ocomment_core::{ByteSpan, Language, TransformResult}; +use ocomment_core::{ + ByteSpan, Comment, CommentKind, Disposition, DispositionExplanation, DispositionPatterns, + Language, Policy, ScanOptions, TransformResult, explain_comment_with, +}; use serde::Serialize; use serde_json::{Value, json}; use similar::{ChangeTag, TextDiff}; use std::{ - io::{self, Write}, - path::{Path, PathBuf}, + collections::BTreeMap, + io::{self, BufWriter, Write}, + path::{Component, Path, PathBuf}, }; +use unicode_width::UnicodeWidthChar; #[derive(Clone, Copy, Debug, Default, Eq, PartialEq, ValueEnum)] pub enum OutputFormat { @@ -32,7 +40,172 @@ pub enum Operation { pub struct Presentation { pub color: bool, pub hyperlinks: bool, - pub progress: bool, +} + +/// How much of the human report a run is allowed to write. +#[derive(Clone, Copy, Debug, Default, Eq, PartialEq)] +pub enum Verbosity { + /// Only errors and diagnostics. + Quiet, + #[default] + Normal, + /// Everything, including the per-kind breakdown and every skipped file. + Verbose, +} + +/// Everything the renderer needs besides the results themselves. +#[derive(Clone, Copy, Debug)] +pub struct RenderOptions { + pub format: OutputFormat, + pub operation: Operation, + pub presentation: Presentation, + pub verbosity: Verbosity, + /// Human lines carry a one-line rendering of the comment text. + pub preview: bool, + /// Human `check` and `scan` lines carry every comment, kept ones included, + /// each under an indented line naming the rule that decided it. + pub explain: bool, + /// The run is `fix --dry-run`: it produces the diff but speaks the + /// vocabulary of the `fix` it is standing in for. + pub dry_run: bool, + /// `--force-invalid` was in effect, so a file that fails to scan still had + /// its provably safe edits applied. + pub force_invalid: bool, + /// The run reached the disk. A `fix` blocked by invalid syntax or an I/O + /// error leaves this false and must not claim any removal. + pub applied: bool, + /// The policy the run was asked for. Only `all` promises to take every + /// comment out, so only `all` owes an explanation for the ones it keeps. + pub policy: Policy, +} + +/// What one run found, counted once for the end-of-run summary. +#[derive(Clone, Debug, Default, Eq, PartialEq)] +pub struct Summary { + pub files_scanned: usize, + pub files_with_removable: usize, + pub removable_comments: usize, + pub kept_comments: usize, + pub files_changed: usize, + pub comments_removed: usize, + pub invalid_files: usize, + /// Non-error skips met while walking, counted under a short stable label + /// rather than the raw reason, which can carry a configured byte limit. + /// A path named on the command line is deliberately absent: it already has + /// its own line on standard output and must not be counted twice. + pub skipped_by_reason: BTreeMap, + /// Non-error skips whose path was named on the command line. + pub named_skips: usize, + pub io_errors: usize, +} + +impl Summary { + pub fn compute(files: &[ProcessedFile], skipped: &[SkippedFile], operation: Operation) -> Self { + let mut summary = Self { + files_scanned: files.len(), + ..Self::default() + }; + for file in files { + let removable = removable_count(file); + summary.removable_comments += removable; + summary.kept_comments += file.result.report.comments.len() - removable; + if removable > 0 { + summary.files_with_removable += 1; + } + if !file.result.report.valid { + summary.invalid_files += 1; + } + if file.source != file.result.output { + summary.files_changed += 1; + if operation == Operation::Fix { + summary.comments_removed += removable; + } + } + } + for item in skipped { + if item.error { + summary.io_errors += 1; + } else if item.explicit { + summary.named_skips += 1; + } else { + *summary + .skipped_by_reason + .entry(skip_label(&item.reason).to_owned()) + .or_default() += 1; + } + } + summary + } + + fn skipped_files(&self) -> usize { + self.skipped_by_reason.values().sum() + } +} + +fn removable_count(file: &ProcessedFile) -> usize { + file.result + .report + .comments + .iter() + .filter(|comment| comment.disposition.is_remove()) + .count() +} + +/// Fold a skip reason onto a short label the summary can group by. +/// +/// The per-file line says what to do about one file; the summary counts many, +/// so it trades the sentence for a key short enough to sit in a list of them. +/// +/// Visible to the crate so the modules that *produce* the reasons — `files` +/// and `git` — can name this function in their own documentation rather than +/// describing a rule they do not own. +pub(crate) fn skip_label(reason: &str) -> &str { + if reason.starts_with("larger than ") { + "too large" + } else if reason.starts_with("binary file") { + "binary" + } else if reason.starts_with("language disabled") { + "language disabled" + } else if reason == NO_LANGUAGE { + "unknown language" + } else { + reason + } +} + +/// The `Keep` reason the core scanner gives a shebang or encoding line that +/// `--force-protected` would have removed. It is one of the six reasons the +/// differential protocol freezes, so matching on it is stable; the end-to-end +/// test `policy_all_says_how_to_remove_a_kept_preamble` is what would catch it +/// drifting apart from the scanner. +const PROTECTED_PREAMBLE: &str = "required source preamble"; + +/// How many comments were kept only because `--force-protected` was absent. +/// +/// Counted from the disposition rather than from the comment kind: a shebang +/// held back by `--keep-kind shebang` stays kept whatever `--force-protected` +/// says, and advertising the flag for it would be a lie. +fn protected_preambles(files: &[ProcessedFile]) -> usize { + files + .iter() + .flat_map(|file| &file.result.report.comments) + .filter(|comment| { + matches!(&comment.disposition, Disposition::Keep { reason } if reason == PROTECTED_PREAMBLE) + }) + .count() +} + +/// `1 file` / `2 files`: the count and its noun, pluralized by the regular +/// rule. Every noun the summary counts goes through this. +fn plural(count: usize, noun: &str) -> String { + format!("{count} {noun}{}", if count == 1 { "" } else { "s" }) +} + +/// `1 comment` / `2 removable comments`: the noun is pluralized and an +/// optional adjective is placed in front of it. +fn comments(count: usize, adjective: &str) -> String { + let space = if adjective.is_empty() { "" } else { " " }; + plural(count, &format!("{adjective}{space}comment")) } #[derive(Clone, Debug)] @@ -53,125 +226,906 @@ struct JsonFile<'a> { source_map: &'a ocomment_core::SourceMap, } +/// The one-line label for a comment OComment would delete. +pub fn removable_label(kind: CommentKind) -> String { + format!("removable {kind} comment") +} + +/// The one-line label for a comment OComment deliberately protects. +pub fn kept_label(kind: CommentKind, reason: &str) -> String { + format!("{}: {reason}", kept_prefix(kind)) +} + +/// The same label without a reason, for a report that gives the reason on a +/// line of its own. +fn kept_prefix(kind: CommentKind) -> String { + format!("kept {kind} comment") +} + +/// What `--explain` needs to account for one file's comments: the options its +/// scan actually ran with, and where each of their settings came from. +#[derive(Clone, Debug)] +pub struct FileExplanation { + pub options: ScanOptions, + pub trace: PolicyTrace, +} + +/// That material for the files of one run, under the path the run reports each +/// file by. A run that was not asked to explain anything carries none. +pub type Explanations = BTreeMap; + +/// One file's explanation material with its policy patterns already compiled. +/// +/// The two regex sets are the same for every comment in the file, so they are +/// built once when the file is reached rather than once per reported line. +struct Explainer<'a> { + material: &'a FileExplanation, + patterns: DispositionPatterns, +} + +impl<'a> Explainer<'a> { + /// An unparseable pattern list is ignored here as the scanner ignores it, + /// which is exactly what `explain_disposition` falls back to on its own. + fn new(material: &'a FileExplanation) -> Self { + Self { + patterns: DispositionPatterns::compile(&material.options) + .unwrap_or_else(|_| DispositionPatterns::empty()), + material, + } + } +} + +/// The indented line under one reported comment: the rule that decided its +/// fate, and either the setting behind that rule or the flag that would +/// overrule it. +/// +/// The pattern a regex explanation quotes and the globs a source names were +/// both written by whoever wrote the configuration, so the composed line gets a +/// comment preview's treatment before it reaches a terminal: one line, no +/// control sequences. The width is not capped — a line that ends in an ellipsis +/// where the pattern was answers nothing. +fn explanation_line( + file: &ProcessedFile, + comment: &Comment, + explainer: &Explainer<'_>, + options: &RenderOptions, +) -> String { + let material = explainer.material; + let start = comment.span.start.min(file.source.len()); + let end = comment.span.end.clamp(start, file.source.len()); + let verdict = explain_comment_with( + &explainer.patterns, + comment, + &file.source[start..end], + file.language, + &material.options, + ); + let tail = match material.trace.origin_of(&verdict, &material.options) { + Some(origin) => format!(" ({origin})"), + None => next_step(&verdict), + }; + format!( + " {}{}{}", + color("\x1b[2m", options.presentation.color), + fold(&format!("{verdict}{tail}")), + color("\x1b[0m", options.presentation.color) + ) +} + +/// Write that line under the comment it is about, when the run has the +/// material to account for it. +fn write_explanation( + output: &mut impl Write, + file: &ProcessedFile, + comment: &Comment, + explainer: Option<&Explainer<'_>>, + options: &RenderOptions, +) -> Result<()> { + let Some(explainer) = explainer else { + return Ok(()); + }; + wrote(writeln!( + output, + "{}", + explanation_line(file, comment, explainer, options) + )) +} + +/// How to overrule a built-in rule, which no setting decided and no table can +/// be pointed at for. +fn next_step(verdict: &DispositionExplanation) -> String { + match verdict { + DispositionExplanation::ProtectedPreamble => { + "; add --force-protected to remove it".to_owned() + } + DispositionExplanation::KeptHtml => format!( + "; use --remove-kind {} or --policy all to remove it", + CommentKind::HtmlComment + ), + DispositionExplanation::KeptDirective { kind, .. } => { + format!("; use --remove-kind {kind} or --policy all to remove it") + } + /* NOTE: The one keep with no flag behind it. `--policy all` does not + * reach it either: what holds the body open is whatever comment is + * still standing under this one, so that is the line to take first. */ + DispositionExplanation::KeptStructural { .. } => { + "; the comment under it has to go first".to_owned() + } + _ => String::new(), + } +} + +/// How many display columns a comment preview may occupy. +const PREVIEW_COLUMNS: usize = 72; + +/// A one-line, terminal-safe rendering of the comment at `span`. +/// +/// Comment text is untrusted input that is about to be written to a terminal, +/// so the whole comment is folded onto one line, every control character — +/// `ESC` above all — is replaced with U+FFFD instead of being forwarded, and +/// the result is cut to `max_columns` display columns. +fn preview(source: &[u8], span: ByteSpan, max_columns: usize) -> String { + let start = span.start.min(source.len()); + let end = span.end.clamp(start, source.len()); + truncate( + fold(&String::from_utf8_lossy(&source[start..end])), + max_columns, + ) +} + +/// The same treatment for a line that did not come out of a source file. +/// +/// What an external tool on `PATH` says about itself is untrusted for exactly +/// the reason a comment is: `doctor` prints it to the same terminal, and a +/// tool planted there could otherwise clear the screen or repaint the report +/// from its own version line. +pub(crate) fn sanitize_line(text: &str) -> String { + truncate(fold(text), PREVIEW_COLUMNS) +} + +/// The same treatment for a message that must not be cut short. +/// +/// A comment preview is commentary and can be trusted to a fixed width, but a +/// diagnostic is the whole answer to a run that produced nothing else. The +/// `regex` crate writes a parse error over several lines, with a caret under +/// the byte it stopped at; the caret means nothing once the lines are joined, +/// yet the sentence after it names what is actually wrong with the pattern. So +/// this one folds — one line, no control characters — and keeps every word. +pub(crate) fn sanitize_message(text: &str) -> String { + fold(text) +} + +/// The same treatment for a name that must not be cut short — or reworded. +/// +/// A directory name is chosen by whoever made the directory, so the rows +/// `doctor` prints one on are untrusted for the same reason a version line is. +/// What they are not is commentary: an absolute path is easily longer than a +/// comment preview may be, and a row that ends in an ellipsis where the reader +/// was looking for the rest of the path answers nothing. +/// +/// Neither is the whitespace in a path commentary, which is why this does not +/// borrow [`fold`]: a name may begin with a space or carry a tab, and a reader +/// who is shown neither cannot type the name back, nor find it in a checkout +/// that has it. So the spacing is left exactly as it was given and every +/// control character — the tab among them — is replaced with U+FFFD, which +/// keeps the promise `fold` was borrowed for in the first place: whatever the +/// name holds, the row stays one row. +pub(crate) fn sanitize_path(text: &str) -> String { + text.chars() + .map(|character| { + if is_control(character) { + '\u{fffd}' + } else { + character + } + }) + .collect() +} + +/// The same treatment for a line of source a prompt has to show as code. +/// +/// A hunk is read for its shape as much as for its text — indentation says +/// what a line belongs to — so unlike a comment preview this one keeps the +/// spaces it was given and expands a tab onto the same eight-column stop the +/// `columns` layout measures a replacement by. What it does not keep is +/// anything that drives the terminal: every control character, `ESC` and the +/// bidirectional overrides above all, still becomes U+FFFD, and the result is +/// still one line cut to a fixed width, because the question underneath it has +/// to stay on the screen with it. +pub(crate) fn sanitize_source_line(text: &str) -> String { + let mut line = String::with_capacity(text.len()); + let mut column = 0usize; + for character in text.chars() { + if character == '\t' { + let width = TAB_WIDTH - (column % TAB_WIDTH); + line.extend(std::iter::repeat_n(' ', width)); + column += width; + } else if is_control(character) { + line.push('\u{fffd}'); + column += 1; + } else { + line.push(character); + column += columns(character); + } + } + truncate(line, PREVIEW_COLUMNS) +} + +/// The tab stop `sanitize_source_line` expands to, the one the `columns` +/// layout already measures a tab by. +const TAB_WIDTH: usize = 8; + +/// Fold `text` onto one control-free line. +fn fold(text: &str) -> String { + let mut folded = String::with_capacity(text.len()); + let mut pending_space = false; + for character in text.chars() { + if matches!(character, ' ' | '\t' | '\r' | '\n' | '\u{c}') { + /* NOTE: Leading whitespace is dropped, and a run only becomes a space + * once something else follows it, so the tail is trimmed too. */ + pending_space = !folded.is_empty(); + continue; + } + if pending_space { + folded.push(' '); + pending_space = false; + } + folded.push(if is_control(character) { + '\u{fffd}' + } else { + character + }); + } + folded +} + +/// C0, DEL, C1, and the bidirectional and separator format controls. None of +/// these may reach the terminal verbatim: C0 drives it, the bidi overrides and +/// isolates can make a comment render as its own reverse, and U+2028/U+2029 +/// break the promise that a preview is one line. U+061C joins the marks it +/// belongs with, and U+FEFF is invisible wherever it lands. +fn is_control(character: char) -> bool { + matches!( + character, + '\u{0}'..='\u{1f}' + | '\u{7f}'..='\u{9f}' + | '\u{61c}' + | '\u{200e}'..='\u{200f}' + | '\u{2028}'..='\u{2029}' + | '\u{202a}'..='\u{202e}' + | '\u{2066}'..='\u{2069}' + | '\u{feff}' + ) +} + +fn columns(character: char) -> usize { + UnicodeWidthChar::width(character).unwrap_or(0) +} + +/// How many characters a preview may carry for each column it may occupy. +/// Zero-width and combining characters cost no columns, so the width budget on +/// its own cannot bound the line a terminal has to hold. +const PREVIEW_CHARS_PER_COLUMN: usize = 4; + +/// Cut `text` to `max_columns` display columns and to a hard character cap, +/// never inside a wide character, leaving room for the ellipsis that marks the +/// cut. +fn truncate(text: String, max_columns: usize) -> String { + let max_chars = max_columns.saturating_mul(PREVIEW_CHARS_PER_COLUMN); + if text.chars().map(columns).sum::() <= max_columns && text.chars().count() <= max_chars + { + return text; + } + let column_budget = max_columns.saturating_sub(1); + let char_budget = max_chars.saturating_sub(1); + let mut cut = String::with_capacity(text.len()); + let mut width = 0usize; + for (taken, character) in text.chars().enumerate() { + if taken >= char_budget { + break; + } + width += columns(character); + if width > column_budget { + break; + } + cut.push(character); + } + cut.push('\u{2026}'); + cut +} + +/// The `: ` tail a human line carries, dimmed when colour is on. +fn preview_suffix(source: &[u8], span: ByteSpan, options: &RenderOptions) -> String { + if !options.preview { + return String::new(); + } + let text = preview(source, span, PREVIEW_COLUMNS); + if text.is_empty() { + return String::new(); + } + format!( + ": {}{text}{}", + color("\x1b[2m", options.presentation.color), + color("\x1b[0m", options.presentation.color) + ) +} + +/// The handle every path that writes the product of a run takes: standard +/// output, locked once for the whole run and buffered. +/// +/// `println!` panics when its write fails, and the release profile aborts on +/// panic, so a reader that stops early — `ocomment … | head` — would end the +/// process with SIGABRT. Writing through a handle that returns its errors lets +/// the caller decide instead, and `main` ends a closed pipe quietly. +pub type Stdout = BufWriter>; + +/// Lock standard output for the rest of the run and buffer it. +pub fn stdout() -> Stdout { + BufWriter::new(io::stdout().lock()) +} + +/// The reader of the program's own output went away mid-run. +/// +/// A broken pipe is only benign when it is *our* report that could not be +/// written; `ocomment … | head` is a reader that finished, not a run that +/// failed. Every other broken pipe — writing a rewritten blob into +/// `git hash-object`, for one — is a real failure, so the benign case is +/// tagged with this marker at the write that raised it instead of being +/// recognized by error kind anywhere in the chain. +#[derive(Debug)] +pub struct OutputPipeClosed; + +impl std::fmt::Display for OutputPipeClosed { + fn fmt(&self, formatter: &mut std::fmt::Formatter<'_>) -> std::fmt::Result { + formatter.write_str("the reader of standard output closed the pipe") + } +} + +impl std::error::Error for OutputPipeClosed {} + +/// Push the last buffered bytes out. +/// +/// A `BufWriter` drops the error of the write it performs while being dropped, +/// so every writer is finished by hand and the failure reaches the caller. +pub fn finish(writer: &mut impl Write) -> Result<()> { + wrote(writer.flush()) +} + +/// Raise one write to the program's own output, tagging the reader that closed +/// the pipe so `main` can end quietly for that case alone. +pub fn wrote(result: io::Result<()>) -> Result<()> { + result.map_err(output_failure) +} + +/// The error one failed write to our own output becomes. +fn output_failure(error: io::Error) -> anyhow::Error { + if error.kind() == io::ErrorKind::BrokenPipe { + return anyhow::Error::new(OutputPipeClosed); + } + anyhow::Error::new(error).context("cannot write standard output") +} + +/// Write one line of commentary to standard error. +/// +/// Commentary — the `-v` trace, the end-of-run summary — is not the product of +/// the run, so a reader that has already gone away is not a failure to report: +/// a closed pipe is dropped and only a real write failure is raised. What must +/// not happen is what `eprintln!` does, which is panic, and so abort under the +/// release profile. +pub fn note(writer: &mut impl Write, line: &str) -> Result<()> { + match writeln!(writer, "{line}") { + Err(error) if error.kind() != io::ErrorKind::BrokenPipe => { + Err(anyhow::Error::new(error).context("cannot write standard error")) + } + _ => Ok(()), + } +} + +/// Turn a serialization failure back into the I/O error it usually is. +/// +/// `serde_json` reports a failed write as an error of its own whose `source` +/// is the *source* of the I/O error rather than the I/O error itself, so a +/// closed pipe would be invisible to anything walking the chain. Its `From` +/// conversion hands the original error back. +fn write_error(error: serde_json::Error) -> anyhow::Error { + output_failure(io::Error::from(error)) +} + pub fn render( files: &[ProcessedFile], skipped: &[SkippedFile], - format: OutputFormat, - operation: Operation, - presentation: Presentation, + options: &RenderOptions, ) -> Result<()> { - match format { - OutputFormat::Human => render_human(files, skipped, operation, presentation), - OutputFormat::Json => render_json(files, skipped), - OutputFormat::Jsonl => render_jsonl(files, skipped), - OutputFormat::Sarif => render_sarif(files, skipped), - OutputFormat::Github => render_github(files, skipped), - } + render_explained(files, skipped, options, &Explanations::new()) +} + +/// The same report, with the material `--explain` needs for the files it has +/// it for. A file with none is reported exactly as `render` reports it. +pub fn render_explained( + files: &[ProcessedFile], + skipped: &[SkippedFile], + options: &RenderOptions, + explanations: &Explanations, +) -> Result<()> { + let mut output = stdout(); + match options.format { + OutputFormat::Human => render_human(&mut output, files, skipped, options, explanations), + OutputFormat::Json => render_json(&mut output, files, skipped), + OutputFormat::Jsonl => render_jsonl(&mut output, files, skipped), + OutputFormat::Sarif => render_sarif(&mut output, files, skipped), + OutputFormat::Github => render_github(&mut output, files, skipped, options.verbosity), + }?; + finish(&mut output) } fn render_human( + output: &mut impl Write, files: &[ProcessedFile], skipped: &[SkippedFile], - operation: Operation, - presentation: Presentation, + options: &RenderOptions, + explanations: &Explanations, ) -> Result<()> { - let stdout = io::stdout(); - let mut output = stdout.lock(); + let operation = options.operation; + let presentation = options.presentation; + let quiet = options.verbosity == Verbosity::Quiet; + let verbose = options.verbosity == Verbosity::Verbose; for file in files { if operation == Operation::Diff && file.source != file.result.output { - write!( + /* NOTE: The patch is the product of `diff`, so `-q` keeps it and drops + * only the summary that follows on standard error. */ + wrote(write!( output, "{}", unified_diff(&file.path, &file.source, &file.result.output) - )?; + ))?; continue; } for diagnostic in &file.result.report.diagnostics { let (line, column) = line_column(&file.source, diagnostic.span.start); - writeln!( + wrote(writeln!( output, - "{}:{line}:{column}: {}{:?}[{}]{}: {}", + "{}:{line}:{column}: {}{}[{}]{}: {}", display_path(&file.path, presentation.hyperlinks), color("\x1b[31m", presentation.color), diagnostic.severity, diagnostic.code, color("\x1b[0m", presentation.color), diagnostic.message - )?; + ))?; } + let explainer = options + .explain + .then(|| explanations.get(&file.path)) + .flatten() + .map(Explainer::new); + let explainer = explainer.as_ref(); if operation == Operation::Scan { + // NOTE: The listing is the product of `scan`; `-q` keeps it too. for comment in &file.result.report.comments { let (line, column) = line_column(&file.source, comment.span.start); - writeln!( + wrote(writeln!( output, - "{}:{line}:{column}: {:?} {:?} {}..{}", + "{}:{line}:{column}: {} {} {}..{}{}", display_path(&file.path, presentation.hyperlinks), comment.kind, comment.disposition, comment.span.start, - comment.span.end - )?; + comment.span.end, + preview_suffix(&file.source, comment.span, options) + ))?; + write_explanation(output, file, comment, explainer, options)?; } - } else if operation != Operation::Fix { - for comment in file - .result - .report - .comments - .iter() - .filter(|comment| comment.disposition.is_remove()) - { + } else if quiet { + continue; + } else if operation == Operation::Fix { + if options.applied && file.source != file.result.output { + wrote(writeln!( + output, + "fixed {}: removed {}", + display_path(&file.path, presentation.hyperlinks), + comments(removable_count(file), "") + ))?; + } + } else { + /* NOTE: `check` reports what it would remove. Asked to explain itself it + * reports the rest too, because a comment it left alone is exactly + * the one the reader is asking about. */ + for comment in &file.result.report.comments { + let removable = comment.disposition.is_remove(); + if !options.explain && !removable { + continue; + } let (line, column) = line_column(&file.source, comment.span.start); - writeln!( + wrote(writeln!( output, - "{}:{line}:{column}: {}removable {:?} comment{}", + "{}:{line}:{column}: {}{}{}{}", display_path(&file.path, presentation.hyperlinks), - color("\x1b[33m", presentation.color), - comment.kind, - color("\x1b[0m", presentation.color) - )?; + color( + if removable { "\x1b[33m" } else { "\x1b[32m" }, + presentation.color + ), + if removable { + removable_label(comment.kind) + } else { + kept_prefix(comment.kind) + }, + color("\x1b[0m", presentation.color), + preview_suffix(&file.source, comment.span, options) + ))?; + write_explanation(output, file, comment, explainer, options)?; } } } + let skips = skip_lines(skipped, presentation, options.verbosity); + /* NOTE: `diff` keeps standard output for the patch alone, so the skips it met + * are left to standard error. `fix --dry-run` is that same `diff` speaking + * for the `fix` it stands in for: a skipped path can be the whole answer + * to the run, so the preview still owes the reader the reason — but beside + * the summary that counts it, because what the preview promises on + * standard output is a patch that has to survive being piped into `git + * apply`. A plain `fix` writes no patch and keeps its skips there. */ if operation != Operation::Diff { - for item in skipped { - writeln!( - output, + for line in &skips { + wrote(writeln!(output, "{line}"))?; + } + } + /* NOTE: The findings are on standard output and the commentary that follows is + * on standard error; a terminal sees both, so the buffer is emptied first + * to keep the report in the order it was written. */ + finish(output)?; + let stderr = io::stderr(); + let mut report = stderr.lock(); + if operation == Operation::Diff && options.dry_run { + for line in &skips { + note(&mut report, line)?; + } + } + if quiet { + return Ok(()); + } + let summary = Summary::compute(files, skipped, operation); + let folded = !verbose && skipped.iter().any(|item| !item.error && !item.explicit); + if verbose && let Some(line) = kind_breakdown(files, options) { + note(&mut report, &line)?; + } + note(&mut report, &summary_report(&summary, options, folded))?; + /* NOTE: Under any other policy a kept preamble is one of many deliberate keeps + * and saying so every run would be noise. `all` said it would take + * everything, so what it left behind is the surprise worth a line. */ + if options.policy == Policy::All { + let protected = protected_preambles(files); + if protected > 0 { + /* NOTE: The line counts what it kept, so the pronoun that stands for it + * has to agree with that count. */ + let pronoun = if protected == 1 { "it" } else { "them" }; + note( + &mut report, + &format!( + "{} kept; add --force-protected to remove {pronoun}.", + comments(protected, "protected preamble") + ), + )?; + } + } + if summary.invalid_files > 0 && !options.force_invalid { + let (verb, pronoun) = if summary.invalid_files == 1 { + ("has", "it") + } else { + ("have", "them") + }; + note( + &mut report, + &format!( + "{} {verb} invalid syntax; nothing was written for {pronoun} \ + (use --force-invalid to apply known-safe edits).", + plural(summary.invalid_files, "file") + ), + )?; + } + Ok(()) +} + +/// The skips one run has to name, in one wording for whichever stream ends up +/// carrying them. An I/O error is named however quiet the run was asked to be: +/// it is a failure, not commentary. +/// +/// Shared with `fix --interactive`, which writes no report of its own and would +/// otherwise be the one command that never says why it passed a file over. +/// Whether a skip is worth a line of the report, in human and in GitHub form. +/// +/// An I/O error decides the exit code, so it is said however quietly the run +/// was asked to speak. A path the caller named is answered on a line of its +/// own, because they asked about that path. What a walk merely wandered past +/// is neither: one unscannable file is a skip, forty of them are noise, and +/// the end-of-run summary counts those instead — `-v` is how a reader asks for +/// the list. Both renderers share this so the two cannot drift apart. +pub(crate) fn skip_is_visible(item: &SkippedFile, verbosity: Verbosity) -> bool { + item.error + || (verbosity != Verbosity::Quiet && (item.explicit || verbosity == Verbosity::Verbose)) +} + +pub(crate) fn skip_lines( + skipped: &[SkippedFile], + presentation: Presentation, + verbosity: Verbosity, +) -> Vec { + skipped + .iter() + .filter(|item| skip_is_visible(item, verbosity)) + .map(|item| { + format!( "{}: {}: {}", display_path(&item.path, presentation.hyperlinks), if item.error { "error" } else { "skipped" }, item.reason - )?; + ) + }) + .collect() +} + +/// The numbers an interactive run's verdict is built from. +/// +/// They count answers rather than findings, which is the one thing the ordinary +/// summary cannot say: it counts what a run *could* have removed. +#[derive(Clone, Copy, Debug, Default, Eq, PartialEq)] +pub(crate) struct InteractiveOutcome { + /// Comments the reader accepted for removal. + pub removed: usize, + /// Questions the reader answered. `a` and `d` answer for every remaining + /// comment in their file, so those count here too. + pub reviewed: usize, + /// Comments the run had to offer, whether or not it got as far as asking. + pub offered: usize, + /// Files an accepted removal is written to. + pub changed: usize, + /// Files the run scanned. + pub scanned: usize, +} + +/// What an interactive run came to, in the vocabulary every other summary uses. +/// +/// A run with nothing to offer borrows the wording the plain `fix` summary +/// gives the same answer, because the only number worth reporting there is how +/// much was looked at. A run stopped by `q` is counted against the questions it +/// actually asked, and says how many it never got to: measuring the acceptances +/// against every comment the run *could* have offered would read as a pile of +/// refusals nobody made. +/// +/// Either way the verdict closes on the `(N files scanned)` every other summary +/// ends with. Answering questions about three files says nothing about how many +/// were opened to find them, and that is the number a reader checks a run +/// against. +pub(crate) fn interactive_summary(outcome: InteractiveOutcome) -> String { + if outcome.offered == 0 { + return format!("Nothing to fix in {}.", plural(outcome.scanned, "file")); + } + let unreviewed = outcome.offered.saturating_sub(outcome.reviewed); + let tail = if unreviewed == 0 { + String::new() + } else { + format!(" ({} not reviewed)", comments(unreviewed, "")) + }; + format!( + "Removed {} of {} in {}{tail} ({} scanned).", + outcome.removed, + comments(outcome.reviewed, ""), + plural(outcome.changed, "file"), + plural(outcome.scanned, "file") + ) +} + +/// The whole end-of-run summary: the verdict for the run, the folded skips, +/// and the I/O errors that were listed one by one above it. +fn summary_report(summary: &Summary, options: &RenderOptions, folded: bool) -> String { + let skips = skip_clause(summary, folded); + let nothing = nothing_to(options); + let mut report = if summary.files_scanned > 0 { + format!("{}{skips}", summary_line(summary, options)) + } else if !skips.is_empty() { + /* NOTE: Nothing was scanned, so the verdict would count zero files; what the + * run actually did was pass every candidate over. */ + format!("Nothing to {nothing}:{skips}") + } else if summary.named_skips > 0 { + format!("Nothing to {nothing}.") + } else { + summary_line(summary, options) + }; + if summary.io_errors > 0 { + report.push_str(&format!(" {}.", plural(summary.io_errors, "I/O error"))); + } + report +} + +/// The verb a run uses for the work it found nothing to do. `fix --dry-run` +/// borrows the vocabulary of the `fix` it is standing in for, as it does +/// everywhere else in the summary. +fn nothing_to(options: &RenderOptions) -> &'static str { + match options.operation { + Operation::Check => "check", + Operation::Fix => "fix", + Operation::Diff if options.dry_run => "fix", + Operation::Diff => "diff", + Operation::Scan => "scan", + } +} + +/// The one-line verdict for the run, without the skipped-file clause. +fn summary_line(summary: &Summary, options: &RenderOptions) -> String { + let scanned = plural(summary.files_scanned, "file"); + let found = || { + format!( + "Found {} in {} ({scanned} scanned).", + comments(summary.removable_comments, "removable"), + plural(summary.files_with_removable, "file") + ) + }; + match options.operation { + /* NOTE: `fix --dry-run` is the diff of a fix: it counts what a real run would + * take out and points back at the run that would write it. */ + Operation::Diff if options.dry_run => { + if summary.removable_comments == 0 { + return format!("Nothing to fix in {scanned}."); + } + format!( + "Would remove {} in {}. Rerun without --dry-run to apply.", + comments(summary.removable_comments, ""), + plural(summary.files_with_removable, "file") + ) + } + Operation::Check | Operation::Diff => { + if summary.removable_comments == 0 { + return format!("No removable comments in {scanned}."); + } + let next = if options.operation == Operation::Diff { + "apply the patch" + } else if summary.removable_comments == 1 { + "remove it" + } else { + "remove them" + }; + format!("{} Run `ocomment fix` to {next}.", found()) + } + Operation::Fix => { + if options.applied && summary.files_changed > 0 { + format!( + "Removed {} in {} ({scanned} scanned).", + comments(summary.comments_removed, ""), + plural(summary.files_changed, "file") + ) + } else if summary.removable_comments == 0 { + format!("Nothing to fix in {scanned}.") + } else { + /* NOTE: The transaction never reached the disk; report what is still + * there rather than claiming a removal. */ + found() + } } + Operation::Scan => format!( + "Scanned {scanned}: {} ({} removable, {} kept).", + comments(summary.removable_comments + summary.kept_comments, ""), + summary.removable_comments, + summary.kept_comments + ), } - if presentation.progress { - eprintln!( - "ocomment: processed {} file(s), skipped {}", - files.len(), - skipped.len() - ); +} + +/// The skipped-file clause appended to the summary line. Only the skips met +/// while walking are folded here; a named path was already reported on its own +/// line. +fn skip_clause(summary: &Summary, folded: bool) -> String { + let total = summary.skipped_files(); + if total == 0 { + return String::new(); } - Ok(()) + let reasons: Vec<_> = summary + .skipped_by_reason + .iter() + .map(|(label, count)| format!("{label}: {count}")) + .collect(); + let hint = if folded { "; use -v to list" } else { "" }; + format!( + " {} skipped ({}{hint}).", + plural(total, "file"), + reasons.join(", ") + ) } -fn color(code: &'static str, enabled: bool) -> &'static str { +/// The `-v` breakdown of what each comment kind contributed. +fn kind_breakdown(files: &[ProcessedFile], options: &RenderOptions) -> Option { + let verb = if options.operation == Operation::Fix && options.applied { + "removed" + } else { + "removable" + }; + let mut removable = [0usize; CommentKind::ALL.len()]; + let mut kept = [0usize; CommentKind::ALL.len()]; + for file in files { + for comment in &file.result.report.comments { + let slot = CommentKind::ALL + .iter() + .position(|kind| *kind == comment.kind) + .expect("CommentKind::ALL lists every kind"); + if comment.disposition.is_remove() { + removable[slot] += 1; + } else { + kept[slot] += 1; + } + } + } + let mut parts = Vec::new(); + for (slot, kind) in CommentKind::ALL.into_iter().enumerate() { + if removable[slot] > 0 { + parts.push(format!("{kind} {} {verb}", removable[slot])); + } + if kept[slot] > 0 { + parts.push(format!("{kind} {} kept", kept[slot])); + } + } + (!parts.is_empty()).then(|| format!("kinds: {}", parts.join(", "))) +} + +pub(crate) fn color(code: &'static str, enabled: bool) -> &'static str { if enabled { code } else { "" } } +/// The path half of a report line, and the hyperlink wrapped around it. +/// +/// A file name is chosen by whoever made the file, so the shown half is +/// untrusted input on its way to a terminal exactly like the preview beside +/// it, and gets `sanitize_path`'s treatment: one line, no control characters, +/// and no width cap, because a path cut to an ellipsis names no file. +/// +/// The link *target* is untrusted for the same reason and by the same route — +/// the frame around it is written in escape bytes, so a name carrying one of +/// its own would close the frame early and the rest of the name would be read +/// as terminal instructions. A URL cannot carry a byte it has no spelling for +/// anyway, so the target is encoded outright rather than patched up for the +/// three characters somebody thought of first. fn display_path(path: &Path, hyperlinks: bool) -> String { - let display = path.display().to_string(); + let display = sanitize_path(&path.display().to_string()); if !hyperlinks { return display; } let absolute = path.canonicalize().unwrap_or_else(|_| path.to_path_buf()); - let target = absolute - .to_string_lossy() - .replace('%', "%25") - .replace(' ', "%20") - .replace('#', "%23"); + let target = percent_encode(&absolute.to_string_lossy()); format!("\x1b]8;;file://{target}\x1b\\{display}\x1b]8;;\x1b\\") } -fn render_json(files: &[ProcessedFile], skipped: &[SkippedFile]) -> Result<()> { +/// The path half of a `file://` URL, with every byte a URL may not carry +/// spelled as the `%XX` a reader of the URL puts back. +/// +/// The unreserved set of RFC 3986 is kept as it stands, and so is the `/` that +/// separates one path segment from the next; everything else — the space and +/// the `#` that used to be special-cased here, the `%` that makes an encoding +/// an encoding, and every control byte — is encoded. A path is bytes rather +/// than characters, so the encoding is done over the UTF-8 the name is spelled +/// in: a `%XX` pair is defined as a byte, and half an encoded character is not +/// a character a terminal can put back together. +fn percent_encode(path: &str) -> String { + let mut encoded = String::with_capacity(path.len()); + for byte in path.bytes() { + if byte.is_ascii_alphanumeric() || matches!(byte, b'-' | b'.' | b'_' | b'~' | b'/') { + encoded.push(char::from(byte)); + } else { + encoded.push('%'); + encoded.push(HEX[usize::from(byte >> 4)]); + encoded.push(HEX[usize::from(byte & 0xf)]); + } + } + encoded +} + +/// The digits a percent-encoded byte is spelled with. RFC 3986 asks for the +/// upper-case ones. +const HEX: [char; 16] = [ + '0', '1', '2', '3', '4', '5', '6', '7', '8', '9', 'A', 'B', 'C', 'D', 'E', 'F', +]; + +fn render_json( + output: &mut impl Write, + files: &[ProcessedFile], + skipped: &[SkippedFile], +) -> Result<()> { let values: Vec<_> = files.iter().map(json_file).collect(); let skipped: Vec<_> = skipped .iter() @@ -180,26 +1134,30 @@ fn render_json(files: &[ProcessedFile], skipped: &[SkippedFile]) -> Result<()> { }) .collect(); serde_json::to_writer_pretty( - io::stdout().lock(), + &mut *output, &json!({"version": 1, "files": values, "skipped": skipped}), - )?; - println!(); + ) + .map_err(write_error)?; + wrote(writeln!(output))?; Ok(()) } -fn render_jsonl(files: &[ProcessedFile], skipped: &[SkippedFile]) -> Result<()> { - let stdout = io::stdout(); - let mut output = stdout.lock(); +fn render_jsonl( + output: &mut impl Write, + files: &[ProcessedFile], + skipped: &[SkippedFile], +) -> Result<()> { for file in files { - serde_json::to_writer(&mut output, &json_file(file))?; - writeln!(output)?; + serde_json::to_writer(&mut *output, &json_file(file)).map_err(write_error)?; + wrote(writeln!(output))?; } for item in skipped { serde_json::to_writer( - &mut output, + &mut *output, &json!({"type": "skip", "path": item.path.to_string_lossy(), "reason": item.reason, "error": item.error}), - )?; - writeln!(output)?; + ) + .map_err(write_error)?; + wrote(writeln!(output))?; } Ok(()) } @@ -215,9 +1173,184 @@ fn json_file(file: &ProcessedFile) -> JsonFile<'_> { } } -fn render_sarif(files: &[ProcessedFile], skipped: &[SkippedFile]) -> Result<()> { +/// Where a SARIF reader is sent to learn what the tool itself is. +const TOOL_INFORMATION_URI: &str = "https://github.com/P4suta/OComment"; + +/// Where a rule about a comment sends a reader asking why that comment is +/// reported — and why the one beside it is not. +const KIND_HELP_URI: &str = "https://github.com/P4suta/OComment#why-was-this-comment-kept"; + +/// The base id a path under the directory the run walked is reported against. +/// SARIF readers, GitHub code scanning among them, resolve `%SRCROOT%` to the +/// root of the checkout. +const SRCROOT: &str = "%SRCROOT%"; + +/// The one sentence every scan diagnostic is described by. The codes are as +/// varied as the languages that raise them, and the result carries the message +/// that says what was actually met. +const DIAGNOSTIC_DESCRIPTION: &str = + "A problem OComment met while scanning the file; the message on the result says what it was."; + +/// The spelling a machine format reports a path under. +/// +/// A SARIF `artifactLocation.uri` and the `file=` of a GitHub annotation are +/// both matched against the paths the repository uses, so a reported path is +/// spelled the way the repository spells it: forward slashes on every +/// platform, and none of the `.` segments a walk root or a typed target leaves +/// behind — `sub/./doc.rs` names a file no checkout has. What a relative path +/// is measured *from* is said separately, by [`artifact_location`]. +fn report_uri(path: &Path) -> String { + let text = path.to_string_lossy().replace('\\', "/"); + let trimmed: Vec<&str> = text.split('/').filter(|segment| *segment != ".").collect(); + if trimmed.is_empty() { + /* NOTE: The path was `.` (or `./`) and naming nothing at all would be worse + * than naming the directory. */ + return text; + } + trimmed.join("/") +} + +/// The SARIF `artifactLocation` for a reported path. +/// +/// A path under the directory the run started in is reported against +/// `%SRCROOT%`: SARIF resolves a relative URI against a base id, and a reader +/// given none has nothing to resolve it against, so the finding lands on no +/// file. An absolute path is not under the checkout as far as the run can +/// tell, one that climbs out through `..` has left it, and the pseudo-path +/// standard input is reported under is not a file at all — each of those is +/// reported as it stands, with no base id claiming otherwise. +fn artifact_location(path: &Path) -> Value { + let uri = report_uri(path); + if under_source_root(path) { + let uri = if reads_as_a_drive_letter(&uri) { + format!("./{uri}") + } else { + uri + }; + json!({"uri": uri, "uriBaseId": SRCROOT}) + } else { + json!({"uri": uri}) + } +} + +/// Whether a repository-relative URI opens with a segment no reader will take +/// for a directory name. +/// +/// A `uri` is read as a URI, and RFC 3986 hands a relative reference's first +/// segment to the scheme as soon as it holds a colon: `c:/a.rs` parses as the +/// scheme `c` over the path `/a.rs`, and a Windows reader sees a drive letter +/// in it besides. A POSIX checkout is free to hold a directory named `c:`, so +/// the path says which it meant with the one `.` segment the standard keeps +/// for exactly this: `./c:/a.rs` is a relative reference whatever reads it, +/// and it still resolves against `%SRCROOT%`. +/// +/// Only a repository-relative path is treated this way. A GitHub annotation is +/// matched against the paths the checkout uses rather than parsed as a URI, so +/// [`report_uri`] leaves the spelling alone and only this document adds to it; +/// `tools/validate_schemas.py` is the other half of the rule and turns down +/// the bare form. +fn reads_as_a_drive_letter(uri: &str) -> bool { + let mut head = uri.split('/').next().unwrap_or_default().chars(); + matches!( + (head.next(), head.next(), head.next()), + (Some(letter), Some(':'), None) if letter.is_ascii_alphabetic() + ) +} + +fn under_source_root(path: &Path) -> bool { + path != Path::new(STDIN_PATH) + && path + .components() + .all(|component| matches!(component, Component::Normal(_) | Component::CurDir)) + && path + .components() + .any(|component| matches!(component, Component::Normal(_))) +} + +/// The rules of one SARIF run, and the index each result points at. +/// +/// A result names its rule twice: by `ruleId`, and by the position of that +/// rule's description in `tool.driver.rules`. A code-scanning UI shows a +/// finding through that description — its title, the sentence under it, and +/// the link it offers — so handing out the id and the index together is what +/// keeps a result from pointing at a description that is not there. +/// +/// Every comment kind is described whether or not the run met one, because the +/// rules a tool reports are also read as the list of what it can find. The +/// rest — a scan diagnostic, a skipped file, a file that could not be read — +/// are described as the run meets them. +struct SarifRules { + entries: Vec, + indices: BTreeMap, +} + +impl SarifRules { + fn new() -> Self { + let mut rules = Self { + entries: Vec::new(), + indices: BTreeMap::new(), + }; + for kind in CommentKind::ALL { + rules.describe( + &format!("removable-{kind}"), + "note", + &format!("Removable {kind} comment"), + &format!( + "A {kind} comment OComment can remove without changing what the file does." + ), + KIND_HELP_URI, + ); + } + rules + } + + /// The index of the rule `id`, describing it first if this run has not + /// reported it before. + fn describe(&mut self, id: &str, level: &str, short: &str, full: &str, help: &str) -> usize { + if let Some(&index) = self.indices.get(id) { + return index; + } + let index = self.entries.len(); + self.entries.push(json!({ + "id": id, + "shortDescription": {"text": short}, + "fullDescription": {"text": full}, + "helpUri": help, + "defaultConfiguration": {"level": level}, + })); + self.indices.insert(id.to_owned(), index); + index + } + + fn kind(&mut self, kind: CommentKind) -> usize { + let id = format!("removable-{kind}"); + *self + .indices + .get(&id) + .expect("every comment kind is described") + } +} + +/// A kebab-cased code read back as the title of a rule: +/// `unterminated-comment` is `Unterminated comment`. +fn sentence_case(code: &str) -> String { + let spelled = code.replace('-', " "); + let mut characters = spelled.chars(); + match characters.next() { + Some(first) => first.to_uppercase().collect::() + characters.as_str(), + None => spelled, + } +} + +fn render_sarif( + output: &mut impl Write, + files: &[ProcessedFile], + skipped: &[SkippedFile], +) -> Result<()> { + let mut rules = SarifRules::new(); let mut results = Vec::new(); for file in files { + let location = artifact_location(&file.path); for comment in file .result .report @@ -227,27 +1360,29 @@ fn render_sarif(files: &[ProcessedFile], skipped: &[SkippedFile]) -> Result<()> { let (line, column) = line_column(&file.source, comment.span.start); let (end_line, end_column) = line_column(&file.source, comment.span.end); - let kind = serde_json::to_value(comment.kind)? - .as_str() - .unwrap_or("comment") - .to_owned(); + let (fix_span, replacement) = fix_for_span(file, comment.span); + let (fix_line, fix_column) = line_column(&file.source, fix_span.start); + let (fix_end_line, fix_end_column) = line_column(&file.source, fix_span.end); + let kind = comment.kind.as_str(); + let index = rules.kind(comment.kind); results.push(json!({ "ruleId": format!("removable-{kind}"), + "ruleIndex": index, "level": "note", - "message": {"text": format!("removable {:?} comment", comment.kind)}, + "message": {"text": removable_label(comment.kind)}, "locations": [{"physicalLocation": { - "artifactLocation": {"uri": file.path.to_string_lossy()}, + "artifactLocation": location.clone(), "region": {"startLine": line, "startColumn": column, "endLine": end_line, "endColumn": end_column} }}], "fixes": [{ "description": {"text": "Remove comment with OComment"}, "artifactChanges": [{ - "artifactLocation": {"uri": file.path.to_string_lossy()}, + "artifactLocation": location.clone(), "replacements": [{"deletedRegion": { - "startLine": line, "startColumn": column, - "endLine": end_line, "endColumn": end_column - }, "insertedContent": {"text": replacement_for_span(file, comment.span)}}] + "startLine": fix_line, "startColumn": fix_column, + "endLine": fix_end_line, "endColumn": fix_end_column + }, "insertedContent": {"text": replacement}}] }] }] })); @@ -260,12 +1395,20 @@ fn render_sarif(files: &[ProcessedFile], skipped: &[SkippedFile]) -> Result<()> ocomment_core::Severity::Warning => "warning", ocomment_core::Severity::Info | ocomment_core::Severity::Hint => "note", }; + let index = rules.describe( + &diagnostic.code, + level, + &sentence_case(&diagnostic.code), + DIAGNOSTIC_DESCRIPTION, + TOOL_INFORMATION_URI, + ); results.push(json!({ "ruleId": diagnostic.code, + "ruleIndex": index, "level": level, "message": {"text": diagnostic.message}, "locations": [{"physicalLocation": { - "artifactLocation": {"uri": file.path.to_string_lossy()}, + "artifactLocation": location.clone(), "region": {"startLine": line, "startColumn": column, "endLine": end_line, "endColumn": end_column} }}] @@ -273,35 +1416,86 @@ fn render_sarif(files: &[ProcessedFile], skipped: &[SkippedFile]) -> Result<()> } } for item in skipped { + let (id, level, short, full) = if item.error { + ( + "io-error", + "error", + "File could not be read", + "A file OComment could not read or write; the message on the result carries the operating-system error.", + ) + } else { + ( + "skipped-file", + "note", + "Skipped file", + "A file OComment did not scan; the message on the result says why it was left alone.", + ) + }; + let index = rules.describe(id, level, short, full, TOOL_INFORMATION_URI); results.push(json!({ - "ruleId": if item.error { "io-error" } else { "skipped-file" }, - "level": if item.error { "error" } else { "note" }, + "ruleId": id, + "ruleIndex": index, + "level": level, "message": {"text": item.reason}, "locations": [{"physicalLocation": { - "artifactLocation": {"uri": item.path.to_string_lossy()} + "artifactLocation": artifact_location(&item.path) }}] })); } let sarif = json!({ "version": "2.1.0", "$schema": "https://json.schemastore.org/sarif-2.1.0.json", - "runs": [{"tool": {"driver": {"name": "ocomment", "informationUri": "https://github.com/P4suta/OComment"}}, "results": results}] + "runs": [{"tool": {"driver": { + "name": "ocomment", + "version": env!("CARGO_PKG_VERSION"), + "informationUri": TOOL_INFORMATION_URI, + "rules": rules.entries + }}, "results": results}] }); - serde_json::to_writer_pretty(io::stdout().lock(), &sarif)?; - println!(); + serde_json::to_writer_pretty(&mut *output, &sarif).map_err(write_error)?; + wrote(writeln!(output))?; Ok(()) } -fn replacement_for_span(file: &ProcessedFile, span: ByteSpan) -> String { +/// The rewrite a removed comment's SARIF fix offers: the bytes it deletes and +/// the bytes that go in their place. +/// +/// A fix is an offer to rewrite the file, so what it deletes has to be what the +/// run would have deleted. Under [`ocomment_core::Layout::Compact`] that is +/// wider than the comment: a comment alone on its line takes the indentation +/// before it and the terminator after it with it, and a fix cut back to the +/// comment's own span would leave behind exactly the blank line that layout +/// exists to close up. So the edit that *contains* the comment is what is +/// reported, rather than one that starts and ends where the comment does. +/// +/// Edits are sorted and non-overlapping and each one spans the comment it +/// removes, so at most one of them can contain a given comment. A file whose +/// report came back invalid has comments but no edits — nothing is rewritten +/// from a source the scanner could not read to the end — and there the +/// comment's own span, with nothing to put in its place, is all there is to +/// offer. +fn fix_for_span(file: &ProcessedFile, span: ByteSpan) -> (ByteSpan, String) { file.result .edits .iter() - .find(|edit| edit.span == span) - .map(|edit| String::from_utf8_lossy(&edit.replacement).into_owned()) - .unwrap_or_default() + .find(|edit| edit.span.start <= span.start && edit.span.end >= span.end) + .map_or_else( + || (span, String::new()), + |edit| { + ( + edit.span, + String::from_utf8_lossy(&edit.replacement).into_owned(), + ) + }, + ) } -fn render_github(files: &[ProcessedFile], skipped: &[SkippedFile]) -> Result<()> { +fn render_github( + output: &mut impl Write, + files: &[ProcessedFile], + skipped: &[SkippedFile], + verbosity: Verbosity, +) -> Result<()> { for file in files { for comment in file .result @@ -311,34 +1505,54 @@ fn render_github(files: &[ProcessedFile], skipped: &[SkippedFile]) -> Result<()> .filter(|comment| comment.disposition.is_remove()) { let (line, column) = line_column(&file.source, comment.span.start); - println!( - "::notice file={},line={line},col={column}::removable {:?} comment", - github_escape(&file.path.to_string_lossy()), - comment.kind - ); + wrote(writeln!( + output, + "::notice file={},line={line},col={column}::{}", + github_escape(&report_uri(&file.path)), + removable_label(comment.kind) + ))?; } for diagnostic in &file.result.report.diagnostics { let (line, column) = line_column(&file.source, diagnostic.span.start); - println!( + wrote(writeln!( + output, "::error file={},line={line},col={column},title={}::{}", - github_escape(&file.path.to_string_lossy()), + github_escape(&report_uri(&file.path)), github_escape(&diagnostic.code), github_escape(&diagnostic.message) - ); + ))?; } } - for item in skipped { - println!( + /* INVARIANT: `-q` trims the human report down to what went wrong, and there is + * no such thing to trim here: an annotation is the *product* of this + * format, not commentary about it, and a hook told to work quietly is + * still owed the notice for the path its caller named and the error for + * the file it could not read. So the visibility rule below is asked at + * `Normal` however quiet the run was, and only `-v` widens it. */ + let visibility = match verbosity { + Verbosity::Quiet => Verbosity::Normal, + loud => loud, + }; + /* NOTE: An annotation costs the reader a line of the checks tab, so a walked + * skip is folded away here exactly as it is in the human report: a run + * over a repository with forty Markdown files in it must not post forty + * notices about them. */ + for item in skipped + .iter() + .filter(|item| skip_is_visible(item, visibility)) + { + wrote(writeln!( + output, "::{} file={},title={}::{}", if item.error { "error" } else { "notice" }, - github_escape(&item.path.to_string_lossy()), + github_escape(&report_uri(&item.path)), if item.error { "OComment I/O error" } else { "OComment skipped file" }, github_escape(&item.reason) - ); + ))?; } Ok(()) } @@ -375,7 +1589,7 @@ pub fn unified_diff(path: &Path, original: &[u8], transformed: &[u8]) -> String output } -fn line_column(source: &[u8], offset: usize) -> (usize, usize) { +pub(crate) fn line_column(source: &[u8], offset: usize) -> (usize, usize) { let offset = offset.min(source.len()); let mut line = 1usize; let mut start = 0usize; @@ -419,3 +1633,350 @@ pub fn invalid(files: &[ProcessedFile]) -> bool { fn _span(_: ByteSpan) -> Value { Value::Null } + +#[cfg(test)] +mod tests { + use super::*; + + /// The frame around a hyperlink target is written in escape bytes, so a + /// name carrying one of its own would close the frame early and be read as + /// terminal instructions from there on. Nor may a URL carry the `%` that + /// makes an encoding an encoding, the space that ends a URL, or the `#` + /// that starts a fragment. + #[test] + fn a_hyperlink_target_encodes_every_byte_a_url_may_not_carry() { + assert_eq!( + percent_encode("/tmp/plain-file_name.rs~"), + "/tmp/plain-file_name.rs~" + ); + assert_eq!(percent_encode("/tmp/a b#c%d.rs"), "/tmp/a%20b%23c%25d.rs"); + assert_eq!( + percent_encode("/tmp/evil\u{1b}[2Jname.rs"), + "/tmp/evil%1B%5B2Jname.rs" + ); + /* NOTE: A path is bytes, and one character is as many `%XX` pairs as it + * takes to spell it. */ + assert_eq!(percent_encode("/tmp/\u{e9}.rs"), "/tmp/%C3%A9.rs"); + } + + /// A name is shown to be typed back, so its own spacing survives; what + /// does not is anything that would drive the terminal or break the row. + #[test] + fn a_sanitized_path_keeps_its_spacing_and_loses_its_controls() { + assert_eq!(sanitize_path(" lead.rs "), " lead.rs "); + assert_eq!(sanitize_path("ta\tb.rs"), "ta\u{fffd}b.rs"); + assert_eq!(sanitize_path("two spaces.rs"), "two spaces.rs"); + assert_eq!(sanitize_path("a\nb\u{1b}c.rs"), "a\u{fffd}b\u{fffd}c.rs"); + } + + /// The reported path is read by a machine that has to find the file again: + /// GitHub matches an annotation by `file=`, and a SARIF reader resolves + /// `artifactLocation.uri` against the checkout. A Windows separator and a + /// `.` segment both name a file no checkout has. + #[test] + fn report_uri_spells_a_path_the_way_a_repository_does() { + assert_eq!(report_uri(Path::new("./a.rs")), "a.rs"); + assert_eq!(report_uri(Path::new("sub/./doc.rs")), "sub/doc.rs"); + assert_eq!(report_uri(Path::new("./sub/./doc.rs")), "sub/doc.rs"); + assert_eq!(report_uri(Path::new(r"sub\doc.rs")), "sub/doc.rs"); + assert_eq!(report_uri(Path::new(r".\sub\.\doc.rs")), "sub/doc.rs"); + /* NOTE: A path that leaves the tree, an absolute one, and standard input are + * all left as they are; only the separators are normalised. */ + assert_eq!(report_uri(Path::new("../sibling/a.rs")), "../sibling/a.rs"); + assert_eq!(report_uri(Path::new("/tmp/a.rs")), "/tmp/a.rs"); + assert_eq!(report_uri(Path::new(r"C:\src\a.rs")), "C:/src/a.rs"); + assert_eq!(report_uri(Path::new(STDIN_PATH)), STDIN_PATH); + // NOTE: Naming the working directory as nothing at all would be worse. + assert_eq!(report_uri(Path::new(".")), "."); + } + + /// `%SRCROOT%` says the path is measured from the root of the checkout, so + /// it is claimed only for the paths that are. + #[test] + fn only_a_path_inside_the_tree_is_reported_against_the_source_root() { + for inside in ["a.rs", "sub/doc.rs", "./sub/doc.rs"] { + assert_eq!( + artifact_location(Path::new(inside))["uriBaseId"], + json!(SRCROOT), + "`{inside}` is not reported against the source root" + ); + } + for outside in ["../sibling/a.rs", "/tmp/a.rs", STDIN_PATH] { + let location = artifact_location(Path::new(outside)); + assert_eq!( + location.get("uriBaseId"), + None, + "`{outside}` claims to be under the source root" + ); + } + } + + /// A relative reference whose first segment holds a colon is read as a + /// scheme, so a checkout that really does hold a directory named `c:` says + /// so with the one `.` segment a URI keeps for the purpose. Nothing else + /// gains one, and a path that is under no base is left exactly as it was. + #[test] + fn a_first_segment_that_reads_as_a_drive_letter_is_disambiguated() { + let location = artifact_location(Path::new("c:/a.rs")); + assert_eq!(location["uri"], json!("./c:/a.rs")); + assert_eq!(location["uriBaseId"], json!(SRCROOT)); + assert_eq!(artifact_location(Path::new("c:"))["uri"], json!("./c:")); + for plain in ["a.rs", "sub/doc.rs", "cc:/a.rs", "sub/c:/a.rs"] { + assert_eq!( + artifact_location(Path::new(plain))["uri"], + json!(report_uri(Path::new(plain))), + "`{plain}` was disambiguated and had no need of it" + ); + } + assert_eq!( + artifact_location(Path::new("/tmp/c:/a.rs"))["uri"], + json!("/tmp/c:/a.rs"), + "a path under no base was rewritten" + ); + } + + /// Every result points into the rules by index, so the two orders have to + /// be the same one. + #[test] + fn a_rule_is_described_once_and_keeps_its_index() { + let mut rules = SarifRules::new(); + assert_eq!(rules.entries.len(), CommentKind::ALL.len()); + assert_eq!(rules.kind(CommentKind::Line), 0); + let first = rules.describe("io-error", "error", "short", "full", TOOL_INFORMATION_URI); + assert_eq!(first, CommentKind::ALL.len()); + let again = rules.describe("io-error", "note", "other", "other", TOOL_INFORMATION_URI); + assert_eq!(first, again, "a second sighting described the rule twice"); + assert_eq!( + rules.entries[first]["defaultConfiguration"]["level"], + "error" + ); + assert_eq!(rules.entries.len(), CommentKind::ALL.len() + 1); + } + + #[test] + fn a_diagnostic_code_reads_back_as_a_title() { + assert_eq!( + sentence_case("unterminated-comment"), + "Unterminated comment" + ); + assert_eq!(sentence_case("nesting-limit"), "Nesting limit"); + assert_eq!(sentence_case(""), ""); + } + + fn preview_of(source: &[u8], max_columns: usize) -> String { + preview(source, ByteSpan::new(0, source.len()), max_columns) + } + + #[test] + fn preview_collapses_every_run_of_whitespace_and_trims() { + assert_eq!( + preview_of(b" /*\r\n\tkeep\t this tidy \x0c*/ ", 72), + "/* keep this tidy */" + ); + } + + #[test] + fn preview_truncates_on_display_width_without_splitting_a_wide_character() { + let source = "ab漢字漢字漢字ab".as_bytes(); + assert_eq!(preview_of(source, 20), "ab漢字漢字漢字ab"); + let cut = preview_of(source, 10); + assert_eq!(cut, "ab漢字漢…"); + assert!(cut.ends_with('…'), "truncation is unmarked: {cut}"); + let width: usize = cut + .chars() + .map(|ch| unicode_width::UnicodeWidthChar::width(ch).unwrap_or(0)) + .sum(); + assert!(width <= 10, "`{cut}` is {width} columns wide"); + } + + #[test] + fn preview_replaces_control_characters_with_the_replacement_character() { + let source = b"// \x1b[31m\x07 \xc2\x9b\x7f bell"; + let rendered = preview_of(source, 72); + assert_eq!(rendered, "// \u{fffd}[31m\u{fffd} \u{fffd}\u{fffd} bell"); + assert!( + !rendered.contains('\x1b'), + "an escape sequence survived: {rendered:?}" + ); + } + + #[test] + fn preview_replaces_invalid_utf8_bytes() { + assert_eq!( + preview_of(b"// \xff\xfe end", 72), + "// \u{fffd}\u{fffd} end" + ); + } + + /// Bidi overrides and isolates can make a comment render as its own + /// reverse, and the line/paragraph separators break the one-line promise. + #[test] + fn preview_replaces_bidirectional_and_separator_controls() { + let source = "// \u{202e}reverse\u{202c} \u{200e}\u{200f} \u{2066}iso\u{2069} \ + \u{2028}\u{2029} \u{61c}\u{feff} end"; + assert_eq!( + preview_of(source.as_bytes(), 72), + "// \u{fffd}reverse\u{fffd} \u{fffd}\u{fffd} \u{fffd}iso\u{fffd} \ + \u{fffd}\u{fffd} \u{fffd}\u{fffd} end" + ); + for character in [ + '\u{61c}', '\u{200e}', '\u{200f}', '\u{202a}', '\u{202b}', '\u{202c}', '\u{202d}', + '\u{202e}', '\u{2066}', '\u{2067}', '\u{2068}', '\u{2069}', '\u{2028}', '\u{2029}', + '\u{feff}', + ] { + assert!( + is_control(character), + "U+{:04X} still reaches the terminal", + character as u32 + ); + } + } + + /// Zero-width characters cost no display columns, so the width budget alone + /// cannot bound the line; a hard character cap must. + #[test] + fn preview_caps_the_character_count_of_a_zero_width_run() { + let source = format!("a{}", "\u{301}".repeat(1000)); + let rendered = preview_of(source.as_bytes(), 8); + assert!( + rendered.chars().count() <= 8 * 4, + "preview is {} characters wide", + rendered.chars().count() + ); + assert!(rendered.ends_with('\u{2026}'), "truncation is unmarked"); + } + + /// A hunk is read as code, so the indentation that says what a line belongs + /// to survives — but nothing that drives the terminal does, because the + /// prompt asking about that line sits directly underneath it. + #[test] + fn a_source_line_keeps_its_shape_and_loses_its_control_characters() { + assert_eq!( + sanitize_source_line(" let x = 1; // note"), + " let x = 1; // note", + "the indentation of a shown line was collapsed" + ); + assert_eq!( + sanitize_source_line("\tif (x) {"), + " if (x) {", + "a tab did not reach its eight-column stop" + ); + assert_eq!( + sanitize_source_line("a\u{1b}[2Jb\u{202e}c"), + "a\u{fffd}[2Jb\u{fffd}c", + "an escape sequence reached the terminal verbatim" + ); + let capped = sanitize_source_line(&"v".repeat(PREVIEW_COLUMNS * 3)); + let width: usize = capped + .chars() + .map(|ch| unicode_width::UnicodeWidthChar::width(ch).unwrap_or(0)) + .sum(); + assert!( + width <= PREVIEW_COLUMNS, + "a shown line ran to {width} columns and pushed the question off the screen" + ); + } + + /// The interactive verdict counts answers, and every noun agrees with the + /// number in front of it. It closes on the same `(N files scanned)` the + /// plain `fix` summary ends with: the reader still has to be told how much + /// was looked at to reach the answers. + #[test] + fn the_interactive_summary_pluralizes_both_of_its_nouns() { + assert_eq!( + interactive_summary(InteractiveOutcome { + removed: 1, + reviewed: 1, + offered: 1, + changed: 1, + scanned: 1, + }), + "Removed 1 of 1 comment in 1 file (1 file scanned)." + ); + assert_eq!( + interactive_summary(InteractiveOutcome { + removed: 2, + reviewed: 5, + offered: 5, + changed: 3, + scanned: 4, + }), + "Removed 2 of 5 comments in 3 files (4 files scanned)." + ); + } + + /// A run that was never asked a question says so in the vocabulary the + /// plain `fix` summary uses for the same answer, and counts the files it + /// scanned — `Removed 0 of 0 comments in 0 files` named three numbers, none + /// of which was the one the reader wanted. + #[test] + fn an_interactive_run_with_nothing_to_offer_borrows_the_fix_wording() { + assert_eq!( + interactive_summary(InteractiveOutcome { + scanned: 3, + ..InteractiveOutcome::default() + }), + "Nothing to fix in 3 files." + ); + assert_eq!( + interactive_summary(InteractiveOutcome { + scanned: 1, + ..InteractiveOutcome::default() + }), + "Nothing to fix in 1 file." + ); + } + + /// `q` stops the questions, so the verdict counts the ones that were + /// answered and says how many were left unasked. Reporting `1 of 9` to a + /// reader who answered twice would read as seven refusals. + #[test] + fn a_stopped_interactive_run_counts_the_questions_it_asked() { + assert_eq!( + interactive_summary(InteractiveOutcome { + removed: 1, + reviewed: 2, + offered: 9, + changed: 1, + scanned: 4, + }), + "Removed 1 of 2 comments in 1 file (7 comments not reviewed) (4 files scanned)." + ); + assert_eq!( + interactive_summary(InteractiveOutcome { + removed: 0, + reviewed: 1, + offered: 2, + changed: 0, + scanned: 1, + }), + "Removed 0 of 1 comment in 0 files (1 comment not reviewed) (1 file scanned)." + ); + } + + /// What a probed tool says about itself gets the preview's treatment: one + /// line, no control sequences, and no more of it than a preview shows. + #[test] + fn sanitize_line_replaces_controls_and_caps_the_width() { + assert_eq!( + sanitize_line("\u{1b}[2J\u{1b}[1;31mv1.0\tPWNED\u{1b}[0m"), + "\u{fffd}[2J\u{fffd}[1;31mv1.0 PWNED\u{fffd}[0m" + ); + let capped = sanitize_line(&"v".repeat(PREVIEW_COLUMNS * 3)); + let width: usize = capped + .chars() + .map(|ch| unicode_width::UnicodeWidthChar::width(ch).unwrap_or(0)) + .sum(); + assert!( + width <= PREVIEW_COLUMNS, + "`{capped}` is {width} columns wide" + ); + assert!(capped.ends_with('\u{2026}'), "truncation is unmarked"); + } + + #[test] + fn preview_reads_only_the_span() { + let source = b"let x = 1; // TODO remove\n"; + assert_eq!(preview(source, ByteSpan::new(11, 25), 72), "// TODO remove"); + } +} diff --git a/rust/ocomment/src/plugin.rs b/rust/ocomment/src/plugin.rs index 76f8a44..63e73bd 100644 --- a/rust/ocomment/src/plugin.rs +++ b/rust/ocomment/src/plugin.rs @@ -1,4 +1,4 @@ -use crate::config::PluginsConfig; +use crate::{config::PluginsConfig, output::wrote}; use anyhow::{Context, Result, anyhow, bail, ensure}; use ocomment_core::{ ByteSpan, CommentKind, Language, TransformOptions, TransformResult, transform_spans, @@ -519,16 +519,26 @@ fn locked_artifact_candidate(root: &Path, artifact: &str) -> Result { } pub fn add( + output: &mut impl Write, root: &Path, source: &str, requested_name: Option<&str>, expected: Option<&str>, identity: Option<&str>, ) -> Result<()> { - install(root, source, requested_name, expected, identity, true) + install( + output, + root, + source, + requested_name, + expected, + identity, + true, + ) } fn install( + output: &mut impl Write, root: &Path, source: &str, requested_name: Option<&str>, @@ -599,11 +609,11 @@ fn install( }, ); save_lock(root, &lock)?; - println!("added plugin {name}"); + wrote(writeln!(output, "added plugin {name}"))?; Ok(()) } -pub fn remove(root: &Path, name: &str) -> Result<()> { +pub fn remove(output: &mut impl Write, root: &Path, name: &str) -> Result<()> { let mut lock = load_lock(root)?; let removed = lock .plugins @@ -623,25 +633,26 @@ pub fn remove(root: &Path, name: &str) -> Result<()> { } } save_lock(root, &lock)?; - println!("removed plugin {name}"); + wrote(writeln!(output, "removed plugin {name}"))?; Ok(()) } -pub fn list(root: &Path) -> Result<()> { +pub fn list(output: &mut impl Write, root: &Path) -> Result<()> { let lock = load_lock(root)?; if lock.plugins.is_empty() { - println!("no plugins locked"); + wrote(writeln!(output, "no plugins locked"))?; } for (name, plugin) in lock.plugins { - println!( + wrote(writeln!( + output, "{name}\t{}\tsha256:{}\tAPI {}", plugin.version, plugin.sha256, plugin.api - ); + ))?; } Ok(()) } -pub fn verify(root: &Path, selected: Option<&str>) -> Result<()> { +pub fn verify(output: &mut impl Write, root: &Path, selected: Option<&str>) -> Result<()> { let lock = load_lock(root)?; for (name, plugin) in lock .plugins @@ -657,7 +668,11 @@ pub fn verify(root: &Path, selected: Option<&str>) -> Result<()> { if actual != plugin.sha256 { bail!("plugin `{name}` digest mismatch"); } - println!("plugin {name}: verified sha256:{}", plugin.sha256); + wrote(writeln!( + output, + "plugin {name}: verified sha256:{}", + plugin.sha256 + ))?; } if let Some(name) = selected && !lock.plugins.contains_key(name) @@ -665,12 +680,12 @@ pub fn verify(root: &Path, selected: Option<&str>) -> Result<()> { bail!("plugin `{name}` is not locked"); } if lock.plugins.is_empty() { - println!("plugins: none (offline lock is valid)"); + wrote(writeln!(output, "plugins: none (offline lock is valid)"))?; } Ok(()) } -pub fn update(root: &Path, selected: Option<&str>) -> Result<()> { +pub fn update(output: &mut impl Write, root: &Path, selected: Option<&str>) -> Result<()> { let lock = load_lock(root)?; let entries: Vec<_> = lock .plugins @@ -683,11 +698,12 @@ pub fn update(root: &Path, selected: Option<&str>) -> Result<()> { } for (name, plugin) in entries { if !is_remote_source(&plugin.source) { - add(root, &plugin.source, Some(&name), None, None)?; + add(output, root, &plugin.source, Some(&name), None, None)?; } else { - // The existing signature identity authorizes a freshly fetched - // artifact. Its new digest is then written to the lockfile. + /* INVARIANT: The existing signature identity authorizes a freshly fetched + * artifact. Its new digest is then written to the lockfile. */ install( + output, root, &plugin.source, Some(&name), @@ -704,9 +720,14 @@ fn is_remote_source(source: &str) -> bool { source.starts_with("https://") || source.starts_with("gh:") || source.starts_with("oci:") } -pub fn new_plugin(path: &Path) -> Result<()> { - fs::create_dir(path) - .with_context(|| format!("refusing to overwrite plugin directory {}", path.display()))?; +pub fn new_plugin(output: &mut impl Write, path: &Path) -> Result<()> { + fs::create_dir(path).with_context(|| { + format!( + "refusing to overwrite plugin directory {}; remove it or run \ + `ocomment plugin remove ` first", + path.display() + ) + })?; fs::create_dir(path.join("src"))?; fs::write( path.join("ocomment-scanner.wit"), @@ -790,10 +811,33 @@ The host provides no WASI, filesystem, network, clock, or random imports. Keep t self-contained and return sorted, non-overlapping, non-empty byte spans. "#, )?; - println!("created plugin scaffold {}", path.display()); + wrote(writeln!( + output, + "created plugin scaffold {}", + path.display() + ))?; Ok(()) } +/// What each external tool is needed for, in the words the command line uses. +/// One constant per purpose keeps the four spawn sites and `doctor` naming the +/// same thing: the line `doctor` prints for a tool that is missing has to be +/// the line the failure would have printed once something needed it. +pub const HTTPS_SOURCES: &str = "https:// plugin sources"; +pub const GH_SOURCES: &str = "gh: plugin sources"; +pub const OCI_SOURCES: &str = "oci: plugin sources"; +pub const SIGNATURE_VERIFICATION: &str = "--identity verification"; + +/// Why a tool OComment shells out to could not be started. +/// +/// The operating system says only "No such file or directory", which names +/// neither the missing binary nor the part of the run that wanted it. This +/// says both, and sends the reader to the command that reports every tool at +/// once instead of making them rediscover the next gap one failure at a time. +fn missing_tool(tool: &str, purpose: &str) -> String { + format!("cannot run `{tool}` (needed for {purpose}); run `ocomment doctor`") +} + fn fetch_remote(source: &str, directory: &Path) -> Result { let temporary = TemporaryPath::new(directory, ".wasm")?; if source.starts_with("https://") { @@ -808,7 +852,7 @@ fn fetch_remote(source: &str, directory: &Path) -> Result { .arg(temporary.path()) .arg(source) .status() - .context("cannot launch HTTPS plugin retrieval")?; + .with_context(|| missing_tool("curl", HTTPS_SOURCES))?; ensure!(status.success(), "plugin retrieval failed with {status}"); } else if let Some(spec) = source.strip_prefix("gh:") { let (repository, tag, asset) = parse_github_source(spec)?; @@ -825,7 +869,7 @@ fn fetch_remote(source: &str, directory: &Path) -> Result { ]) .arg(temporary.path()) .status() - .context("cannot launch GitHub plugin retrieval")?; + .with_context(|| missing_tool("gh", GH_SOURCES))?; ensure!(status.success(), "plugin retrieval failed with {status}"); } else { let specification = source.strip_prefix("oci:").expect("remote kind checked"); @@ -840,7 +884,7 @@ fn fetch_remote(source: &str, directory: &Path) -> Result { .args(["pull", reference, "--output"]) .arg(pulled.path()) .status() - .context("cannot launch OCI plugin retrieval")?; + .with_context(|| missing_tool("oras", OCI_SOURCES))?; ensure!(status.success(), "plugin retrieval failed with {status}"); let artifact = if let Some(relative) = artifact_path { let relative = Path::new(relative); @@ -893,7 +937,7 @@ fn verify_sigstore(source: &str, artifact: &Path, identity: &str, directory: &Pa ]) .arg(reference) .status() - .context("cannot launch cosign")?; + .with_context(|| missing_tool("cosign", SIGNATURE_VERIFICATION))?; ensure!(status.success(), "Sigstore verification failed"); return Ok(()); } @@ -913,7 +957,7 @@ fn verify_sigstore(source: &str, artifact: &Path, identity: &str, directory: &Pa ]) .arg(bundle.path()) .status() - .context("cannot retrieve GitHub Sigstore bundle")? + .with_context(|| missing_tool("gh", GH_SOURCES))? } else { let bundle_url = format!("{source}.sigstore.json"); Command::new("curl") @@ -927,7 +971,7 @@ fn verify_sigstore(source: &str, artifact: &Path, identity: &str, directory: &Pa .arg(bundle.path()) .arg(&bundle_url) .status() - .with_context(|| format!("cannot retrieve Sigstore bundle {bundle_url}"))? + .with_context(|| missing_tool("curl", HTTPS_SOURCES))? }; ensure!(download.success(), "cannot retrieve Sigstore bundle"); let status = Command::new("cosign") @@ -941,7 +985,7 @@ fn verify_sigstore(source: &str, artifact: &Path, identity: &str, directory: &Pa ]) .arg(artifact) .status() - .context("cannot launch cosign")?; + .with_context(|| missing_tool("cosign", SIGNATURE_VERIFICATION))?; if !status.success() { bail!("Sigstore verification failed"); } @@ -1238,6 +1282,37 @@ mod tests { assert!(result.report.comments.is_empty()); } + /// A tool that is not installed is the most common way a plugin command + /// fails, and the shell's "No such file or directory" names neither the + /// binary nor the reason this run wanted it. Every spawn site says both, + /// and points at the one command that reports the whole environment. + /// + /// `cosign` runs only after an artifact has already been fetched and + /// validated, which no offline test can arrange, so its wording is pinned + /// here rather than through the command line. + #[test] + fn a_missing_tool_names_itself_its_purpose_and_doctor() { + for (tool, purpose) in [ + ("curl", HTTPS_SOURCES), + ("gh", GH_SOURCES), + ("oras", OCI_SOURCES), + ("cosign", SIGNATURE_VERIFICATION), + ] { + assert_eq!( + missing_tool(tool, purpose), + format!("cannot run `{tool}` (needed for {purpose}); run `ocomment doctor`") + ); + } + assert_eq!( + missing_tool("gh", GH_SOURCES), + "cannot run `gh` (needed for gh: plugin sources); run `ocomment doctor`" + ); + assert_eq!( + missing_tool("cosign", SIGNATURE_VERIFICATION), + "cannot run `cosign` (needed for --identity verification); run `ocomment doctor`" + ); + } + #[test] fn rejects_core_modules_and_component_imports() { let module = wat::parse_str("(module)").unwrap(); diff --git a/rust/ocomment/src/values.rs b/rust/ocomment/src/values.rs new file mode 100644 index 0000000..544cbe9 --- /dev/null +++ b/rust/ocomment/src/values.rs @@ -0,0 +1,259 @@ +//! `clap::ValueEnum` wrappers around the core vocabulary enums. +//! +//! The orphan rule forbids implementing `clap::ValueEnum` on the types owned by +//! `ocomment-core`, so every user-facing enum gets a transparent newtype here. +//! Names, aliases, and the variant list all come from the core enum, which stays +//! the single source of truth; this module only adds the per-value help clap +//! needs for `--help`, error messages, and shell completions. + +use clap::{ValueEnum, builder::PossibleValue}; +use ocomment_core::{CommentKind, Dialect, Language, Layout, Policy}; +use std::ops::Deref; + +/// Register the canonical spelling plus every accepted alias. +/// +/// `clap` matches a command-line value through `PossibleValue::matches`, so an +/// alias that is not registered here is not accepted, however well the core +/// `FromStr` understands it. Core aliases are stored with `-` as the separator; +/// the `_` spelling is registered too so both keep working. +fn possible_value( + name: &'static str, + aliases: &'static [&'static str], + help: &'static str, +) -> PossibleValue { + let mut value = PossibleValue::new(name).help(help); + for alias in aliases { + value = value.alias(*alias); + } + for spelling in std::iter::once(&name).chain(aliases) { + if spelling.contains('-') { + value = value.alias(spelling.replace('-', "_")); + } + } + value +} + +macro_rules! value_enum_wrapper { + ($name:ident, $inner:ty, $help:expr) => { + #[doc = concat!("A `clap::ValueEnum` view of [`", stringify!($inner), "`].")] + #[derive(Clone, Copy, Debug, Eq, PartialEq)] + pub struct $name(pub $inner); + + impl From<$name> for $inner { + fn from(value: $name) -> Self { + value.0 + } + } + + impl From<$inner> for $name { + fn from(value: $inner) -> Self { + Self(value) + } + } + + impl Deref for $name { + type Target = $inner; + fn deref(&self) -> &Self::Target { + &self.0 + } + } + + impl ValueEnum for $name { + fn value_variants<'a>() -> &'a [Self] { + static VARIANTS: [$name; <$inner>::ALL.len()] = { + let mut variants = [$name(<$inner>::ALL[0]); <$inner>::ALL.len()]; + let mut index = 0; + while index < variants.len() { + variants[index] = $name(<$inner>::ALL[index]); + index += 1; + } + variants + }; + &VARIANTS + } + + fn to_possible_value(&self) -> Option { + let help: fn($inner) -> &'static str = $help; + Some(possible_value( + self.0.as_str(), + self.0.aliases(), + help(self.0), + )) + } + } + }; +} + +value_enum_wrapper!(PolicyArg, Policy, |value| match value { + Policy::Safe => "Remove ordinary and doc comments; keep preambles and directives", + Policy::Legal => "Like safe, and keep licence and copyright comments as well", + Policy::All => "Remove every comment that no keep override protects", +}); + +value_enum_wrapper!(LayoutArg, Layout, |value| match value { + Layout::Lines => "Keep the line structure and separate tokens that would otherwise join", + Layout::Columns => "Pad each removed comment so the following columns do not shift", + Layout::Compact => + "Drop lines that held only a removed comment, and the whitespace it left behind", +}); + +/* NOTE: The CLI is deliberately stricter than the core `FromStr`, which folds case, + * dashes, and underscores away before it looks a name up: only the canonical + * spelling, the pinned aliases, and their underscore variants are registered + * here, so `--language r-u-s-t` stays an error even though the core accepts it. */ +value_enum_wrapper!(LanguageArg, Language, |value| match value { + Language::Rust => "Rust source files", + Language::Ocaml => "OCaml implementation and interface files", + Language::C => "C source and header files", + Language::Cpp => "C++ source and header files", + Language::Go => "Go source files", + Language::Java => "Java source files, including Unicode escape translation", + Language::JavaScript => "JavaScript modules and scripts, including JSX", + Language::TypeScript => "TypeScript modules and scripts, including TSX", + Language::Python => "Python source and stub files", + Language::Shell => "POSIX sh, Bash, and zsh scripts", + Language::Html => "HTML documents, including nested script and style elements", + Language::Css => "CSS stylesheets", + Language::Jsonc => "JSON with comments, including JSON5", + Language::Sql => "SQL for every supported database dialect", + Language::Kotlin => "Kotlin source and script files", + Language::Toml => "TOML documents, including the lock files written in it", + Language::Lua => "Lua chunks and LuaRocks rockspecs", + Language::Yaml => "YAML documents, including the tool configurations written in it", + Language::Php => "PHP scripts and templates; the inline HTML around the tags is content", + Language::Ruby => "Ruby scripts, gem manifests, and the project files named after their tool", + Language::Zig => "Zig source files and Zig Object Notation data", + Language::R => "R scripts and the `.Rprofile` an R session sources at start-up", + Language::Dart => "Dart source files, whose block comments nest", + Language::Swift => + "Swift source files, whose block comments nest and whose `#/../#` is a regex", + Language::CSharp => "C# source and script files, whose `#` lines are preprocessor directives", + Language::Scala => + "Scala source and script files, whose block comments nest and whose XML literals are opaque", + Language::Vue => "Vue single-file components, whose templates are HTML with `{{ ... }}` code", + Language::Svelte => "Svelte components, whose templates are HTML with `{ ... }` code", + Language::Markdown => + "Markdown documents, whose fenced code blocks are scanned as their named languages", + Language::Perl => "Perl scripts and modules, whose quote words and regexes hide a `#`", + Language::Unknown => "An undetected language", +}); + +value_enum_wrapper!(DialectArg, Dialect, |value| match value { + Dialect::Standard => "The default lexical rules of the language", + Dialect::Jsx => "JavaScript with JSX elements", + Dialect::Tsx => "TypeScript with JSX elements", + Dialect::ObjectiveC => "Objective-C extensions to C", + Dialect::ObjectiveCpp => "Objective-C++ extensions to C++", + Dialect::GnuC => "GNU extensions to C", + Dialect::GnuCpp => "GNU extensions to C++", + Dialect::Cuda => "CUDA extensions to C++", + Dialect::PosixSh => "The POSIX shell command language", + Dialect::Bash53 => "Bash 5.3", + Dialect::Zsh => "The Z shell", + Dialect::PostgreSql => "PostgreSQL, with dollar-quoted bodies", + Dialect::MySql => "MySQL, including its executable versioned comments", + Dialect::Sqlite => "SQLite", + Dialect::TSql => "Microsoft Transact-SQL", + Dialect::Oracle => "Oracle SQL and PL/SQL", + Dialect::Scss => "SCSS and the indented Sass syntax", +}); + +value_enum_wrapper!(CommentKindArg, CommentKind, |value| match value { + CommentKind::Line => "An ordinary comment running to the end of the line", + CommentKind::Block => "An ordinary delimited comment", + CommentKind::DocLine => "A documentation comment running to the end of the line", + CommentKind::DocBlock => "A delimited documentation comment", + CommentKind::Directive => "A tool or language directive such as a pragma or lint control", + CommentKind::License => "A licence or copyright preamble", + CommentKind::HtmlComment => "A DOM-observable HTML comment", + CommentKind::Shebang => "The interpreter line starting an executable script", + CommentKind::Encoding => "A source encoding declaration", + CommentKind::OptimizerHint => "A compiler or database optimizer hint", + CommentKind::VersionComment => "A MySQL versioned comment that the server executes", +}); + +#[cfg(test)] +mod tests { + use super::*; + + /// Every spelling the core enum accepts must reach clap, which matches only + /// through the registered name and aliases. + fn round_trip(canonical: &'static str, aliases: &'static [&'static str]) + where + T: ValueEnum + Copy + PartialEq + std::fmt::Debug, + { + let expected = T::from_str(canonical, false) + .unwrap_or_else(|_| panic!("clap rejects the canonical name `{canonical}`")); + for spelling in std::iter::once(&canonical).chain(aliases) { + for candidate in [ + (*spelling).to_owned(), + spelling.to_ascii_uppercase(), + spelling.replace('-', "_"), + ] { + let parsed = T::from_str(&candidate, true) + .unwrap_or_else(|_| panic!("clap rejects `{candidate}`")); + assert_eq!( + parsed, expected, + "`{candidate}` resolved to the wrong value" + ); + } + } + } + + #[test] + fn every_core_spelling_reaches_clap() { + for value in Policy::ALL { + round_trip::(value.as_str(), value.aliases()); + } + for value in Layout::ALL { + round_trip::(value.as_str(), value.aliases()); + } + for value in Language::ALL { + round_trip::(value.as_str(), value.aliases()); + } + for value in Dialect::ALL { + round_trip::(value.as_str(), value.aliases()); + } + for value in CommentKind::ALL { + round_trip::(value.as_str(), value.aliases()); + } + } + + #[test] + fn every_value_carries_help_and_the_core_variant_order() { + assert_eq!( + LanguageArg::value_variants() + .iter() + .map(|value| value.0) + .collect::>(), + Language::ALL.to_vec() + ); + for value in DialectArg::value_variants() { + let possible = value + .to_possible_value() + .expect("dialects are never hidden"); + assert_eq!(possible.get_name(), value.0.as_str()); + assert!( + possible.get_help().is_some(), + "`{}` has no help text", + value.0 + ); + } + } + + #[test] + fn punctuated_aliases_survive() { + assert_eq!( + LanguageArg::from_str("c++", false), + Ok(LanguageArg(Language::Cpp)) + ); + assert_eq!( + DialectArg::from_str("objective-c++", false), + Ok(DialectArg(Dialect::ObjectiveCpp)) + ); + assert_eq!( + DialectArg::from_str("bash-5.3", false), + Ok(DialectArg(Dialect::Bash53)) + ); + } +} diff --git a/rust/ocomment/tests/cli.rs b/rust/ocomment/tests/cli.rs index 18acd93..f9b3c8a 100644 --- a/rust/ocomment/tests/cli.rs +++ b/rust/ocomment/tests/cli.rs @@ -1,7 +1,9 @@ use std::{ + collections::BTreeSet, fs, + io::{Read, Write}, path::Path, - process::{Command, Output}, + process::{Command, ExitStatus, Output, Stdio}, }; use tempfile::TempDir; @@ -18,6 +20,26 @@ fn run(directory: &Path, arguments: &[&str]) -> Output { .unwrap() } +/// Run the binary with `input` piped to its standard input. +fn run_stdin(directory: &Path, arguments: &[&str], input: &[u8]) -> Output { + let mut child = Command::new(binary()) + .current_dir(directory) + .env("PATH", "/usr/bin:/bin") + .args(arguments) + .stdin(Stdio::piped()) + .stdout(Stdio::piped()) + .stderr(Stdio::piped()) + .spawn() + .unwrap(); + child + .stdin + .take() + .expect("standard input was piped") + .write_all(input) + .unwrap(); + child.wait_with_output().unwrap() +} + fn git(directory: &Path, arguments: &[&str]) -> Vec { let output = Command::new("/usr/bin/git") .current_dir(directory) @@ -50,6 +72,12 @@ fn git_with_path(directory: &Path, arguments: &[&str], path: &std::ffi::OsStr) - output.stdout } +/// What a file OComment has no built-in scanner for is skipped with. The +/// sentence is pinned literally by `an_unknown_language_skip_says_how_to_force_one`; +/// every other test names it through this constant. +const NO_LANGUAGE: &str = + "no built-in language for this file (see `ocomment languages`; use --language to force)"; + fn repository() -> TempDir { let directory = tempfile::tempdir().unwrap(); git(directory.path(), &["init", "-q"]); @@ -96,6 +124,464 @@ fn check_diff_and_fix_follow_the_exit_contract() { ); } +/* NOTE: A lock file carries no extension the detector can use, so the whole + * name has to reach it through the binary for the run to scan the file at all. + * `Cargo.lock` is the one every Rust checkout has. */ +#[test] +fn a_toml_lock_file_is_scanned_under_its_reserved_name() { + let directory = tempfile::tempdir().unwrap(); + let path = directory.path().join("Cargo.lock"); + fs::write(&path, b"# generated\nname = \"# opaque\" # remove\n").unwrap(); + + let scanned = run( + directory.path(), + &["scan", "Cargo.lock", "--format", "json"], + ); + assert_eq!( + scanned.status.code(), + Some(0), + "{}", + String::from_utf8_lossy(&scanned.stderr) + ); + let document: serde_json::Value = serde_json::from_slice(&scanned.stdout).unwrap(); + let report = &document["files"][0]["report"]; + assert_eq!(report["language"], "toml"); + assert_eq!(report["comments"].as_array().unwrap().len(), 2); + + let fixed = run(directory.path(), &["fix", "Cargo.lock"]); + assert_eq!( + fixed.status.code(), + Some(0), + "{}", + String::from_utf8_lossy(&fixed.stderr) + ); + assert_eq!(fs::read(&path).unwrap(), b"\nname = \"# opaque\" \n"); +} + +/* NOTE: A `.clang-format` file is YAML with no extension for the detector to go + * on and a hidden name besides, so naming it is what gets it scanned at all: + * the whole name reaches the detector, and an explicitly named path lifts the + * hidden-file rule the walk applies on its own. */ +#[test] +fn a_yaml_configuration_is_scanned_under_its_reserved_name() { + let directory = tempfile::tempdir().unwrap(); + let path = directory.path().join(".clang-format"); + fs::write( + &path, + b"# yamllint disable-line rule:line-length\nColumnLimit: 100 # remove\n", + ) + .unwrap(); + + let scanned = run( + directory.path(), + &["scan", ".clang-format", "--format", "json"], + ); + assert_eq!( + scanned.status.code(), + Some(0), + "{}", + String::from_utf8_lossy(&scanned.stderr) + ); + let document: serde_json::Value = serde_json::from_slice(&scanned.stdout).unwrap(); + let report = &document["files"][0]["report"]; + assert_eq!(report["language"], "yaml"); + assert_eq!(report["comments"].as_array().unwrap().len(), 2); + + let fixed = run(directory.path(), &["fix", ".clang-format"]); + assert_eq!( + fixed.status.code(), + Some(0), + "{}", + String::from_utf8_lossy(&fixed.stderr) + ); + assert_eq!( + fs::read(&path).unwrap(), + b"# yamllint disable-line rule:line-length\nColumnLimit: 100 \n" + ); +} + +/* NOTE: A Lua script installed as a command carries no extension at all, so the + * `#!` line is the only evidence the run has; this is the path from the file + * name through the detector and out the other side as a Lua scan. */ +#[test] +fn a_lua_script_is_scanned_from_its_shebang_alone() { + let directory = tempfile::tempdir().unwrap(); + let path = directory.path().join("hook"); + fs::write( + &path, + b"#!/usr/bin/env lua\nprint(\"-- opaque\") -- remove\n", + ) + .unwrap(); + + let scanned = run(directory.path(), &["scan", "hook", "--format", "json"]); + assert_eq!( + scanned.status.code(), + Some(0), + "{}", + String::from_utf8_lossy(&scanned.stderr) + ); + let document: serde_json::Value = serde_json::from_slice(&scanned.stdout).unwrap(); + let report = &document["files"][0]["report"]; + assert_eq!(report["language"], "lua"); + assert_eq!(report["comments"].as_array().unwrap().len(), 2); + + let fixed = run(directory.path(), &["fix", "hook"]); + assert_eq!( + fixed.status.code(), + Some(0), + "{}", + String::from_utf8_lossy(&fixed.stderr) + ); + assert_eq!( + fs::read(&path).unwrap(), + b"#!/usr/bin/env lua\nprint(\"-- opaque\") \n" + ); +} + +/* NOTE: A PHP template is two languages in one file and only the code half is + * scanned: the inline HTML around the tags is content, so the `` comment + * in it survives a run that removes the `//` comment inside them. */ +#[test] +fn a_php_template_is_scanned_only_inside_its_tags() { + let directory = tempfile::tempdir().unwrap(); + let path = directory.path().join("page.phtml"); + fs::write( + &path, + b"\n\n", + ) + .unwrap(); + + let scanned = run( + directory.path(), + &["scan", "page.phtml", "--format", "json"], + ); + assert_eq!( + scanned.status.code(), + Some(0), + "{}", + String::from_utf8_lossy(&scanned.stderr) + ); + let document: serde_json::Value = serde_json::from_slice(&scanned.stdout).unwrap(); + let report = &document["files"][0]["report"]; + assert_eq!(report["language"], "php"); + assert_eq!(report["comments"].as_array().unwrap().len(), 1); + + let fixed = run(directory.path(), &["fix", "page.phtml"]); + assert_eq!( + fixed.status.code(), + Some(0), + "{}", + String::from_utf8_lossy(&fixed.stderr) + ); + assert_eq!( + fs::read(&path).unwrap(), + b"\n\n" + ); +} + +/* NOTE: Zig is the one built-in language with no block comment, and this is what + * that costs a run end to end: `// zig fmt: off` is the only instruction the + * formatter reads out of a comment and is kept, the `//` written on a + * multiline string literal line is content the way one inside a quoted string + * is, and only the ordinary comment beside them is removed. `zig ast-check` + * (0.16.0) accepts the file below. */ +#[test] +fn a_zig_file_keeps_its_fmt_directive_and_its_multiline_string() { + let directory = tempfile::tempdir().unwrap(); + let path = directory.path().join("main.zig"); + fs::write( + &path, + b"// zig fmt: off\nconst text =\n \\\\a // not a comment\n;\nconst n: u32 = 1; // remove\n", + ) + .unwrap(); + + let scanned = run(directory.path(), &["scan", "main.zig", "--format", "json"]); + assert_eq!( + scanned.status.code(), + Some(0), + "{}", + String::from_utf8_lossy(&scanned.stderr) + ); + let document: serde_json::Value = serde_json::from_slice(&scanned.stdout).unwrap(); + let report = &document["files"][0]["report"]; + assert_eq!(report["language"], "zig"); + assert_eq!(report["comments"].as_array().unwrap().len(), 2); + assert_eq!(report["comments"][0]["kind"], "directive"); + assert_eq!(report["comments"][0]["disposition"]["action"], "keep"); + assert_eq!(report["comments"][1]["disposition"]["action"], "remove"); + + let fixed = run(directory.path(), &["fix", "main.zig"]); + assert_eq!( + fixed.status.code(), + Some(0), + "{}", + String::from_utf8_lossy(&fixed.stderr) + ); + assert_eq!( + fs::read(&path).unwrap(), + b"// zig fmt: off\nconst text =\n \\\\a // not a comment\n;\nconst n: u32 = 1; \n" + ); +} + +/* NOTE: Dart is the one built-in C-family language whose block comment nests, and + * this is what that plus its interpolation costs a run end to end: the outer + * `/*` is closed by the second `*/` and not the first, `${ ... }` is code so + * the comment written inside the string is a comment of its own, and + * `// dart format off` is one of the four instructions a Dart tool reads and is + * kept. Ground truth, Dart SDK 3.13.2 `scanString`: `SINGLE_LINE_COMMENT` at + * [0,18), `MULTI_LINE_COMMENT "/* who */ +"` at [48,57) inside the interpolation, + * `SINGLE_LINE_COMMENT` at [61,70), and `MULTI_LINE_COMMENT` at [71,106). + * `dart analyze` accepts both the file below and the bytes `fix` leaves. */ +#[test] +fn a_dart_file_keeps_its_format_directive_and_nests_its_block_comment() { + let directory = tempfile::tempdir().unwrap(); + let path = directory.path().join("main.dart"); + fs::write( + &path, + b"// dart format off\nvar greeting = 'hi ${'there' /* who */}'; // remove\n/* outer /* inner */ still outer */\n", + ) + .unwrap(); + + let scanned = run(directory.path(), &["scan", "main.dart", "--format", "json"]); + assert_eq!( + scanned.status.code(), + Some(0), + "{}", + String::from_utf8_lossy(&scanned.stderr) + ); + let document: serde_json::Value = serde_json::from_slice(&scanned.stdout).unwrap(); + let report = &document["files"][0]["report"]; + assert_eq!(report["language"], "dart"); + assert_eq!(report["comments"].as_array().unwrap().len(), 4); + assert_eq!(report["comments"][0]["kind"], "directive"); + assert_eq!(report["comments"][0]["disposition"]["action"], "keep"); + assert_eq!(report["comments"][1]["span"]["start"], 48); + assert_eq!(report["comments"][1]["span"]["end"], 57); + assert_eq!(report["comments"][3]["span"]["start"], 71); + assert_eq!(report["comments"][3]["span"]["end"], 106); + + let fixed = run(directory.path(), &["fix", "main.dart"]); + assert_eq!( + fixed.status.code(), + Some(0), + "{}", + String::from_utf8_lossy(&fixed.stderr) + ); + assert_eq!( + fs::read(&path).unwrap(), + b"// dart format off\nvar greeting = 'hi ${'there' }'; \n\n" + ); +} + +/* NOTE: Swift is the one built-in language whose regular expression literal can + * carry a `//` with no quote in front of it, and this is what that costs a run + * end to end: `#/https://x/#` holds two slashes that are pattern rather than + * comment, `\( ... )` is code so the block comment written inside the string is + * a comment of its own, the outer `/*` is closed by the second `*/` and not the + * first, and `// swift-tools-version:` is kept because SwiftPM reads it before + * it reads a manifest at all. Ground truth, the SwiftSyntax parser of the Swift + * 6.3.3 toolchain: `lineComment` at [0,26), a `regexLiteralPattern` at [39,48), + * `lineComment` at [52,61), `blockComment` at [92,101) inside the + * interpolation, `lineComment` at [105,114), and `blockComment` at [115,150). + * `swift-frontend -dump-parse -swift-version 6` accepts both the file below and + * the bytes `fix` leaves. */ +#[test] +fn a_swift_file_keeps_its_tools_version_and_hides_a_slash_pair_in_a_regex() { + let directory = tempfile::tempdir().unwrap(); + let path = directory.path().join("Package.swift"); + fs::write( + &path, + b"// swift-tools-version:5.9\nlet url = #/https://x/# // remove\nlet greeting = \"hi \\( \"there\" /* who */ )\" // remove\n/* outer /* inner */ still outer */\n", + ) + .unwrap(); + + let scanned = run( + directory.path(), + &["scan", "Package.swift", "--format", "json"], + ); + assert_eq!( + scanned.status.code(), + Some(0), + "{}", + String::from_utf8_lossy(&scanned.stderr) + ); + let document: serde_json::Value = serde_json::from_slice(&scanned.stdout).unwrap(); + let report = &document["files"][0]["report"]; + assert_eq!(report["language"], "swift"); + assert_eq!(report["comments"].as_array().unwrap().len(), 5); + assert_eq!(report["comments"][0]["kind"], "directive"); + assert_eq!(report["comments"][0]["disposition"]["action"], "keep"); + assert_eq!(report["comments"][1]["span"]["start"], 52); + assert_eq!(report["comments"][2]["span"]["start"], 92); + assert_eq!(report["comments"][2]["span"]["end"], 101); + assert_eq!(report["comments"][4]["span"]["start"], 115); + assert_eq!(report["comments"][4]["span"]["end"], 150); + + let fixed = run(directory.path(), &["fix", "Package.swift"]); + assert_eq!( + fixed.status.code(), + Some(0), + "{}", + String::from_utf8_lossy(&fixed.stderr) + ); + assert_eq!( + fs::read(&path).unwrap(), + b"// swift-tools-version:5.9\nlet url = #/https://x/# \nlet greeting = \"hi \\( \"there\" )\" \n\n" + ); +} + +/* NOTE: C# is the one built-in language whose *lines* are lexed two ways, and + * this is what that costs a run end to end: `#region` takes the rest of its line + * as the label an editor folds under, so the `//` in it is not a comment, while + * the `//` behind `#endregion` is one; the format clause behind the `:` of an + * interpolation hole is text, so the `//` in it is not a comment either; a + * verbatim string carries its `\` and hides the `//` inside it; and a block + * comment does not nest, so its first closing delimiter ends it and the `//` + * behind the leftovers opens a comment of its own. `// ` is kept because Roslyn exempts a + * file carrying one from every analyzer that opts out of generated code. Ground + * truth, the Roslyn lexer the .NET SDK 10.0.400 ships: + * `SingleLineCommentTrivia` at [0,20), `PreprocessingMessageTrivia` at [29,53), + * `SingleLineCommentTrivia` at [125,134), `MultiLineCommentTrivia` at [135,155), + * and `SingleLineCommentTrivia` at [156,165) and [177,186). It reports no + * lexical diagnostic for the file below, nor for the bytes `fix` leaves. */ +#[test] +fn a_csharp_file_keeps_its_generated_marker_and_its_directive_lines() { + let directory = tempfile::tempdir().unwrap(); + let path = directory.path().join("Program.cs"); + fs::write( + &path, + b"// \n#region Helpers // not a comment\nvar path = @\"C:\\dir // no\";\nvar text = $\"{path.Length:D4 // no} tail\"; // remove\n/* outer /* inner */ // remove\n#endregion // remove\n", + ) + .unwrap(); + + let scanned = run( + directory.path(), + &["scan", "Program.cs", "--format", "json"], + ); + assert_eq!( + scanned.status.code(), + Some(0), + "{}", + String::from_utf8_lossy(&scanned.stderr) + ); + let document: serde_json::Value = serde_json::from_slice(&scanned.stdout).unwrap(); + let report = &document["files"][0]["report"]; + assert_eq!(report["language"], "csharp"); + assert_eq!(report["comments"].as_array().unwrap().len(), 5); + assert_eq!(report["comments"][0]["kind"], "directive"); + assert_eq!(report["comments"][0]["disposition"]["action"], "keep"); + assert_eq!(report["comments"][1]["span"]["start"], 125); + assert_eq!(report["comments"][2]["kind"], "block"); + assert_eq!(report["comments"][2]["span"]["start"], 135); + assert_eq!(report["comments"][2]["span"]["end"], 155); + assert_eq!(report["comments"][4]["span"]["start"], 177); + assert_eq!(report["comments"][4]["span"]["end"], 186); + + let fixed = run(directory.path(), &["fix", "Program.cs"]); + assert_eq!( + fixed.status.code(), + Some(0), + "{}", + String::from_utf8_lossy(&fixed.stderr) + ); + assert_eq!( + fs::read(&path).unwrap(), + b"// \n#region Helpers // not a comment\nvar path = @\"C:\\dir // no\";\nvar text = $\"{path.Length:D4 // no} tail\"; \n \n#endregion \n" + ); +} + +/* NOTE: R is the one built-in language whose extension is written in upper case + * as often as in lower — `analysis.R` and `analysis.r` are the same kind of + * file — so this is the run that proves the suffix is folded before it is + * looked up. It is also what a roxygen comment costs end to end: `#'` is + * documentation and the default policy takes it, `# nolint` is lintr's + * instruction and is kept, and the `#` inside the raw string is content. R + * 4.3.3 `getParseData` reads the file below as `COMMENT` at [0,19), [20,30), + * [60,68) and [116,124), with `STR_CONST` covering [80,104). */ +#[test] +fn an_r_file_keeps_its_lint_directive_and_its_raw_string() { + let directory = tempfile::tempdir().unwrap(); + let path = directory.path().join("analysis.R"); + fs::write( + &path, + b"#' Add two numbers.\n#' @export\nadd <- function(a, b) a + b # nolint\npattern <- r\"(\\d+ # not a comment)\"\ntotal <- 1 # remove\n", + ) + .unwrap(); + + let scanned = run( + directory.path(), + &["scan", "analysis.R", "--format", "json"], + ); + assert_eq!( + scanned.status.code(), + Some(0), + "{}", + String::from_utf8_lossy(&scanned.stderr) + ); + let document: serde_json::Value = serde_json::from_slice(&scanned.stdout).unwrap(); + let report = &document["files"][0]["report"]; + assert_eq!(report["language"], "r"); + assert_eq!(report["comments"].as_array().unwrap().len(), 4); + assert_eq!(report["comments"][0]["kind"], "doc-line"); + assert_eq!(report["comments"][2]["kind"], "directive"); + assert_eq!(report["comments"][2]["disposition"]["action"], "keep"); + assert_eq!(report["comments"][3]["disposition"]["action"], "remove"); + + let fixed = run(directory.path(), &["fix", "analysis.R"]); + assert_eq!( + fixed.status.code(), + Some(0), + "{}", + String::from_utf8_lossy(&fixed.stderr) + ); + assert_eq!( + fs::read(&path).unwrap(), + b"\n\nadd <- function(a, b) a + b # nolint\npattern <- r\"(\\d+ # not a comment)\"\ntotal <- 1 \n" + ); +} + +/* NOTE: A `Gemfile` carries no extension, so it reaches the Ruby scanner by its + * whole name alone — and once there, the magic comment at the head of it is a + * directive the default policy keeps, where the embedded document below it is + * an ordinary comment the same run removes. */ +#[test] +fn a_gemfile_is_scanned_as_ruby_by_its_name_alone() { + let directory = tempfile::tempdir().unwrap(); + let path = directory.path().join("Gemfile"); + fs::write( + &path, + b"# frozen_string_literal: true\nsource '# opaque' # remove\n=begin\nnotes\n=end\n", + ) + .unwrap(); + + let scanned = run(directory.path(), &["scan", "Gemfile", "--format", "json"]); + assert_eq!( + scanned.status.code(), + Some(0), + "{}", + String::from_utf8_lossy(&scanned.stderr) + ); + let document: serde_json::Value = serde_json::from_slice(&scanned.stdout).unwrap(); + let report = &document["files"][0]["report"]; + assert_eq!(report["language"], "ruby"); + assert_eq!(report["comments"].as_array().unwrap().len(), 3); + assert_eq!(report["comments"][0]["kind"], "directive"); + assert_eq!(report["comments"][0]["disposition"]["action"], "keep"); + + let fixed = run(directory.path(), &["fix", "Gemfile"]); + assert_eq!( + fixed.status.code(), + Some(0), + "{}", + String::from_utf8_lossy(&fixed.stderr) + ); + assert_eq!( + fs::read(&path).unwrap(), + b"# frozen_string_literal: true\nsource '# opaque' \n\n\n\n" + ); +} + #[test] fn invalid_input_returns_two_and_fix_is_non_destructive() { let directory = tempfile::tempdir().unwrap(); @@ -193,20 +679,216 @@ fn init_lefthook_preserves_partial_stage_contract() { assert!(!generated.contains("stage_fixed")); } +/// The starter file is a decision the reader may already have made +/// differently: a second `init` must not quietly replace the config they have +/// been editing, and the refusal has to name both ways out. #[test] -fn staged_fix_does_not_stage_unrelated_working_tree_changes() { - let directory = repository(); - let path = directory.path().join("partial stage.rs"); - fs::write(&path, b"let a = 1; // existing\nlet b = 2;\n").unwrap(); - git(directory.path(), &["add", "partial stage.rs"]); - git( - directory.path(), - &["commit", "--quiet", "--message", "base"], - ); +fn init_config_refuses_to_overwrite_an_existing_file() { + let directory = tempfile::tempdir().unwrap(); + let path = directory.path().join(".ocomment.toml"); + let mine = b"version = 1\n[policy]\nmode = \"all\"\n"; + fs::write(&path, mine).unwrap(); - fs::write(&path, b"let a = 1; // existing\nlet b = 2; // staged\n").unwrap(); - git(directory.path(), &["add", "partial stage.rs"]); - fs::write( + let output = run(directory.path(), &["init", "config"]); + assert_eq!( + output.status.code(), + Some(2), + "{}", + String::from_utf8_lossy(&output.stderr) + ); + let error = String::from_utf8_lossy(&output.stderr); + assert!( + error.contains( + ".ocomment.toml already exists; use --force to overwrite or --stdout to print the \ + template" + ), + "{error}" + ); + assert_eq!( + fs::read(&path).unwrap(), + mine, + "the refusal edited the file" + ); +} + +/// The same refusal guards the hook file, and `--force` is the way past it. +#[test] +fn init_force_overwrites_an_existing_file() { + let directory = tempfile::tempdir().unwrap(); + let path = directory.path().join(".ocomment.toml"); + fs::write(&path, b"stale\n").unwrap(); + + let output = run(directory.path(), &["init", "config", "--force"]); + assert_eq!( + output.status.code(), + Some(0), + "{}", + String::from_utf8_lossy(&output.stderr) + ); + let written = fs::read_to_string(&path).unwrap(); + assert!(written.contains("version = 1"), "{written}"); + assert!( + !written.contains("stale"), + "the old bytes survived: {written}" + ); + + let hook = directory.path().join("lefthook.yml"); + fs::write(&hook, b"stale\n").unwrap(); + let refused = run(directory.path(), &["init", "lefthook"]); + assert_eq!(refused.status.code(), Some(2)); + assert!( + String::from_utf8_lossy(&refused.stderr).contains("lefthook.yml already exists"), + "{}", + String::from_utf8_lossy(&refused.stderr) + ); + assert_eq!(fs::read(&hook).unwrap(), b"stale\n"); + let forced = run(directory.path(), &["init", "lefthook", "--force"]); + assert_eq!(forced.status.code(), Some(0)); + assert!(fs::read_to_string(&hook).unwrap().contains("pre-commit:")); +} + +/// `--stdout` is the read-only door: the template goes to the pipe and the +/// working directory is left exactly as it was found. +#[test] +fn init_stdout_prints_the_template_and_writes_nothing() { + let directory = tempfile::tempdir().unwrap(); + let output = run(directory.path(), &["init", "config", "--stdout"]); + assert_eq!( + output.status.code(), + Some(0), + "{}", + String::from_utf8_lossy(&output.stderr) + ); + let printed = String::from_utf8(output.stdout).unwrap(); + assert!(printed.contains("version = 1"), "{printed}"); + assert!( + !printed.contains("created "), + "nothing was created, so nothing may claim it was: {printed}" + ); + assert!(!directory.path().join(".ocomment.toml").exists()); + + let hook = run(directory.path(), &["init", "lefthook", "--fix", "--stdout"]); + assert_eq!(hook.status.code(), Some(0)); + let printed = String::from_utf8(hook.stdout).unwrap(); + assert!(printed.contains("ocomment fix --staged"), "{printed}"); + assert!(!directory.path().join("lefthook.yml").exists()); +} + +/// A config in a parent directory already governs this one, so a new starter +/// file here layers over it rather than starting from nothing. The note says +/// so once the file exists, and does not stop it being written. +#[test] +fn init_notes_a_project_config_that_already_applies() { + let directory = tempfile::tempdir().unwrap(); + fs::write( + directory.path().join(".ocomment.toml"), + b"version = 1\n[policy]\nmode = \"all\"\n", + ) + .unwrap(); + let nested = directory.path().join("crate"); + fs::create_dir(&nested).unwrap(); + + let output = run(&nested, &["init", "config"]); + assert_eq!( + output.status.code(), + Some(0), + "{}", + String::from_utf8_lossy(&output.stderr) + ); + let note = String::from_utf8_lossy(&output.stderr); + assert!( + note.contains("note: ") + && note.contains(".ocomment.toml already applies to this directory"), + "{note}" + ); + assert!( + nested.join(".ocomment.toml").is_file(), + "the note replaced the file" + ); + + /* NOTE: The config the run itself just created is this directory's own, not an + * inherited one, so a first `init` in a bare directory says nothing. */ + let bare = tempfile::tempdir().unwrap(); + let quiet = run(bare.path(), &["init", "config"]); + assert_eq!(quiet.status.code(), Some(0)); + assert!( + !String::from_utf8_lossy(&quiet.stderr).contains("already applies"), + "{}", + String::from_utf8_lossy(&quiet.stderr) + ); + + /* NOTE: The note is advice about a file that was just created. A refused `init` + * created nothing, so it has nothing to advise about: the error stands + * alone rather than trailing guidance for a file that does not exist. */ + let refused = run(&nested, &["init", "config"]); + assert_eq!(refused.status.code(), Some(2)); + let error = String::from_utf8_lossy(&refused.stderr); + assert!(error.contains(".ocomment.toml already exists"), "{error}"); + assert!( + !error.contains("already applies"), + "a refused init still advised about the inherited config:\n{error}" + ); +} + +/// Creating the file is not the end of the task, so the line that reports it +/// names the step that is. +#[test] +fn init_success_messages_name_the_next_step() { + let directory = tempfile::tempdir().unwrap(); + let config = run(directory.path(), &["init", "config"]); + assert_eq!(config.status.code(), Some(0)); + assert_eq!( + String::from_utf8_lossy(&config.stdout).trim_end(), + "created .ocomment.toml \u{2014} edit [policy] and run `ocomment check`" + ); + + let hook = run(directory.path(), &["init", "lefthook"]); + assert_eq!(hook.status.code(), Some(0)); + assert_eq!( + String::from_utf8_lossy(&hook.stdout).trim_end(), + "created lefthook.yml \u{2014} run `lefthook install` to activate the hook" + ); +} + +/// One writes a file and the other refuses to; asking for both is a mistake +/// clap can catch before anything is opened. +#[test] +fn init_refuses_force_together_with_stdout() { + let directory = tempfile::tempdir().unwrap(); + let output = run(directory.path(), &["init", "config", "--force", "--stdout"]); + assert_eq!(output.status.code(), Some(2)); + let error = String::from_utf8_lossy(&output.stderr); + assert!(error.contains("--force"), "{error}"); + assert!(error.contains("--stdout"), "{error}"); + assert!(!directory.path().join(".ocomment.toml").exists()); +} + +/// The two new switches are documented where a reader looks for them. +#[test] +fn init_help_documents_force_and_stdout() { + let directory = tempfile::tempdir().unwrap(); + let output = run(directory.path(), &["init", "--help"]); + assert_eq!(output.status.code(), Some(0)); + let help = String::from_utf8(output.stdout).unwrap(); + for needle in ["--force", "--stdout"] { + assert!(help.contains(needle), "`{needle}` is missing from:\n{help}"); + } +} + +#[test] +fn staged_fix_does_not_stage_unrelated_working_tree_changes() { + let directory = repository(); + let path = directory.path().join("partial stage.rs"); + fs::write(&path, b"let a = 1; // existing\nlet b = 2;\n").unwrap(); + git(directory.path(), &["add", "partial stage.rs"]); + git( + directory.path(), + &["commit", "--quiet", "--message", "base"], + ); + + fs::write(&path, b"let a = 1; // existing\nlet b = 2; // staged\n").unwrap(); + git(directory.path(), &["add", "partial stage.rs"]); + fs::write( &path, b"let a = 1; // existing\nlet b = 2; // staged\nlet c = 3; // unstaged\n", ) @@ -236,225 +918,5149 @@ fn staged_fix_does_not_stage_unrelated_working_tree_changes() { assert!(!cached.contains("// unstaged")); } +/// `--staged` reads its paths from the index rather than from a walk, but +/// `[files]` says which of the project's files OComment is allowed to touch +/// either way. A path the configuration excludes is not the commit hook's +/// business: it is not reported, and `fix --staged` leaves its blob alone. #[test] -fn staged_fix_stops_on_existing_block_comment_interior() { +fn staged_runs_honour_the_files_exclude_globs() { let directory = repository(); - let path = directory.path().join("block.c"); - fs::write(&path, b"/* existing\nbase\n*/\nint x;\n").unwrap(); - git(directory.path(), &["add", "block.c"]); + fs::write( + directory.path().join(".ocomment.toml"), + b"version = 1\n\n[files]\nexclude = [\"vendor/**\"]\n", + ) + .unwrap(); + fs::create_dir(directory.path().join("vendor")).unwrap(); + fs::create_dir(directory.path().join("src")).unwrap(); + fs::write( + directory.path().join("vendor/x.rs"), + b"let a = 1; // vendored\n", + ) + .unwrap(); + fs::write(directory.path().join("src/y.rs"), b"let b = 2; // ours\n").unwrap(); + git(directory.path(), &["add", "vendor/x.rs", "src/y.rs"]); + + let checked = run(directory.path(), &["check", "--staged"]); + let report = String::from_utf8(checked.stdout).unwrap(); + assert_eq!( + checked.status.code(), + Some(1), + "`check --staged` said:\n{report}" + ); + assert!(report.contains("src/y.rs"), "{report}"); + assert!( + !report.contains("vendor/x.rs"), + "an excluded path was reported:\n{report}" + ); + + let fixed = run(directory.path(), &["fix", "--staged"]); + assert_eq!( + fixed.status.code(), + Some(0), + "{}", + String::from_utf8_lossy(&fixed.stderr) + ); + assert_eq!( + git(directory.path(), &["show", ":vendor/x.rs"]), + b"let a = 1; // vendored\n", + "an excluded blob was rewritten" + ); + assert_eq!( + git(directory.path(), &["show", ":src/y.rs"]), + b"let b = 2; \n" + ); +} + +/// The other half of the same rule: an `include` list narrows a staged run to +/// the paths it names, exactly as it narrows a walk. +#[test] +fn staged_runs_honour_the_files_include_globs() { + let directory = repository(); + fs::write( + directory.path().join(".ocomment.toml"), + b"version = 1\n\n[files]\ninclude = [\"src/**\"]\n", + ) + .unwrap(); + fs::create_dir(directory.path().join("vendor")).unwrap(); + fs::create_dir(directory.path().join("src")).unwrap(); + fs::write( + directory.path().join("vendor/x.rs"), + b"let a = 1; // vendored\n", + ) + .unwrap(); + fs::write(directory.path().join("src/y.rs"), b"let b = 2; // ours\n").unwrap(); + git(directory.path(), &["add", "vendor/x.rs", "src/y.rs"]); + + let checked = run(directory.path(), &["check", "--staged"]); + let report = String::from_utf8(checked.stdout).unwrap(); + assert_eq!( + checked.status.code(), + Some(1), + "`check --staged` said:\n{report}" + ); + assert!(report.contains("src/y.rs"), "{report}"); + assert!( + !report.contains("vendor/x.rs"), + "a path outside the include list was reported:\n{report}" + ); +} + +/// `git` names a staged path relative to the repository root and a `[files]` +/// glob is written relative to the project root, so the two meet wherever the +/// command was typed. A run from a subdirectory must reach the same verdict as +/// a run from the top. +#[test] +fn staged_globs_stay_root_relative_from_a_subdirectory() { + let directory = repository(); + fs::write( + directory.path().join(".ocomment.toml"), + b"version = 1\n\n[files]\nexclude = [\"vendor/**\"]\n", + ) + .unwrap(); + fs::create_dir(directory.path().join("vendor")).unwrap(); + fs::create_dir(directory.path().join("src")).unwrap(); + fs::write( + directory.path().join("vendor/x.rs"), + b"let a = 1; // vendored\n", + ) + .unwrap(); + fs::write(directory.path().join("src/y.rs"), b"let b = 2; // ours\n").unwrap(); + git(directory.path(), &["add", "vendor/x.rs", "src/y.rs"]); + + let checked = run(&directory.path().join("src"), &["check", "--staged"]); + let report = String::from_utf8(checked.stdout).unwrap(); + assert_eq!( + checked.status.code(), + Some(1), + "`check --staged` said:\n{report}" + ); + assert!(report.contains("src/y.rs"), "{report}"); + assert!( + !report.contains("vendor/x.rs"), + "an excluded path was reported from a subdirectory:\n{report}" + ); +} + +/// `[files]` bounds a walk with more than its two glob lists: `hidden` decides +/// whether a dot-directory is looked into at all, and `max_size` decides how +/// much of a file is worth reading. A staged path is a walked path rather than +/// a named one, so a staged run is bounded by the same two settings — a commit +/// that touches `.cache/generated.rs` or a two-megabyte fixture must not put +/// either through a hook that would never have walked into them. +#[test] +fn staged_runs_honour_the_hidden_and_size_limits() { + let directory = repository(); + fs::write( + directory.path().join(".ocomment.toml"), + b"version = 1\n\n[files]\nmax_size = 20\n", + ) + .unwrap(); + fs::create_dir(directory.path().join(".hidden")).unwrap(); + fs::create_dir(directory.path().join("src")).unwrap(); + fs::write(directory.path().join(".hidden/x.rs"), b"let a = 1; // x\n").unwrap(); + fs::write( + directory.path().join("big.rs"), + b"let big = 3; // past the limit\n", + ) + .unwrap(); + fs::write(directory.path().join("src/y.rs"), b"let b = 2; // ours\n").unwrap(); git( directory.path(), - &["commit", "--quiet", "--message", "base"], + &["add", ".hidden/x.rs", "big.rs", "src/y.rs"], ); - let staged_source = b"/* existing\nbase\nadded\n*/\nint x;\n"; - fs::write(&path, staged_source).unwrap(); - git(directory.path(), &["add", "block.c"]); - let before_index = git(directory.path(), &["show", ":block.c"]); - let output = run(directory.path(), &["fix", "--staged"]); - assert_eq!(output.status.code(), Some(2)); - assert!(String::from_utf8_lossy(&output.stdout).contains("staged-existing-block-comment")); - assert_eq!(git(directory.path(), &["show", ":block.c"]), before_index); - assert_eq!(fs::read(path).unwrap(), staged_source); + let checked = run(directory.path(), &["check", "--staged"]); + let report = String::from_utf8(checked.stdout).unwrap(); + let summary = String::from_utf8(checked.stderr).unwrap(); + assert_eq!( + checked.status.code(), + Some(1), + "`check --staged` said:\n{report}{summary}" + ); + assert!(report.contains("src/y.rs"), "{report}"); + assert!( + !report.contains(".hidden/x.rs"), + "a hidden staged path was reported while `hidden` is off:\n{report}" + ); + assert!( + !report.contains("big.rs"), + "a staged blob past `max_size` was reported:\n{report}" + ); + /* NOTE: A size skip is a fact about the run, so it is folded into the summary + * under the same short label a walk gives it; a hidden path was never a + * candidate and is not a skip at all. */ + assert!( + summary.contains("1 file skipped (too large: 1"), + "the oversized staged blob was passed over silently:\n{summary}" + ); + + fs::write( + directory.path().join(".ocomment.toml"), + b"version = 1\n\n[files]\nmax_size = 20\nhidden = true\n", + ) + .unwrap(); + let visible = run(directory.path(), &["check", "--staged"]); + let report = String::from_utf8(visible.stdout).unwrap(); + assert_eq!( + visible.status.code(), + Some(1), + "`check --staged` said:\n{report}" + ); + assert!( + report.contains(".hidden/x.rs"), + "`hidden = true` did not reach the staged run:\n{report}" + ); + assert!( + !report.contains("big.rs"), + "`hidden = true` also lifted the size limit:\n{report}" + ); } +/// The other half of the same rule. `[files]` bounds what a run *finds*, and +/// a path the caller typed was never found: they named it, so it is checked +/// whatever `hidden` and `max_size` would have said about it. A walk has said +/// so since `discover_with_scope`, and a staged run says it about the same two +/// settings — otherwise `ocomment check --staged .hidden/x.rs` answers about +/// nothing at all, which reads as a clean file rather than as a path outside +/// the project's bounds. #[test] -fn staged_new_rename_delete_and_unusual_paths_are_handled_from_index_blobs() { +fn staged_paths_the_caller_names_bypass_the_hidden_and_size_limits() { let directory = repository(); - fs::write(directory.path().join("renamed.rs"), b"let base = 1;\n").unwrap(); - fs::write(directory.path().join("deleted.rs"), b"let deleted = 1;\n").unwrap(); - git(directory.path(), &["add", "."]); + fs::write( + directory.path().join(".ocomment.toml"), + b"version = 1\n\n[files]\nmax_size = 20\n", + ) + .unwrap(); + fs::create_dir(directory.path().join(".hidden")).unwrap(); + fs::write(directory.path().join(".hidden/x.rs"), b"let a = 1; // x\n").unwrap(); + fs::write( + directory.path().join("big.rs"), + b"let big = 3; // past the limit\n", + ) + .unwrap(); + git(directory.path(), &["add", ".hidden/x.rs", "big.rs"]); + + for named in [".hidden/x.rs", "big.rs"] { + let checked = run(directory.path(), &["check", "--staged", named]); + let report = String::from_utf8(checked.stdout).unwrap(); + assert_eq!( + checked.status.code(), + Some(1), + "`check --staged {named}` said:\n{report}{}", + String::from_utf8_lossy(&checked.stderr) + ); + assert!( + report.contains(named), + "the caller named {named} and it was filtered out anyway:\n{report}" + ); + } + + /* NOTE: A named directory carries the same licence to everything under it, + * exactly as an explicitly walked directory does. */ + let named_directory = run(directory.path(), &["check", "--staged", ".hidden"]); + let report = String::from_utf8(named_directory.stdout).unwrap(); + assert_eq!(named_directory.status.code(), Some(1), "{report}"); + assert!( + report.contains(".hidden/x.rs"), + "a staged path under a named directory was filtered out:\n{report}" + ); + + /* NOTE: Nobody named anything here, so both limits apply again. */ + let bare = run(directory.path(), &["check", "--staged"]); + let report = String::from_utf8(bare.stdout).unwrap(); + assert_eq!( + bare.status.code(), + Some(0), + "a bare staged run reported a hidden or oversized path:\n{report}" + ); + assert!(!report.contains(".hidden/x.rs"), "{report}"); + assert!(!report.contains("big.rs"), "{report}"); +} + +/// A pathspec is not always the prefix of the path `git` answers with. +/// +/// `git diff --cached` names a staged path relative to the repository root, +/// while the pathspec beside it is written however the caller found it +/// convenient: as an absolute path, or with a wildcard `git` expands itself. +/// Comparing the two as text answers "nobody named this" for both spellings, +/// and the project's limits then hide the very file the caller asked about, so +/// the question goes to `git` instead — the only party that knows what a +/// pathspec covers. +#[test] +fn a_staged_pathspec_names_its_paths_however_it_is_spelled() { + let directory = repository(); + fs::write( + directory.path().join(".ocomment.toml"), + b"version = 1\n\n[files]\nmax_size = 20\n", + ) + .unwrap(); + fs::create_dir(directory.path().join(".hidden")).unwrap(); + fs::write(directory.path().join(".hidden/x.rs"), b"let a = 1; // x\n").unwrap(); + git(directory.path(), &["add", ".hidden/x.rs"]); + + let absolute = directory.path().join(".hidden/x.rs"); + let named = run( + directory.path(), + &["check", "--staged", absolute.to_str().unwrap()], + ); + let report = String::from_utf8(named.stdout).unwrap(); + assert_eq!( + named.status.code(), + Some(1), + "an absolute staged pathspec named nothing:\n{report}" + ); + assert!(report.contains(".hidden/x.rs"), "{report}"); + + let expanded = run(directory.path(), &["check", "--staged", ".hidden/*.rs"]); + let report = String::from_utf8(expanded.stdout).unwrap(); + assert_eq!( + expanded.status.code(), + Some(1), + "a wildcard staged pathspec named nothing:\n{report}" + ); + assert!(report.contains(".hidden/x.rs"), "{report}"); +} + +/// A repository whose staged paths sit on both sides of every `[files]` limit: +/// one hidden, one oversized at the top, one oversized under `src`, and one +/// ordinary file `src` is scanned for. +fn repository_with_limits() -> TempDir { + let directory = repository(); + fs::write( + directory.path().join(".ocomment.toml"), + b"version = 1\n\n[files]\nmax_size = 20\n", + ) + .unwrap(); + fs::create_dir(directory.path().join(".hidden")).unwrap(); + fs::create_dir(directory.path().join("src")).unwrap(); + fs::write(directory.path().join(".hidden/x.rs"), b"let a = 1; // x\n").unwrap(); + fs::write( + directory.path().join("wide.rs"), + b"let wide = 3; // past the limit\n", + ) + .unwrap(); + fs::write(directory.path().join("src/y.rs"), b"let b = 2; // ours\n").unwrap(); + fs::write( + directory.path().join("src/tall.rs"), + b"let tall = 4; // past the limit\n", + ) + .unwrap(); git( directory.path(), - &["commit", "--quiet", "--message", "base"], + &["add", ".hidden/x.rs", "wide.rs", "src/y.rs", "src/tall.rs"], ); + directory +} - git(directory.path(), &["mv", "renamed.rs", "renamed target.rs"]); +/// Where a pathspec was typed decides what it covers. +/// +/// `ocomment check .` from `src/` walks `src/`, so `--staged .` from the same +/// directory has to mean the same subtree — and to mean it in both directions: +/// what sits above `src` is no business of the run, and what sits under it was +/// named, so the project's limits are lifted from it exactly as a walk lifts +/// them from a directory the caller pointed at. +#[test] +fn a_staged_pathspec_is_resolved_where_it_was_typed() { + let directory = repository_with_limits(); + + let checked = run(&directory.path().join("src"), &["check", "--staged", "."]); + let report = String::from_utf8(checked.stdout).unwrap(); + assert_eq!( + checked.status.code(), + Some(1), + "`check --staged .` from a subdirectory said:\n{report}" + ); + assert!(report.contains("src/y.rs"), "{report}"); + assert!( + report.contains("src/tall.rs"), + "`.` named the subtree and the size limit was applied to it anyway:\n{report}" + ); + assert!( + !report.contains(".hidden/x.rs") && !report.contains("wide.rs"), + "`.` typed in src/ reached outside it:\n{report}" + ); +} + +/// The whole tree is what a staged run already covers, so naming it says +/// nothing. +/// +/// A hook that spells its run `ocomment check --staged .` from the top of the +/// repository is asking for the same run as `ocomment check --staged`, and it +/// must get the same answer: every `[files]` limit still applies. Only a +/// pathspec that narrows the run is a request about particular paths, which is +/// what earns the licence to look past those limits. +#[test] +fn a_whole_tree_staged_pathspec_keeps_the_project_limits() { + let directory = repository_with_limits(); + + let whole_tree = run(directory.path(), &["check", "--staged", "."]); + let report = String::from_utf8(whole_tree.stdout).unwrap(); + let summary = String::from_utf8(whole_tree.stderr).unwrap(); + assert_eq!( + whole_tree.status.code(), + Some(1), + "`check --staged .` said:\n{report}{summary}" + ); + assert!(report.contains("src/y.rs"), "{report}"); + assert!( + !report.contains(".hidden/x.rs"), + "a bare `.` lifted `hidden` from the whole tree:\n{report}" + ); + assert!( + !report.contains("wide.rs") && !report.contains("src/tall.rs"), + "a bare `.` lifted `max_size` from the whole tree:\n{report}" + ); + assert!( + summary.contains("2 files skipped (too large: 2"), + "the oversized staged blobs were passed over silently:\n{summary}" + ); + + let bare = run(directory.path(), &["check", "--staged"]); + assert_eq!(bare.status.code(), whole_tree.status.code()); + assert_eq!( + String::from_utf8(bare.stdout).unwrap(), + report, + "`check --staged .` and `check --staged` disagreed about the same tree" + ); + assert_eq!(String::from_utf8(bare.stderr).unwrap(), summary); +} + +/// A staged path the caller named and nothing could read is answered on a line +/// of its own. +/// +/// The rule is the walk's: what a run merely came across is folded into the +/// end-of-run summary, and what the caller asked about is answered directly. +/// `ocomment check --staged notes.unknownext` that says only "nothing to check" reads +/// as a clean file rather than as a file with no scanner. +#[test] +fn a_named_staged_path_without_a_scanner_gets_its_own_line() { + let directory = repository(); + fs::write(directory.path().join("notes.unknownext"), b"# notes\n").unwrap(); + git(directory.path(), &["add", "notes.unknownext"]); + + let named = run(directory.path(), &["check", "--staged", "notes.unknownext"]); + let report = String::from_utf8(named.stdout).unwrap(); + assert_eq!(named.status.code(), Some(0), "{report}"); + assert!( + report.contains(&format!("notes.unknownext: skipped: {NO_LANGUAGE}")), + "a staged path the caller named was passed over without a word:\n{report}" + ); + + let bare = run(directory.path(), &["check", "--staged"]); + let folded = String::from_utf8(bare.stdout).unwrap(); + assert!( + !folded.contains("notes.unknownext: skipped"), + "a staged path nobody named was listed per file:\n{folded}" + ); +} + +/// A staged blob OComment has no scanner for is passed over, and a run says so +/// the way a walk says it: folded onto the end-of-run summary under the same +/// short label, rather than dropped without a word. A pre-commit hook that +/// stages a PNG and a Markdown file is otherwise indistinguishable from one +/// that scanned them and found nothing. +#[test] +fn staged_blobs_without_a_scanner_are_counted_in_the_summary() { + let directory = repository(); + fs::write(directory.path().join("notes.unknownext"), b"# notes\n").unwrap(); + fs::write(directory.path().join("image.dat"), b"\x89PNG\0\r\n").unwrap(); + fs::write(directory.path().join("y.rs"), b"let b = 2; // ours\n").unwrap(); + git( + directory.path(), + &["add", "notes.unknownext", "image.dat", "y.rs"], + ); + + let checked = run(directory.path(), &["check", "--staged"]); + let report = String::from_utf8(checked.stdout).unwrap(); + let summary = String::from_utf8(checked.stderr).unwrap(); + assert_eq!( + checked.status.code(), + Some(1), + "`check --staged` said:\n{report}{summary}" + ); + assert!(report.contains("y.rs"), "{report}"); + assert!( + summary.contains("2 files skipped (binary: 1, unknown language: 1"), + "the staged blobs nothing could read were passed over silently:\n{summary}" + ); + /* NOTE: Nobody typed either path, so neither is annotated on a line of its + * own until `-v` asks for the list. */ + assert!( + !report.contains("notes.unknownext") && !report.contains("image.dat"), + "a folded skip was reported per file:\n{report}" + ); + + let verbose = run(directory.path(), &["check", "--staged", "-v"]); + let listed = String::from_utf8(verbose.stdout).unwrap(); + assert!( + listed.contains(&format!("notes.unknownext: skipped: {NO_LANGUAGE}")), + "-v did not list the staged path with no language:\n{listed}" + ); + assert!( + listed.contains("image.dat: skipped: binary file (NUL byte)"), + "-v did not list the staged binary blob:\n{listed}" + ); +} + +#[test] +fn staged_fix_stops_on_existing_block_comment_interior() { + let directory = repository(); + let path = directory.path().join("block.c"); + fs::write(&path, b"/* existing\nbase\n*/\nint x;\n").unwrap(); + git(directory.path(), &["add", "block.c"]); + git( + directory.path(), + &["commit", "--quiet", "--message", "base"], + ); + let staged_source = b"/* existing\nbase\nadded\n*/\nint x;\n"; + fs::write(&path, staged_source).unwrap(); + git(directory.path(), &["add", "block.c"]); + let before_index = git(directory.path(), &["show", ":block.c"]); + + let output = run(directory.path(), &["fix", "--staged"]); + assert_eq!(output.status.code(), Some(2)); + assert!(String::from_utf8_lossy(&output.stdout).contains("staged-existing-block-comment")); + assert_eq!(git(directory.path(), &["show", ":block.c"]), before_index); + assert_eq!(fs::read(path).unwrap(), staged_source); +} + +#[test] +fn staged_new_rename_delete_and_unusual_paths_are_handled_from_index_blobs() { + let directory = repository(); + fs::write(directory.path().join("renamed.rs"), b"let base = 1;\n").unwrap(); + fs::write(directory.path().join("deleted.rs"), b"let deleted = 1;\n").unwrap(); + git(directory.path(), &["add", "."]); + git( + directory.path(), + &["commit", "--quiet", "--message", "base"], + ); + + git(directory.path(), &["mv", "renamed.rs", "renamed target.rs"]); + fs::write( + directory.path().join("renamed target.rs"), + b"let base = 1;\nlet renamed = 2; // staged rename\n", + ) + .unwrap(); + git(directory.path(), &["add", "renamed target.rs"]); + fs::remove_file(directory.path().join("deleted.rs")).unwrap(); + git(directory.path(), &["add", "deleted.rs"]); + let unusual = "odd\n名前.rs"; + fs::write( + directory.path().join(unusual), + b"let new = 1; // new file\n", + ) + .unwrap(); + git(directory.path(), &["add", unusual]); + + let output = run(directory.path(), &["fix", "--staged", "--index-only"]); + assert_eq!( + output.status.code(), + Some(0), + "{}", + String::from_utf8_lossy(&output.stderr) + ); + assert_eq!( + git(directory.path(), &["show", ":renamed target.rs"]), + b"let base = 1;\nlet renamed = 2; \n" + ); + assert_eq!( + git(directory.path(), &["show", &format!(":{unusual}")]), + b"let new = 1; \n" + ); + assert_eq!( + fs::read(directory.path().join("renamed target.rs")).unwrap(), + b"let base = 1;\nlet renamed = 2; // staged rename\n" + ); + assert_eq!( + fs::read(directory.path().join(unusual)).unwrap(), + b"let new = 1; // new file\n" + ); +} + +#[cfg(unix)] +#[test] +fn staged_non_utf8_paths_remain_os_native() { + use std::os::unix::ffi::{OsStrExt, OsStringExt}; + + let directory = repository(); + let name = std::ffi::OsString::from_vec(b"non-\xff.rs".to_vec()); + let path = directory.path().join(&name); + fs::write(&path, b"let value = 1; // remove\n").unwrap(); + git_with_path(directory.path(), &["add", "--"], &name); + + let output = run(directory.path(), &["fix", "--staged", "--index-only"]); + assert_eq!( + output.status.code(), + Some(0), + "{}", + String::from_utf8_lossy(&output.stderr) + ); + let mut specification = std::ffi::OsString::from(":"); + specification.push(&name); + assert_eq!( + git_with_path( + directory.path(), + &["cat-file", "blob"], + specification.as_os_str() + ), + b"let value = 1; \n" + ); + assert_eq!(name.as_bytes(), b"non-\xff.rs"); +} + +#[test] +fn ambiguous_staged_mapping_changes_nothing_and_suggests_index_only() { + let directory = repository(); + let path = directory.path().join("ambiguous.rs"); + fs::write(&path, b"let base = 1;\n").unwrap(); + git(directory.path(), &["add", "ambiguous.rs"]); + git( + directory.path(), + &["commit", "--quiet", "--message", "base"], + ); + + let staged = b"let base = 1;\nlet staged = 2; // remove\n"; + fs::write(&path, staged).unwrap(); + git(directory.path(), &["add", "ambiguous.rs"]); + let working = [staged.as_slice(), staged.as_slice()].concat(); + fs::write(&path, &working).unwrap(); + let before_index = git(directory.path(), &["show", ":ambiguous.rs"]); + + let output = run(directory.path(), &["fix", "--staged"]); + assert_eq!(output.status.code(), Some(2)); + assert!(String::from_utf8_lossy(&output.stderr).contains("--index-only")); + assert_eq!( + git(directory.path(), &["show", ":ambiguous.rs"]), + before_index + ); + assert_eq!(fs::read(&path).unwrap(), working); + + let index_only = run(directory.path(), &["fix", "--staged", "--index-only"]); + assert_eq!(index_only.status.code(), Some(0)); + assert_eq!( + git(directory.path(), &["show", ":ambiguous.rs"]), + b"let base = 1;\nlet staged = 2; \n" + ); + assert_eq!(fs::read(path).unwrap(), working); +} + +#[test] +fn normal_walk_respects_gitignore() { + let directory = repository(); + fs::write(directory.path().join(".gitignore"), b"ignored.rs\n").unwrap(); + fs::write(directory.path().join("ignored.rs"), b"// ignored\n").unwrap(); + fs::write(directory.path().join("seen.rs"), b"// seen\n").unwrap(); + let output = run(directory.path(), &[]); + assert_eq!(output.status.code(), Some(1)); + let text = String::from_utf8_lossy(&output.stdout); + assert!(text.contains("seen.rs")); + assert!(!text.contains("ignored.rs")); +} + +#[test] +fn explicit_io_failure_returns_two_and_blocks_the_whole_fix() { + let directory = tempfile::tempdir().unwrap(); + let path = directory.path().join("good.rs"); + let original = b"let x = 1; // remove\n"; + fs::write(&path, original).unwrap(); + let output = run(directory.path(), &["fix", "good.rs", "missing.rs"]); + assert_eq!(output.status.code(), Some(2)); + assert_eq!(fs::read(path).unwrap(), original); + assert!(String::from_utf8_lossy(&output.stdout).contains("path does not exist")); +} + +/// A command that names no path checks the current directory, the way every +/// other file-walking developer tool does. The repository root is still where +/// the configuration is discovered and where the override globs are anchored, +/// but it is no longer what a bare `ocomment` walks: run from a subdirectory, +/// the command must not reach back up to files the caller cannot see. +#[test] +fn no_argument_scan_uses_the_current_directory_not_the_repository_root() { + let directory = repository(); + fs::write(directory.path().join("root.rs"), b"// root comment\n").unwrap(); + let nested = directory.path().join("nested/deeper"); + fs::create_dir_all(&nested).unwrap(); + fs::write(nested.join("deep.rs"), b"// deep comment\n").unwrap(); + + let output = run(&nested, &[]); + assert_eq!(output.status.code(), Some(1)); + let report = String::from_utf8(output.stdout).unwrap(); + assert!( + !report.contains("root.rs"), + "the bare command reached above the current directory:\n{report}" + ); + assert!( + report.contains("deep.rs:1:1: removable"), + "the bare command never checked the current directory:\n{report}" + ); + /* NOTE: The implicit target is `.`, and a walk rooted there prefixes every entry + * with `./`. `ocomment` and `ocomment check deep.rs` report one file under + * one name, so that prefix is not part of it. */ + assert!( + !report.contains("./"), + "the implicit target leaked its `./` into the report:\n{report}" + ); + + /* NOTE: `-v` names both halves of the answer: the root the configuration came + * from, and the target that root no longer decides. */ + let traced = run(&nested, &["-v"]); + let trace = String::from_utf8(traced.stderr).unwrap(); + let repository_name = directory.path().file_name().unwrap().to_str().unwrap(); + assert!( + trace + .lines() + .any(|line| line.starts_with("root: ") && line.ends_with(repository_name)), + "the trace did not root the run at the repository:\n{trace}" + ); + assert!( + trace.lines().any(|line| line == "target: ."), + "the trace did not name the implicit target:\n{trace}" + ); +} + +/// The root keeps the two jobs it did not lose: it is where `.ocomment.toml` +/// is found, and it is what `files.include`, `files.exclude`, and every +/// `[[overrides]].paths` glob is written relative to. A path named on the +/// command line is relative to the working directory instead, so the two only +/// line up once a path is resolved against the directory it was typed in — +/// whichever of the three ways the file was named. +#[test] +fn project_config_and_overrides_apply_from_a_subdirectory() { + let directory = tempfile::tempdir().unwrap(); + fs::write( + directory.path().join(".ocomment.toml"), + b"version = 1\n\n[files]\nexclude = [\"nested/skip/**\"]\n\n[[overrides]]\npaths = [\"nested/**\"]\npolicy = \"all\"\n", + ) + .unwrap(); + let nested = directory.path().join("nested"); + fs::create_dir_all(nested.join("skip")).unwrap(); + /* NOTE: A directive is kept under the default `safe` policy and removed under + * `all`, so the line it is reported on is the override speaking. */ + fs::write(nested.join("kept.rs"), b"let x = 1; // rustfmt::skip\n").unwrap(); + fs::write(nested.join("skip/ignored.rs"), b"let y = 2; // remove\n").unwrap(); + + for arguments in [&[][..], &["check", "."][..], &["check", "kept.rs"][..]] { + let output = run(&nested, arguments); + let report = String::from_utf8(output.stdout).unwrap(); + assert_eq!( + output.status.code(), + Some(1), + "`ocomment {}` did not apply the override:\n{report}", + arguments.join(" ") + ); + assert!( + report.contains("kept.rs:1:12: removable directive comment"), + "`ocomment {}` did not apply the override:\n{report}", + arguments.join(" ") + ); + assert!( + !report.contains("ignored.rs"), + "`ocomment {}` walked into the excluded directory:\n{report}", + arguments.join(" ") + ); + } +} + +/// `fix` is the command that writes, so the change of target matters most +/// there: run from a subdirectory it rewrites that subdirectory, and the +/// files above it are none of its business. +#[test] +fn fix_from_a_subdirectory_leaves_the_repository_root_alone() { + let directory = repository(); + let untouched = directory.path().join("root.rs"); + let original = b"let a = 1; // root comment\n"; + fs::write(&untouched, original).unwrap(); + let nested = directory.path().join("nested"); + fs::create_dir(&nested).unwrap(); + let rewritten = nested.join("deep.rs"); + fs::write(&rewritten, b"let b = 2; // deep comment\n").unwrap(); + + let output = run(&nested, &["fix"]); + assert_eq!( + output.status.code(), + Some(0), + "{}", + String::from_utf8_lossy(&output.stderr) + ); + assert_eq!( + fs::read(&untouched).unwrap(), + original, + "`fix` from a subdirectory rewrote the repository root" + ); + assert_eq!(fs::read(&rewritten).unwrap(), b"let b = 2; \n"); +} + +/// A reader who has only ever run `ocomment fix` from the top of a repository +/// can read the bare command as "fix the project", so the one run that writes +/// says where it is pointed and where the project it belongs to starts. From +/// the root the two are the same directory and the note would be noise. +#[test] +fn fix_from_a_subdirectory_notes_the_project_root() { + let directory = repository(); + let nested = directory.path().join("nested"); + fs::create_dir(&nested).unwrap(); + fs::write(nested.join("deep.rs"), b"let b = 2; // deep comment\n").unwrap(); + + let output = run(&nested, &["fix"]); + assert_eq!(output.status.code(), Some(0)); + let note = String::from_utf8(output.stderr).unwrap(); + assert!( + note.contains("note: fixing files under . (project root: "), + "`fix` never said what it was pointed at:\n{note}" + ); + assert_eq!( + note.matches("note: fixing files under").count(), + 1, + "the scope was noted more than once:\n{note}" + ); + + let from_root = run(directory.path(), &["fix"]); + assert_eq!(from_root.status.code(), Some(0)); + let quiet = String::from_utf8(from_root.stderr).unwrap(); + assert!( + !quiet.contains("note: fixing files under"), + "the note was printed where the target and the root agree:\n{quiet}" + ); +} + +#[test] +fn explicit_directory_bypasses_hidden_and_size_limits() { + let directory = tempfile::tempdir().unwrap(); + fs::write( + directory.path().join(".ocomment.toml"), + b"version = 1\n[files]\nmax_size = 1\n", + ) + .unwrap(); + fs::write( + directory.path().join(".hidden.rs"), + b"// hidden and larger than one byte\n", + ) + .unwrap(); + let output = run(directory.path(), &["check", "."]); + assert_eq!(output.status.code(), Some(1)); + assert!(String::from_utf8_lossy(&output.stdout).contains(".hidden.rs")); +} + +/// The target a command with no PATH stands in for is not an explicitly named +/// one: `.` substituted for a missing argument walks with the ordinary hidden +/// and size limits, so a bare run reports what a run naming its files would. +/// Naming the same directory is a request, and still bypasses both. +#[test] +fn an_implicit_target_keeps_the_hidden_and_size_limits() { + let directory = tempfile::tempdir().unwrap(); + fs::write( + directory.path().join(".ocomment.toml"), + b"version = 1\n[files]\nmax_size = 1000\n", + ) + .unwrap(); + fs::create_dir(directory.path().join(".hidden")).unwrap(); + fs::write( + directory.path().join(".hidden/b.rs"), + b"let b = 1; // hidden\n", + ) + .unwrap(); + fs::create_dir(directory.path().join("src")).unwrap(); + fs::write( + directory.path().join("src/a.rs"), + b"let a = 1; // remove me\n", + ) + .unwrap(); + let mut big = String::from("// oversized\n"); + while big.len() <= 100_000 { + big.push_str("let x = 1;\n"); + } + fs::write(directory.path().join("src/big.rs"), big.as_bytes()).unwrap(); + + let bare = run(directory.path(), &[]); + let stdout = String::from_utf8_lossy(&bare.stdout).into_owned(); + let stderr = String::from_utf8_lossy(&bare.stderr).into_owned(); + assert_eq!( + bare.status.code(), + Some(1), + "bare run said:\n{stdout}{stderr}" + ); + assert!( + stdout.contains("src/a.rs:1:12: removable line comment"), + "a bare run missed the one file it should report:\n{stdout}" + ); + assert!( + !stdout.contains(".hidden"), + "a bare run reached into a hidden directory:\n{stdout}" + ); + assert!( + !stdout.contains("big.rs"), + "a bare run scanned a file over files.max_size:\n{stdout}" + ); + assert!( + stderr.contains("Found 1 removable comment in 1 file (1 file scanned)."), + "a bare run counted more than the one file it may walk:\n{stderr}" + ); + assert!( + stderr.contains("1 file skipped (too large: 1"), + "a bare run did not fold the oversized file into its skips:\n{stderr}" + ); +} + +/// `.git` is hidden, so nothing a bare run does may look inside it — and `fix` +/// least of all: the sample hooks git writes into a fresh repository are full +/// of comments, and rewriting them is not what "fix my project" asked for. +#[test] +fn a_bare_run_never_reaches_into_the_git_directory() { + let directory = tempfile::tempdir().unwrap(); + git(directory.path(), &["init"]); + fs::write(directory.path().join(".ocomment.toml"), b"version = 1\n").unwrap(); + let hook = directory.path().join(".git/hooks/x.sample"); + fs::write(&hook, b"let x = 1; // sample hook comment\n").unwrap(); + let before = fs::read(&hook).unwrap(); + fs::write(directory.path().join("a.rs"), b"let a = 1; // remove me\n").unwrap(); + + let check = run(directory.path(), &[]); + let listing = String::from_utf8_lossy(&check.stdout).into_owned(); + assert!( + !listing.contains(".git"), + "a bare check listed something under .git:\n{listing}" + ); + + let fixed = run(directory.path(), &["fix"]); + let report = String::from_utf8_lossy(&fixed.stdout).into_owned(); + assert!( + !report.contains(".git"), + "a bare fix reported something under .git:\n{report}" + ); + assert_eq!( + fs::read(&hook).unwrap(), + before, + "a bare fix rewrote a file under .git" + ); +} + +/// Naming the directory lifts the hidden-file rule, and so does `files.hidden`; +/// neither may lift the one that keeps git's own storage out of a walk. `git` +/// itself never offers `.git` as a candidate for anything, and a tool that +/// rewrites files in place may do so least of all: `ocomment fix .` in a fresh +/// repository would otherwise rewrite every sample hook git had just written. +#[test] +fn a_named_directory_never_reaches_into_the_git_directory() { + for configuration in ["version = 1\n", "version = 1\n[files]\nhidden = true\n"] { + let directory = tempfile::tempdir().unwrap(); + git(directory.path(), &["init", "-q"]); + fs::write(directory.path().join(".ocomment.toml"), configuration).unwrap(); + let hook = directory.path().join(".git/hooks/x.sample"); + fs::write(&hook, b"#!/bin/sh\necho hi # sample hook comment\n").unwrap(); + let before = fs::read(&hook).unwrap(); + /* NOTE: A submodule or a linked worktree keeps its `.git` as a *file*; it + * points at git's storage and is no more a candidate than the + * directory it stands in for. */ + fs::create_dir(directory.path().join("vendor")).unwrap(); + fs::write( + directory.path().join("vendor/.git"), + b"gitdir: ../.git/modules/vendor\n", + ) + .unwrap(); + fs::write(directory.path().join("a.rs"), b"let a = 1; // remove me\n").unwrap(); + + let checked = run(directory.path(), &["check", "-v", "."]); + let listing = format!( + "{}{}", + String::from_utf8_lossy(&checked.stdout), + String::from_utf8_lossy(&checked.stderr) + ); + assert!( + !listing.contains(".git"), + "`check .` under {configuration:?} reached into git's storage:\n{listing}" + ); + assert!( + listing.contains("a.rs:1:12: removable line comment"), + "`check .` under {configuration:?} missed the project file:\n{listing}" + ); + + let fixed = run(directory.path(), &["fix", "."]); + let report = format!( + "{}{}", + String::from_utf8_lossy(&fixed.stdout), + String::from_utf8_lossy(&fixed.stderr) + ); + assert!( + !report.contains(".git"), + "`fix .` under {configuration:?} reported something under .git:\n{report}" + ); + assert_eq!( + fs::read(&hook).unwrap(), + before, + "`fix .` under {configuration:?} rewrote a file under .git" + ); + } +} + +/// The exclusion is about where a walk may wander, not about what a caller may +/// ask for. A path typed on the command line is a request, so a hook the +/// caller pointed at is still reported. +#[test] +fn a_path_named_inside_the_git_directory_is_still_honoured() { + let directory = tempfile::tempdir().unwrap(); + git(directory.path(), &["init", "-q"]); + let hook = directory.path().join(".git/hooks/x.sample"); + fs::write(&hook, b"#!/bin/sh\necho hi # sample hook comment\n").unwrap(); + + let output = run(directory.path(), &["check", ".git/hooks/x.sample"]); + let listing = String::from_utf8_lossy(&output.stdout).into_owned(); + assert_eq!( + output.status.code(), + Some(1), + "a named hook was not reported:\n{listing}" + ); + assert!( + listing.contains(".git/hooks/x.sample:2:9: removable line comment"), + "a named hook was not reported:\n{listing}" + ); +} + +#[cfg(unix)] +#[test] +fn symlink_following_is_explicitly_configurable() { + use std::os::unix::fs::symlink; + + let directory = tempfile::tempdir().unwrap(); + fs::write( + directory.path().join("source.txt"), + b"let x = 1; // remove\n", + ) + .unwrap(); + symlink("source.txt", directory.path().join("link.rs")).unwrap(); + + let skipped = run(directory.path(), &["check", "link.rs"]); + assert_eq!(skipped.status.code(), Some(0)); + assert!(String::from_utf8_lossy(&skipped.stdout).contains("symbolic link")); + + fs::write( + directory.path().join(".ocomment.toml"), + b"version = 1\n[files]\nfollow_symlinks = true\n", + ) + .unwrap(); + let followed = run(directory.path(), &["check", "link.rs"]); + assert_eq!(followed.status.code(), Some(1)); + assert!(String::from_utf8_lossy(&followed.stdout).contains("removable")); +} + +fn subcommand_lines(help: &str) -> Vec<(String, String)> { + let mut lines = help.lines().skip_while(|line| *line != "Commands:"); + lines.next(); + lines + .take_while(|line| !line.trim().is_empty()) + .map(|line| { + let trimmed = line.trim_start(); + match trimmed.split_once(" ") { + Some((name, description)) => (name.to_owned(), description.trim().to_owned()), + None => (trimmed.to_owned(), String::new()), + } + }) + .collect() +} + +#[test] +fn help_describes_every_subcommand() { + let directory = tempfile::tempdir().unwrap(); + let output = run(directory.path(), &["--help"]); + assert_eq!(output.status.code(), Some(0)); + let help = String::from_utf8(output.stdout).unwrap(); + let listed = subcommand_lines(&help); + assert!(!listed.is_empty(), "no Commands section in:\n{help}"); + for (name, description) in &listed { + assert!( + !description.is_empty(), + "subcommand `{name}` has no description in:\n{help}" + ); + } + let expected = [ + ("check", "Report removable comments (default command)"), + ( + "fix", + "Remove comments in place through an atomic, rollback-backed transaction", + ), + ("diff", "Print a unified diff of the changes fix would make"), + ( + "scan", + "List every comment with its kind, disposition and byte span", + ), + ( + "strip", + "Read source on stdin and write the stripped result to stdout", + ), + ("lsp", "Run the LSP 3.18 server over stdio"), + ( + "init", + "Write a starter .ocomment.toml or Lefthook configuration", + ), + ( + "config", + "Show, locate, explain, or export the resolved configuration", + ), + ( + "languages", + "List built-in languages, extensions, and dialects", + ), + ("plugin", "Manage sandboxed WASM scanner plugins"), + ("completions", "Generate shell completions"), + ( + "doctor", + "Diagnose the environment (config, git, plugins, tools)", + ), + ("man", "Render the roff manual page to stdout"), + ]; + for (name, description) in expected { + let found = listed + .iter() + .find(|(listed_name, _)| listed_name == name) + .unwrap_or_else(|| panic!("subcommand `{name}` is missing from:\n{help}")); + assert_eq!(found.1, description, "wrong description for `{name}`"); + } + assert!( + listed.iter().any(|(name, _)| name == "man"), + "the `man` subcommand must be documented in:\n{help}" + ); +} + +#[test] +fn help_documents_exit_status_files_and_examples() { + let directory = tempfile::tempdir().unwrap(); + let output = run(directory.path(), &["--help"]); + assert_eq!(output.status.code(), Some(0)); + let help = String::from_utf8(output.stdout).unwrap(); + for needle in [ + "EXIT STATUS", + "FILES", + "EXAMPLES", + ".ocomment.toml", + ".ocommentignore", + ".ocomment.lock", + ] { + assert!(help.contains(needle), "`--help` lacks {needle}:\n{help}"); + } +} + +#[test] +fn check_help_groups_options_and_lists_possible_values() { + let directory = tempfile::tempdir().unwrap(); + let short = run(directory.path(), &["check", "-h"]); + assert_eq!(short.status.code(), Some(0)); + let short = String::from_utf8(short.stdout).unwrap(); + assert!( + short.contains("[possible values: safe, legal, all]"), + "`check -h` lacks the policy values:\n{short}" + ); + assert!(short.contains("Policy:"), "no Policy heading:\n{short}"); + assert!(short.contains("Output:"), "no Output heading:\n{short}"); + + let long = run(directory.path(), &["check", "--help"]); + assert_eq!(long.status.code(), Some(0)); + let long = String::from_utf8(long.stdout).unwrap(); + assert!(long.contains("Policy:"), "no Policy heading:\n{long}"); + assert!(long.contains("Output:"), "no Output heading:\n{long}"); + for needle in [ + "- safe:", + "- legal:", + "- all:", + "- lines:", + "- rust:", + "- doc-line:", + ] { + assert!( + long.contains(needle), + "`check --help` lacks documented value {needle}:\n{long}" + ); + } +} + +#[test] +fn unknown_policy_value_reports_the_possible_values() { + let directory = tempfile::tempdir().unwrap(); + let output = run(directory.path(), &["--policy", "foo"]); + assert_eq!(output.status.code(), Some(2)); + let error = String::from_utf8_lossy(&output.stderr); + assert!(error.contains("invalid value 'foo'"), "{error}"); + assert!( + error.contains("[possible values: safe, legal, all]"), + "{error}" + ); +} + +#[test] +fn language_aliases_are_accepted_on_the_command_line() { + let directory = tempfile::tempdir().unwrap(); + fs::write( + directory.path().join("sample.rs"), + b"let x = 1; // remove\n", + ) + .unwrap(); + for alias in ["rs", "c++", "RUST"] { + let output = run( + directory.path(), + &["check", "sample.rs", "--language", alias], + ); + let error = String::from_utf8_lossy(&output.stderr); + assert!( + !error.contains("invalid value"), + "`--language {alias}` was rejected: {error}" + ); + assert_eq!( + output.status.code(), + Some(1), + "`--language {alias}`: {error}" + ); + } +} + +#[test] +fn unsupported_dialect_names_the_supported_ones() { + let directory = tempfile::tempdir().unwrap(); + fs::write( + directory.path().join("sample.rs"), + b"let x = 1; // remove\n", + ) + .unwrap(); + let output = run( + directory.path(), + &[ + "check", + "sample.rs", + "--language", + "rust", + "--dialect", + "jsx", + ], + ); + assert_eq!(output.status.code(), Some(2)); + let error = String::from_utf8_lossy(&output.stderr); + assert!( + error.contains("unsupported dialect `jsx` for rust"), + "{error}" + ); + assert!(error.contains("supported: standard"), "{error}"); +} + +#[test] +fn man_subcommand_renders_a_roff_page() { + let directory = tempfile::tempdir().unwrap(); + let output = run(directory.path(), &["man"]); + assert_eq!( + output.status.code(), + Some(0), + "{}", + String::from_utf8_lossy(&output.stderr) + ); + let page = String::from_utf8(output.stdout).unwrap(); + /* NOTE: roff requires the `\*(Aq` string definition before the title macro, so + * `.TH` is the first macro that is not a string definition. */ + let header = page + .lines() + .find(|line| !line.starts_with(".ie ") && !line.starts_with(".el ")) + .unwrap_or_default(); + assert!( + header.starts_with(".TH"), + "man page starts with:\n{page:.120}" + ); + assert!( + header.contains("ocomment"), + "the .TH header does not name the tool" + ); + assert!(page.contains(".SH NAME"), "man page has no NAME section"); +} + +/// The shipped page had an uppercase title, the "User Commands" manual, and a +/// SEE ALSO pointer; the generated page must keep all three. +#[test] +fn man_page_keeps_the_shipped_title_manual_and_see_also() { + let directory = tempfile::tempdir().unwrap(); + let output = run(directory.path(), &["man"]); + assert_eq!( + output.status.code(), + Some(0), + "{}", + String::from_utf8_lossy(&output.stderr) + ); + let page = String::from_utf8(output.stdout).unwrap(); + for needle in [ + ".TH OCOMMENT 1", + "User Commands", + "SEE ALSO", + "The complete schemas and guides are available in the OComment repository.", + ] { + assert!(page.contains(needle), "man page lacks {needle}:\n{page}"); + } +} + +#[test] +fn bash_completions_carry_the_policy_values() { + let directory = tempfile::tempdir().unwrap(); + let output = run(directory.path(), &["completions", "bash"]); + assert_eq!(output.status.code(), Some(0)); + let script = String::from_utf8(output.stdout).unwrap(); + for value in ["safe", "legal", "all"] { + assert!( + script.contains(value), + "bash completions lack the policy value {value}" + ); + } +} + +/// Rust `Debug` spellings that must never reach a terminal again. Bare +/// `Remove` is checked separately: SARIF legitimately says "Remove comment +/// with OComment" in its fix description. +const DEBUG_LEAKS: [&str; 3] = ["DocBlock", "Keep {", "Shebang"]; + +fn assert_no_debug_leak(context: &str, text: &str) { + for leak in DEBUG_LEAKS { + assert!( + !text.contains(leak), + "{context} leaks the Rust Debug token `{leak}`:\n{text}" + ); + } +} + +#[test] +fn human_check_names_comment_kinds_in_canonical_spelling() { + let directory = tempfile::tempdir().unwrap(); + fs::write( + directory.path().join("doc.rs"), + b"/** doc */\nfn main() {}\n", + ) + .unwrap(); + let output = run(directory.path(), &["check", "doc.rs"]); + assert_eq!(output.status.code(), Some(1)); + let stdout = String::from_utf8(output.stdout).unwrap(); + assert!( + stdout.contains("removable doc-block comment"), + "check output is:\n{stdout}" + ); + assert_no_debug_leak("human check output", &stdout); + assert!(!stdout.contains("Remove"), "check output is:\n{stdout}"); +} + +#[test] +fn human_scan_lines_use_canonical_kinds_and_dispositions() { + let directory = tempfile::tempdir().unwrap(); + fs::write( + directory.path().join("a.py"), + b"#!/usr/bin/env python3\nx = 1 # remove\n", + ) + .unwrap(); + let output = run(directory.path(), &["scan", "a.py"]); + assert_eq!(output.status.code(), Some(0)); + let stdout = String::from_utf8(output.stdout).unwrap(); + assert_eq!( + stdout, + "a.py:1:1: shebang keep (required source preamble) 0..22: #!/usr/bin/env python3\n\ + a.py:2:8: line remove 30..38: # remove\n" + ); + assert_no_debug_leak("human scan output", &stdout); +} + +#[test] +fn human_diagnostics_lowercase_the_severity() { + let directory = tempfile::tempdir().unwrap(); + fs::write(directory.path().join("broken.c"), b"int x; /* open").unwrap(); + let output = run(directory.path(), &["check", "broken.c"]); + assert_eq!(output.status.code(), Some(2)); + let stdout = String::from_utf8(output.stdout).unwrap(); + assert!( + stdout.contains("error[unterminated-comment]: unterminated block comment"), + "check output is:\n{stdout}" + ); + assert!( + !stdout.contains("Error["), + "check output still Debug-prints the severity:\n{stdout}" + ); +} + +#[test] +fn config_explain_prints_canonical_policy_and_layout() { + let directory = tempfile::tempdir().unwrap(); + let output = run(directory.path(), &["config", "explain"]); + assert_eq!(output.status.code(), Some(0)); + let stdout = String::from_utf8(output.stdout).unwrap(); + assert!( + stdout.contains("policy: safe; layout: lines"), + "config explain output is:\n{stdout}" + ); + /* NOTE: Only the policy line is pinned: the surrounding lines print filesystem + * paths that may legitimately contain any spelling. */ + let policy_line = stdout + .lines() + .find(|line| line.starts_with("policy:")) + .unwrap_or_else(|| panic!("config explain has no policy line:\n{stdout}")); + assert!( + !policy_line.contains("Safe") && !policy_line.contains("Lines"), + "config explain still Debug-prints the enums:\n{policy_line}" + ); +} + +#[test] +fn github_annotations_use_kebab_comment_kinds() { + let directory = tempfile::tempdir().unwrap(); + fs::write( + directory.path().join("doc.rs"), + b"/** doc */\nfn main() {}\n", + ) + .unwrap(); + let output = run(directory.path(), &["check", "doc.rs", "--format", "github"]); + assert_eq!(output.status.code(), Some(1)); + let stdout = String::from_utf8(output.stdout).unwrap(); + assert_eq!( + stdout, + "::notice file=doc.rs,line=1,col=1::removable doc-block comment\n" + ); + assert_no_debug_leak("github annotations", &stdout); + assert!(!stdout.contains("Remove"), "github output is:\n{stdout}"); +} + +#[test] +fn sarif_keeps_kebab_rule_ids_and_canonical_messages() { + let directory = tempfile::tempdir().unwrap(); + fs::write( + directory.path().join("doc.rs"), + b"/** doc */\nfn main() {}\n", + ) + .unwrap(); + let output = run(directory.path(), &["check", "doc.rs", "--format", "sarif"]); + assert_eq!(output.status.code(), Some(1)); + let value: serde_json::Value = serde_json::from_slice(&output.stdout).unwrap(); + let result = &value["runs"][0]["results"][0]; + assert_eq!(result["ruleId"], "removable-doc-block"); + assert_eq!(result["message"]["text"], "removable doc-block comment"); + assert_no_debug_leak("SARIF report", &String::from_utf8(output.stdout).unwrap()); +} + +/// A SARIF `fix` is an offer to rewrite the file, and a tool that takes it up +/// has to end with the bytes `ocomment fix` would have written. Under +/// `--layout compact` the edit is wider than the comment — a comment alone on +/// its line takes the whole line with it — so a fix cut to the comment's own +/// span would delete the comment and leave the blank line behind, which is the +/// output of a layout nobody asked for. The region reported is the edit's. +#[test] +fn a_sarif_fix_reproduces_what_fix_writes_under_compact_layout() { + let directory = tempfile::tempdir().unwrap(); + let source = "fn main() {\n // note\n let x = 1; // trailing\n}\n"; + fs::write(directory.path().join("main.rs"), source).unwrap(); + let reported = run( + directory.path(), + &[ + "check", "main.rs", "--layout", "compact", "--format", "sarif", + ], + ); + assert_eq!(reported.status.code(), Some(1)); + let value: serde_json::Value = serde_json::from_slice(&reported.stdout).unwrap(); + let results = value["runs"][0]["results"].as_array().unwrap(); + let mut replacements: Vec<(usize, usize, String)> = Vec::new(); + for result in results { + let replacement = &result["fixes"][0]["artifactChanges"][0]["replacements"][0]; + let region = &replacement["deletedRegion"]; + let number = |name: &str| usize::try_from(region[name].as_u64().unwrap()).unwrap(); + replacements.push(( + byte_offset(source, number("startLine"), number("startColumn")), + byte_offset(source, number("endLine"), number("endColumn")), + replacement["insertedContent"]["text"] + .as_str() + .unwrap() + .to_owned(), + )); + } + // NOTE: The whole-line comment: from the start of its line to the start of + // NOTE: the next one, so the line itself goes rather than being blanked. + assert_eq!( + replacements[0], + (12, 24, String::new()), + "the whole-line comment's fix is not the compact edit" + ); + let mut patched = String::new(); + let mut cursor = 0; + for (start, end, inserted) in &replacements { + assert!(*start >= cursor, "the SARIF fixes overlap"); + patched.push_str(&source[cursor..*start]); + patched.push_str(inserted); + cursor = *end; + } + patched.push_str(&source[cursor..]); + let fixed = run(directory.path(), &["fix", "main.rs", "--layout", "compact"]); + assert_eq!( + fixed.status.code(), + Some(0), + "{}", + String::from_utf8_lossy(&fixed.stderr) + ); + let written = fs::read_to_string(directory.path().join("main.rs")).unwrap(); + assert_eq!( + patched, written, + "applying the SARIF fixes is not what `ocomment fix --layout compact` writes" + ); +} + +/// Where a 1-based line and a 1-based byte column land in the source, so a +/// SARIF region can be turned back into the bytes it names. +fn byte_offset(source: &str, line: usize, column: usize) -> usize { + let mut offset = 0; + for _ in 1..line { + offset += source[offset..] + .find('\n') + .expect("the region names a line the source has") + + 1; + } + offset + column - 1 +} + +/// `strip` writes the stripped source and `config` answers a question about +/// the configuration; neither has a report to render, so a format that +/// describes one is refused rather than silently ignored — the way `languages` +/// refuses the same flags. +#[test] +fn strip_and_config_refuse_the_formats_they_cannot_honour() { + let directory = tempfile::tempdir().unwrap(); + for format in ["json", "jsonl", "sarif", "github"] { + let stripped = run_stdin( + directory.path(), + &["strip", "--language", "rust", "--format", format], + b"let x = 1; // note\n", + ); + assert_eq!( + stripped.status.code(), + Some(2), + "`strip --format {format}` was accepted" + ); + assert!( + stripped.stdout.is_empty(), + "`strip --format {format}` stripped the source anyway" + ); + let error = String::from_utf8(stripped.stderr).unwrap(); + assert!( + error.contains("`ocomment strip` is only available with --format human"), + "`strip --format {format}` said:\n{error}" + ); + for action in ["show", "locate", "explain", "schema"] { + let answered = run(directory.path(), &["config", action, "--format", format]); + assert_eq!( + answered.status.code(), + Some(2), + "`config {action} --format {format}` was accepted" + ); + assert!( + answered.stdout.is_empty(), + "`config {action} --format {format}` answered anyway" + ); + let error = String::from_utf8(answered.stderr).unwrap(); + assert!( + error.contains("`ocomment config` is only available with --format human"), + "`config {action} --format {format}` said:\n{error}" + ); + } + } + let stripped = run_stdin( + directory.path(), + &["strip", "--language", "rust", "--format", "human"], + b"let x = 1; // note\n", + ); + assert_eq!( + stripped.status.code(), + Some(0), + "{}", + String::from_utf8_lossy(&stripped.stderr) + ); + assert_eq!(stripped.stdout, b"let x = 1; \n"); + for action in ["show", "locate", "explain", "schema"] { + let answered = run(directory.path(), &["config", action, "--format", "human"]); + assert_eq!( + answered.status.code(), + Some(0), + "`config {action}` was refused: {}", + String::from_utf8_lossy(&answered.stderr) + ); + assert!( + !answered.stdout.is_empty(), + "`config {action}` answered with nothing" + ); + } +} + +/// Every comment kind a rule id can name, in the spelling `CommentKind` +/// serialises. A kind added without a rule to describe it fails this test. +const SARIF_KINDS: [&str; 11] = [ + "line", + "block", + "doc-line", + "doc-block", + "directive", + "license", + "html-comment", + "shebang", + "encoding", + "optimizer-hint", + "version-comment", +]; + +/// The SARIF failure levels OComment reports at. `none` is a level too, but +/// nothing OComment writes uses it. +const SARIF_LEVELS: [&str; 3] = ["error", "warning", "note"]; + +/// A code-scanning UI shows a finding through the rule it names: the title, the +/// sentence under it, and the link it offers all come from +/// `tool.driver.rules`, which a result reaches by `ruleIndex`. A rule the tool +/// never describes leaves the finding with nothing but its id, so every id a +/// run can emit is described and every result points at its own description. +#[test] +fn sarif_describes_every_rule_it_reports() { + let directory = tempfile::tempdir().unwrap(); + fs::write(directory.path().join("a.rs"), b"let value = 1; // remove\n").unwrap(); + fs::write(directory.path().join("bad.rs"), b"/* never ends\n").unwrap(); + fs::write(directory.path().join("plain.txt"), b"nothing to scan\n").unwrap(); + let output = run( + directory.path(), + &["check", "a.rs", "bad.rs", "plain.txt", "--format", "sarif"], + ); + assert_eq!(output.status.code(), Some(2)); + let report = String::from_utf8(output.stdout).unwrap(); + let document: serde_json::Value = serde_json::from_str(&report).unwrap(); + let driver = &document["runs"][0]["tool"]["driver"]; + assert_eq!(driver["name"], "ocomment"); + assert_eq!( + driver["version"], + env!("CARGO_PKG_VERSION"), + "the driver does not report the version that produced the run:\n{report}" + ); + assert_eq!( + driver["informationUri"], + "https://github.com/P4suta/OComment" + ); + + let rules = driver["rules"] + .as_array() + .unwrap_or_else(|| panic!("tool.driver.rules is not an array:\n{report}")); + let mut described = BTreeSet::new(); + for rule in rules { + let id = rule["id"] + .as_str() + .unwrap_or_else(|| panic!("a rule has no string id:\n{report}")); + assert!(described.insert(id.to_owned()), "`{id}` is described twice"); + for field in ["shortDescription", "fullDescription"] { + let text = rule[field]["text"].as_str().unwrap_or_default(); + assert!(!text.is_empty(), "rule `{id}` has no {field}:\n{report}"); + } + let help = rule["helpUri"].as_str().unwrap_or_default(); + assert!( + help.starts_with("https://"), + "rule `{id}` links nowhere: {help:?}" + ); + let level = rule["defaultConfiguration"]["level"] + .as_str() + .unwrap_or_default(); + assert!( + SARIF_LEVELS.contains(&level), + "rule `{id}` defaults to {level:?}, which is not a SARIF level" + ); + } + let removable: BTreeSet = described + .iter() + .filter(|id| id.starts_with("removable-")) + .cloned() + .collect(); + let expected: BTreeSet = SARIF_KINDS + .iter() + .map(|kind| format!("removable-{kind}")) + .collect(); + assert_eq!( + removable, expected, + "the rules do not describe exactly one removable kind each" + ); + let doc_block = rules + .iter() + .find(|rule| rule["id"] == "removable-doc-block") + .expect("`removable-doc-block` is described"); + assert_eq!( + doc_block["shortDescription"]["text"], + "Removable doc-block comment" + ); + assert_eq!(doc_block["defaultConfiguration"]["level"], "note"); + + let results = document["runs"][0]["results"].as_array().unwrap(); + let mut reported = BTreeSet::new(); + for result in results { + let id = result["ruleId"] + .as_str() + .unwrap_or_else(|| panic!("a result has no ruleId:\n{report}")); + let index = result["ruleIndex"] + .as_u64() + .unwrap_or_else(|| panic!("the `{id}` result has no ruleIndex:\n{report}")); + assert_eq!( + rules[index as usize]["id"], id, + "the `{id}` result points at rule {index}, which describes something else" + ); + let level = result["level"].as_str().unwrap_or_default(); + assert!( + SARIF_LEVELS.contains(&level), + "the `{id}` result is at level {level:?}, which is not a SARIF level" + ); + reported.insert(id.to_owned()); + } + for id in [ + "removable-line", + "removable-block", + "unterminated-comment", + "skipped-file", + ] { + assert!( + reported.contains(id), + "the run reported no `{id}` result:\n{report}" + ); + } + assert_no_debug_leak("SARIF report", &report); +} + +/// A file OComment cannot read is reported as a result too, and it names a rule +/// like any other finding. +#[cfg(unix)] +#[test] +fn sarif_describes_the_io_error_rule_when_it_reports_one() { + use std::os::unix::fs::PermissionsExt; + + let directory = tempfile::tempdir().unwrap(); + let unreadable = directory.path().join("locked.rs"); + fs::write(&unreadable, b"let value = 1; // remove\n").unwrap(); + fs::set_permissions(&unreadable, fs::Permissions::from_mode(0o000)).unwrap(); + let output = run( + directory.path(), + &["check", "locked.rs", "--format", "sarif"], + ); + assert_eq!(output.status.code(), Some(2)); + let report = String::from_utf8(output.stdout).unwrap(); + let document: serde_json::Value = serde_json::from_str(&report).unwrap(); + let result = &document["runs"][0]["results"][0]; + assert_eq!(result["ruleId"], "io-error"); + assert_eq!(result["level"], "error"); + let index = result["ruleIndex"] + .as_u64() + .unwrap_or_else(|| panic!("the io-error result has no ruleIndex:\n{report}")); + let rule = &document["runs"][0]["tool"]["driver"]["rules"][index as usize]; + assert_eq!(rule["id"], "io-error"); + assert_eq!(rule["defaultConfiguration"]["level"], "error"); +} + +/// A code-scanning UI resolves `artifactLocation.uri` against the checkout, so +/// a reported path has to be spelled the way the repository spells it: forward +/// slashes, no `./` standing in for the directory the run started in, and +/// `%SRCROOT%` saying what the rest is relative to. Every location in the +/// document is read that way, the ones under `fixes` included. +#[test] +fn sarif_locates_reported_files_under_the_source_root() { + let directory = tempfile::tempdir().unwrap(); + fs::create_dir(directory.path().join("sub")).unwrap(); + fs::write( + directory.path().join("sub/doc.rs"), + b"/** doc */\nfn main() {}\n", + ) + .unwrap(); + let output = run( + directory.path(), + &["check", "sub/./doc.rs", "--format", "sarif"], + ); + assert_eq!(output.status.code(), Some(1)); + let report = String::from_utf8(output.stdout).unwrap(); + let document: serde_json::Value = serde_json::from_str(&report).unwrap(); + let locations = artifact_locations(&document); + assert!( + !locations.is_empty(), + "the report locates nothing:\n{report}" + ); + for location in &locations { + let uri = location["uri"].as_str().unwrap_or_default(); + assert_eq!(uri, "sub/doc.rs", "a location is spelled {uri:?}"); + assert_eq!( + location["uriBaseId"], "%SRCROOT%", + "a relative location says nothing about what it is relative to:\n{report}" + ); + } +} + +/// A path the user typed as an absolute one is not under the checkout, so it +/// keeps its absolute spelling and names no base id — a base id would say it is +/// relative to the source root, which it is not. +#[test] +fn sarif_leaves_an_absolute_path_absolute_and_unbased() { + let directory = tempfile::tempdir().unwrap(); + let absolute = directory.path().join("a.rs"); + fs::write(&absolute, b"let value = 1; // remove\n").unwrap(); + let output = run( + directory.path(), + &["check", absolute.to_str().unwrap(), "--format", "sarif"], + ); + assert_eq!(output.status.code(), Some(1)); + let report = String::from_utf8(output.stdout).unwrap(); + let document: serde_json::Value = serde_json::from_str(&report).unwrap(); + let expected = absolute.to_string_lossy().replace('\\', "/"); + for location in artifact_locations(&document) { + assert_eq!(location["uri"], expected); + assert!( + location.get("uriBaseId").is_none(), + "an absolute location claims a base id:\n{report}" + ); + } +} + +/// Standard input has no place in the checkout either, so the pseudo-path it is +/// reported under is left alone rather than resolved against the source root. +#[test] +fn sarif_leaves_the_stdin_pseudo_path_unbased() { + let directory = tempfile::tempdir().unwrap(); + let output = run_stdin( + directory.path(), + &["check", "-", "--language", "rust", "--format", "sarif"], + b"let value = 1; // remove\n", + ); + assert_eq!(output.status.code(), Some(1)); + let report = String::from_utf8(output.stdout).unwrap(); + let document: serde_json::Value = serde_json::from_str(&report).unwrap(); + for location in artifact_locations(&document) { + assert_eq!(location["uri"], ""); + assert!( + location.get("uriBaseId").is_none(), + "the standard-input pseudo-path claims a base id:\n{report}" + ); + } +} + +/// A relative URI is read as a URI, and RFC 3986 gives a first segment holding +/// a colon back to the scheme: `c:/a.rs` parses as the scheme `c` rather than +/// as a path, and a Windows reader sees a drive letter in it besides. A POSIX +/// checkout is free to hold a directory named `c:`, so the emitter puts the one +/// `.` segment a URI is allowed to keep in front of that path — `./c:/a.rs`, +/// still measured from `%SRCROOT%` — and no reader can misread it. +/// +/// A GitHub annotation is matched against the paths the repository uses rather +/// than parsed as a URI, so `file=` keeps the plain spelling — with the `%3A` +/// the annotation format already owes a colon, which is a property delimiter +/// there. +#[cfg(unix)] +#[test] +fn sarif_disambiguates_a_leading_segment_that_reads_as_a_drive_letter() { + let directory = tempfile::tempdir().unwrap(); + fs::create_dir(directory.path().join("c:")).unwrap(); + fs::write( + directory.path().join("c:/a.rs"), + b"let value = 1; // remove\n", + ) + .unwrap(); + + let output = run(directory.path(), &["check", "c:/a.rs", "--format", "sarif"]); + let report = String::from_utf8(output.stdout).unwrap(); + assert_eq!( + output.status.code(), + Some(1), + "`check --format sarif` said:\n{report}" + ); + let document: serde_json::Value = serde_json::from_str(&report).unwrap(); + let locations = artifact_locations(&document); + assert!( + !locations.is_empty(), + "the report locates nothing:\n{report}" + ); + for location in &locations { + assert_eq!( + location["uri"], "./c:/a.rs", + "a drive-letter first segment was left ambiguous:\n{report}" + ); + assert_eq!( + location["uriBaseId"], "%SRCROOT%", + "the disambiguated path lost the base it is measured from:\n{report}" + ); + } + + let annotated = run( + directory.path(), + &["check", "c:/a.rs", "--format", "github"], + ); + let stdout = String::from_utf8(annotated.stdout).unwrap(); + assert!( + stdout.contains("::notice file=c%3A/a.rs,"), + "a GitHub annotation lost the path the repository spells:\n{stdout}" + ); +} + +/// Every `artifactLocation` in a SARIF document, from the locations a result +/// reports and from the changes its fix would make. +fn artifact_locations(document: &serde_json::Value) -> Vec { + let mut found = Vec::new(); + for run in document["runs"].as_array().into_iter().flatten() { + for result in run["results"].as_array().into_iter().flatten() { + for location in result["locations"].as_array().into_iter().flatten() { + found.push(location["physicalLocation"]["artifactLocation"].clone()); + } + for fix in result["fixes"].as_array().into_iter().flatten() { + for change in fix["artifactChanges"].as_array().into_iter().flatten() { + found.push(change["artifactLocation"].clone()); + } + } + } + } + found +} + +/// GitHub matches an annotation to a line of the diff by the path in `file=`, +/// and matches it against the paths the repository uses. A `./` the walk left +/// behind is enough to lose the annotation. +#[test] +fn github_annotations_report_repository_paths() { + let directory = tempfile::tempdir().unwrap(); + fs::create_dir(directory.path().join("sub")).unwrap(); + fs::write( + directory.path().join("sub/doc.rs"), + b"/** doc */\nfn main() {}\n", + ) + .unwrap(); + let output = run( + directory.path(), + &["check", "sub/./doc.rs", "--format", "github"], + ); + assert_eq!(output.status.code(), Some(1)); + let stdout = String::from_utf8(output.stdout).unwrap(); + assert_eq!( + stdout, + "::notice file=sub/doc.rs,line=1,col=1::removable doc-block comment\n" + ); +} + +#[test] +fn json_and_jsonl_serde_names_are_frozen() { + let directory = tempfile::tempdir().unwrap(); + fs::write( + directory.path().join("sample.py"), + b"#!/usr/bin/env python3\n# SPDX-License-Identifier: MIT\nx = 1 # remove\n", + ) + .unwrap(); + + let jsonl = run( + directory.path(), + &["scan", "sample.py", "--format", "jsonl"], + ); + assert_eq!(jsonl.status.code(), Some(0)); + assert_eq!( + String::from_utf8(jsonl.stdout).unwrap(), + concat!( + r#"{"path":"sample.py","language":"python","changed":true,"report":{"language":"python","#, + r#""comments":[{"span":{"start":0,"end":22},"kind":"shebang","disposition":{"action":"keep","#, + r#""reason":"required source preamble"}},{"span":{"start":23,"end":53},"kind":"license","#, + r#""disposition":{"action":"remove"}},{"span":{"start":61,"end":69},"kind":"line","#, + r#""disposition":{"action":"remove"}}],"diagnostics":[],"valid":true},"#, + r#""edits":[{"span":{"start":23,"end":53},"replacement":""},"#, + r#"{"span":{"start":61,"end":69},"replacement":""}],"#, + r#""source_map":{"segments":[{"original":{"start":0,"end":23},"output":{"start":0,"end":23},"exact":true},"#, + r#"{"original":{"start":23,"end":53},"output":{"start":23,"end":23},"exact":false},"#, + r#"{"original":{"start":53,"end":61},"output":{"start":23,"end":31},"exact":true},"#, + r#"{"original":{"start":61,"end":69},"output":{"start":31,"end":31},"exact":false},"#, + r#"{"original":{"start":69,"end":70},"output":{"start":31,"end":32},"exact":true}]}}"#, + "\n" + ), + "the JSONL protocol changed" + ); + + let json = run(directory.path(), &["scan", "sample.py", "--format", "json"]); + assert_eq!(json.status.code(), Some(0)); + let value: serde_json::Value = serde_json::from_slice(&json.stdout).unwrap(); + let comments = value["files"][0]["report"]["comments"].as_array().unwrap(); + assert_eq!( + comments + .iter() + .map(|comment| comment["kind"].as_str().unwrap()) + .collect::>(), + ["shebang", "license", "line"] + ); + assert_eq!(comments[0]["disposition"]["action"], "keep"); + assert_eq!( + comments[0]["disposition"]["reason"], + "required source preamble" + ); + assert_eq!(comments[1]["disposition"]["action"], "remove"); + assert_eq!(value["files"][0]["language"], "python"); +} + +#[test] +fn json_diagnostics_keep_the_lower_case_serde_severity() { + let directory = tempfile::tempdir().unwrap(); + fs::write(directory.path().join("broken.c"), b"int x; /* open").unwrap(); + let output = run(directory.path(), &["scan", "broken.c", "--format", "json"]); + assert_eq!(output.status.code(), Some(2)); + let value: serde_json::Value = serde_json::from_slice(&output.stdout).unwrap(); + let diagnostic = &value["files"][0]["report"]["diagnostics"][0]; + assert_eq!(diagnostic["severity"], "error"); + assert_eq!(diagnostic["code"], "unterminated-comment"); +} + +/// The end-of-run summary belongs on standard error so that `check` keeps a +/// grep-able `path:line:col` stream on standard output. +#[test] +fn check_writes_its_summary_to_standard_error() { + let directory = tempfile::tempdir().unwrap(); + fs::write( + directory.path().join("sample.rs"), + b"let x = 1; // remove\n", + ) + .unwrap(); + let output = run(directory.path(), &["check", "sample.rs"]); + assert_eq!(output.status.code(), Some(1)); + let stdout = String::from_utf8(output.stdout).unwrap(); + let stderr = String::from_utf8(output.stderr).unwrap(); + assert!( + stdout.contains("removable line comment"), + "check output is:\n{stdout}" + ); + assert!( + !stdout.contains("Found"), + "the summary leaked onto stdout:\n{stdout}" + ); + assert_eq!( + stderr, + "Found 1 removable comment in 1 file (1 file scanned). \ + Run `ocomment fix` to remove it.\n" + ); +} + +#[test] +fn a_clean_check_summarizes_the_files_it_scanned() { + let directory = tempfile::tempdir().unwrap(); + fs::write(directory.path().join("sample.rs"), b"let x = 1;\n").unwrap(); + let output = run(directory.path(), &["check", "sample.rs"]); + assert_eq!(output.status.code(), Some(0)); + assert_eq!( + String::from_utf8(output.stderr).unwrap(), + "No removable comments in 1 file.\n" + ); + assert_eq!(String::from_utf8(output.stdout).unwrap(), ""); +} + +/// `diff` must keep standard output a clean patch. +#[test] +fn diff_keeps_the_patch_on_stdout_and_summarizes_on_stderr() { + let directory = tempfile::tempdir().unwrap(); + fs::write( + directory.path().join("sample.rs"), + b"let x = 1; // remove\n", + ) + .unwrap(); + let output = run(directory.path(), &["diff", "sample.rs"]); + assert_eq!(output.status.code(), Some(1)); + let stdout = String::from_utf8(output.stdout).unwrap(); + assert!(stdout.starts_with("--- a/sample.rs"), "diff is:\n{stdout}"); + assert!( + !stdout.contains("Found"), + "the summary leaked into the patch:\n{stdout}" + ); + assert_eq!( + String::from_utf8(output.stderr).unwrap(), + "Found 1 removable comment in 1 file (1 file scanned). \ + Run `ocomment fix` to apply the patch.\n" + ); +} + +#[test] +fn fix_reports_every_changed_file_and_summarizes_on_stderr() { + let directory = tempfile::tempdir().unwrap(); + fs::write( + directory.path().join("sample.rs"), + b"let x = 1; // remove\n", + ) + .unwrap(); + let output = run(directory.path(), &["fix", "sample.rs"]); + assert_eq!( + output.status.code(), + Some(0), + "{}", + String::from_utf8_lossy(&output.stderr) + ); + assert_eq!( + String::from_utf8(output.stdout).unwrap(), + "fixed sample.rs: removed 1 comment\n" + ); + assert_eq!( + String::from_utf8(output.stderr).unwrap(), + "Removed 1 comment in 1 file (1 file scanned).\n" + ); +} + +#[test] +fn a_clean_fix_says_there_was_nothing_to_do() { + let directory = tempfile::tempdir().unwrap(); + fs::write(directory.path().join("sample.rs"), b"let x = 1;\n").unwrap(); + let output = run(directory.path(), &["fix", "sample.rs"]); + assert_eq!(output.status.code(), Some(0)); + assert_eq!(String::from_utf8(output.stdout).unwrap(), ""); + assert_eq!( + String::from_utf8(output.stderr).unwrap(), + "Nothing to fix in 1 file.\n" + ); +} + +#[test] +fn scan_summarizes_the_comment_counts() { + let directory = tempfile::tempdir().unwrap(); + fs::write( + directory.path().join("a.py"), + b"#!/usr/bin/env python3\nx = 1 # remove\n", + ) + .unwrap(); + let output = run(directory.path(), &["scan", "a.py"]); + assert_eq!(output.status.code(), Some(0)); + assert_eq!( + String::from_utf8(output.stderr).unwrap(), + "Scanned 1 file: 2 comments (1 removable, 1 kept).\n" + ); +} + +#[test] +fn quiet_silences_a_check_that_still_exits_one() { + let directory = tempfile::tempdir().unwrap(); + fs::write( + directory.path().join("sample.rs"), + b"let x = 1; // remove\n", + ) + .unwrap(); + let output = run(directory.path(), &["check", "-q", "sample.rs"]); + assert_eq!(output.status.code(), Some(1)); + assert_eq!(String::from_utf8(output.stdout).unwrap(), ""); + assert_eq!(String::from_utf8(output.stderr).unwrap(), ""); +} + +#[test] +fn quiet_and_verbose_cannot_be_combined() { + let directory = tempfile::tempdir().unwrap(); + let output = run(directory.path(), &["check", "-q", "-v"]); + assert_eq!(output.status.code(), Some(2)); + assert!( + String::from_utf8_lossy(&output.stderr).contains("cannot be used with"), + "{}", + String::from_utf8_lossy(&output.stderr) + ); +} + +#[test] +fn verbose_traces_the_root_target_config_and_kinds() { + let directory = tempfile::tempdir().unwrap(); + fs::write( + directory.path().join("sample.rs"), + b"let x = 1; // remove\n", + ) + .unwrap(); + let output = run(directory.path(), &["check", "-v", "sample.rs"]); + assert_eq!(output.status.code(), Some(1)); + let stderr = String::from_utf8(output.stderr).unwrap(); + assert!(stderr.contains("root: "), "verbose trace is:\n{stderr}"); + assert!( + stderr.contains("target: sample.rs"), + "verbose trace is:\n{stderr}" + ); + assert!(stderr.contains("config: "), "verbose trace is:\n{stderr}"); + assert!( + stderr.contains("kinds: line 1 removable"), + "verbose trace is:\n{stderr}" + ); +} + +#[test] +fn directory_walks_fold_skipped_files_into_the_summary() { + let directory = tempfile::tempdir().unwrap(); + fs::write( + directory.path().join("sample.rs"), + b"let x = 1; // remove\n", + ) + .unwrap(); + fs::write(directory.path().join("notes.unknownext"), b"# notes\n").unwrap(); + let output = run(directory.path(), &["check", "."]); + assert_eq!(output.status.code(), Some(1)); + let stdout = String::from_utf8(output.stdout).unwrap(); + let stderr = String::from_utf8(output.stderr).unwrap(); + assert!( + !stdout.contains("notes.unknownext"), + "a walked skip was listed individually:\n{stdout}" + ); + assert!( + stderr.contains("1 file skipped (unknown language: 1; use -v to list)."), + "summary is:\n{stderr}" + ); +} + +#[test] +fn verbose_lists_the_folded_skips() { + let directory = tempfile::tempdir().unwrap(); + fs::write( + directory.path().join("sample.rs"), + b"let x = 1; // remove\n", + ) + .unwrap(); + fs::write(directory.path().join("notes.unknownext"), b"# notes\n").unwrap(); + let output = run(directory.path(), &["check", "-v", "."]); + assert_eq!(output.status.code(), Some(1)); + let stdout = String::from_utf8(output.stdout).unwrap(); + assert!( + stdout.contains(&format!("notes.unknownext: skipped: {NO_LANGUAGE}")), + "verbose check output is:\n{stdout}" + ); +} + +#[test] +fn an_explicit_unknown_language_argument_is_still_listed() { + let directory = tempfile::tempdir().unwrap(); + fs::write(directory.path().join("notes.unknownext"), b"# notes\n").unwrap(); + let output = run(directory.path(), &["check", "notes.unknownext"]); + assert_eq!(output.status.code(), Some(0)); + let stdout = String::from_utf8(output.stdout).unwrap(); + assert!( + stdout.contains(&format!("notes.unknownext: skipped: {NO_LANGUAGE}")), + "check output is:\n{stdout}" + ); +} + +#[test] +fn machine_formats_never_emit_the_summary() { + let directory = tempfile::tempdir().unwrap(); + fs::write( + directory.path().join("sample.rs"), + b"let x = 1; // remove\n", + ) + .unwrap(); + for format in ["json", "jsonl", "sarif", "github"] { + let output = run( + directory.path(), + &[ + "check", + "sample.rs", + "-v", + "--progress", + "always", + "--format", + format, + ], + ); + assert_eq!( + String::from_utf8(output.stderr).unwrap(), + "", + "the {format} format wrote to standard error" + ); + } +} + +/// A single invalid file blocks the whole transaction, so the summary must not +/// claim removals that never reached the disk. +#[test] +fn a_blocked_fix_does_not_claim_removals() { + let directory = tempfile::tempdir().unwrap(); + let good = directory.path().join("good.rs"); + let original = b"let x = 1; // remove\n"; + fs::write(&good, original).unwrap(); + fs::write(directory.path().join("broken.c"), b"int x; /* open").unwrap(); + let output = run(directory.path(), &["fix", "."]); + assert_eq!(output.status.code(), Some(2)); + assert_eq!(fs::read(&good).unwrap(), original); + let stdout = String::from_utf8(output.stdout).unwrap(); + let stderr = String::from_utf8(output.stderr).unwrap(); + assert!( + !stdout.contains("fixed "), + "fix claimed a write that was blocked:\n{stdout}" + ); + assert!( + stderr.contains("1 file has invalid syntax; nothing was written for it (use --force-invalid to apply known-safe edits)."), + "summary is:\n{stderr}" + ); + assert!( + !stderr.contains("Removed "), + "summary claims removals that were blocked:\n{stderr}" + ); +} + +#[test] +fn staged_runs_also_emit_the_summary() { + let directory = repository(); + fs::write( + directory.path().join("sample.rs"), + b"let x = 1; // remove\n", + ) + .unwrap(); + git(directory.path(), &["add", "sample.rs"]); + let output = run(directory.path(), &["check", "--staged"]); + assert_eq!(output.status.code(), Some(1)); + assert_eq!( + String::from_utf8(output.stderr).unwrap(), + "Found 1 removable comment in 1 file (1 file scanned). \ + Run `ocomment fix` to remove it.\n" + ); +} + +#[test] +fn help_lists_the_verbosity_and_progress_flags() { + let directory = tempfile::tempdir().unwrap(); + let output = run(directory.path(), &["check", "--help"]); + assert_eq!(output.status.code(), Some(0)); + let help = String::from_utf8(output.stdout).unwrap(); + for needle in ["-q, --quiet", "-v, --verbose", "--progress "] { + assert!( + help.contains(needle), + "`check --help` lacks {needle}:\n{help}" + ); + } + assert!( + help.contains("live scanning counter"), + "`--progress` does not say that it draws the live counter:\n{help}" + ); + assert!( + !help.contains("progress indicator"), + "`--progress` describes the live counter it draws, not a vague \ + indicator:\n{help}" + ); +} + +/// `-q` does not print nothing: it drops the commentary and keeps whatever the +/// command was asked to produce — the findings, the patch, the listing. A help +/// line that claims otherwise sends a reader hunting for output that was never +/// dropped, or piping a run they think is silent. +#[test] +fn quiet_help_says_what_it_keeps() { + let directory = tempfile::tempdir().unwrap(); + let output = run(directory.path(), &["check", "--help"]); + assert_eq!(output.status.code(), Some(0)); + let help = String::from_utf8(output.stdout).unwrap(); + assert!( + help.contains("Drop the run summary and notes"), + "`-q` does not say what it drops:\n{help}" + ); + assert!( + help.contains("still written"), + "`-q` does not say what it keeps:\n{help}" + ); + assert!( + !help.contains("Print nothing but errors"), + "`-q` still claims to print nothing:\n{help}" + ); +} + +/// Fill a directory with `count` one-comment Rust files. +fn many_files(count: usize) -> TempDir { + let directory = tempfile::tempdir().unwrap(); + for index in 0..count { + fs::write( + directory.path().join(format!("file{index:03}.rs")), + b"let x = 1; // remove\n", + ) + .unwrap(); + } + directory +} + +/// `--progress always` draws a live counter on standard error and still leaves +/// the end-of-run summary readable once the counter line is cleared. +#[test] +fn progress_always_draws_a_live_counter_and_keeps_the_summary() { + let directory = many_files(120); + let output = run(directory.path(), &["check", "--progress", "always", "."]); + assert_eq!(output.status.code(), Some(1)); + let stderr = String::from_utf8(output.stderr).unwrap(); + assert!( + stderr.contains("ocomment: scanning 120/120 files"), + "progress counter is missing from:\n{stderr:?}" + ); + assert!( + stderr.contains("\r\x1b[2K"), + "the counter line was never cleared:\n{stderr:?}" + ); + assert!( + stderr.ends_with( + "Found 120 removable comments in 120 files (120 files scanned). \ + Run `ocomment fix` to remove them.\n" + ), + "summary is missing from:\n{stderr:?}" + ); +} + +#[test] +fn progress_never_draws_nothing_and_quiet_wins_over_progress() { + let directory = many_files(120); + let output = run(directory.path(), &["check", "--progress", "never", "."]); + assert_eq!(output.status.code(), Some(1)); + let stderr = String::from_utf8(output.stderr).unwrap(); + assert!( + !stderr.contains("scanning"), + "`--progress never` still drew a counter:\n{stderr:?}" + ); + let quiet = run( + directory.path(), + &["check", "-q", "--progress", "always", "."], + ); + assert_eq!(quiet.status.code(), Some(1)); + assert_eq!( + String::from_utf8(quiet.stderr).unwrap(), + "", + "`-q` did not silence the progress counter" + ); +} + +/// The human report says what the comment is, not merely that one is there. +#[test] +fn check_previews_the_comment_text_on_the_reported_line() { + let directory = tempfile::tempdir().unwrap(); + fs::write( + directory.path().join("sample.rs"), + b"let x = 1; // TODO remove\n", + ) + .unwrap(); + let output = run(directory.path(), &["check", "sample.rs"]); + assert_eq!(output.status.code(), Some(1)); + let stdout = String::from_utf8(output.stdout).unwrap(); + assert!( + stdout.contains(": removable line comment: // TODO remove"), + "check output is:\n{stdout}" + ); + assert_eq!( + stdout, + "sample.rs:1:12: removable line comment: // TODO remove\n" + ); +} + +#[test] +fn no_preview_restores_the_bare_reported_line() { + let directory = tempfile::tempdir().unwrap(); + fs::write( + directory.path().join("sample.rs"), + b"let x = 1; // TODO remove\n", + ) + .unwrap(); + let output = run(directory.path(), &["check", "--no-preview", "sample.rs"]); + assert_eq!(output.status.code(), Some(1)); + assert_eq!( + String::from_utf8(output.stdout).unwrap(), + "sample.rs:1:12: removable line comment\n" + ); +} + +/// `scan` previews too, and a multi-line comment stays on one line. +#[test] +fn scan_previews_the_comment_text_folded_onto_one_line() { + let directory = tempfile::tempdir().unwrap(); + fs::write( + directory.path().join("block.c"), + b"int x; /* first\n second */\n", + ) + .unwrap(); + let output = run(directory.path(), &["scan", "block.c"]); + assert_eq!(output.status.code(), Some(0)); + let stdout = String::from_utf8(output.stdout).unwrap(); + assert!( + stdout.starts_with("block.c:1:8: block "), + "scan output is:\n{stdout}" + ); + assert!( + stdout.contains("7..28: /* first second */\n"), + "scan output is:\n{stdout}" + ); + assert_eq!(stdout.lines().count(), 1, "scan output is:\n{stdout}"); +} + +/// A comment carrying terminal escapes must not be able to drive the terminal. +#[test] +fn a_previewed_comment_cannot_inject_escape_sequences() { + let directory = tempfile::tempdir().unwrap(); + fs::write( + directory.path().join("evil.rs"), + b"let x = 1; // \x1b[31mred\x1b[0m boom\n", + ) + .unwrap(); + let output = run(directory.path(), &["check", "evil.rs"]); + assert_eq!(output.status.code(), Some(1)); + let stdout = String::from_utf8(output.stdout).unwrap(); + assert!( + !stdout.contains('\x1b'), + "an escape byte reached the terminal:\n{stdout:?}" + ); + assert!( + stdout.contains("// \u{fffd}[31mred\u{fffd}[0m boom"), + "check output is:\n{stdout:?}" + ); +} + +/// A file name is chosen by whoever made the file, so the half of a report +/// line that shows a path is untrusted input on its way to a terminal exactly +/// like the preview beside it. It gets the same treatment, and is cut nowhere: +/// a path ending in an ellipsis names no file. +#[cfg(unix)] +#[test] +fn a_reported_path_cannot_inject_escape_sequences() { + use std::os::unix::ffi::OsStringExt; + + let directory = tempfile::tempdir().unwrap(); + let name = std::ffi::OsString::from_vec(b"evil\x1b[2Jname.rs".to_vec()); + fs::write(directory.path().join(&name), b"let x = 1; // remove me\n").unwrap(); + let unreadable = std::ffi::OsString::from_vec(b"evil\x1b[2Jskip.bin".to_vec()); + fs::write(directory.path().join(&unreadable), b"\x00\x01binary\n").unwrap(); + + let output = run(directory.path(), &["check", "-v", "."]); + assert_eq!( + output.status.code(), + Some(1), + "{}", + String::from_utf8_lossy(&output.stderr) + ); + assert!( + !output.stdout.contains(&0x1b), + "an escape byte reached the report: {:?}", + String::from_utf8_lossy(&output.stdout) + ); + assert!( + !output.stderr.contains(&0x1b), + "an escape byte reached the summary: {:?}", + String::from_utf8_lossy(&output.stderr) + ); + let stdout = String::from_utf8(output.stdout).unwrap(); + assert!( + stdout.contains("evil\u{fffd}[2Jname.rs:1:12: removable line comment"), + "the report lost the file it names:\n{stdout}" + ); + assert!( + stdout.contains("evil\u{fffd}[2Jskip.bin: skipped: binary file"), + "the skip lost the file it names:\n{stdout}" + ); + + /* NOTE: Asked for hyperlinks, the report writes escape bytes of its own: the + * OSC 8 frame is delimited by them. They are the only ones it may write. + * The name goes into the frame's *target* as well as its text, so the + * target percent-encodes what it is given rather than forwarding it. */ + let linked = run( + directory.path(), + &["check", "-v", ".", "--hyperlinks", "always"], + ); + assert_eq!( + linked.status.code(), + Some(1), + "{}", + String::from_utf8_lossy(&linked.stderr) + ); + let linked_stdout = String::from_utf8(linked.stdout).unwrap(); + assert!( + linked_stdout.contains("\x1b]8;;file://"), + "no hyperlink was written to link a path with:\n{linked_stdout}" + ); + let unframed = linked_stdout.replace("\x1b]8;;", "").replace("\x1b\\", ""); + assert!( + !unframed.contains('\x1b'), + "an escape byte reached the report outside the hyperlink frame: {unframed:?}" + ); + assert!( + linked_stdout.contains("%1B%5B2Jname.rs"), + "the link target forwarded the name instead of encoding it:\n{linked_stdout}" + ); +} + +/// A file name is not commentary. The spaces and tabs in it are the name — a +/// reader who cannot see them cannot type the name back, and a report that +/// quietly drops them names a file the checkout does not have. Every control +/// character is still replaced, the tab included, so the row stays one row. +#[cfg(unix)] +#[test] +fn a_reported_path_keeps_the_spacing_of_the_name_it_reports() { + let directory = tempfile::tempdir().unwrap(); + fs::write( + directory.path().join("ta\tb.rs"), + b"let x = 1; // remove me\n", + ) + .unwrap(); + fs::write( + directory.path().join(" lead.rs"), + b"let x = 1; // remove me\n", + ) + .unwrap(); + + let output = run(directory.path(), &["check", "."]); + assert_eq!( + output.status.code(), + Some(1), + "{}", + String::from_utf8_lossy(&output.stderr) + ); + let stdout = String::from_utf8(output.stdout).unwrap(); + assert!( + stdout.contains("ta\u{fffd}b.rs:1:12: removable line comment"), + "the tab in a file name vanished from the report:\n{stdout}" + ); + assert!( + stdout + .lines() + .any(|line| line == " lead.rs:1:12: removable line comment: // remove me"), + "the leading space in a file name vanished from the report:\n{stdout}" + ); +} + +/// A directory and a file inside it are both named, so the walk meets the file +/// twice. It is one file: the report says so once, exactly as it does for a +/// file it can scan. +#[test] +fn a_file_reached_twice_is_skipped_once() { + let directory = tempfile::tempdir().unwrap(); + fs::write(directory.path().join("plain.txt"), b"nothing to scan\n").unwrap(); + + let output = run( + directory.path(), + &["check", ".", "plain.txt", "--format", "github"], + ); + assert_eq!( + output.status.code(), + Some(0), + "{}", + String::from_utf8_lossy(&output.stderr) + ); + let stdout = String::from_utf8(output.stdout).unwrap(); + let annotations: Vec<_> = stdout + .lines() + .filter(|line| line.contains("plain.txt")) + .collect(); + assert_eq!( + annotations.len(), + 1, + "one file was annotated more than once:\n{stdout}" + ); + assert!( + annotations[0].starts_with("::notice file=plain.txt,title=OComment skipped file::"), + "the skip was not reported as a notice:\n{stdout}" + ); +} + +/// A configuration file is read from the project, and the pattern in it is +/// echoed back on the line that rejects it. That makes it untrusted input on +/// its way to a terminal, and it is folded like every other one. +#[test] +fn an_invalid_policy_regex_cannot_inject_escape_sequences() { + let directory = tempfile::tempdir().unwrap(); + fs::write(directory.path().join("a.rs"), b"let x = 1; // remove me\n").unwrap(); + fs::write( + directory.path().join(".ocomment.toml"), + "version = 1\n[policy]\nkeep_regex = [\"\\u001B[2J(\"]\n", + ) + .unwrap(); + + let output = run(directory.path(), &["check", "a.rs"]); + assert_eq!( + output.status.code(), + Some(2), + "an invalid regex was accepted:\n{}", + String::from_utf8_lossy(&output.stderr) + ); + assert!( + !output.stderr.contains(&0x1b), + "an escape byte reached the terminal: {:?}", + String::from_utf8_lossy(&output.stderr) + ); + let error = String::from_utf8(output.stderr).unwrap(); + assert!( + error.contains("invalid comment policy regex `\u{fffd}[2J(`"), + "the error lost the pattern it rejects:\n{error}" + ); +} + +/// The same rule for the other pattern a project file carries. A `[files]` +/// glob is echoed back on the line that rejects it — twice, because `globset` +/// quotes the glob inside its own parse error — so both halves are folded +/// before either reaches a terminal. +#[test] +fn an_invalid_file_glob_cannot_inject_escape_sequences() { + let directory = tempfile::tempdir().unwrap(); + fs::write(directory.path().join("a.rs"), b"let x = 1; // remove me\n").unwrap(); + fs::write( + directory.path().join(".ocomment.toml"), + "version = 1\n[files]\nexclude = [\"\\u001B[2J[\"]\n", + ) + .unwrap(); + + let output = run(directory.path(), &["check", "a.rs"]); + assert_eq!( + output.status.code(), + Some(2), + "an invalid glob was accepted:\n{}", + String::from_utf8_lossy(&output.stderr) + ); + assert!( + !output.stderr.contains(&0x1b), + "an escape byte reached the terminal: {:?}", + String::from_utf8_lossy(&output.stderr) + ); + let error = String::from_utf8(output.stderr).unwrap(); + assert!( + error.contains("invalid file glob `\u{fffd}[2J[`"), + "the error lost the pattern it rejects:\n{error}" + ); + assert!( + error.lines().count() == 1, + "the error spread over more than one line:\n{error}" + ); +} + +/// The GitHub renderer annotates a pull request, and an annotation costs the +/// reader a line in the checks tab. So it folds a skip away exactly as the +/// human renderer does: an I/O error and a path the caller named are always +/// worth saying, while a file a walk merely wandered past is `-v` material. +#[test] +fn github_annotations_fold_walked_skips_away_unless_asked() { + let directory = tempfile::tempdir().unwrap(); + fs::write(directory.path().join("a.rs"), b"let x = 1; // remove\n").unwrap(); + fs::write(directory.path().join("notes.unknownext"), b"# notes\n").unwrap(); + + let quiet = run(directory.path(), &["check", "--format", "github"]); + let stdout = String::from_utf8(quiet.stdout).unwrap(); + assert!(stdout.contains("::notice file=a.rs"), "{stdout}"); + assert!( + !stdout.contains("notes.unknownext"), + "a walked skip was annotated without -v:\n{stdout}" + ); + + let loud = run(directory.path(), &["check", "--format", "github", "-v"]); + let verbose = String::from_utf8(loud.stdout).unwrap(); + assert!( + verbose.contains("::notice file=notes.unknownext,title=OComment skipped file::"), + "-v lost the walked skip:\n{verbose}" + ); + + let named = run( + directory.path(), + &["check", "notes.unknownext", "--format", "github"], + ); + let explicit = String::from_utf8(named.stdout).unwrap(); + assert!( + explicit.contains("::notice file=notes.unknownext,title=OComment skipped file::"), + "a path the caller named lost its annotation:\n{explicit}" + ); + + let missing = run( + directory.path(), + &["check", "gone.rs", "--format", "github"], + ); + let failure = String::from_utf8(missing.stdout).unwrap(); + assert!( + failure.contains("::error file=gone.rs,title=OComment I/O error::"), + "an I/O error lost its annotation:\n{failure}" + ); +} + +/// `-q` trims the human report down to what went wrong; it is a human-format +/// concept and has no business reaching a machine format. A GitHub annotation +/// is the product of `--format github`, so a hook that runs quietly still +/// annotates the path the caller named and the file it could not read — the +/// walked skip stays folded because `-v`, not `-q`, is what decides that. +#[test] +fn quiet_does_not_take_annotations_off_a_machine_format() { + let directory = tempfile::tempdir().unwrap(); + fs::write(directory.path().join("a.rs"), b"let x = 1; // remove\n").unwrap(); + fs::write(directory.path().join("notes.unknownext"), b"# notes\n").unwrap(); + + let named = run( + directory.path(), + &["check", "notes.unknownext", "--format", "github", "-q"], + ); + let explicit = String::from_utf8(named.stdout).unwrap(); + assert!( + explicit.contains("::notice file=notes.unknownext,title=OComment skipped file::"), + "-q took the annotation off a path the caller named:\n{explicit}" + ); + + let missing = run( + directory.path(), + &["check", "gone.rs", "--format", "github", "-q"], + ); + let failure = String::from_utf8(missing.stdout).unwrap(); + assert!( + failure.contains("::error file=gone.rs,title=OComment I/O error::"), + "-q took the annotation off an I/O error:\n{failure}" + ); + + let walked = run(directory.path(), &["check", "--format", "github", "-q"]); + let stdout = String::from_utf8(walked.stdout).unwrap(); + assert!(stdout.contains("::notice file=a.rs"), "{stdout}"); + assert!( + !stdout.contains("notes.unknownext"), + "a walked skip was annotated without -v:\n{stdout}" + ); +} + +/// The `regex` crate writes a parse error over several lines, with a caret +/// under the byte it stopped at. The caret means nothing once the pattern is +/// folded, but the sentence after it is the whole answer, so the report keeps +/// every word of it and puts the lot on the one line an error is. +#[test] +fn an_invalid_policy_regex_is_reported_whole_on_one_line() { + let directory = tempfile::tempdir().unwrap(); + fs::write( + directory.path().join(".ocomment.toml"), + "version = 1\n[policy]\nkeep_regex = [\"[\\u001Ba-\"]\n", + ) + .unwrap(); + + let output = run(directory.path(), &["check"]); + assert_eq!(output.status.code(), Some(2)); + assert!( + !output.stderr.contains(&0x1b), + "an escape byte reached the terminal: {:?}", + String::from_utf8_lossy(&output.stderr) + ); + let error = String::from_utf8(output.stderr).unwrap(); + assert_eq!( + error.trim_end().lines().count(), + 1, + "the parse error spilled over more than one line:\n{error}" + ); + assert!( + error.contains( + "invalid comment policy regex `[\u{fffd}a-`: \ + regex parse error: [\u{fffd}a- ^ error: unclosed character class" + ), + "the parse error was not folded, or was cut short of the reason:\n{error}" + ); +} + +/// The same rule for the file as a whole. A `toml` parse error quotes the +/// line it stopped on, and that line came out of a project file, so it carries +/// whatever bytes the file carries — an escape sequence among them — over four +/// lines of caret diagram. The verdict is one line, and every byte of it is +/// printable. +#[test] +fn an_invalid_configuration_is_reported_whole_on_one_line() { + let directory = tempfile::tempdir().unwrap(); + fs::write( + directory.path().join(".ocomment.toml"), + b"version = 1\n\n[files]\ninclude = [\"a\x07 b\"]\n", + ) + .unwrap(); + + let output = run(directory.path(), &["check"]); + assert_eq!(output.status.code(), Some(2)); + assert!( + !output.stderr.contains(&0x07), + "a control byte reached the terminal: {:?}", + String::from_utf8_lossy(&output.stderr) + ); + let error = String::from_utf8(output.stderr).unwrap(); + assert_eq!( + error.trim_end().lines().count(), + 1, + "the parse error spilled over more than one line:\n{error}" + ); + assert!( + error.contains("invalid configuration "), + "the error lost the file it rejects:\n{error}" + ); + assert!( + error.contains("include = [\"a\u{fffd} b\"]"), + "the quoted line was not folded, or lost the byte that is wrong with it:\n{error}" + ); + assert!( + error.contains("invalid basic string"), + "the error was cut short of the reason:\n{error}" + ); +} + +/// The other half of that line: the path in front of the colon. +/// +/// A configuration file is named by the directory it was found in, and a +/// directory name carries whatever bytes the file system allowed — a `\x07` +/// that rings the terminal's bell among them. The name is still the answer to +/// "which file?", so it is printed rather than withheld, and it gets the +/// treatment every other path in the report gets. +#[test] +fn an_invalid_configuration_names_its_file_without_ringing_the_terminal() { + let parent = tempfile::tempdir().unwrap(); + let directory = parent.path().join("ring\u{7}ing"); + fs::create_dir(&directory).unwrap(); + fs::write( + directory.join(".ocomment.toml"), + b"version = 1\n\n[files]\ninclude = [\n", + ) + .unwrap(); + + let output = run(&directory, &["check"]); + assert_eq!( + output.status.code(), + Some(2), + "{}", + String::from_utf8_lossy(&output.stderr) + ); + assert!( + !output.stderr.contains(&0x07), + "a control byte from the path reached the terminal: {:?}", + String::from_utf8_lossy(&output.stderr) + ); + let error = String::from_utf8(output.stderr).unwrap(); + assert!( + error.contains("invalid configuration ") && error.contains(".ocomment.toml"), + "the error lost the file it rejects:\n{error}" + ); + assert!( + error.contains("ring\u{fffd}ing"), + "the directory the file was found in was dropped from the error:\n{error}" + ); +} + +/// A long comment is cut to a readable width rather than flooding the report. +#[test] +fn a_long_comment_preview_is_truncated_with_an_ellipsis() { + let directory = tempfile::tempdir().unwrap(); + let comment = "x".repeat(200); + fs::write( + directory.path().join("long.rs"), + format!("let x = 1; // {comment}\n"), + ) + .unwrap(); + let output = run(directory.path(), &["check", "long.rs"]); + assert_eq!(output.status.code(), Some(1)); + let stdout = String::from_utf8(output.stdout).unwrap(); + let (_, preview) = stdout.trim_end().rsplit_once(": ").unwrap(); + assert!(preview.ends_with('…'), "preview is:\n{preview}"); + assert_eq!(preview.chars().count(), 72, "preview is:\n{preview}"); +} + +#[test] +fn help_documents_the_preview_switch() { + let directory = tempfile::tempdir().unwrap(); + let output = run(directory.path(), &["check", "--help"]); + assert_eq!(output.status.code(), Some(0)); + let help = String::from_utf8(output.stdout).unwrap(); + assert!( + help.contains("--no-preview"), + "`check --help` lacks --no-preview:\n{help}" + ); +} + +/// `clap_mangen` dumps `after_long_help` as one opaque `.SH EXTRA` blob; the +/// manual must carry the same content as real roff sections instead. +#[test] +fn man_page_renders_real_sections_instead_of_one_extra_blob() { + let directory = tempfile::tempdir().unwrap(); + let output = run(directory.path(), &["man"]); + assert_eq!( + output.status.code(), + Some(0), + "{}", + String::from_utf8_lossy(&output.stderr) + ); + let page = String::from_utf8(output.stdout).unwrap(); + for needle in [ + ".SH EXIT STATUS", + ".SH FILES", + ".SH EXAMPLES", + ".SH SEE ALSO", + ] { + assert!(page.contains(needle), "man page lacks {needle}:\n{page}"); + } + assert!( + !page.contains(".SH EXTRA"), + "the help blob is still dumped verbatim:\n{page}" + ); +} + +/// A bidirectional override can make a comment render as its own reverse; the +/// preview must neutralize the whole format-control class, not only C0. +#[test] +fn a_previewed_comment_cannot_reorder_the_line_with_bidi_controls() { + let directory = tempfile::tempdir().unwrap(); + fs::write( + directory.path().join("bidi.rs"), + "let x = 1; // \u{202e}drowssap\u{202c} end\n".as_bytes(), + ) + .unwrap(); + let output = run(directory.path(), &["check", "bidi.rs"]); + assert_eq!(output.status.code(), Some(1)); + let stdout = String::from_utf8(output.stdout).unwrap(); + assert!( + !stdout.contains('\u{202e}'), + "a bidi override reached the terminal:\n{stdout:?}" + ); + assert_eq!( + stdout, + "bidi.rs:1:12: removable line comment: // \u{fffd}drowssap\u{fffd} end\n" + ); +} + +/// An explicitly named skip already has its own line on standard output, so +/// the folded clause must not count it a second time. +#[test] +fn a_named_skip_is_not_counted_twice_in_the_summary() { + let directory = tempfile::tempdir().unwrap(); + fs::write( + directory.path().join("sample.rs"), + b"let x = 1; // remove\n", + ) + .unwrap(); + fs::write(directory.path().join("notes.unknownext"), b"# notes\n").unwrap(); + let output = run( + directory.path(), + &["check", "sample.rs", "notes.unknownext"], + ); + assert_eq!(output.status.code(), Some(1)); + let stdout = String::from_utf8(output.stdout).unwrap(); + let stderr = String::from_utf8(output.stderr).unwrap(); + assert!( + stdout.contains(&format!("notes.unknownext: skipped: {NO_LANGUAGE}")), + "check output is:\n{stdout}" + ); + assert_eq!( + stderr, + "Found 1 removable comment in 1 file (1 file scanned). \ + Run `ocomment fix` to remove it.\n" + ); +} + +/// Scanning nothing at all is not "no removable comments in 0 files": say what +/// actually happened to the files that were passed over. +#[test] +fn a_run_that_scans_nothing_reports_the_skips_instead() { + let directory = tempfile::tempdir().unwrap(); + fs::write(directory.path().join("notes.unknownext"), b"# notes\n").unwrap(); + let output = run(directory.path(), &["check", "."]); + assert_eq!(output.status.code(), Some(0)); + assert_eq!( + String::from_utf8(output.stderr).unwrap(), + "Nothing to check: 1 file skipped (unknown language: 1; use -v to list).\n" + ); +} + +/// Every noun in the summary is pluralized; `file(s)` never reaches a user. +#[test] +fn the_summary_pluralizes_every_noun() { + let directory = tempfile::tempdir().unwrap(); + fs::write( + directory.path().join("a.rs"), + b"let x = 1; // one\nlet y = 2; // two\n", + ) + .unwrap(); + fs::write(directory.path().join("b.rs"), b"let z = 3; // three\n").unwrap(); + let output = run(directory.path(), &["check", "a.rs", "b.rs"]); + assert_eq!(output.status.code(), Some(1)); + let stderr = String::from_utf8(output.stderr).unwrap(); + assert_eq!( + stderr, + "Found 3 removable comments in 2 files (2 files scanned). \ + Run `ocomment fix` to remove them.\n" + ); + assert!(!stderr.contains("(s)"), "summary is:\n{stderr}"); +} + +/// An unreadable path is an I/O error, and the summary must own up to it. +#[cfg(unix)] +#[test] +fn the_summary_counts_io_errors() { + let directory = tempfile::tempdir().unwrap(); + fs::write( + directory.path().join("sample.rs"), + b"let x = 1; // remove\n", + ) + .unwrap(); + let output = run(directory.path(), &["check", "sample.rs", "missing.rs"]); + assert_eq!(output.status.code(), Some(2)); + let stderr = String::from_utf8(output.stderr).unwrap(); + assert!( + stderr.contains("1 I/O error."), + "the summary hides the I/O error:\n{stderr}" + ); +} + +/// `-q` silences the chatter, never the product: a patch is the whole point of +/// `diff`, so it survives. +#[test] +fn quiet_diff_still_writes_the_patch() { + let directory = tempfile::tempdir().unwrap(); + fs::write( + directory.path().join("sample.rs"), + b"let x = 1; // remove\n", + ) + .unwrap(); + let output = run(directory.path(), &["diff", "-q", "sample.rs"]); + assert_eq!(output.status.code(), Some(1)); + let stdout = String::from_utf8(output.stdout).unwrap(); + assert!(stdout.starts_with("--- a/sample.rs"), "diff is:\n{stdout}"); + assert!( + stdout.contains("-let x = 1; // remove"), + "diff is:\n{stdout}" + ); + assert_eq!(String::from_utf8(output.stderr).unwrap(), ""); +} + +/// The same rule for `scan`: the listing is the product. +#[test] +fn quiet_scan_still_writes_the_listing() { + let directory = tempfile::tempdir().unwrap(); + fs::write( + directory.path().join("a.py"), + b"#!/usr/bin/env python3\nx = 1 # remove\n", + ) + .unwrap(); + let output = run(directory.path(), &["scan", "-q", "a.py"]); + assert_eq!(output.status.code(), Some(0)); + let stdout = String::from_utf8(output.stdout).unwrap(); + assert_eq!(stdout.lines().count(), 2, "scan output is:\n{stdout}"); + assert!( + stdout.contains("a.py:2:8: line remove "), + "scan output is:\n{stdout}" + ); + assert_eq!(String::from_utf8(output.stderr).unwrap(), ""); +} + +/// Nothing was scanned and the only skip was named on the command line, where +/// it already has its own line: the summary says so without repeating it. +#[test] +fn a_run_of_only_named_skips_does_not_repeat_them() { + let directory = tempfile::tempdir().unwrap(); + fs::write(directory.path().join("notes.unknownext"), b"# notes\n").unwrap(); + let output = run(directory.path(), &["check", "notes.unknownext"]); + assert_eq!(output.status.code(), Some(0)); + assert!( + String::from_utf8(output.stdout) + .unwrap() + .contains(&format!("notes.unknownext: skipped: {NO_LANGUAGE}")) + ); + assert_eq!( + String::from_utf8(output.stderr).unwrap(), + "Nothing to check.\n" + ); +} + +/// `-` in the PATH list is standard input: it is scanned like any other file +/// and reported under the pseudo path ``. +#[test] +fn a_dash_reads_standard_input_as_a_file() { + let directory = tempfile::tempdir().unwrap(); + let output = run_stdin( + directory.path(), + &["check", "--language", "rust", "-"], + b"let x = 1; // note\n", + ); + assert_eq!( + output.status.code(), + Some(1), + "{}", + String::from_utf8_lossy(&output.stderr) + ); + assert_eq!( + String::from_utf8(output.stdout).unwrap(), + ":1:12: removable line comment: // note\n" + ); +} + +/// The patch for standard input names the same pseudo path. +#[test] +fn a_dash_diffs_standard_input_under_the_pseudo_path() { + let directory = tempfile::tempdir().unwrap(); + let output = run_stdin( + directory.path(), + &["diff", "--language", "rust", "-"], + b"let x = 1; // note\n", + ); + assert_eq!(output.status.code(), Some(1)); + let stdout = String::from_utf8(output.stdout).unwrap(); + assert!(stdout.starts_with("--- a/\n"), "diff is:\n{stdout}"); + assert!(stdout.contains("-let x = 1; // note"), "diff is:\n{stdout}"); +} + +/// The machine formats carry the pseudo path too, so a piped run is as +/// scriptable as a walked one. +#[test] +fn a_dash_names_standard_input_in_json() { + let directory = tempfile::tempdir().unwrap(); + let output = run_stdin( + directory.path(), + &["check", "--format", "json", "--language", "rust", "-"], + b"let x = 1; // note\n", + ); + assert_eq!(output.status.code(), Some(1)); + let stdout = String::from_utf8(output.stdout).unwrap(); + assert!( + stdout.contains("\"path\": \"\""), + "json is:\n{stdout}" + ); +} + +/// Standard input has no name to detect a language from, so bytes that carry +/// no signature are a usage error with an actionable message. +#[test] +fn undetectable_standard_input_asks_for_a_language() { + let directory = tempfile::tempdir().unwrap(); + let output = run_stdin(directory.path(), &["check", "-"], b"let x = 1; // note\n"); + assert_eq!(output.status.code(), Some(2)); + assert_eq!( + String::from_utf8(output.stderr).unwrap(), + "ocomment: cannot detect the language of standard input; \ + pass --language (see `ocomment languages`)\n" + ); +} + +/// `strip` and every command that accepts `-` read the same standard input, +/// so they must fail with the same words when they cannot tell what it is. +/// One constant is what makes that true; this test is what keeps it true. +#[test] +fn strip_and_check_agree_on_the_undetectable_input_message() { + let directory = tempfile::tempdir().unwrap(); + let expected = "ocomment: cannot detect the language of standard input; \ + pass --language (see `ocomment languages`)\n"; + for arguments in [ + vec!["strip"], + vec!["check", "-"], + vec!["diff", "-"], + vec!["scan", "-"], + ] { + let output = run_stdin(directory.path(), &arguments, b"let x = 1; // note\n"); + assert_eq!(output.status.code(), Some(2), "`ocomment {arguments:?}`"); + assert_eq!( + String::from_utf8(output.stderr).unwrap(), + expected, + "`ocomment {arguments:?}`" + ); + } +} + +/// A pipe cannot be rewritten in place; `fix` says so and names the command +/// that does write a stripped stream. +#[test] +fn fix_refuses_standard_input() { + let directory = tempfile::tempdir().unwrap(); + let output = run_stdin( + directory.path(), + &["fix", "--language", "rust", "-"], + b"let x = 1; // note\n", + ); + assert_eq!(output.status.code(), Some(2)); + assert_eq!( + String::from_utf8(output.stderr).unwrap(), + "ocomment: cannot rewrite standard input in place; use `ocomment strip`\n" + ); +} + +/// There is only one standard input, so naming it twice is a usage error +/// rather than a silently deduplicated target. +#[test] +fn standard_input_may_be_named_only_once() { + let directory = tempfile::tempdir().unwrap(); + let output = run_stdin( + directory.path(), + &["check", "--language", "rust", "-", "-"], + b"let x = 1; // note\n", + ); + assert_eq!(output.status.code(), Some(2)); + assert!( + String::from_utf8(output.stderr) + .unwrap() + .contains("cannot read standard input twice"), + "the second `-` was accepted" + ); +} + +/// `--staged` reads the Git index; a pipe cannot be one of its entries. +#[test] +fn standard_input_conflicts_with_staged() { + let directory = repository(); + let output = run_stdin( + directory.path(), + &["check", "--staged", "--language", "rust", "-"], + b"let x = 1; // note\n", + ); + assert_eq!(output.status.code(), Some(2)); + assert_eq!( + String::from_utf8(output.stderr).unwrap(), + "ocomment: cannot read standard input with --staged; the index is the source\n" + ); +} + +/// `fix --dry-run` is `diff` with fix vocabulary: the patch goes to standard +/// output, the file keeps every byte, and the exit code still reports a +/// pending change. +#[test] +fn fix_dry_run_writes_a_patch_and_leaves_the_file_alone() { + let directory = tempfile::tempdir().unwrap(); + let path = directory.path().join("sample.rs"); + let before = b"let x = 1; // remove\n"; + fs::write(&path, before).unwrap(); + + let output = run(directory.path(), &["fix", "--dry-run", "sample.rs"]); + assert_eq!(output.status.code(), Some(1)); + assert_eq!(fs::read(&path).unwrap(), before); + let stdout = String::from_utf8(output.stdout).unwrap(); + assert!( + stdout.starts_with("--- a/sample.rs\n"), + "diff is:\n{stdout}" + ); + assert_eq!( + String::from_utf8(output.stderr).unwrap(), + "Would remove 1 comment in 1 file. Rerun without --dry-run to apply.\n" + ); +} + +/// With nothing to take out, the preview says what a real `fix` would say. +#[test] +fn fix_dry_run_on_a_clean_file_reports_nothing_to_fix() { + let directory = tempfile::tempdir().unwrap(); + let path = directory.path().join("clean.rs"); + fs::write(&path, b"let x = 1;\n").unwrap(); + + let output = run(directory.path(), &["fix", "--dry-run", "clean.rs"]); + assert_eq!(output.status.code(), Some(0)); + assert_eq!(String::from_utf8(output.stdout).unwrap(), ""); + assert_eq!( + String::from_utf8(output.stderr).unwrap(), + "Nothing to fix in 1 file.\n" + ); +} + +/// A skipped path can be the whole answer to the run, so the preview still has +/// to name it — but `fix --dry-run` promises a patch on standard output, and a +/// reader piping that into `git apply` cannot be handed a prose line in the +/// middle of it. The reason goes to standard error instead, directly above the +/// summary that counts it, word for word what the `fix` it stands in for says. +/// +/// Spec change: the preview used to print that line on standard output, where +/// it corrupted the patch. Plain `fix` writes no patch and keeps its skips on +/// standard output; plain `diff` folds them into the summary as before. +#[test] +fn fix_dry_run_lists_a_skipped_path_the_way_fix_does() { + let directory = tempfile::tempdir().unwrap(); + fs::write(directory.path().join("notes.unknownext"), b"# notes\n").unwrap(); + + let skip = format!("notes.unknownext: skipped: {NO_LANGUAGE}\n"); + let previewed = run(directory.path(), &["fix", "--dry-run", "notes.unknownext"]); + assert_eq!(previewed.status.code(), Some(0)); + assert_eq!( + String::from_utf8(previewed.stdout).unwrap(), + "", + "the preview put prose on the standard output it promises as a patch" + ); + assert_eq!( + String::from_utf8(previewed.stderr).unwrap(), + format!("{skip}Nothing to fix.\n"), + "the preview never said why it had nothing to fix" + ); + + let fixed = run(directory.path(), &["fix", "notes.unknownext"]); + assert_eq!( + String::from_utf8(fixed.stdout).unwrap(), + skip, + "`fix` stopped listing the skip on standard output" + ); + + let diffed = run(directory.path(), &["diff", "notes.unknownext"]); + assert_eq!(diffed.status.code(), Some(0)); + assert_eq!( + String::from_utf8(diffed.stdout).unwrap(), + "", + "`diff` reserves its standard output for the patch" + ); + assert_eq!( + String::from_utf8(diffed.stderr).unwrap(), + "Nothing to diff.\n", + "plain `diff` folds a skip into its summary rather than listing it" + ); +} + +/// Both new entry points are discoverable from `--help`. +#[test] +fn help_documents_standard_input_and_the_dry_run() { + let directory = tempfile::tempdir().unwrap(); + let checked = run(directory.path(), &["check", "--help"]); + assert_eq!(checked.status.code(), Some(0)); + let help = String::from_utf8(checked.stdout).unwrap(); + assert!( + help.contains("`-` reads standard input"), + "`check --help` does not document `-`:\n{help}" + ); + let fixed = run(directory.path(), &["fix", "--help"]); + assert_eq!(fixed.status.code(), Some(0)); + let help = String::from_utf8(fixed.stdout).unwrap(); + assert!( + help.contains("--dry-run"), + "`fix --help` does not document --dry-run:\n{help}" + ); +} + +/// Standard input is one target among others, not a mode: a piped file and a +/// named one are reported by the same run. +#[test] +fn a_dash_can_be_mixed_with_named_paths() { + let directory = tempfile::tempdir().unwrap(); + fs::write(directory.path().join("sample.rs"), b"let y = 2; // named\n").unwrap(); + let output = run_stdin( + directory.path(), + &["check", "--language", "rust", "sample.rs", "-"], + b"let x = 1; // piped\n", + ); + assert_eq!(output.status.code(), Some(1)); + assert_eq!( + String::from_utf8(output.stdout).unwrap(), + ":1:12: removable line comment: // piped\n\ + sample.rs:1:12: removable line comment: // named\n" + ); +} + +/// The default command takes the same PATH list, so `-` works without naming +/// `check` at all. +#[test] +fn the_default_command_also_reads_a_dash() { + let directory = tempfile::tempdir().unwrap(); + let output = run_stdin( + directory.path(), + &["--language", "rust", "-"], + b"let x = 1; // note\n", + ); + assert_eq!(output.status.code(), Some(1)); + assert_eq!( + String::from_utf8(output.stdout).unwrap(), + ":1:12: removable line comment: // note\n" + ); +} + +/// `--dry-run` previews the staged run too: the patch is the one `--staged` +/// would apply, and the index keeps every byte. +#[test] +fn fix_dry_run_previews_the_staged_patch_without_writing_the_index() { + let directory = repository(); + fs::write( + directory.path().join("sample.rs"), + b"let x = 1; // remove\n", + ) + .unwrap(); + git(directory.path(), &["add", "sample.rs"]); + let output = run(directory.path(), &["fix", "--dry-run", "--staged"]); + assert_eq!(output.status.code(), Some(1)); + let stdout = String::from_utf8(output.stdout).unwrap(); + assert!( + stdout.starts_with("--- a/sample.rs\n"), + "diff is:\n{stdout}" + ); + assert_eq!( + String::from_utf8(output.stderr).unwrap(), + "Would remove 1 comment in 1 file. Rerun without --dry-run to apply.\n" + ); + assert_eq!( + git(directory.path(), &["show", ":sample.rs"]), + b"let x = 1; // remove\n" + ); +} + +/// A tree whose report is far larger than any pipe buffer, so a reader that +/// stops early is guaranteed to close the pipe while the run is still writing. +fn wide_tree(files: usize, comments: usize) -> TempDir { + let directory = tempfile::tempdir().unwrap(); + let mut source = String::new(); + for index in 0..comments { + source.push_str(&format!("let value{index} = {index}; // remove {index}\n")); + } + for index in 0..files { + fs::write(directory.path().join(format!("file{index}.rs")), &source).unwrap(); + } + directory +} + +/// Run the binary, take `head` bytes of its output, then close the pipe and +/// report how the run ended and what it said on standard error. +fn run_closed_pipe(directory: &Path, arguments: &[&str], head: usize) -> (ExitStatus, String) { + let mut child = Command::new(binary()) + .current_dir(directory) + .env("PATH", "/usr/bin:/bin") + .args(arguments) + .stdin(Stdio::null()) + .stdout(Stdio::piped()) + .stderr(Stdio::piped()) + .spawn() + .unwrap(); + let mut output = child.stdout.take().expect("standard output was piped"); + let mut taken = vec![0u8; head]; + if head > 0 { + output.read_exact(&mut taken).unwrap(); + } + /* NOTE: The reader has what it wanted; from here every write the run attempts + * fails with EPIPE. */ + drop(output); + let mut message = String::new(); + child + .stderr + .take() + .expect("standard error was piped") + .read_to_string(&mut message) + .unwrap(); + (child.wait().unwrap(), message) +} + +/// `ocomment check --format json . | head` is a reader that stops early, not a +/// failure: the run ends quietly with status 0 and says nothing. +#[test] +fn a_closed_pipe_ends_the_json_report_quietly() { + let directory = wide_tree(100, 50); + let (status, stderr) = + run_closed_pipe(directory.path(), &["check", "--format", "json", "."], 10); + assert!( + status.success(), + "expected a quiet exit, got {status:?} with stderr:\n{stderr}" + ); + assert_eq!(stderr, ""); +} + +/// The human report is written the same way, so it ends the same way. +#[test] +fn a_closed_pipe_ends_the_human_report_quietly() { + let directory = wide_tree(100, 50); + let (status, stderr) = run_closed_pipe(directory.path(), &["check", "."], 10); + assert!( + status.success(), + "expected a quiet exit, got {status:?} with stderr:\n{stderr}" + ); + assert_eq!(stderr, ""); +} + +/// So are the machine formats that serialize straight into standard output. +#[test] +fn a_closed_pipe_ends_the_sarif_report_quietly() { + let directory = wide_tree(100, 50); + let (status, stderr) = + run_closed_pipe(directory.path(), &["check", "--format", "sarif", "."], 10); + assert!( + status.success(), + "expected a quiet exit, got {status:?} with stderr:\n{stderr}" + ); + assert_eq!(stderr, ""); +} + +/// A short report can lose its reader before it writes its first byte. The +/// listing commands must survive that too. +#[test] +fn a_pipe_closed_before_the_first_byte_ends_languages_quietly() { + let directory = tempfile::tempdir().unwrap(); + let (status, stderr) = run_closed_pipe(directory.path(), &["languages"], 0); + assert!( + status.success(), + "expected a quiet exit, got {status:?} with stderr:\n{stderr}" + ); + assert_eq!(stderr, ""); +} + +/// `clap_complete` writes straight into the handle it is handed and panics if +/// that write fails, so the completion script is buffered before it is written. +#[test] +fn a_pipe_closed_before_the_first_byte_ends_completions_quietly() { + let directory = tempfile::tempdir().unwrap(); + let (status, stderr) = run_closed_pipe(directory.path(), &["completions", "zsh"], 0); + assert!( + status.success(), + "expected a quiet exit, got {status:?} with stderr:\n{stderr}" + ); + assert_eq!(stderr, ""); +} + +/// Run the binary with its standard error piped to a reader that closes at +/// once, and report how it ended. +fn run_closed_error_pipe(directory: &Path, arguments: &[&str]) -> ExitStatus { + let mut child = Command::new(binary()) + .current_dir(directory) + .env("PATH", "/usr/bin:/bin") + .args(arguments) + .stdin(Stdio::null()) + .stdout(Stdio::null()) + .stderr(Stdio::piped()) + .spawn() + .unwrap(); + drop(child.stderr.take().expect("standard error was piped")); + child.wait().unwrap() +} + +/// Standard error carries commentary, not the product of the run, so losing +/// its reader changes nothing: `-v` still reports its verdict through the exit +/// status instead of dying on the trace it could not write. +#[test] +fn a_closed_error_pipe_does_not_end_the_run() { + let directory = tempfile::tempdir().unwrap(); + fs::write( + directory.path().join("sample.rs"), + b"let x = 1; // remove\n", + ) + .unwrap(); + let status = run_closed_error_pipe(directory.path(), &["check", "-v", "."]); + assert_eq!( + status.code(), + Some(1), + "a closed standard error changed the verdict: {status:?}" + ); +} + +/// The real `git` on this machine, found on `PATH` the way a shell finds it. +/// +/// A fake `git` planted ahead of it has to hand every other subcommand to the +/// genuine one by absolute path: the fake is first on `PATH` itself, so `exec +/// git` would only call it back. `/usr/bin/git` is the fallback for a `PATH` +/// that names none. +#[cfg(unix)] +fn real_git() -> std::path::PathBuf { + std::env::var_os("PATH") + .and_then(|path| { + std::env::split_paths(&path) + .map(|directory| directory.join("git")) + .find(|candidate| candidate.is_file()) + }) + .unwrap_or_else(|| std::path::PathBuf::from("/usr/bin/git")) +} + +/// A closed pipe is benign only when it is *our* report that lost its reader. +/// `git hash-object` exiting before it reads the rewritten blob breaks a pipe +/// the run owns in the other direction: the index was never updated, so the +/// run must report the failure instead of ending quietly with success. +#[cfg(unix)] +#[test] +fn a_broken_pipe_from_git_hash_object_fails_the_staged_fix() { + use std::os::unix::fs::PermissionsExt; + + let directory = repository(); + /* NOTE: What travels down the pipe is the blob with the comments already taken + * out, so it is that which has to outgrow the pipe buffer — 64 KiB on + * Linux — for the write to still be in flight when the fake + * `git hash-object` drops the reading end. Half again as much is margin + * enough. The file is therefore sized by the bytes that survive the fix + * rather than by its own length, and it carries them on a few long lines + * instead of many short ones: the run costs time per comment, and this + * test needs bytes. */ + let padding = "x".repeat(200); + let mut source = String::new(); + let mut stripped = 0; + let mut index = 0; + while stripped < 96 * 1024 { + let code = format!("let value{index} = \"{padding}\";"); + stripped += code.len() + 1; + source.push_str(&format!("{code} // remove {index}\n")); + index += 1; + } + let path = directory.path().join("wide.rs"); + fs::write(&path, &source).unwrap(); + git(directory.path(), &["add", "wide.rs"]); + let staged_before = git(directory.path(), &["show", ":wide.rs"]); + + /* NOTE: Every invocation reaches the real Git except `hash-object`, which closes + * its standard input and fails without reading a byte. */ + let fake = tempfile::tempdir().unwrap(); + let script = fake.path().join("git"); + fs::write( + &script, + format!( + "#!/bin/sh\n\ + if [ \"$1\" = hash-object ]; then\n\ + exec 0<&-\n\ + exit 1\n\ + fi\n\ + exec {} \"$@\"\n", + real_git().display() + ), + ) + .unwrap(); + fs::set_permissions(&script, fs::Permissions::from_mode(0o755)).unwrap(); + + let output = Command::new(binary()) + .current_dir(directory.path()) + .env("PATH", format!("{}:/usr/bin:/bin", fake.path().display())) + .args(["fix", "--staged"]) + .output() + .unwrap(); + assert_eq!( + output.status.code(), + Some(2), + "a failed blob write ended the run quietly; stderr was:\n{}", + String::from_utf8_lossy(&output.stderr) + ); + let stderr = String::from_utf8_lossy(&output.stderr); + assert!( + !stderr.is_empty(), + "the failed staged fix was never reported on standard error" + ); + assert!( + stderr.contains("git hash-object"), + "the report does not say which write failed:\n{stderr}" + ); + /* NOTE: `hash-object` also exits non-zero, and that failure carries the same + * name. This is the test for the broken pipe, so the blob must have been + * in flight when the reader went away, not sitting whole in the buffer. */ + assert!( + stderr.contains("cannot write the rewritten blob"), + "the run failed before the blob was ever written, so the broken pipe \ + went untested:\n{stderr}" + ); + assert_eq!( + git(directory.path(), &["show", ":wide.rs"]), + staged_before, + "the index changed although no blob was written" + ); + assert_eq!( + fs::read(&path).unwrap(), + source.as_bytes(), + "the working tree changed although no blob was written" + ); +} + +/// `fix` refuses standard input, so its `--help` must not offer it as a target. +#[test] +fn fix_help_does_not_advertise_the_standard_input_it_refuses() { + let directory = tempfile::tempdir().unwrap(); + let fixed = run(directory.path(), &["fix", "--help"]); + assert_eq!(fixed.status.code(), Some(0)); + let help = String::from_utf8(fixed.stdout).unwrap(); + assert!( + !help.contains("reads standard input"), + "`fix --help` advertises a target it refuses:\n{help}" + ); + assert!( + help.contains("Files or directories to rewrite"), + "`fix --help` does not describe its PATH list:\n{help}" + ); + let checked = run(directory.path(), &["check", "--help"]); + assert_eq!(checked.status.code(), Some(0)); + let help = String::from_utf8(checked.stdout).unwrap(); + assert!( + help.contains("reads standard input"), + "`check --help` stopped documenting `-`:\n{help}" + ); +} + +/// The counter line is erased only if one was ever drawn: a run that scans +/// nothing must not write an escape sequence to a terminal that saw no counter. +#[test] +fn progress_clears_the_counter_only_when_one_was_drawn() { + let empty = tempfile::tempdir().unwrap(); + let output = run(empty.path(), &["check", "--progress", "always", "."]); + assert_eq!(output.status.code(), Some(0)); + let stderr = String::from_utf8(output.stderr).unwrap(); + assert!( + !stderr.contains("\r\x1b[2K"), + "a counter that was never drawn was cleared anyway:\n{stderr:?}" + ); + + let directory = many_files(120); + let output = run(directory.path(), &["check", "--progress", "always", "."]); + assert_eq!(output.status.code(), Some(1)); + let stderr = String::from_utf8(output.stderr).unwrap(); + assert!( + stderr.contains("\r\x1b[2K"), + "the counter line was never cleared:\n{stderr:?}" + ); +} + +/// "Nothing to check" is the vocabulary of `check`. Every command has its own +/// verb for the run that found nothing to work on. +#[test] +fn an_empty_run_summarizes_itself_in_the_vocabulary_of_its_command() { + let directory = tempfile::tempdir().unwrap(); + fs::write(directory.path().join("notes.unknownext"), b"# notes\n").unwrap(); + /* NOTE: `fix --dry-run` keeps its standard output for the patch, so the named + * skip it met stands on standard error directly above the summary this + * test is about; every other command reports the skip elsewhere. */ + let skip = format!("notes.unknownext: skipped: {NO_LANGUAGE}\n"); + for (arguments, expected) in [ + ( + vec!["check", "notes.unknownext"], + "Nothing to check.\n".to_owned(), + ), + ( + vec!["fix", "notes.unknownext"], + "Nothing to fix.\n".to_owned(), + ), + ( + vec!["fix", "--dry-run", "notes.unknownext"], + format!("{skip}Nothing to fix.\n"), + ), + ( + vec!["diff", "notes.unknownext"], + "Nothing to diff.\n".to_owned(), + ), + ( + vec!["scan", "notes.unknownext"], + "Nothing to scan.\n".to_owned(), + ), + ] { + let output = run(directory.path(), &arguments); + assert_eq!(output.status.code(), Some(0), "`ocomment {arguments:?}`"); + assert_eq!( + String::from_utf8(output.stderr).unwrap(), + expected, + "`ocomment {arguments:?}`" + ); + } + let output = run(directory.path(), &["scan", "."]); + assert_eq!(output.status.code(), Some(0)); + assert_eq!( + String::from_utf8(output.stderr).unwrap(), + "Nothing to scan: 1 file skipped (unknown language: 1; use -v to list).\n" + ); +} + +/// A file OComment has no scanner for is not "unknown": the skip line names +/// the list to consult and the flag that forces a language anyway. The folded +/// summary clause keeps the short key, so a walk over a hundred unreadable +/// extensions still reads as one clause instead of a hundred sentences. +#[test] +fn an_unknown_language_skip_says_how_to_force_one() { + let directory = tempfile::tempdir().unwrap(); + fs::write( + directory.path().join("sample.rs"), + b"let x = 1; // remove\n", + ) + .unwrap(); + fs::write(directory.path().join("notes.unknownext"), b"# notes\n").unwrap(); + let output = run(directory.path(), &["check", "-v", "."]); + assert_eq!(output.status.code(), Some(1)); + let stdout = String::from_utf8(output.stdout).unwrap(); + let stderr = String::from_utf8(output.stderr).unwrap(); + assert!( + stdout.contains( + "notes.unknownext: skipped: no built-in language for this file \ + (see `ocomment languages`; use --language to force)" + ), + "the skip line never said what to do about it:\n{stdout}" + ); + assert!( + stderr.contains("1 file skipped (unknown language: 1)."), + "the folded clause stopped using the short key:\n{stderr}" + ); +} + +/// A path that was named and is not there says where it was looked for, so a +/// typo, a wrong working directory, and a deleted file are told apart without +/// a second run. +#[test] +fn a_missing_path_says_where_it_was_looked_for() { + let directory = tempfile::tempdir().unwrap(); + let cwd = fs::canonicalize(directory.path()).unwrap(); + let output = run(&cwd, &["check", "missing.rs"]); + assert_eq!(output.status.code(), Some(2)); + let stdout = String::from_utf8(output.stdout).unwrap(); + assert!( + stdout.contains(&format!( + "missing.rs: error: path does not exist (checked relative to {})", + cwd.display() + )), + "check output is:\n{stdout}" + ); +} + +/// A project configuration without the version key is refused; saying which +/// line to add, and to which file, is the whole fix. +#[test] +fn a_configuration_without_a_version_says_how_to_add_one() { + let directory = tempfile::tempdir().unwrap(); + let config = directory.path().join(".ocomment.toml"); + fs::write(&config, b"[policy]\nmode = \"all\"\n").unwrap(); + fs::write( + directory.path().join("sample.rs"), + b"let x = 1; // remove\n", + ) + .unwrap(); + let output = run(directory.path(), &["check", "sample.rs"]); + assert_eq!(output.status.code(), Some(2)); + let stderr = String::from_utf8(output.stderr).unwrap(); + assert!( + stderr.contains(&format!( + "must contain `version = 1` (add `version = 1` at the top of {})", + config.display() + )), + "the version error never said what to write:\n{stderr}" + ); +} + +/// A misspelled `[languages.*]` key is refused by name; the fix is the list of +/// the languages that do exist. +#[test] +fn an_unknown_language_key_points_at_the_language_list() { + let directory = tempfile::tempdir().unwrap(); + fs::write( + directory.path().join(".ocomment.toml"), + b"version = 1\n[languages.klingon]\n", + ) + .unwrap(); + fs::write( + directory.path().join("sample.rs"), + b"let x = 1; // remove\n", + ) + .unwrap(); + let output = run(directory.path(), &["check", "sample.rs"]); + assert_eq!(output.status.code(), Some(2)); + assert_eq!( + String::from_utf8(output.stderr).unwrap(), + "ocomment: unknown language configuration key `klingon`; see `ocomment languages`\n" + ); +} + +/// `--staged` reads the index, so outside a repository the flag is the thing +/// to drop. Git's own words are kept: they say which directory was searched. +#[test] +fn staged_outside_a_repository_names_the_flag_and_quotes_git() { + let directory = tempfile::tempdir().unwrap(); + let output = run(directory.path(), &["check", "--staged"]); + assert_eq!(output.status.code(), Some(2)); + let stderr = String::from_utf8(output.stderr).unwrap(); + assert!( + stderr.contains("--staged needs a Git repository:"), + "the failure never named the flag that needed one:\n{stderr}" + ); + assert!( + stderr.contains("not a git repository"), + "Git's own explanation was dropped:\n{stderr}" + ); +} + +/// A lock file left behind by a crashed Git is indistinguishable from a Git +/// that is running right now, so the message offers both readings and the +/// path to delete. +#[test] +fn a_locked_git_index_says_what_to_do_about_the_lock() { + let directory = repository(); + fs::write( + directory.path().join("sample.rs"), + b"let x = 1; // remove\n", + ) + .unwrap(); + git(directory.path(), &["add", "sample.rs"]); + fs::write(directory.path().join(".git/index.lock"), b"").unwrap(); + let output = run(directory.path(), &["fix", "--staged"]); + assert_eq!(output.status.code(), Some(2)); + let stderr = String::from_utf8(output.stderr).unwrap(); + assert!( + stderr.contains( + "Git index is locked; no files were modified; another Git process may be \ + running, or remove a stale .git/index.lock" + ), + "the lock failure never said what to do about it:\n{stderr}" + ); +} + +/// The plugin commands shell out to four tools. A missing one must name the +/// binary, say what this run wanted it for, and point at the command that +/// reports the whole environment at once. +#[test] +fn a_missing_plugin_tool_names_it_its_purpose_and_doctor() { + let directory = tempfile::tempdir().unwrap(); + /* NOTE: Pin the project root: without a configuration the walk upwards can find + * a repository marker above the temporary directory and install there. */ + fs::write(directory.path().join(".ocomment.toml"), b"version = 1\n").unwrap(); + let empty = tempfile::tempdir().unwrap(); + for (source, expected) in [ + ( + "https://example.invalid/scanner.wasm", + "cannot run `curl` (needed for https:// plugin sources); run `ocomment doctor`", + ), + ( + "gh:owner/repo@v1#scanner.wasm", + "cannot run `gh` (needed for gh: plugin sources); run `ocomment doctor`", + ), + ( + "oci:example.invalid/scanner:1", + "cannot run `oras` (needed for oci: plugin sources); run `ocomment doctor`", + ), + ] { + let output = Command::new(binary()) + .current_dir(directory.path()) + .env("PATH", empty.path()) + .args([ + "plugin", + "add", + source, + "--sha256", + "0000000000000000000000000000000000000000000000000000000000000000", + "--identity", + "publisher@example.test", + ]) + .output() + .unwrap(); + assert_eq!(output.status.code(), Some(2), "`plugin add {source}`"); + let stderr = String::from_utf8(output.stderr).unwrap(); + assert!( + stderr.contains(expected), + "`plugin add {source}` said:\n{stderr}" + ); + } +} + +/// Write an executable stand-in for one external tool, printing `lines` and +/// nothing else. `doctor` reports whatever a tool says about itself, so a fake +/// that says something recognizable is enough to pin the row it produces. +#[cfg(unix)] +fn fake_tool_lines(directory: &Path, name: &str, lines: &[&str]) { + use std::os::unix::fs::PermissionsExt; + let path = directory.join(name); + let arguments: String = lines + .iter() + .map(|line| { + assert!( + !line.contains(['"', '\\', '$', '`']), + "the fake tool writes a shell script, so `{line}` needs quoting it does not do" + ); + format!(" \"{line}\"") + }) + .collect(); + fs::write(&path, format!("#!/bin/sh\nprintf '%s\\n'{arguments}\n")).unwrap(); + fs::set_permissions(&path, fs::Permissions::from_mode(0o755)).unwrap(); +} + +/// The common case: a tool that answers with one line. +#[cfg(unix)] +fn fake_tool(directory: &Path, name: &str, line: &str) { + fake_tool_lines(directory, name, &[line]); +} + +/// Run the binary with `PATH` pointing at `tools` and nothing else, so a probe +/// sees exactly the tools the test installed there. +#[cfg(unix)] +fn run_with_tools(directory: &Path, tools: &Path, arguments: &[&str]) -> Output { + Command::new(binary()) + .current_dir(directory) + .env("PATH", tools) + .args(arguments) + .output() + .unwrap() +} + +/// `doctor` is the command every missing-tool failure points at, so it has to +/// probe the tools the plugin commands and `--staged` shell out to instead of +/// assuming them. A tool that is not installed is reported with the very +/// purpose the failure would have named, and is not itself a failure: all five +/// are optional, and a run that never touches a plugin never needs one. +#[cfg(unix)] +#[test] +fn doctor_probes_the_optional_tools_it_shells_out_to() { + let directory = tempfile::tempdir().unwrap(); + let tools = tempfile::tempdir().unwrap(); + fake_tool(tools.path(), "git", "git version 9.9.9"); + + let output = run_with_tools(directory.path(), tools.path(), &["doctor"]); + assert_eq!( + output.status.code(), + Some(0), + "a missing optional tool failed the run: {}", + String::from_utf8_lossy(&output.stderr) + ); + let report = String::from_utf8(output.stdout).unwrap(); + assert!( + report.contains("git: git version 9.9.9"), + "doctor did not report the one tool on PATH:\n{report}" + ); + for missing in [ + "curl: not found (needed for https:// plugin sources)", + "gh: not found (needed for gh: plugin sources)", + "oras: not found (needed for oci: plugin sources)", + "cosign: not found (needed for --identity verification)", + ] { + assert!( + report.contains(missing), + "doctor never reported `{missing}`:\n{report}" + ); + } +} + +/// The row carries the tool's own version line, whatever the tool chose to +/// say: `doctor` reports the environment rather than parsing it. `git` is +/// probed for the same reason as the rest — `--staged` is the part of the run +/// that stops working without it. +#[cfg(unix)] +#[test] +fn doctor_reports_a_probed_tools_own_version_line() { + let directory = tempfile::tempdir().unwrap(); + let tools = tempfile::tempdir().unwrap(); + fake_tool(tools.path(), "cosign", "cosign v9.9.9"); + + let output = run_with_tools(directory.path(), tools.path(), &["doctor"]); + assert_eq!(output.status.code(), Some(0)); + let report = String::from_utf8(output.stdout).unwrap(); + assert!( + report.contains("cosign: cosign v9.9.9"), + "doctor did not carry the tool's own version line:\n{report}" + ); + assert!( + report.contains("git: not found (needed for --staged)"), + "a missing `git` never named the flag that needs it:\n{report}" + ); +} + +/// `cosign version` draws several lines of ASCII art before it says anything +/// about itself, and a row carrying the top of that banner would tell a reader +/// nothing at all. A version has a number in it, so that is the line the row +/// carries — sanitised like every probed line, so the run of spaces the tool +/// aligned its banner with is collapsed to one. +#[cfg(unix)] +#[test] +fn doctor_looks_past_a_banner_for_the_version_line() { + let directory = tempfile::tempdir().unwrap(); + let tools = tempfile::tempdir().unwrap(); + fake_tool_lines( + tools.path(), + "cosign", + &[ + " ______ ______", + " | | | __ |", + "cosign: A tool for Container Signing", + "", + "GitVersion: v9.9.9", + ], + ); + + let output = run_with_tools(directory.path(), tools.path(), &["doctor"]); + assert_eq!(output.status.code(), Some(0)); + let report = String::from_utf8(output.stdout).unwrap(); + assert!( + report.contains("cosign: GitVersion: v9.9.9"), + "doctor reported the banner instead of the version behind it:\n{report}" + ); + assert!( + !report.contains("GitVersion: v9.9.9"), + "the tool's own alignment survived into the row:\n{report}" + ); + assert!( + !report.contains("cosign: ______"), + "the top of the banner reached the report:\n{report}" + ); +} + +/// A probed tool chooses the bytes `doctor` prints, so a version line is +/// untrusted input on its way to a terminal: a tool planted on `PATH` could +/// clear the screen or repaint the report from its own banner. The row carries +/// what the tool said with every control sequence replaced, and a tool that +/// answers at all is still a healthy row rather than a failing run. +#[cfg(unix)] +#[test] +fn doctor_strips_control_sequences_from_a_probed_tools_version_line() { + let directory = tempfile::tempdir().unwrap(); + let tools = tempfile::tempdir().unwrap(); + fake_tool( + tools.path(), + "cosign", + "\u{1b}[2J\u{1b}[1;31mv1.0 PWNED\u{1b}[0m", + ); + + let output = run_with_tools(directory.path(), tools.path(), &["doctor"]); + assert_eq!( + output.status.code(), + Some(0), + "a tool that answered with control sequences failed the run: {}", + String::from_utf8_lossy(&output.stderr) + ); + assert!( + !output.stdout.contains(&0x1b), + "an escape byte reached the report: {:?}", + String::from_utf8_lossy(&output.stdout) + ); + let report = String::from_utf8(output.stdout).unwrap(); + assert!( + report.contains("cosign: \u{fffd}[2J\u{fffd}[1;31mv1.0 PWNED\u{fffd}[0m\n"), + "doctor did not report the version line with its controls replaced:\n{report}" + ); +} + +/// The other half of "why did that run do that?" is the environment the run +/// resolved for itself: where it stood, what it took as the root, which +/// configuration files it merged, and whether its output is decorated. +#[test] +fn doctor_reports_the_environment_it_resolved() { + let directory = tempfile::tempdir().unwrap(); + let empty = tempfile::tempdir().unwrap(); + let doctor = |no_color: Option<&str>| { + let mut command = Command::new(binary()); + command + .current_dir(directory.path()) + .env("PATH", "/usr/bin:/bin") + /* NOTE: Pin the user layer away from whoever is running the tests: the + * trace this reports has to be the one this run resolved. */ + .env("XDG_CONFIG_HOME", empty.path()) + .arg("doctor"); + match no_color { + Some(value) => command.env("NO_COLOR", value), + None => command.env_remove("NO_COLOR"), + }; + let output = command.output().unwrap(); + assert_eq!( + output.status.code(), + Some(0), + "{}", + String::from_utf8_lossy(&output.stderr) + ); + String::from_utf8(output.stdout).unwrap() + }; + + let report = doctor(None); + let cwd = fs::canonicalize(directory.path()).unwrap(); + assert!( + report.contains(&format!("cwd: {}", cwd.display())), + "doctor never said where it was standing:\n{report}" + ); + assert!( + report.contains("root: "), + "doctor never said what it took as the root:\n{report}" + ); + assert!( + report.contains("config: built-in defaults"), + "doctor never traced the configuration it merged:\n{report}" + ); + /* NOTE: The report is read through a pipe, so the decoration it describes is the + * decoration this very run chose. */ + assert!( + report.contains("stdout: not a terminal"), + "doctor never said whether its output is a terminal:\n{report}" + ); + assert!( + report.contains("NO_COLOR: unset"), + "doctor never said whether NO_COLOR is set:\n{report}" + ); + assert!( + doctor(Some("1")).contains("NO_COLOR: set"), + "doctor ignored the NO_COLOR that silences its colour" + ); + + let created = run(directory.path(), &["init", "config"]); + assert_eq!(created.status.code(), Some(0)); + let report = doctor(None); + assert!( + report.contains(&format!( + "config: project {}", + cwd.join(".ocomment.toml").display() + )), + "doctor did not name the project configuration it found:\n{report}" + ); + assert!( + report.contains(&format!("root: {}", cwd.display())), + "the project file did not move the root with it:\n{report}" + ); +} + +/// A directory name is chosen by whoever made the directory, not by OComment, +/// so the two rows that print one are untrusted input on their way to a +/// terminal exactly like a probed tool's version line. They are sanitised the +/// same way and cut nowhere: a path is the answer the reader came for, and one +/// ending in an ellipsis names no directory at all. +#[cfg(unix)] +#[test] +fn doctor_sanitises_the_directories_it_reports_without_cutting_them_short() { + /* NOTE: Long enough that a preview-width cap would have to cut it, and carrying + * the escape that would let a directory name repaint the report. */ + let name = format!("ocomment\u{1b}{}", "a".repeat(90)); + let directory = tempfile::Builder::new() + .prefix(&name) + .tempdir() + .expect("a directory name may carry an escape on this platform"); + /* NOTE: A project file of its own makes this directory the root as well, so both + * rows name it and both are pinned by one run. */ + fs::write(directory.path().join(".ocomment.toml"), b"version = 1\n").unwrap(); + let empty = tempfile::tempdir().unwrap(); + let output = Command::new(binary()) + .current_dir(directory.path()) + .env("PATH", "/usr/bin:/bin") + .env("XDG_CONFIG_HOME", empty.path()) + .arg("doctor") + .output() + .unwrap(); + assert_eq!( + output.status.code(), + Some(0), + "{}", + String::from_utf8_lossy(&output.stderr) + ); + assert!( + !output.stdout.contains(&0x1b), + "an escape byte reached the report: {:?}", + String::from_utf8_lossy(&output.stdout) + ); + let report = String::from_utf8(output.stdout).unwrap(); + let sanitised = name.replace('\u{1b}', "\u{fffd}"); + for row in ["cwd", "root"] { + let prefix = format!("{row}: "); + let named = report + .lines() + .find(|line| line.starts_with(&prefix)) + .unwrap_or_else(|| panic!("doctor printed no `{row}` row:\n{report}")); + assert!( + named.contains(&sanitised), + "the `{row}` row lost the directory it names:\n{report}" + ); + // NOTE: A version line may be cut to the preview width; a path may not. + assert!( + !named.contains('\u{2026}'), + "the `{row}` row was cut to the preview width:\n{report}" + ); + } +} + +/// The scaffold refuses to write into a directory that already exists, and a +/// refusal that only says "refusing" leaves the reader to guess. There are two +/// ways out — take the directory away, or take the plugin that owns it away — +/// and the message names both. +#[test] +fn plugin_new_refuses_an_existing_directory_and_says_what_to_do() { + let directory = tempfile::tempdir().unwrap(); + let taken = directory.path().join("scanner"); + fs::create_dir(&taken).unwrap(); + + let output = run(directory.path(), &["plugin", "new", "scanner"]); + assert_eq!(output.status.code(), Some(2)); + let error = String::from_utf8_lossy(&output.stderr); + assert!( + error.contains("refusing to overwrite plugin directory"), + "{error}" + ); + assert!( + error.contains("remove it or run `ocomment plugin remove ` first"), + "the refusal never said what to do about it:\n{error}" + ); + assert_eq!( + fs::read_dir(&taken).unwrap().count(), + 0, + "the refusal wrote into the directory it refused" + ); +} + +/// `--policy all` means "take everything out", so the one thing it deliberately +/// leaves behind has to explain itself: the summary counts the kept preambles +/// and names the flag that removes them too. +#[test] +fn policy_all_says_how_to_remove_a_kept_preamble() { + let directory = tempfile::tempdir().unwrap(); + fs::write( + directory.path().join("a.py"), + b"#!/usr/bin/env python3\n# note\nx = 1\n", + ) + .unwrap(); + let hint = "1 protected preamble comment kept; add --force-protected to remove it."; + + let all = run(directory.path(), &["check", "--policy", "all", "a.py"]); + assert_eq!(all.status.code(), Some(1)); + let stderr = String::from_utf8(all.stderr).unwrap(); + assert!( + stderr.contains(hint), + "`--policy all` never explained the comment it kept:\n{stderr}" + ); + + // NOTE: Nothing is protected any more, so there is nothing to explain. + let forced = run( + directory.path(), + &["check", "--policy", "all", "--force-protected", "a.py"], + ); + let stderr = String::from_utf8(forced.stderr).unwrap(); + assert!( + !stderr.contains("--force-protected"), + "the hint outlived the flag that answers it:\n{stderr}" + ); + + /* NOTE: Under `safe` the preamble is one of many deliberate keeps; singling it + * out would be noise on every run. */ + let safe = run(directory.path(), &["check", "a.py"]); + let stderr = String::from_utf8(safe.stderr).unwrap(); + assert!( + !stderr.contains("--force-protected"), + "a policy that keeps much more than preambles advertised the flag:\n{stderr}" + ); + + // NOTE: A file with no preamble at all never mentions it. + fs::write(directory.path().join("b.py"), b"# note\nx = 1\n").unwrap(); + let plain = run(directory.path(), &["check", "--policy", "all", "b.py"]); + let stderr = String::from_utf8(plain.stderr).unwrap(); + assert!( + !stderr.contains("--force-protected"), + "a run that kept no preamble advertised the flag anyway:\n{stderr}" + ); + + /* NOTE: The hint counts what it kept, so its pronoun has to agree with the + * count: one preamble is removed with "it", several with "them". */ + fs::write( + directory.path().join("c.py"), + b"#!/usr/bin/env python3\nx = 2\n", + ) + .unwrap(); + let both = run( + directory.path(), + &["check", "--policy", "all", "a.py", "c.py"], + ); + let stderr = String::from_utf8(both.stderr).unwrap(); + assert!( + stderr + .contains("2 protected preamble comments kept; add --force-protected to remove them."), + "the plural hint does not agree with the two preambles it counted:\n{stderr}" + ); +} + +/// A file that ships with the repository, resolved from the crate directory so +/// a test can read it from whatever temporary directory it runs in. +fn shipped(relative: &str) -> std::path::PathBuf { + Path::new(env!("CARGO_MANIFEST_DIR")) + .join("../..") + .join(relative) +} + +/// How `clap_mangen` writes one option name: every `-` escaped, the name in +/// bold. Help text that merely mentions a flag is rendered in roman, so this +/// matches a real entry rather than a passing reference in someone else's +/// description. +fn roff_option(flag: &str) -> String { + format!("\\fB{}\\fR", flag.replace('-', "\\-")) +} + +/// Every long flag the CLI shows a user, gathered by walking `--help` down +/// every subcommand. Descriptions are scanned too: a flag a description names +/// is a flag the reader will look up. +fn long_flags_in_help(directory: &Path, path: &[&str], found: &mut BTreeSet) { + let mut arguments = path.to_vec(); + arguments.push("--help"); + let output = run(directory, &arguments); + assert_eq!( + output.status.code(), + Some(0), + "`ocomment {}` did not print help", + arguments.join(" ") + ); + let help = String::from_utf8(output.stdout).unwrap(); + let mut rest = help.as_str(); + while let Some(start) = rest.find("--") { + let tail = &rest[start + 2..]; + let end = tail + .find(|character: char| { + !(character.is_ascii_lowercase() || character.is_ascii_digit() || character == '-') + }) + .unwrap_or(tail.len()); + let name = tail[..end].trim_end_matches('-'); + if name.starts_with(|character: char| character.is_ascii_lowercase()) { + found.insert(format!("--{name}")); + } + rest = tail; + } + let children = subcommand_lines(&help); + for (name, _) in &children { + if name == "help" { + continue; + } + let mut child = path.to_vec(); + child.push(name.as_str()); + long_flags_in_help(directory, &child, found); + } +} + +/// A flag nobody can look up is a flag nobody knows about. The manual page is +/// the reference the `man` subcommand and the release archives both hand out, +/// so every flag `--help` mentions anywhere in the command tree has to have an +/// entry there. +#[test] +fn the_manual_page_documents_every_long_flag() { + let directory = tempfile::tempdir().unwrap(); + let mut flags = BTreeSet::new(); + long_flags_in_help(directory.path(), &[], &mut flags); + assert!( + flags.len() >= 20, + "the help walk stopped finding flags, so this test proves nothing: {flags:?}" + ); + /* NOTE: A walk that stopped at the root would still collect enough flags to look + * healthy, so it is pinned to one flag from each depth it has to reach. */ + for reached in ["--dry-run", "--sha256"] { + assert!( + flags.contains(reached), + "the help walk never reached `{reached}`, so it is not descending: {flags:?}" + ); + } + let page = fs::read_to_string(shipped("docs/ocomment.1")).unwrap(); + let missing: Vec<&String> = flags + .iter() + .filter(|flag| !page.contains(&roff_option(flag))) + .collect(); + assert!( + missing.is_empty(), + "docs/ocomment.1 has no entry for {missing:?}; regenerate it with `ocomment man`" + ); +} + +/// The manual page is generated from the parser, and the generated bytes are +/// checked in twice: once for `man -l docs/ocomment.1` and once for the release +/// archives. A page that drifted from the binary documents a tool nobody ships. +#[test] +fn the_checked_in_manual_page_is_the_one_the_binary_renders() { + let directory = tempfile::tempdir().unwrap(); + let output = run(directory.path(), &["man"]); + assert_eq!( + output.status.code(), + Some(0), + "{}", + String::from_utf8_lossy(&output.stderr) + ); + for path in ["docs/ocomment.1", "release-extras/ocomment.1"] { + let checked_in = fs::read(shipped(path)).unwrap(); + assert!( + checked_in == output.stdout, + "{path} is stale; regenerate it with \ + `python3 tools/release_extras.py --binary rust/target/debug/ocomment` \ + and copy release-extras/ocomment.1 to docs/" + ); + } +} + +/// The completion scripts ship from the same generator and go stale the same +/// way, so they are pinned to the binary too. +#[test] +fn the_checked_in_completions_are_the_ones_the_binary_generates() { + let directory = tempfile::tempdir().unwrap(); + for (shell, path) in [ + ("bash", "release-extras/ocomment.bash"), + ("zsh", "release-extras/_ocomment"), + ("fish", "release-extras/ocomment.fish"), + ("powershell", "release-extras/_ocomment.ps1"), + ("elvish", "release-extras/ocomment.elv"), + ] { + let output = run(directory.path(), &["completions", shell]); + assert_eq!( + output.status.code(), + Some(0), + "{}", + String::from_utf8_lossy(&output.stderr) + ); + let checked_in = fs::read(shipped(path)).unwrap(); + assert!( + checked_in == output.stdout, + "{path} is stale; regenerate it with \ + `python3 tools/release_extras.py --binary rust/target/debug/ocomment`" + ); + } +} + +/// `--explain` answers "why was this comment kept?": it lists every comment, +/// kept ones included, and names both the rule that decided each one and the +/// table that rule was written in. A plain `check` still reports only what it +/// would remove. +#[test] +fn check_explain_names_the_override_and_the_pattern_that_kept_a_comment() { + let directory = tempfile::tempdir().unwrap(); + fs::write( + directory.path().join(".ocomment.toml"), + b"version = 1\n\n[policy]\nkeep_regex = [\"(?i)^// api\"]\n\n\ + [[overrides]]\npaths = [\"gen/**\"]\nkeep_regex = [\"(?i)generated\"]\n", + ) + .unwrap(); + fs::create_dir(directory.path().join("gen")).unwrap(); + fs::write( + directory.path().join("gen/b.rs"), + b"/* generated */\nlet x = 1; // TODO\n", + ) + .unwrap(); + fs::write(directory.path().join("a.rs"), b"// API stays\n").unwrap(); + + let plain = run(directory.path(), &["check"]); + assert_eq!(plain.status.code(), Some(1)); + let plain = String::from_utf8(plain.stdout).unwrap(); + assert!( + !plain.contains("kept"), + "a plain `check` listed a kept comment:\n{plain}" + ); + + let output = run(directory.path(), &["check", "--explain"]); + assert_eq!( + output.status.code(), + Some(1), + "{}", + String::from_utf8_lossy(&output.stderr) + ); + let report = String::from_utf8(output.stdout).unwrap(); + for needle in [ + "gen/b.rs:1:1: kept block comment: /* generated */", + "kept: matched keep_regex #1 `(?i)generated` ([[overrides]] #0, paths = [\"gen/**\"])", + "gen/b.rs:2:12: removable line comment: // TODO", + /* NOTE: Nothing set `[policy] mode`, so the reader is told it is a default + * rather than sent to a file that never mentions it. The pattern the + * same file does set is named with the file, spelled the way the reader + * typed their way into the directory. */ + "removed: policy `safe` removes ordinary comments (built-in defaults)", + "a.rs:1:1: kept line comment: // API stays", + "kept: matched keep_regex #0 `(?i)^// api` ([policy] in .ocomment.toml)", + ] { + assert!( + report.contains(needle), + "`check --explain` lacks {needle:?}:\n{report}" + ); + } +} + +/// A comment no setting decided is explained by the flag that would change its +/// fate, because there is no table to send the reader to. +#[test] +fn check_explain_says_what_would_remove_a_preamble_or_a_directive() { + let directory = tempfile::tempdir().unwrap(); fs::write( - directory.path().join("renamed target.rs"), - b"let base = 1;\nlet renamed = 2; // staged rename\n", + directory.path().join("script.py"), + b"#!/usr/bin/env python3\n# note\n", ) .unwrap(); - git(directory.path(), &["add", "renamed target.rs"]); - fs::remove_file(directory.path().join("deleted.rs")).unwrap(); - git(directory.path(), &["add", "deleted.rs"]); - let unusual = "odd\n名前.rs"; fs::write( - directory.path().join(unusual), - b"let new = 1; // new file\n", + directory.path().join("app.js"), + b"// eslint-disable-next-line\nlet x = 1;\n", ) .unwrap(); - git(directory.path(), &["add", unusual]); - let output = run(directory.path(), &["fix", "--staged", "--index-only"]); + let output = run(directory.path(), &["check", "--explain"]); assert_eq!( output.status.code(), - Some(0), + Some(1), "{}", String::from_utf8_lossy(&output.stderr) ); + let report = String::from_utf8(output.stdout).unwrap(); + for needle in [ + "script.py:1:1: kept shebang comment: #!/usr/bin/env python3", + "required source preamble", + "add --force-protected to remove it", + "app.js:1:1: kept directive comment: // eslint-disable-next-line", + "kept: tool or language directive `eslint`; use --remove-kind directive \ + or --policy all to remove it", + ] { + assert!( + report.contains(needle), + "`check --explain` lacks {needle:?}:\n{report}" + ); + } +} + +/// The one keep no setting decided and no flag overrules: a YAML block scalar +/// ends at the comment above the directive it would otherwise swallow. The +/// explanation names the block scalar and the line that has to go first, and +/// the run reports nothing removable at all. +#[test] +fn check_explain_names_the_block_scalar_a_kept_comment_separates() { + let directory = tempfile::tempdir().unwrap(); + fs::write( + directory.path().join("values.yaml"), + b"k: |\n a\n# ends the block\n # yamllint disable\nz: 1\n", + ) + .unwrap(); + + let output = run(directory.path(), &["check", "--explain"]); assert_eq!( - git(directory.path(), &["show", ":renamed target.rs"]), - b"let base = 1;\nlet renamed = 2; \n" + output.status.code(), + Some(0), + "{}", + String::from_utf8_lossy(&output.stderr) ); - assert_eq!( - git(directory.path(), &["show", &format!(":{unusual}")]), - b"let new = 1; \n" + let report = String::from_utf8(output.stdout).unwrap(); + for needle in [ + "values.yaml:3:1: kept line comment: # ends the block", + "kept: it separates a `yaml` block scalar from the kept comment below \ + it; the comment under it has to go first", + ] { + assert!( + report.contains(needle), + "`check --explain` lacks {needle:?}:\n{report}" + ); + } + /* NOTE: `all` takes the directive out, and with nothing left standing under + * the body the comment above it is ordinary again. */ + let widened = run(directory.path(), &["check", "--explain", "--policy", "all"]); + let widened = String::from_utf8(widened.stdout).unwrap(); + assert!( + widened.contains("values.yaml:3:1: removable line comment"), + "`--policy all` still kept it:\n{widened}" ); - assert_eq!( - fs::read(directory.path().join("renamed target.rs")).unwrap(), - b"let base = 1;\nlet renamed = 2; // staged rename\n" +} + +/// A setting the command line supplied is named as the command line, not as +/// the file it would otherwise have been written in. +#[test] +fn explain_names_the_command_line_when_a_flag_set_the_policy() { + let directory = tempfile::tempdir().unwrap(); + fs::write( + directory.path().join("notice.rs"), + b"// Copyright 2026 Example\nlet x = 1; // TODO\n", + ) + .unwrap(); + + let output = run( + directory.path(), + &["check", "--explain", "--policy", "legal"], ); assert_eq!( - fs::read(directory.path().join(unusual)).unwrap(), - b"let new = 1; // new file\n" + output.status.code(), + Some(1), + "{}", + String::from_utf8_lossy(&output.stderr) ); + let report = String::from_utf8(output.stdout).unwrap(); + for needle in [ + "notice.rs:1:1: kept license comment: // Copyright 2026 Example", + "kept: policy legal protects license comments, and this one says `copyright` \ + (--policy on the command line)", + "removed: policy `legal` removes ordinary comments (--policy on the command line)", + ] { + assert!( + report.contains(needle), + "`check --explain --policy legal` lacks {needle:?}:\n{report}" + ); + } } -#[cfg(unix)] +/// The machine formats are schemas, not prose, and none of them has a place to +/// put an explanation. Asking for one is a usage error rather than a flag that +/// quietly does nothing. #[test] -fn staged_non_utf8_paths_remain_os_native() { - use std::os::unix::ffi::{OsStrExt, OsStringExt}; +fn explain_is_refused_by_every_machine_format() { + let directory = tempfile::tempdir().unwrap(); + fs::write(directory.path().join("a.rs"), b"let x = 1; // TODO\n").unwrap(); + for format in ["json", "jsonl", "sarif", "github"] { + let output = run( + directory.path(), + &["check", "--explain", "--format", format], + ); + assert_eq!( + output.status.code(), + Some(2), + "`--format {format} --explain` was accepted" + ); + let error = String::from_utf8_lossy(&output.stderr); + assert!( + error.contains("--explain is only available with --format human"), + "`--format {format} --explain` said:\n{error}" + ); + assert!( + output.stdout.is_empty(), + "`--format {format} --explain` wrote a report anyway" + ); + } +} - let directory = repository(); - let name = std::ffi::OsString::from_vec(b"non-\xff.rs".to_vec()); - let path = directory.path().join(&name); - fs::write(&path, b"let value = 1; // remove\n").unwrap(); - git_with_path(directory.path(), &["add", "--"], &name); +/// `--explain` annotates a report of comments, and only `check` and `scan` +/// write one: `fix` reports the files it rewrote, `diff` writes a patch, +/// `strip` writes the stripped source, and the rest of the commands are not +/// about comments at all. The flag is global, so asking for it anywhere else +/// is a usage error rather than a flag that quietly does nothing. +#[test] +fn explain_is_refused_by_the_commands_that_write_no_report() { + let directory = tempfile::tempdir().unwrap(); + fs::write(directory.path().join("a.rs"), b"let x = 1; // TODO\n").unwrap(); + let refused: [&[&str]; 12] = [ + &["fix", "--explain", "--dry-run"], + &["fix", "--explain"], + &["diff", "--explain"], + &["strip", "--explain", "--language", "rust"], + &["lsp", "--explain"], + &["init", "--explain"], + &["config", "--explain"], + &["languages", "--explain"], + &["plugin", "--explain", "list"], + &["completions", "--explain", "bash"], + &["doctor", "--explain"], + &["man", "--explain"], + ]; + for arguments in refused { + let output = run_stdin(directory.path(), arguments, b"let x = 1; // TODO\n"); + assert_eq!( + output.status.code(), + Some(2), + "`ocomment {}` was accepted", + arguments.join(" ") + ); + let error = String::from_utf8_lossy(&output.stderr); + assert!( + error.contains("--explain is only available with `check` and `scan`"), + "`ocomment {}` said:\n{error}", + arguments.join(" ") + ); + assert!( + output.stdout.is_empty(), + "`ocomment {}` wrote a report anyway", + arguments.join(" ") + ); + } + assert_eq!( + fs::read(directory.path().join("a.rs")).unwrap(), + b"let x = 1; // TODO\n", + "a refused run rewrote the file anyway" + ); + assert!( + !directory.path().join(".ocomment.toml").exists(), + "a refused `init` wrote its starter file anyway" + ); + // NOTE: The two commands the flag is for still take it. + for command in ["check", "scan"] { + let output = run(directory.path(), &[command, "--explain"]); + assert!( + output.status.code() != Some(2), + "`ocomment {command} --explain` was refused:\n{}", + String::from_utf8_lossy(&output.stderr) + ); + assert!( + String::from_utf8_lossy(&output.stdout).contains("a.rs"), + "`ocomment {command} --explain` wrote no report" + ); + } +} - let output = run(directory.path(), &["fix", "--staged", "--index-only"]); +/// `scan` already lists every comment; `--explain` puts the reason under each +/// of its lines without disturbing the listing itself. +#[test] +fn scan_explain_annotates_every_listed_comment() { + let directory = tempfile::tempdir().unwrap(); + fs::write( + directory.path().join("a.py"), + b"#!/usr/bin/env python3\nx = 1 # remove\n", + ) + .unwrap(); + let output = run(directory.path(), &["scan", "a.py", "--explain"]); assert_eq!( output.status.code(), Some(0), "{}", String::from_utf8_lossy(&output.stderr) ); - let mut specification = std::ffi::OsString::from(":"); - specification.push(&name); + let stdout = String::from_utf8(output.stdout).unwrap(); + let lines: Vec<&str> = stdout.lines().collect(); assert_eq!( - git_with_path( - directory.path(), - &["cat-file", "blob"], - specification.as_os_str() + lines.first().copied(), + Some("a.py:1:1: shebang keep (required source preamble) 0..22: #!/usr/bin/env python3"), + "`scan --explain` changed the listing:\n{stdout}" + ); + assert!( + lines + .get(1) + .is_some_and(|line| line.starts_with(" kept: required source preamble")), + "`scan --explain` did not explain the shebang:\n{stdout}" + ); + assert_eq!( + lines.get(2).copied(), + Some("a.py:2:8: line remove 30..38: # remove"), + "`scan --explain` changed the listing:\n{stdout}" + ); + assert!( + lines.get(3).is_some_and( + |line| line.starts_with(" removed: policy `safe` removes ordinary comments") ), - b"let value = 1; \n" + "`scan --explain` did not explain the removal:\n{stdout}" ); - assert_eq!(name.as_bytes(), b"non-\xff.rs"); + assert_no_debug_leak("human scan --explain output", &stdout); } +/// A staged run reads index blobs through a path that carries no policy trace, +/// so it says so rather than printing a listing with every explanation missing. #[test] -fn ambiguous_staged_mapping_changes_nothing_and_suggests_index_only() { +fn explain_is_refused_by_a_staged_run() { let directory = repository(); - let path = directory.path().join("ambiguous.rs"); - fs::write(&path, b"let base = 1;\n").unwrap(); - git(directory.path(), &["add", "ambiguous.rs"]); - git( - directory.path(), - &["commit", "--quiet", "--message", "base"], + fs::write(directory.path().join("a.rs"), b"let x = 1; // TODO\n").unwrap(); + git(directory.path(), &["add", "a.rs"]); + let output = run(directory.path(), &["check", "--staged", "--explain"]); + assert_eq!(output.status.code(), Some(2)); + let error = String::from_utf8_lossy(&output.stderr); + assert!( + error.contains("--explain is not available with --staged"), + "`check --staged --explain` said:\n{error}" ); +} - let staged = b"let base = 1;\nlet staged = 2; // remove\n"; - fs::write(&path, staged).unwrap(); - git(directory.path(), &["add", "ambiguous.rs"]); - let working = [staged.as_slice(), staged.as_slice()].concat(); - fs::write(&path, &working).unwrap(); - let before_index = git(directory.path(), &["show", ":ambiguous.rs"]); +/// `fix -i` asks a question per comment, so it needs somebody there to answer +/// it. A piped or redirected run would otherwise read the prompt's answer out +/// of whatever the pipe carried — a script's own data — and start writing +/// files from it. The refusal names both ways out and touches nothing. +#[test] +fn fix_interactive_without_a_terminal_refuses_and_writes_nothing() { + let directory = tempfile::tempdir().unwrap(); + let path = directory.path().join("sample.rs"); + let before = b"let x = 1; // remove\n"; + fs::write(&path, before).unwrap(); - let output = run(directory.path(), &["fix", "--staged"]); + let output = run_stdin(directory.path(), &["fix", "-i", "sample.rs"], b"y\n"); assert_eq!(output.status.code(), Some(2)); - assert!(String::from_utf8_lossy(&output.stderr).contains("--index-only")); + assert_eq!(fs::read(&path).unwrap(), before); + assert_eq!(String::from_utf8(output.stdout).unwrap(), ""); assert_eq!( - git(directory.path(), &["show", ":ambiguous.rs"]), - before_index + String::from_utf8(output.stderr).unwrap(), + "ocomment: --interactive needs a terminal; run without -i or use `ocomment diff`\n" ); - assert_eq!(fs::read(&path).unwrap(), working); +} - let index_only = run(directory.path(), &["fix", "--staged", "--index-only"]); - assert_eq!(index_only.status.code(), Some(0)); +/// The long spelling refuses the same way, so a script that uses it is not +/// told something different from one that uses `-i`. +#[test] +fn fix_interactive_long_spelling_refuses_without_a_terminal() { + let directory = tempfile::tempdir().unwrap(); + fs::write( + directory.path().join("sample.rs"), + b"let x = 1; // remove\n", + ) + .unwrap(); + let output = run(directory.path(), &["fix", "--interactive", "sample.rs"]); + assert_eq!(output.status.code(), Some(2)); assert_eq!( - git(directory.path(), &["show", ":ambiguous.rs"]), - b"let base = 1;\nlet staged = 2; \n" + String::from_utf8(output.stderr).unwrap(), + "ocomment: --interactive needs a terminal; run without -i or use `ocomment diff`\n" ); - assert_eq!(fs::read(path).unwrap(), working); } +/// A machine format has no prompt to put a question on and no place to put the +/// answer, so the combination is refused rather than quietly ignoring one of +/// the two flags. It is refused before the terminal is looked at, because the +/// flag combination is wrong however the run was started. #[test] -fn normal_walk_respects_gitignore() { +fn fix_interactive_refuses_a_machine_format() { + let directory = tempfile::tempdir().unwrap(); + let path = directory.path().join("sample.rs"); + let before = b"let x = 1; // remove\n"; + fs::write(&path, before).unwrap(); + + let output = run( + directory.path(), + &["fix", "-i", "--format", "json", "sample.rs"], + ); + assert_eq!(output.status.code(), Some(2)); + assert_eq!(fs::read(&path).unwrap(), before); + assert_eq!( + String::from_utf8(output.stderr).unwrap(), + "ocomment: --interactive is only available with --format human\n" + ); +} + +/// Each of these describes a run that cannot also be interactive: the index +/// carries no working-tree file to show a hunk from, `--dry-run` writes +/// nothing whatever the answers were, and `-q` asks for a run with no +/// commentary at all. Clap refuses the pair at parse time, before any file is +/// read. +#[test] +fn fix_interactive_conflicts_with_the_flags_that_contradict_it() { let directory = repository(); - fs::write(directory.path().join(".gitignore"), b"ignored.rs\n").unwrap(); - fs::write(directory.path().join("ignored.rs"), b"// ignored\n").unwrap(); - fs::write(directory.path().join("seen.rs"), b"// seen\n").unwrap(); - let output = run(directory.path(), &[]); - assert_eq!(output.status.code(), Some(1)); - let text = String::from_utf8_lossy(&output.stdout); - assert!(text.contains("seen.rs")); - assert!(!text.contains("ignored.rs")); + let path = directory.path().join("sample.rs"); + let before = b"let x = 1; // remove\n"; + fs::write(&path, before).unwrap(); + git(directory.path(), &["add", "sample.rs"]); + + for conflicting in ["--staged", "--dry-run", "--quiet", "-q"] { + let output = run(directory.path(), &["fix", "-i", conflicting]); + assert_eq!( + output.status.code(), + Some(2), + "`fix -i {conflicting}` was accepted" + ); + let error = String::from_utf8(output.stderr).unwrap(); + assert!( + error.contains("cannot be used with"), + "`fix -i {conflicting}` did not report a conflict:\n{error}" + ); + assert_eq!( + fs::read(&path).unwrap(), + before, + "`fix -i {conflicting}` reached the file" + ); + } } +/// The flag is discoverable where the reader looks for it. #[test] -fn explicit_io_failure_returns_two_and_blocks_the_whole_fix() { +fn help_documents_the_interactive_fix() { let directory = tempfile::tempdir().unwrap(); - let path = directory.path().join("good.rs"); - let original = b"let x = 1; // remove\n"; - fs::write(&path, original).unwrap(); - let output = run(directory.path(), &["fix", "good.rs", "missing.rs"]); - assert_eq!(output.status.code(), Some(2)); - assert_eq!(fs::read(path).unwrap(), original); - assert!(String::from_utf8_lossy(&output.stdout).contains("path does not exist")); + let fixed = run(directory.path(), &["fix", "--help"]); + assert_eq!(fixed.status.code(), Some(0)); + let help = String::from_utf8(fixed.stdout).unwrap(); + assert!( + help.contains("-i, --interactive"), + "`fix --help` does not document --interactive:\n{help}" + ); } +/// The listing is a table of languages, not a report of comments, so the +/// formats that describe a report have nowhere to put it. Each is refused with +/// the pair that does work rather than answered with the human table, which is +/// what `--format json` used to be given. #[test] -fn no_argument_scan_uses_repository_root_from_a_subdirectory() { - let directory = repository(); - fs::write(directory.path().join("root.rs"), b"// root comment\n").unwrap(); - let nested = directory.path().join("nested/deeper"); - fs::create_dir_all(&nested).unwrap(); - let output = run(&nested, &[]); - assert_eq!(output.status.code(), Some(1)); - assert!(String::from_utf8_lossy(&output.stdout).contains("root.rs")); +fn languages_refuses_the_formats_that_carry_no_table() { + let directory = tempfile::tempdir().unwrap(); + for format in ["jsonl", "sarif", "github"] { + let output = run(directory.path(), &["languages", "--format", format]); + assert_eq!( + output.status.code(), + Some(2), + "`languages --format {format}` was accepted" + ); + assert!( + output.stdout.is_empty(), + "`languages --format {format}` wrote a listing anyway" + ); + let error = String::from_utf8(output.stderr).unwrap(); + assert!( + error.contains("only available with --format human or --format json"), + "`languages --format {format}` said:\n{error}" + ); + } + for format in ["human", "json"] { + let output = run(directory.path(), &["languages", "--format", format]); + assert_eq!( + output.status.code(), + Some(0), + "`languages --format {format}` was refused: {}", + String::from_utf8_lossy(&output.stderr) + ); + let listing = String::from_utf8(output.stdout).unwrap(); + for language in ["rust", "objective-cpp", "xhtml", "kotlin", "toml"] { + assert!( + listing.contains(language), + "`languages --format {format}` omits `{language}`:\n{listing}" + ); + } + } } +/// A Scala file keeps its scala-cli directive and hides a `//` inside an XML +/// literal's text, while comments inside an interpolation and a nested block +/// comment are removed. #[test] -fn explicit_directory_bypasses_hidden_and_size_limits() { +fn a_scala_file_keeps_its_directive_and_hides_xml_text() { let directory = tempfile::tempdir().unwrap(); + let path = directory.path().join("Main.scala"); fs::write( - directory.path().join(".ocomment.toml"), - b"version = 1\n[files]\nmax_size = 1\n", + &path, + b"//> using scala \"3.3.0\"\nval a =
// text\nval b = s\"${1 /* keep */}\" // remove\n/* outer /* inner */ */\n", ) .unwrap(); + + let scanned = run( + directory.path(), + &["scan", "Main.scala", "--format", "json"], + ); + assert_eq!( + scanned.status.code(), + Some(0), + "{}", + String::from_utf8_lossy(&scanned.stderr) + ); + let document: serde_json::Value = serde_json::from_slice(&scanned.stdout).unwrap(); + let report = &document["files"][0]["report"]; + assert_eq!(report["language"], "scala"); + assert_eq!(report["comments"].as_array().unwrap().len(), 4); + assert_eq!(report["comments"][0]["kind"], "directive"); + assert_eq!(report["comments"][0]["disposition"]["action"], "keep"); + assert_eq!(report["comments"][0]["span"]["start"], 0); + assert_eq!(report["comments"][0]["span"]["end"], 23); + assert_eq!(report["comments"][1]["span"]["start"], 61); + assert_eq!(report["comments"][1]["span"]["end"], 71); + assert_eq!(report["comments"][2]["span"]["start"], 74); + assert_eq!(report["comments"][2]["span"]["end"], 83); + assert_eq!(report["comments"][3]["span"]["start"], 84); + assert_eq!(report["comments"][3]["span"]["end"], 107); + + let fixed = run(directory.path(), &["fix", "Main.scala"]); + assert_eq!( + fixed.status.code(), + Some(0), + "{}", + String::from_utf8_lossy(&fixed.stderr) + ); + assert_eq!( + fs::read(&path).unwrap(), + b"//> using scala \"3.3.0\"\nval a = // text\nval b = s\"${1 }\" \n\n" + ); +} + +/// A Vue file scans its template, script and style blocks, keeps the +/// template's HTML comment, and removes a comment from the mustache and from +/// each embedded language. +#[test] +fn a_vue_file_scans_its_template_script_and_style_blocks() { + let directory = tempfile::tempdir().unwrap(); + let path = directory.path().join("App.vue"); fs::write( - directory.path().join(".hidden.rs"), - b"// hidden and larger than one byte\n", + &path, + b"\n\n\n", ) .unwrap(); - let output = run(directory.path(), &["check", "."]); - assert_eq!(output.status.code(), Some(1)); - assert!(String::from_utf8_lossy(&output.stdout).contains(".hidden.rs")); + + let scanned = run(directory.path(), &["scan", "App.vue", "--format", "json"]); + assert_eq!( + scanned.status.code(), + Some(0), + "{}", + String::from_utf8_lossy(&scanned.stderr) + ); + let document: serde_json::Value = serde_json::from_slice(&scanned.stdout).unwrap(); + let report = &document["files"][0]["report"]; + assert_eq!(report["language"], "vue"); + assert_eq!(report["comments"].as_array().unwrap().len(), 4); + assert_eq!(report["comments"][0]["kind"], "html-comment"); + assert_eq!(report["comments"][0]["disposition"]["action"], "keep"); + assert_eq!(report["comments"][1]["span"]["start"], 35); + assert_eq!(report["comments"][1]["span"]["end"], 42); + assert_eq!(report["comments"][2]["span"]["start"], 89); + assert_eq!(report["comments"][3]["span"]["start"], 125); + + let fixed = run(directory.path(), &["fix", "App.vue"]); + assert_eq!( + fixed.status.code(), + Some(0), + "{}", + String::from_utf8_lossy(&fixed.stderr) + ); + assert_eq!( + fs::read(&path).unwrap(), + b"\n\n\n" + ); } -#[cfg(unix)] +/// A Markdown file scans its fenced code blocks as their named languages and +/// keeps its HTML comment, while inline code stays opaque. #[test] -fn symlink_following_is_explicitly_configurable() { - use std::os::unix::fs::symlink; - +fn a_markdown_file_scans_its_fenced_code_blocks() { let directory = tempfile::tempdir().unwrap(); + let path = directory.path().join("README.md"); fs::write( - directory.path().join("source.txt"), - b"let x = 1; // remove\n", + &path, + b"# notes\n\n```rust\n// c\n```\n`// inline`\n", ) .unwrap(); - symlink("source.txt", directory.path().join("link.rs")).unwrap(); - let skipped = run(directory.path(), &["check", "link.rs"]); - assert_eq!(skipped.status.code(), Some(0)); - assert!(String::from_utf8_lossy(&skipped.stdout).contains("symbolic link")); + let scanned = run(directory.path(), &["scan", "README.md", "--format", "json"]); + assert_eq!( + scanned.status.code(), + Some(0), + "{}", + String::from_utf8_lossy(&scanned.stderr) + ); + let document: serde_json::Value = serde_json::from_slice(&scanned.stdout).unwrap(); + let report = &document["files"][0]["report"]; + assert_eq!(report["language"], "markdown"); + assert_eq!(report["comments"].as_array().unwrap().len(), 2); + assert_eq!(report["comments"][0]["kind"], "html-comment"); + assert_eq!(report["comments"][0]["disposition"]["action"], "keep"); + assert_eq!(report["comments"][1]["span"]["start"], 30); + assert_eq!(report["comments"][1]["span"]["end"], 34); + + let fixed = run(directory.path(), &["fix", "README.md"]); + assert_eq!( + fixed.status.code(), + Some(0), + "{}", + String::from_utf8_lossy(&fixed.stderr) + ); + assert_eq!( + fs::read(&path).unwrap(), + b"# notes\n\n```rust\n\n```\n`// inline`\n" + ); +} +/// A Perl file hides the `#` in its quote words and regexes and keeps its POD +/// opaque, while a division's `#` is a comment. +#[test] +fn a_perl_file_hides_quote_words_and_keeps_pod_opaque() { + let directory = tempfile::tempdir().unwrap(); + let path = directory.path().join("script.pl"); fs::write( - directory.path().join(".ocomment.toml"), - b"version = 1\n[files]\nfollow_symlinks = true\n", + &path, + b"=head1 NAME\n# not a comment\n=cut\nmy $x = 'a#b';\nmy $y = $x / 2; # division\n", ) .unwrap(); - let followed = run(directory.path(), &["check", "link.rs"]); - assert_eq!(followed.status.code(), Some(1)); - assert!(String::from_utf8_lossy(&followed.stdout).contains("removable")); + + let scanned = run(directory.path(), &["scan", "script.pl", "--format", "json"]); + assert_eq!( + scanned.status.code(), + Some(0), + "{}", + String::from_utf8_lossy(&scanned.stderr) + ); + let document: serde_json::Value = serde_json::from_slice(&scanned.stdout).unwrap(); + let report = &document["files"][0]["report"]; + assert_eq!(report["language"], "perl"); + assert_eq!(report["comments"].as_array().unwrap().len(), 1); + assert_eq!(report["comments"][0]["span"]["start"], 64); + assert_eq!(report["comments"][0]["span"]["end"], 74); + + let fixed = run(directory.path(), &["fix", "script.pl"]); + assert_eq!( + fixed.status.code(), + Some(0), + "{}", + String::from_utf8_lossy(&fixed.stderr) + ); + assert_eq!( + fs::read(&path).unwrap(), + b"=head1 NAME\n# not a comment\n=cut\nmy $x = 'a#b';\nmy $y = $x / 2; \n" + ); } diff --git a/rust/ocomment/tests/lsp.rs b/rust/ocomment/tests/lsp.rs index 2bc5456..a46785a 100644 --- a/rust/ocomment/tests/lsp.rs +++ b/rust/ocomment/tests/lsp.rs @@ -302,7 +302,7 @@ fn protocol_supports_utf8_pull_workspace_actions_and_stale_versions() { })); assert!(client.response(6)["result"].is_null()); - // A stale update must not replace the current document snapshot. + // NOTE: A stale update must not replace the current document snapshot. client.send(json!({ "jsonrpc": "2.0", "method": "textDocument/didChange", "params": { @@ -449,3 +449,133 @@ fn on_save_is_opt_in_and_returns_annotated_safe_edits() { assert_eq!(response["result"][0]["range"]["start"]["character"], 11); client.stop(); } + +#[test] +fn diagnostics_and_hover_name_comment_kinds_in_canonical_spelling() { + let workspace = tempfile::tempdir().unwrap(); + let uri = Url::from_file_path(workspace.path().join("doc.rs")).unwrap(); + let mut client = LspClient::start(workspace.path()); + let _ = client.initialize(workspace.path(), &["utf-8"]); + client.send(json!({ + "jsonrpc": "2.0", "method": "textDocument/didOpen", + "params": { "textDocument": { + "uri": uri, "languageId": "rust", "version": 1, + "text": "/** doc */\n// SPDX-License-Identifier: MIT\n" + }} + })); + let pushed = client.notification("textDocument/publishDiagnostics"); + let diagnostics = pushed["params"]["diagnostics"].as_array().unwrap(); + assert_eq!(diagnostics[0]["message"], "removable doc-block comment"); + assert_eq!(diagnostics[0]["code"], "removable-comment"); + + client.send(json!({ + "jsonrpc": "2.0", "id": 60, "method": "textDocument/hover", + "params": { + "textDocument": { "uri": uri }, + "position": { "line": 0, "character": 3 } + } + })); + assert_eq!( + client.response(60)["result"]["contents"], + "OComment: removable doc-block comment" + ); + + client.stop(); +} + +#[test] +fn hover_over_a_protected_comment_names_the_kind_and_reason() { + let workspace = tempfile::tempdir().unwrap(); + let uri = Url::from_file_path(workspace.path().join("preamble.py")).unwrap(); + let mut client = LspClient::start(workspace.path()); + let _ = client.initialize(workspace.path(), &["utf-8"]); + client.send(json!({ + "jsonrpc": "2.0", "method": "textDocument/didOpen", + "params": { "textDocument": { + "uri": uri, "languageId": "python", "version": 1, + "text": "#!/usr/bin/env python3\nx = 1\n" + }} + })); + let _ = client.notification("textDocument/publishDiagnostics"); + client.send(json!({ + "jsonrpc": "2.0", "id": 61, "method": "textDocument/hover", + "params": { + "textDocument": { "uri": uri }, + "position": { "line": 0, "character": 3 } + } + })); + assert_eq!( + client.response(61)["result"]["contents"], + "OComment: kept shebang comment: required source preamble" + ); + client.stop(); +} + +#[test] +fn editor_language_ids_name_languages_the_path_alone_would_not() { + let workspace = tempfile::tempdir().unwrap(); + let mut client = LspClient::start(workspace.path()); + let _ = client.initialize(workspace.path(), &["utf-8"]); + + /* NOTE: Neither name carries an extension and neither buffer opens with a + * shebang, so the client's `languageId` is the only thing that can say + * what the bytes are. An id the server cannot place leaves the document + * `unknown`, which is an error diagnostic rather than a comment, so both + * assertions below check the comment and not merely the count. */ + let shell = Url::from_file_path(workspace.path().join("hook")).unwrap(); + client.send(json!({ + "jsonrpc": "2.0", "method": "textDocument/didOpen", + "params": { "textDocument": { + "uri": shell, "languageId": "shellscript", "version": 1, + "text": "echo hi # remove\n" + }} + })); + let pushed = client.notification("textDocument/publishDiagnostics"); + assert_eq!(pushed["params"]["uri"], shell.as_str()); + let diagnostics = pushed["params"]["diagnostics"].as_array().unwrap(); + assert_eq!(diagnostics.len(), 1, "unexpected diagnostics: {pushed}"); + assert_eq!(diagnostics[0]["code"], "removable-comment"); + assert_eq!(diagnostics[0]["range"]["start"]["character"], 8); + + let cuda = Url::from_file_path(workspace.path().join("kernel")).unwrap(); + client.send(json!({ + "jsonrpc": "2.0", "method": "textDocument/didOpen", + "params": { "textDocument": { + "uri": cuda, "languageId": "cuda-cpp", "version": 1, + "text": "int x = 1; // remove\n" + }} + })); + let pushed = client.notification("textDocument/publishDiagnostics"); + assert_eq!(pushed["params"]["uri"], cuda.as_str()); + let diagnostics = pushed["params"]["diagnostics"].as_array().unwrap(); + assert_eq!(diagnostics.len(), 1, "unexpected diagnostics: {pushed}"); + assert_eq!(diagnostics[0]["code"], "removable-comment"); + assert_eq!(diagnostics[0]["range"]["start"]["character"], 11); + + client.stop(); +} + +#[test] +fn shellscript_keeps_the_dialect_the_path_implies() { + let workspace = tempfile::tempdir().unwrap(); + let uri = Url::from_file_path(workspace.path().join("script.bash")).unwrap(); + let mut client = LspClient::start(workspace.path()); + let _ = client.initialize(workspace.path(), &["utf-8"]); + /* NOTE: `$'...'` is ANSI-C quoting in Bash and zsh only. Read as POSIX sh + * the string ends at the escaped quote and the comment starts eleven + * columns earlier, on `#1'`. One editor id, `shellscript`, covers all + * three shells, so the dialect still has to come from the path. */ + client.send(json!({ + "jsonrpc": "2.0", "method": "textDocument/didOpen", + "params": { "textDocument": { + "uri": uri, "languageId": "shellscript", "version": 1, + "text": "printf $'it\\'s #1' # remove\n" + }} + })); + let pushed = client.notification("textDocument/publishDiagnostics"); + let diagnostics = pushed["params"]["diagnostics"].as_array().unwrap(); + assert_eq!(diagnostics.len(), 1, "unexpected diagnostics: {pushed}"); + assert_eq!(diagnostics[0]["code"], "removable-comment"); + assert_eq!(diagnostics[0]["range"]["start"]["character"], 19); + client.stop(); +} diff --git a/rust/ocomment/tests/source_guards.rs b/rust/ocomment/tests/source_guards.rs new file mode 100644 index 0000000..14995cd --- /dev/null +++ b/rust/ocomment/tests/source_guards.rs @@ -0,0 +1,209 @@ +//! Guards that read this crate's own source text. +//! +//! A test that runs the binary can only catch a bypass on the paths it +//! happens to exercise. These read the sources instead, so an invariant that +//! holds today cannot be broken quietly by a line added tomorrow. + +use std::{collections::BTreeSet, fs, path::PathBuf}; + +/// Every source file of the crate, embedded at compile time so the scan does +/// not depend on the directory the test runs in. `the_guard_reads_every_source` +/// keeps this list equal to what is on disk. +const SOURCES: [(&str, &str); 11] = [ + ("atomic.rs", include_str!("../src/atomic.rs")), + ("cli.rs", include_str!("../src/cli.rs")), + ("config.rs", include_str!("../src/config.rs")), + ("files.rs", include_str!("../src/files.rs")), + ("git.rs", include_str!("../src/git.rs")), + ("interactive.rs", include_str!("../src/interactive.rs")), + ("lsp.rs", include_str!("../src/lsp.rs")), + ("main.rs", include_str!("../src/main.rs")), + ("output.rs", include_str!("../src/output.rs")), + ("plugin.rs", include_str!("../src/plugin.rs")), + ("values.rs", include_str!("../src/values.rs")), +]; + +/// The names this crate gives a handle on the program's standard output: the +/// locked writer `output::stdout()` returns is bound as `stdout`, and every +/// function that is handed it takes it as `output`. Nothing else in the crate +/// is written to under either name. +const STDOUT_HANDLES: [&str; 2] = ["stdout", "output"]; + +/// The write macros, matched with their opening parenthesis so the target is +/// the text that follows. +const MACROS: [&str; 2] = ["write!(", "writeln!("]; + +/// The method form of the same write. +const METHOD: &str = ".write_all("; + +/// The call every write to standard output is raised through. +const WRAPPER: &str = "wrote("; + +/// One write whose target is a standard-output handle. +struct StdoutWrite { + line: usize, + /// The call sits directly inside `wrote(` — or `output::wrote(`. + wrapped: bool, +} + +/// Every write to the program's own standard output is raised through +/// [`output::wrote`], which tags a lost reader as `OutputPipeClosed` so `main` +/// can end quietly for that case and only that case. A raw `writeln!` would +/// return a bare `BrokenPipe` that the chain cannot tell apart from a real +/// failure — `git hash-object` dropping the blob it was being handed, say — +/// and `ocomment … | head` would start failing runs, or a failed staged fix +/// would start passing. +/// +/// The invariant checked here is textual: a `write!`, `writeln!`, or +/// `write_all` whose target names a standard-output handle must have `wrote(` +/// immediately in front of it. It is deliberately syntactic rather than +/// semantic — it cannot know what a handle is, only what it is called — so it +/// leans on the naming convention above and on +/// `standard_output_is_locked_in_exactly_one_place`, which keeps a writer from +/// being conjured anonymously under some other name. +#[test] +fn every_write_to_standard_output_goes_through_wrote() { + let mut wrapped = 0; + let mut bare = Vec::new(); + for (name, source) in SOURCES { + for call in stdout_writes(source) { + if call.wrapped { + wrapped += 1; + } else { + bare.push(format!("src/{name}:{}", call.line)); + } + } + } + assert!( + bare.is_empty(), + "these writes to standard output bypass `output::wrote`, so a reader \ + that closed the pipe would surface as an unrecognizable I/O failure: \ + {bare:?}" + ); + // NOTE: A scan that matches nothing would pass this test forever. + assert!( + wrapped >= 30, + "the guard recognized only {wrapped} writes to standard output, far \ + fewer than the crate makes; the naming convention it reads must have \ + changed, and the guard with it" + ); +} + +/// The handle can only be watched by name if it is only ever made in one +/// place. `output::stdout()` locks standard output for the whole run; nowhere +/// else may turn it into a writer, whether by locking it, writing to it, or +/// flushing it. (Naming it to ask whether it is a terminal is not writing to +/// it, and neither is handing the LSP server its own protocol channel.) +#[test] +fn standard_output_is_locked_in_exactly_one_place() { + let mut offenders = Vec::new(); + for (name, source) in SOURCES { + if name == "output.rs" { + continue; + } + for (at, _) in source.match_indices("stdout()") { + let rest = &source[at + "stdout()".len()..]; + if [".lock()", ".write", ".flush"] + .iter() + .any(|call| rest.starts_with(call)) + { + offenders.push(format!("src/{name}:{}", line_of(source, at))); + } + } + } + assert!( + offenders.is_empty(), + "standard output is written through a handle made outside \ + `output::stdout`, where the pipe guard cannot see it: {offenders:?}" + ); +} + +/// The embedded list is the whole crate, so a module added later is scanned +/// rather than silently exempt. +#[test] +fn the_guard_reads_every_source() { + let root = PathBuf::from(env!("CARGO_MANIFEST_DIR")).join("src"); + let mut on_disk = BTreeSet::new(); + let mut pending = vec![(String::new(), root)]; + while let Some((prefix, directory)) = pending.pop() { + for entry in fs::read_dir(&directory).unwrap() { + let entry = entry.unwrap(); + let name = entry.file_name().to_str().unwrap().to_owned(); + let name = format!("{prefix}{name}"); + if entry.file_type().unwrap().is_dir() { + pending.push((format!("{name}/"), entry.path())); + } else if name.ends_with(".rs") { + on_disk.insert(name); + } + } + } + let scanned: BTreeSet = SOURCES.iter().map(|(name, _)| (*name).to_owned()).collect(); + assert_eq!( + scanned, on_disk, + "the source list this file scans is not the source list on disk" + ); +} + +/// Every write in `source` whose target names a standard-output handle. +fn stdout_writes(source: &str) -> Vec { + let mut writes = Vec::new(); + for marker in MACROS { + for (at, _) in source.match_indices(marker) { + let target = handle(first_argument(&source[at + marker.len()..])); + if STDOUT_HANDLES.contains(&target) { + writes.push(StdoutWrite { + line: line_of(source, at), + wrapped: wraps(&source[..at]), + }); + } + } + } + for (at, _) in source.match_indices(METHOD) { + let receiver = identifier_before(source, at); + if STDOUT_HANDLES.contains(&receiver) { + writes.push(StdoutWrite { + line: line_of(source, at), + wrapped: wraps(&source[..at - receiver.len()]), + }); + } + } + writes +} + +/// Whether the call that follows `prefix` sits directly inside `wrote(`. +fn wraps(prefix: &str) -> bool { + prefix.trim_end().ends_with(WRAPPER) +} + +/// The 1-based line byte `at` falls on. +fn line_of(source: &str, at: usize) -> usize { + source[..at].matches('\n').count() + 1 +} + +/// The first argument of a call, given everything after its opening +/// parenthesis. A target too involved to end at the first comma — a call of +/// its own, say — comes back as something no handle is named, and the write is +/// left to `standard_output_is_locked_in_exactly_one_place`. +fn first_argument(rest: &str) -> &str { + let end = rest.find([',', ')']).unwrap_or(rest.len()); + rest[..end].trim() +} + +/// A target expression reduced to the name it writes through, so that +/// `&mut output` and `output` are the same handle. +fn handle(target: &str) -> &str { + let target = target.trim_start_matches('&').trim_start(); + target.strip_prefix("mut ").unwrap_or(target).trim() +} + +/// The identifier ending at byte `at`, empty when the byte before it is not +/// part of one. +fn identifier_before(source: &str, at: usize) -> &str { + let start = source[..at] + .char_indices() + .rev() + .take_while(|(_, character)| character.is_ascii_alphanumeric() || *character == '_') + .last() + .map_or(at, |(index, _)| index); + &source[start..at] +} diff --git a/rust/ocomment/tests/spec_languages.rs b/rust/ocomment/tests/spec_languages.rs new file mode 100644 index 0000000..458e22d --- /dev/null +++ b/rust/ocomment/tests/spec_languages.rs @@ -0,0 +1,713 @@ +//! The shared language table, the detector, and the command that prints it. +//! +//! `spec/languages.toml` is what this repository publishes as the list of +//! languages OComment understands: the `files:` pattern of the pre-commit +//! hooks, the documentation page, and `ocomment languages` all come out of it. +//! A table like that is only worth publishing while it is true, so every claim +//! it makes is checked here against the code that would have to honour it — +//! `ocomment_core::detect_language` for the file names, the binary itself for +//! the dialects and for the listing — and the JSON listing is checked against +//! the table byte for byte, so the two cannot drift apart quietly. + +use ocomment_core::{Dialect, Language, detect_language}; +use serde::Deserialize; +use serde_json::{Map, Value, json}; +use std::{ + collections::{BTreeMap, BTreeSet}, + io::Write, + path::Path, + process::{Command, Output, Stdio}, +}; + +/// The canonical table, as it sits in `spec/`. +const SPEC: &str = include_str!("../../../spec/languages.toml"); + +/// The copy the `ocomment` crate publishes and reads at run time. +/// `tools/check_embedded_specs.py` guards the same equality outside `cargo`. +const EMBEDDED: &str = include_str!("../assets/languages.toml"); + +/// The detector's own source, read as text so the table can be checked in the +/// other direction as well: what the spec claims is checked by running the +/// detector, and what the detector knows is read out of the file it is written +/// in, since nothing enumerates it at run time. +const DETECT: &str = include_str!("../../ocomment-core/src/detect.rs"); + +/// The prose that states, in words or in figures, how many languages OComment +/// scans or how many editor language identifiers the extension attaches to. +/// Nothing derives these sentences, so nothing but a test stops the next +/// language from leaving them behind — and a description that undercounts is +/// read by everyone who installs the extension. +const COMPARISON: &str = include_str!("../../../docs/comparison.md"); +const EDITORS: &str = include_str!("../../../docs/editors.md"); +const CHANGELOG: &str = include_str!("../../../CHANGELOG.md"); +const VSCODE_PACKAGE: &str = include_str!("../../../editors/vscode/package.json"); +const VSCODE_README: &str = include_str!("../../../editors/vscode/README.md"); +const VSCODE_CHANGELOG: &str = include_str!("../../../editors/vscode/CHANGELOG.md"); + +/// The extensions `detect_language` knows that `spec/languages.toml` does not +/// publish. There are none: an extension the detector answers to is one the +/// hooks match, the documentation lists and `ocomment languages` prints. +/// +/// The list stays here rather than the assertion being narrowed to "empty", +/// because it is what an extension added to the detector alone lands in, and +/// what it costs to put one here is the point: publishing one means +/// regenerating the `files:` pattern of `.pre-commit-hooks.yaml` from the spec +/// — `python3 tools/check_hooks.py --print-pattern` — and the languages page +/// with it, so an extension left out is left out on purpose and in writing. +const UNPUBLISHED_EXTENSIONS: [&str; 0] = []; + +/// One language of the shared table. +/// +/// `deny_unknown_fields` is what makes a typo in the spec a failing test rather +/// than a key that silently claims nothing. +#[derive(Debug, Deserialize)] +#[serde(deny_unknown_fields)] +struct Entry { + /// The canonical language name, which is also its serde spelling. + name: String, + /// Every file extension that selects this language, without the dot. + extensions: Vec, + /// Every dialect the language accepts, in the order the binary lists them. + dialects: Vec, + /// The extensions that select a dialect other than `standard`. + #[serde(default)] + extension_dialects: BTreeMap, + /// Whole file names that select the language when the extension does not. + #[serde(default)] + reserved_names: Vec, + /// Interpreter names that select the language from a `#!` line. + #[serde(default)] + shebangs: Vec, + /// A short remark for the human listing. + #[serde(default)] + notes: Option, +} + +/// The shared table as a whole. +#[derive(Debug, Deserialize)] +#[serde(deny_unknown_fields)] +struct Table { + /// The schema version of the file; only `1` exists. + version: u32, + /// One entry per built-in language, in the order the binary lists them. + languages: Vec, +} + +fn table() -> Table { + let parsed: Table = toml::from_str(SPEC).expect("spec/languages.toml is valid TOML"); + assert_eq!(parsed.version, 1, "unknown spec/languages.toml version"); + parsed +} + +/// The dialect the spec claims for an extension: the one it names, or +/// `standard` when it names none. +fn claimed_dialect(entry: &Entry, extension: &str) -> Dialect { + let name = entry + .extension_dialects + .get(extension) + .map_or("standard", String::as_str); + name.parse() + .unwrap_or_else(|error| panic!("`{name}` is not a dialect: {error}")) +} + +fn language(name: &str) -> Language { + name.parse() + .unwrap_or_else(|error| panic!("`{name}` is not a language: {error}")) +} + +/// Run the built binary somewhere no configuration file of this machine can +/// reach it, with `input` on its standard input. +fn run(arguments: &[&str], input: &[u8]) -> Output { + let home = tempfile::tempdir().unwrap(); + let mut child = Command::new(env!("CARGO_BIN_EXE_ocomment")) + .current_dir(home.path()) + .env("PATH", "/usr/bin:/bin") + .env("HOME", home.path()) + .env("XDG_CONFIG_HOME", home.path().join("config")) + .env("NO_COLOR", "1") + .args(arguments) + .stdin(Stdio::piped()) + .stdout(Stdio::piped()) + .stderr(Stdio::piped()) + .spawn() + .unwrap(); + child + .stdin + .take() + .expect("standard input was piped") + .write_all(input) + .unwrap(); + child.wait_with_output().unwrap() +} + +/// The spec lists every language the binary has, under the name the binary +/// uses for it, in the same order. A language added to the core enum without a +/// row here would ship undocumented and unhooked. +#[test] +fn the_spec_lists_every_built_in_language() { + let listed: Vec = table() + .languages + .iter() + .map(|entry| entry.name.clone()) + .collect(); + let built_in: Vec = Language::ALL + .iter() + .map(|value| value.as_str().to_owned()) + .collect(); + assert_eq!( + listed, built_in, + "spec/languages.toml and `Language::ALL` disagree" + ); +} + +/// Every extension in the spec really selects the language it is listed under, +/// and the dialect the spec claims for it. `.m` is Objective-C and `.cu` is +/// CUDA, and a table that says so has to be right about it. +#[test] +fn every_listed_extension_detects_its_language() { + for entry in table().languages { + let expected = language(&entry.name); + for extension in &entry.extensions { + let name = format!("sample.{extension}"); + let found = detect_language(Some(Path::new(&name)), b"") + .unwrap_or_else(|| panic!("`{name}` is detected as nothing")); + let dialect = claimed_dialect(&entry, extension); + assert_eq!( + (found.language, found.dialect, found.reason), + (expected, dialect, "extension"), + "`{name}` is not what spec/languages.toml says it is" + ); + assert!( + entry.dialects.contains(&dialect.as_str().to_owned()), + "`.{extension}` selects `{dialect}`, which `{}` does not list", + entry.name + ); + } + for extension in entry.extension_dialects.keys() { + assert!( + entry.extensions.contains(extension), + "`{}` names a dialect for `.{extension}`, which it does not list", + entry.name + ); + } + } +} + +/// Every whole file name in the spec selects the language it is listed under. +/// `Dockerfile` has no extension to go on, so this is the only claim there is. +#[test] +fn every_listed_reserved_name_detects_its_language() { + let mut checked = 0; + for entry in table().languages { + let expected = language(&entry.name); + for reserved in &entry.reserved_names { + let found = detect_language(Some(Path::new(reserved)), b"") + .unwrap_or_else(|| panic!("`{reserved}` is detected as nothing")); + assert_eq!( + (found.language, found.reason), + (expected, "reserved-filename"), + "`{reserved}` is not what spec/languages.toml says it is" + ); + checked += 1; + } + } + assert!(checked > 0, "the spec claims no reserved file names"); +} + +/// Every interpreter name in the spec selects the language it is listed under +/// when it turns up in a `#!` line, which is all a piped script has to go on. +#[test] +fn every_listed_shebang_detects_its_language() { + let mut checked = 0; + for entry in table().languages { + let expected = language(&entry.name); + for interpreter in &entry.shebangs { + let line = format!("#!/usr/bin/env {interpreter}\n"); + let found = detect_language(None, line.as_bytes()) + .unwrap_or_else(|| panic!("`{line:?}` is detected as nothing")); + assert_eq!( + (found.language, found.reason), + (expected, "shebang"), + "`{interpreter}` is not what spec/languages.toml says it is" + ); + checked += 1; + } + } + assert!(checked > 0, "the spec claims no shebangs"); +} + +/// The dialects the spec lists for a language are exactly the ones the binary +/// accepts for it, in the same order. +/// +/// Both halves are read out of the binary rather than out of a second table in +/// this file: naming a dialect the language does not have is refused with the +/// list of the ones it does, which is `config::supported_dialects` verbatim, so +/// one refused run per language proves the whole row — and every dialect the +/// row lists is then run to prove the refusal was not lying about it. +#[test] +fn the_listed_dialects_are_the_dialects_the_binary_accepts() { + for entry in table().languages { + let listed: BTreeSet<&str> = entry.dialects.iter().map(String::as_str).collect(); + let absent = Dialect::ALL + .iter() + .find(|value| !listed.contains(value.as_str())) + .expect("no language accepts every dialect"); + let refused = run( + &[ + "strip", + "--language", + &entry.name, + "--dialect", + absent.as_str(), + ], + b"", + ); + assert_eq!( + refused.status.code(), + Some(2), + "`--dialect {absent}` was accepted for {}", + entry.name + ); + let error = String::from_utf8_lossy(&refused.stderr); + let supported = error + .split_once("supported: ") + .unwrap_or_else(|| panic!("{} refused `{absent}` with: {error}", entry.name)) + .1 + .trim_end() + .to_owned(); + assert_eq!( + supported, + entry.dialects.join(", "), + "spec/languages.toml lists the wrong dialects for {}", + entry.name + ); + for dialect in &entry.dialects { + let accepted = run( + &["strip", "--language", &entry.name, "--dialect", dialect], + b"", + ); + assert_eq!( + accepted.status.code(), + Some(0), + "`--language {} --dialect {dialect}` was refused: {}", + entry.name, + String::from_utf8_lossy(&accepted.stderr) + ); + } + } +} + +/// The language and dialect enumerations of the published JSON schemas are the +/// same vocabulary as the spec table. `result.schema.json` describes what a run +/// reports and so also carries `unknown`, which is the only difference allowed. +#[test] +fn the_schemas_enumerate_the_same_vocabulary() { + let config: Value = serde_json::from_str(include_str!("../../../spec/config.schema.json")) + .expect("spec/config.schema.json is valid JSON"); + let result: Value = serde_json::from_str(include_str!("../../../spec/result.schema.json")) + .expect("spec/result.schema.json is valid JSON"); + let names = |schema: &Value, definition: &str| -> Vec { + schema["$defs"][definition]["enum"] + .as_array() + .unwrap_or_else(|| panic!("`{definition}` has no enum")) + .iter() + .map(|value| value.as_str().expect("enum values are strings").to_owned()) + .collect() + }; + let spec = table(); + let languages: Vec = spec + .languages + .iter() + .map(|entry| entry.name.clone()) + .collect(); + assert_eq!( + names(&config, "language"), + languages, + "spec/config.schema.json and spec/languages.toml disagree" + ); + let mut reported = languages.clone(); + reported.push(Language::Unknown.as_str().to_owned()); + assert_eq!( + names(&result, "language"), + reported, + "spec/result.schema.json and spec/languages.toml disagree" + ); + let dialects: BTreeSet = spec + .languages + .iter() + .flat_map(|entry| entry.dialects.iter().cloned()) + .collect(); + assert_eq!( + names(&config, "dialect") + .into_iter() + .collect::>(), + dialects, + "spec/config.schema.json and spec/languages.toml disagree about dialects" + ); + assert_eq!( + names(&config, "dialect"), + Dialect::ALL + .iter() + .map(|value| value.as_str().to_owned()) + .collect::>(), + "spec/config.schema.json and `Dialect::ALL` disagree" + ); +} + +/// The crate publishes and reads the spec table itself, so what a released +/// binary prints cannot be a copy that was edited on its own. +#[test] +fn the_embedded_table_is_the_spec_table() { + assert_eq!( + EMBEDDED, SPEC, + "rust/ocomment/assets/languages.toml differs from spec/languages.toml" + ); +} + +/// `ocomment languages --format json` is the shared table, rendered. Every key +/// the spec sets is in the JSON, and every key it leaves out is absent rather +/// than empty. +#[test] +fn the_json_listing_is_the_spec_table() { + let listed = run(&["languages", "--format", "json"], b""); + assert_eq!( + listed.status.code(), + Some(0), + "{}", + String::from_utf8_lossy(&listed.stderr) + ); + let printed: Value = + serde_json::from_slice(&listed.stdout).expect("`languages --format json` writes JSON"); + let expected: Vec = table() + .languages + .iter() + .map(|entry| { + let mut object = Map::new(); + object.insert("name".to_owned(), json!(entry.name)); + object.insert("extensions".to_owned(), json!(entry.extensions)); + object.insert("dialects".to_owned(), json!(entry.dialects)); + if !entry.extension_dialects.is_empty() { + object.insert( + "extension_dialects".to_owned(), + json!(entry.extension_dialects), + ); + } + if !entry.reserved_names.is_empty() { + object.insert("reserved_names".to_owned(), json!(entry.reserved_names)); + } + if !entry.shebangs.is_empty() { + object.insert("shebangs".to_owned(), json!(entry.shebangs)); + } + if let Some(notes) = &entry.notes { + object.insert("notes".to_owned(), json!(notes)); + } + Value::Object(object) + }) + .collect(); + assert_eq!( + printed, + Value::Array(expected), + "`ocomment languages --format json` is not spec/languages.toml" + ); +} + +/// The human listing is the same table in columns: one row per language, in +/// spec order, carrying the extensions and dialects the spec gives it. +#[test] +fn the_human_listing_is_the_spec_table() { + let listed = run(&["languages"], b""); + assert_eq!(listed.status.code(), Some(0)); + let text = String::from_utf8(listed.stdout).expect("the listing is text"); + let mut rows = text.lines(); + assert_eq!( + rows.next(), + Some("language\textensions\tdialects\tnotes"), + "the listing has no header" + ); + for entry in table().languages { + let row = rows + .next() + .unwrap_or_else(|| panic!("`{}` has no row", entry.name)); + let columns: Vec<&str> = row.split('\t').collect(); + assert_eq!(columns[0], entry.name, "rows are out of spec order: {row}"); + assert_eq!( + columns[1], + entry.extensions.join(","), + "wrong extensions for `{}`", + entry.name + ); + assert_eq!( + columns[2], + entry.dialects.join(","), + "wrong dialects for `{}`", + entry.name + ); + assert_eq!( + columns.get(3).copied(), + entry.notes.as_deref(), + "wrong notes for `{}`", + entry.name + ); + } + assert_eq!(rows.next(), None, "the listing has a row the spec does not"); +} + +/// Every string literal of one `match` block of the detector, which for the +/// two blocks read here is exactly the set of names that block answers to. +/// +/// The block is found by the line that opens it and ends at the first line +/// that closes a `let` binding, so a rewrite of `detect.rs` that moves either +/// one fails this loudly rather than passing on an empty set. +fn match_keys(header: &str) -> BTreeSet { + let opened = DETECT + .split_once(header) + .unwrap_or_else(|| panic!("detect.rs no longer contains `{header}`")) + .1; + let body = opened + .split_once("\n };") + .unwrap_or_else(|| panic!("the block opened by `{header}` is not closed as expected")) + .0; + let mut keys = BTreeSet::new(); + let mut rest = body; + while let Some((_, after)) = rest.split_once('"') { + let (key, tail) = after + .split_once('"') + .unwrap_or_else(|| panic!("an unterminated literal follows `{header}`")); + assert!( + !key.contains('\\'), + "`{key}` is escaped, which this reader cannot undo" + ); + keys.insert(key.to_owned()); + rest = tail; + } + assert!(!keys.is_empty(), "`{header}` matches on nothing"); + keys +} + +/// The published table is checked against the detector above; this checks the +/// detector against the published table, which nothing else does. An extension +/// the detector answers to is either in `spec/languages.toml` — and so in the +/// hooks, the documentation, and the listing — or named as one that is not. +#[test] +fn the_detector_knows_no_unrecorded_extension() { + let published: BTreeSet = table() + .languages + .iter() + .flat_map(|entry| entry.extensions.iter().cloned()) + .collect(); + let known = match_keys("let by_extension = match extension.as_str() {"); + let missing: BTreeSet<&str> = known + .iter() + .map(String::as_str) + .filter(|extension| !published.contains(*extension)) + .collect(); + assert_eq!( + missing, + UNPUBLISHED_EXTENSIONS.into_iter().collect::>(), + "the detector and spec/languages.toml disagree about which extensions exist" + ); + let unknown: Vec<&String> = published.difference(&known).collect(); + assert!( + unknown.is_empty(), + "spec/languages.toml publishes extensions the detector does not know: {unknown:?}" + ); +} + +/// The same, for the interpreter names a `#!` line is read for. Unlike the +/// extensions and the reserved names, this set is not read out of the +/// detector's source: `ocomment_core::shebang_interpreters` publishes it, so +/// what is compared here is the table the detector actually searches rather +/// than a reading of the file it is written in. +/// +/// `every_listed_shebang_detects_its_language` runs the detector over every +/// name the spec claims; this is the other direction, and it is the one that +/// catches an interpreter taught to the detector and never written down — +/// which would leave a piped script detected as a language `ocomment +/// languages` says nothing about. +#[test] +fn the_detector_knows_no_unrecorded_shebang() { + let published: BTreeSet = table() + .languages + .iter() + .flat_map(|entry| entry.shebangs.iter().cloned()) + .collect(); + let known: BTreeSet = ocomment_core::shebang_interpreters() + .map(ToOwned::to_owned) + .collect(); + assert_eq!( + known, published, + "the detector and spec/languages.toml disagree about which interpreters exist" + ); +} + +/// The same, for the whole file names that carry no extension. Every one the +/// detector answers to is published, so this difference is empty in both +/// directions. +#[test] +fn the_detector_knows_no_unrecorded_file_name() { + let published: BTreeSet = table() + .languages + .iter() + .flat_map(|entry| entry.reserved_names.iter()) + .map(|name| name.to_ascii_lowercase()) + .collect(); + let known = match_keys("let reserved = match lower.as_str() {"); + assert_eq!( + known, published, + "the detector and spec/languages.toml disagree about which file names are reserved" + ); +} + +/// The English word for a small number, so a count written out in prose can be +/// checked against the number it means. The list stops where the prose does: a +/// repository with more than thirty-nine languages needs another entry here, +/// which is the same edit as the sentence it guards. +fn number_word(value: usize) -> String { + const UNITS: [&str; 20] = [ + "zero", + "one", + "two", + "three", + "four", + "five", + "six", + "seven", + "eight", + "nine", + "ten", + "eleven", + "twelve", + "thirteen", + "fourteen", + "fifteen", + "sixteen", + "seventeen", + "eighteen", + "nineteen", + ]; + const TENS: [&str; 4] = ["twenty", "thirty", "forty", "fifty"]; + if value < UNITS.len() { + return UNITS[value].to_owned(); + } + let tens = TENS + .get(value / 10 - 2) + .unwrap_or_else(|| panic!("no English word for {value}")); + match value % 10 { + 0 => (*tens).to_owned(), + unit => format!("{tens}-{}", UNITS[unit]), + } +} + +/// One text with its runs of whitespace collapsed, so a claim can be searched +/// for without the line wrapping of the file it lives in being part of the +/// assertion. +fn unwrapped(text: &str) -> String { + text.split_whitespace().collect::>().join(" ") +} + +/// The VS Code language identifiers the extension attaches to, taken from the +/// default of its `ocomment.languages` setting. +fn vscode_language_identifiers() -> Vec { + let manifest: Value = serde_json::from_str(VSCODE_PACKAGE).expect("package.json parses"); + manifest["contributes"]["configuration"]["properties"]["ocomment.languages"]["default"] + .as_array() + .expect("`ocomment.languages` has an array default") + .iter() + .map(|value| { + value + .as_str() + .expect("a language identifier is a string") + .to_owned() + }) + .collect() +} + +/// The extension activates on exactly the identifiers it attaches the server +/// to, in the same order. The two lists sit in one file and are read by +/// different parts of VS Code, so an identifier added to one alone is an +/// extension that either never wakes up for a language or wakes up for one it +/// then ignores. +#[test] +fn the_vscode_activation_events_are_the_languages_it_attaches_to() { + let manifest: Value = serde_json::from_str(VSCODE_PACKAGE).expect("package.json parses"); + let activated: Vec = manifest["activationEvents"] + .as_array() + .expect("`activationEvents` is an array") + .iter() + .filter_map(|value| value.as_str()?.strip_prefix("onLanguage:")) + .map(ToOwned::to_owned) + .collect(); + assert_eq!( + activated, + vscode_language_identifiers(), + "editors/vscode/package.json activates on a different set of languages than it attaches to" + ); +} + +/// Every written-out count of languages or of editor language identifiers is +/// the count it claims to be. `Language::ALL` and the extension's own selector +/// are the two things being counted, so adding a language cannot leave a +/// sentence, a Marketplace description, or a changelog entry quietly wrong. +/// +/// The claims are searched for in the file with its line wrapping collapsed, +/// so re-flowing a paragraph is not a failure and changing what it says is. +#[test] +fn every_written_language_count_matches_what_it_counts() { + let languages = Language::ALL.len(); + let dialects = Dialect::ALL.len(); + let identifiers = vscode_language_identifiers().len(); + let claims = [ + ( + "docs/comparison.md", + COMPARISON, + format!("[{languages} languages and {dialects} dialects](languages.md)"), + ), + ( + "editors/vscode/package.json", + VSCODE_PACKAGE, + format!("comment checker and remover for {languages} languages."), + ), + ( + "editors/vscode/README.md", + VSCODE_README, + format!("the {identifiers} identifiers above"), + ), + ( + "docs/editors.md", + EDITORS, + format!( + "It attaches to {} language identifiers", + number_word(identifiers) + ), + ), + ( + "editors/vscode/CHANGELOG.md", + VSCODE_CHANGELOG, + format!( + "attaches it to the {} language identifiers OComment scans", + number_word(identifiers) + ), + ), + ( + "CHANGELOG.md", + CHANGELOG, + format!( + "attaches it to the {} language identifiers OComment scans", + number_word(identifiers) + ), + ), + ( + "CHANGELOG.md", + CHANGELOG, + format!("transformations for {languages} built-in languages"), + ), + ]; + for (name, text, claim) in claims { + assert!( + unwrapped(text).contains(&claim), + "{name} does not say `{claim}`; \ + {languages} language(s) and {identifiers} editor language identifier(s) are what \ + `Language::ALL` and editors/vscode/package.json hold" + ); + } +} diff --git a/spec/config.schema.json b/spec/config.schema.json index dae4bf1..7e64936 100644 --- a/spec/config.schema.json +++ b/spec/config.schema.json @@ -70,7 +70,7 @@ "$defs": { "strings": { "type": "array", "items": { "type": "string" }, "default": [] }, "policy": { "enum": ["safe", "legal", "all"], "default": "safe" }, - "language": { "enum": ["rust", "ocaml", "c", "cpp", "go", "java", "javascript", "typescript", "python", "shell", "html", "css", "jsonc", "sql", "kotlin"] }, + "language": { "enum": ["rust", "ocaml", "c", "cpp", "go", "java", "javascript", "typescript", "python", "shell", "html", "css", "jsonc", "sql", "kotlin", "toml", "lua", "yaml", "php", "ruby", "zig", "r", "dart", "swift", "csharp", "scala", "vue", "svelte", "markdown", "perl"] }, "kind": { "enum": ["line", "block", "doc-line", "doc-block", "directive", "license", "html-comment", "shebang", "encoding", "optimizer-hint", "version-comment"] }, "kinds": { "type": "array", "items": { "$ref": "#/$defs/kind" }, "default": [] }, "languageConfig": { @@ -97,7 +97,7 @@ } }, "dialect": { - "enum": ["standard", "jsx", "tsx", "objective-c", "objective-cpp", "gnu-c", "gnu-cpp", "cuda", "posix-sh", "bash53", "zsh", "postgresql", "mysql", "sqlite", "t-sql", "oracle"] + "enum": ["standard", "jsx", "tsx", "objective-c", "objective-cpp", "gnu-c", "gnu-cpp", "cuda", "posix-sh", "bash53", "zsh", "postgresql", "mysql", "sqlite", "t-sql", "oracle", "scss"] }, "nonEmptyStrings": { "type": "array", "minItems": 1, "items": { "type": "string", "minLength": 1 } diff --git a/spec/differential-protocol.md b/spec/differential-protocol.md index d9e47e7..032a716 100644 --- a/spec/differential-protocol.md +++ b/spec/differential-protocol.md @@ -4,12 +4,44 @@ Each JSONL request is one object with `id`, `operation` (`scan`, `transform`, `apply_edits`, `transform-spans`, `scan-profile`, or `transform-profile`), `language`, byte-preserving `source_base64`, and options. External-span requests add ordered `{start,end,kind}` entries; profile requests add a declarative -profile object. Each +profile object; `apply_edits` requests add ordered +`{span:{start,end},replacement_base64}` entries. Each response repeats `id` and contains either `ok` with the normalized result or `error`. Spans are half-open byte offsets; comments and edits are ordered by `(start,end)`; object keys are serialized in protocol order. Binary payloads use base64 so invalid UTF-8 is never normalized by a JSON implementation. +`id` is echoed back unchanged and is not interpreted. `tools/differential.py` +sends the fixture id, so a response names the case it belongs to. + The Rust conformance driver and `ocomment-ref` must compare comment spans, classification, dispositions, diagnostics, edits, safe/all output, and source map segments byte-for-byte. + +## Where the requests come from + +Every request is built from one case of the fixture corpus in +`spec/fixtures/v1/*.json`, which is the single source of truth for what the two +implementations are asked and what they are expected to answer. No fixture +bytes live in `tools/differential.py`, in the Rust test suite, or in the OCaml +reference; adding a hazard to a runner instead of to the corpus hides it from +the other runner. + +A case maps onto a request directly, under the same names: + +| Case field | Request | +| --- | --- | +| `id` | `id` | +| `language` | `language` | +| `dialect` | `options.dialect` | +| `operation` | `operation`, defaulting to `transform` | +| `options` | `options`, over the defaults `{"policy": "safe", "layout": "lines"}` | +| `source_utf8` or `source_base64` | `source_base64` | +| `spans`, `edits`, `profile` | the same key, unchanged | + +`note` and `expect` are not sent. `expect` is the result the corpus has +recorded for the case: `tools/differential.py` checks it against the agreed +response, so the corpus pins absolute behaviour and not only agreement, and +`rust/ocomment-core/tests/spec_fixtures.rs` checks the same blocks with no +OCaml toolchain in sight. `spec/fixtures/README.md` documents the schema and +how a block is recorded. diff --git a/spec/directives.toml b/spec/directives.toml index 34a9e65..f9b09f5 100644 --- a/spec/directives.toml +++ b/spec/directives.toml @@ -3,7 +3,16 @@ version = 1 protected = [ "shebang", "encoding", "go:", "+build", "triple-slash-reference", "sourceMappingURL", "sourceURL", "#__PURE__", "@__PURE__", - "lint-and-formatter", "type-checker", "optimizer-hint", "version-comment" + "lint-and-formatter", "type-checker", "optimizer-hint", "version-comment", + "syntax=", "hadolint", ":schema", "taplo:", "---@diagnostic", "luacheck:", + "selene:", "stylua:", "luacov:", "yaml-language-server:", "yamllint", + "renovate:", "checkov:skip", "trivy:ignore", "nosec", "kics-scan", "@schema", + "phpcs:", "@phpstan-ignore", "@psalm-suppress", "@codeCoverageIgnore", + "frozen_string_literal:", "warn_indent:", "shareable_constant_value:", + "rubocop:", "standard:", "typed:", "zig fmt:", "styler:", "nocov", "@dart", + "dart format", "ignore:", "ignore_for_file:", "swift-tools-version:", + "swiftlint:", "swiftformat:", "swift-format-ignore", " using" ] [policy.safe] diff --git a/spec/fixtures/README.md b/spec/fixtures/README.md index 57b6422..8a39192 100644 --- a/spec/fixtures/README.md +++ b/spec/fixtures/README.md @@ -1,11 +1,159 @@ # Shared fixtures -`v1/builtins.json` is consumed unchanged by the Rust/OCaml differential runner. -Every source is encoded to UTF-8 bytes only after JSON parsing; byte offsets are -then compared by the normalized JSONL protocol. The harness runs both `safe` -and `all` policies for every built-in language and adds binary, dialect, -malformed-input, external-span, and declarative-profile cases. - -Fixture changes are specification changes. Add the corresponding official -lexical-spec reference and expected behavior to the case note before changing a -source form. +`v1/*.json` is the fixture corpus, and it is the single source of truth for +what OComment does to a hazardous source. Nothing about a case lives anywhere +else: no fixture bytes are hardcoded in `tools/differential.py`, in the Rust +test suite, or in the OCaml reference. + +Two consumers read it, and both must pass: + +- `tools/differential.py` turns every case into one request of the + [differential protocol](../differential-protocol.md), feeds the corpus to the + Rust engine and to the OCaml reference, and requires the two responses to be + equal byte for byte. That says the pair agree. +- `rust/ocomment-core/tests/spec_fixtures.rs` runs every case against the Rust + engine alone, so the corpus still holds on a machine with no OCaml toolchain. + That says *what* they agree on. + +Both also check a case's `expect` block, and both refuse to run a corpus that +has shrunk. The floors live in `v1/floor.txt` and nowhere else, so the two +runners cannot drift apart: `cases` is the least number of cases the corpus may +hold, and `expectations` the least number of those that must carry a recorded +expectation. `differential.py` enforces `cases` and requires `expectations` to +be present but does not enforce it — it is also the runner that *records* a +missing block, and a floor it enforced would refuse to run on the way to +putting one back. The Rust test, which never records, enforces both. + +Two documents are in `v1/` today. The split is editorial, not structural — the +loaders concatenate every `*.json` in file-name order and require ids to be +unique across all of them. `floor.txt` sits beside them and is not a document: +both loaders read only `*.json`. + +- `builtins.json` — one small source per built-in language, under both the + `safe` and the `all` policy. +- `hazards.json` — the lexical hazards: raw strings, nested comments, + heredocs, regex-versus-division, translation-phase escapes, dialect + differences, malformed input, layout arithmetic, external spans, and + declarative profiles. + +## Case schema + +```json +{ + "version": 1, + "cases": [ + { + "id": "sql-mysql-dash-boundary", + "language": "sql", + "dialect": "mysql", + "operation": "transform", + "options": { "policy": "safe", "layout": "lines" }, + "source_utf8": "select 1--2; -- remove\n", + "note": "MySQL comment syntax requires `--` to be followed by whitespace...", + "expect": { + "valid": true, + "comments": [{ "start": 13, "end": 22, "kind": "line", "action": "remove" }], + "diagnostics": [], + "output_utf8": "select 1--2; \n" + } + } + ] +} +``` + +| Field | Required | Meaning | +| --- | --- | --- | +| `id` | yes | Unique across the whole corpus, and what a failure names. Meaningful, kebab-case, and stable: renaming one rewrites history for no gain. | +| `language` | yes | A built-in language name. `apply_edits` and the profile operations still need one; it is not read there. | +| `dialect` | no | The vendor rules to lex with. It sits beside `language` because the two together choose the grammar, while `options` holds the policy knobs. | +| `operation` | no | `transform` when absent. One of the protocol operations: `scan`, `transform`, `apply_edits`, `transform-spans`, `scan-profile`, `transform-profile`. | +| `options` | no | `ScanOptions` fields under their frozen serde names — `policy`, `force_invalid`, `force_protected`, `keep_kinds`, `remove_kinds`, `keep_regex`, `remove_regex` — plus `layout`. `policy` defaults to `safe` and `layout` to `lines`; everything else defaults as `ScanOptions::default` does. An unknown key is an error, not a no-op. | +| `source_utf8` / `source_base64` | exactly one | The source bytes. | +| `spans` | `transform-spans` | Ordered `{start, end, kind}` comment spans an external scanner is pretending to have found. | +| `edits` | `apply_edits` | Ordered `{span: {start, end}, replacement_base64}` edits. | +| `profile` | `*-profile` | A declarative profile object. | +| `note` | yes | The official lexical specification the case comes from, and what is supposed to happen. | +| `expect` | no, but see below | The recorded result. | + +### Source encoding + +Use `source_utf8` when the bytes are valid UTF-8 and every character is +ordinary text — newline, carriage return and tab are fine, and so are CJK +characters and emoji. Use `source_base64` otherwise: bytes that are not UTF-8 +at all, C0 controls, U+2028 and U+2029, and format, surrogate, private-use or +unassigned characters. Those are exactly the characters a JSON tool, an editor, +or a terminal is liable to normalise, and a fixture whose bytes drift is worse +than no fixture. The same rule governs `expect.output_utf8` against +`expect.output_base64`. + +A case whose source is *generated* — a sweep over a range of scalar values, a +long repeated tag — is recorded as fixed base64 rather than as the generator +that produced it. The point of the corpus is that both implementations see the +same bytes forever, and a generator is one refactor away from producing +different ones. + +### `expect` + +Every field is optional and every field present is checked, so a partial block +is a partial assertion rather than a weaker one. + +| Field | Checked against | +| --- | --- | +| `valid` | `ScanReport::valid`. | +| `comments` | Every comment, in order, as `{start, end, kind, action}`. `action` is `keep` or `remove`; the human-readable keep reason is deliberately not pinned here. | +| `diagnostics` | Every diagnostic, in order, as `{code, start, end}`. An empty array asserts that there are none. | +| `output_utf8` / `output_base64` | The transformed bytes. | + +An operation that reports nothing (`apply_edits`) takes only the output fields; +an operation that writes nothing (`scan`, `scan-profile`) takes only the report +fields. + +Whether or not a case carries an `expect` block, the Rust test holds it to the +engine's structural promises: the run must not panic, and a transformation's +edits must be sorted, non-overlapping, inside the source, and must reproduce +the output when applied in one pass. + +## Adding a case + +1. Add the case to `hazards.json` — or to `builtins.json` if it is the plain + one-source-per-language coverage. Never to a Python or Rust file: a hazard + hardcoded in a runner is invisible to the other runner. +2. Write the `note` first. Name the clause of the official lexical + specification the case exercises and say what is supposed to happen. A case + whose expected behaviour cannot be sourced is a bug report, not a fixture. +3. Leave `expect` out for the moment and run + `opam exec -- ./tools/differential.sh`. A mismatch between the two + implementations is a real finding; settle it before recording anything. +4. Record `expect` with `python3 tools/differential.py --record` once that run + is green, then run both consumers again. + +Raising `cases` and `expectations` in `v1/floor.txt` is optional when adding a +case and mandatory when the corpus is reorganised: the floors exist so that +neither a case nor its recorded expectation can quietly disappear. + +## Recording an `expect` block + +An `expect` block is *recorded*, never hand-written: + +```sh +cargo build --manifest-path rust/Cargo.toml -p ocomment-core --example ref_driver --locked +opam exec -- dune build --root ocaml bin/main.exe +python3 tools/differential.py --record +``` + +`--record` runs the whole corpus through both implementations first and records +nothing unless every case agreed, so a recorded block is a record of the +specification and not of whichever implementation was consulted. It fills in +only the cases that have no `expect` block, and rewrites a document only when +something changed; on an already-recorded corpus it is a no-op. + +An output longer than about a kilobyte is left unrecorded rather than inlined. +`column-unicode-width-scalar-sample` is the one such case, and the comment span +and validity still pin it while the differential comparison covers the bytes. + +Re-recording is deliberate. `--record` will not overwrite a block that is +already there, so an intentional behaviour change means deleting that case's +`expect` block and re-recording it in the same commit that argues for the +change. Changing a recorded value is a specification change, and needs what any +other one needs: the clause that now says otherwise, in the case `note` and in +`CHANGELOG.md`. diff --git a/spec/fixtures/v1/builtins.json b/spec/fixtures/v1/builtins.json index 7003cb7..4301b22 100644 --- a/spec/fixtures/v1/builtins.json +++ b/spec/fixtures/v1/builtins.json @@ -2,79 +2,1684 @@ "version": 1, "cases": [ { + "id": "rust-builtin-safe", "language": "rust", + "operation": "transform", + "options": { + "policy": "safe", + "layout": "lines" + }, "source_utf8": "r#\"// string\"# /* block */\r\n// line\r\n", - "note": "raw string, block and line comments, CRLF" + "note": "Rust Reference, Tokens and Comments: a raw string hides the comment delimiters inside it, a block comment and a line comment are found beside it, and the CRLF endings survive the transformation.", + "expect": { + "valid": true, + "comments": [ + { + "start": 15, + "end": 26, + "kind": "block", + "action": "remove" + }, + { + "start": 28, + "end": 35, + "kind": "line", + "action": "remove" + } + ], + "diagnostics": [], + "output_utf8": "r#\"// string\"# \r\n\r\n" + } }, { + "id": "rust-builtin-all", + "language": "rust", + "operation": "transform", + "options": { + "policy": "all", + "layout": "lines" + }, + "source_utf8": "r#\"// string\"# /* block */\r\n// line\r\n", + "note": "Rust Reference, Tokens and Comments: a raw string hides the comment delimiters inside it, a block comment and a line comment are found beside it, and the CRLF endings survive the transformation.", + "expect": { + "valid": true, + "comments": [ + { + "start": 15, + "end": 26, + "kind": "block", + "action": "remove" + }, + { + "start": 28, + "end": 35, + "kind": "line", + "action": "remove" + } + ], + "diagnostics": [], + "output_utf8": "r#\"// string\"# \r\n\r\n" + } + }, + { + "id": "ocaml-builtin-safe", + "language": "ocaml", + "operation": "transform", + "options": { + "policy": "safe", + "layout": "lines" + }, + "source_utf8": "\"(* string *)\" (* outer (* nested *) end *)\n", + "note": "OCaml manual, Lexical conventions: a string literal hides `(*`, and comments nest, so the outer comment closes once.", + "expect": { + "valid": true, + "comments": [ + { + "start": 15, + "end": 43, + "kind": "block", + "action": "remove" + } + ], + "diagnostics": [], + "output_utf8": "\"(* string *)\" \n" + } + }, + { + "id": "ocaml-builtin-all", "language": "ocaml", + "operation": "transform", + "options": { + "policy": "all", + "layout": "lines" + }, "source_utf8": "\"(* string *)\" (* outer (* nested *) end *)\n", - "note": "string opacity and nested comments" + "note": "OCaml manual, Lexical conventions: a string literal hides `(*`, and comments nest, so the outer comment closes once.", + "expect": { + "valid": true, + "comments": [ + { + "start": 15, + "end": 43, + "kind": "block", + "action": "remove" + } + ], + "diagnostics": [], + "output_utf8": "\"(* string *)\" \n" + } }, { + "id": "c-builtin-safe", "language": "c", + "operation": "transform", + "options": { + "policy": "safe", + "layout": "lines" + }, "source_utf8": "char *s = \"// string\"; /* block */\n// line\n", - "note": "quoted delimiter opacity" + "note": "ISO C17 6.4.5 string literals and 6.4.9 comments: the `//` inside the literal is opaque, and both comment forms beside it are found.", + "expect": { + "valid": true, + "comments": [ + { + "start": 23, + "end": 34, + "kind": "block", + "action": "remove" + }, + { + "start": 35, + "end": 42, + "kind": "line", + "action": "remove" + } + ], + "diagnostics": [], + "output_utf8": "char *s = \"// string\"; \n\n" + } + }, + { + "id": "c-builtin-all", + "language": "c", + "operation": "transform", + "options": { + "policy": "all", + "layout": "lines" + }, + "source_utf8": "char *s = \"// string\"; /* block */\n// line\n", + "note": "ISO C17 6.4.5 string literals and 6.4.9 comments: the `//` inside the literal is opaque, and both comment forms beside it are found.", + "expect": { + "valid": true, + "comments": [ + { + "start": 23, + "end": 34, + "kind": "block", + "action": "remove" + }, + { + "start": 35, + "end": 42, + "kind": "line", + "action": "remove" + } + ], + "diagnostics": [], + "output_utf8": "char *s = \"// string\"; \n\n" + } + }, + { + "id": "cpp-builtin-safe", + "language": "cpp", + "operation": "transform", + "options": { + "policy": "safe", + "layout": "lines" + }, + "source_utf8": "auto s = \"/* string */\"; // line\n", + "note": "ISO C++ [lex.string] and [lex.comment]: the `/* */` inside the literal is opaque, and the line comment beside it is found.", + "expect": { + "valid": true, + "comments": [ + { + "start": 25, + "end": 32, + "kind": "line", + "action": "remove" + } + ], + "diagnostics": [], + "output_utf8": "auto s = \"/* string */\"; \n" + } }, { + "id": "cpp-builtin-all", "language": "cpp", + "operation": "transform", + "options": { + "policy": "all", + "layout": "lines" + }, "source_utf8": "auto s = \"/* string */\"; // line\n", - "note": "quoted delimiter opacity" + "note": "ISO C++ [lex.string] and [lex.comment]: the `/* */` inside the literal is opaque, and the line comment beside it is found.", + "expect": { + "valid": true, + "comments": [ + { + "start": 25, + "end": 32, + "kind": "line", + "action": "remove" + } + ], + "diagnostics": [], + "output_utf8": "auto s = \"/* string */\"; \n" + } }, { + "id": "go-builtin-safe", "language": "go", + "operation": "transform", + "options": { + "policy": "safe", + "layout": "lines" + }, "source_utf8": "var s = `// raw`; /* block */\n", - "note": "raw string opacity" + "note": "Go specification, raw string literals: a back-quoted string takes no escapes and hides `//`; the block comment beside it is found.", + "expect": { + "valid": true, + "comments": [ + { + "start": 18, + "end": 29, + "kind": "block", + "action": "remove" + } + ], + "diagnostics": [], + "output_utf8": "var s = `// raw`; \n" + } + }, + { + "id": "go-builtin-all", + "language": "go", + "operation": "transform", + "options": { + "policy": "all", + "layout": "lines" + }, + "source_utf8": "var s = `// raw`; /* block */\n", + "note": "Go specification, raw string literals: a back-quoted string takes no escapes and hides `//`; the block comment beside it is found.", + "expect": { + "valid": true, + "comments": [ + { + "start": 18, + "end": 29, + "kind": "block", + "action": "remove" + } + ], + "diagnostics": [], + "output_utf8": "var s = `// raw`; \n" + } + }, + { + "id": "java-builtin-safe", + "language": "java", + "operation": "transform", + "options": { + "policy": "safe", + "layout": "lines" + }, + "source_utf8": "String s = \"// raw\"; // line\n", + "note": "JLS 3.10.5 string literals and 3.7 comments: the `//` inside the literal is opaque, and the line comment beside it is found.", + "expect": { + "valid": true, + "comments": [ + { + "start": 21, + "end": 28, + "kind": "line", + "action": "remove" + } + ], + "diagnostics": [], + "output_utf8": "String s = \"// raw\"; \n" + } }, { + "id": "java-builtin-all", "language": "java", + "operation": "transform", + "options": { + "policy": "all", + "layout": "lines" + }, "source_utf8": "String s = \"// raw\"; // line\n", - "note": "string opacity" + "note": "JLS 3.10.5 string literals and 3.7 comments: the `//` inside the literal is opaque, and the line comment beside it is found.", + "expect": { + "valid": true, + "comments": [ + { + "start": 21, + "end": 28, + "kind": "line", + "action": "remove" + } + ], + "diagnostics": [], + "output_utf8": "String s = \"// raw\"; \n" + } + }, + { + "id": "javascript-builtin-safe", + "language": "javascript", + "operation": "transform", + "options": { + "policy": "safe", + "layout": "lines" + }, + "source_utf8": "const s = \"// raw\"; /* block */\n", + "note": "ECMA-262 string literals and comments: the `//` inside the literal is opaque, and the block comment beside it is found.", + "expect": { + "valid": true, + "comments": [ + { + "start": 20, + "end": 31, + "kind": "block", + "action": "remove" + } + ], + "diagnostics": [], + "output_utf8": "const s = \"// raw\"; \n" + } }, { + "id": "javascript-builtin-all", "language": "javascript", + "operation": "transform", + "options": { + "policy": "all", + "layout": "lines" + }, "source_utf8": "const s = \"// raw\"; /* block */\n", - "note": "string opacity" + "note": "ECMA-262 string literals and comments: the `//` inside the literal is opaque, and the block comment beside it is found.", + "expect": { + "valid": true, + "comments": [ + { + "start": 20, + "end": 31, + "kind": "block", + "action": "remove" + } + ], + "diagnostics": [], + "output_utf8": "const s = \"// raw\"; \n" + } }, { + "id": "typescript-builtin-safe", "language": "typescript", + "operation": "transform", + "options": { + "policy": "safe", + "layout": "lines" + }, "source_utf8": "const s: string = \"// raw\"; // line\n", - "note": "typed source and string opacity" + "note": "TypeScript inherits the ECMA-262 lexical grammar: a type annotation changes nothing, the `//` inside the literal is opaque, and the line comment is found.", + "expect": { + "valid": true, + "comments": [ + { + "start": 28, + "end": 35, + "kind": "line", + "action": "remove" + } + ], + "diagnostics": [], + "output_utf8": "const s: string = \"// raw\"; \n" + } }, { + "id": "typescript-builtin-all", + "language": "typescript", + "operation": "transform", + "options": { + "policy": "all", + "layout": "lines" + }, + "source_utf8": "const s: string = \"// raw\"; // line\n", + "note": "TypeScript inherits the ECMA-262 lexical grammar: a type annotation changes nothing, the `//` inside the literal is opaque, and the line comment is found.", + "expect": { + "valid": true, + "comments": [ + { + "start": 28, + "end": 35, + "kind": "line", + "action": "remove" + } + ], + "diagnostics": [], + "output_utf8": "const s: string = \"// raw\"; \n" + } + }, + { + "id": "python-builtin-safe", + "language": "python", + "operation": "transform", + "options": { + "policy": "safe", + "layout": "lines" + }, + "source_utf8": "s = \"# raw\" # line\n", + "note": "Python Reference 2.1.3 comments: a `#` inside a string literal is not a comment, and the trailing one is.", + "expect": { + "valid": true, + "comments": [ + { + "start": 13, + "end": 19, + "kind": "line", + "action": "remove" + } + ], + "diagnostics": [], + "output_utf8": "s = \"# raw\" \n" + } + }, + { + "id": "python-builtin-all", "language": "python", + "operation": "transform", + "options": { + "policy": "all", + "layout": "lines" + }, "source_utf8": "s = \"# raw\" # line\n", - "note": "string opacity" + "note": "Python Reference 2.1.3 comments: a `#` inside a string literal is not a comment, and the trailing one is.", + "expect": { + "valid": true, + "comments": [ + { + "start": 13, + "end": 19, + "kind": "line", + "action": "remove" + } + ], + "diagnostics": [], + "output_utf8": "s = \"# raw\" \n" + } }, { + "id": "shell-builtin-safe", "language": "shell", + "operation": "transform", + "options": { + "policy": "safe", + "layout": "lines" + }, "source_utf8": "s='# raw' # line\n", - "note": "single quote opacity and token boundary" + "note": "POSIX 2.2.2 single quotes and 2.3 token recognition: `#` inside single quotes is text, and the one starting a word is a comment.", + "expect": { + "valid": true, + "comments": [ + { + "start": 10, + "end": 16, + "kind": "line", + "action": "remove" + } + ], + "diagnostics": [], + "output_utf8": "s='# raw' \n" + } + }, + { + "id": "shell-builtin-all", + "language": "shell", + "operation": "transform", + "options": { + "policy": "all", + "layout": "lines" + }, + "source_utf8": "s='# raw' # line\n", + "note": "POSIX 2.2.2 single quotes and 2.3 token recognition: `#` inside single quotes is text, and the one starting a word is a comment.", + "expect": { + "valid": true, + "comments": [ + { + "start": 10, + "end": 16, + "kind": "line", + "action": "remove" + } + ], + "diagnostics": [], + "output_utf8": "s='# raw' \n" + } + }, + { + "id": "html-builtin-safe", + "language": "html", + "operation": "transform", + "options": { + "policy": "safe", + "layout": "lines" + }, + "source_utf8": "", + "note": "HTML Standard 13.1.6 comments and 13.2.5 script data: a `` comment is exposed to scripts through the DOM, and a `" + } }, { + "id": "html-builtin-all", "language": "html", + "operation": "transform", + "options": { + "policy": "all", + "layout": "lines" + }, "source_utf8": "", - "note": "DOM-observable HTML comment and recursive script" + "note": "HTML Standard 13.1.6 comments and 13.2.5 script data: a `` comment is exposed to scripts through the DOM, and a `" + } }, { + "id": "css-builtin-safe", "language": "css", + "operation": "transform", + "options": { + "policy": "safe", + "layout": "lines" + }, "source_utf8": "a{content:\"/* raw */\";/* block */}\n", - "note": "string opacity" + "note": "CSS Syntax Level 3, 4.3.1 consume a token: a `/* */` inside a string is opaque, and the one beside it is a comment.", + "expect": { + "valid": true, + "comments": [ + { + "start": 22, + "end": 33, + "kind": "block", + "action": "remove" + } + ], + "diagnostics": [], + "output_utf8": "a{content:\"/* raw */\"; }\n" + } + }, + { + "id": "css-builtin-all", + "language": "css", + "operation": "transform", + "options": { + "policy": "all", + "layout": "lines" + }, + "source_utf8": "a{content:\"/* raw */\";/* block */}\n", + "note": "CSS Syntax Level 3, 4.3.1 consume a token: a `/* */` inside a string is opaque, and the one beside it is a comment.", + "expect": { + "valid": true, + "comments": [ + { + "start": 22, + "end": 33, + "kind": "block", + "action": "remove" + } + ], + "diagnostics": [], + "output_utf8": "a{content:\"/* raw */\"; }\n" + } + }, + { + "id": "jsonc-builtin-safe", + "language": "jsonc", + "operation": "transform", + "options": { + "policy": "safe", + "layout": "lines" + }, + "source_utf8": "{\"x\":\"// raw\" // line\n}\n", + "note": "JSON with comments: a JSON string hides `//`, and the comment after the value is found.", + "expect": { + "valid": true, + "comments": [ + { + "start": 14, + "end": 21, + "kind": "line", + "action": "remove" + } + ], + "diagnostics": [], + "output_utf8": "{\"x\":\"// raw\" \n}\n" + } }, { + "id": "jsonc-builtin-all", "language": "jsonc", + "operation": "transform", + "options": { + "policy": "all", + "layout": "lines" + }, "source_utf8": "{\"x\":\"// raw\" // line\n}\n", - "note": "JSON string opacity" + "note": "JSON with comments: a JSON string hides `//`, and the comment after the value is found.", + "expect": { + "valid": true, + "comments": [ + { + "start": 14, + "end": 21, + "kind": "line", + "action": "remove" + } + ], + "diagnostics": [], + "output_utf8": "{\"x\":\"// raw\" \n}\n" + } + }, + { + "id": "sql-builtin-safe", + "language": "sql", + "operation": "transform", + "options": { + "policy": "safe", + "layout": "lines" + }, + "source_utf8": "select '-- raw'; -- line\n/* block */\n", + "note": "SQL standard string literals and both comment forms: `--` inside a literal is opaque, and the `--` and `/* */` beside it are comments.", + "expect": { + "valid": true, + "comments": [ + { + "start": 17, + "end": 24, + "kind": "line", + "action": "remove" + }, + { + "start": 25, + "end": 36, + "kind": "block", + "action": "remove" + } + ], + "diagnostics": [], + "output_utf8": "select '-- raw'; \n\n" + } }, { + "id": "sql-builtin-all", "language": "sql", + "operation": "transform", + "options": { + "policy": "all", + "layout": "lines" + }, "source_utf8": "select '-- raw'; -- line\n/* block */\n", - "note": "SQL string opacity and both comment forms" + "note": "SQL standard string literals and both comment forms: `--` inside a literal is opaque, and the `--` and `/* */` beside it are comments.", + "expect": { + "valid": true, + "comments": [ + { + "start": 17, + "end": 24, + "kind": "line", + "action": "remove" + }, + { + "start": 25, + "end": 36, + "kind": "block", + "action": "remove" + } + ], + "diagnostics": [], + "output_utf8": "select '-- raw'; \n\n" + } }, { + "id": "kotlin-builtin-safe", "language": "kotlin", + "operation": "transform", + "options": { + "policy": "safe", + "layout": "lines" + }, "source_utf8": "val s = \"// raw\" // line\n/* outer /* nested */ end */", - "note": "string opacity and nested comments" + "note": "Kotlin specification, string literals and comments: the `//` inside the literal is opaque, the line comment is found, and block comments nest.", + "expect": { + "valid": true, + "comments": [ + { + "start": 17, + "end": 24, + "kind": "line", + "action": "remove" + }, + { + "start": 25, + "end": 53, + "kind": "block", + "action": "remove" + } + ], + "diagnostics": [], + "output_utf8": "val s = \"// raw\" \n" + } + }, + { + "id": "kotlin-builtin-all", + "language": "kotlin", + "operation": "transform", + "options": { + "policy": "all", + "layout": "lines" + }, + "source_utf8": "val s = \"// raw\" // line\n/* outer /* nested */ end */", + "note": "Kotlin specification, string literals and comments: the `//` inside the literal is opaque, the line comment is found, and block comments nest.", + "expect": { + "valid": true, + "comments": [ + { + "start": 17, + "end": 24, + "kind": "line", + "action": "remove" + }, + { + "start": 25, + "end": 53, + "kind": "block", + "action": "remove" + } + ], + "diagnostics": [], + "output_utf8": "val s = \"// raw\" \n" + } + }, + { + "id": "toml-builtin-safe", + "language": "toml", + "operation": "transform", + "options": { + "policy": "safe", + "layout": "lines" + }, + "source_utf8": "#:schema https://example.test/pyproject.json\nkey = \"# opaque\" # remove\n", + "note": "TOML v1.0.0, Comment and String: a `#` inside a basic string is one of its bytes, and only the one outside opens a comment. Taplo's `#:schema` line is a directive `safe` keeps and `all` removes.", + "expect": { + "valid": true, + "comments": [ + { + "start": 0, + "end": 44, + "kind": "directive", + "action": "keep" + }, + { + "start": 62, + "end": 70, + "kind": "line", + "action": "remove" + } + ], + "diagnostics": [], + "output_utf8": "#:schema https://example.test/pyproject.json\nkey = \"# opaque\" \n" + } + }, + { + "id": "toml-builtin-all", + "language": "toml", + "operation": "transform", + "options": { + "policy": "all", + "layout": "lines" + }, + "source_utf8": "#:schema https://example.test/pyproject.json\nkey = \"# opaque\" # remove\n", + "note": "TOML v1.0.0, Comment and String: a `#` inside a basic string is one of its bytes, and only the one outside opens a comment. Taplo's `#:schema` line is a directive `safe` keeps and `all` removes.", + "expect": { + "valid": true, + "comments": [ + { + "start": 0, + "end": 44, + "kind": "directive", + "action": "remove" + }, + { + "start": 62, + "end": 70, + "kind": "line", + "action": "remove" + } + ], + "diagnostics": [], + "output_utf8": "\nkey = \"# opaque\" \n" + } + }, + { + "id": "lua-builtin-safe", + "language": "lua", + "operation": "transform", + "options": { + "policy": "safe", + "layout": "lines" + }, + "source_utf8": "---@diagnostic disable-next-line: undefined-global\nprint([[-- opaque]]) -- remove\n", + "note": "Lua 5.4 reference manual, 3.1: a long string carries the comment opener inside it as content, and only the `--` outside one opens a comment. The language server's `---@diagnostic` line is a directive `safe` keeps and `all` removes.", + "expect": { + "valid": true, + "comments": [ + { + "start": 0, + "end": 50, + "kind": "directive", + "action": "keep" + }, + { + "start": 72, + "end": 81, + "kind": "line", + "action": "remove" + } + ], + "diagnostics": [], + "output_utf8": "---@diagnostic disable-next-line: undefined-global\nprint([[-- opaque]]) \n" + } + }, + { + "id": "lua-builtin-all", + "language": "lua", + "operation": "transform", + "options": { + "policy": "all", + "layout": "lines" + }, + "source_utf8": "---@diagnostic disable-next-line: undefined-global\nprint([[-- opaque]]) -- remove\n", + "note": "Lua 5.4 reference manual, 3.1: a long string carries the comment opener inside it as content, and only the `--` outside one opens a comment. The language server's `---@diagnostic` line is a directive `safe` keeps and `all` removes.", + "expect": { + "valid": true, + "comments": [ + { + "start": 0, + "end": 50, + "kind": "directive", + "action": "remove" + }, + { + "start": 72, + "end": 81, + "kind": "line", + "action": "remove" + } + ], + "diagnostics": [], + "output_utf8": "\nprint([[-- opaque]]) \n" + } + }, + { + "id": "yaml-builtin-safe", + "language": "yaml", + "operation": "transform", + "options": { + "policy": "safe", + "layout": "lines" + }, + "source_utf8": "# yamllint disable-line rule:line-length\nkey: \"# opaque\" # remove\n", + "note": "YAML 1.2.2, 6.6 (Comment) and 7.3.1 (Double-Quoted Style): a `#` between the quotes of a double-quoted scalar is one of its bytes, and a comment needs white space in front of it, so only the `#` outside the quotes opens one. The `# yamllint` line is a directive `safe` keeps and `all` removes.", + "expect": { + "valid": true, + "comments": [ + { + "start": 0, + "end": 40, + "kind": "directive", + "action": "keep" + }, + { + "start": 57, + "end": 65, + "kind": "line", + "action": "remove" + } + ], + "diagnostics": [], + "output_utf8": "# yamllint disable-line rule:line-length\nkey: \"# opaque\" \n" + } + }, + { + "id": "yaml-builtin-all", + "language": "yaml", + "operation": "transform", + "options": { + "policy": "all", + "layout": "lines" + }, + "source_utf8": "# yamllint disable-line rule:line-length\nkey: \"# opaque\" # remove\n", + "note": "YAML 1.2.2, 6.6 (Comment) and 7.3.1 (Double-Quoted Style): a `#` between the quotes of a double-quoted scalar is one of its bytes, and a comment needs white space in front of it, so only the `#` outside the quotes opens one. The `# yamllint` line is a directive `safe` keeps and `all` removes.", + "expect": { + "valid": true, + "comments": [ + { + "start": 0, + "end": 40, + "kind": "directive", + "action": "remove" + }, + { + "start": 57, + "end": 65, + "kind": "line", + "action": "remove" + } + ], + "diagnostics": [], + "output_utf8": "\nkey: \"# opaque\" \n" + } + }, + { + "id": "php-builtin-safe", + "language": "php", + "operation": "transform", + "options": { + "policy": "safe", + "layout": "lines" + }, + "source_utf8": "\r\n

# html

\r\n", + "note": "PHP manual, Basic syntax and Comments: the inline HTML behind the closing tag is output verbatim and holds no comment, a `//` inside a single-quoted string is one of its bytes, and `phpcs:disable` is a directive `safe` keeps and `all` removes. The CRLF endings survive the transformation.", + "expect": { + "valid": true, + "comments": [ + { + "start": 6, + "end": 22, + "kind": "directive", + "action": "keep" + }, + { + "start": 42, + "end": 51, + "kind": "line", + "action": "remove" + } + ], + "diagnostics": [], + "output_utf8": "\r\n

# html

\r\n" + } + }, + { + "id": "php-builtin-all", + "language": "php", + "operation": "transform", + "options": { + "policy": "all", + "layout": "lines" + }, + "source_utf8": "\r\n

# html

\r\n", + "note": "PHP manual, Basic syntax and Comments: the inline HTML behind the closing tag is output verbatim and holds no comment, a `//` inside a single-quoted string is one of its bytes, and `phpcs:disable` is a directive `safe` keeps and `all` removes. The CRLF endings survive the transformation.", + "expect": { + "valid": true, + "comments": [ + { + "start": 6, + "end": 22, + "kind": "directive", + "action": "remove" + }, + { + "start": 42, + "end": 51, + "kind": "line", + "action": "remove" + } + ], + "diagnostics": [], + "output_utf8": "\r\n

# html

\r\n" + } + }, + { + "id": "ruby-builtin-safe", + "language": "ruby", + "operation": "transform", + "options": { + "policy": "safe", + "layout": "lines" + }, + "source_utf8": "#!/usr/bin/env ruby\r\n# frozen_string_literal: true\r\nx = '# opaque' # remove\r\n=begin\r\ndoc\r\n=end\r\n", + "note": "Ruby 3.3 documentation, doc/syntax/comments.rdoc, literals.rdoc, and \"Magic comments\": the `#!` line is the interpreter preamble, `# frozen_string_literal:` is a magic comment the interpreter reads, a `#` inside a single-quoted string is one of its bytes, and `=begin`/`=end` at column zero delimit an embedded document that `safe` and `all` both remove. The CRLF endings survive the transformation.", + "expect": { + "valid": true, + "comments": [ + { + "start": 0, + "end": 19, + "kind": "shebang", + "action": "keep" + }, + { + "start": 21, + "end": 50, + "kind": "directive", + "action": "keep" + }, + { + "start": 67, + "end": 75, + "kind": "line", + "action": "remove" + }, + { + "start": 77, + "end": 94, + "kind": "block", + "action": "remove" + } + ], + "diagnostics": [], + "output_utf8": "#!/usr/bin/env ruby\r\n# frozen_string_literal: true\r\nx = '# opaque' \r\n\r\n\r\n\r\n" + } + }, + { + "id": "ruby-builtin-all", + "language": "ruby", + "operation": "transform", + "options": { + "policy": "all", + "layout": "lines" + }, + "source_utf8": "#!/usr/bin/env ruby\r\n# frozen_string_literal: true\r\nx = '# opaque' # remove\r\n=begin\r\ndoc\r\n=end\r\n", + "note": "Ruby 3.3 documentation, doc/syntax/comments.rdoc, literals.rdoc, and \"Magic comments\": the `#!` line is the interpreter preamble, `# frozen_string_literal:` is a magic comment the interpreter reads, a `#` inside a single-quoted string is one of its bytes, and `=begin`/`=end` at column zero delimit an embedded document that `safe` and `all` both remove. The CRLF endings survive the transformation.", + "expect": { + "valid": true, + "comments": [ + { + "start": 0, + "end": 19, + "kind": "shebang", + "action": "keep" + }, + { + "start": 21, + "end": 50, + "kind": "directive", + "action": "remove" + }, + { + "start": 67, + "end": 75, + "kind": "line", + "action": "remove" + }, + { + "start": 77, + "end": 94, + "kind": "block", + "action": "remove" + } + ], + "diagnostics": [], + "output_utf8": "#!/usr/bin/env ruby\r\n\r\nx = '# opaque' \r\n\r\n\r\n\r\n" + } + }, + { + "id": "zig-builtin-safe", + "language": "zig", + "operation": "transform", + "options": { + "policy": "safe", + "layout": "lines" + }, + "source_utf8": "// zig fmt: off\r\nconst s = \"// string\";\r\n/// doc\r\n// line\r\n", + "note": "Zig Language Reference, Comments and String Literals: a string literal hides the `//` inside it, `///` is a documentation comment and `//` an ordinary one, `// zig fmt: off` is the one instruction the formatter reads out of a comment, and the CRLF endings survive the transformation. Ground truth, `std.zig.Tokenizer` 0.16.0: `string_literal \"\\\"// string\\\"\"` and `doc_comment \"/// doc\"`, with no token at all for the two ordinary comment lines.", + "expect": { + "valid": true, + "comments": [ + { + "start": 0, + "end": 15, + "kind": "directive", + "action": "keep" + }, + { + "start": 41, + "end": 48, + "kind": "doc-line", + "action": "remove" + }, + { + "start": 50, + "end": 57, + "kind": "line", + "action": "remove" + } + ], + "diagnostics": [], + "output_utf8": "// zig fmt: off\r\nconst s = \"// string\";\r\n\r\n\r\n" + } + }, + { + "id": "zig-builtin-all", + "language": "zig", + "operation": "transform", + "options": { + "policy": "all", + "layout": "lines" + }, + "source_utf8": "// zig fmt: off\r\nconst s = \"// string\";\r\n/// doc\r\n// line\r\n", + "note": "Zig Language Reference, Comments and String Literals: a string literal hides the `//` inside it, `///` is a documentation comment and `//` an ordinary one, `// zig fmt: off` is the one instruction the formatter reads out of a comment, and the CRLF endings survive the transformation. Ground truth, `std.zig.Tokenizer` 0.16.0: `string_literal \"\\\"// string\\\"\"` and `doc_comment \"/// doc\"`, with no token at all for the two ordinary comment lines.", + "expect": { + "valid": true, + "comments": [ + { + "start": 0, + "end": 15, + "kind": "directive", + "action": "remove" + }, + { + "start": 41, + "end": 48, + "kind": "doc-line", + "action": "remove" + }, + { + "start": 50, + "end": 57, + "kind": "line", + "action": "remove" + } + ], + "diagnostics": [], + "output_utf8": "\r\nconst s = \"// string\";\r\n\r\n\r\n" + } + }, + { + "id": "r-builtin-safe", + "language": "r", + "operation": "transform", + "options": { + "policy": "safe", + "layout": "lines" + }, + "source_utf8": "# styler: off\r\nx <- \"# string\"\r\n#' doc\r\n# line\r\n", + "note": "R Language Definition, 10.2 Comments and `?Quotes`: a string hides the `#` inside it, `#'` is roxygen2's documentation marker and `#` an ordinary comment, `# styler: off` is the instruction the formatter reads out of a comment, and the CRLF endings survive the transformation. Ground truth, R 4.3.3 `utils::getParseData(parse(file =, keep.source = TRUE))`: `COMMENT` at [0,13), [32,38) and [40,46), and `STR_CONST \"\\\"# string\\\"\"` at [20,30).", + "expect": { + "valid": true, + "comments": [ + { + "start": 0, + "end": 13, + "kind": "directive", + "action": "keep" + }, + { + "start": 32, + "end": 38, + "kind": "doc-line", + "action": "remove" + }, + { + "start": 40, + "end": 46, + "kind": "line", + "action": "remove" + } + ], + "diagnostics": [], + "output_utf8": "# styler: off\r\nx <- \"# string\"\r\n\r\n\r\n" + } + }, + { + "id": "r-builtin-all", + "language": "r", + "operation": "transform", + "options": { + "policy": "all", + "layout": "lines" + }, + "source_utf8": "# styler: off\r\nx <- \"# string\"\r\n#' doc\r\n# line\r\n", + "note": "R Language Definition, 10.2 Comments and `?Quotes`: a string hides the `#` inside it, `#'` is roxygen2's documentation marker and `#` an ordinary comment, `# styler: off` is the instruction the formatter reads out of a comment, and the CRLF endings survive the transformation. Ground truth, R 4.3.3 `utils::getParseData(parse(file =, keep.source = TRUE))`: `COMMENT` at [0,13), [32,38) and [40,46), and `STR_CONST \"\\\"# string\\\"\"` at [20,30).", + "expect": { + "valid": true, + "comments": [ + { + "start": 0, + "end": 13, + "kind": "directive", + "action": "remove" + }, + { + "start": 32, + "end": 38, + "kind": "doc-line", + "action": "remove" + }, + { + "start": 40, + "end": 46, + "kind": "line", + "action": "remove" + } + ], + "diagnostics": [], + "output_utf8": "\r\nx <- \"# string\"\r\n\r\n\r\n" + } + }, + { + "id": "dart-builtin-safe", + "language": "dart", + "operation": "transform", + "options": { + "policy": "safe", + "layout": "lines" + }, + "source_utf8": "// dart format off\r\nvar s = '// string';\r\n/// doc\r\n// line\r\n", + "note": "Dart Language Specification, 17.1 Comments and 17.6 Strings: a string literal hides the `//` inside it, `///` is a documentation comment and `//` an ordinary one, `// dart format off` is the whole phrase `dart_style` compares against, and the CRLF endings survive the transformation. Ground truth, Dart SDK 3.13.2 `scanString` of `package:_fe_analyzer_shared`: `SINGLE_LINE_COMMENT \"// dart format off\"` at [0,18), `STRING \"'// string'\"` at [28,39), `DartDocToken \"/// doc\"` at [42,49), and `SINGLE_LINE_COMMENT \"// line\"` at [51,58).", + "expect": { + "valid": true, + "comments": [ + { + "start": 0, + "end": 18, + "kind": "directive", + "action": "keep" + }, + { + "start": 42, + "end": 49, + "kind": "doc-line", + "action": "remove" + }, + { + "start": 51, + "end": 58, + "kind": "line", + "action": "remove" + } + ], + "diagnostics": [], + "output_utf8": "// dart format off\r\nvar s = '// string';\r\n\r\n\r\n" + } + }, + { + "id": "dart-builtin-all", + "language": "dart", + "operation": "transform", + "options": { + "policy": "all", + "layout": "lines" + }, + "source_utf8": "// dart format off\r\nvar s = '// string';\r\n/// doc\r\n// line\r\n", + "note": "Dart Language Specification, 17.1 Comments and 17.6 Strings: a string literal hides the `//` inside it, `///` is a documentation comment and `//` an ordinary one, `// dart format off` is the whole phrase `dart_style` compares against, and the CRLF endings survive the transformation. Ground truth, Dart SDK 3.13.2 `scanString` of `package:_fe_analyzer_shared`: `SINGLE_LINE_COMMENT \"// dart format off\"` at [0,18), `STRING \"'// string'\"` at [28,39), `DartDocToken \"/// doc\"` at [42,49), and `SINGLE_LINE_COMMENT \"// line\"` at [51,58).", + "expect": { + "valid": true, + "comments": [ + { + "start": 0, + "end": 18, + "kind": "directive", + "action": "remove" + }, + { + "start": 42, + "end": 49, + "kind": "doc-line", + "action": "remove" + }, + { + "start": 51, + "end": 58, + "kind": "line", + "action": "remove" + } + ], + "diagnostics": [], + "output_utf8": "\r\nvar s = '// string';\r\n\r\n\r\n" + } + }, + { + "id": "swift-builtin-safe", + "language": "swift", + "operation": "transform", + "options": { + "policy": "safe", + "layout": "lines" + }, + "source_utf8": "// swift-tools-version:5.9\r\nlet s = \"// string\"\r\n/// doc\r\n// line\r\n", + "note": "The Swift Programming Language, Lexical Structure (Comments, String Literals): a string literal hides the `//` inside it, `///` is a documentation comment and `//` an ordinary one, `// swift-tools-version:` is the line SwiftPM reads before it reads the manifest at all, and the CRLF endings survive the transformation. Ground truth, the SwiftSyntax parser of the Swift 6.3.3 toolchain (`SwiftParser.Parser.parse`, read for comment trivia and their UTF-8 offsets): `lineComment` at [0,26), the string `\"// string\"` as `stringQuote`/`stringSegment`/`stringQuote` at [36,47), `docLineComment` at [49,56) and `lineComment` at [58,65), with no parser diagnostic.", + "expect": { + "valid": true, + "comments": [ + { + "start": 0, + "end": 26, + "kind": "directive", + "action": "keep" + }, + { + "start": 49, + "end": 56, + "kind": "doc-line", + "action": "remove" + }, + { + "start": 58, + "end": 65, + "kind": "line", + "action": "remove" + } + ], + "diagnostics": [], + "output_utf8": "// swift-tools-version:5.9\r\nlet s = \"// string\"\r\n\r\n\r\n" + } + }, + { + "id": "swift-builtin-all", + "language": "swift", + "operation": "transform", + "options": { + "policy": "all", + "layout": "lines" + }, + "source_utf8": "// swift-tools-version:5.9\r\nlet s = \"// string\"\r\n/// doc\r\n// line\r\n", + "note": "The Swift Programming Language, Lexical Structure (Comments, String Literals): a string literal hides the `//` inside it, `///` is a documentation comment and `//` an ordinary one, `// swift-tools-version:` is the line SwiftPM reads before it reads the manifest at all, and the CRLF endings survive the transformation. Ground truth, the SwiftSyntax parser of the Swift 6.3.3 toolchain (`SwiftParser.Parser.parse`, read for comment trivia and their UTF-8 offsets): `lineComment` at [0,26), the string `\"// string\"` as `stringQuote`/`stringSegment`/`stringQuote` at [36,47), `docLineComment` at [49,56) and `lineComment` at [58,65), with no parser diagnostic.", + "expect": { + "valid": true, + "comments": [ + { + "start": 0, + "end": 26, + "kind": "directive", + "action": "remove" + }, + { + "start": 49, + "end": 56, + "kind": "doc-line", + "action": "remove" + }, + { + "start": 58, + "end": 65, + "kind": "line", + "action": "remove" + } + ], + "diagnostics": [], + "output_utf8": "\r\nlet s = \"// string\"\r\n\r\n\r\n" + } + }, + { + "id": "csharp-builtin-safe", + "language": "csharp", + "operation": "transform", + "options": { + "policy": "safe", + "layout": "lines" + }, + "source_utf8": "// \r\nvar s = \"// string\";\r\n/// doc\r\n// line\r\n", + "note": "ECMA-334 6.3.3 Comments and 6.4.5 String literals: a string literal hides the `//` inside it, `///` is a documentation comment and `//` an ordinary one, `// ` is the marker Roslyn exempts a generated file from every analyzer by, and the CRLF endings survive the transformation. Ground truth, the Roslyn lexer the .NET SDK 10.0.400 ships (`CSharpSyntaxTree.ParseText` with `LanguageVersion.Preview`, read for the comment trivia and their UTF-8 offsets): `SingleLineCommentTrivia` at [0,20), the string `\"// string\"` as one `StringLiteralToken` at [30,41), `SingleLineDocumentationCommentTrivia` at [44,51) and `SingleLineCommentTrivia` at [53,60), with no lexical diagnostic.", + "expect": { + "valid": true, + "comments": [ + { + "start": 0, + "end": 20, + "kind": "directive", + "action": "keep" + }, + { + "start": 44, + "end": 51, + "kind": "doc-line", + "action": "remove" + }, + { + "start": 53, + "end": 60, + "kind": "line", + "action": "remove" + } + ], + "diagnostics": [], + "output_utf8": "// \r\nvar s = \"// string\";\r\n\r\n\r\n" + } + }, + { + "id": "csharp-builtin-all", + "language": "csharp", + "operation": "transform", + "options": { + "policy": "all", + "layout": "lines" + }, + "source_utf8": "// \r\nvar s = \"// string\";\r\n/// doc\r\n// line\r\n", + "note": "ECMA-334 6.3.3 Comments and 6.4.5 String literals: a string literal hides the `//` inside it, `///` is a documentation comment and `//` an ordinary one, `// ` is the marker Roslyn exempts a generated file from every analyzer by, and the CRLF endings survive the transformation. Ground truth, the Roslyn lexer the .NET SDK 10.0.400 ships (`CSharpSyntaxTree.ParseText` with `LanguageVersion.Preview`, read for the comment trivia and their UTF-8 offsets): `SingleLineCommentTrivia` at [0,20), the string `\"// string\"` as one `StringLiteralToken` at [30,41), `SingleLineDocumentationCommentTrivia` at [44,51) and `SingleLineCommentTrivia` at [53,60), with no lexical diagnostic.", + "expect": { + "valid": true, + "comments": [ + { + "start": 0, + "end": 20, + "kind": "directive", + "action": "remove" + }, + { + "start": 44, + "end": 51, + "kind": "doc-line", + "action": "remove" + }, + { + "start": 53, + "end": 60, + "kind": "line", + "action": "remove" + } + ], + "diagnostics": [], + "output_utf8": "\r\nvar s = \"// string\";\r\n\r\n\r\n" + } + }, + { + "id": "scala-builtin-safe", + "language": "scala", + "operation": "transform", + "options": { + "policy": "safe", + "layout": "lines" + }, + "source_utf8": "//> using scala \"3.3.0\"\nval a = s\"${1 /* in */}\" // line\n/** doc */\nval b = // text\n", + "note": "The Scala 3.8 compiler's comment reader and lexer: a block comment nests and `/**` is the documentation marker, `//> using` is the directive scala-cli reads before it reads a manifest at all, an interpolated string's `${ ... }` is code so the `/* in */` inside it is a comment, and an XML literal's text is not code so the `// text` inside `` is protected. Ground truth, scalac 3.8.4: the source parses (only `scala.xml` is missing on the classpath), the compiler reports `//> using scala \"3.3.0\"` as a line comment at [0,23), `/* in */` at [38,46), `// line` at [50,57) and `/** doc */` at [58,68), and the XML text as none of them.", + "expect": { + "valid": true, + "comments": [ + { + "start": 0, + "end": 23, + "kind": "directive", + "action": "keep" + }, + { + "start": 38, + "end": 46, + "kind": "block", + "action": "remove" + }, + { + "start": 50, + "end": 57, + "kind": "line", + "action": "remove" + }, + { + "start": 58, + "end": 68, + "kind": "doc-block", + "action": "remove" + } + ], + "diagnostics": [], + "output_utf8": "//> using scala \"3.3.0\"\nval a = s\"${1 }\" \n\nval b = // text\n" + } + }, + { + "id": "scala-builtin-all", + "language": "scala", + "operation": "transform", + "options": { + "policy": "all", + "layout": "lines" + }, + "source_utf8": "val a = \"\"\"a\"\"\"\"\nval b = s\"\"\"${1 // in\n} \"\"\"\n//> using scala \"3\"\nval c = `a//b`\n// line\n", + "note": "The Scala 3.8 compiler's lexer: a triple-quoted string closes on the first three quotes of a run and makes a fourth part of its value, so `\"\"\"a\"\"\"\"` is the string `a\"`; a `//` inside a multi-line string's interpolation is a comment; a backquoted identifier may hold `//` without it being one; and under `Policy::All` the `//> using` directive is removed like any other comment. Ground truth, scalac 3.8.4: `// in` at [33,38), `//> using scala \"3\"` at [45,64) and `// line` at [80,87), with the two string forms lexed as single tokens and no diagnostic.", + "expect": { + "valid": true, + "comments": [ + { + "start": 33, + "end": 38, + "kind": "line", + "action": "remove" + }, + { + "start": 45, + "end": 64, + "kind": "directive", + "action": "remove" + }, + { + "start": 80, + "end": 87, + "kind": "line", + "action": "remove" + } + ], + "diagnostics": [], + "output_utf8": "val a = \"\"\"a\"\"\"\"\nval b = s\"\"\"${1 \n} \"\"\"\n\nval c = `a//b`\n\n" + } + }, + { + "id": "vue-builtin-safe", + "language": "vue", + "operation": "transform", + "options": { + "policy": "safe", + "layout": "lines" + }, + "source_utf8": "\n\n\n", + "note": "Vue single-file components, @vue/compiler-sfc 3.5: the template's `` is an HTML comment that `safe` keeps as DOM-observable, `{{ ... }}` opens an expression whose `/* c */` is a comment, and the `\n\n" + } + }, + { + "id": "svelte-builtin-safe", + "language": "svelte", + "operation": "transform", + "options": { + "policy": "safe", + "layout": "lines" + }, + "source_utf8": "\n\n

{x /* c */}

\n\n", + "note": "Svelte components, svelte/compiler 5.56: the `\n\n

{x }

\n\n" + } + }, + { + "id": "markdown-builtin-safe", + "language": "markdown", + "operation": "transform", + "options": { + "policy": "safe", + "layout": "lines" + }, + "source_utf8": "text\n\nmore\n```rust\n// c\n```\n`// inline`\n", + "note": "CommonMark 0.31: `` is an HTML block that `safe` keeps as DOM-observable, a ```rust fence is scanned as Rust so the `// c` inside is a comment, and an inline code span is opaque. Ground truth, the `commonmark` package: the comment parses as one `html_block` and the fence as a `code_block` with info `rust`.", + "expect": { + "valid": true, + "comments": [ + { + "start": 5, + "end": 18, + "kind": "html-comment", + "action": "keep" + }, + { + "start": 32, + "end": 36, + "kind": "line", + "action": "remove" + } + ], + "diagnostics": [], + "output_utf8": "text\n\nmore\n```rust\n\n```\n`// inline`\n" + } + }, + { + "id": "perl-builtin-safe", + "language": "perl", + "operation": "transform", + "options": { + "policy": "safe", + "layout": "lines" + }, + "source_utf8": "=head1 NAME\n# not a comment\n=cut\nmy $x = 'a#b';\nif ($x =~ /a#b/) { print \"yes\\n\" }\nmy $y = $x / 2; # division\n", + "note": "Perl 5.38: the POD block and the string and regex are opaque \u2014 the `#` inside them is content \u2014 and `$x / 2` is a division whose `# division` is a comment. Ground truth, `perl -c`: the source compiles.", + "expect": { + "valid": true, + "comments": [ + { + "start": 99, + "end": 109, + "kind": "line", + "action": "remove" + } + ], + "diagnostics": [], + "output_utf8": "=head1 NAME\n# not a comment\n=cut\nmy $x = 'a#b';\nif ($x =~ /a#b/) { print \"yes\\n\" }\nmy $y = $x / 2; \n" + } } ] } diff --git a/spec/fixtures/v1/floor.txt b/spec/fixtures/v1/floor.txt new file mode 100644 index 0000000..3f36a2d --- /dev/null +++ b/spec/fixtures/v1/floor.txt @@ -0,0 +1,20 @@ +# The floors both corpus runners hold this corpus to, in one file so that they +# cannot drift apart. `tools/differential.py` and +# `rust/ocomment-core/tests/spec_fixtures.rs` both read it; neither carries a +# number of its own. +# +# cases the least number of cases `v1/*.json` may hold. +# expectations the least number of those that must carry a recorded `expect` +# block. A case with none is still held to the structural promises, +# so this floor is what stops the corpus from degrading into that +# weaker check. +# +# Raising either is optional when a case is added and mandatory when the corpus +# is reorganised: the floors exist so that neither a case nor its recorded +# expectation can quietly disappear. +# +# Blank lines and `#` lines are ignored; every other line is a name and a +# decimal count separated by white space. + +cases 466 +expectations 466 diff --git a/spec/fixtures/v1/hazards.json b/spec/fixtures/v1/hazards.json new file mode 100644 index 0000000..6ee8281 --- /dev/null +++ b/spec/fixtures/v1/hazards.json @@ -0,0 +1,11198 @@ +{ + "version": 1, + "cases": [ + { + "id": "rust-nested-raw", + "language": "rust", + "operation": "transform", + "options": { + "policy": "safe", + "layout": "lines" + }, + "source_utf8": "r#\"// opaque\"# /* outer /* inner */ end */\\n// rustfmt::skip\\n", + "note": "Rust Reference, Tokens: raw string literals take no escapes, and block comments nest. The raw string hides `//`, `/* outer /* inner */ end */` is one comment, and `// rustfmt::skip` is a directive `safe` keeps.", + "expect": { + "valid": true, + "comments": [ + { + "start": 15, + "end": 42, + "kind": "block", + "action": "remove" + }, + { + "start": 44, + "end": 62, + "kind": "directive", + "action": "keep" + } + ], + "diagnostics": [], + "output_utf8": "r#\"// opaque\"# \\n// rustfmt::skip\\n" + } + }, + { + "id": "rust-raw-c-string", + "language": "rust", + "operation": "transform", + "options": { + "policy": "safe", + "layout": "lines" + }, + "source_utf8": "cr#\"inner \" // opaque\"#; // remove\n", + "note": "Rust Reference, Tokens (raw C string literals, Rust 1.77): `cr#\"...\"#` ends only at `\"#`, so the unescaped `\"` and the `//` inside it are opaque and only the trailing line comment is found.", + "expect": { + "valid": true, + "comments": [ + { + "start": 25, + "end": 34, + "kind": "line", + "action": "remove" + } + ], + "diagnostics": [], + "output_utf8": "cr#\"inner \" // opaque\"#; \n" + } + }, + { + "id": "rust-multiline-string", + "language": "rust", + "operation": "transform", + "options": { + "policy": "safe", + "layout": "lines" + }, + "source_utf8": "const A: &str = \"a\n// opaque\nb\"; // remove\n", + "note": "Rust Reference, Tokens: a string literal may contain newlines, so `//` on an inner line is opaque and only the trailing line comment is found.", + "expect": { + "valid": true, + "comments": [ + { + "start": 33, + "end": 42, + "kind": "line", + "action": "remove" + } + ], + "diagnostics": [], + "output_utf8": "const A: &str = \"a\n// opaque\nb\"; \n" + } + }, + { + "id": "ocaml-nested-quoted", + "language": "ocaml", + "operation": "transform", + "options": { + "policy": "safe", + "layout": "lines" + }, + "source_utf8": "{tag| (* opaque *) |tag} (* outer \"*)\" (* inner *) *)", + "note": "OCaml manual, Lexical conventions: quoted string literals `{id|...|id}` take no escapes, and comments nest. The quoted string hides `(*`, and the nested comment closes once.", + "expect": { + "valid": true, + "comments": [ + { + "start": 25, + "end": 53, + "kind": "block", + "action": "remove" + } + ], + "diagnostics": [], + "output_utf8": "{tag| (* opaque *) |tag} " + } + }, + { + "id": "ocaml-comment-quoted", + "language": "ocaml", + "operation": "transform", + "options": { + "policy": "safe", + "layout": "lines" + }, + "source_utf8": "(* outer {tag| *) opaque |tag} end *)", + "note": "OCaml manual, Comments: a comment may contain a string literal, and the literal's contents are not scanned for `*)`, so the `*)` inside `{tag|...|tag}` does not close the comment.", + "expect": { + "valid": true, + "comments": [ + { + "start": 0, + "end": 37, + "kind": "block", + "action": "remove" + } + ], + "diagnostics": [], + "output_utf8": "" + } + }, + { + "id": "ocaml-long-quoted-id", + "language": "ocaml", + "operation": "transform", + "options": { + "policy": "safe", + "layout": "lines" + }, + "source_utf8": "{aaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaa|(* opaque *)|aaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaa} (* remove *)", + "note": "OCaml manual, Lexical conventions: the identifier of a quoted string literal has no length limit, so an 80-character tag must still be matched exactly at the close.", + "expect": { + "valid": true, + "comments": [ + { + "start": 177, + "end": 189, + "kind": "block", + "action": "remove" + } + ], + "diagnostics": [], + "output_utf8": "{aaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaa|(* opaque *)|aaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaa} " + } + }, + { + "id": "invalid-ocaml-quoted", + "language": "ocaml", + "operation": "transform", + "options": { + "policy": "safe", + "layout": "lines" + }, + "source_utf8": "{tag| unterminated (* opaque *)", + "note": "OCaml manual, Lexical conventions: an unterminated quoted string literal is a lexical error, reported as an error diagnostic that makes the report invalid.", + "expect": { + "valid": false, + "comments": [], + "diagnostics": [ + { + "code": "unterminated-string", + "start": 0, + "end": 31 + } + ], + "output_utf8": "{tag| unterminated (* opaque *)" + } + }, + { + "id": "c-line-splice", + "language": "c", + "operation": "transform", + "options": { + "policy": "safe", + "layout": "lines" + }, + "source_utf8": "int x; /\\\n/ comment\\\ncontinued\nint y;", + "note": "ISO C17 5.1.1.2 translation phases: backslash-newline splicing (phase 2) happens before comments are recognised (phase 3), so `/\\/` opens a line comment and a spliced line comment runs on to the next line.", + "expect": { + "valid": true, + "comments": [ + { + "start": 7, + "end": 30, + "kind": "line", + "action": "remove" + } + ], + "diagnostics": [], + "output_utf8": "int x; \n\n\nint y;" + } + }, + { + "id": "cpp-raw", + "language": "cpp", + "operation": "transform", + "options": { + "policy": "safe", + "layout": "lines" + }, + "source_utf8": "R\"tag(/* opaque */ // opaque)tag\" // remove", + "note": "ISO C++ [lex.string]: a raw string literal `R\"tag(...)tag\"` ends only at the matching delimiter, so both comment forms inside it are opaque.", + "expect": { + "valid": true, + "comments": [ + { + "start": 34, + "end": 43, + "kind": "line", + "action": "remove" + } + ], + "diagnostics": [], + "output_utf8": "R\"tag(/* opaque */ // opaque)tag\" " + } + }, + { + "id": "go-directives", + "language": "go", + "operation": "transform", + "options": { + "policy": "safe", + "layout": "lines" + }, + "source_utf8": "//go:build linux\n// +build linux\n//line generated.go:1\n// remove\n", + "note": "Go compiler directives: `//go:build`, the legacy `// +build` constraint, and `//line` are directives `safe` keeps; the plain line comment is removed.", + "expect": { + "valid": true, + "comments": [ + { + "start": 0, + "end": 16, + "kind": "directive", + "action": "keep" + }, + { + "start": 17, + "end": 32, + "kind": "directive", + "action": "keep" + }, + { + "start": 33, + "end": 54, + "kind": "directive", + "action": "keep" + }, + { + "start": 55, + "end": 64, + "kind": "line", + "action": "remove" + } + ], + "diagnostics": [], + "output_utf8": "//go:build linux\n// +build linux\n//line generated.go:1\n\n" + } + }, + { + "id": "java-unicode", + "language": "java", + "operation": "transform", + "options": { + "policy": "safe", + "layout": "lines" + }, + "source_utf8": "int x; \\u002f\\u002f comment\\u000aint y;", + "note": "JLS 3.3 Unicode escapes: `\\uXXXX` is translated before lexing, so `\\u002f\\u002f` opens a line comment and `\\u000a` is the line terminator that ends it.", + "expect": { + "valid": true, + "comments": [ + { + "start": 7, + "end": 27, + "kind": "line", + "action": "remove" + } + ], + "diagnostics": [], + "output_utf8": "int x; \\u000aint y;" + } + }, + { + "id": "java-unicode-surrogates", + "language": "java", + "operation": "transform", + "options": { + "policy": "safe", + "layout": "lines" + }, + "source_utf8": "String s = \"\\uD83D\\uDE00 // opaque\"; // remove", + "note": "JLS 3.3: a surrogate pair written as two Unicode escapes is one code point; it is inside a string literal, so the `//` after it is opaque and only the trailing comment is found.", + "expect": { + "valid": true, + "comments": [ + { + "start": 37, + "end": 46, + "kind": "line", + "action": "remove" + } + ], + "diagnostics": [], + "output_utf8": "String s = \"\\uD83D\\uDE00 // opaque\"; " + } + }, + { + "id": "invalid-java-unicode", + "language": "java", + "operation": "transform", + "options": { + "policy": "safe", + "layout": "lines" + }, + "source_utf8": "int x = 1; \\u00G0 // known", + "note": "JLS 3.3: an eligible backslash not followed by four hexadecimal digits is a compile-time error, reported as an error diagnostic while the trailing comment is still located.", + "expect": { + "valid": false, + "comments": [ + { + "start": 18, + "end": 26, + "kind": "line", + "action": "remove" + } + ], + "diagnostics": [ + { + "code": "invalid-unicode-escape", + "start": 11, + "end": 17 + } + ], + "output_utf8": "int x = 1; \\u00G0 // known" + } + }, + { + "id": "forced-invalid-java-unicode", + "language": "java", + "operation": "transform", + "options": { + "policy": "safe", + "layout": "lines", + "force_invalid": true + }, + "source_utf8": "int x = 1; \\u00G0 // known", + "note": "The `invalid-java-unicode` source with `force_invalid`: the error diagnostic stands, and edits are produced anyway.", + "expect": { + "valid": false, + "comments": [ + { + "start": 18, + "end": 26, + "kind": "line", + "action": "remove" + } + ], + "diagnostics": [ + { + "code": "invalid-unicode-escape", + "start": 11, + "end": 17 + } + ], + "output_utf8": "int x = 1; \\u00G0 " + } + }, + { + "id": "java-text-block-escape", + "language": "java", + "operation": "transform", + "options": { + "policy": "safe", + "layout": "lines" + }, + "source_utf8": "String s = \"\"\"\n\\\"\"\" // opaque\nend\n\"\"\"; // remove\n", + "note": "JLS 3.10.6 text blocks: an escaped `\\\"\"\"` does not close the block, so the `//` inside it is opaque.", + "expect": { + "valid": true, + "comments": [ + { + "start": 39, + "end": 48, + "kind": "line", + "action": "remove" + } + ], + "diagnostics": [], + "output_utf8": "String s = \"\"\"\n\\\"\"\" // opaque\nend\n\"\"\"; \n" + } + }, + { + "id": "java-inner-doc-marker", + "language": "java", + "operation": "transform", + "options": { + "policy": "safe", + "layout": "lines" + }, + "source_utf8": "/// javadoc\n//! plain\n/** javadoc */\n/*! plain */\nclass A {}\n", + "note": "JLS 3.7 and JEP 467 (Markdown documentation comments): Java's documentation comments are `/** ... */` and, since JDK 23, `///`. `//!` and `/*!` are the Rust inner-doc and Doxygen markers, which Java has no convention for, so a comment opening with either is an ordinary line or block comment.", + "expect": { + "valid": true, + "comments": [ + { + "start": 0, + "end": 11, + "kind": "doc-line", + "action": "remove" + }, + { + "start": 12, + "end": 21, + "kind": "line", + "action": "remove" + }, + { + "start": 22, + "end": 36, + "kind": "doc-block", + "action": "remove" + }, + { + "start": 37, + "end": 49, + "kind": "block", + "action": "remove" + } + ], + "diagnostics": [], + "output_utf8": "\n\n\n\nclass A {}\n" + } + }, + { + "id": "javascript-goals", + "language": "javascript", + "operation": "transform", + "options": { + "policy": "safe", + "layout": "lines" + }, + "source_utf8": "#!/usr/bin/env node\nconst r = /\\/\\/* opaque/;\nconst t = `literal // opaque ${1 /* remove */}`;\n// remove\n", + "note": "ECMA-262 lexical goal symbols: a `/` after `=` starts a regular expression, a template literal's text is opaque, and its substitution returns to the code goal. The hashbang (ECMA-262 Hashbang Grammar) is protected.", + "expect": { + "valid": true, + "comments": [ + { + "start": 0, + "end": 19, + "kind": "shebang", + "action": "keep" + }, + { + "start": 79, + "end": 91, + "kind": "block", + "action": "remove" + }, + { + "start": 95, + "end": 104, + "kind": "line", + "action": "remove" + } + ], + "diagnostics": [], + "output_utf8": "#!/usr/bin/env node\nconst r = /\\/\\/* opaque/;\nconst t = `literal // opaque ${1 }`;\n\n" + } + }, + { + "id": "javascript-control-regex", + "language": "javascript", + "operation": "transform", + "options": { + "policy": "safe", + "layout": "lines" + }, + "source_utf8": "if (ready) /https?:\\/\\/example\\.test/.test(value); // remove", + "note": "ECMA-262 InputElementRegExp: after the `)` closing an `if` condition a `/` starts a regular expression, so the escaped `\\/\\/` in it is opaque.", + "expect": { + "valid": true, + "comments": [ + { + "start": 51, + "end": 60, + "kind": "line", + "action": "remove" + } + ], + "diagnostics": [], + "output_utf8": "if (ready) /https?:\\/\\/example\\.test/.test(value); " + } + }, + { + "id": "javascript-brace-goals", + "language": "javascript", + "operation": "transform", + "options": { + "policy": "safe", + "layout": "lines" + }, + "source_utf8": "const ratio = {} / 2; // remove\nif (ready) {} /[/*]/.test(value); // remove\n", + "note": "ECMA-262 lexical goal symbols: `{}` after `=` is an object literal and the following `/` is division, while `{}` after `if (...)` is a block and the following `/` starts a regular expression.", + "expect": { + "valid": true, + "comments": [ + { + "start": 22, + "end": 31, + "kind": "line", + "action": "remove" + }, + { + "start": 66, + "end": 75, + "kind": "line", + "action": "remove" + } + ], + "diagnostics": [], + "output_utf8": "const ratio = {} / 2; \nif (ready) {} /[/*]/.test(value); \n" + } + }, + { + "id": "javascript-html-like-comments", + "language": "javascript", + "operation": "transform", + "options": { + "policy": "safe", + "layout": "lines" + }, + "source_utf8": "const x = 1; remove\nconst text = '` is one at the start of a line, but `", + "note": "Layout `columns` after a removed HTML comment that spans lines: column positions are recomputed from the current line, not from the original offsets.", + "expect": { + "valid": true, + "comments": [ + { + "start": 2, + "end": 20, + "kind": "html-comment", + "action": "remove" + }, + { + "start": 36, + "end": 41, + "kind": "block", + "action": "remove" + } + ], + "diagnostics": [], + "output_utf8": "ab" + } + }, + { + "id": "non-utf8-bytes", + "language": "c", + "operation": "transform", + "options": { + "policy": "safe", + "layout": "lines" + }, + "source_base64": "/y8qIHJlbW92ZSAqL4ANCg==", + "note": "Bytes that are not UTF-8 around a comment are copied through untouched, and the CRLF ending survives.", + "expect": { + "valid": true, + "comments": [ + { + "start": 1, + "end": 13, + "kind": "block", + "action": "remove" + } + ], + "diagnostics": [], + "output_base64": "/yCADQo=" + } + }, + { + "id": "compact-layout", + "language": "c", + "operation": "transform", + "options": { + "policy": "safe", + "layout": "compact" + }, + "source_utf8": "left/* remove */right\n", + "note": "Layout `compact` over a comment between two tokens on one line: nothing about the line changes but the comment, so `compact` leaves exactly what `lines` leaves, the separating space included.", + "expect": { + "valid": true, + "comments": [ + { + "start": 4, + "end": 16, + "kind": "block", + "action": "remove" + } + ], + "diagnostics": [], + "output_utf8": "left right\n" + } + }, + { + "id": "compact-whole-line-run", + "language": "rust", + "operation": "transform", + "options": { + "policy": "safe", + "layout": "compact" + }, + "source_utf8": "fn main() {}\n// one\n// two\nlet x = 1;\n", + "note": "Layout `compact` takes a line that held nothing but a removed comment away entirely, its terminator included, so a run of whole-line comments disappears instead of leaving a run of blank lines. Layout `lines` keeps all four lines.", + "expect": { + "valid": true, + "comments": [ + { + "start": 13, + "end": 19, + "kind": "line", + "action": "remove" + }, + { + "start": 20, + "end": 26, + "kind": "line", + "action": "remove" + } + ], + "diagnostics": [], + "output_utf8": "fn main() {}\nlet x = 1;\n" + } + }, + { + "id": "compact-indented-line", + "language": "rust", + "operation": "transform", + "options": { + "policy": "safe", + "layout": "compact" + }, + "source_utf8": "fn main() {\n // note\n let x = 1;\n}\n", + "note": "Layout `compact`: only whitespace stood before the comment on its line, so the indentation goes with the line rather than staying behind as trailing whitespace.", + "expect": { + "valid": true, + "comments": [ + { + "start": 16, + "end": 23, + "kind": "line", + "action": "remove" + } + ], + "diagnostics": [], + "output_utf8": "fn main() {\n let x = 1;\n}\n" + } + }, + { + "id": "compact-crlf-line", + "language": "rust", + "operation": "transform", + "options": { + "policy": "safe", + "layout": "compact" + }, + "source_utf8": "let x = 1;\r\n// note\r\nlet y = 2;\r\n", + "note": "Layout `compact` over CRLF endings: the terminator removed with the line is that line's own CRLF, and the lines that survive keep theirs byte for byte.", + "expect": { + "valid": true, + "comments": [ + { + "start": 12, + "end": 19, + "kind": "line", + "action": "remove" + } + ], + "diagnostics": [], + "output_utf8": "let x = 1;\r\nlet y = 2;\r\n" + } + }, + { + "id": "compact-trailing-whitespace", + "language": "rust", + "operation": "transform", + "options": { + "policy": "safe", + "layout": "compact" + }, + "source_utf8": "let x = 1; \t // note\nlet y = 2;\t/* two */\t\nlet z = 3;\n", + "note": "Layout `compact` trims the whitespace a removal would otherwise leave at the end of a line, on both sides of the comment. Code survives on each line, so each keeps its terminator.", + "expect": { + "valid": true, + "comments": [ + { + "start": 13, + "end": 20, + "kind": "line", + "action": "remove" + }, + { + "start": 32, + "end": 41, + "kind": "block", + "action": "remove" + } + ], + "diagnostics": [], + "output_utf8": "let x = 1;\nlet y = 2;\nlet z = 3;\n" + } + }, + { + "id": "compact-no-final-newline", + "language": "rust", + "operation": "transform", + "options": { + "policy": "safe", + "layout": "compact" + }, + "source_utf8": "let x = 1; // note", + "note": "Layout `compact` over a file that ends without a final newline: the trailing comment and the space before it go, and no terminator is invented for the line that survives.", + "expect": { + "valid": true, + "comments": [ + { + "start": 11, + "end": 18, + "kind": "line", + "action": "remove" + } + ], + "diagnostics": [], + "output_utf8": "let x = 1;" + } + }, + { + "id": "compact-last-line-only-comment", + "language": "rust", + "operation": "transform", + "options": { + "policy": "safe", + "layout": "compact" + }, + "source_utf8": "let x = 1;\n// note", + "note": "Layout `compact`: the last line held nothing but the comment and had no terminator of its own, so the line goes and the line before it keeps the terminator it always had.", + "expect": { + "valid": true, + "comments": [ + { + "start": 11, + "end": 18, + "kind": "line", + "action": "remove" + } + ], + "diagnostics": [], + "output_utf8": "let x = 1;\n" + } + }, + { + "id": "compact-block-shares-lines-with-code", + "language": "c", + "operation": "transform", + "options": { + "policy": "safe", + "layout": "compact" + }, + "source_utf8": "int a = 1; /* one\ntwo\nthree */ int b = 2;\n", + "note": "ISO C17 6.4.9p1: `/*` ... `*/` is one comment however many lines it spans. Layout `compact` keeps the code before it on a line of its own, ends that line with the terminator that ended it in the source, and drops the lines that held only comment.", + "expect": { + "valid": true, + "comments": [ + { + "start": 11, + "end": 30, + "kind": "block", + "action": "remove" + } + ], + "diagnostics": [], + "output_utf8": "int a = 1;\n int b = 2;\n" + } + }, + { + "id": "compact-block-alone-on-its-lines", + "language": "c", + "operation": "transform", + "options": { + "policy": "safe", + "layout": "compact" + }, + "source_utf8": "int a = 1;\n/* one\ntwo */\nint b = 2;\n", + "note": "ISO C17 6.4.9p1. Layout `compact`: every line the block comment covered held nothing else, so all of them go, terminators included.", + "expect": { + "valid": true, + "comments": [ + { + "start": 11, + "end": 24, + "kind": "block", + "action": "remove" + } + ], + "diagnostics": [], + "output_utf8": "int a = 1;\nint b = 2;\n" + } + }, + { + "id": "compact-block-at-end-without-newline", + "language": "c", + "operation": "transform", + "options": { + "policy": "safe", + "layout": "compact" + }, + "source_utf8": "int x = 1; /* one\ntwo */", + "note": "ISO C17 6.4.9p1. Layout `compact` over a block comment that runs to the end of a file with no final newline: the line the code is on kept its terminator inside the comment, so that terminator comes back, and the last line held only comment and goes.", + "expect": { + "valid": true, + "comments": [ + { + "start": 11, + "end": 24, + "kind": "block", + "action": "remove" + } + ], + "diagnostics": [], + "output_utf8": "int x = 1;\n" + } + }, + { + "id": "compact-two-comments-on-one-line", + "language": "c", + "operation": "transform", + "options": { + "policy": "safe", + "layout": "compact" + }, + "source_utf8": "a/* one */ /* two */\n", + "note": "Layout `compact` judges being alone on a line from the original bytes: neither comment was, so the line keeps its terminator, and the trim closes the gap the first removal left before the second.", + "expect": { + "valid": true, + "comments": [ + { + "start": 1, + "end": 10, + "kind": "block", + "action": "remove" + }, + { + "start": 11, + "end": 20, + "kind": "block", + "action": "remove" + } + ], + "diagnostics": [], + "output_utf8": "a\n" + } + }, + { + "id": "compact-html-comment", + "language": "html", + "operation": "transform", + "options": { + "policy": "all", + "layout": "compact" + }, + "source_utf8": "

a

\n\n

b

\n", + "note": "HTML 13.2.5.4x: `` is one comment token. An HTML comment closes up completely under every layout, the newlines it spanned included; `compact` adds the whole-line rule and the trim, and never takes the terminator of a line where code survives.", + "expect": { + "valid": true, + "comments": [ + { + "start": 9, + "end": 22, + "kind": "html-comment", + "action": "remove" + }, + { + "start": 32, + "end": 48, + "kind": "html-comment", + "action": "remove" + } + ], + "diagnostics": [], + "output_utf8": "

a

\n

b

\n" + } + }, + { + "id": "compact-javascript-line-separator", + "language": "javascript", + "operation": "transform", + "options": { + "policy": "safe", + "layout": "compact" + }, + "source_base64": "bGV0IGEgPSAxO+KAqC8vIG5vdGXigKhsZXQgYiA9IDI7Cg==", + "note": "ECMA-262 12.3: U+2028 LINE SEPARATOR is a LineTerminator, so it ends a SingleLineComment. Layout `compact` removes it with the line it terminated and leaves the other one alone.", + "expect": { + "valid": true, + "comments": [ + { + "start": 13, + "end": 20, + "kind": "line", + "action": "remove" + } + ], + "diagnostics": [], + "output_base64": "bGV0IGEgPSAxO+KAqGxldCBiID0gMjsK" + } + }, + { + "id": "compact-kept-comment-holds-its-line", + "language": "rust", + "operation": "transform", + "options": { + "policy": "safe", + "layout": "compact" + }, + "source_utf8": "// rustfmt::skip\n// note\nfn main() {}\n", + "note": "Layout `compact` over a kept directive above a removed comment: a comment that survives is content on its line, so only the line of the removed one goes.", + "expect": { + "valid": true, + "comments": [ + { + "start": 0, + "end": 16, + "kind": "directive", + "action": "keep" + }, + { + "start": 17, + "end": 24, + "kind": "line", + "action": "remove" + } + ], + "diagnostics": [], + "output_utf8": "// rustfmt::skip\nfn main() {}\n" + } + }, + { + "id": "invalid-cpp-raw", + "language": "cpp", + "operation": "transform", + "options": { + "policy": "safe", + "layout": "lines" + }, + "source_utf8": "R\"tag(unterminated /* opaque */", + "note": "ISO C++ [lex.string]: an unterminated raw string literal is a lexical error, reported as an error diagnostic that makes the report invalid.", + "expect": { + "valid": false, + "comments": [], + "diagnostics": [ + { + "code": "unterminated-string", + "start": 0, + "end": 31 + } + ], + "output_utf8": "R\"tag(unterminated /* opaque */" + } + }, + { + "id": "invalid-shell-quote", + "language": "shell", + "operation": "transform", + "options": { + "policy": "safe", + "layout": "lines" + }, + "source_utf8": "echo 'unterminated", + "note": "POSIX 2.2.2 single quotes: an unterminated single quote is a lexical error, reported as an error diagnostic that makes the report invalid.", + "expect": { + "valid": false, + "comments": [], + "diagnostics": [ + { + "code": "unterminated-string", + "start": 5, + "end": 18 + } + ], + "output_utf8": "echo 'unterminated" + } + }, + { + "id": "invalid-shell-heredoc", + "language": "shell", + "operation": "transform", + "options": { + "policy": "safe", + "layout": "lines" + }, + "source_utf8": "cat <:`, and the colon is the boundary that ends the tool's name, so each of the first four lines is a directive and only the last is removed.", + "expect": { + "valid": true, + "comments": [ + { + "start": 0, + "end": 23, + "kind": "directive", + "action": "keep" + }, + { + "start": 24, + "end": 57, + "kind": "directive", + "action": "keep" + }, + { + "start": 58, + "end": 75, + "kind": "directive", + "action": "keep" + }, + { + "start": 76, + "end": 94, + "kind": "directive", + "action": "keep" + }, + { + "start": 95, + "end": 106, + "kind": "line", + "action": "remove" + } + ], + "diagnostics": [], + "output_utf8": "-- luacheck: ignore 212\n-- selene: allow(unused_variable)\n-- stylua: ignore\n-- luacov: disable\n\n" + } + }, + { + "id": "lua-shebang-first-line", + "language": "lua", + "operation": "transform", + "options": { + "policy": "safe", + "layout": "lines" + }, + "source_utf8": "#!/usr/bin/env lua\nprint(1) -- remove\n", + "note": "Lua 5.4 reference manual, 7 (the standalone interpreter): the loader skips a first line that opens with `#`, which is what lets a chunk carry a `#!` line. It is a required preamble that only `--force-protected` gives up.", + "expect": { + "valid": true, + "comments": [ + { + "start": 0, + "end": 18, + "kind": "shebang", + "action": "keep" + }, + { + "start": 28, + "end": 37, + "kind": "line", + "action": "remove" + } + ], + "diagnostics": [], + "output_utf8": "#!/usr/bin/env lua\nprint(1) \n" + } + }, + { + "id": "lua-hash-first-line", + "language": "lua", + "operation": "transform", + "options": { + "policy": "safe", + "layout": "lines" + }, + "source_utf8": "# the loader skips this\nx = 1\n#not a comment\n", + "note": "Lua 5.4 reference manual, 7 (the standalone interpreter): the loader skips the whole of a first line that opens with `#`, whether or not a `!` follows, so a bare one is a comment Lua never reads and a `safe` run may remove it. On any later line `#` is the length operator, which is why the third line holds no comment at all.", + "expect": { + "valid": true, + "comments": [ + { + "start": 0, + "end": 23, + "kind": "line", + "action": "remove" + } + ], + "diagnostics": [], + "output_utf8": "\nx = 1\n#not a comment\n" + } + }, + { + "id": "lua-crlf", + "language": "lua", + "operation": "transform", + "options": { + "policy": "safe", + "layout": "lines" + }, + "source_utf8": "--[[ long\r\ncomment ]]\r\ns = \"x \\\r\ny\"\r\nz = 1 -- yes\r\n", + "note": "Lua 5.4 reference manual, 3.1: a long comment carries its line breaks as content and a backslash before one carries it into a short string; `\\r\\n` is one line break to the lexer, so the CRLF endings survive the transformation.", + "expect": { + "valid": true, + "comments": [ + { + "start": 0, + "end": 21, + "kind": "block", + "action": "remove" + }, + { + "start": 43, + "end": 49, + "kind": "line", + "action": "remove" + } + ], + "diagnostics": [], + "output_utf8": "\r\n\r\ns = \"x \\\r\ny\"\r\nz = 1 \r\n" + } + }, + { + "id": "lua-columns-layout", + "language": "lua", + "operation": "transform", + "options": { + "policy": "safe", + "layout": "columns" + }, + "source_utf8": "x = 1 -- note\ny = 2\n", + "note": "Lua 5.4 reference manual, 3.1: a short comment runs to the end of the line, and the `columns` layout pads what it removes so the following columns do not shift.", + "expect": { + "valid": true, + "comments": [ + { + "start": 6, + "end": 13, + "kind": "line", + "action": "remove" + } + ], + "diagnostics": [], + "output_utf8": "x = 1 \ny = 2\n" + } + }, + { + "id": "lua-compact-layout", + "language": "lua", + "operation": "transform", + "options": { + "policy": "safe", + "layout": "compact" + }, + "source_utf8": "-- alone\nx = 1 -- trailing\ny = 2\n", + "note": "Lua 5.4 reference manual, 3.1: the `compact` layout drops a line that held only a removed comment and the whitespace a trailing one left behind.", + "expect": { + "valid": true, + "comments": [ + { + "start": 0, + "end": 8, + "kind": "line", + "action": "remove" + }, + { + "start": 15, + "end": 26, + "kind": "line", + "action": "remove" + } + ], + "diagnostics": [], + "output_utf8": "x = 1\ny = 2\n" + } + }, + { + "id": "parity-unterminated-message-rust", + "language": "rust", + "operation": "scan", + "options": { + "policy": "safe", + "layout": "lines" + }, + "source_utf8": "let s = \"unclosed // not a comment\n", + "note": "Rust Reference, Tokens (string literals): a `\"` literal carries a bare newline as content, so this one is closed by nothing and the `//` inside it is never a comment. The diagnostic names the construct that was left open -- `unterminated string` -- rather than saying `literal`, and the two implementations must spell the message the same way, because a message is what a user reads. The recorded block pins the code and the span; the byte-for-byte comparison in `tools/differential.py` is what pins the text.", + "expect": { + "valid": false, + "comments": [], + "diagnostics": [ + { + "code": "unterminated-string", + "start": 8, + "end": 35 + } + ] + } + }, + { + "id": "parity-unterminated-message-go", + "language": "go", + "operation": "scan", + "options": { + "policy": "safe", + "layout": "lines" + }, + "source_utf8": "s := \"unclosed // not a comment\n", + "note": "Go specification, String literals: an interpreted string literal may not contain a newline, so this one ends unterminated at the line break. Go spells one diagnostic for both of its quoted forms -- `unterminated string or rune literal` -- because `\"` and `'` are the same failure to a reader scanning the line.", + "expect": { + "valid": false, + "comments": [], + "diagnostics": [ + { + "code": "unterminated-string", + "start": 5, + "end": 31 + } + ] + } + }, + { + "id": "parity-unterminated-message-go-rune", + "language": "go", + "operation": "scan", + "options": { + "policy": "safe", + "layout": "lines" + }, + "source_utf8": "r := 'x // not a comment\n", + "note": "Go specification, Rune literals: a rune literal may not contain a newline either, and it carries the same message as the string literal above, so the pair cannot drift apart.", + "expect": { + "valid": false, + "comments": [], + "diagnostics": [ + { + "code": "unterminated-string", + "start": 5, + "end": 24 + } + ] + } + }, + { + "id": "parity-unterminated-message-css", + "language": "css", + "operation": "scan", + "options": { + "policy": "safe", + "layout": "lines" + }, + "source_utf8": "a::before { content: \"unclosed\n", + "note": "CSS Syntax Module Level 3, 4.3.5: consuming a string that reaches the end of input is a parse error. The diagnostic names the language as well as the construct -- `unterminated CSS string` -- because a CSS string is also what an HTML `\n\n", + "note": "@vue/compiler-sfc 3.5 accepts a `", "", + "
", "
", "", +] +SHELL_STRUCTURE = ["<-", "|2-", "|+", "\n ", "%YAML 1.2", "...", + "key:\n", "\n-\n", "!!str ", "&a ", +] + +# NOTE: The shapes that reach PHP mode at all. Nothing but a whole `", "", "// ReSharper disable once X", "// csharpier-ignore", +] + +SCALA_STRUCTURE = [ + "s\"", "f\"", "raw\"", "xml\"", "\"\"\"", "\"\"\"\"", "\"\"\"\"\"", "$$", "$\"", + "${", "$x", "$_", "`a//b`", "//> using scala ", "//> using", "return\"", + "", "", "", "", "", "", "x <", + "> <", "\\u0022", "'c'", "'/", "'sym", +] + +VUE_STRUCTURE = [ + "{{", "}}", "", "", "", + "", 'lang="ts"', 'lang="scss"', 'lang="less"', "v-pre", + "", ":title=", "// text", "/* c */", +] + +SVELTE_STRUCTURE = [ + "{", "}", "{#if", "{/if}", "{#each", "{/each}", "", + "", "", 'lang="ts"', 'lang="scss"', + "

", "

", "title=", "// text", "/* c */", +] + +SCSS_STRUCTURE = [ + "//", "/*", "*/", "#{", "}", "url(", ")", "$x:", 'content: "', "//cdn/", + "// c", "/* c */", "a {", "}", +] + +PERL_STRUCTURE = [ + "#", "q{", "q}", "qq{", "qw(", "qx`", "m/", "s/", "tr{", "y{", "/", + "<<'EOF'", "<", + "$x", "@y", "%h", "# not", "#c", "\\", "(" , ")", "{", "}", "[", "]", +] + +# NOTE: The bytes a lexer is liable to mishandle: NUL, DEL, a byte order mark, a +# NOTE: no-break space, the two Unicode line terminators, and two characters +# NOTE: wider than one byte. +AWKWARD_BYTES = [ + "\x00", "\x7f", "", " ", "
", "
", "é", "中", +] + +TOKENS = ( + COMMENT_MARKERS + + LINE_STRUCTURE + + QUOTES_AND_ESCAPES + + CODE_BYTES + + TRANSLATION_PHASE + + DIRECTIVE_WORDS + + MARKUP + + SHELL_STRUCTURE + + YAML_STRUCTURE + + PHP_STRUCTURE + + RUBY_STRUCTURE + + ZIG_STRUCTURE + + R_STRUCTURE + + DART_STRUCTURE + + SWIFT_STRUCTURE + + CSHARP_STRUCTURE + + SCALA_STRUCTURE + + VUE_STRUCTURE + + SVELTE_STRUCTURE + + SCSS_STRUCTURE + + PERL_STRUCTURE + + AWKWARD_BYTES +) + +INVALID_UTF8 = [b"\xff", b"\xc0\xa0", b"\xed\xa0\x80", b"\xf5\x80\x80\x80"] + +OPERATIONS = ["scan", "transform"] +POLICIES = ["safe", "all", "legal"] +LAYOUTS = ["lines", "columns", "compact"] + + +def built_in_languages(): + """The languages and dialects `spec/languages.toml` declares, so the sweep + cannot fall behind a language someone adds.""" + table = tomllib.loads(LANGUAGES.read_text(encoding="utf-8")) + return [(entry["name"], entry["dialects"]) for entry in table["languages"]] + + +def random_tokens(rng): + """One source as the list of tokens it was built from, so it can be shrunk.""" + tokens = [rng.choice(TOKENS) for _ in range(rng.randint(1, 24))] + if rng.random() < 0.1: + tokens[rng.randrange(len(tokens))] = rng.choice(INVALID_UTF8) + return tokens + + +def assemble(tokens): + """The bytes a token list stands for.""" + return b"".join( + token if isinstance(token, bytes) else token.encode("utf-8") for token in tokens + ) + + +def request(identifier, language, source, options, operation): + """One protocol request, in the shape `spec/differential-protocol.md` fixes.""" + return { + "id": identifier, + "operation": operation, + "language": language, + "source_base64": base64.b64encode(source).decode(), + "options": options, + } + + +def random_options(rng, dialects): + """Policy knobs for one request; `layout` is read only by `transform`.""" + options = {"policy": rng.choice(POLICIES), "layout": rng.choice(LAYOUTS)} + if len(dialects) > 1: + options["dialect"] = rng.choice(dialects) + if rng.random() < 0.1: + options["force_invalid"] = True + if rng.random() < 0.1: + options["force_protected"] = True + return options + + +def run(executable, requests): + """Feed a batch to one implementation and parse the responses.""" + payload = "".join(json.dumps(item, separators=(",", ":")) + "\n" for item in requests) + completed = subprocess.run( + [str(executable)], input=payload, text=True, capture_output=True, check=True + ) + return [json.loads(line) for line in completed.stdout.splitlines()] + + +def compare(requests): + """Every request of the batch whose two responses differ, as + `(request, rust, ocaml)`.""" + rust = run(RUST, requests) + ocaml = run(OCAML, requests) + if not len(rust) == len(ocaml) == len(requests): + raise SystemExit( + f"response count {len(rust)} (rust) vs {len(ocaml)} (ocaml) " + f"for {len(requests)} request(s)" + ) + return [ + (item, left, right) + for item, left, right in zip(requests, rust, ocaml) + if left != right + ] + + +def difference_paths(left, right, prefix=""): + """Where two responses differ, as dotted paths with list indices collapsed. + + Collapsing the indices is what makes a signature: the same bug reached from + a hundred sources names the same fields, however many comments happened to + precede the one it went wrong on. + """ + if type(left) is not type(right): + return [f"{prefix}:type"] + if isinstance(left, dict): + paths = [] + for key in sorted(set(left) | set(right)): + if key not in left or key not in right: + paths.append(f"{prefix}.{key}:absent") + else: + paths.extend(difference_paths(left[key], right[key], f"{prefix}.{key}")) + return paths + if isinstance(left, list): + if len(left) != len(right): + return [f"{prefix}:length"] + paths = [] + for item, other in zip(left, right): + paths.extend(difference_paths(item, other, f"{prefix}[]")) + return sorted(set(paths)) + return [] if left == right else [prefix] + + +def signature(language, left, right): + """What tells one divergence from another: the language, the fields that + disagree, and -- where the field is a message or a kind -- the two values, + because those name the rule that went wrong.""" + paths = tuple(sorted(set(difference_paths(left, right)))) + named = [] + for path in paths: + if path.endswith((".message", ".kind", ".code", ".action", ".reason")): + named.append((path, extract(left, path), extract(right, path))) + return (language, paths, tuple(named)) + + +def extract(value, path): + """The value at a dotted path, or `None` where the path does not lead + anywhere -- a list index was collapsed, or a key is absent.""" + for step in path.lstrip(".").split("."): + if step.endswith("[]"): + step = step[:-2] + if not isinstance(value, dict) or step not in value: + return None + value = value[step] + if not isinstance(value, list) or not value: + return None + value = value[0] + elif isinstance(value, dict) and step in value: + value = value[step] + else: + return None + return value + + +def shrink(item, tokens, language, budget): + """A shorter token list that still diverges, by dropping one token at a time. + + Greedy and bounded per signature: a divergence is meant to be rare, and the + point of the repro is to be short enough to paste into a fixture `note`, + not to be minimal. Every pass over the list is repeated until one removes + nothing, so a token that only became removable after its neighbour went is + still reached. + """ + current = list(tokens) + probes = budget + changed = True + while changed and probes > 0: + changed = False + index = 0 + while index < len(current) and probes > 0: + candidate = current[:index] + current[index + 1 :] + if not candidate: + break + probes -= 1 + probe = request(item["id"], language, assemble(candidate), item["options"], + item["operation"]) + if compare([probe]): + current = candidate + changed = True + else: + index += 1 + # NOTE: The two answers are read back from the shrunken source, so what the + # NOTE: report prints is what the source it prints really produces. + probe = request(item["id"], language, assemble(current), item["options"], + item["operation"]) + _, left, right = compare([probe])[0] + return assemble(current), left, right + + +def sweep(seed, cases, languages): + """Every divergence one seed turns up, keyed by signature.""" + rng = random.Random(seed) + requests = [] + sources = {} + for language, dialects in languages: + for index in range(cases): + tokens = random_tokens(rng) + identifier = f"{language}-{seed}-{index}" + sources[identifier] = tokens + requests.append( + request( + identifier, + language, + assemble(tokens), + random_options(rng, dialects), + rng.choice(OPERATIONS), + ) + ) + return requests, sources, compare(requests) + + +def main(argv): + parser = argparse.ArgumentParser( + description="Fuzz the Rust engine against the OCaml reference.", + epilog="An on-demand check. What it finds belongs in spec/fixtures/v1/hazards.json.", + ) + parser.add_argument( + "--seed", type=int, action="append", metavar="N", + help="a seed to sweep with; repeat for more than one (default: 1)", + ) + parser.add_argument( + "--cases", type=int, default=DEFAULT_CASES, metavar="N", + help=f"random sources per language per seed (default: {DEFAULT_CASES})", + ) + parser.add_argument( + "--shrink-probes", type=int, default=400, metavar="N", + help="single-token removals the shrinker may try per signature (default: 400)", + ) + arguments = parser.parse_args(argv) + for executable in (RUST, OCAML): + if not executable.exists(): + parser.error(f"{executable.relative_to(ROOT)} is not built; see the module docstring") + seeds = arguments.seed or [1] + languages = built_in_languages() + + found = {} + total = 0 + diverged = 0 + for seed in seeds: + requests, sources, divergences = sweep(seed, arguments.cases, languages) + total += len(requests) + diverged += len(divergences) + for item, left, right in divergences: + key = signature(item["language"], left.get("ok", left), right.get("ok", right)) + if key in found and len(assemble(sources[item["id"]])) >= len(found[key][0]): + continue + repro, shrunk_left, shrunk_right = shrink( + item, sources[item["id"]], item["language"], arguments.shrink_probes + ) + if key not in found or len(repro) < len(found[key][0]): + found[key] = (repro, item, shrunk_left, shrunk_right) + + print( + f"{total} request(s) over {len(seeds)} seed(s) x {len(languages)} language(s) " + f"x {arguments.cases} source(s); {diverged} divergent, " + f"{len(found)} distinct signature(s)" + ) + for key, (repro, item, left, right) in sorted(found.items(), key=lambda entry: str(entry[0])): + print("=" * 78) + print(f"language={item['language']} operation={item['operation']} options={item['options']}") + print(f"fields ={' '.join(key[1])}") + print(f"source ={repro!r}") + print(f"rust ={json.dumps(left.get('ok', left), sort_keys=True)[:800]}") + print(f"ocaml ={json.dumps(right.get('ok', right), sort_keys=True)[:800]}") + return 1 if found else 0 + + +if __name__ == "__main__": + raise SystemExit(main(sys.argv[1:])) diff --git a/tools/gen_docs.py b/tools/gen_docs.py new file mode 100644 index 0000000..01733dc --- /dev/null +++ b/tools/gen_docs.py @@ -0,0 +1,1215 @@ +#!/usr/bin/env python3 +"""Regenerate the documentation pages that are derived rather than written. + +Four pages under `docs/` restate something this repository already defines: the +CLI's own `--help` text, the shared language table, what each policy and layout +does to a fixed sample, and which markers survive a removal. A page like that is +stale the moment the thing it restates changes, and a stale page is worse than +no page at all, so it is generated here and `--check` fails the build when the +checked-in bytes and the regenerated bytes differ. + +`--check` also compares `docs/ocomment.1` with the page ` man` renders. +A Rust test pins the same file; both are kept, because that test proves the +shipped manual page still matches the parser, while this proves the +documentation site ships the page that was shipped. + +Every example comes out of the built binary or out of `spec/`, never out of +prose, so nothing on the generated pages can claim behaviour the binary does not +have. Only the standard library is used, because this runs in a job that +installs nothing beyond the toolchain. +""" + +from __future__ import annotations + +import argparse +import difflib +import itertools +import json +import os +import pathlib +import re +import subprocess +import tempfile +import tomllib + + +ROOT = pathlib.Path(__file__).resolve().parents[1] +DOCS = ROOT / "docs" +LANGUAGES = ROOT / "spec/languages.toml" +DIRECTIVES = ROOT / "spec/directives.toml" +MAN_PAGE = DOCS / "ocomment.1" +DEFAULT_BINARIES = ( + ROOT / "rust/target/debug/ocomment", + ROOT / "rust/target/release/ocomment", +) + +# INVARIANT: One sample for every name in the `protected` list of +# INVARIANT: `spec/directives.toml`, written the way a project really writes it +# INVARIANT: and holding exactly one comment. `protected_table` fails when the +# INVARIANT: spec names a marker that has no sample here, so a marker added to +# INVARIANT: the shared spec cannot reach a release undocumented, and it fails +# INVARIANT: again when the binary does not in fact keep one, so the table can +# INVARIANT: never promise a protection that is not there. +# NOTE: `tools/check_directives.py` proves the same names are protected, with +# NOTE: negative controls this page has no use for. The two lists are kept +# NOTE: apart on purpose: each is checked against the shared spec, and a docs +# NOTE: generator that imported a checker would fail for two different reasons +# NOTE: at once. +PROTECTED_SAMPLES: dict[str, tuple[str, str | None, bytes]] = { + "shebang": ("shell", None, b"#!/bin/sh\n"), + "encoding": ("python", None, b"# -*- coding: utf-8 -*-\n"), + "go:": ("go", None, b"//go:build linux\n"), + "+build": ("go", None, b"// +build linux\n"), + "triple-slash-reference": ( + "typescript", + None, + b'/// \n', + ), + "sourceMappingURL": ("javascript", None, b"//# sourceMappingURL=bundle.js.map\n"), + "sourceURL": ("javascript", None, b"//# sourceURL=bundle.js\n"), + "#__PURE__": ("javascript", None, b"const value = /*#__PURE__*/ factory();\n"), + "@__PURE__": ("javascript", None, b"const value = /*@__PURE__*/ factory();\n"), + "lint-and-formatter": ( + "javascript", + None, + b"// eslint-disable-next-line no-eval\n", + ), + "type-checker": ("python", None, b"value = 1 # type: ignore\n"), + "optimizer-hint": ("sql", "oracle", b"select /*+ index(t) */ 1 from dual;\n"), + "version-comment": ("sql", "mysql", b"/*!40101 SET NAMES utf8 */\n"), + "syntax=": ("shell", None, b"# syntax=docker/dockerfile:1\n"), + "hadolint": ("shell", None, b"# hadolint ignore=DL3018\n"), + ":schema": ("toml", None, b"#:schema https://example.test/pyproject.json\n"), + "taplo:": ("toml", None, b"# taplo: array_auto_expand = false\n"), + "---@diagnostic": ( + "lua", + None, + b"---@diagnostic disable-next-line: undefined-global\n", + ), + "luacheck:": ("lua", None, b"-- luacheck: ignore 212\n"), + "selene:": ("lua", None, b"-- selene: allow(unused_variable)\n"), + "stylua:": ("lua", None, b"-- stylua: ignore\n"), + "luacov:": ("lua", None, b"-- luacov: disable\n"), + "yaml-language-server:": ( + "yaml", + None, + b"# yaml-language-server: $schema=https://example.test/schema.json\n", + ), + "yamllint": ("yaml", None, b"# yamllint disable-line rule:line-length\n"), + "renovate:": ("yaml", None, b"# renovate: datasource=docker depName=alpine\n"), + "checkov:skip": ("yaml", None, b"# checkov:skip=CKV_AWS_20:public by design\n"), + "trivy:ignore": ("yaml", None, b"# trivy:ignore:AVD-AWS-0089\n"), + "nosec": ("yaml", None, b"# nosec\n"), + "kics-scan": ("yaml", None, b"# kics-scan ignore-line\n"), + "@schema": ("yaml", None, b"# @schema type: string\n"), + "phpcs:": ("php", None, b"\n"), + "ReSharper": ("csharp", None, b"// ReSharper disable once UnusedMember.Local\n"), + "csharpier-ignore": ("csharp", None, b"// csharpier-ignore\n"), + "//> using": ("scala", None, b'//> using scala "3.3.0"\n'), +} + +BANNER = ( + "" +) +REGENERATE = "python3 tools/gen_docs.py" + + +class Cli: + """The built binary, run with an environment that cannot vary by machine. + + `HOME` and `XDG_CONFIG_HOME` are pointed at an empty directory so a user + configuration file on the machine running this cannot reach the examples, + and `NO_COLOR` is set so no run can escape into the page. + """ + + def __init__(self, binary: pathlib.Path, home: pathlib.Path) -> None: + self.binary = binary + self.environment = dict(os.environ) + self.environment.update( + { + "NO_COLOR": "1", + "TERM": "dumb", + "HOME": str(home), + "XDG_CONFIG_HOME": str(home / "config"), + } + ) + + def run( + self, + arguments: list[str], + *, + source: bytes | None = None, + cwd: pathlib.Path | None = None, + expected: tuple[int, ...] = (0,), + ) -> subprocess.CompletedProcess[bytes]: + """Run the binary and refuse to document an exit code nobody expected.""" + completed = subprocess.run( + [str(self.binary), *arguments], + input=source, + cwd=None if cwd is None else str(cwd), + capture_output=True, + env=self.environment, + ) + if completed.returncode not in expected: + raise SystemExit( + f"`ocomment {' '.join(arguments)}` exited {completed.returncode}," + f" expected one of {expected}\n" + + completed.stderr.decode("utf-8", "replace") + ) + return completed + + def text(self, arguments: list[str], **keywords) -> str: + """The standard output of one run, decoded.""" + return self.run(arguments, **keywords).stdout.decode("utf-8") + + +def fence(language: str, body: str) -> str: + """A fenced block that always ends in exactly one newline.""" + return f"```{language}\n{body.rstrip(chr(10))}\n```" + + +def anchor(heading: str) -> str: + """The identifier mdBook gives a heading, so a link to it can be built.""" + return re.sub(r"[^a-z0-9]+", "-", heading.lower()).strip("-") + + +def cell(text: str) -> str: + """One table cell: a pipe inside it would end the column early.""" + return text.replace("|", "\\|") + + +def count(number: int, singular: str, plural: str | None = None) -> str: + """`1 file` and `2 files`, so the generated prose is never `1 files`.""" + return f"{number} {singular if number == 1 else (plural or singular + 's')}" + + +def commands_in(help_text: str) -> list[str]: + """The subcommand names one `--help` page lists, in the order it lists them.""" + names: list[str] = [] + lines = help_text.splitlines() + if "Commands:" not in lines: + return names + for line in lines[lines.index("Commands:") + 1 :]: + if not line.strip(): + break + match = re.match(r"^ {2}([a-z][a-z0-9-]*)(?:\s|$)", line) + # NOTE: `help` prints the same page as `--help` and takes no options of + # NOTE: its own, so documenting it would repeat every block on the page. + if match is not None and match.group(1) != "help": + names.append(match.group(1)) + return names + + +def command_paths(cli: Cli, path: tuple[str, ...] = ()) -> list[tuple[str, ...]]: + """Every command the binary offers, parents before children. + + The list is discovered rather than written down, so a new subcommand is + documented by the same change that adds it. + """ + found = [path] + for name in commands_in(cli.text([*path, "--help"])): + found.extend(command_paths(cli, (*path, name))) + return found + + +COMMANDS_INTRO = """\ +# Commands + +Each block below is the exact `--help` text of the binary this page was +generated from, so a flag documented here is a flag the tool has, and a flag +that is missing here does not exist. + +`ocomment` with no command is `ocomment check`, and a command with no path is +the current directory. Findings, patches, listings, and every machine format go +to standard output; the run summary and every note go to standard error, so +`ocomment diff src > fix.patch` keeps the patch clean. + +`check` exits `0` when nothing removable was found, `1` when removable comments +were reported or a diff was printed, and `2` for an invalid source, +configuration, plugin, or I/O failure. + +The `Policy` and `Output` option groups are global. Every command that can act +on them accepts them, which is why the same two groups appear under most of the +blocks below. `ocomment man` renders the same material as a manual page.""" + + +def commands_page(cli: Cli) -> str: + """`docs/commands.md`: one `--help` block for every command, discovered.""" + paths = command_paths(cli) + titles = [" ".join(("ocomment", *path)) for path in paths] + lines = [BANNER, "", COMMANDS_INTRO, "", "## Contents", ""] + lines.extend(f"- [`{title}`](#{anchor(title)})" for title in titles) + for path, title in zip(paths, titles): + help_text = cli.text([*path, "--help"]) + lines.extend( + ["", f"## `{title}`", "", fence("console", f"$ {title} --help\n{help_text}")] + ) + return "\n".join(lines) + "\n" + + +LANGUAGES_INTRO = """\ +# Languages and dialects + +`spec/languages.toml` is the canonical table. The `files:` pattern of the +published pre-commit hooks is generated from it, `tools/check_hooks.py` fails +when the two drift apart, and the table below is generated from the same file. +The binary embeds that same file, so `ocomment languages` prints these rows in +columns and `ocomment languages --format json` prints them as JSON. + +A dialect changes the lexical rules rather than the file type: `--dialect +mysql` is still SQL, and only that dialect treats `/*!40101 ... */` as +something the server executes rather than as a comment. + +A language is chosen from the file extension, and `--language` overrides that +for a run — which is what `ocomment strip` needs, because standard input has no +name. `--dialect` picks the dialect for the same run, and `[languages.] +dialect = "..."` in `.ocomment.toml` picks one for everybody working in the +repository. An incompatible pair is an error rather than a silent fallback: + +```console +$ ocomment strip --language rust --dialect mysql +ocomment: unsupported dialect `mysql` for rust; supported: standard +``` +""" + + +NAMED_INTRO = """\ +A file whose extension decides nothing is looked up by its whole name, and a +file with no name at all — a script on standard input — is read from its `#!` +line. A name is matched without regard to case, and a shebang matches when the +interpreter name appears anywhere on the line.""" + + +def languages_page() -> str: + """`docs/languages.md`: the shared language table, rendered.""" + with LANGUAGES.open("rb") as stream: + table = tomllib.load(stream) + entries = table["languages"] + extensions = sorted({item for entry in entries for item in entry["extensions"]}) + dialects = sorted({item for entry in entries for item in entry["dialects"]}) + lines = [ + BANNER, + "", + LANGUAGES_INTRO, + f"OComment has {count(len(entries), 'built-in language')} covering", + f"{count(len(extensions), 'file extension')} and" + f" {count(len(dialects), 'named dialect')}.", + "", + "| Language | Extensions | Dialects |", + "| --- | --- | --- |", + ] + for entry in entries: + dialect_of = entry.get("extension_dialects", {}) + suffixes = ", ".join( + f"`.{item}` (`{dialect_of[item]}`)" if item in dialect_of else f"`.{item}`" + for item in entry["extensions"] + ) + forms = ", ".join(f"`{item}`" for item in entry["dialects"]) + lines.append(f"| `{cell(entry['name'])}` | {cell(suffixes)} | {cell(forms)} |") + lines.extend(["", "## Detected without an extension", ""]) + lines.extend(NAMED_INTRO.splitlines()) + lines.extend(["", "| Language | File names | Shebangs |", "| --- | --- | --- |"]) + for entry in entries: + names = ", ".join(f"`{item}`" for item in entry.get("reserved_names", ())) + shebangs = ", ".join(f"`{item}`" for item in entry.get("shebangs", ())) + if not names and not shebangs: + continue + lines.append( + f"| `{cell(entry['name'])}` | {cell(names) or '—'} | {cell(shebangs) or '—'} |" + ) + lines.extend( + [ + "", + "## Anything else", + "", + "HTML is scanned recursively: the contents of a `