diff --git a/.github/actions/check-new-seeds/action.yml b/.github/actions/check-new-seeds/action.yml new file mode 100644 index 000000000..1c15fdbe0 --- /dev/null +++ b/.github/actions/check-new-seeds/action.yml @@ -0,0 +1,27 @@ +name: Verify new seeds +description: Run libfuzzer -merge=1 over (baseline + new) seeds for each harness with newly-added seeds in the PR. Hard-fail if any new seed is redundant. + +inputs: + project: + description: Project name. + required: true + base_sha: + description: Git base SHA to diff against (typically merge-base with the default branch). + required: true + out_base: + description: Directory containing the per-project oss-fuzz out tree (e.g. build/out, /tmp/oss-fuzz/build/out). + required: true + seeds_dir: + description: Workspace-relative seeds root grouped by harness (e.g. projects//seeds, .github/fuzz/seeds). + required: true + +runs: + using: composite + steps: + - shell: bash + env: + PROJECT: ${{ inputs.project }} + BASE_SHA: ${{ inputs.base_sha }} + OUT_BASE: ${{ inputs.out_base }} + SEEDS_DIR: ${{ inputs.seeds_dir }} + run: python3 "$GITHUB_ACTION_PATH/check_new_seeds.py" diff --git a/.github/actions/check-new-seeds/check_new_seeds.py b/.github/actions/check-new-seeds/check_new_seeds.py new file mode 100644 index 000000000..3e7d5bfbf --- /dev/null +++ b/.github/actions/check-new-seeds/check_new_seeds.py @@ -0,0 +1,330 @@ +# This file ships into oss-fuzz review-repo PRs via the check-new-seeds +# composite action; oss-fuzz's `infra/presubmit.py` check_license greps every +# Python file under non-projects/ paths for the Apache 2.0 LICENSE-2.0 URL, +# so we carry the standard header here even though it's our own infra code. +# Copyright 2026 fuzz-for-me contributors +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. +"""Verify newly-added fuzz seeds aren't redundant given the baseline corpus. + +For each harness whose seed dir got new files in this PR, run libfuzzer's +`-merge=1` over (baseline + new) and check every new seed survived. A seed +is redundant if any of: + - its content already exists in baseline (exact duplicate of preexisting), + - its content matches another newly-added seed (dup within the PR), + - libfuzzer's merge dropped it (covers no new edges beyond what's loaded). + +Hard-fails (exit 1) with a per-seed reason and a copy-paste reproducer. + +Inputs (env vars): + PROJECT project name (used for OUT_BASE//) + BASE_SHA git SHA to diff against (typically merge-base with main/master) + OUT_BASE dir containing built harness binaries + (e.g. build/out for oss-fuzz, /tmp/oss-fuzz/build/out for upstream) + SEEDS_DIR workspace-relative seeds root, grouped by harness + (e.g. projects//seeds for oss-fuzz, .github/fuzz/seeds for upstream) + BASE_RUNNER (optional) override base-runner image, default + gcr.io/oss-fuzz-base/base-runner +""" + +import hashlib +import os +import pathlib +import shutil +import subprocess +import sys +import tempfile +from collections import defaultdict + +DEFAULT_BASE_RUNNER = "gcr.io/oss-fuzz-base/base-runner" + + +def file_sha(p: pathlib.Path) -> str: + return hashlib.sha1(p.read_bytes()).hexdigest() + + +def hashes_in(d: pathlib.Path) -> set[str]: + return {file_sha(p) for p in d.iterdir() if p.is_file()} + + +def added_paths(base_sha: str, prefix: str) -> list[str]: + """Files added in commits between base_sha and HEAD, restricted to prefix/.""" + result = subprocess.run( + [ + "git", + "diff", + "--name-only", + "--diff-filter=A", + f"{base_sha}...HEAD", + "--", + prefix, + ], + capture_output=True, + text=True, + check=True, + ) + return [line for line in result.stdout.splitlines() if line] + + +def group_by_harness(paths: list[str], seeds_dir: str) -> dict[str, list[str]]: + """Group seed paths by their harness subdir. + + e.g. projects/foo/seeds/parser/a.in -> {"parser": ["projects/foo/seeds/parser/a.in"]} + """ + prefix = seeds_dir.rstrip("/") + "/" + groups: dict[str, list[str]] = defaultdict(list) + for p in paths: + if not p.startswith(prefix): + continue + rest = p[len(prefix):] + parts = rest.split("/", 1) + if len(parts) != 2: # seed file directly under seeds/, no harness subdir + continue + groups[parts[0]].append(p) + return groups + + +def docker_merge( + image: str, + out_dir: pathlib.Path, + harness: str, + minimized: pathlib.Path, + baseline: pathlib.Path, + new: pathlib.Path, +) -> subprocess.CompletedProcess: + """Run ` -merge=1 /minimized /baseline /new` inside the base-runner.""" + cmd = [ + "docker", + "run", + "--rm", + "--privileged", + "-e", + "FUZZING_ENGINE=libfuzzer", + "-e", + "SANITIZER=address", + "-e", + "ARCHITECTURE=x86_64", + "-e", + "OUT=/out", + "-v", + f"{out_dir.resolve()}:/out:ro", + "-v", + f"{baseline.resolve()}:/baseline:ro", + "-v", + f"{new.resolve()}:/new:ro", + "-v", + f"{minimized.resolve()}:/minimized", + "--entrypoint", + "/out/" + harness, + image, + "-merge=1", + "/minimized", + "/baseline", + "/new", + ] + return subprocess.run(cmd, capture_output=True, text=True) + + +def check_harness( + harness: str, + out_dir: pathlib.Path, + seeds_root: pathlib.Path, + new_names: set[str], + image: str, + work_root: pathlib.Path, +) -> tuple[list[tuple[str, str]], str]: + """Returns (redundant_seeds, debug_log). + + redundant_seeds: list of (filename, reason) for each redundant new seed. + debug_log: stdout+stderr of the merge run (always returned for diagnostics). + """ + harness_seeds = seeds_root / harness + if not harness_seeds.is_dir(): + return [], f"{harness}: seeds dir {harness_seeds} missing — skipping" + + binary = out_dir / harness + if not binary.is_file(): + return [], ( + f"{harness}: harness binary not built ({binary} missing) — skipping. " + "If this is unexpected, check the build_fuzzers step.") + + work = work_root / harness + baseline = work / "baseline" + new = work / "new" + minimized = work / "minimized" + for d in (baseline, new, minimized): + d.mkdir(parents=True, exist_ok=True) + + for f in harness_seeds.iterdir(): + if not f.is_file(): + continue + dst = (new if f.name in new_names else baseline) / f.name + shutil.copyfile(f, dst) + + baseline_hashes = hashes_in(baseline) + + proc = docker_merge(image, out_dir, harness, minimized, baseline, new) + log = ( + f"$ {' '.join(['', '-merge=1', '/minimized', '/baseline', '/new'])}\n" + f"--- exit {proc.returncode} ---\n" + f"--- stdout ---\n{proc.stdout}\n" + f"--- stderr ---\n{proc.stderr}") + if proc.returncode != 0: + return [(harness, f"libfuzzer merge failed (exit {proc.returncode})")], log + + minimized_hashes = hashes_in(minimized) + + # Detect merge-incompatible harnesses. libFuzzer's -merge=1 keeps a minimal + # covering set; with an empty baseline, *any* input that executes and records + # features yields a non-empty `minimized`. If the merge succeeds but keeps + # ZERO files even though there was ≥1 input, libFuzzer never recorded + # per-input features — i.e. the harness doesn't return cleanly to libFuzzer + # between inputs so the merge control file gets no FT lines. (u-boot's + # sandbox fuzzer is the canonical case: it runs the target on a separate + # thread via a coroutine handoff and os_abort()s when sandbox_main returns, + # killing the process before libFuzzer's merge bookkeeping completes.) + # Flagging every seed "redundant" here would be a false positive, so only + # report true byte-duplicates and skip the edge-coverage verdict. + inputs_present = any(baseline.iterdir()) or any(new.iterdir()) + merge_recorded_nothing = not minimized_hashes + incompatible = inputs_present and merge_recorded_nothing + + redundant: list[tuple[str, str]] = [] + seen_in_new: set[str] = set() + for src in sorted(new.iterdir()): + h = file_sha(src) + if h in baseline_hashes: + redundant.append((src.name, "duplicate of an existing baseline seed")) + elif h in seen_in_new: + redundant.append((src.name, "duplicate of another newly-added seed")) + elif incompatible: + # Not byte-duplicate; can't judge edge-coverage on a merge-incompatible + # harness — accept it. + seen_in_new.add(h) + elif h not in minimized_hashes: + redundant.append( + (src.name, "covers no new edges beyond baseline + earlier new seeds")) + else: + seen_in_new.add(h) + + if incompatible: + log += ( + "\n--- note ---\n" + "libFuzzer -merge=1 kept 0 files despite ≥1 input seed; the harness " + "does not return cleanly to libFuzzer between inputs (merge control " + "file has no FT lines), so per-seed edge-coverage cannot be verified. " + "Skipping the redundancy verdict for this harness; only exact " + "byte-duplicate seeds are reported.") + + return redundant, log + + +def _emit(line: str = ""): + print(line, flush=True) + + +def report_failure( + findings: list[tuple[str, list[tuple[str, str]], str]], + seeds_dir: str, + out_base: str, + project: str, +): + """Print a GH-Actions error block + per-harness fix instructions.""" + total = sum(len(rs) for _, rs, _ in findings) + _emit(f"::error::{total} redundant new seed(s) — see details below.") + _emit() + _emit("=" * 70) + _emit("Redundant new seeds detected") + _emit("=" * 70) + _emit() + _emit("libfuzzer's -merge=1 keeps only inputs that add new code coverage. " + "Each seed below either duplicated an existing seed or covered nothing " + "new on top of the corpus loaded before it.") + _emit() + for harness, redundant, log in findings: + _emit(f"### {harness}") + for name, reason in redundant: + _emit(f" - {seeds_dir}/{harness}/{name}") + _emit(f" -> {reason}") + _emit() + _emit(" merge log (last 20 lines of stderr):") + tail = "\n".join(log.splitlines()[-20:]) + for line in tail.splitlines(): + _emit(f" | {line}") + _emit() + _emit("=" * 70) + _emit("How to make CI go green") + _emit("=" * 70) + _emit() + _emit("Pick one per redundant file:") + _emit(" (a) Delete the file from the PR.") + _emit(" (b) Replace its content with input that exercises a new code path") + _emit(" (different parser branch, new field, edge value, etc.).") + _emit() + _emit("To reproduce locally on a built project:") + _emit(f" cd {out_base}/{project}") + _emit(" mkdir -p baseline new minimized") + _emit( + " # populate baseline/ with preexisting seeds, new/ with the added ones") + _emit(" ./ -merge=1 minimized/ baseline/ new/") + _emit( + " # any seed in new/ whose content-hash is not in minimized/ is redundant." + ) + + +def main() -> int: + project = os.environ["PROJECT"] + base_sha = os.environ["BASE_SHA"] + out_base = os.environ["OUT_BASE"] + seeds_dir = os.environ["SEEDS_DIR"] + image = os.environ.get("BASE_RUNNER", DEFAULT_BASE_RUNNER) + + added = added_paths(base_sha, seeds_dir) + if not added: + _emit(f"No new seed files under {seeds_dir} in this PR — skipping check.") + return 0 + + groups = group_by_harness(added, seeds_dir) + if not groups: + _emit(f"Seeds added under {seeds_dir} but none in a per-harness subdir — " + "skipping. (Expected layout: //.)") + return 0 + + out_dir = pathlib.Path(out_base) / project + seeds_root = pathlib.Path(seeds_dir) + + findings: list[tuple[str, list[tuple[str, str]], str]] = [] + with tempfile.TemporaryDirectory() as tmp: + tmp_root = pathlib.Path(tmp) + for harness, paths in sorted(groups.items()): + new_names = {pathlib.Path(p).name for p in paths} + _emit(f"::group::Verify new seeds: {harness} ({len(new_names)} new)") + redundant, log = check_harness(harness, out_dir, seeds_root, new_names, + image, tmp_root) + if redundant: + findings.append((harness, redundant, log)) + _emit(f" -> {len(redundant)} redundant") + else: + _emit(" -> all new seeds non-redundant") + _emit("::endgroup::") + + if findings: + report_failure(findings, seeds_dir, out_base, project) + return 1 + + _emit("All new seeds are non-redundant.") + return 0 + + +if __name__ == "__main__": + sys.exit(main()) diff --git a/.github/actions/collect-fuzz-stats/action.yml b/.github/actions/collect-fuzz-stats/action.yml new file mode 100644 index 000000000..cc2af2c8c --- /dev/null +++ b/.github/actions/collect-fuzz-stats/action.yml @@ -0,0 +1,69 @@ +name: Collect fuzz stats +description: Dump coverage summaries (project-wide + per-harness) and corpus file counts into stats-/ for the report job. + +inputs: + project: + description: Project name. + required: true + variant: + description: Variant label — "baseline" or "current". + required: true + sha: + description: Commit SHA measured in this variant. + required: true + has_project: + description: String "true" if the project was present and built for this variant; otherwise only meta.json is written. + required: true + out_base: + description: Directory containing the oss-fuzz out tree (e.g. build/out for oss-fuzz template, /tmp/oss-fuzz/build/out for upstream). + required: true + +runs: + using: composite + steps: + - shell: bash + env: + PROJECT: ${{ inputs.project }} + VARIANT: ${{ inputs.variant }} + SHA: ${{ inputs.sha }} + HAS_PROJECT: ${{ inputs.has_project }} + OUT_BASE: ${{ inputs.out_base }} + run: | + OUT="stats-$VARIANT" + mkdir -p "$OUT/harness" + + jq -n \ + --arg variant "$VARIANT" \ + --arg sha "${SHA:-}" \ + --arg project "$PROJECT" \ + --arg has_project "${HAS_PROJECT:-false}" \ + '{variant: $variant, sha: $sha, project: $project, has_project: ($has_project == "true")}' \ + > "$OUT/meta.json" + + if [ "$HAS_PROJECT" != "true" ]; then + echo "No project at this variant; stats collection skipped" + exit 0 + fi + + PROJ_SUM=$(find "$OUT_BASE/$PROJECT/report/linux" -maxdepth 1 -name summary.json 2>/dev/null | head -1 || true) + if [ -n "$PROJ_SUM" ]; then + cp "$PROJ_SUM" "$OUT/project.summary.json" + fi + + if [ -d "$OUT_BASE/$PROJECT/report_target" ]; then + for d in "$OUT_BASE/$PROJECT/report_target"/*/linux; do + [ -d "$d" ] || continue + HNAME=$(basename "$(dirname "$d")") + [ -f "$d/summary.json" ] && cp "$d/summary.json" "$OUT/harness/$HNAME.summary.json" + done + fi + + # summary.json carries only the harness-inclusive scalar; the per-function + # list is what lets the report exclude harness functions, so ship that. + INSPECTOR_FUNCS="$OUT_BASE/$PROJECT/inspector/all-fuzz-introspector-functions.json" + if [ -f "$INSPECTOR_FUNCS" ]; then + cp "$INSPECTOR_FUNCS" "$OUT/reachability.json" + fi + + echo "---- $OUT ----" + find "$OUT" -type f -printf '%p (%s bytes)\n' diff --git a/.github/actions/fuzz-all/action.yml b/.github/actions/fuzz-all/action.yml new file mode 100644 index 000000000..f43779ddf --- /dev/null +++ b/.github/actions/fuzz-all/action.yml @@ -0,0 +1,48 @@ +name: Fuzz all harnesses +description: Run every libFuzzer harness in $OUT concurrently for fuzz_seconds wall-time, capturing per-harness logs. + +inputs: + project: + description: oss-fuzz project name (or path, in external mode). + required: true + helper_py: + description: Path to oss-fuzz infra/helper.py. + required: true + out_dir: + description: Path to build/out/ with the harness binaries. + required: true + corpus_root: + description: Parent directory for per-harness corpus dirs. + required: true + logs_dir: + description: Directory for per-harness raw fuzz logs. + required: true + fuzz_seconds: + description: Wall-time budget per harness (passed as -max_total_time). + required: true + external: + description: Pass --external to helper.py run_fuzzer (for ClusterFuzzLite-style projects). + required: false + default: 'false' + +runs: + using: composite + steps: + - shell: bash + env: + PROJECT: ${{ inputs.project }} + HELPER_PY: ${{ inputs.helper_py }} + OUT_DIR: ${{ inputs.out_dir }} + CORPUS_ROOT: ${{ inputs.corpus_root }} + LOGS_DIR: ${{ inputs.logs_dir }} + FUZZ_SECONDS: ${{ inputs.fuzz_seconds }} + EXTERNAL_FLAG: ${{ inputs.external == 'true' && '--external' || '' }} + run: | + python3 "$GITHUB_ACTION_PATH/run_fuzzers.py" \ + --helper-py "$HELPER_PY" \ + --project "$PROJECT" \ + --out-dir "$OUT_DIR" \ + --corpus-root "$CORPUS_ROOT" \ + --logs-dir "$LOGS_DIR" \ + --max-total-time "$FUZZ_SECONDS" \ + $EXTERNAL_FLAG diff --git a/.github/actions/fuzz-all/run_fuzzers.py b/.github/actions/fuzz-all/run_fuzzers.py new file mode 100644 index 000000000..8864739d2 --- /dev/null +++ b/.github/actions/fuzz-all/run_fuzzers.py @@ -0,0 +1,159 @@ +#!/usr/bin/env python3 +# This file ships into oss-fuzz review-repo PRs via the fuzz-all composite +# action; oss-fuzz's `infra/presubmit.py` check_license greps every Python +# file under non-projects/ paths for the Apache 2.0 LICENSE-2.0 URL, so we +# carry the standard header here even though it's our own infra code. +# Copyright 2026 fuzz-for-me contributors +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. +"""Run all libFuzzer harnesses for an oss-fuzz project concurrently. + +Shared by CI workflows (the fuzz-verify reusable workflow) and the agent +CLI. stdlib-only so it can run in a stock oss-fuzz CI environment without the +fuzz_for_me package installed. +""" + +import argparse +import os +import shlex +import stat +import subprocess +import sys +from pathlib import Path + +SKIP_SUFFIXES = (".options", ".dict", "_seed_corpus.zip") +SKIP_NAMES = {"llvm-symbolizer", "jazzer_agent_deploy.jar", "jazzer_driver"} + + +def discover_harnesses(out_dir): + out = [] + for entry in sorted(out_dir.iterdir()): + if not entry.is_file(): + continue + if entry.name in SKIP_NAMES: + continue + if any(entry.name.endswith(s) for s in SKIP_SUFFIXES): + continue + if not (entry.stat().st_mode & stat.S_IXUSR): + continue + out.append(entry.name) + return out + + +def build_cmd(args, harness, corpus_dir): + # --out-dir is intentionally NOT passed to helper.py run_fuzzer: not all + # versions of upstream helper.py accept it. The default location + # (oss-fuzz/build/out/) is what build_fuzzers writes to. + cmd = [sys.executable, str(args.helper_py), "run_fuzzer"] + if args.external: + cmd.append("--external") + cmd += ["--corpus-dir", str(corpus_dir), args.project, harness] + if args.max_total_time > 0: + cmd += ["--", f"-max_total_time={args.max_total_time}"] + return cmd + + +def main(): + ap = argparse.ArgumentParser() + ap.add_argument("--helper-py", required=True, type=Path) + ap.add_argument("--project", + required=True, + help="oss-fuzz project name, or path in --external mode") + ap.add_argument("--out-dir", + required=True, + type=Path, + help="build/out/ — where the harness binaries are") + ap.add_argument("--corpus-root", required=True, type=Path) + ap.add_argument("--logs-dir", required=True, type=Path) + ap.add_argument("--max-total-time", + type=int, + default=0, + help="seconds per harness; 0 = unbounded (use --detach)") + ap.add_argument("--external", + action="store_true", + help="pass --external to helper.py") + ap.add_argument("--detach", + action="store_true", + help="launch and exit immediately, printing PIDs") + args = ap.parse_args() + + args.out_dir = args.out_dir.resolve() + args.corpus_root = args.corpus_root.resolve() + args.logs_dir = args.logs_dir.resolve() + args.helper_py = args.helper_py.resolve() + + if not args.detach and args.max_total_time <= 0: + ap.error("--max-total-time must be > 0 unless --detach is given") + + harnesses = discover_harnesses(args.out_dir) + if not harnesses: + print(f"::error::no harness binaries found in {args.out_dir}", + file=sys.stderr) + return 1 + + args.corpus_root.mkdir(parents=True, exist_ok=True) + args.logs_dir.mkdir(parents=True, exist_ok=True) + + print( + f"Harnesses ({len(harnesses)}): {' '.join(harnesses)} — " + f"max_total_time={args.max_total_time}s, detach={args.detach}", + flush=True) + + procs = [] + for h in harnesses: + corpus_dir = args.corpus_root / h + corpus_dir.mkdir(parents=True, exist_ok=True) + log_path = args.logs_dir / f"{h}.log" + log_fp = open(log_path, "w") + cmd = build_cmd(args, h, corpus_dir) + proc = subprocess.Popen(cmd, + stdout=log_fp, + stderr=subprocess.STDOUT, + start_new_session=True) + procs.append((h, proc, log_fp, log_path, corpus_dir)) + + if args.detach: + for h, proc, log_fp, log_path, _ in procs: + log_fp.close() + print(f"{h} pid={proc.pid} log={log_path}") + return 0 + + rc = 0 + timeout = args.max_total_time + 60 + for h, proc, log_fp, _, _ in procs: + try: + proc.wait(timeout=timeout) + except subprocess.TimeoutExpired: + print(f"::warning::{h} exceeded timeout, killing", file=sys.stderr) + proc.kill() + proc.wait() + rc = max(rc, 1) + finally: + log_fp.close() + + for h, proc, _, log_path, corpus_dir in procs: + n_corpus = sum( + 1 for _ in corpus_dir.iterdir()) if corpus_dir.is_dir() else 0 + print(f"::group::{h}", flush=True) + try: + sys.stdout.write(log_path.read_text(errors="replace")) + except OSError as e: + print(f"(failed to read log: {e})") + print(f"\nCorpus: {n_corpus} files (exit={proc.returncode})", flush=True) + print("::endgroup::", flush=True) + + return rc + + +if __name__ == "__main__": + sys.exit(main()) diff --git a/.github/actions/post-fuzz-report/action.yml b/.github/actions/post-fuzz-report/action.yml new file mode 100644 index 000000000..2fa513615 --- /dev/null +++ b/.github/actions/post-fuzz-report/action.yml @@ -0,0 +1,82 @@ +name: Post fuzz coverage report +description: Download baseline/current stats artifacts, render markdown, post or update a sticky PR comment. + +inputs: + footer: + description: Footer flavor — "generic" for oss-fuzz, "upstream" for upstream source PRs. + required: false + default: generic + fuzz_seconds: + description: Total fuzz budget used (for display only). + required: true + github_token: + description: Token with pull-requests:write to post the comment. + required: true + +runs: + using: composite + steps: + - uses: actions/download-artifact@3e5f45b2cfb9172054b4087a40e8e0b5a5461e7c # v8.0.1 + with: + name: stats-baseline + path: stats/baseline + continue-on-error: true + + - uses: actions/download-artifact@3e5f45b2cfb9172054b4087a40e8e0b5a5461e7c # v8.0.1 + with: + name: stats-current + path: stats/current + continue-on-error: true + + # Reachability is produced by a separate parallel job (see the + # `reachability` job). Land its reachability.json next to the other + # stats so render_comment.py finds it at the same path as before — + # no renderer change. Soft: a missing artifact (introspector failed) + # just renders the reachability row as unavailable. + - uses: actions/download-artifact@3e5f45b2cfb9172054b4087a40e8e0b5a5461e7c # v8.0.1 + with: + name: reachability-baseline + path: stats/baseline + continue-on-error: true + + - uses: actions/download-artifact@3e5f45b2cfb9172054b4087a40e8e0b5a5461e7c # v8.0.1 + with: + name: reachability-current + path: stats/current + continue-on-error: true + + - name: Render comment + shell: bash + env: + RUN_URL: ${{ github.server_url }}/${{ github.repository }}/actions/runs/${{ github.run_id }} + FUZZ_SECONDS: ${{ inputs.fuzz_seconds }} + FOOTER: ${{ inputs.footer }} + STATS_ROOT: stats + run: | + python3 "$GITHUB_ACTION_PATH/render_comment.py" > comment.md + echo "---- comment.md ----" + cat comment.md + + - name: Post or update PR comment + shell: bash + env: + GH_TOKEN: ${{ inputs.github_token }} + REPO: ${{ github.repository }} + BRANCH: ${{ github.ref_name }} + run: | + PR=$(gh pr list --repo "$REPO" --head "$BRANCH" --state open --json number --jq '.[0].number // empty') + if [ -z "$PR" ]; then + echo "No open PR for branch $BRANCH — skipping comment" + exit 0 + fi + MARKER='' + EXISTING=$(gh api "repos/$REPO/issues/$PR/comments" --paginate \ + --jq ".[] | select(.body | contains(\"$MARKER\")) | .id" | head -1) + jq -Rs '{body: .}' < comment.md > payload.json + if [ -n "$EXISTING" ]; then + gh api --method PATCH "repos/$REPO/issues/comments/$EXISTING" --input payload.json > /dev/null + echo "Updated existing comment $EXISTING on PR #$PR" + else + gh api --method POST "repos/$REPO/issues/$PR/comments" --input payload.json > /dev/null + echo "Created new comment on PR #$PR" + fi diff --git a/.github/actions/post-fuzz-report/cov.py b/.github/actions/post-fuzz-report/cov.py new file mode 100644 index 000000000..2920e01bb --- /dev/null +++ b/.github/actions/post-fuzz-report/cov.py @@ -0,0 +1,87 @@ +# This file ships into oss-fuzz review-repo PRs via the post-fuzz-report +# composite action (next to render_comment.py); oss-fuzz's +# `infra/presubmit.py` check_license greps every Python file under +# non-projects/ paths for the Apache 2.0 LICENSE-2.0 URL, so we carry the +# standard header here even though it's our own infra code. +# Copyright 2026 fuzz-for-me contributors +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. +"""Single source of truth for "project coverage". + +Project coverage = the aggregate over in-scope source files, *excluding* the +fuzz harnesses. Harness bodies run on every input, so they're ~100% covered by +construction; counting them inflates the headline and is pure noise. llvm-cov's +precomputed ``data[0]["totals"]`` and Fuzz Introspector's +``MergedProjectProfile.stats`` are both harness-inclusive, so the headline must +be re-derived from the per-file / per-function lists instead of trusting those +free aggregates. + +stdlib-only and dependency-free: this module ships standalone next to +``render_comment.py`` into bare CI, and is also imported in-tree by +``manager.py``. +""" + +HARNESS_PREFIX = "/src/harnesses/" + +_METRICS = ("lines", "branches", "functions") + + +def _triple(covered, count): + return (covered, count, (100.0 * covered / count) if count else 0.0) + + +def coverage_totals(summary): + """Harness-excluded totals from an llvm-cov export ``summary.json``. + + Sums the per-file ``summary`` blocks (``data[0]["files"]``) for files not + under ``HARNESS_PREFIX``. Returns ``{metric: (covered, count, pct)}`` for + lines/branches/functions, or ``None`` when no usable data is present. + """ + if not summary: + return None + try: + files = summary["data"][0]["files"] + except (KeyError, IndexError, TypeError): + return None + acc = {m: [0, 0] for m in _METRICS} + for f in files: + if f["filename"].startswith(HARNESS_PREFIX): + continue + s = f["summary"] + for m in _METRICS: + sm = s.get(m) + if sm: + acc[m][0] += sm["covered"] + acc[m][1] += sm["count"] + return {m: _triple(*acc[m]) for m in _METRICS} + + +def reach_totals(functions): + """Harness-excluded static reachability from Fuzz Introspector. + + ``functions`` is the parsed ``all-fuzz-introspector-functions.json`` list + (``summary.json`` itself carries only the harness-inclusive scalar). Counts + entries whose ``Functions filename`` is not under ``HARNESS_PREFIX``; an + entry is "reached" when ``Combined reached by Fuzzers`` is non-empty. + Returns ``{"reach": (reached, total, pct)}`` or ``None``. + """ + if not functions: + return None + reached = total = 0 + for fn in functions: + if fn["Functions filename"].startswith(HARNESS_PREFIX): + continue + total += 1 + if fn["Combined reached by Fuzzers"]: + reached += 1 + return {"reach": _triple(reached, total)} diff --git a/.github/actions/post-fuzz-report/render_comment.py b/.github/actions/post-fuzz-report/render_comment.py new file mode 100644 index 000000000..4b2eace9e --- /dev/null +++ b/.github/actions/post-fuzz-report/render_comment.py @@ -0,0 +1,252 @@ +# This file ships into oss-fuzz review-repo PRs via the post-fuzz-report +# composite action; oss-fuzz's `infra/presubmit.py` check_license greps every +# Python file under non-projects/ paths for the Apache 2.0 LICENSE-2.0 URL, +# so we carry the standard header here even though it's our own infra code. +# Copyright 2026 fuzz-for-me contributors +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. +"""Render before/after coverage comparison as a sticky PR comment body. + +Reads artifacts downloaded by the calling workflow: + stats/baseline/meta.json, project.summary.json, corpus.json, harness/*.summary.json + stats/current/ ... (same layout) + +Writes the markdown body to stdout. +""" + +import datetime +import json +import os +import pathlib +import sys + +try: + # In-tree (tests, host package). + from fuzz_for_me.ci import cov +except ImportError: # pragma: no cover - only the shipped bare-CI layout + # Shipped standalone next to this file in the composite action; exercised + # end-to-end by TestShippedStandalone via a subprocess. + import cov # ty: ignore[unresolved-import] + +# Must match the `reachability` job's `timeout-minutes` in +# ci/fuzz-verify.yml. Pinned by test_reach_timeout_label_matches_workflow_cap +# so the rendered cell can't advertise a cap that differs from the one +# GitHub actually enforces. +_REACH_TIMEOUT_MIN = 45 + +MARKER = "" + +# Δ is computed on covered counts (not percent points) so the same metric is +# meaningful when the denominator changes — e.g. when new code lands the +# instrumented line count grows, so comparing raw percentages understates real +# coverage gains. "new" / "deleted" cover the divide-by-zero edges. +FOOTER_DELTA_NOTE = ( + "Δ = (after − before) / before, to accommodate that denominators " + 'may change. "new" when before is 0; "deleted" when after is 0.') +FOOTER_GENERIC = FOOTER_DELTA_NOTE +FOOTER_UPSTREAM = ("Same harness config applied to both sides " + "(baseline = base source + PR harness).\n" + + FOOTER_DELTA_NOTE) + + +def _load_json(path: pathlib.Path): + if not path.exists(): + return None + try: + return json.loads(path.read_text()) + except json.JSONDecodeError: + return None + + +def _load_variant(base: pathlib.Path) -> dict: + harness: dict = {} + hd = base / "harness" + if hd.is_dir(): + for f in sorted(hd.glob("*.summary.json")): + data = _load_json(f) + if data is not None: + harness[f.name.removesuffix(".summary.json")] = data + return { + "meta": _load_json(base / "meta.json"), + "project": _load_json(base / "project.summary.json"), + "harness": harness, + "reachability": _load_json(base / "reachability.json"), + } + + +def _totals(summary): + """Harness-excluded coverage totals from an llvm-cov ``summary.json``.""" + return cov.coverage_totals(summary) + + +def _reach_totals(functions): + """Harness-excluded static reachability from the introspector per-function + list (``all-fuzz-introspector-functions.json``, carried by the + reachability.json artifact). ``None`` when no data is available.""" + return cov.reach_totals(functions) + + +def _fmt_cov(tot, key): + if not tot: + return "0%" + cov, n, pct = tot[key] + return f"{pct:.1f}% ({cov}/{n})" + + +def _fmt_delta(b, a, key, c_has=True): + if not b and not a: + return "—" + if not a: + # Baseline has data, current doesn't. + return "**removed**" if c_has else "**build failed**" + cov_a = a[key][0] + cov_b = b[key][0] if b else 0 + if cov_b == 0 and cov_a == 0: + return "—" + if cov_b == 0: + return "**new**" + if cov_a == 0: + return "**deleted**" + d = (cov_a - cov_b) / cov_b * 100 + sign = "+" if d >= 0 else "" + return f"**{sign}{d:.1f}%**" + + +def _fmt_reach_cell(reach, variant_has, run_url): + """Format a reachability value cell. + + ``>{_REACH_TIMEOUT_MIN}m`` (linked to the workflow run) when the variant + ran but the introspector build produced no summary — overwhelmingly + because it didn't finish within the soft job's guard timeout (an unbounded + Fuzz Introspector analysis; upstream OSS-Fuzz hits the same wall even with + a 20h budget). ``0%`` when the variant didn't run at all (matching the + coverage-row convention). + """ + if reach: + return _fmt_cov(reach, "reach") + if variant_has: + return f"[>{_REACH_TIMEOUT_MIN}m]({run_url})" + return "0%" + + +def _fmt_reach_delta(br, cr): + """Delta for the reachability row. Like ``_fmt_delta`` but treats a missing + variant as a build failure (``continue-on-error: true``) rather than a + removed metric — there is no "remove static reachability" intent.""" + if not br and not cr: + return "—" + if not br: + d = cr["reach"][2] + return f"**+{d:.1f}%**" + if not cr: + return "**build failed**" + d = cr["reach"][2] - br["reach"][2] + sign = "+" if d >= 0 else "" + return f"**{sign}{d:.1f}%**" + + +def render( + stats_root: pathlib.Path, + run_url: str, + fuzz_seconds: str, + now_utc: str, + footer: str, +) -> str: + b = _load_variant(stats_root / "baseline") + c = _load_variant(stats_root / "current") + + b_meta = b["meta"] or {} + c_meta = c["meta"] or {} + b_sha_full = b_meta.get("sha") or "" + c_sha_full = c_meta.get("sha") or "" + b_sha = b_sha_full[:7] if b_sha_full else "unknown" + c_sha = c_sha_full[:7] if c_sha_full else "unknown" + project = c_meta.get("project") or b_meta.get("project") or "?" + b_has = bool(b_meta.get("has_project")) + c_has = bool(c_meta.get("has_project")) + + out = [MARKER, "", "## Fuzzing Coverage Report", ""] + + tested = f"**Tested:** project `{project}` · base `{b_sha}`" + if not b_has: + tested += ( + " _(no baseline — project not present at base or baseline build failed)_" + ) + tested += f" → head `{c_sha}`" + if not c_has: + tested += " _(current measurement failed)_" + tested += (f" · {fuzz_seconds}s total fuzz budget" + f" · updated {now_utc}" + f" · [workflow run]({run_url})") + out += [tested, ""] + + bt = _totals(b["project"]) + ct = _totals(c["project"]) + br = _reach_totals(b["reachability"]) + cr = _reach_totals(c["reachability"]) + if b_has or c_has or bt or ct or br or cr: + out += [ + "| Metric | Before | After | Delta |", + "|---|---|---|---|", + f"| Static reachability | {_fmt_reach_cell(br, b_has, run_url)} | " + f"{_fmt_reach_cell(cr, c_has, run_url)} | " + f"{_fmt_reach_delta(br, cr)} |", + ] + if bt or ct: + out += [ + f"| Line coverage | {_fmt_cov(bt, 'lines')} | {_fmt_cov(ct, 'lines')} | {_fmt_delta(bt, ct, 'lines', c_has)} |", + f"| Branch coverage | {_fmt_cov(bt, 'branches')} | {_fmt_cov(ct, 'branches')} | {_fmt_delta(bt, ct, 'branches', c_has)} |", + f"| Function coverage | {_fmt_cov(bt, 'functions')} | {_fmt_cov(ct, 'functions')} | {_fmt_delta(bt, ct, 'functions', c_has)} |", + ] + out.append("") + + all_h = sorted(set(b["harness"].keys()) | set(c["harness"].keys())) + if all_h: + out += [ + "### Per-harness", + "", + "| Harness | Lines before | Lines after | Δ |", + "|---|---|---|---|", + ] + for h in all_h: + bh = _totals(b["harness"].get(h)) + ch = _totals(c["harness"].get(h)) + out.append( + f"| `{h}` | {_fmt_cov(bh, 'lines')} | {_fmt_cov(ch, 'lines')} | " + f"{_fmt_delta(bh, ch, 'lines', c_has)} |") + out.append("") + + if not (b_has or c_has or bt or ct or all_h or br or cr): + out += [ + "_No coverage data collected. Check the workflow run for build errors._", + "", + ] + + out.append(footer) + return "\n".join(out) + + +def main(): + stats_root = pathlib.Path(os.environ.get("STATS_ROOT", "stats")) + run_url = os.environ["RUN_URL"] + fuzz_seconds = os.environ.get("FUZZ_SECONDS", "300") + footer_kind = os.environ.get("FOOTER", "generic") + now_utc = datetime.datetime.now( + datetime.timezone.utc).strftime("%Y-%m-%d %H:%M UTC") + footer = FOOTER_UPSTREAM if footer_kind == "upstream" else FOOTER_GENERIC + sys.stdout.write(render(stats_root, run_url, fuzz_seconds, now_utc, footer)) + sys.stdout.write("\n") + + +if __name__ == "__main__": + main() diff --git a/.github/actions/prepare/action.yml b/.github/actions/prepare/action.yml new file mode 100644 index 000000000..777d344f6 --- /dev/null +++ b/.github/actions/prepare/action.yml @@ -0,0 +1,45 @@ +name: Prepare fuzz-verify +description: Resolve project/variant, baseline checkout/overlay, oss-fuzz clone, pristine snapshot. Emits normalized outputs. + +inputs: + mode: + description: upstream | oss-fuzz + required: true + variant: + description: baseline | current + required: true + project_name: + description: Project name (upstream); empty for oss-fuzz (auto-detected). + required: false + default: '' + +outputs: + project: + value: ${{ steps.run.outputs.project }} + sha: + value: ${{ steps.run.outputs.sha }} + merge_base_sha: + value: ${{ steps.run.outputs.merge_base_sha }} + has_project: + value: ${{ steps.run.outputs.has_project }} + helper_dir: + value: ${{ steps.run.outputs.helper_dir }} + out_base: + value: ${{ steps.run.outputs.out_base }} + corpus_root: + value: ${{ steps.run.outputs.corpus_root }} + seeds_dir: + value: ${{ steps.run.outputs.seeds_dir }} + pristine_dir: + value: ${{ steps.run.outputs.pristine_dir }} + +runs: + using: composite + steps: + - id: run + shell: bash + env: + MODE: ${{ inputs.mode }} + VARIANT: ${{ inputs.variant }} + PROJECT_NAME: ${{ inputs.project_name }} + run: python3 "$GITHUB_ACTION_PATH/prepare.py" diff --git a/.github/actions/prepare/prepare.py b/.github/actions/prepare/prepare.py new file mode 100644 index 000000000..13bd6bc0e --- /dev/null +++ b/.github/actions/prepare/prepare.py @@ -0,0 +1,228 @@ +#!/usr/bin/env python3 +# This file ships into oss-fuzz review-repo PRs via the prepare composite +# action; oss-fuzz's `infra/presubmit.py` check_license greps every Python +# file under non-projects/ paths for the Apache 2.0 LICENSE-2.0 URL, so we +# carry the standard header here even though it's our own infra code. +# Copyright 2026 fuzz-for-me contributors +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. +"""Mode-specific prep for the fuzz-verify reusable workflow. + +Resolves the project + variant SHA, performs baseline checkout/overlay, +clones oss-fuzz (upstream mode), snapshots a pristine source tree +(upstream mode, bug #2 fix), and emits normalized outputs the +mode-agnostic reusable workflow consumes. +""" + +import os +import pathlib +import re +import shutil +import subprocess +import sys + +_OSS_FUZZ_DIR = "/tmp/oss-fuzz" +_PRISTINE_DIR = "/tmp/fuzz-verify-pristine" + + +def _run(args, cwd=None, check=True): + return subprocess.run(args, + cwd=cwd, + check=check, + capture_output=True, + text=True) + + +def detect_oss_fuzz_project(repo: pathlib.Path) -> str: + out = _run( + ["git", "-C", + str(repo), "diff", "--name-only", "origin/master...HEAD"], + check=False, + ).stdout + for line in out.splitlines(): + if line.startswith("projects/"): + return line.split("/")[1] + # Fallback for local test repos without an `origin` remote. + out = _run( + ["git", "-C", + str(repo), "diff", "--name-only", "master...HEAD"], + check=False, + ).stdout + for line in out.splitlines(): + if line.startswith("projects/"): + return line.split("/")[1] + return "" + + +def compute_outputs(mode: str, project: str) -> dict[str, str]: + if mode == "oss-fuzz": + return { + "helper_dir": ".", + "out_base": "build/out", + "corpus_root": f"corpus/{project}", + "seeds_dir": f"projects/{project}/seeds", + } + if mode == "upstream": + return { + "helper_dir": _OSS_FUZZ_DIR, + "out_base": f"{_OSS_FUZZ_DIR}/build/out", + "corpus_root": f"{_OSS_FUZZ_DIR}/corpus/{project}", + "seeds_dir": ".github/fuzz/seeds", + } + raise ValueError(f"unknown mode: {mode}") + + +def _merge_base(repo: pathlib.Path, ref_a: str, ref_b: str) -> str: + return _run(["git", "-C", str(repo), "merge-base", ref_a, + ref_b]).stdout.strip() + + +def _default_remote_branch(repo: pathlib.Path) -> str: + r = _run( + [ + "git", "-C", + str(repo), "symbolic-ref", "--short", "refs/remotes/origin/HEAD" + ], + check=False, + ).stdout.strip() + return r.split("origin/")[-1] if r else "" + + +def _compute_merge_base(repo: pathlib.Path) -> str: + base_ref = "origin/master" + if _run(["git", "-C", + str(repo), "rev-parse", "--verify", "-q", base_ref], + check=False).returncode != 0: + base_ref = "master" + sha = _merge_base(repo, base_ref, "HEAD") + if not re.fullmatch(r"[0-9a-f]{7,40}", sha): + raise RuntimeError( + f"merge-base produced no usable SHA (base_ref={base_ref})") + return sha + + +def resolve_variant(mode: str, variant: str, project: str, + repo: pathlib.Path) -> dict[str, str]: + head = _run(["git", "-C", str(repo), "rev-parse", "HEAD"]).stdout.strip() + if variant == "current": + return {"sha": os.environ.get("GITHUB_SHA", head), "has_project": "true"} + + sha = _compute_merge_base(repo) + + if mode == "oss-fuzz": + present = _run( + [ + "git", "-C", + str(repo), "cat-file", "-e", f"{sha}:projects/{project}/Dockerfile" + ], + check=False, + ).returncode == 0 + if not present: + return {"sha": sha, "has_project": "false"} + shutil.rmtree(repo / "projects" / project, ignore_errors=True) + _run(["git", "-C", str(repo), "checkout", sha, "--", f"projects/{project}"]) + return {"sha": sha, "has_project": "true"} + + # upstream baseline: stash overlay, hard-reset to merge-base, restore. + fuzz = repo / ".github" / "fuzz" + if not fuzz.is_dir(): + print("::error::.github/fuzz not present on PR — " + "cannot overlay onto baseline") + return {"sha": sha, "has_project": "false"} + actions = repo / ".github" / "actions" + tmp = pathlib.Path(_run(["mktemp", "-d"]).stdout.strip()) + shutil.copytree(fuzz, tmp / "fuzz") + if actions.is_dir(): + shutil.copytree(actions, tmp / "actions") + _run(["git", "-C", str(repo), "reset", "--hard", sha]) + _run( + ["git", "-C", + str(repo), "submodule", "update", "--init", "--recursive"], + check=False, + ) + shutil.rmtree(repo / ".github" / "fuzz", ignore_errors=True) + shutil.rmtree(repo / ".github" / "actions", ignore_errors=True) + (repo / ".github").mkdir(exist_ok=True) + shutil.copytree(tmp / "fuzz", repo / ".github" / "fuzz") + if (tmp / "actions").is_dir(): + shutil.copytree(tmp / "actions", repo / ".github" / "actions") + return {"sha": sha, "has_project": "true"} + + +def _copy_tree(src: pathlib.Path, dst: pathlib.Path) -> str: + if shutil.which("rsync"): + dst.mkdir(parents=True, exist_ok=True) + _run(["rsync", "-a", "--delete", f"{src}/", f"{dst}/"]) + else: + if dst.exists(): + shutil.rmtree(dst) + shutil.copytree(src, dst, symlinks=True) + return str(dst) + + +def snapshot_pristine(workspace: pathlib.Path | str, + dest_root: pathlib.Path | str) -> str: + return _copy_tree(pathlib.Path(workspace), pathlib.Path(dest_root)) + + +def materialize_working_copy(pristine: pathlib.Path | str, + dest: pathlib.Path | str) -> str: + return _copy_tree(pathlib.Path(pristine), pathlib.Path(dest)) + + +def _emit(outputs: dict[str, str]): + path = os.environ["GITHUB_OUTPUT"] + with open(path, "a") as fh: + for k, v in outputs.items(): + fh.write(f"{k}={v}\n") + + +def main(): + mode = os.environ["MODE"] + variant = os.environ["VARIANT"] + ws = pathlib.Path(os.environ["GITHUB_WORKSPACE"]) + project = os.environ.get("PROJECT_NAME", "") or detect_oss_fuzz_project(ws) + + var = resolve_variant(mode, variant, project, ws) + out = {"project": project, **var, "merge_base_sha": _compute_merge_base(ws)} + is_upstream_project = var["has_project"] == "true" and mode == "upstream" + + if is_upstream_project: + fuzz = ws / ".github" / "fuzz" + if not fuzz.is_dir(): + print("::error::.github/fuzz not present — " + "cannot overlay onto baseline") + return 1 + _run([ + "git", "clone", "--depth", "1", + "https://github.com/google/oss-fuzz.git", _OSS_FUZZ_DIR + ], + check=False) + dest = pathlib.Path(f"{_OSS_FUZZ_DIR}/projects/{project}") + dest.mkdir(parents=True, exist_ok=True) + for item in fuzz.iterdir(): + tgt = dest / item.name + (shutil.copytree if item.is_dir() else shutil.copy)(item, tgt) + + paths = compute_outputs(mode, project) + if is_upstream_project: + out["pristine_dir"] = snapshot_pristine(ws, _PRISTINE_DIR) + else: + out["pristine_dir"] = "" + out.update(paths) + _emit(out) + return 0 + + +if __name__ == "__main__": + sys.exit(main()) diff --git a/.github/workflows/_fuzz-verify.yml b/.github/workflows/_fuzz-verify.yml new file mode 100644 index 000000000..6937f74c1 --- /dev/null +++ b/.github/workflows/_fuzz-verify.yml @@ -0,0 +1,274 @@ +# Reusable workflow injected as .github/workflows/_fuzz-verify.yml by pr_server. +# Mode-agnostic: every path/arg comes from the `prepare` action's outputs. +name: Fuzz Verify +on: + workflow_call: + inputs: + mode: { required: true, type: string } + project_name: { required: false, type: string, default: '' } + footer: { required: true, type: string } + fuzz_seconds: { required: true, type: string } + +permissions: + # Default to read-only at the workflow level; the report job elevates to pull-requests:write. + contents: read + +concurrency: + group: ${{ github.workflow }}-${{ github.ref }} + cancel-in-progress: true + +jobs: + measure: + name: Measure (${{ matrix.variant }}) + runs-on: ubuntu-latest + timeout-minutes: 120 + continue-on-error: ${{ matrix.variant == 'baseline' }} + strategy: + fail-fast: false + matrix: + variant: [baseline, current] + steps: + - uses: actions/checkout@de0fac2e4500dabe0009e67214ff5f5447ce83dd # v6.0.2 + with: { fetch-depth: 0, persist-credentials: false, submodules: recursive } + + - id: prep + uses: ./.github/actions/prepare + with: + mode: ${{ inputs.mode }} + variant: ${{ matrix.variant }} + project_name: ${{ inputs.project_name }} + + - name: Install oss-fuzz deps + if: steps.prep.outputs.has_project == 'true' + env: + HELPER_DIR: ${{ steps.prep.outputs.helper_dir }} + run: pip install -r "$HELPER_DIR/infra/ci/requirements.txt" 2>/dev/null || pip install docker + + - name: Working copy (address) + id: wc-addr + if: steps.prep.outputs.has_project == 'true' && steps.prep.outputs.pristine_dir != '' + env: + PRISTINE_DIR: ${{ steps.prep.outputs.pristine_dir }} + run: | + rsync -a --delete "$PRISTINE_DIR/" /tmp/ws-address/ + echo "src=/tmp/ws-address" >> "$GITHUB_OUTPUT" + + - name: Build image + id: bi + if: steps.prep.outputs.has_project == 'true' + continue-on-error: ${{ matrix.variant == 'baseline' }} + env: + HELPER_DIR: ${{ steps.prep.outputs.helper_dir }} + PROJECT: ${{ steps.prep.outputs.project }} + run: echo n | python3 "$HELPER_DIR/infra/helper.py" build_image "$PROJECT" + + - name: Build fuzzers + id: bf + if: steps.prep.outputs.has_project == 'true' && steps.bi.outcome == 'success' + continue-on-error: ${{ matrix.variant == 'baseline' }} + env: + HELPER_DIR: ${{ steps.prep.outputs.helper_dir }} + PROJECT: ${{ steps.prep.outputs.project }} + WC_SRC: ${{ steps.wc-addr.outputs.src }} + run: | + if [ -n "$WC_SRC" ]; then + echo n | python3 "$HELPER_DIR/infra/helper.py" build_fuzzers \ + --sanitizer address "$PROJECT" "$WC_SRC" + else + echo n | python3 "$HELPER_DIR/infra/helper.py" build_fuzzers \ + --sanitizer address "$PROJECT" + fi + + - name: Verify new seeds + if: matrix.variant == 'current' && steps.prep.outputs.has_project == 'true' && steps.bf.outcome == 'success' + uses: ./.github/actions/check-new-seeds + with: + project: ${{ steps.prep.outputs.project }} + base_sha: ${{ steps.prep.outputs.merge_base_sha }} + out_base: ${{ steps.prep.outputs.out_base }} + seeds_dir: ${{ steps.prep.outputs.seeds_dir }} + + - name: Fuzz + if: steps.prep.outputs.has_project == 'true' && steps.bf.outcome == 'success' + uses: ./.github/actions/fuzz-all + with: + project: ${{ steps.prep.outputs.project }} + helper_py: ${{ steps.prep.outputs.helper_dir }}/infra/helper.py + out_dir: ${{ steps.prep.outputs.out_base }}/${{ steps.prep.outputs.project }} + corpus_root: ${{ steps.prep.outputs.corpus_root }} + logs_dir: /tmp/fuzz-logs + fuzz_seconds: ${{ inputs.fuzz_seconds }} + + - name: Coverage + if: steps.prep.outputs.has_project == 'true' && steps.bf.outcome == 'success' + env: + HELPER_DIR: ${{ steps.prep.outputs.helper_dir }} + PROJECT: ${{ steps.prep.outputs.project }} + PRISTINE_DIR: ${{ steps.prep.outputs.pristine_dir }} + CORPUS_ROOT: ${{ steps.prep.outputs.corpus_root }} + run: | + if [ -n "$PRISTINE_DIR" ]; then + rsync -a --delete "$PRISTINE_DIR/" /tmp/ws-cov/ + echo n | python3 "$HELPER_DIR/infra/helper.py" build_fuzzers \ + --sanitizer coverage "$PROJECT" /tmp/ws-cov + else + echo n | python3 "$HELPER_DIR/infra/helper.py" build_fuzzers \ + --sanitizer coverage "$PROJECT" + fi + CORPUS_LINK_DIR="$HELPER_DIR/build/corpus" + mkdir -p "$CORPUS_LINK_DIR" + case "$CORPUS_ROOT" in + /*) CORPUS_TGT="$CORPUS_ROOT" ;; + *) CORPUS_TGT="$(pwd)/$CORPUS_ROOT" ;; + esac + ln -sfn "$CORPUS_TGT" "$CORPUS_LINK_DIR/$PROJECT" + python3 "$HELPER_DIR/infra/helper.py" coverage --no-corpus-download --no-serve "$PROJECT" 2>&1 || true + + - name: Collect stats + if: always() + continue-on-error: ${{ matrix.variant == 'baseline' }} + uses: ./.github/actions/collect-fuzz-stats + with: + project: ${{ steps.prep.outputs.project }} + variant: ${{ matrix.variant }} + sha: ${{ steps.prep.outputs.sha }} + has_project: ${{ steps.prep.outputs.has_project }} + out_base: ${{ steps.prep.outputs.out_base }} + + - name: Upload stats + if: always() + uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7.0.1 + with: + name: stats-${{ matrix.variant }} + path: stats-${{ matrix.variant }}/ + if-no-files-found: error + + - name: Upload coverage report + if: always() && steps.prep.outputs.has_project == 'true' + uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7.0.1 + with: + name: coverage-report-${{ matrix.variant }} + path: ${{ steps.prep.outputs.out_base }}/*coverage*/report/ + if-no-files-found: ignore + + # 45-min guard: Fuzz Introspector's analysis is unbounded over /src, so + # projects vendoring large trees (libprotobuf-mutator, fuzzer-test-suite) + # never finish — upstream OSS-Fuzz grants its own introspector build 20h + # and nginx/capnproto STILL time out. More time is futile; ~68% of + # successful C/C++ projects finish under 45 min, so cap there and + # fast-fail the doomed long tail instead of burning 2h × 2 variants. + # Soft signal: continue-on-error, so a cancel never reds the PR. + reachability: + name: Reachability (${{ matrix.variant }}) + runs-on: ubuntu-latest + # No job-level timeout-minutes — a timeout-cancelled job propagates to + # the workflow conclusion, reds the PR even though continue-on-error is + # set. The 45-min cap is enforced at the step level instead (see + # `timeout 45m` in "Build introspector") and the step absorbs the exit. + continue-on-error: true + strategy: + fail-fast: false + matrix: + variant: [baseline, current] + steps: + - uses: actions/checkout@de0fac2e4500dabe0009e67214ff5f5447ce83dd # v6.0.2 + with: { fetch-depth: 0, persist-credentials: false, submodules: recursive } + + - id: prep + uses: ./.github/actions/prepare + with: + mode: ${{ inputs.mode }} + variant: ${{ matrix.variant }} + project_name: ${{ inputs.project_name }} + + - name: Install oss-fuzz deps + if: steps.prep.outputs.has_project == 'true' + env: + HELPER_DIR: ${{ steps.prep.outputs.helper_dir }} + run: pip install -r "$HELPER_DIR/infra/ci/requirements.txt" 2>/dev/null || pip install docker + + - name: Build image + id: bi + if: steps.prep.outputs.has_project == 'true' + continue-on-error: true + env: + HELPER_DIR: ${{ steps.prep.outputs.helper_dir }} + PROJECT: ${{ steps.prep.outputs.project }} + run: echo n | python3 "$HELPER_DIR/infra/helper.py" build_image "$PROJECT" + + - name: Working copy (introspector) + id: wc-int + if: steps.prep.outputs.has_project == 'true' && steps.prep.outputs.pristine_dir != '' && steps.bi.outcome == 'success' + env: + PRISTINE_DIR: ${{ steps.prep.outputs.pristine_dir }} + run: | + rsync -a --delete "$PRISTINE_DIR/" /tmp/ws-introspector/ + echo "src=/tmp/ws-introspector" >> "$GITHUB_OUTPUT" + + - name: Build introspector + if: steps.prep.outputs.has_project == 'true' && steps.bi.outcome == 'success' + env: + HELPER_DIR: ${{ steps.prep.outputs.helper_dir }} + PROJECT: ${{ steps.prep.outputs.project }} + WC_SRC: ${{ steps.wc-int.outputs.src }} + # 45-min cap on the unbounded introspector analysis. We absorb the + # exit code so a timeout (124) or build failure doesn't red the + # step — Extract reachability handles missing output gracefully and + # render_comment renders ">45m" for a missing variant. Keeping the + # job-level timeout-minutes here would cancel the job on expiry, + # which propagates to the workflow conclusion despite + # continue-on-error and reds the PR. + run: | + set +e + if [ -n "$WC_SRC" ]; then + echo n | timeout 45m python3 "$HELPER_DIR/infra/helper.py" build_fuzzers \ + --sanitizer introspector "$PROJECT" "$WC_SRC" + else + echo n | timeout 45m python3 "$HELPER_DIR/infra/helper.py" build_fuzzers \ + --sanitizer introspector "$PROJECT" + fi + rc=$? + echo "::notice::introspector build exited rc=$rc (124=timeout)" + exit 0 + + - name: Extract reachability + if: always() + env: + VARIANT: ${{ matrix.variant }} + OUT_BASE: ${{ steps.prep.outputs.out_base }} + PROJECT: ${{ steps.prep.outputs.project }} + run: | + mkdir -p "reachability-$VARIANT" + SRC="$OUT_BASE/$PROJECT/inspector/all-fuzz-introspector-functions.json" + if [ -f "$SRC" ]; then + cp "$SRC" "reachability-$VARIANT/reachability.json" + else + echo "::notice::No introspector output for $VARIANT — reachability unavailable" + fi + + - name: Upload reachability + if: always() + uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7.0.1 + with: + name: reachability-${{ matrix.variant }} + path: reachability-${{ matrix.variant }}/ + if-no-files-found: ignore + + report: + name: Report + needs: [measure, reachability] + if: always() + runs-on: ubuntu-latest + permissions: + contents: read + pull-requests: write # post sticky fuzz-verify comment + steps: + - uses: actions/checkout@de0fac2e4500dabe0009e67214ff5f5447ce83dd # v6.0.2 + with: + persist-credentials: false + sparse-checkout: .github/actions/post-fuzz-report + - uses: ./.github/actions/post-fuzz-report + with: + footer: ${{ inputs.footer }} + fuzz_seconds: ${{ inputs.fuzz_seconds }} + github_token: ${{ secrets.GITHUB_TOKEN }} diff --git a/.github/workflows/fuzz-verify.yml b/.github/workflows/fuzz-verify.yml new file mode 100644 index 000000000..3032670b2 --- /dev/null +++ b/.github/workflows/fuzz-verify.yml @@ -0,0 +1,15 @@ +# Injected by pr_server. Drop before forwarding upstream. +name: Fuzz Verify +on: [push] +jobs: + fuzz-verify: + uses: ./.github/workflows/_fuzz-verify.yml + permissions: + contents: read + pull-requests: write + with: + mode: oss-fuzz + project_name: "" + footer: generic + fuzz_seconds: "300" + secrets: inherit diff --git a/projects/nginx/build.sh b/projects/nginx/build.sh index 39da173db..165b6b971 100644 --- a/projects/nginx/build.sh +++ b/projects/nginx/build.sh @@ -19,6 +19,35 @@ git apply $SRC/add_fuzzers.diff || patch -p1 < $SRC/add_fuzzers.diff cp -r $SRC/fuzz src/ cp $SRC/make_fuzzers auto/make_fuzzers +# Expose the static DNS-response parser so resolver_fuzzer can drive it +# directly. Drops `static` from the five interesting parser entry points +# in src/core/ngx_resolver.c (declarations on lines ~86-104 and definitions +# scattered through the file). All other `static`s in the file are left +# alone. +python3 - <<'PY' +import re +path = "src/core/ngx_resolver.c" +with open(path) as f: + s = f.read() +fns = ( + "ngx_resolver_process_response", + "ngx_resolver_process_a", + "ngx_resolver_process_srv", + "ngx_resolver_process_ptr", + "ngx_resolver_copy", +) +for fn in fns: + # Declarations: "static void ngx_resolver_foo(" / "static ngx_int_t ngx_resolver_foo(" + s = re.sub(r"^static (void|ngx_int_t) " + fn + r"\(", + r"\1 " + fn + "(", s, flags=re.M) + # Definitions: "static void\nngx_resolver_foo(" / "static ngx_int_t\nngx_resolver_foo(" + s = re.sub(r"^static (void|ngx_int_t)\n" + fn + r"\(", + r"\1\n" + fn + "(", s, flags=re.M) +with open(path, "w") as f: + f.write(s) +print("resolver statics exposed") +PY + cd src/fuzz rm -rf genfiles && mkdir genfiles && $SRC/LPM/external.protobuf/bin/protoc http_request_proto.proto --cpp_out=genfiles cd ../.. @@ -31,3 +60,163 @@ make -f objs/Makefile fuzzers cp objs/*_fuzzer $OUT/ cp $SRC/fuzz/*.dict $OUT/ +cp $SRC/fuzz/*.options $OUT/ 2>/dev/null || true + +################################################################################ +# Seed corpora +################################################################################ +SEEDS_DIR=$(mktemp -d) + +# ---- pp_fuzzer seeds (v1 text + v2 binary, incl. one with TLVs) ---- +mkdir -p "$SEEDS_DIR/pp" +printf 'PROXY TCP4 1.2.3.4 5.6.7.8 80 81\r\nGET / HTTP/1.0\r\n\r\n' > "$SEEDS_DIR/pp/v1_tcp4" +printf 'PROXY TCP6 ::1 ::2 12345 443\r\n' > "$SEEDS_DIR/pp/v1_tcp6" +printf 'PROXY UNKNOWN\r\n' > "$SEEDS_DIR/pp/v1_unknown" +# v2 minimal IPv4: magic(12) + ver/cmd(0x21) + fam/proto(0x11) + len(12 BE) +printf '\r\n\r\n\x00\r\nQUIT\n\x21\x11\x00\x0c\x01\x02\x03\x04\x05\x06\x07\x08\x00\x50\x00\x51' > "$SEEDS_DIR/pp/v2_tcp4" +# v2 with one TLV (ALPN=h2): header + 12-byte addrs + tlv type(0x01) + len(2) + "h2" = 17 bytes +printf '\r\n\r\n\x00\r\nQUIT\n\x21\x11\x00\x11\x01\x02\x03\x04\x05\x06\x07\x08\x00\x50\x00\x51\x01\x00\x02h2' > "$SEEDS_DIR/pp/v2_alpn" + +(cd "$SEEDS_DIR/pp" && zip -q -j "$OUT/pp_fuzzer_seed_corpus.zip" ./*) + +# ---- parser_fuzzer seeds (selector byte + payload) ---- +mkdir -p "$SEEDS_DIR/hp" +# selector 0: request line +printf '\x00GET /index.html?x=1 HTTP/1.1\r\n' > "$SEEDS_DIR/hp/sel0_get" +printf '\x00POST http://example.com:80/path HTTP/1.0\r\n' > "$SEEDS_DIR/hp/sel0_abs_uri" +printf '\x00CONNECT example.com:443 HTTP/1.1\r\n' > "$SEEDS_DIR/hp/sel0_connect" +# selector 1: header line (no underscores) +printf '\x01Host: example.com\r\nUser-Agent: ua\r\nContent-Length: 5\r\n\r\n' > "$SEEDS_DIR/hp/sel1_hdrs" +# selector 2: header line (allow underscores) +printf '\x02X_Custom: value\r\n\r\n' > "$SEEDS_DIR/hp/sel2_under" +# selector 3: status line +printf '\x03HTTP/1.1 200 OK\r\n' > "$SEEDS_DIR/hp/sel3_ok" +printf '\x03HTTP/1.0 404 Not Found\r\n' > "$SEEDS_DIR/hp/sel3_404" +# selector 4: chunked +printf '\x045\r\nhello\r\n0\r\n\r\n' > "$SEEDS_DIR/hp/sel4_simple" +printf '\x04a;ext=foo\r\n0123456789\r\n0\r\nTrailer: x\r\n\r\n' > "$SEEDS_DIR/hp/sel4_ext" +# selector 5: complex uri +printf '\x05/a/b/../c/./d?x=1&y=2' > "$SEEDS_DIR/hp/sel5_dotdot" +printf '\x05/path%%20with%%2fencoded' > "$SEEDS_DIR/hp/sel5_enc" +# selector 6: unsafe uri +printf '\x06/abc%%20def' > "$SEEDS_DIR/hp/sel6_safe" +printf '\x06../etc/passwd' > "$SEEDS_DIR/hp/sel6_dotdot" +# selector 7: ngx_http_parse_uri (URI char validator) +printf '\x07/path/with?query=1#frag' > "$SEEDS_DIR/hp/sel7_uri" +printf '\x07/with+plus/' > "$SEEDS_DIR/hp/sel7_plus" +# selector 8: ngx_http_arg — "name\0arg1=v1&arg2=v2" +printf '\x08arg1\x00arg1=hello&arg2=world&arg3=' > "$SEEDS_DIR/hp/sel8_arg" +# selector 9: ngx_http_split_args — uri with '?' +printf '\x09/foo/bar?a=1&b=2' > "$SEEDS_DIR/hp/sel9_split" +# selector 10/11/12: header chains "name\0Key1: v1\nKey2: v2" +printf '\x0aHost\x00Host: example.com\nHost: other.com\n' > "$SEEDS_DIR/hp/sel10_multi" +printf '\x0bsid\x00Cookie: sid=abc; foo=bar\nCookie: x=y\n' > "$SEEDS_DIR/hp/sel11_cookie" +printf '\x0csid\x00Set-Cookie: sid=abc; Path=/; HttpOnly\n' > "$SEEDS_DIR/hp/sel12_setcookie" + +(cd "$SEEDS_DIR/hp" && zip -q -j "$OUT/parser_fuzzer_seed_corpus.zip" ./*) + +# ---- inet_fuzzer seeds (selector byte + payload) ---- +mkdir -p "$SEEDS_DIR/in" +# selector 0: ngx_inet_addr (IPv4 text) +printf '\x00127.0.0.1' > "$SEEDS_DIR/in/sel0_v4_loop" +printf '\x000.0.0.0' > "$SEEDS_DIR/in/sel0_v4_any" +printf '\x00255.255.255.255' > "$SEEDS_DIR/in/sel0_v4_bcast" +# selector 1: ngx_inet6_addr +printf '\x01::1' > "$SEEDS_DIR/in/sel1_v6_loop" +printf '\x01fe80::1%%eth0' > "$SEEDS_DIR/in/sel1_v6_scope" +printf '\x012001:db8::1' > "$SEEDS_DIR/in/sel1_v6" +# selector 2: ngx_ptocidr +printf '\x0210.0.0.0/8' > "$SEEDS_DIR/in/sel2_cidr4" +printf '\x02::/0' > "$SEEDS_DIR/in/sel2_cidr6" +# selector 3: ngx_parse_addr +printf '\x03192.168.1.1' > "$SEEDS_DIR/in/sel3_v4" +printf '\x03::1' > "$SEEDS_DIR/in/sel3_v6" +# selector 4: ngx_parse_addr_port +printf '\x04127.0.0.1:8080' > "$SEEDS_DIR/in/sel4_v4port" +printf '\x04[::1]:443' > "$SEEDS_DIR/in/sel4_v6port" +# selector 5: ngx_parse_url (no listen) +printf '\x05http://example.com:80/p' > "$SEEDS_DIR/in/sel5_url" +printf '\x05unix:/tmp/sock:' > "$SEEDS_DIR/in/sel5_unix" +# selector 6: ngx_parse_url (listen=1) +printf '\x06*:8080' > "$SEEDS_DIR/in/sel6_listen" +printf '\x060.0.0.0:80' > "$SEEDS_DIR/in/sel6_listen2" +# selector 7: numeric parsers +printf '\x071234567890' > "$SEEDS_DIR/in/sel7_num" +printf '\x070x1aBcDeF' > "$SEEDS_DIR/in/sel7_hex" +# selector 8: unescape_uri +printf '\x08/path%%20with%%2Fenc' > "$SEEDS_DIR/in/sel8_unesc" +# selector 9: escape_uri +printf '\x09a b/c?d=e f' > "$SEEDS_DIR/in/sel9_esc" +# selector 10: base64 decode +printf '\x0aSGVsbG8gV29ybGQ=' > "$SEEDS_DIR/in/sel10_b64" + +(cd "$SEEDS_DIR/in" && zip -q -j "$OUT/inet_fuzzer_seed_corpus.zip" ./*) + +# ---- h2_fuzzer seeds: minimal HTTP/2 frame sequences ---- +mkdir -p "$SEEDS_DIR/h2" +# Empty SETTINGS frame (server sends one back, then client should ACK) +printf '\x00\x00\x00\x04\x00\x00\x00\x00\x00' > "$SEEDS_DIR/h2/settings_empty" +# SETTINGS ack +printf '\x00\x00\x00\x04\x01\x00\x00\x00\x00' > "$SEEDS_DIR/h2/settings_ack" +# SETTINGS with one entry (HEADER_TABLE_SIZE=4096) +printf '\x00\x00\x06\x04\x00\x00\x00\x00\x00\x00\x01\x00\x00\x10\x00' > "$SEEDS_DIR/h2/settings_one" +# SETTINGS + HEADERS (:method GET, :path /, :scheme http, :authority localhost) +# HEADERS frame: length=?, type=0x01, flags=0x05 (END_HEADERS|END_STREAM), stream=1 +# HPACK: indexed :method GET = 0x82, indexed :path / = 0x84, indexed :scheme http = 0x86, +# literal :authority "localhost" = 0x41 0x09 "localhost" +printf '\x00\x00\x00\x04\x00\x00\x00\x00\x00\x00\x00\x0e\x01\x05\x00\x00\x00\x01\x82\x84\x86\x41\x09localhost' > "$SEEDS_DIR/h2/headers_get" +# PING +printf '\x00\x00\x00\x04\x00\x00\x00\x00\x00\x00\x00\x08\x06\x00\x00\x00\x00\x00\x01\x02\x03\x04\x05\x06\x07\x08' > "$SEEDS_DIR/h2/ping" +# WINDOW_UPDATE for connection (+1000) +printf '\x00\x00\x00\x04\x00\x00\x00\x00\x00\x00\x00\x04\x08\x00\x00\x00\x00\x00\x00\x00\x03\xe8' > "$SEEDS_DIR/h2/window_update" +# GOAWAY +printf '\x00\x00\x00\x04\x00\x00\x00\x00\x00\x00\x00\x08\x07\x00\x00\x00\x00\x00\x00\x00\x00\x00\x00\x00\x00\x00' > "$SEEDS_DIR/h2/goaway" +# HEADERS with dynamic-table indexing (literal incremental) +printf '\x00\x00\x00\x04\x00\x00\x00\x00\x00\x00\x00\x14\x01\x05\x00\x00\x00\x01\x82\x84\x86\x40\x07custom1\x05valueA' > "$SEEDS_DIR/h2/headers_dyn" + +(cd "$SEEDS_DIR/h2" && zip -q -j "$OUT/h2_fuzzer_seed_corpus.zip" ./*) + +# ---- upstream_fuzzer seeds: valid HTTP/1 upstream responses ---- +mkdir -p "$SEEDS_DIR/up" +printf 'HTTP/1.1 200 OK\r\nContent-Length: 5\r\n\r\nhello' > "$SEEDS_DIR/up/200_cl" +printf 'HTTP/1.1 200 OK\r\nTransfer-Encoding: chunked\r\n\r\n5\r\nhello\r\n0\r\n\r\n' > "$SEEDS_DIR/up/200_chunked" +printf 'HTTP/1.1 204 No Content\r\n\r\n' > "$SEEDS_DIR/up/204" +printf 'HTTP/1.1 301 Moved\r\nLocation: /new\r\nContent-Length: 0\r\n\r\n' > "$SEEDS_DIR/up/301" +printf 'HTTP/1.1 304 Not Modified\r\nETag: "abc"\r\n\r\n' > "$SEEDS_DIR/up/304" +printf 'HTTP/1.1 500 Internal Server Error\r\nContent-Length: 0\r\n\r\n' > "$SEEDS_DIR/up/500" +printf 'HTTP/1.1 100 Continue\r\n\r\nHTTP/1.1 200 OK\r\nContent-Length: 2\r\n\r\nok' > "$SEEDS_DIR/up/100_continue" +# gzip body +printf 'HTTP/1.1 200 OK\r\nContent-Encoding: gzip\r\nContent-Length: 20\r\n\r\n\x1f\x8b\x08\x00\x00\x00\x00\x00\x00\x03\xcb\x48\xcd\xc9\xc9\x07\x00\x86\xa6\x10\x36' > "$SEEDS_DIR/up/gzip" +# Set-Cookie + cache headers (filter pipeline) +printf 'HTTP/1.1 200 OK\r\nSet-Cookie: id=x; Path=/\r\nCache-Control: max-age=60\r\nETag: "v1"\r\nLast-Modified: Wed, 14 May 2026 00:00:00 GMT\r\nContent-Length: 3\r\n\r\nfoo' > "$SEEDS_DIR/up/cookies" +# X-Accel-Redirect (internal redirect) +printf 'HTTP/1.1 200 OK\r\nX-Accel-Redirect: /internal\r\n\r\n' > "$SEEDS_DIR/up/x_accel_redirect" +# Trailer headers +printf 'HTTP/1.1 200 OK\r\nTrailer: X-Trailer\r\nTransfer-Encoding: chunked\r\n\r\n5\r\nhello\r\n0\r\nX-Trailer: v\r\n\r\n' > "$SEEDS_DIR/up/trailer" +# Connection: upgrade (tunnel module) +printf 'HTTP/1.1 101 Switching\r\nUpgrade: websocket\r\nConnection: Upgrade\r\n\r\n' > "$SEEDS_DIR/up/upgrade" +# Multiple Set-Cookie +printf 'HTTP/1.1 200 OK\r\nSet-Cookie: a=1\r\nSet-Cookie: b=2\r\nSet-Cookie: c=3\r\nContent-Length: 0\r\n\r\n' > "$SEEDS_DIR/up/multi_cookie" +# WWW-Authenticate +printf 'HTTP/1.1 401 Unauthorized\r\nWWW-Authenticate: Basic realm="x"\r\nContent-Length: 0\r\n\r\n' > "$SEEDS_DIR/up/auth" + +(cd "$SEEDS_DIR/up" && zip -q -j "$OUT/upstream_fuzzer_seed_corpus.zip" ./*) + +# ---- resolver_fuzzer seeds: DNS responses ---- +# First byte = tcp_flag (0=UDP, 1=TCP), remainder = DNS packet. +mkdir -p "$SEEDS_DIR/dns" +# NOERROR response with one A answer for "example.com": id, flags=0x8180, qd=1, an=1. +printf '\x00\x00\x12\x34\x81\x80\x00\x01\x00\x01\x00\x00\x00\x00\x07example\x03com\x00\x00\x01\x00\x01\xc0\x0c\x00\x01\x00\x01\x00\x00\x00\x3c\x00\x04\x5d\xb8\xd8\x22' > "$SEEDS_DIR/dns/udp_a" +printf '\x00\x00\x12\x34\x81\x80\x00\x01\x00\x01\x00\x00\x00\x00\x07example\x03com\x00\x00\x1c\x00\x01\xc0\x0c\x00\x1c\x00\x01\x00\x00\x00\x3c\x00\x10\x20\x01\x0d\xb8\x00\x00\x00\x00\x00\x00\x00\x00\x00\x00\x00\x01' > "$SEEDS_DIR/dns/udp_aaaa" +printf '\x00\x00\x12\x34\x81\x80\x00\x01\x00\x01\x00\x00\x00\x00\x03www\x07example\x03com\x00\x00\x01\x00\x01\xc0\x0c\x00\x05\x00\x01\x00\x00\x00\x3c\x00\x10\x06target\x07example\x03com\x00' > "$SEEDS_DIR/dns/udp_cname" +printf '\x00\x00\x12\x34\x81\x80\x00\x01\x00\x01\x00\x00\x00\x00\x05_http\x04_tcp\x07example\x03com\x00\x00\x21\x00\x01\xc0\x0c\x00\x21\x00\x01\x00\x00\x00\x3c\x00\x16\x00\x0a\x00\x14\x00\x50\x06target\x07example\x03com\x00' > "$SEEDS_DIR/dns/udp_srv" +printf '\x00\x00\x12\x34\x81\x80\x00\x01\x00\x01\x00\x00\x00\x00\x011\x011\x011\x0127\x07in-addr\x04arpa\x00\x00\x0c\x00\x01\xc0\x0c\x00\x0c\x00\x01\x00\x00\x00\x3c\x00\x0d\x09localhost\x00' > "$SEEDS_DIR/dns/udp_ptr" +printf '\x00\x00\x12\x34\x81\x83\x00\x01\x00\x00\x00\x00\x00\x00\x05nohost\x03com\x00\x00\x01\x00\x01' > "$SEEDS_DIR/dns/udp_nxdomain" +printf '\x00\x00\x12\x34\x81\x82\x00\x01\x00\x00\x00\x00\x00\x00\x05nohost\x03com\x00\x00\x01\x00\x01' > "$SEEDS_DIR/dns/udp_servfail" +printf '\x00\x00\x12\x34\x83\x00\x00\x01\x00\x01\x00\x00\x00\x00\x07example\x03com\x00\x00\x01\x00\x01\xc0\x0c\x00\x01\x00\x01\x00\x00\x00\x3c\x00\x04\x01\x02\x03\x04' > "$SEEDS_DIR/dns/udp_truncated" +printf '\x00\x00\x12\x34\x81\x80\x00\x01\x00\x02\x00\x00\x00\x00\x03www\x07example\x03com\x00\x00\x01\x00\x01\xc0\x0c\x00\x05\x00\x01\x00\x00\x00\x3c\x00\x02\xc0\x10\xc0\x10\x00\x01\x00\x01\x00\x00\x00\x3c\x00\x04\x01\x02\x03\x04' > "$SEEDS_DIR/dns/udp_compress" +printf '\x00\x00\x12\x34\x81\x80\x00\x01\x00\x03\x00\x00\x00\x00\x07example\x03com\x00\x00\x01\x00\x01\xc0\x0c\x00\x01\x00\x01\x00\x00\x00\x3c\x00\x04\x01\x02\x03\x04\xc0\x0c\x00\x01\x00\x01\x00\x00\x00\x3c\x00\x04\x05\x06\x07\x08\xc0\x0c\x00\x01\x00\x01\x00\x00\x00\x3c\x00\x04\x09\x0a\x0b\x0c' > "$SEEDS_DIR/dns/udp_multi_a" +# TCP variants of the same packets (first byte = 1) +printf '\x01\x00\x12\x34\x81\x80\x00\x01\x00\x01\x00\x00\x00\x00\x07example\x03com\x00\x00\x01\x00\x01\xc0\x0c\x00\x01\x00\x01\x00\x00\x00\x3c\x00\x04\x5d\xb8\xd8\x22' > "$SEEDS_DIR/dns/tcp_a" + +(cd "$SEEDS_DIR/dns" && zip -q -j "$OUT/resolver_fuzzer_seed_corpus.zip" ./*) diff --git a/projects/nginx/fuzz/h2_fuzzer.cc b/projects/nginx/fuzz/h2_fuzzer.cc new file mode 100644 index 000000000..e239288c9 --- /dev/null +++ b/projects/nginx/fuzz/h2_fuzzer.cc @@ -0,0 +1,256 @@ +// Copyright 2026 Google LLC +// +// Licensed under the Apache License, Version 2.0 (the "License"); +// you may not use this file except in compliance with the License. +// You may obtain a copy of the License at +// +// http://www.apache.org/licenses/LICENSE-2.0 +// +// Unless required by applicable law or agreed to in writing, software +// distributed under the License is distributed on an "AS IS" BASIS, +// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +// See the License for the specific language governing permissions and +// limitations under the License. +// +//////////////////////////////////////////////////////////////////////////////// +// +// HTTP/2 connection fuzzer. Brings up an in-process nginx cycle with an +// `http2`-enabled unix listener and feeds each fuzz input prefixed with +// the HTTP/2 connection preface so that ngx_http_v2_init() is reached and +// the H2 frame parser in src/http/v2/ngx_http_v2.c (including HPACK +// decoding in ngx_http_v2_table.c) is driven. +// +extern "C" { +#include +#include +#include +#include +} +#include +#include +#include +#include +#include +#include + +static const char configuration[] = +"error_log stderr emerg;\n" +"worker_rlimit_nofile 8192;\n" +"events {\n" +" use epoll;\n" +" worker_connections 2;\n" +" multi_accept off;\n" +" accept_mutex off;\n" +"}\n" +"http {\n" +" server_tokens off;\n" +" default_type application/octet-stream;\n" +" error_log stderr emerg;\n" +" access_log off;\n" +" client_max_body_size 256M;\n" +" client_body_temp_path /tmp/;\n" +" gzip on;\n" +" gzip_types *;\n" +" gzip_min_length 1;\n" +" charset_types *;\n" +" source_charset utf-8;\n" +" charset utf-8;\n" +" keepalive_timeout 1;\n" +" add_header X-Test value;\n" +" server {\n" +" listen unix:nginx_h2.sock http2;\n" +" server_name localhost;\n" +" userid on;\n" +" location / {\n" +" return 200 'hello world';\n" +" }\n" +" location /redir {\n" +" return 302 /target;\n" +" }\n" +" location /addsuf {\n" +" return 200 'hello hello hello hello';\n" +" }\n" +" }\n" +"}\n"; + +#define H2_PREFACE "PRI * HTTP/2.0\r\n\r\nSM\r\n\r\n" +#define H2_PREFACE_LEN (sizeof(H2_PREFACE) - 1) + +static ngx_cycle_t *cycle; +static ngx_log_t ngx_log; +static ngx_open_file_t ngx_log_file; +static char *my_argv[2]; +static char arg1[] = {0, 0xA, 0}; +extern char **environ; +static const char *config_file = "/tmp/h2_config.conf"; + +struct fuzz_buf { + const uint8_t *data; + size_t len; +}; +static struct fuzz_buf g_request; + +static size_t g_preface_offset; + +// First serve the H2 preface, then the fuzz bytes. preface_offset is reset +// at the top of every LLVMFuzzerTestOneInput so partial-preface deliveries +// from one iteration cannot bleed into the next. +static ssize_t h2_recv(ngx_connection_t *c, u_char *buf, size_t size) { + size_t n = 0; + if (g_preface_offset < H2_PREFACE_LEN) { + size_t avail = H2_PREFACE_LEN - g_preface_offset; + size_t take = size < avail ? size : avail; + memcpy(buf, H2_PREFACE + g_preface_offset, take); + g_preface_offset += take; + buf += take; + size -= take; + n += take; + if (size == 0) return (ssize_t) n; + } + if (g_request.len == 0) { + if (n) return (ssize_t) n; + c->read->ready = 0; + return 0; + } + size_t take = size < g_request.len ? size : g_request.len; + memcpy(buf, g_request.data, take); + g_request.data += take; + g_request.len -= take; + n += take; + return (ssize_t) n; +} + +static ngx_int_t add_event(ngx_event_t *ev, ngx_int_t event, ngx_uint_t flags) { + return NGX_OK; +} +static ngx_int_t init_event(ngx_cycle_t *cycle, ngx_msec_t timer) { + return NGX_OK; +} +static ngx_chain_t *send_chain(ngx_connection_t *c, ngx_chain_t *in, + off_t limit) { + // Pretend the entire chain was written. + while (in && in->next) in = in->next; + return NULL; +} +static ssize_t send_void(ngx_connection_t *c, u_char *buf, size_t size) { + return (ssize_t) size; +} +static ssize_t recv_chain_void(ngx_connection_t *c, ngx_chain_t *in, + off_t limit) { + return 0; +} + +static int initialize_nginx(void) { + ngx_cycle_t init_cycle; + + if (access("nginx_h2.sock", F_OK) != -1) { + remove("nginx_h2.sock"); + } + + ngx_debug_init(); + ngx_strerror_init(); + ngx_time_init(); + ngx_regex_init(); + + ngx_log.file = &ngx_log_file; + ngx_log.log_level = NGX_LOG_EMERG; + ngx_log_file.fd = ngx_stderr; + + ngx_memzero(&init_cycle, sizeof(ngx_cycle_t)); + init_cycle.log = &ngx_log; + ngx_cycle = &init_cycle; + init_cycle.pool = ngx_create_pool(1024, &ngx_log); + + my_argv[0] = arg1; + my_argv[1] = NULL; + ngx_argv = ngx_os_argv = my_argv; + ngx_argc = 0; + + char *env_before = environ[0]; + environ[0] = my_argv[0] + 1; + ngx_os_init(&ngx_log); + free(environ[0]); + environ[0] = env_before; + + ngx_crc32_table_init(); + ngx_preinit_modules(); + + FILE *fptr = fopen(config_file, "w"); + fprintf(fptr, "%s", configuration); + fclose(fptr); + init_cycle.conf_file.len = strlen(config_file); + init_cycle.conf_file.data = (unsigned char *) config_file; + + cycle = ngx_init_cycle(&init_cycle); + if (cycle == NULL) return 1; + + ngx_os_status(cycle->log); + ngx_cycle = cycle; + + ngx_event_actions.add = add_event; + ngx_event_actions.init = init_event; + ngx_io.send_chain = send_chain; + ngx_event_flags = 1; + ngx_queue_init(&ngx_posted_accept_events); + ngx_queue_init(&ngx_posted_next_events); + ngx_queue_init(&ngx_posted_events); + ngx_event_timer_init(cycle->log); + return 0; +} + +extern "C" int LLVMFuzzerTestOneInput(const uint8_t *data, size_t size) { + static int init = initialize_nginx(); + if (init != 0) return 0; + if (size > 16384) return 0; + + g_request.data = data; + g_request.len = size; + g_preface_offset = 0; + + // Provide a small pool of free connections so H2 streams (each needing + // a fresh ngx_connection_t) can be allocated. The H2 server caps streams + // at SETTINGS_MAX_CONCURRENT_STREAMS (default 128), so a few extra + // connections meaningfully widen handler/filter coverage. + enum { N_FREE = 8 }; + ngx_event_t read_events[N_FREE] = {}; + ngx_event_t write_events[N_FREE] = {}; + ngx_connection_t pool[N_FREE] = {}; + for (int i = 0; i < N_FREE; i++) { + pool[i].read = &read_events[i]; + pool[i].write = &write_events[i]; + pool[i].data = (i + 1 < N_FREE) ? &pool[i + 1] : NULL; + } + ngx_listening_t *ls = (ngx_listening_t *) ngx_cycle->listening.elts; + ngx_cycle->free_connections = &pool[0]; + ngx_cycle->free_connection_n = N_FREE; + + ngx_connection_t *c = ngx_get_connection(254, &ngx_log); + c->shared = 1; + c->destroyed = 0; + c->type = SOCK_STREAM; + c->pool = ngx_create_pool(256, ngx_cycle->log); + c->sockaddr = ls->sockaddr; + c->listening = ls; + c->recv = h2_recv; + c->send = send_void; + c->send_chain = send_chain; + c->recv_chain = recv_chain_void; + c->log = &ngx_log; + c->pool->log = &ngx_log; + c->read->log = &ngx_log; + c->write->log = &ngx_log; + c->socklen = ls->socklen; + c->local_sockaddr = ls->sockaddr; + c->local_socklen = ls->socklen; + c->data = NULL; + + c->read->ready = 1; + c->write->ready = c->write->delayed = 1; + + ngx_http_init_connection(c); + + if (c->destroyed != 1) { + ngx_close_connection(c); + } + return 0; +} diff --git a/projects/nginx/fuzz/h2_fuzzer.dict b/projects/nginx/fuzz/h2_fuzzer.dict new file mode 100644 index 000000000..d07c76209 --- /dev/null +++ b/projects/nginx/fuzz/h2_fuzzer.dict @@ -0,0 +1,45 @@ +# HTTP/2 frame type bytes +"\x00" +"\x01" +"\x02" +"\x03" +"\x04" +"\x05" +"\x06" +"\x07" +"\x08" +"\x09" + +# Common HPACK literal header names/values +":method" +":path" +":authority" +":scheme" +":status" +"GET" +"POST" +"HEAD" +"/" +"/index.html" +"http" +"https" +"200" +"304" +"404" +"500" + +# HPACK indexed-table single bytes (1xxxxxxx = indexed) +"\x82" +"\x83" +"\x84" +"\x85" +"\x86" +"\x87" +"\x88" + +# Common frame shells +"\x00\x00\x00\x04\x00\x00\x00\x00\x00" +"\x00\x00\x00\x04\x01\x00\x00\x00\x00" +"\x00\x00\x04\x08\x00\x00\x00\x00\x00\x00\x00\x10\x00" +"\x00\x00\x08\x06\x00\x00\x00\x00\x00\x00\x00\x00\x00\x00\x00\x00\x00" +"\x00\x00\x08\x07\x00\x00\x00\x00\x00" diff --git a/projects/nginx/fuzz/inet_fuzzer.cc b/projects/nginx/fuzz/inet_fuzzer.cc new file mode 100644 index 000000000..2e4398de2 --- /dev/null +++ b/projects/nginx/fuzz/inet_fuzzer.cc @@ -0,0 +1,149 @@ +// Copyright 2026 Google LLC +// +// Licensed under the Apache License, Version 2.0 (the "License"); +// you may not use this file except in compliance with the License. +// You may obtain a copy of the License at +// +// http://www.apache.org/licenses/LICENSE-2.0 +// +// Unless required by applicable law or agreed to in writing, software +// distributed under the License is distributed on an "AS IS" BASIS, +// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +// See the License for the specific language governing permissions and +// limitations under the License. +// +//////////////////////////////////////////////////////////////////////////////// +// +// Direct-call fuzzer for nginx's address/URL/string parsing helpers in +// src/core/ngx_inet.c and src/core/ngx_string.c. Each input begins with +// a one-byte selector that picks which parser is exercised. +// +extern "C" { +#include +#include +#include +} +#include +#include +#include +#include + +static ngx_log_t g_log; +static ngx_open_file_t g_log_file; +static int g_init_done; + +static void init_once(void) { + if (g_init_done) return; + g_init_done = 1; + ngx_pagesize = getpagesize(); + for (ngx_uint_t n = ngx_pagesize; n >>= 1; ngx_pagesize_shift++) {} + ngx_cacheline_size = 64; + ngx_time_init(); + ngx_debug_init(); + ngx_strerror_init(); + + g_log.file = &g_log_file; + g_log.log_level = NGX_LOG_EMERG; + g_log_file.fd = ngx_stderr; +} + +extern "C" int LLVMFuzzerTestOneInput(const uint8_t *data, size_t size) { + init_once(); + if (size < 2 || size > 4096) return 0; + + uint8_t selector = data[0]; + const uint8_t *p = data + 1; + size_t n = size - 1; + + // NUL-terminated, writable copy for parsers that expect a C string. + ngx_pool_t *pool = ngx_create_pool(2048, &g_log); + if (pool == NULL) return 0; + + u_char *buf = (u_char *) ngx_palloc(pool, n + 1); + if (buf == NULL) { ngx_destroy_pool(pool); return 0; } + memcpy(buf, p, n); + buf[n] = 0; + + switch (selector % 11) { + case 0: + (void) ngx_inet_addr(buf, n); + break; + case 1: { + u_char out[16]; + (void) ngx_inet6_addr(buf, n, out); + break; + } + case 2: { + ngx_str_t text = { n, buf }; + ngx_cidr_t cidr; + memset(&cidr, 0, sizeof(cidr)); + (void) ngx_ptocidr(&text, &cidr); + break; + } + case 3: { + ngx_addr_t addr; + memset(&addr, 0, sizeof(addr)); + (void) ngx_parse_addr(pool, &addr, buf, n); + break; + } + case 4: { + ngx_addr_t addr; + memset(&addr, 0, sizeof(addr)); + (void) ngx_parse_addr_port(pool, &addr, buf, n); + break; + } + case 5: { + ngx_url_t u; + memset(&u, 0, sizeof(u)); + u.url.data = buf; + u.url.len = n; + u.default_port = 80; + u.no_resolve = 1; + (void) ngx_parse_url(pool, &u); + break; + } + case 6: { + ngx_url_t u; + memset(&u, 0, sizeof(u)); + u.url.data = buf; + u.url.len = n; + u.default_port = 443; + u.no_resolve = 1; + u.listen = 1; + (void) ngx_parse_url(pool, &u); + break; + } + case 7: { + (void) ngx_atoi(buf, n); + (void) ngx_atosz(buf, n); + (void) ngx_atoof(buf, n); + (void) ngx_atotm(buf, n); + (void) ngx_hextoi(buf, n); + break; + } + case 8: { + u_char *dst = (u_char *) ngx_palloc(pool, n + 1); + if (dst != NULL) { + u_char *src = buf; + ngx_unescape_uri(&dst, &src, n, 0); + } + break; + } + case 9: { + uintptr_t need = ngx_escape_uri(NULL, buf, n, NGX_ESCAPE_URI); + u_char *dst = (u_char *) ngx_palloc(pool, n + 2 * need + 1); + if (dst) (void) ngx_escape_uri(dst, buf, n, NGX_ESCAPE_URI); + break; + } + case 10: { + ngx_str_t dst = { 0, NULL }; + ngx_str_t src = { n, buf }; + dst.data = (u_char *) ngx_palloc(pool, ngx_base64_decoded_length(n) + 1); + if (dst.data) (void) ngx_decode_base64(&dst, &src); + break; + } + } + + ngx_destroy_pool(pool); + return 0; +} diff --git a/projects/nginx/fuzz/inet_fuzzer.dict b/projects/nginx/fuzz/inet_fuzzer.dict new file mode 100644 index 000000000..c2fdef5ec --- /dev/null +++ b/projects/nginx/fuzz/inet_fuzzer.dict @@ -0,0 +1,33 @@ +"127.0.0.1" +"0.0.0.0" +"255.255.255.255" +"::" +"::1" +"::ffff:" +"fe80::" +"2001:db8::" +"%eth0" +"/0" +"/8" +"/16" +"/24" +"/32" +"/64" +"/128" +"http://" +"https://" +"unix:" +"localhost" +"example.com" +":80" +":443" +":8080" +"%20" +"%2F" +"%2f" +"%00" +"+" +"=" +"?" +"&" +"#" diff --git a/projects/nginx/fuzz/parser_fuzzer.cc b/projects/nginx/fuzz/parser_fuzzer.cc new file mode 100644 index 000000000..04d0080d7 --- /dev/null +++ b/projects/nginx/fuzz/parser_fuzzer.cc @@ -0,0 +1,289 @@ +// Copyright 2026 Google LLC +// +// Licensed under the Apache License, Version 2.0 (the "License"); +// you may not use this file except in compliance with the License. +// You may obtain a copy of the License at +// +// http://www.apache.org/licenses/LICENSE-2.0 +// +// Unless required by applicable law or agreed to in writing, software +// distributed under the License is distributed on an "AS IS" BASIS, +// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +// See the License for the specific language governing permissions and +// limitations under the License. +// +//////////////////////////////////////////////////////////////////////////////// +// +// Direct-call fuzzer for the HTTP/1 parser primitives in +// src/http/ngx_http_parse.c. The first input byte selects which parser +// to drive; the rest is the parser input. This bypasses the full nginx +// listening / cycle machinery so we can exercise byte-level state +// transitions cheaply. +// +extern "C" { +#include +#include +#include +} +#include +#include +#include +#include + +static ngx_log_t g_log; +static ngx_open_file_t g_log_file; +static int g_init_done; + +static void init_once(void) { + if (g_init_done) return; + g_init_done = 1; + ngx_pagesize = getpagesize(); + for (ngx_uint_t n = ngx_pagesize; n >>= 1; ngx_pagesize_shift++) {} + ngx_cacheline_size = 64; + ngx_time_init(); + ngx_debug_init(); + ngx_strerror_init(); + + g_log.file = &g_log_file; + g_log.log_level = NGX_LOG_EMERG; + g_log_file.fd = ngx_stderr; +} + +// Iteratively feed bytes to a parser that returns NGX_AGAIN until either +// it terminates or we run out of data, so a single fuzz input can drive +// state transitions across the whole input rather than aborting on first +// byte boundary. +static void drive_request_line(ngx_http_request_t *r, u_char *buf, size_t n) { + ngx_buf_t b; + memset(&b, 0, sizeof(b)); + b.start = buf; + b.pos = buf; + b.last = buf + n; + b.end = buf + n; + (void) ngx_http_parse_request_line(r, &b); +} + +static void drive_header_line(ngx_http_request_t *r, u_char *buf, size_t n, + ngx_uint_t allow_underscores) { + ngx_buf_t b; + memset(&b, 0, sizeof(b)); + b.start = buf; + b.pos = buf; + b.last = buf + n; + b.end = buf + n; + + while (b.pos < b.last) { + u_char *prev = b.pos; + ngx_int_t rc = ngx_http_parse_header_line(r, &b, allow_underscores); + if (rc == NGX_HTTP_PARSE_HEADER_DONE || rc == NGX_ERROR + || rc == NGX_HTTP_PARSE_INVALID_HEADER) break; + if (rc == NGX_OK) continue; + if (rc == NGX_AGAIN && b.pos == prev) break; + } +} + +static void drive_status_line(ngx_http_request_t *r, u_char *buf, size_t n) { + ngx_buf_t b; + ngx_http_status_t st; + memset(&st, 0, sizeof(st)); + memset(&b, 0, sizeof(b)); + b.start = buf; b.pos = buf; b.last = buf + n; b.end = buf + n; + + (void) ngx_http_parse_status_line(r, &b, &st); +} + +static void drive_chunked(ngx_http_request_t *r, u_char *buf, size_t n, + ngx_uint_t keep_trailers) { + ngx_buf_t b; + ngx_http_chunked_t ctx; + memset(&ctx, 0, sizeof(ctx)); + memset(&b, 0, sizeof(b)); + b.start = buf; b.pos = buf; b.last = buf + n; b.end = buf + n; + + // Loop until the parser stops making progress so trailers, multiple + // chunks, etc., all get exercised within one input. + for (;;) { + u_char *prev = b.pos; + ngx_int_t rc = ngx_http_parse_chunked(r, &b, &ctx, keep_trailers); + if (rc == NGX_DONE || rc == NGX_ERROR) break; + if (b.pos == prev) break; + } +} + +static void drive_complex_uri(ngx_http_request_t *r, u_char *buf, size_t n, + ngx_uint_t merge_slashes) { + u_char *dst = (u_char *) ngx_palloc(r->pool, n + 1); + if (dst == NULL) return; + r->uri.data = dst; + r->uri_start = buf; + r->uri_end = buf + n; + (void) ngx_http_parse_complex_uri(r, merge_slashes); +} + +static void drive_unsafe_uri(ngx_http_request_t *r, u_char *buf, size_t n) { + ngx_str_t uri = { n, buf }; + ngx_str_t args = { 0, NULL }; + ngx_uint_t flags = 0; + (void) ngx_http_parse_unsafe_uri(r, &uri, &args, &flags); +} + +// ngx_http_parse_uri walks r->uri_start..r->uri_end validating URI chars, +// setting r->complex_uri / quoted_uri / plus_in_uri / empty_path_in_uri. +static void drive_parse_uri(ngx_http_request_t *r, u_char *buf, size_t n) { + r->uri_start = buf; + r->uri_end = buf + n; + (void) ngx_http_parse_uri(r); +} + +// ngx_http_arg(r, name, len, value) finds an arg in r->args. Split input as +// "name\0args" — name has byte count up to first NUL, args is the rest. +static void drive_arg(ngx_http_request_t *r, u_char *buf, size_t n) { + size_t name_len = 0; + while (name_len < n && buf[name_len] != 0) name_len++; + if (name_len == 0 || name_len == n) return; + r->args.data = buf + name_len + 1; + r->args.len = n - name_len - 1; + ngx_str_t value; + (void) ngx_http_arg(r, buf, name_len, &value); +} + +// ngx_http_split_args operates on r and a uri ngx_str_t; it splits at '?' +// into r->args and shrinks the uri. +static void drive_split_args(ngx_http_request_t *r, u_char *buf, size_t n) { + ngx_str_t uri = { n, buf }; + ngx_str_t args = { 0, NULL }; + ngx_http_split_args(r, &uri, &args); +} + +// Build a small linked list of ngx_table_elt_t from input bytes, with the +// first token used as the lookup `name`. Drives one of the multi-header +// scanners. Input layout: "\0:\n:\n..." +static ngx_table_elt_t *build_header_chain(ngx_pool_t *pool, u_char *buf, + size_t n, ngx_str_t *name_out) { + size_t name_len = 0; + while (name_len < n && buf[name_len] != 0) name_len++; + if (name_len == 0 || name_len == n) return NULL; + name_out->len = name_len; + name_out->data = buf; + + u_char *p = buf + name_len + 1; + u_char *end = buf + n; + ngx_table_elt_t *head = NULL, *tail = NULL; + while (p < end) { + u_char *colon = (u_char *) memchr(p, ':', end - p); + if (colon == NULL) break; + u_char *nl = (u_char *) memchr(colon, '\n', end - colon); + if (nl == NULL) nl = end; + ngx_table_elt_t *h = (ngx_table_elt_t *) ngx_pcalloc(pool, sizeof(*h)); + if (h == NULL) break; + h->key.data = p; + h->key.len = colon - p; + h->value.data = colon + 1; + h->value.len = nl - colon - 1; + h->hash = 1; + if (head == NULL) head = h; + if (tail) tail->next = h; + tail = h; + if (nl >= end) break; + p = nl + 1; + } + return head; +} + +static void drive_multi_header(ngx_http_request_t *r, u_char *buf, size_t n) { + ngx_str_t name; + ngx_table_elt_t *head = build_header_chain(r->pool, buf, n, &name); + if (head == NULL) return; + ngx_str_t value; + (void) ngx_http_parse_multi_header_lines(r, head, &name, &value); +} + +static void drive_cookie(ngx_http_request_t *r, u_char *buf, size_t n) { + ngx_str_t name; + ngx_table_elt_t *head = build_header_chain(r->pool, buf, n, &name); + if (head == NULL) return; + ngx_str_t value; + (void) ngx_http_parse_cookie_lines(r, head, &name, &value); +} + +static void drive_set_cookie(ngx_http_request_t *r, u_char *buf, size_t n) { + ngx_str_t name; + ngx_table_elt_t *head = build_header_chain(r->pool, buf, n, &name); + if (head == NULL) return; + ngx_str_t value; + (void) ngx_http_parse_set_cookie_lines(r, head, &name, &value); +} + +extern "C" int LLVMFuzzerTestOneInput(const uint8_t *data, size_t size) { + init_once(); + if (size < 2 || size > 16384) return 0; + + uint8_t selector = data[0]; + const uint8_t *payload = data + 1; + size_t payload_len = size - 1; + + ngx_pool_t *pool = ngx_create_pool(4096, &g_log); + if (pool == NULL) return 0; + + // Make a writable copy (parsers may write through r->lowcase_header, etc., + // but the input buffer itself is only read; still, copy to avoid lifetime + // surprises and to be consistent across selectors). + u_char *buf = (u_char *) ngx_palloc(pool, payload_len + 1); + if (buf == NULL) { ngx_destroy_pool(pool); return 0; } + memcpy(buf, payload, payload_len); + + ngx_http_request_t *r = + (ngx_http_request_t *) ngx_pcalloc(pool, sizeof(ngx_http_request_t)); + ngx_connection_t *c = + (ngx_connection_t *) ngx_pcalloc(pool, sizeof(ngx_connection_t)); + if (r == NULL || c == NULL) { ngx_destroy_pool(pool); return 0; } + r->pool = pool; + r->connection = c; + c->log = &g_log; + c->pool = pool; + + switch (selector % 13) { + case 0: + drive_request_line(r, buf, payload_len); + break; + case 1: + drive_header_line(r, buf, payload_len, /*allow_underscores=*/0); + break; + case 2: + drive_header_line(r, buf, payload_len, /*allow_underscores=*/1); + break; + case 3: + drive_status_line(r, buf, payload_len); + break; + case 4: + drive_chunked(r, buf, payload_len, /*keep_trailers=*/0); + break; + case 5: + drive_complex_uri(r, buf, payload_len, /*merge_slashes=*/1); + break; + case 6: + drive_unsafe_uri(r, buf, payload_len); + break; + case 7: + drive_parse_uri(r, buf, payload_len); + break; + case 8: + drive_arg(r, buf, payload_len); + break; + case 9: + drive_split_args(r, buf, payload_len); + break; + case 10: + drive_multi_header(r, buf, payload_len); + break; + case 11: + drive_cookie(r, buf, payload_len); + break; + case 12: + drive_set_cookie(r, buf, payload_len); + break; + } + + ngx_destroy_pool(pool); + return 0; +} diff --git a/projects/nginx/fuzz/parser_fuzzer.dict b/projects/nginx/fuzz/parser_fuzzer.dict new file mode 100644 index 000000000..1c7d12af3 --- /dev/null +++ b/projects/nginx/fuzz/parser_fuzzer.dict @@ -0,0 +1,60 @@ +# Methods +"GET " +"POST " +"HEAD " +"PUT " +"DELETE " +"OPTIONS " +"PATCH " +"CONNECT " +"PROPFIND " +"PROPPATCH " +"MKCOL " +"COPY " +"MOVE " +"LOCK " +"UNLOCK " +"TRACE " + +# HTTP versions / line endings +" HTTP/1.0\x0d\x0a" +" HTTP/1.1\x0d\x0a" +" HTTP/0.9\x0d\x0a" +"\x0d\x0a" + +# Common headers +"Host:" +"User-Agent:" +"Content-Length:" +"Content-Type:" +"Transfer-Encoding: chunked\x0d\x0a" +"Connection:" +"Connection: keep-alive\x0d\x0a" +"Connection: close\x0d\x0a" +"Cookie:" +"Authorization:" +"X-Forwarded-For:" +"Expect: 100-continue\x0d\x0a" +"Accept-Encoding: gzip\x0d\x0a" + +# URI patterns +"http://" +"https://" +"://" +"%20" +"%2e" +"%2f" +"/../" +"/./" +"//" + +# Chunked transfer encoding tokens +"0\x0d\x0a" +"\x0d\x0a0\x0d\x0a\x0d\x0a" +";" + +# Status lines +"HTTP/1.1 200 OK\x0d\x0a" +"HTTP/1.0 404 Not Found\x0d\x0a" +"HTTP/1.1 301 " +"HTTP/1.1 500 " diff --git a/projects/nginx/fuzz/pp_fuzzer.cc b/projects/nginx/fuzz/pp_fuzzer.cc new file mode 100644 index 000000000..0d4bcfc38 --- /dev/null +++ b/projects/nginx/fuzz/pp_fuzzer.cc @@ -0,0 +1,82 @@ +// Copyright 2026 Google LLC +// +// Licensed under the Apache License, Version 2.0 (the "License"); +// you may not use this file except in compliance with the License. +// You may obtain a copy of the License at +// +// http://www.apache.org/licenses/LICENSE-2.0 +// +// Unless required by applicable law or agreed to in writing, software +// distributed under the License is distributed on an "AS IS" BASIS, +// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +// See the License for the specific language governing permissions and +// limitations under the License. +// +//////////////////////////////////////////////////////////////////////////////// +extern "C" { +#include +#include +#include +} +#include +#include +#include +#include + +static ngx_log_t g_log; +static ngx_open_file_t g_log_file; +static int g_init_done; + +static const char *tlv_names[] = { + "alpn", "authority", "unique_id", "ssl_version", "ssl_cn", "ssl_cipher", + "ssl_sig_alg", "ssl_key_alg", "netns", "0xea", +}; + +static void init_once(void) { + if (g_init_done) return; + g_init_done = 1; + ngx_pagesize = getpagesize(); + for (ngx_uint_t n = ngx_pagesize; n >>= 1; ngx_pagesize_shift++) {} + ngx_cacheline_size = 64; + ngx_time_init(); + ngx_debug_init(); + ngx_strerror_init(); + + g_log.file = &g_log_file; + g_log.log_level = NGX_LOG_EMERG; + g_log_file.fd = ngx_stderr; +} + +extern "C" int LLVMFuzzerTestOneInput(const uint8_t *data, size_t size) { + init_once(); + if (size == 0 || size > 4096) return 0; + + ngx_pool_t *pool = ngx_create_pool(2048, &g_log); + if (pool == NULL) return 0; + + ngx_connection_t c; + memset(&c, 0, sizeof(c)); + c.log = &g_log; + c.pool = pool; + + // Make a writable copy of the input for ngx_proxy_protocol_read, + // which may overwrite bytes (e.g. NUL-terminating the v1 header line). + u_char *buf = (u_char *) ngx_palloc(pool, size); + if (buf == NULL) { ngx_destroy_pool(pool); return 0; } + memcpy(buf, data, size); + + u_char *last = buf + size; + u_char *p = ngx_proxy_protocol_read(&c, buf, last); + + if (p != NULL && c.proxy_protocol != NULL) { + // Probe TLV lookup; this also exercises get_tlv parsing for v2. + for (size_t i = 0; i < sizeof(tlv_names)/sizeof(tlv_names[0]); i++) { + ngx_str_t name = { strlen(tlv_names[i]), (u_char *) tlv_names[i] }; + ngx_str_t value; + ngx_proxy_protocol_get_tlv(&c, &name, &value); + } + } + + ngx_destroy_pool(pool); + return 0; +} diff --git a/projects/nginx/fuzz/pp_fuzzer.dict b/projects/nginx/fuzz/pp_fuzzer.dict new file mode 100644 index 000000000..c7737b250 --- /dev/null +++ b/projects/nginx/fuzz/pp_fuzzer.dict @@ -0,0 +1,32 @@ +# PROXY protocol v1 magic + minimal v1 examples +"PROXY " +"TCP4 " +"TCP6 " +"UNKNOWN" +"\x0d\x0a" +"PROXY TCP4 1.1.1.1 2.2.2.2 80 81\x0d\x0a" +"PROXY TCP6 ::1 ::1 80 81\x0d\x0a" + +# PROXY protocol v2 magic ("\r\n\r\n\0\r\nQUIT\n") +"\x0d\x0a\x0d\x0a\x00\x0d\x0aQUIT\x0a" + +# v2 TLV type bytes (PP2_TYPE_ALPN/AUTHORITY/CRC32C/NOOP/UNIQUE_ID/SSL/...) +"\x01" +"\x02" +"\x03" +"\x04" +"\x05" +"\x20" +"\x21" +"\x22" +"\x23" +"\x24" +"\x25" +"\x30" +"\xea" + +# v2 family/protocol bytes +"\x11" +"\x12" +"\x21" +"\x22" diff --git a/projects/nginx/fuzz/resolver_fuzzer.cc b/projects/nginx/fuzz/resolver_fuzzer.cc new file mode 100644 index 000000000..aa080eb1c --- /dev/null +++ b/projects/nginx/fuzz/resolver_fuzzer.cc @@ -0,0 +1,111 @@ +// Copyright 2026 Google LLC +// +// Licensed under the Apache License, Version 2.0 (the "License"); +// you may not use this file except in compliance with the License. +// You may obtain a copy of the License at +// +// http://www.apache.org/licenses/LICENSE-2.0 +// +// Unless required by applicable law or agreed to in writing, software +// distributed under the License is distributed on an "AS IS" BASIS, +// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +// See the License for the specific language governing permissions and +// limitations under the License. +// +//////////////////////////////////////////////////////////////////////////////// +// +// DNS-response fuzzer for src/core/ngx_resolver.c. Drives +// ngx_resolver_process_response() directly with a fuzz-controlled DNS packet +// against a minimally-mocked ngx_resolver_t. build.sh patches `static` off +// ngx_resolver_process_response so it's reachable externally. +// +extern "C" { +#include +#include +#include +#include + +// Exposed by the build.sh sed patch (was static). +void ngx_resolver_process_response(ngx_resolver_t *r, u_char *buf, + size_t n, ngx_uint_t tcp); +} +#include +#include +#include +#include + +static ngx_log_t g_log; +static ngx_open_file_t g_log_file; +static int g_init_done; + +static void init_once(void) { + if (g_init_done) return; + g_init_done = 1; + ngx_pagesize = getpagesize(); + for (ngx_uint_t n = ngx_pagesize; n >>= 1; ngx_pagesize_shift++) {} + ngx_cacheline_size = 64; + ngx_time_init(); + ngx_debug_init(); + ngx_strerror_init(); + g_log.file = &g_log_file; + g_log.log_level = NGX_LOG_EMERG; + g_log_file.fd = ngx_stderr; +} + +static void noop_rbtree_insert(ngx_rbtree_node_t *temp, + ngx_rbtree_node_t *node, + ngx_rbtree_node_t *sentinel) { + // The resolver inserts via specific insert_value callbacks; for fuzzing + // we never insert nodes, so this is unused. Stub keeps init happy. + (void) temp; (void) node; (void) sentinel; +} + +extern "C" int LLVMFuzzerTestOneInput(const uint8_t *data, size_t size) { + init_once(); + if (size < 2 || size > 65536) return 0; + + uint8_t tcp_flag = data[0] & 1; + const uint8_t *payload = data + 1; + size_t payload_len = size - 1; + + ngx_pool_t *pool = ngx_create_pool(8192, &g_log); + if (pool == NULL) return 0; + + ngx_resolver_t *r = (ngx_resolver_t *) ngx_pcalloc(pool, sizeof(*r)); + if (r == NULL) { ngx_destroy_pool(pool); return 0; } + r->log = &g_log; + r->log_level = NGX_LOG_EMERG; + r->ipv4 = 1; +#if (NGX_HAVE_INET6) + r->ipv6 = 1; +#endif + + ngx_rbtree_init(&r->name_rbtree, &r->name_sentinel, noop_rbtree_insert); + ngx_rbtree_init(&r->srv_rbtree, &r->srv_sentinel, noop_rbtree_insert); + ngx_rbtree_init(&r->addr_rbtree, &r->addr_sentinel, noop_rbtree_insert); +#if (NGX_HAVE_INET6) + ngx_rbtree_init(&r->addr6_rbtree, &r->addr6_sentinel, noop_rbtree_insert); +#endif + + ngx_queue_init(&r->name_resend_queue); + ngx_queue_init(&r->srv_resend_queue); + ngx_queue_init(&r->addr_resend_queue); + ngx_queue_init(&r->name_expire_queue); + ngx_queue_init(&r->srv_expire_queue); + ngx_queue_init(&r->addr_expire_queue); +#if (NGX_HAVE_INET6) + ngx_queue_init(&r->addr6_resend_queue); + ngx_queue_init(&r->addr6_expire_queue); +#endif + + // ngx_resolver_process_response writes to the buffer in some paths + // (e.g. unescaping during name parsing), so feed a writable copy. + u_char *buf = (u_char *) ngx_palloc(pool, payload_len); + if (buf == NULL) { ngx_destroy_pool(pool); return 0; } + memcpy(buf, payload, payload_len); + + ngx_resolver_process_response(r, buf, payload_len, tcp_flag); + + ngx_destroy_pool(pool); + return 0; +} diff --git a/projects/nginx/fuzz/resolver_fuzzer.dict b/projects/nginx/fuzz/resolver_fuzzer.dict new file mode 100644 index 000000000..aede85cd4 --- /dev/null +++ b/projects/nginx/fuzz/resolver_fuzzer.dict @@ -0,0 +1,44 @@ +# DNS header: id(2) flags(2) qdcount(2) ancount(2) nscount(2) arcount(2) +# Standard query response, no error: flags = 0x8000 +"\x80\x00" +# Standard query response, NOERROR with RD/RA: flags = 0x8180 +"\x81\x80" +# Truncated response: flags = 0x8200 +"\x82\x00" +# Response with NXDOMAIN (rcode=3) +"\x81\x83" +# Response with FORMERR (rcode=1) +"\x80\x01" +# Response with SERVFAIL (rcode=2) +"\x80\x02" + +# QTYPEs (big-endian) +"\x00\x01" +"\x00\x05" +"\x00\x0c" +"\x00\x0f" +"\x00\x10" +"\x00\x1c" +"\x00\x21" + +# QCLASS IN +"\x00\x01" + +# Compression pointer (top 2 bits = 11) +"\xc0\x0c" +"\xc0\x00" + +# Common labels (length-prefixed) +"\x03www" +"\x07example" +"\x03com" +"\x03org" +"\x03net" +"\x00" + +# Terminator +"\x00" + +# Common TTLs +"\x00\x00\x00\x3c" +"\x00\x00\x01\x2c" diff --git a/projects/nginx/fuzz/upstream_fuzzer.cc b/projects/nginx/fuzz/upstream_fuzzer.cc new file mode 100644 index 000000000..4d66f1252 --- /dev/null +++ b/projects/nginx/fuzz/upstream_fuzzer.cc @@ -0,0 +1,302 @@ +// Copyright 2026 Google LLC +// +// Licensed under the Apache License, Version 2.0 (the "License"); +// you may not use this file except in compliance with the License. +// You may obtain a copy of the License at +// +// http://www.apache.org/licenses/LICENSE-2.0 +// +// Unless required by applicable law or agreed to in writing, software +// distributed under the License is distributed on an "AS IS" BASIS, +// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +// See the License for the specific language governing permissions and +// limitations under the License. +// +//////////////////////////////////////////////////////////////////////////////// +// +// Upstream-side HTTP/1 response fuzzer. Brings up an in-process nginx cycle +// that proxies every request to an upstream group, then for each iteration: +// 1. Feeds a *fixed* valid HTTP/1.1 request from the "client" side so the +// request reliably reaches the upstream pipeline. +// 2. Feeds the fuzz bytes verbatim as the "upstream" HTTP/1 response. +// +// This concentrates fuzzing on ngx_http_upstream.c + ngx_http_proxy_module.c +// (and the response filter chain) rather than the inbound HTTP parser. +// +extern "C" { +#include +#include +#include +#include +} +#include +#include +#include +#include +#include +#include + +static const char configuration[] = +"error_log stderr emerg;\n" +"worker_rlimit_nofile 8192;\n" +"events {\n" +" use epoll;\n" +" worker_connections 4;\n" +" multi_accept off;\n" +" accept_mutex off;\n" +"}\n" +"http {\n" +" server_tokens off;\n" +" default_type application/octet-stream;\n" +" error_log stderr emerg;\n" +" access_log off;\n" +" client_max_body_size 256M;\n" +" client_body_temp_path /tmp/;\n" +" proxy_temp_path /tmp/;\n" +" proxy_buffer_size 24K;\n" +" proxy_max_temp_file_size 0;\n" +" proxy_buffers 8 4K;\n" +" proxy_busy_buffers_size 28K;\n" +" proxy_buffering off;\n" +" gzip on;\n" +" gzip_types *;\n" +" upstream up { server 127.0.0.1:1010 max_fails=0; }\n" +" server {\n" +" listen unix:nginx_up.sock;\n" +" proxy_next_upstream off;\n" +" proxy_read_timeout 5m;\n" +" proxy_http_version 1.1;\n" +" location / {\n" +" proxy_pass http://up;\n" +" proxy_set_header Host upstream.local;\n" +" proxy_set_header Connection '';\n" +" chunked_transfer_encoding off;\n" +" proxy_buffering off;\n" +" proxy_cache off;\n" +" }\n" +" }\n" +"}\n"; + +// A valid HTTP/1.1 request that reliably routes to the upstream location. +static const char kClientRequest[] = +"GET /resource?x=1 HTTP/1.1\r\n" +"Host: front.local\r\n" +"User-Agent: ufuzz\r\n" +"Accept: */*\r\n" +"X-Forwarded-For: 1.2.3.4\r\n" +"Cookie: session=abc\r\n" +"\r\n"; + +static ngx_cycle_t *cycle; +static ngx_log_t ngx_log; +static ngx_open_file_t ngx_log_file; +static char *my_argv[2]; +static char arg1[] = {0, 0xA, 0}; +extern char **environ; +static const char *config_file = "/tmp/upstream_config.conf"; + +struct fuzz_buf { + const uint8_t *data; + size_t len; +}; +static struct fuzz_buf g_client; // fixed request bytes (re-served each iter) +static size_t g_client_off; +static struct fuzz_buf g_reply; // fuzzed upstream response + +static ngx_http_upstream_t *upstream_ctx; +static ngx_http_request_t *req_for_reply; +static ngx_http_cleanup_t cln_new; +static int cln_added; + +static void cleanup_reply(void *data) { + req_for_reply = NULL; +} + +// Client → nginx (replays kClientRequest verbatim per iteration). +static ssize_t client_recv(ngx_connection_t *c, u_char *buf, size_t size) { + if (g_client_off >= g_client.len) { + c->read->ready = 0; + return 0; + } + size_t take = g_client.len - g_client_off; + if (take > size) take = size; + memcpy(buf, g_client.data + g_client_off, take); + g_client_off += take; + return (ssize_t) take; +} + +// Upstream → nginx (fuzz input). +static ssize_t reply_recv(ngx_connection_t *c, u_char *buf, size_t size) { + req_for_reply = (ngx_http_request_t *) c->data; + if (!cln_added && req_for_reply) { + cln_added = 1; + cln_new.handler = cleanup_reply; + cln_new.data = NULL; + cln_new.next = req_for_reply->cleanup; + req_for_reply->cleanup = &cln_new; + } + if (req_for_reply) upstream_ctx = req_for_reply->upstream; + if (g_reply.len == 0) { + c->read->ready = 0; + return 0; + } + size_t take = g_reply.len; + if (take > size) take = size; + memcpy(buf, g_reply.data, take); + g_reply.data += take; + g_reply.len -= take; + return (ssize_t) take; +} + +static ngx_int_t add_event(ngx_event_t *ev, ngx_int_t e, ngx_uint_t f) { + return NGX_OK; +} +static ngx_int_t init_event(ngx_cycle_t *cycle, ngx_msec_t t) { + return NGX_OK; +} + +// When nginx writes (request to upstream OR response to client), we don't +// actually send. For the upstream side, mark its read as ready and swap in +// reply_recv so the next read consumes our fuzz bytes. +static ngx_chain_t *send_chain(ngx_connection_t *c, ngx_chain_t *in, + off_t limit) { + c->read->ready = 1; + c->recv = reply_recv; + while (in && in->next) in = in->next; + return NULL; +} + +extern "C" long int invalid_call(ngx_connection_s *a, ngx_chain_s *b, + long int cc) { + return 0; +} + +static int initialize_nginx(void) { + ngx_cycle_t init_cycle; + + if (access("nginx_up.sock", F_OK) != -1) { + remove("nginx_up.sock"); + } + + ngx_debug_init(); + ngx_strerror_init(); + ngx_time_init(); + ngx_regex_init(); + + ngx_log.file = &ngx_log_file; + ngx_log.log_level = NGX_LOG_EMERG; + ngx_log_file.fd = ngx_stderr; + + ngx_memzero(&init_cycle, sizeof(ngx_cycle_t)); + init_cycle.log = &ngx_log; + ngx_cycle = &init_cycle; + init_cycle.pool = ngx_create_pool(1024, &ngx_log); + + my_argv[0] = arg1; + my_argv[1] = NULL; + ngx_argv = ngx_os_argv = my_argv; + ngx_argc = 0; + + char *env_before = environ[0]; + environ[0] = my_argv[0] + 1; + ngx_os_init(&ngx_log); + free(environ[0]); + environ[0] = env_before; + + ngx_crc32_table_init(); + ngx_preinit_modules(); + + FILE *fptr = fopen(config_file, "w"); + fprintf(fptr, "%s", configuration); + fclose(fptr); + init_cycle.conf_file.len = strlen(config_file); + init_cycle.conf_file.data = (unsigned char *) config_file; + + cycle = ngx_init_cycle(&init_cycle); + if (cycle == NULL) return 1; + + ngx_os_status(cycle->log); + ngx_cycle = cycle; + + ngx_event_actions.add = add_event; + ngx_event_actions.init = init_event; + ngx_io.send_chain = send_chain; + ngx_event_flags = 1; + ngx_queue_init(&ngx_posted_accept_events); + ngx_queue_init(&ngx_posted_next_events); + ngx_queue_init(&ngx_posted_events); + ngx_event_timer_init(cycle->log); + return 0; +} + +extern "C" int LLVMFuzzerTestOneInput(const uint8_t *data, size_t size) { + static int init = initialize_nginx(); + if (init != 0) return 0; + if (size > 65536) return 0; + + g_client.data = (const uint8_t *) kClientRequest; + g_client.len = sizeof(kClientRequest) - 1; + g_client_off = 0; + g_reply.data = data; + g_reply.len = size; + req_for_reply = NULL; + upstream_ctx = NULL; + cln_added = 0; + + ngx_event_t read_event1 = {}; + ngx_event_t write_event1 = {}; + ngx_connection_t local1 = {}; + ngx_event_t read_event2 = {}; + ngx_event_t write_event2 = {}; + ngx_connection_t local2 = {}; + + ngx_listening_t *ls = (ngx_listening_t *) ngx_cycle->listening.elts; + + local1.read = &read_event1; + local1.write = &write_event1; + local2.read = &read_event2; + local2.write = &write_event2; + local2.send_chain = send_chain; + + ngx_cycle->free_connections = &local1; + local1.data = &local2; + ngx_cycle->free_connection_n = 2; + + ngx_connection_t *c = ngx_get_connection(253, &ngx_log); + c->shared = 1; + c->destroyed = 0; + c->type = SOCK_STREAM; + c->pool = ngx_create_pool(256, ngx_cycle->log); + c->sockaddr = ls->sockaddr; + c->listening = ls; + c->recv = client_recv; + c->send_chain = send_chain; + c->send = (ngx_send_pt) invalid_call; + c->recv_chain = (ngx_recv_chain_pt) invalid_call; + c->log = &ngx_log; + c->pool->log = &ngx_log; + c->read->log = &ngx_log; + c->write->log = &ngx_log; + c->socklen = ls->socklen; + c->local_sockaddr = ls->sockaddr; + c->local_socklen = ls->socklen; + c->data = NULL; + + read_event1.ready = 1; + write_event1.ready = write_event1.delayed = 1; + + ngx_http_init_connection(c); + + if (c->destroyed != 1) { + if (c->read->data != NULL) { + ngx_connection_t *c2 = (ngx_connection_t *) c->read->data; + ngx_http_request_t *r = (ngx_http_request_t *) c2->data; + if (r) { + r->cleanup = NULL; + ngx_http_finalize_request(r, NGX_DONE); + } + } + ngx_close_connection(c); + } + return 0; +} diff --git a/projects/nginx/fuzz/upstream_fuzzer.dict b/projects/nginx/fuzz/upstream_fuzzer.dict new file mode 100644 index 000000000..103120769 --- /dev/null +++ b/projects/nginx/fuzz/upstream_fuzzer.dict @@ -0,0 +1,49 @@ +# Status lines +"HTTP/1.0 " +"HTTP/1.1 " +"HTTP/1.1 200 OK\x0d\x0a" +"HTTP/1.1 204 No Content\x0d\x0a" +"HTTP/1.1 301 Moved Permanently\x0d\x0a" +"HTTP/1.1 302 Found\x0d\x0a" +"HTTP/1.1 304 Not Modified\x0d\x0a" +"HTTP/1.1 400 Bad Request\x0d\x0a" +"HTTP/1.1 404 Not Found\x0d\x0a" +"HTTP/1.1 500 Internal Server Error\x0d\x0a" +"HTTP/1.1 502 Bad Gateway\x0d\x0a" +"HTTP/1.1 100 Continue\x0d\x0a\x0d\x0a" + +# Response headers (upstream-relevant) +"Content-Length:" +"Content-Length: 0\x0d\x0a" +"Content-Encoding:" +"Content-Encoding: gzip\x0d\x0a" +"Transfer-Encoding: chunked\x0d\x0a" +"Connection: close\x0d\x0a" +"Connection: keep-alive\x0d\x0a" +"Server:" +"Date:" +"Last-Modified:" +"ETag:" +"Cache-Control:" +"Set-Cookie:" +"Location:" +"WWW-Authenticate:" +"Content-Type:" +"Content-Type: text/html\x0d\x0a" +"X-Accel-Redirect:" +"X-Accel-Limit-Rate:" +"X-Accel-Buffering:" +"X-Accel-Charset:" +"X-Accel-Expires:" +"Vary:" +"Refresh:" +"Trailer:" +"Upgrade:" + +# Chunked-encoding tokens +"\x0d\x0a" +"0\x0d\x0a\x0d\x0a" +";chunk-ext=val" + +# gzip magic so the gzip-content path is reachable +"\x1f\x8b\x08\x00" diff --git a/projects/nginx/fuzz/upstream_fuzzer.options b/projects/nginx/fuzz/upstream_fuzzer.options new file mode 100644 index 000000000..d5de8e7bc --- /dev/null +++ b/projects/nginx/fuzz/upstream_fuzzer.options @@ -0,0 +1,5 @@ +[libfuzzer] +detect_leaks=0 + +[asan] +detect_leaks=0 diff --git a/projects/nginx/make_fuzzers b/projects/nginx/make_fuzzers index 433c491a8..80d60de55 100644 --- a/projects/nginx/make_fuzzers +++ b/projects/nginx/make_fuzzers @@ -16,7 +16,7 @@ ngx_objs=`echo objs/src/fuzz/wrappers.o $ngx_all_objs $ngx_modules_obj \ cat << END >> $NGX_MAKEFILE -fuzzers: objs/http_request_fuzzer +fuzzers: objs/http_request_fuzzer objs/pp_fuzzer objs/parser_fuzzer objs/inet_fuzzer objs/h2_fuzzer objs/upstream_fuzzer objs/resolver_fuzzer objs/src/fuzz/wrappers.o: \$(CC) $ngx_compile_opt \$(CFLAGS) -o objs/src/fuzz/wrappers.o src/fuzz/wrappers.c @@ -36,4 +36,52 @@ objs/http_request_fuzzer: $ngx_deps_fuzz \$(LIB_FUZZING_ENGINE) $ngx_libs$ngx_link$ngx_main_link -lcrypt $ngx_long_end +objs/pp_fuzzer: $ngx_deps_fuzz + \$(CXX) \$(CXXFLAGS) -DNDEBUG src/fuzz/pp_fuzzer.cc \ + -o objs/pp_fuzzer \ + \$(CORE_INCS) \$(HTTP_INCS) \ + $ngx_binexit$ngx_long_cont$ngx_objs \ + \$(LIB_FUZZING_ENGINE) $ngx_libs$ngx_link$ngx_main_link -lcrypt +$ngx_long_end + +objs/parser_fuzzer: $ngx_deps_fuzz + \$(CXX) \$(CXXFLAGS) -DNDEBUG src/fuzz/parser_fuzzer.cc \ + -o objs/parser_fuzzer \ + \$(CORE_INCS) \$(HTTP_INCS) \ + $ngx_binexit$ngx_long_cont$ngx_objs \ + \$(LIB_FUZZING_ENGINE) $ngx_libs$ngx_link$ngx_main_link -lcrypt +$ngx_long_end + +objs/inet_fuzzer: $ngx_deps_fuzz + \$(CXX) \$(CXXFLAGS) -DNDEBUG src/fuzz/inet_fuzzer.cc \ + -o objs/inet_fuzzer \ + \$(CORE_INCS) \$(HTTP_INCS) \ + $ngx_binexit$ngx_long_cont$ngx_objs \ + \$(LIB_FUZZING_ENGINE) $ngx_libs$ngx_link$ngx_main_link -lcrypt +$ngx_long_end + +objs/h2_fuzzer: $ngx_deps_fuzz + \$(CXX) \$(CXXFLAGS) -DNDEBUG src/fuzz/h2_fuzzer.cc \ + -o objs/h2_fuzzer \ + \$(CORE_INCS) \$(HTTP_INCS) \ + $ngx_binexit$ngx_long_cont$ngx_objs \ + \$(LIB_FUZZING_ENGINE) $ngx_libs$ngx_link$ngx_main_link -lcrypt +$ngx_long_end + +objs/upstream_fuzzer: $ngx_deps_fuzz + \$(CXX) \$(CXXFLAGS) -DNDEBUG src/fuzz/upstream_fuzzer.cc \ + -o objs/upstream_fuzzer \ + \$(CORE_INCS) \$(HTTP_INCS) \ + $ngx_binexit$ngx_long_cont$ngx_objs \ + \$(LIB_FUZZING_ENGINE) $ngx_libs$ngx_link$ngx_main_link -lcrypt +$ngx_long_end + +objs/resolver_fuzzer: $ngx_deps_fuzz + \$(CXX) \$(CXXFLAGS) -DNDEBUG src/fuzz/resolver_fuzzer.cc \ + -o objs/resolver_fuzzer \ + \$(CORE_INCS) \$(HTTP_INCS) \ + $ngx_binexit$ngx_long_cont$ngx_objs \ + \$(LIB_FUZZING_ENGINE) $ngx_libs$ngx_link$ngx_main_link -lcrypt +$ngx_long_end + END