Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
31 changes: 31 additions & 0 deletions .github/workflows/ci.yaml
Original file line number Diff line number Diff line change
Expand Up @@ -117,6 +117,37 @@ jobs:
token: ${{ env.CODECOV_TOKEN }}
fail_ci_if_error: false

# Go fuzzes one target per invocation, and its corpus of "interesting" inputs is
# a machine-local cache by design — not something to commit. The seed corpus
# (f.Add, plus any crasher checked in) already runs as an ordinary test in the
# `test` job; this job is the part that explores inputs nobody has seen yet.
#
# Time-boxed rather than exhaustive: the point is steady pressure on every PR.
# On a failure the toolchain writes the reproducer under testdata/fuzz/, so it
# is uploaded — committing that file turns the crash into a permanent test.
fuzz:
name: fuzz
runs-on: ubuntu-latest
timeout-minutes: 10
steps:
- uses: actions/checkout@v4

- uses: actions/setup-go@v5
with:
go-version-file: go.mod
cache: true

- name: Fuzz
run: make fuzz FUZZTIME=60s

- name: Upload crash reproducers
if: failure()
uses: actions/upload-artifact@v4
with:
name: fuzz-crashers
path: "**/testdata/fuzz/**"
if-no-files-found: ignore

# Lint results are platform-independent, so this runs once rather than three
# times. It also exercises a Makefile target, which keeps CI and the local
# `make check` entry point from drifting apart.
Expand Down
18 changes: 16 additions & 2 deletions Makefile
Original file line number Diff line number Diff line change
Expand Up @@ -54,9 +54,23 @@ fmt: ## Format the tree
fmt-check: ## Fail if anything is unformatted
@out=$$(gofmt -l .); if [ -n "$$out" ]; then echo "unformatted:"; echo "$$out"; exit 1; fi

# Go fuzzes one target per invocation, so discover them rather than naming one:
# a hardcoded target silently passes once the package it names moves or is not
# written yet. Exits non-zero if it finds nothing, for the same reason.
FUZZTIME ?= 30s

.PHONY: fuzz
fuzz: ## Short fuzz run over the JSONC parser (DCL-10)
go test -run=NONE -fuzz=FuzzParse -fuzztime=60s ./pkg/jsonc
fuzz: ## Bounded fuzz run over every Fuzz target (FUZZTIME=30s)
@set -e; \
found=0; \
for pkg in $$(go list ./...); do \
for fn in $$(go test -list='Fuzz.*' $$pkg 2>/dev/null | grep '^Fuzz' || true); do \
found=1; \
echo "==> $$pkg $$fn ($(FUZZTIME))"; \
go test -run='^$$' -fuzz="^$$fn$$" -fuzztime=$(FUZZTIME) $$pkg; \
done; \
done; \
if [ $$found -eq 0 ]; then echo "no fuzz targets found"; exit 1; fi

.PHONY: check
check: fmt-check vet lint test ## Everything CI runs
Expand Down
43 changes: 43 additions & 0 deletions pkg/position/bench_test.go
Original file line number Diff line number Diff line change
@@ -0,0 +1,43 @@
package position_test

import (
"os"
"path/filepath"
"testing"

"github.com/lonhutt/dcx/pkg/position"
)

// These exist to answer one question if it is ever raised: is the column scan
// worth optimising? Measured first, optimised only if the numbers say so.
//
// For scale: the worst line in the 86 KB fixture is 186 characters, so the scan
// from line start to offset is bounded by that regardless of file size.

func BenchmarkNew(b *testing.B) {
src := loadFixture(b)
b.ReportAllocs()
b.SetBytes(int64(len(src)))
for b.Loop() {
_ = position.New(src)
}
}

func BenchmarkUTF16(b *testing.B) {
src := loadFixture(b)
ix := position.New(src)
off := position.Offset(len(src) / 2)
b.ReportAllocs()
for b.Loop() {
_, _ = ix.UTF16(off)
}
}

func loadFixture(tb testing.TB) []byte {
tb.Helper()
src, err := os.ReadFile(filepath.Join("testdata", "large-commented.jsonc"))
if err != nil {
tb.Fatalf("read fixture: %v", err)
}
return src
}
73 changes: 73 additions & 0 deletions pkg/position/fuzz_test.go
Original file line number Diff line number Diff line change
@@ -0,0 +1,73 @@
package position_test

import (
"testing"

"github.com/lonhutt/dcx/pkg/position"
)

// FuzzLineIndex asserts the two properties that must hold for arbitrary bytes:
// nothing panics, and every result stays in bounds and round trips.
//
// Malformed UTF-8 is a normal input for a linter, not an exotic one. The oracle
// and the implementation agree on it as long as both advance one byte per
// invalid byte, which is what utf8.DecodeRune does when it returns RuneError.
func FuzzLineIndex(f *testing.F) {
for _, seed := range []string{
"",
"{}",
"a\n",
"a\r\nb\rc",
mixed,
"😀\n中\né",
"a\xffb", // invalid byte
"\xe4", // truncated multi-byte sequence
"\ufeff{}\n", // BOM
} {
f.Add([]byte(seed))
}

f.Fuzz(func(t *testing.T, src []byte) {
ix := position.New(src)
orc := newOracle(src)

if n := ix.LineCount(); n < 1 {
t.Fatalf("LineCount() = %d, must be at least 1 even for empty input", n)
}
if got, want := ix.LineCount(), orc.lineCount(); got != want {
t.Fatalf("LineCount() = %d, oracle says %d", got, want)
}

for _, off := range runeBoundaries(src) {
// Checked against the oracle as well as through its own inverse: a
// defect shared by UTF8 and OffsetUTF8 satisfies the round trip while
// still producing the wrong column.
wantLine, wantCol := orc.utf8At(off)
l, c := ix.UTF8(position.Offset(off))
if l != wantLine || c != wantCol {
t.Fatalf("UTF8(%d) = (%d,%d), oracle says (%d,%d)", off, l, c, wantLine, wantCol)
}
if back := ix.OffsetUTF8(l, c); back != position.Offset(off) {
t.Fatalf("utf8 round trip: offset %d -> (%d,%d) -> %d", off, l, c, back)
}
Comment thread
coderabbitai[bot] marked this conversation as resolved.

line, char := ix.UTF16(position.Offset(off))
if line < 0 || line >= ix.LineCount() {
t.Fatalf("UTF16(%d) line = %d, out of range [0,%d)", off, line, ix.LineCount())
}
if char < 0 {
t.Fatalf("UTF16(%d) char = %d, negative", off, char)
}

if back := ix.OffsetUTF16(line, char); back != position.Offset(off) {
t.Fatalf("round trip: offset %d -> (%d,%d) -> %d", off, line, char, back)
}

// Agreement with the reference implementation must hold here too,
// not just on the curated corpus.
if wl, wc := orc.utf16At(off); wl != line || wc != char {
t.Fatalf("UTF16(%d) = (%d,%d), oracle says (%d,%d)", off, line, char, wl, wc)
}
}
})
}
106 changes: 106 additions & 0 deletions pkg/position/oracle_test.go
Original file line number Diff line number Diff line change
@@ -0,0 +1,106 @@
package position_test

import (
"unicode/utf16"
"unicode/utf8"
)

// This file is the reference implementation the real one is tested against.
//
// The per-offset arithmetic here is deliberately the slowest, most obviously
// correct code that could work: it re-encodes the line prefix through []rune and
// utf16.Encode on every call, because that is the *definition* of an LSP column
// rather than an optimisation of it. Nothing in that path should ever be made
// clever. Its only job is to be so simple that when it disagrees with
// pkg/position, the bug is in pkg/position.
//
// The one concession is that line splitting is hoisted into newOracle instead of
// being redone per call. That is not cleverness, it is the difference between a
// 0.3s test and a 30s one on the 86 KB fixture: splitting is O(n) and was being
// run once per offset, making the whole check O(n^2).

type oracle struct {
src []byte
starts []int // byte offset at which each line begins
}

// newOracle splits src into lines, treating "\n", "\r\n" and a lone "\r" as
// terminators — which is what VS Code and other LSP clients do. If this
// disagrees with the real implementation, every position after the first stray
// carriage return is silently wrong.
func newOracle(src []byte) *oracle {
starts := []int{0}
for i := 0; i < len(src); {
switch src[i] {
case '\n':
i++
starts = append(starts, i)
case '\r':
i++
if i < len(src) && src[i] == '\n' {
i++
}
starts = append(starts, i)
default:
i++
}
}
// A source ending in a terminator has a final, empty line whose start offset
// equals len(src). That line is addressable and must not be dropped.
return &oracle{src: src, starts: starts}
}

func (o *oracle) lineCount() int { return len(o.starts) }

// line returns the 0-based line containing off.
func (o *oracle) line(off int) int {
line := 0
for i, s := range o.starts {
if s > off {
break
}
line = i
}
return line
}

// utf8At returns the 0-based line and the column measured in runes.
func (o *oracle) utf8At(off int) (line, col int) {
line = o.line(off)
return line, utf8.RuneCount(o.src[o.starts[line]:off])
}

// utf16At returns the 0-based line and the column measured in UTF-16 code units,
// which is what the Language Server Protocol means by Position.character.
//
// Invalid UTF-8 becomes one U+FFFD per bad byte here, which is also how
// utf8.DecodeRune advances, so the two agree even on malformed input.
func (o *oracle) utf16At(off int) (line, char int) {
line = o.line(off)
return line, len(utf16.Encode([]rune(string(o.src[o.starts[line]:off]))))
}

// runeBoundaries returns every offset in [0, len(src)] that names a position.
//
// Offsets inside a multi-byte rune have no meaningful column and cannot round
// trip, so tests must not assert on them. The interior of a "\r\n" pair is
// excluded for the same reason: the terminator is one unit of line structure,
// and LSP columns are measured over a line's visible text, which stops before
// it. An offset between the two bytes therefore names no column at all.
func runeBoundaries(src []byte) []int {
offs := make([]int, 0, len(src)+1)
for i := 0; i < len(src); {
if src[i] == '\n' && i > 0 && src[i-1] == '\r' {
i++
continue
}
offs = append(offs, i)
// Advance by what DecodeRune consumes, not by one byte with a RuneStart
// filter. The two agree on valid UTF-8, but inside a truncated sequence
// DecodeRune yields a one-byte RuneError per continuation byte, and those
// offsets are reachable positions the filter would skip.
_, size := utf8.DecodeRune(src[i:])
i += size
}
return append(offs, len(src)) // EOF is always a valid position
}
Loading
Loading