Repository navigation
Position #2
New issue
Have a question about this project? Sign up for a free GitHub account to open an issue and contact its maintainers and the community.
By clicking “Sign up for GitHub”, you agree to our terms of service and privacy statement. We’ll occasionally send you account related emails.
Already on GitHub? Sign in to your account
Merged
Merged
Position #2
Changes from all commits
Commits
File filter
Filter by extension
Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
There are no files selected for viewing
This file contains hidden or bidirectional Unicode text that may be interpreted or compiled differently than what appears below. To review, open the file in an editor that reveals hidden Unicode characters.
Learn more about bidirectional Unicode characters
This file contains hidden or bidirectional Unicode text that may be interpreted or compiled differently than what appears below. To review, open the file in an editor that reveals hidden Unicode characters.
Learn more about bidirectional Unicode characters
This file contains hidden or bidirectional Unicode text that may be interpreted or compiled differently than what appears below. To review, open the file in an editor that reveals hidden Unicode characters.
Learn more about bidirectional Unicode characters
| Original file line number | Diff line number | Diff line change |
|---|---|---|
| @@ -0,0 +1,43 @@ | ||
| package position_test | ||
|
|
||
| import ( | ||
| "os" | ||
| "path/filepath" | ||
| "testing" | ||
|
|
||
| "github.com/lonhutt/dcx/pkg/position" | ||
| ) | ||
|
|
||
| // These exist to answer one question if it is ever raised: is the column scan | ||
| // worth optimising? Measured first, optimised only if the numbers say so. | ||
| // | ||
| // For scale: the worst line in the 86 KB fixture is 186 characters, so the scan | ||
| // from line start to offset is bounded by that regardless of file size. | ||
|
|
||
| func BenchmarkNew(b *testing.B) { | ||
| src := loadFixture(b) | ||
| b.ReportAllocs() | ||
| b.SetBytes(int64(len(src))) | ||
| for b.Loop() { | ||
| _ = position.New(src) | ||
| } | ||
| } | ||
|
|
||
| func BenchmarkUTF16(b *testing.B) { | ||
| src := loadFixture(b) | ||
| ix := position.New(src) | ||
| off := position.Offset(len(src) / 2) | ||
| b.ReportAllocs() | ||
| for b.Loop() { | ||
| _, _ = ix.UTF16(off) | ||
| } | ||
| } | ||
|
|
||
| func loadFixture(tb testing.TB) []byte { | ||
| tb.Helper() | ||
| src, err := os.ReadFile(filepath.Join("testdata", "large-commented.jsonc")) | ||
| if err != nil { | ||
| tb.Fatalf("read fixture: %v", err) | ||
| } | ||
| return src | ||
| } |
This file contains hidden or bidirectional Unicode text that may be interpreted or compiled differently than what appears below. To review, open the file in an editor that reveals hidden Unicode characters.
Learn more about bidirectional Unicode characters
| Original file line number | Diff line number | Diff line change |
|---|---|---|
| @@ -0,0 +1,73 @@ | ||
| package position_test | ||
|
|
||
| import ( | ||
| "testing" | ||
|
|
||
| "github.com/lonhutt/dcx/pkg/position" | ||
| ) | ||
|
|
||
| // FuzzLineIndex asserts the two properties that must hold for arbitrary bytes: | ||
| // nothing panics, and every result stays in bounds and round trips. | ||
| // | ||
| // Malformed UTF-8 is a normal input for a linter, not an exotic one. The oracle | ||
| // and the implementation agree on it as long as both advance one byte per | ||
| // invalid byte, which is what utf8.DecodeRune does when it returns RuneError. | ||
| func FuzzLineIndex(f *testing.F) { | ||
| for _, seed := range []string{ | ||
| "", | ||
| "{}", | ||
| "a\n", | ||
| "a\r\nb\rc", | ||
| mixed, | ||
| "😀\n中\né", | ||
| "a\xffb", // invalid byte | ||
| "\xe4", // truncated multi-byte sequence | ||
| "\ufeff{}\n", // BOM | ||
| } { | ||
| f.Add([]byte(seed)) | ||
| } | ||
|
|
||
| f.Fuzz(func(t *testing.T, src []byte) { | ||
| ix := position.New(src) | ||
| orc := newOracle(src) | ||
|
|
||
| if n := ix.LineCount(); n < 1 { | ||
| t.Fatalf("LineCount() = %d, must be at least 1 even for empty input", n) | ||
| } | ||
| if got, want := ix.LineCount(), orc.lineCount(); got != want { | ||
| t.Fatalf("LineCount() = %d, oracle says %d", got, want) | ||
| } | ||
|
|
||
| for _, off := range runeBoundaries(src) { | ||
| // Checked against the oracle as well as through its own inverse: a | ||
| // defect shared by UTF8 and OffsetUTF8 satisfies the round trip while | ||
| // still producing the wrong column. | ||
| wantLine, wantCol := orc.utf8At(off) | ||
| l, c := ix.UTF8(position.Offset(off)) | ||
| if l != wantLine || c != wantCol { | ||
| t.Fatalf("UTF8(%d) = (%d,%d), oracle says (%d,%d)", off, l, c, wantLine, wantCol) | ||
| } | ||
| if back := ix.OffsetUTF8(l, c); back != position.Offset(off) { | ||
| t.Fatalf("utf8 round trip: offset %d -> (%d,%d) -> %d", off, l, c, back) | ||
| } | ||
|
|
||
| line, char := ix.UTF16(position.Offset(off)) | ||
| if line < 0 || line >= ix.LineCount() { | ||
| t.Fatalf("UTF16(%d) line = %d, out of range [0,%d)", off, line, ix.LineCount()) | ||
| } | ||
| if char < 0 { | ||
| t.Fatalf("UTF16(%d) char = %d, negative", off, char) | ||
| } | ||
|
|
||
| if back := ix.OffsetUTF16(line, char); back != position.Offset(off) { | ||
| t.Fatalf("round trip: offset %d -> (%d,%d) -> %d", off, line, char, back) | ||
| } | ||
|
|
||
| // Agreement with the reference implementation must hold here too, | ||
| // not just on the curated corpus. | ||
| if wl, wc := orc.utf16At(off); wl != line || wc != char { | ||
| t.Fatalf("UTF16(%d) = (%d,%d), oracle says (%d,%d)", off, line, char, wl, wc) | ||
| } | ||
| } | ||
| }) | ||
| } | ||
This file contains hidden or bidirectional Unicode text that may be interpreted or compiled differently than what appears below. To review, open the file in an editor that reveals hidden Unicode characters.
Learn more about bidirectional Unicode characters
| Original file line number | Diff line number | Diff line change |
|---|---|---|
| @@ -0,0 +1,106 @@ | ||
| package position_test | ||
|
|
||
| import ( | ||
| "unicode/utf16" | ||
| "unicode/utf8" | ||
| ) | ||
|
|
||
| // This file is the reference implementation the real one is tested against. | ||
| // | ||
| // The per-offset arithmetic here is deliberately the slowest, most obviously | ||
| // correct code that could work: it re-encodes the line prefix through []rune and | ||
| // utf16.Encode on every call, because that is the *definition* of an LSP column | ||
| // rather than an optimisation of it. Nothing in that path should ever be made | ||
| // clever. Its only job is to be so simple that when it disagrees with | ||
| // pkg/position, the bug is in pkg/position. | ||
| // | ||
| // The one concession is that line splitting is hoisted into newOracle instead of | ||
| // being redone per call. That is not cleverness, it is the difference between a | ||
| // 0.3s test and a 30s one on the 86 KB fixture: splitting is O(n) and was being | ||
| // run once per offset, making the whole check O(n^2). | ||
|
|
||
| type oracle struct { | ||
| src []byte | ||
| starts []int // byte offset at which each line begins | ||
| } | ||
|
|
||
| // newOracle splits src into lines, treating "\n", "\r\n" and a lone "\r" as | ||
| // terminators — which is what VS Code and other LSP clients do. If this | ||
| // disagrees with the real implementation, every position after the first stray | ||
| // carriage return is silently wrong. | ||
| func newOracle(src []byte) *oracle { | ||
| starts := []int{0} | ||
| for i := 0; i < len(src); { | ||
| switch src[i] { | ||
| case '\n': | ||
| i++ | ||
| starts = append(starts, i) | ||
| case '\r': | ||
| i++ | ||
| if i < len(src) && src[i] == '\n' { | ||
| i++ | ||
| } | ||
| starts = append(starts, i) | ||
| default: | ||
| i++ | ||
| } | ||
| } | ||
| // A source ending in a terminator has a final, empty line whose start offset | ||
| // equals len(src). That line is addressable and must not be dropped. | ||
| return &oracle{src: src, starts: starts} | ||
| } | ||
|
|
||
| func (o *oracle) lineCount() int { return len(o.starts) } | ||
|
|
||
| // line returns the 0-based line containing off. | ||
| func (o *oracle) line(off int) int { | ||
| line := 0 | ||
| for i, s := range o.starts { | ||
| if s > off { | ||
| break | ||
| } | ||
| line = i | ||
| } | ||
| return line | ||
| } | ||
|
|
||
| // utf8At returns the 0-based line and the column measured in runes. | ||
| func (o *oracle) utf8At(off int) (line, col int) { | ||
| line = o.line(off) | ||
| return line, utf8.RuneCount(o.src[o.starts[line]:off]) | ||
| } | ||
|
|
||
| // utf16At returns the 0-based line and the column measured in UTF-16 code units, | ||
| // which is what the Language Server Protocol means by Position.character. | ||
| // | ||
| // Invalid UTF-8 becomes one U+FFFD per bad byte here, which is also how | ||
| // utf8.DecodeRune advances, so the two agree even on malformed input. | ||
| func (o *oracle) utf16At(off int) (line, char int) { | ||
| line = o.line(off) | ||
| return line, len(utf16.Encode([]rune(string(o.src[o.starts[line]:off])))) | ||
| } | ||
|
|
||
| // runeBoundaries returns every offset in [0, len(src)] that names a position. | ||
| // | ||
| // Offsets inside a multi-byte rune have no meaningful column and cannot round | ||
| // trip, so tests must not assert on them. The interior of a "\r\n" pair is | ||
| // excluded for the same reason: the terminator is one unit of line structure, | ||
| // and LSP columns are measured over a line's visible text, which stops before | ||
| // it. An offset between the two bytes therefore names no column at all. | ||
| func runeBoundaries(src []byte) []int { | ||
| offs := make([]int, 0, len(src)+1) | ||
| for i := 0; i < len(src); { | ||
| if src[i] == '\n' && i > 0 && src[i-1] == '\r' { | ||
| i++ | ||
| continue | ||
| } | ||
| offs = append(offs, i) | ||
| // Advance by what DecodeRune consumes, not by one byte with a RuneStart | ||
| // filter. The two agree on valid UTF-8, but inside a truncated sequence | ||
| // DecodeRune yields a one-byte RuneError per continuation byte, and those | ||
| // offsets are reachable positions the filter would skip. | ||
| _, size := utf8.DecodeRune(src[i:]) | ||
| i += size | ||
| } | ||
| return append(offs, len(src)) // EOF is always a valid position | ||
| } |
Oops, something went wrong.
Oops, something went wrong.
Add this suggestion to a batch that can be applied as a single commit.
This suggestion is invalid because no changes were made to the code.
Suggestions cannot be applied while the pull request is closed.
Suggestions cannot be applied while viewing a subset of changes.
Only one suggestion per line can be applied in a batch.
Add this suggestion to a batch that can be applied as a single commit.
Applying suggestions on deleted lines is not supported.
You must change the existing code in this line in order to create a valid suggestion.
Outdated suggestions cannot be applied.
This suggestion has been applied or marked resolved.
Suggestions cannot be applied from pending reviews.
Suggestions cannot be applied on multi-line comments.
Suggestions cannot be applied while the pull request is queued to merge.
Suggestion cannot be applied right now. Please check back later.
Uh oh!
There was an error while loading. Please reload this page.