diff --git a/README.md b/README.md index 42f52c9..79f3a5a 100644 --- a/README.md +++ b/README.md @@ -241,6 +241,14 @@ limited to the first 8 KiB and 40 lines. Go retains its strict generated marker check, ending at the first non-comment token. Go tokenization prevents markers in strings or block comments from becoming generated evidence. +`ClassifyPrefix(path, prefix)` accepts bytes starting at file offset zero when +EOF is unknown, such as the roughly 1 KiB prefixes read by +[peek](https://github.com/git-pkgs/peek). Its classifier method retains +vendor-root context. Inspected prefixes always set `HeaderLimited`, including +empty prefixes. An unfinished line comment cannot supply a generated marker or +source-map directive; a closed block comment can. Use `ClassifyBlob` when the +bytes contain the complete file. + Other supported source files match `Code generated by`, `DO NOT EDIT` or `@generated` in leading comments, ignoring case. Hash comments are checked for `.py`, `.rb` and `.sh`; line and block comments are checked for `.js`, `.mjs`, @@ -256,13 +264,15 @@ These are heuristics: short minified files and source-map directives beyond the inspection limits can be missed. No tail reads are performed. The result includes `HeaderChecked`, `BytesExamined` and `HeaderLimited`. -Unsupported file types receive path-only classification. Missing generated -evidence does not prove a file was handwritten, and a generated lockfile -may still be essential dependency evidence. - -Content caches should include the blob ID, `ContentVersion` and the occurrence -filename extension, including its casing. The same blob can receive path-only -classification at one occurrence and generated-header evidence at another. +`Result.Set()` converts the labels to a `Set`, so content results reach the same +filters as `Match`. Unsupported file types receive path-only classification. +Missing generated evidence does not prove a file was handwritten, and a +generated lockfile may still be essential dependency evidence. + +Content caches should include the blob ID, `ContentVersion`, input length, +whether the input is complete, and the occurrence filename extension, including +its casing. The same blob can receive path-only classification at one +occurrence and generated-header evidence at another. ## Corpus diff --git a/blob.go b/blob.go index 36a6017..71fad88 100644 --- a/blob.go +++ b/blob.go @@ -33,6 +33,14 @@ func ClassifyBlob(name string, contents []byte) (BlobResult, error) { return defaults.ClassifyBlob(name, contents) } +// ClassifyPrefix adds content evidence from a prefix starting at file offset zero. +// EOF is unknown, so an inspected prefix always has HeaderLimited set, even if +// empty. At most MaxHeaderBytes and MaxHeaderLines are inspected. +// Use ClassifyBlob when the contents are known to be complete. +func ClassifyPrefix(name string, prefix []byte) (BlobResult, error) { + return defaults.ClassifyPrefix(name, prefix) +} + // ClassifyReader adds content evidence using bounded input. // It reads at most MaxHeaderBytes plus one byte used to detect truncation. func ClassifyReader(name string, reader io.Reader) (BlobResult, error) { @@ -45,7 +53,16 @@ func (c *Classifier) ClassifyBlob(name string, contents []byte) (BlobResult, err if err != nil || !inspect { return blob, err } - return inspectContent(blob, name, contents), nil + return inspectContent(blob, name, contents, false), nil +} + +// ClassifyPrefix preserves this classifier's vendor-root context. +func (c *Classifier) ClassifyPrefix(name string, prefix []byte) (BlobResult, error) { + blob, inspect, err := c.blobResult(name) + if err != nil || !inspect { + return blob, err + } + return inspectContent(blob, name, prefix, true), nil } // ClassifyReader preserves this classifier's vendor-root context while @@ -65,7 +82,7 @@ func (c *Classifier) ClassifyReader(name string, reader io.Reader) (BlobResult, if err != nil { return BlobResult{}, fmt.Errorf("read content header: %w", err) } - return inspectContent(blob, name, contents), nil + return inspectContent(blob, name, contents, false), nil } func (c *Classifier) blobResult(name string) (BlobResult, bool, error) { @@ -77,7 +94,7 @@ func (c *Classifier) blobResult(name string) (BlobResult, bool, error) { return blob, contentType(name) != "", nil } -func inspectContent(blob BlobResult, name string, contents []byte) BlobResult { +func inspectContent(blob BlobResult, name string, contents []byte, truncated bool) BlobResult { blob.HeaderChecked = true header := contents[:min(len(contents), MaxHeaderBytes)] lines := 0 @@ -90,7 +107,7 @@ func inspectContent(blob BlobResult, name string, contents []byte) BlobResult { break } } - blob.HeaderLimited = len(header) < len(contents) + blob.HeaderLimited = truncated || len(header) < len(contents) blob.BytesExamined = len(header) if contentType(name) == "go" { return inspectGoHeader(blob, name, header) diff --git a/content_version_test.go b/content_version_test.go index bf60abb..8ff1c7c 100644 --- a/content_version_test.go +++ b/content_version_test.go @@ -51,3 +51,44 @@ func TestContentVersionFingerprint(t *testing.T) { t.Fatalf("content semantics changed without a ContentVersion bump: version=%q fingerprint=%s", roles.ContentVersion, got) } } + +func TestContentVersionPrefixFingerprint(t *testing.T) { + const window = 1024 + cases := []struct { + path, prefix string + }{ + {clientPython, ""}, + {clientPython, "# DO NOT EDIT"}, + {clientPython, "# DO NOT EDIT\n"}, + {clientPython, "# DO NOT EDIT\n# caf\xc3"}, + {clientGo, "// Code generated by tool. DO NOT EDIT."}, + {clientGo, "// Code generated by tool. DO NOT EDIT.\npackage client\n"}, + {appJS, "// @generated"}, + {appJS, "// @generated\n"}, + {appJS, "// @generated" + strings.Repeat(" ", window)}, + {appJS, "//# sourceMappingURL=app.js.map"}, + {appJS, mapDirective}, + {appJS, strings.Repeat("a();", window)}, + {appCSS, "/* DO NOT EDIT"}, + {appCSS, "/* DO NOT EDIT */"}, + {"client.txt", "# DO NOT EDIT\n"}, + } + hash := sha256.New() + for _, tc := range cases { + result, err := roles.ClassifyPrefix(tc.path, []byte(tc.prefix)) + if err != nil { + t.Fatal(err) + } + _, _ = fmt.Fprintf(hash, "%s\n", tc.path) + if err := json.NewEncoder(hash).Encode(result); err != nil { + t.Fatal(err) + } + } + got := hex.EncodeToString(hash.Sum(nil)) + want, ok := map[string]string{ + "2": "996a0bbef0c963d66e9b1a7ad0b3906f5e7955b46c2bf6cc89d65ef4e87a7698", + }[roles.ContentVersion] + if !ok || got != want { + t.Fatalf("prefix content semantics changed without a ContentVersion bump: version=%q fingerprint=%s", roles.ContentVersion, got) + } +} diff --git a/prefix_test.go b/prefix_test.go new file mode 100644 index 0000000..169948a --- /dev/null +++ b/prefix_test.go @@ -0,0 +1,116 @@ +package roles_test + +import ( + "bytes" + "errors" + "fmt" + "io" + "reflect" + "strings" + "testing" + + "github.com/git-pkgs/roles" +) + +func TestClassifyPrefix(t *testing.T) { + for _, tc := range []struct { + name, path, prefix string + generated bool + completeGenerated bool + }{ + {"empty", clientPython, "", false, false}, + {"partial hash comment", clientPython, "# DO NOT EDIT", false, true}, + {"complete hash comment", clientPython, "# DO NOT EDIT\n", true, true}, + {"partial line comment", appJS, "// @generated", false, true}, + {"complete line comment", appJS, "// @generated\n", true, true}, + {"partial block comment", appCSS, "/* DO NOT EDIT", false, false}, + {"complete block comment", appCSS, "/* DO NOT EDIT */", true, true}, + {"partial go comment", clientGo, "// Code generated by tool. DO NOT EDIT.", false, false}, + {"complete go comment", clientGo, "// Code generated by tool. DO NOT EDIT.\n", true, true}, + {"partial source map", appJS, "//# sourceMappingURL=app.js.map", false, true}, + {"complete source map", appJS, mapDirective, true, true}, + {"block source map", appCSS, "/*# sourceMappingURL=app.css.map */", true, true}, + {"split rune", clientPython, "# DO NOT EDIT\n# caf\xc3", true, false}, + {"invalid utf8", clientPython, "# DO NOT EDIT\n\xff\n", false, false}, + {"binary", clientPython, "# DO NOT EDIT\n\x00", false, false}, + {"marker after code", appJS, "const x = 1;\n// @generated\n", false, false}, + } { + t.Run(tc.name, func(t *testing.T) { + input := []byte(tc.prefix) + before := bytes.Clone(input) + got, err := roles.ClassifyPrefix(tc.path, input) + if err != nil || !got.HeaderChecked || !got.HeaderLimited || got.BytesExamined != len(input) || got.Has(roles.Generated) != tc.generated { + t.Fatalf("ClassifyPrefix = %+v, %v", got, err) + } + complete, err := roles.ClassifyBlob(tc.path, input) + if err != nil || complete.HeaderLimited || complete.Has(roles.Generated) != tc.completeGenerated { + t.Fatalf("ClassifyBlob = %+v, %v", complete, err) + } + if !bytes.Equal(input, before) { + t.Fatal("input changed") + } + }) + } +} + +func TestClassifyCachedPrefix(t *testing.T) { + const window = 1024 + for _, tc := range []struct { + name, content string + generated bool + }{ + {"unterminated marker comment", "// @generated" + strings.Repeat(" ", window), false}, + {"complete marker", "// @generated\n" + strings.Repeat("const x = 1;\n", window), true}, + {"minified", strings.Repeat("a();", window), true}, + } { + t.Run(tc.name, func(t *testing.T) { + cached := make([]byte, window) + if _, err := io.ReadFull(strings.NewReader(tc.content), cached); err != nil { + t.Fatal(err) + } + got, err := roles.ClassifyPrefix(appJS, cached) + if err != nil || !got.HeaderLimited || got.Has(roles.Generated) != tc.generated || got.Has(roles.Minified) != (tc.name == "minified") { + t.Fatalf("cached prefix = %+v, %v", got, err) + } + }) + } +} + +func TestClassifyPrefixBoundsAndContext(t *testing.T) { + classifier, err := roles.New([]roles.VendorRoot{{Path: "deps"}}) + if err != nil { + t.Fatal(err) + } + for _, input := range []string{ + strings.Repeat(" ", roles.MaxHeaderBytes) + "// @generated\n", + strings.Repeat("\n", roles.MaxHeaderLines) + "// @generated\n", + } { + got, err := classifier.ClassifyPrefix("deps/app.js", []byte(input)) + if err != nil || !got.Has(roles.Vendor) || got.Has(roles.Generated) || !got.HeaderLimited { + t.Fatalf("ClassifyPrefix = %+v, %v", got, err) + } + complete, err := classifier.ClassifyBlob("deps/app.js", []byte(input)) + if err != nil || !reflect.DeepEqual(got, complete) { + t.Fatalf("prefix=%+v complete=%+v error=%v", got, complete, err) + } + } + for _, name := range []string{"deps/README.md", "deps/app.js/"} { + got, err := classifier.ClassifyPrefix(name, []byte("// @generated\n")) + if err != nil || !got.Has(roles.Vendor) || got.Has(roles.Generated) || got.HeaderChecked || got.HeaderLimited || got.BytesExamined != 0 { + t.Fatalf("path-only result = %+v, %v", got, err) + } + } + if _, err := roles.ClassifyPrefix("../app.js", nil); !errors.Is(err, roles.ErrInvalidPath) { + t.Fatalf("invalid path error = %v", err) + } +} + +func ExampleClassifyPrefix() { + prefix := []byte("# DO NOT EDIT") + result, err := roles.ClassifyPrefix("client.py", prefix) + if err != nil { + panic(err) + } + fmt.Println(result.Has(roles.Generated), result.HeaderLimited) + // Output: false true +} diff --git a/roles.go b/roles.go index eaf3704..4e260aa 100644 --- a/roles.go +++ b/roles.go @@ -133,6 +133,10 @@ func (r Result) Has(role Role) bool { return false } +// Set returns the labels as a bit set, for callers that cache them or combine +// them with path-based Set results. +func (r Result) Set() Set { return setOf(r.Roles) } + // ErrInvalidPath indicates an absolute, unclean, empty, or NUL-containing path. var ErrInvalidPath = errors.New("expected a clean repository-relative path") diff --git a/roles_test.go b/roles_test.go index 227b9d9..a4cb7c3 100644 --- a/roles_test.go +++ b/roles_test.go @@ -314,6 +314,31 @@ func TestVendorContextMultipleEvidence(t *testing.T) { } } +func TestResultSet(t *testing.T) { + const name = "vendor/src/app.min.js" + result, err := roles.Classify(name) + if err != nil { + t.Fatal(err) + } + set, err := roles.Match(name) + if err != nil { + t.Fatal(err) + } + if result.Set() != set { + t.Fatalf("Set = %v, want %v", result.Set().List(), set.List()) + } + blob, err := roles.ClassifyBlob("src/app.js", []byte("// Code generated by tool\nexport const version = 1;\n")) + if err != nil { + t.Fatal(err) + } + if !blob.Set().Has(roles.Generated) { + t.Fatalf("content set = %v, want generated", blob.Set().List()) + } + if (roles.Result{}).Set() != 0 { + t.Fatalf("empty result set = %v", (roles.Result{}).Set().List()) + } +} + func TestMatchAllocations(t *testing.T) { if got := testing.AllocsPerRun(100, func() { _, _ = roles.Match("packages/parser/vendor/src/parser_test.go") }); got != 0 { t.Fatalf("allocations = %v", got) @@ -333,7 +358,7 @@ func FuzzClassify(f *testing.F) { if err != nil { return } - if !slices.Equal(result.Roles, set.List()) { + if !slices.Equal(result.Roles, set.List()) || result.Set() != set { t.Fatal("inconsistent labels") } for _, e := range result.Evidence {