diff --git a/README.md b/README.md index 6e9f73e..279145b 100644 --- a/README.md +++ b/README.md @@ -58,6 +58,8 @@ slices rather than nil slices. | `testdata/package-lock.json` | `test`, `fixture`, `generated` | | `packages/api/uv.lock` | `generated` | | `src/Form.Designer.cs` | `source`, `generated` | +| `src/app.min.js` | `source`, `generated`, `minified` | +| `service.pb.go` | `generated` | | `.pytest_cache/v/cache/nodeids` | `generated`, `cache` | | `CMakeFiles/app.dir/main.o` | `generated`, `build-output` | | `Pods/library/main.swift` | `vendor` | @@ -76,7 +78,7 @@ slices rather than nil slices. The vocabulary is `source`, `test`, `fixture`, `example`, `benchmark`, `fuzz`, `vendor`, `generated`, `build-output`, `cache`, `documentation`, `legal`, -`build`, `ci`, `packaging`, `tooling` and `configuration`. `build` identifies +`build`, `ci`, `packaging`, `tooling`, `configuration` and `minified`. `build` identifies build instructions and wrappers. `build-output` identifies specific emitted build-system files and directories. These labels describe path conventions; they do not establish that a tracked file is safe to delete. Coverage is based @@ -210,21 +212,35 @@ The caller supplies a rooted filesystem; `os.Root.FS` provides containment for a directory on disk. No files are read for content classification during `Walk`, and no directory names are automatically excluded. -`ClassifyReader(path, reader)` adds optional generated Go header detection and +`ClassifyReader(path, reader)` adds optional generated and minified detection and reads at most 8 KiB plus one byte used to detect truncation. `ClassifyBlob` provides the same result for callers that already hold the complete contents. Both have classifier methods that retain vendor-root context. Inspection is -limited to the first 8 KiB and 40 lines, ending at the first non-comment token. -Go tokenization prevents markers in strings or block comments from becoming -generated evidence. +limited to the first 8 KiB and 40 lines. Go retains its strict generated marker +check, ending at the first non-comment token. Go tokenization prevents markers +in strings or block comments from becoming generated evidence. + +Other supported source files match `Code generated by`, `DO NOT EDIT` or +`@generated` in leading comments, ignoring case. Hash comments are checked for +`.py`, `.rb` and `.sh`; line and block comments are checked for `.js`, `.mjs`, +`.cjs`, `.jsx`, `.ts`, `.tsx`, `.c`, `.h`, `.cc`, `.cpp`, `.hpp`, `.cs`, `.java`, +`.rs`, `.swift`, `.kt` and `.dart`. CSS uses block comments. Inspection stops at +code, so a marker in a string or a later comment is not a generated header. + +JavaScript, TypeScript and CSS also receive `generated` for source-map comment +directives within the inspected prefix. A line with at least 1,000 code bytes +and fewer than one space or tab per ten code bytes receives both `minified` +and `generated`. Strings and comments do not count toward that threshold. +These are heuristics: short minified files and source-map directives beyond +the inspection limits can be missed. No tail reads are performed. The result includes `HeaderChecked`, `BytesExamined` and `HeaderLimited`. -Other file types receive path-only classification. Missing generated +Unsupported file types receive path-only classification. Missing generated evidence does not prove a file was handwritten, and a generated lockfile may still be essential dependency evidence. -Content caches should include the blob ID, `ContentVersion` and whether the -occurrence name has a `.go` suffix. The same blob can receive path-only +Content caches should include the blob ID, `ContentVersion` and the occurrence +filename extension, including its casing. The same blob can receive path-only classification at one occurrence and generated-header evidence at another. ## Corpus @@ -262,8 +278,10 @@ any roles matched inside them. Named minified JavaScript/CSS files, source maps and .NET designer files also receive `generated`. Source-map and designer suffixes accept mixed case; minified suffixes use lowercase `.min.js`, `-min.js`, `.min.css` and -`-min.css`. A plain `.d.ts` filename or a name such as `jquery.js` does not -establish generated or vendored content. +`-min.css` and also receive `minified`. Protobuf output names ending in `.pb.go`, +`_pb2.py`, `_pb2_grpc.py`, `_pb2.pyi` or `_pb2_grpc.pyi` receive `generated`. +A plain `.d.ts` filename or a name such as `jquery.js` does not establish +generated or vendored content. Named community documents include support, governance, maintainers, authors and roadmaps, using case-insensitive stems and selected document diff --git a/blob.go b/blob.go index 38cce2f..36a6017 100644 --- a/blob.go +++ b/blob.go @@ -13,11 +13,11 @@ import ( const ( MaxHeaderBytes = 8192 MaxHeaderLines = 40 - ContentVersion = "1" + ContentVersion = "2" headerReadLimit = MaxHeaderBytes + 1 ) -// BlobResult describes the bounded Go header inspection, where applicable. +// BlobResult describes bounded content inspection, where applicable. // Absence of Generated is not proof that a file was handwritten. type BlobResult struct { Result @@ -26,14 +26,14 @@ type BlobResult struct { HeaderLimited bool `json:"header_limited"` } -// ClassifyBlob adds generated Go header evidence to path classification. +// ClassifyBlob adds generated and minified content evidence to path classification. // Contents must be the complete file. At most MaxHeaderBytes and MaxHeaderLines -// are inspected. Other languages receive path-only classification. +// are inspected. Unsupported file types receive path-only classification. func ClassifyBlob(name string, contents []byte) (BlobResult, error) { return defaults.ClassifyBlob(name, contents) } -// ClassifyReader adds generated Go header evidence using bounded input. +// ClassifyReader adds content evidence using bounded input. // It reads at most MaxHeaderBytes plus one byte used to detect truncation. func ClassifyReader(name string, reader io.Reader) (BlobResult, error) { return defaults.ClassifyReader(name, reader) @@ -45,7 +45,7 @@ func (c *Classifier) ClassifyBlob(name string, contents []byte) (BlobResult, err if err != nil || !inspect { return blob, err } - return inspectGoHeader(blob, name, contents), nil + return inspectContent(blob, name, contents), nil } // ClassifyReader preserves this classifier's vendor-root context while @@ -63,9 +63,9 @@ func (c *Classifier) ClassifyReader(name string, reader io.Reader) (BlobResult, } contents, err := io.ReadAll(io.LimitReader(reader, headerReadLimit)) if err != nil { - return BlobResult{}, fmt.Errorf("read Go header: %w", err) + return BlobResult{}, fmt.Errorf("read content header: %w", err) } - return inspectGoHeader(blob, name, contents), nil + return inspectContent(blob, name, contents), nil } func (c *Classifier) blobResult(name string) (BlobResult, bool, error) { @@ -74,10 +74,10 @@ func (c *Classifier) blobResult(name string) (BlobResult, bool, error) { return BlobResult{}, false, err } blob := BlobResult{Result: result} - return blob, strings.HasSuffix(name, ".go"), nil + return blob, contentType(name) != "", nil } -func inspectGoHeader(blob BlobResult, name string, contents []byte) BlobResult { +func inspectContent(blob BlobResult, name string, contents []byte) BlobResult { blob.HeaderChecked = true header := contents[:min(len(contents), MaxHeaderBytes)] lines := 0 @@ -92,6 +92,13 @@ func inspectGoHeader(blob BlobResult, name string, contents []byte) BlobResult { } blob.HeaderLimited = len(header) < len(contents) blob.BytesExamined = len(header) + if contentType(name) == "go" { + return inspectGoHeader(blob, name, header) + } + return inspectSourceHeader(blob, name, header) +} + +func inspectGoHeader(blob BlobResult, name string, header []byte) BlobResult { file := token.NewFileSet().AddFile(name, -1, len(header)) var lexer scanner.Scanner lexer.Init(file, header, nil, scanner.ScanComments) diff --git a/blob_test.go b/blob_test.go index 8c1dde8..afd1603 100644 --- a/blob_test.go +++ b/blob_test.go @@ -98,13 +98,15 @@ func (r *countingReader) Read(buffer []byte) (int, error) { func FuzzBlob(f *testing.F) { f.Add([]byte("// Code generated by test. DO NOT EDIT.\n")) f.Fuzz(func(t *testing.T, data []byte) { - got, err := roles.ClassifyBlob("vendor/client.go", data) - if err != nil || !got.Has(roles.Vendor) || got.BytesExamined > min(len(data), roles.MaxHeaderBytes) { - t.Fatal(got, err) - } - fromReader, err := roles.ClassifyReader("vendor/client.go", bytes.NewReader(data)) - if err != nil || !reflect.DeepEqual(fromReader, got) { - t.Fatalf("reader differs: %+v, %v", fromReader, err) + for _, name := range []string{"vendor/client.go", "vendor/client.py", "vendor/client.js", "vendor/client.css"} { + got, err := roles.ClassifyBlob(name, data) + if err != nil || !got.Has(roles.Vendor) || got.BytesExamined > min(len(data), roles.MaxHeaderBytes) { + t.Fatal(got, err) + } + fromReader, err := roles.ClassifyReader(name, bytes.NewReader(data)) + if err != nil || !reflect.DeepEqual(fromReader, got) { + t.Fatalf("reader differs: %+v, %v", fromReader, err) + } } }) } diff --git a/cmd/roles/main_test.go b/cmd/roles/main_test.go index c4852ff..7ad2506 100644 --- a/cmd/roles/main_test.go +++ b/cmd/roles/main_test.go @@ -37,6 +37,32 @@ func TestRun(t *testing.T) { } } +func TestRunGeneratedAndMinifiedPaths(t *testing.T) { + for _, name := range []string{"src/app.min.js", "src/app-min.css", "src/service.pb.go", "src/service_pb2.py"} { + for _, labelsOnly := range []bool{false, true} { + args := []string{name} + if labelsOnly { + args = append([]string{labelsOnlyFlag}, args...) + } + var out bytes.Buffer + if err := run(args, &out); err != nil { + t.Fatal(err) + } + var got roles.Result + if err := json.Unmarshal(out.Bytes(), &got); err != nil { + t.Fatal(err) + } + want := []roles.Role{roles.Source, roles.Generated} + if strings.Contains(name, "app") { + want = append(want, roles.Minified) + } + if !slices.Equal(got.Roles, want) { + t.Fatalf("run(%v) = %v, want %v", args, got.Roles, want) + } + } + } +} + func TestRunJSONSchema(t *testing.T) { var out bytes.Buffer if err := run([]string{"LICENSE"}, &out); err != nil { diff --git a/content.go b/content.go new file mode 100644 index 0000000..afc054b --- /dev/null +++ b/content.go @@ -0,0 +1,182 @@ +package roles + +import ( + "bytes" + "path" + "regexp" + "strings" + "unicode/utf8" +) + +const ( + minifiedLineBytes = 1000 + cssContent = "css" +) + +var generatedMarker = regexp.MustCompile(`(?i)\bcode generated by\b|\bdo not edit\b|(?:^|\s)@generated(?:\s|$)`) + +func contentType(name string) string { + switch path.Ext(name) { + case ".go": + return "go" + case ".py", ".rb", ".sh": + return "hash" + case ".js", ".mjs", ".cjs", ".jsx", ".ts", ".tsx": + return "javascript" + case ".css": + return cssContent + case ".c", ".h", ".cc", ".cpp", ".hpp", ".cs", ".java", ".rs", ".swift", ".kt", ".dart": + return "c" + default: + return "" + } +} + +func inspectSourceHeader(blob BlobResult, name string, header []byte) BlobResult { + if blob.HeaderLimited { + header = trimPartialRune(header) + } + if bytes.IndexByte(header, 0) >= 0 || !utf8.Valid(header) { + return blob + } + kind := contentType(name) + if generatedHeader(header, kind, blob.HeaderLimited) { + blob.addContentEvidence(name, "generated.header", Generated, "header") + } + if kind == "javascript" || kind == cssContent { + minified, sourceMap := inspectWebContent(header, kind, blob.HeaderLimited) + if sourceMap { + blob.addContentEvidence(name, "generated.source-map", Generated, "source-map") + } + if minified { + blob.addContentEvidence(name, "generated.minified-content", Generated, "minified") + blob.addContentEvidence(name, "minified.content", Minified, "long-line") + } + } + return blob +} + +func (b *BlobResult) addContentEvidence(name, rule string, role Role, subtype string) { + b.Roles = (setOf(b.Roles) | roleBit(role)).List() + b.Evidence = append(b.Evidence, Evidence{Rule: rule, Role: role, Path: name, Subtype: subtype, Source: "roles"}) +} + +func generatedHeader(header []byte, kind string, limited bool) bool { + text := strings.TrimPrefix(string(header), "\ufeff") + if strings.HasPrefix(text, "#!") { + _, text, _ = strings.Cut(text, "\n") + } + for { + text = strings.TrimLeft(text, " \t\r\n\v\f") + comment, rest, ok := leadingComment(text, kind, limited) + if !ok { + return false + } + if generatedMarker.MatchString(comment) { + return true + } + text = rest + } +} + +func leadingComment(text, kind string, limited bool) (comment, rest string, ok bool) { + switch { + case kind == "hash": + if !strings.HasPrefix(text, "#") { + return "", "", false + } + text = text[1:] + case strings.HasPrefix(text, "/*"): + comment, rest, ok = strings.Cut(text[2:], "*/") + return comment, rest, ok + case kind != cssContent && strings.HasPrefix(text, "//"): + text = text[2:] + default: + return "", "", false + } + comment, rest, ok = strings.Cut(text, "\n") + return comment, rest, ok || !limited +} + +func inspectWebContent(header []byte, kind string, limited bool) (minified, sourceMap bool) { + text := string(header) + code, spaces := 0, 0 + finishLine := func() { + minified = minified || minifiedLine(code, spaces) + code, spaces = 0, 0 + } + for len(text) > 0 { + comment, rest, ok := leadingComment(text, kind, limited) + if ok { + sourceMap = sourceMap || sourceMapComment(comment) + if strings.Contains(text[:len(text)-len(rest)], "\n") { + finishLine() + } + text = rest + continue + } + if strings.HasPrefix(text, "/*") || (kind != cssContent && strings.HasPrefix(text, "//")) { + break + } + if text[0] == '\'' || text[0] == '"' || text[0] == '`' { + end := quotedEnd(text) + if strings.Contains(text[:end], "\n") { + finishLine() + } + text = text[end:] + continue + } + switch text[0] { + case '\n', '\r': + finishLine() + case ' ', '\t': + spaces++ + default: + code++ + } + text = text[1:] + } + finishLine() + return minified, sourceMap +} + +func minifiedLine(code, spaces int) bool { + const whitespaceRatio = 10 + return code >= minifiedLineBytes && spaces*whitespaceRatio < code +} + +func quotedEnd(text string) int { + quote := text[0] + for i := 1; i < len(text); i++ { + switch text[i] { + case '\\': + i++ + case quote: + return i + 1 + } + } + return len(text) +} + +// trimPartialRune drops a multibyte rune split by the header byte limit so +// truncation does not disqualify an otherwise valid UTF-8 header. +func trimPartialRune(header []byte) []byte { + for i := 1; i < utf8.UTFMax && i <= len(header); i++ { + if !utf8.RuneStart(header[len(header)-i]) { + continue + } + if !utf8.Valid(header[len(header)-i:]) { + return header[:len(header)-i] + } + break + } + return header +} + +func sourceMapComment(comment string) bool { + if len(comment) == 0 || (comment[0] != '#' && comment[0] != '@') { + return false + } + value, ok := strings.CutPrefix(strings.TrimSpace(comment[1:]), "sourceMappingURL=") + return ok && len(strings.Fields(value)) == 1 +} diff --git a/content_test.go b/content_test.go new file mode 100644 index 0000000..22e6848 --- /dev/null +++ b/content_test.go @@ -0,0 +1,145 @@ +package roles_test + +import ( + "bytes" + "reflect" + "slices" + "strings" + "testing" + + "github.com/git-pkgs/roles" +) + +const ( + clientPython = "client.py" + appJS = "src/app.js" + appCSS = "src/app.css" + mapDirective = "//# sourceMappingURL=app.js.map\n" +) + +func TestGeneratedSourceHeaders(t *testing.T) { + for _, tc := range []struct { + name, path, content string + generated bool + }{ + {"python", clientPython, "# -*- coding: utf-8 -*-\n# Generated by protoc. DO NOT EDIT!\nfrom google.protobuf import descriptor\n", true}, + {"ruby", "client.rb", "#!/usr/bin/env ruby\n# Code generated by schema compiler\nrequire 'json'\n", true}, + {"shell", "client.sh", "#!/bin/sh\n# DO NOT EDIT\necho hello\n", true}, + {"java", "Client.java", "/*\n * @generated\n */\npackage client;\n", true}, + {"javascript", appJS, "/** @generated */\nexport const version = 1;\n", true}, + {"typescript", "client.ts", "// Code generated by schema compiler\nexport interface Client {}\n", true}, + {"css", appCSS, "/* DO NOT EDIT */\nbody { color: red; }\n", true}, + {"bom and crlf", "client.cs", "\ufeff// Code generated by tool\r\nclass Client {}\r\n", true}, + {"comment at eof", clientPython, "# DO NOT EDIT", true}, + {"lowercase", "client.rs", "// do not edit\nstruct Client;\n", true}, + {"string", clientPython, "marker = '# DO NOT EDIT'\n", false}, + {"docstring", clientPython, "\"\"\"\n# DO NOT EDIT\n\"\"\"\n", false}, + {"after code", appJS, "const example = 1;\n// DO NOT EDIT\n", false}, + {"unsupported", "README.md", "// Code generated by tool\n", false}, + {"directory path", "client.py/", "# DO NOT EDIT\n", false}, + {"binary", appJS, "// DO NOT EDIT\n\x00", false}, + {"invalid utf8", appJS, "// DO NOT EDIT\n\xff", false}, + {"partial tag", appJS, "// @generatedValue\n", false}, + {"incomplete block", appJS, "/* DO NOT EDIT", false}, + {"last allowed line", clientPython, strings.Repeat("\n", roles.MaxHeaderLines-1) + "# DO NOT EDIT\nprint(1)\n", true}, + {"past lines", clientPython, strings.Repeat("\n", roles.MaxHeaderLines) + "# DO NOT EDIT\n", false}, + {"past bytes", clientPython, strings.Repeat(" ", roles.MaxHeaderBytes) + "# DO NOT EDIT\n", false}, + {"rune split at limit", clientPython, "# DO NOT EDIT\n#" + strings.Repeat("x", roles.MaxHeaderBytes-16) + "é\nprint(1)\n", true}, + {"truncated comment", clientPython, "# DO NOT EDIT" + strings.Repeat(" ", roles.MaxHeaderBytes), false}, + } { + t.Run(tc.name, func(t *testing.T) { + got := classifyContent(t, tc.path, tc.content) + if got.Has(roles.Generated) != tc.generated || got.Has(roles.Minified) { + t.Fatalf("unexpected roles: %+v", got) + } + }) + } +} + +func TestWebContent(t *testing.T) { + compact := strings.Repeat("function f(a){return a+1;}", 100) + for _, tc := range []struct { + name, path, content string + minified, sourceMap bool + }{ + {"javascript", appJS, compact, true, false}, + {"module", "app.mjs", compact, true, false}, + {"commonjs", "app.cjs", compact, true, false}, + {"css", appCSS, strings.Repeat(".foo{color:red}", 100), true, false}, + {"under threshold", appJS, strings.Repeat("a();", 249), false, false}, + {"at threshold", appJS, strings.Repeat("a();", 250), true, false}, + {"pretty printed", appJS, strings.Repeat("function f(a) {\n return a + 1;\n}\n", 100), false, false}, + {"whitespace", appJS, strings.Repeat("a(); ", 300), false, false}, + {"trailing whitespace", appJS, strings.Repeat("a();", 250) + strings.Repeat(" ", 100), false, false}, + {"long string", appJS, "const data = '" + compact + "';", false, false}, + {"long comment", appJS, "/*" + compact + "*/", false, false}, + {"line comment", appJS, "//" + compact, false, false}, + {"unsupported", "data.json", compact, false, false}, + {"large bundle", appJS, strings.Repeat(compact, 1000), true, false}, + {"js source map", appJS, "console.log(1);\n" + mapDirective, false, true}, + {"css source map", appCSS, "body{color:red}\n/*# sourceMappingURL=app.css.map */", false, true}, + {"legacy source map", appJS, "//@ sourceMappingURL=app.js.map", false, true}, + {"inline source map", appJS, "//# sourceMappingURL=data:application/json;base64,e30=\n", false, true}, + {"minified with map", appJS, compact + "\n" + mapDirective, true, true}, + {"empty map", appJS, "//# sourceMappingURL=\n", false, false}, + {"map prose", appJS, "// see sourceMappingURL=app.js.map\n", false, false}, + {"quoted map", appJS, "const text = '//# sourceMappingURL=app.js.map';", false, false}, + {"template map", appJS, "const text = `\n" + mapDirective + "`;", false, false}, + {"escaped template", appJS, "const text = `\\`\n" + mapDirective + "`;", false, false}, + {"map outside limit", appJS, strings.Repeat("\n", roles.MaxHeaderLines) + mapDirective, false, false}, + {"truncated map", appJS, "//# sourceMappingURL=" + strings.Repeat("x", roles.MaxHeaderBytes), false, false}, + {"rune split at limit", appJS, strings.Repeat("a", roles.MaxHeaderBytes-2) + "€\nb();\n", true, false}, + } { + t.Run(tc.name, func(t *testing.T) { + got := classifyContent(t, tc.path, tc.content) + if got.Has(roles.Minified) != tc.minified || got.Has(roles.Generated) != (tc.minified || tc.sourceMap) { + t.Fatalf("unexpected roles: %+v", got) + } + hasMap := slices.ContainsFunc(got.Evidence, func(e roles.Evidence) bool { return e.Rule == "generated.source-map" }) + if hasMap != tc.sourceMap { + t.Fatalf("source map evidence: %+v", got.Evidence) + } + }) + } +} + +func classifyContent(t *testing.T, name, content string) roles.BlobResult { + t.Helper() + got, err := roles.ClassifyBlob(name, []byte(content)) + if err != nil { + t.Fatal(err) + } + reader := &countingReader{Reader: strings.NewReader(content)} + fromReader, err := roles.ClassifyReader(name, reader) + if err != nil || !reflect.DeepEqual(got, fromReader) { + t.Fatalf("blob=%+v reader=%+v error=%v", got, fromReader, err) + } + if reader.read > roles.MaxHeaderBytes+1 || got.BytesExamined > roles.MaxHeaderBytes { + t.Fatal("content read exceeded limit") + } + if got.HeaderChecked && got.HeaderLimited != (got.BytesExamined < len(content)) { + t.Fatalf("incorrect inspection bounds: %+v", got) + } + return got +} + +func TestContentVendorContext(t *testing.T) { + classifier, err := roles.New([]roles.VendorRoot{{Path: "deps"}}) + if err != nil { + t.Fatal(err) + } + name := "deps/src/app.min.js" + content := []byte("// @generated\n" + strings.Repeat("a();", 300) + "\n" + mapDirective) + got, err := classifier.ClassifyReader(name, bytes.NewReader(content)) + if err != nil { + t.Fatal(err) + } + want := []roles.Role{roles.Source, roles.Vendor, roles.Generated, roles.Minified} + if !slices.Equal(got.Roles, want) { + t.Fatalf("roles = %v, want %v", got.Roles, want) + } + fromBlob, err := classifier.ClassifyBlob(name, content) + if err != nil || !reflect.DeepEqual(got, fromBlob) { + t.Fatalf("blob=%+v reader=%+v error=%v", fromBlob, got, err) + } +} diff --git a/content_version_test.go b/content_version_test.go index a68f83c..bf60abb 100644 --- a/content_version_test.go +++ b/content_version_test.go @@ -23,6 +23,13 @@ func TestContentVersionFingerprint(t *testing.T) { {clientGo, strings.Repeat("\n", roles.MaxHeaderLines) + "// Code generated by tool. DO NOT EDIT.\n"}, {clientGo, strings.Repeat("x", roles.MaxHeaderBytes) + "\n// Code generated by tool. DO NOT EDIT.\n"}, {"client.txt", "// Code generated by tool. DO NOT EDIT.\n"}, + {clientPython, "# DO NOT EDIT\nprint(1)\n"}, + {appJS, "/* @generated */\nexport const version = 1;\n"}, + {appJS, "console.log(1);\n" + mapDirective}, + {appCSS, strings.Repeat(".foo{color:red}", 100)}, + {appJS, "const data = '" + strings.Repeat("a();", 300) + "';"}, + {appJS, strings.Repeat("a();", roles.MaxHeaderBytes)}, + {appJS, strings.Repeat("a", roles.MaxHeaderBytes-2) + "€\nb();\n"}, } hash := sha256.New() for _, tc := range cases { @@ -38,6 +45,7 @@ func TestContentVersionFingerprint(t *testing.T) { got := hex.EncodeToString(hash.Sum(nil)) want, ok := map[string]string{ "1": "e5085bec1fa3969071c52a75f04c064dd1af9ca18f034603b29cd14d869508d0", + "2": "8897d21dc9c850d31ba3dab64542c0c05bb423bfcd0253e8e6df39786005aff3", }[roles.ContentVersion] if !ok || got != want { t.Fatalf("content semantics changed without a ContentVersion bump: version=%q fingerprint=%s", roles.ContentVersion, got) diff --git a/corpus.go b/corpus.go index 73d58ef..40073d3 100644 --- a/corpus.go +++ b/corpus.go @@ -10,7 +10,7 @@ import ( const suffixFoldKind = "suffix-fold" // CorpusVersion identifies classification semantics and evidence ordering. -const CorpusVersion = "5" +const CorpusVersion = "6" //go:embed corpus/rules.json var corpusData []byte diff --git a/corpus/rules.json b/corpus/rules.json index 1541101..d9fad2f 100644 --- a/corpus/rules.json +++ b/corpus/rules.json @@ -2327,5 +2327,82 @@ "pattern": ".teamcity", "role": "ci", "source": "linguist" + }, + { + "id": "minified.suffix..min.js", + "kind": "suffix", + "pattern": ".min.js", + "role": "minified", + "source": "roles", + "subtype": "minified" + }, + { + "id": "minified.suffix.-min.js", + "kind": "suffix", + "pattern": "-min.js", + "role": "minified", + "source": "roles", + "subtype": "minified" + }, + { + "id": "minified.suffix..min.css", + "kind": "suffix", + "pattern": ".min.css", + "role": "minified", + "source": "roles", + "subtype": "minified" + }, + { + "id": "minified.suffix.-min.css", + "kind": "suffix", + "pattern": "-min.css", + "role": "minified", + "source": "roles", + "subtype": "minified" + }, + { + "id": "generated.suffix..pb.go", + "kind": "suffix", + "pattern": ".pb.go", + "role": "generated", + "source": "roles", + "subtype": "protobuf", + "ecosystem": "go" + }, + { + "id": "generated.suffix._pb2.py", + "kind": "suffix", + "pattern": "_pb2.py", + "role": "generated", + "source": "roles", + "subtype": "protobuf", + "ecosystem": "python" + }, + { + "id": "generated.suffix._pb2_grpc.py", + "kind": "suffix", + "pattern": "_pb2_grpc.py", + "role": "generated", + "source": "roles", + "subtype": "protobuf", + "ecosystem": "python" + }, + { + "id": "generated.suffix._pb2.pyi", + "kind": "suffix", + "pattern": "_pb2.pyi", + "role": "generated", + "source": "roles", + "subtype": "protobuf", + "ecosystem": "python" + }, + { + "id": "generated.suffix._pb2_grpc.pyi", + "kind": "suffix", + "pattern": "_pb2_grpc.pyi", + "role": "generated", + "source": "roles", + "subtype": "protobuf", + "ecosystem": "python" } ] diff --git a/corpus_version_test.go b/corpus_version_test.go index 95cc473..ef839b3 100644 --- a/corpus_version_test.go +++ b/corpus_version_test.go @@ -116,6 +116,7 @@ func TestCorpusVersionFingerprint(t *testing.T) { "3": "4364c21af3469ad9cd44262ec5d00c03b902211f7c51d016d83df1e9cbc4c0c8", "4": "63ed99b7d7119058e151d8261c5d10babe89e42d617079e2aaa453e72fe9c003", "5": "11cd35cb6c340dc94b387348e37c8db2a54fcf79779e4b09fde46c750d0740c3", + "6": "9363bd9a8b80ef83a38ffacb052c49b1d68543551ae75a6bf74f1934ba67dfae", }[roles.CorpusVersion] if !ok || got != want { t.Fatalf("classification semantics changed without a CorpusVersion bump: version=%q fingerprint=%s", roles.CorpusVersion, got) diff --git a/repositories_test.go b/repositories_test.go index 56f6e96..063d4ee 100644 --- a/repositories_test.go +++ b/repositories_test.go @@ -67,7 +67,7 @@ func TestRepositoryRoleSummaries(t *testing.T) { wantCounts := map[string]string{ "brief": `{"build":5,"ci":3,"configuration":23,"documentation":3,"example":4,"fixture":106,"generated":3,"legal":2,"packaging":23,"source":29,"test":128}`, "git-pkgs": `{"ci":2,"configuration":3,"documentation":14,"fixture":9,"legal":7,"source":155,"test":79,"tooling":3}`, - "scrutineer": `{"ci":5,"configuration":4,"documentation":44,"fixture":21,"generated":8,"legal":2,"source":421,"test":193,"tooling":18,"vendor":9}`, + "scrutineer": `{"ci":5,"configuration":4,"documentation":44,"fixture":21,"generated":8,"legal":2,"minified":8,"source":421,"test":193,"tooling":18,"vendor":9}`, } for _, repo := range loadRepositories(t) { if repo.Revision != wantRevision[repo.Name] { diff --git a/role_mapping_test.go b/role_mapping_test.go index bef32d3..44428f3 100644 --- a/role_mapping_test.go +++ b/role_mapping_test.go @@ -11,7 +11,7 @@ import ( func TestRoleConstants(t *testing.T) { declared := declaredRoles(t) - wantOrder := []Role{Source, Test, Fixture, Example, Benchmark, Fuzz, Vendor, Generated, BuildOutput, Cache, Documentation, Legal, Build, CI, Packaging, Tooling, Configuration} + wantOrder := []Role{Source, Test, Fixture, Example, Benchmark, Fuzz, Vendor, Generated, BuildOutput, Cache, Documentation, Legal, Build, CI, Packaging, Tooling, Configuration, Minified} if !slices.Equal(roleOrder[:], wantOrder) { t.Fatalf("role order = %v, want %v", roleOrder, wantOrder) } diff --git a/roles.go b/roles.go index 1045805..bb3574b 100644 --- a/roles.go +++ b/roles.go @@ -37,9 +37,10 @@ const ( Packaging Role = "packaging" Tooling Role = "tooling" Configuration Role = "configuration" + Minified Role = "minified" ) -var roleOrder = [...]Role{Source, Test, Fixture, Example, Benchmark, Fuzz, Vendor, Generated, BuildOutput, Cache, Documentation, Legal, Build, CI, Packaging, Tooling, Configuration} +var roleOrder = [...]Role{Source, Test, Fixture, Example, Benchmark, Fuzz, Vendor, Generated, BuildOutput, Cache, Documentation, Legal, Build, CI, Packaging, Tooling, Configuration, Minified} // Set is a compact collection of roles in corpus-defined order. type Set uint32 diff --git a/testdata/linguist-paths.json b/testdata/linguist-paths.json index dcfa569..6f59e8f 100644 --- a/testdata/linguist-paths.json +++ b/testdata/linguist-paths.json @@ -473,49 +473,57 @@ { "path": "app.min.js", "roles": [ - "generated" + "generated", + "minified" ], "subtypes": [ "minified" ], "sources": [ - "linguist" + "linguist", + "roles" ] }, { "path": "app-min.js", "roles": [ - "generated" + "generated", + "minified" ], "subtypes": [ "minified" ], "sources": [ - "linguist" + "linguist", + "roles" ] }, { "path": "app.min.css", "roles": [ - "generated" + "generated", + "minified" ], "subtypes": [ "minified" ], "sources": [ - "linguist" + "linguist", + "roles" ] }, { "path": "app-min.css", "roles": [ - "generated" + "generated", + "minified" ], "subtypes": [ "minified" ], "sources": [ - "linguist" + "linguist", + "roles" ] }, { @@ -716,13 +724,15 @@ "path": "vendor/app.min.js", "roles": [ "vendor", - "generated" + "generated", + "minified" ], "subtypes": [ "minified" ], "sources": [ - "linguist" + "linguist", + "roles" ] }, { @@ -943,7 +953,15 @@ }, { "path": "file.pb.go", - "roles": [] + "roles": [ + "generated" + ], + "subtypes": [ + "protobuf" + ], + "sources": [ + "roles" + ] }, { "path": "module.rbi", diff --git a/testdata/paths.json b/testdata/paths.json index 7d4ce11..1d658b9 100644 --- a/testdata/paths.json +++ b/testdata/paths.json @@ -2606,5 +2606,86 @@ { "path": "pipeline.yml", "roles": [] + }, + { + "path": "src/service_pb2.py", + "roles": [ + "source", + "generated" + ], + "subtypes": [ + "protobuf" + ], + "sources": [ + "roles" + ] + }, + { + "path": "vendor/src/service.pb.go", + "roles": [ + "source", + "vendor", + "generated" + ], + "subtypes": [ + "protobuf" + ] + }, + { + "path": "src/service_pb2_grpc.py", + "roles": [ + "source", + "generated" + ], + "subtypes": [ + "protobuf" + ], + "sources": [ + "roles" + ] + }, + { + "path": "service_pb2.pyi", + "roles": [ + "generated" + ], + "subtypes": [ + "protobuf" + ], + "sources": [ + "roles" + ] + }, + { + "path": "service_pb2_grpc.pyi", + "roles": [ + "generated" + ], + "subtypes": [ + "protobuf" + ], + "sources": [ + "roles" + ] + }, + { + "path": "service_pb2.py.bak", + "roles": [] + }, + { + "path": "service.pb.go.bak", + "roles": [] + }, + { + "path": "service.proto", + "roles": [] + }, + { + "path": "service.pb.go/", + "roles": [] + }, + { + "path": "app.min.js/", + "roles": [] } ]