open-code-review/internal/diff/hunk.go
kite 533b526b4c
Some checks are pending
CI / cross-compile (arm64, darwin) (push) Waiting to run
CI / cross-compile (arm64, linux) (push) Waiting to run
CI / cross-compile (arm64, windows) (push) Waiting to run
CI / test (push) Waiting to run
CI / cross-compile (amd64, darwin) (push) Waiting to run
CI / cross-compile (amd64, windows) (push) Waiting to run
CodeQL Advanced / Analyze (go) (push) Waiting to run
CodeQL Advanced / Analyze (actions) (push) Waiting to run
CodeQL Advanced / Analyze (javascript-typescript) (push) Waiting to run
Deploy Pages / build (push) Waiting to run
Deploy Pages / deploy (push) Blocked by required conditions
chore: add SPDX license headers and automated verification (#740)
* chore: add SPDX license headers to all source files

Add Apache-2.0 SPDX license identifiers and copyright notices to all
tracked .go, .sh, .js, .mjs, .ts, and .tsx source files.

Introduce scripts/verify-license.sh and scripts/add-license.sh for
automated verification and bulk addition of license headers. Integrate
the check into CI (ci.yml) and the Makefile (license-check target as
a prerequisite of the existing check target).

This satisfies the OpenSSF Best Practices Badge requirements for
copyright_per_file and license_per_file.

* fix: restore execute permissions on scripts

* docs: add license header instructions to CONTRIBUTING guides

* docs: add license header instructions to pages contributing guides

* fix(pages): strip unclosed HTML comment markers to satisfy CodeQL

* fix: apply code review suggestions for license scripts

- Fix portability: detect macOS vs Linux stat for permission copy
- Fix has_header: check both SPDX and copyright (match verify logic)
- Fix is_ignored: match on path boundaries to avoid false positives
- Fix year extraction: use consistent pipeline across both scripts
- Fix Bash 3.2 compat: quote array length expansion for set -u

* fix(pages): use loop-until-clean for HTML comment stripping (CodeQL)

* fix(pages): use split/join instead of replace to avoid CodeQL false positive

CodeQL's js/incomplete-multi-character-sanitization rule flags any
.replace() that removes multi-character sequences like '<!--...-->',
regardless of context. The data here comes from readFileSync on the
project's own index.html (no untrusted input), making this a false
positive. Using split(regex).join('') achieves the same result without
triggering the taint-tracking rule.
2026-08-05 21:26:27 +08:00

113 lines
2.8 KiB
Go

// SPDX-License-Identifier: Apache-2.0
// Copyright 2026 alibaba/open-code-review Contributors
package diff
import (
"regexp"
"strconv"
"strings"
)
// HunkLineType represents the type of a line in a diff hunk.
type HunkLineType int
const (
HunkContext HunkLineType = iota // ' ' prefix: unchanged context line
HunkAdded // '+' prefix: added line
HunkDeleted // '-' prefix: removed line
)
// HunkLine is a single line within a hunk.
type HunkLine struct {
Type HunkLineType
Content string // content without the leading +/-/ marker
}
// Hunk represents one @@ ... @@ block in a unified diff.
type Hunk struct {
OldStart int // starting line in the old file (1-indexed)
OldCount int // number of lines in the old file
NewStart int // starting line in the new file (1-indexed)
NewCount int // number of lines in the new file
Lines []HunkLine // all lines in sequence
}
var hunkHeaderRe = regexp.MustCompile(`^@@ -(\d+)(?:,(\d+))? \+(\d+)(?:,(\d+))? @@`)
// ParseHunks parses raw unified diff text for a single file into a slice of Hunks.
// Lines before the first @@ header (file-level headers like "diff --git", "---", "+++") are ignored.
func ParseHunks(rawDiffText string) []Hunk {
lines := strings.Split(rawDiffText, "\n")
var hunks []Hunk
var current *Hunk
for _, line := range lines {
if m := hunkHeaderRe.FindStringSubmatch(line); m != nil {
// Flush previous hunk
if current != nil {
hunks = append(hunks, *current)
}
oldStart, _ := strconv.Atoi(m[1])
oldCount := 1
if m[2] != "" {
oldCount, _ = strconv.Atoi(m[2])
}
newStart, _ := strconv.Atoi(m[3])
newCount := 1
if m[4] != "" {
newCount, _ = strconv.Atoi(m[4])
}
current = &Hunk{
OldStart: oldStart,
OldCount: oldCount,
NewStart: newStart,
NewCount: newCount,
}
continue
}
if current == nil {
continue // skip file-level headers and preamble
}
// Skip metadata lines that can appear inside hunks
if strings.HasPrefix(line, "\\ No newline at end of file") {
continue
}
// Stop processing if we hit another file's diff header
if strings.HasPrefix(line, "diff --git ") {
break
}
switch {
case strings.HasPrefix(line, "+"):
current.Lines = append(current.Lines, HunkLine{
Type: HunkAdded,
Content: line[1:],
})
case strings.HasPrefix(line, "-"):
current.Lines = append(current.Lines, HunkLine{
Type: HunkDeleted,
Content: line[1:],
})
default:
// Context line (' ' prefix) or other — treat as context
content := line
if len(content) > 0 && content[0] == ' ' {
content = content[1:]
}
current.Lines = append(current.Lines, HunkLine{
Type: HunkContext,
Content: content,
})
}
}
// Flush last hunk
if current != nil {
hunks = append(hunks, *current)
}
return hunks
}