1
+#!/usr/bin/env perl
2
+
3
+# Detect bare 'grep' used as a test assertion where 'test_grep'
4
+# should be used, and '! test_grep' where 'test_grep !' should
5
+# be used.
6
+#
7
+# The shared shell parser tokenizes test bodies so that 'grep'
8
+# inside heredocs, command substitutions like $(grep ...), and
9
+# quoted strings is collapsed into a single token and never seen
10
+# by our check. A line-oriented approach would need to track
11
+# heredoc delimiters, nested $() depth, and cross-line pipe
12
+# state to avoid false positives on patterns like:
13
+#
14
+# write_script foo.sh <<-\EOF
15
+# grep pattern file # data, not an assertion
16
+# EOF
17
+#
18
+# The Lexer already handles these.
19
+
20
+use warnings;
21
+use strict;
22
+use File::Basename;
23
+do(dirname($0) . "/lib-shell-parser.pl")
24
+ or die "$0: failed to load lib-shell-parser.pl: $@$!\n";
25
+
26
+my $exit_code = 0;
27
+
28
+# GrepLintParser inherits ScriptParser's ability to find
29
+# test_expect_success/failure blocks and call check_test()
30
+# on each body. We override check_test() to walk the token
31
+# stream looking for bare grep assertions.
32
+package GrepLintParser;
33
+
34
+our @ISA = ('ScriptParser');
35
+
36
+# After these tokens, the next token is a command word.
37
+# For example, in 'echo foo && grep bar file', the 'grep'
38
+# after '&&' is at command position and should be flagged.
39
+my %cmd_start = map { $_ => 1 } qw(&& || ; ;; do then else elif), "\n", '{', '(';
40
+
41
+# Tokens indicating grep's output is piped or redirected.
42
+my %filter_op = map { $_ => 1 } qw(| > >> <);
43
+
44
+# A token is at "command word" position if the shell would
45
+# interpret it as a program name rather than an argument.
46
+# Only 'grep' at command position is an assertion we should
47
+# flag; 'grep' as an argument ('test_must_fail grep') or
48
+# value ('for cmd in grep sed') is not.
49
+sub is_command_word {
50
+ my ($tokens, $pos) = @_;
51
+ return 1 if $pos == 0;
52
+ for (my $j = $pos - 1; $j >= 0; $j--) {
53
+ my $t = $tokens->[$j]->[0];
54
+ # After a separator or pipe, a new command starts.
55
+ return 1 if $cmd_start{$t} || $t eq '|';
56
+ # After '}' or ')', what follows is a separator or
57
+ # redirect on the compound command, not a new command.
58
+ return 0 if $t eq '}' || $t eq ')';
59
+ # '!' is a prefix that does not consume command
60
+ # position; keep scanning to find what precedes it.
61
+ next if $t eq '!';
62
+ # Any other word means we are past the command word.
63
+ return 0;
64
+ }
65
+ return 1;
66
+}
67
+
68
+# lint_ok() reports whether a bare grep carries a trailing
69
+# '# lint-ok' comment telling this linter to skip it.
70
+#
71
+# In practice this is needed for just one case: a grep acting
72
+# as a data filter whose output is consumed by a redirect or
73
+# pipe on an enclosing compound command (such as a subshell or
74
+# brace group) rather than by grep's own pipeline, e.g.
75
+#
76
+# ( grep ... && # lint-ok
77
+# sed ... ) >out
78
+#
79
+# { grep ... || : # lint-ok
80
+# } >out
81
+#
82
+# is_filter() only scans grep's own pipeline: it stops at the
83
+# separator before the compound command closes and never sees
84
+# the outer redirect, so it would flag such a grep as an
85
+# assertion. A grep that really is an assertion is better
86
+# written as test_grep (or a guarded test_grep when the file's
87
+# presence is conditional) than annotated with lint-ok.
88
+sub lint_ok {
89
+ my ($raw_lines, $ln) = @_;
90
+ if ($ln < 1 || $ln > @$raw_lines) {
91
+ warn "lint_ok: line number $ln out of range (1.." .
92
+ scalar(@$raw_lines) . ")\n";
93
+ return 0;
94
+ }
95
+ return $raw_lines->[$ln - 1] =~ /lint-ok/;
96
+}
97
+
98
+# Grep is a filter (not an assertion) if it receives piped
99
+# input or sends its output to a pipe or redirect. Check
100
+# both directions from grep's position in the token stream.
101
+sub is_filter {
102
+ my ($tokens, $pos) = @_;
103
+ # Backward: is grep receiving piped input?
104
+ # Newlines don't break pipes ('cmd |\n grep' is one
105
+ # pipeline), so skip past them.
106
+ for (my $j = $pos - 1; $j >= 0; $j--) {
107
+ my $t = $tokens->[$j]->[0];
108
+ return 1 if $t eq '|';
109
+ next if $t eq "\n";
110
+ last if $cmd_start{$t} || $t eq '}' || $t eq ')';
111
+ }
112
+ # Forward: is grep piping or redirecting output?
113
+ # Unlike the backward scan, we do not skip newlines here:
114
+ # a bare newline is a command boundary, and redirects or
115
+ # pipes must appear on the same line as grep (or after a
116
+ # line continuation, which the Lexer consumes).
117
+ for (my $j = $pos + 1; $j < @$tokens; $j++) {
118
+ my $t = $tokens->[$j]->[0];
119
+ return 0 if $cmd_start{$t};
120
+ return 1 if $filter_op{$t};
121
+ }
122
+ return 0;
123
+}
124
+
125
+# Map a body-relative line number to a file line number.
126
+# For double-quoted bodies, backslash-continuation lines
127
+# (\<newline>) are consumed by the Lexer without appearing
128
+# in the body text, so the inner parser sees fewer lines
129
+# than the source file has. We walk the source lines to
130
+# count continuations and adjust accordingly.
131
+sub body_to_file_line {
132
+ my ($body_lineno, $body_token, $raw_lines, $body_start) = @_;
133
+ my $body_text = $body_token->[0];
134
+ my $body_end_line = $body_token->[4];
135
+ unless ($body_start && $body_start >= 1) {
136
+ warn "body_start is not a positive integer\n";
137
+ return $body_lineno;
138
+ }
139
+ my $file_lineno = $body_lineno + $body_start - 1;
140
+ # Only double-quoted bodies have line splices.
141
+ return $file_lineno unless $body_text =~ /^"/;
142
+ my $adj = 0;
143
+ my $lines_seen = 0;
144
+ unless ($body_end_line && $body_end_line >= $body_start) {
145
+ warn "body_end_line is not set for double-quoted body\n";
146
+ return $file_lineno;
147
+ }
148
+ my $end = $body_end_line;
149
+ if ($end > @$raw_lines) {
150
+ warn "body_end_line ($end) exceeds file length (" .
151
+ scalar(@$raw_lines) . ")\n";
152
+ return $file_lineno;
153
+ }
154
+ my $src_ln = $body_start;
155
+ while ($src_ln <= $end && $lines_seen < $body_lineno) {
156
+ my $line = $raw_lines->[$src_ln - 1];
157
+ # Odd trailing backslashes = continuation (\<nl>).
158
+ # Even = escaped backslashes (\\), not a continuation.
159
+ if ($line =~ /(\\*)$/ && length($1) % 2 == 1) {
160
+ $adj++;
161
+ } else {
162
+ $lines_seen++;
163
+ }
164
+ $src_ln++;
165
+ }
166
+ if ($lines_seen < $body_lineno) {
167
+ warn "body_lineno ($body_lineno) not found within body range " .
168
+ "($body_start..$end)\n";
169
+ }
170
+ return $file_lineno + $adj;
171
+}
172
+
173
+# ScriptParser calls this for each test body found in the script.
174
+sub check_test {
175
+ my $self = shift @_;
176
+ my $title = ScriptParser::unwrap(shift @_);
177
+ my $body_token = shift @_;
178
+ my $body_start = $body_token->[3];
179
+ my $body = ScriptParser::unwrap($body_token);
180
+ # Handle heredoc-style test bodies:
181
+ # test_expect_success 'title' - <<\EOF
182
+ # grep pattern file
183
+ # EOF
184
+ # The '-' signals that the body follows as a heredoc.
185
+ if ($body eq '-') {
186
+ my $herebody = shift @_;
187
+ if ($herebody) {
188
+ $body = $herebody->{content};
189
+ $body_start = $herebody->{start_line};
190
+ }
191
+ }
192
+ return unless $body;
193
+
194
+ my $raw_lines = $self->{raw_lines};
195
+
196
+ # The outer parser gives us the body as an opaque string.
197
+ # Parse it to get individual tokens with command boundaries.
198
+ my $parser = ShellParser->new(\$body);
199
+ my @tokens = $parser->parse();
200
+
201
+ my $file = $self->{file};
202
+
203
+ for (my $i = 0; $i < @tokens; $i++) {
204
+ my $text = $tokens[$i]->[0];
205
+ next unless is_command_word(\@tokens, $i);
206
+
207
+ my $token_lineno = $tokens[$i]->[3];
208
+ unless (defined($token_lineno) && $token_lineno >= 1) {
209
+ warn "token has no line number\n";
210
+ next;
211
+ }
212
+ my $file_lineno = body_to_file_line(
213
+ $token_lineno,
214
+ $body_token, $raw_lines, $body_start);
215
+
216
+ # '!' negates the exit code without consuming command
217
+ # position. '! test_grep' is an anti-pattern because
218
+ # test_grep only prints diagnostics on grep failure,
219
+ # and '!' inverts after that decision is already made.
220
+ if ($text eq '!') {
221
+ if ($i + 1 < @tokens &&
222
+ $tokens[$i + 1]->[0] eq 'test_grep' &&
223
+ !lint_ok($raw_lines, $file_lineno)) {
224
+ print "$file:$file_lineno: error: ",
225
+ 'use "test_grep !" instead of ',
226
+ '"! test_grep"', "\n";
227
+ $exit_code = 1;
228
+ }
229
+ next;
230
+ }
231
+
232
+ # Bare grep as a command (not a filter) is a test
233
+ # assertion that should use test_grep for better
234
+ # failure diagnostics.
235
+ if ($text eq 'grep' &&
236
+ !is_filter(\@tokens, $i) &&
237
+ !lint_ok($raw_lines, $file_lineno)) {
238
+ print "$file:$file_lineno: error: ",
239
+ "bare grep outside pipeline ",
240
+ "(use test_grep)\n";
241
+ $exit_code = 1;
242
+ }
243
+ }
244
+}
245
+
246
+package main;
247
+
248
+for my $file (@ARGV) {
249
+ open(my $fh, '<:unix:crlf', $file) or die "$0: $file: $!\n";
250
+ my @raw_lines = <$fh>;
251
+ close $fh;
252
+ my $s = join('', @raw_lines);
253
+ my $parser = GrepLintParser->new(\$s);
254
+ $parser->{file} = $file;
255
+ $parser->{raw_lines} = \@raw_lines;
256
+ $parser->parse();
257
+}
258
+exit $exit_code;