miller/pkg/transformers/having_fields.go
John Kerl 40ac1b309e
Remove os.Exit callsites below the entrypoint: phase 1 (#2198)
* Rebuild docs: help-catalog index count drifted on main (666 -> 667)

Pre-existing drift from a recent catalog addition; surfaced by make dev.

Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>

* Remove os.Exit callsites below the entrypoint: phase 1 (plans/exit.md)

Phase 1 of plans/exit.md: the mechanical swaps, plus the single-exit-point
scaffolding.

- New sentinel lib.ExitRequest{Code} (io.EOF-style control flow): returned by
  code paths that have already printed what they wanted (--version, -h,
  'put --explain' on a valid expression, 'put -x') instead of exiting mid-stack.
- pkg/entrypoint: new exitOnError() maps errors to exit codes in one place:
  ErrHelpRequested -> 0, ErrUsagePrinted -> 1, ExitRequest -> its code,
  anything else printed (JSON if --errors-json) -> 1.
- pkg/climain: version/help/usage/validation exits become returned errors;
  the ErrHelpRequested/ErrUsagePrinted -> exit translation moves up to the
  entrypoint; loadMlrrcOrDie becomes error-returning loadMlrrcFiles.
- Transformer ParseCLI/constructor exits -> returned errors (cut,
  having_fields, nest, reorder, merge_fields, reshape, rename, split, tee,
  subs, put_or_filter). merge_fields' bad-regex message formerly mis-reported
  itself as coming from the cut verb; tee's unrecognized-option exit was
  formerly silent; split/tee constructor failures formerly exited without
  printing the constructor's error.
- YAML/JSON record-writer marshal errors -> returned through Write, matching
  the CSV writer's error path.
- lib.WriteTempFileOrDie -> WriteTempFile (string, error); its sole caller
  (regtest diff helper) degrades gracefully.

Stderr messages and exit codes are byte-identical for all regression-covered
paths (4779 cases pass); runtime Transform-path exits are phase 3.

Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>

* Add lib.NewExitZeroRequest() constructor for exit-0 sentinel returns

Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>

---------

Co-authored-by: Claude Fable 5 <noreply@anthropic.com>
2026-07-15 11:34:01 -04:00

359 lines
11 KiB
Go

package transformers
import (
"fmt"
"os"
"regexp"
"strings"
"github.com/johnkerl/miller/v6/pkg/cli"
"github.com/johnkerl/miller/v6/pkg/lib"
"github.com/johnkerl/miller/v6/pkg/types"
)
type tHavingFieldsCriterion int
const (
havingFieldsCriterionUnspecified tHavingFieldsCriterion = iota
havingFieldsAtLeast
havingFieldsWhichAre
havingFieldsAtMost
havingAllFieldsMatching
havingAnyFieldsMatching
havingNoFieldsMatching
)
const verbNameHavingFields = "having-fields"
var havingFieldsOptions = []OptionSpec{
{Flag: "--at-least", Arg: "{comma-separated names}", Type: "csv-list", Desc: "Pass records that have at least these field names."},
{Flag: "--which-are", Arg: "{comma-separated names}", Type: "csv-list", Desc: "Pass records whose field names are exactly these."},
{Flag: "--at-most", Arg: "{comma-separated names}", Type: "csv-list", Desc: "Pass records that have at most these field names."},
{Flag: "--all-matching", Arg: "{regular expression}", Type: "regex", Desc: "Pass records where all field names match the regex."},
{Flag: "--any-matching", Arg: "{regular expression}", Type: "regex", Desc: "Pass records where any field name matches the regex."},
{Flag: "--none-matching", Arg: "{regular expression}", Type: "regex", Desc: "Pass records where no field name matches the regex."},
}
var HavingFieldsSetup = TransformerSetup{
Verb: verbNameHavingFields,
UsageFunc: transformerHavingFieldsUsage,
ParseCLIFunc: transformerHavingFieldsParseCLI,
IgnoresInput: false,
Options: havingFieldsOptions,
}
func transformerHavingFieldsUsage(
o *os.File,
) {
exeName := "mlr"
verb := verbNameHavingFields
fmt.Fprintf(o, "Usage: %s %s [options]\n", "mlr", verbNameHavingFields)
fmt.Fprintf(o, "Conditionally passes through records depending on each record's field names.\n")
WriteVerbOptions(o, havingFieldsOptions)
fmt.Fprintf(o, "Examples:\n")
fmt.Fprintf(o, " %s %s --which-are amount,status,owner\n", exeName, verb)
fmt.Fprintf(o, " %s %s --any-matching 'sda[0-9]'\n", exeName, verb)
fmt.Fprintf(o, " %s %s --any-matching '\"sda[0-9]\"'\n", exeName, verb)
fmt.Fprintf(o, " %s %s --any-matching '\"sda[0-9]\"i' (this is case-insensitive)\n", exeName, verb)
}
func transformerHavingFieldsParseCLI(
pargi *int,
argc int,
args []string,
_ *cli.TOptions,
doConstruct bool, // false for first pass of CLI-parse, true for second pass
) (RecordTransformer, error) {
havingFieldsCriterion := havingFieldsCriterionUnspecified
var fieldNames []string = nil
regexString := ""
// Skip the verb name from the current spot in the mlr command line
argi := *pargi
verb := args[argi]
argi++
var err error
for argi < argc /* variable increment: 1 or 2 depending on flag */ {
opt := args[argi]
if !strings.HasPrefix(opt, "-") {
break // No more flag options to process
}
if args[argi] == "--" {
break // All transformers must do this so main-flags can follow verb-flags
}
argi++
switch opt {
case "-h", "--help":
transformerHavingFieldsUsage(os.Stdout)
return nil, cli.ErrHelpRequested
case "--at-least":
havingFieldsCriterion = havingFieldsAtLeast
fieldNames, err = cli.VerbGetStringArrayArg(verb, opt, args, &argi, argc)
if err != nil {
return nil, err
}
regexString = ""
case "--which-are":
havingFieldsCriterion = havingFieldsWhichAre
fieldNames, err = cli.VerbGetStringArrayArg(verb, opt, args, &argi, argc)
if err != nil {
return nil, err
}
regexString = ""
case "--at-most":
havingFieldsCriterion = havingFieldsAtMost
fieldNames, err = cli.VerbGetStringArrayArg(verb, opt, args, &argi, argc)
if err != nil {
return nil, err
}
regexString = ""
case "--all-matching":
havingFieldsCriterion = havingAllFieldsMatching
regexString, err = cli.VerbGetStringArg(verb, opt, args, &argi, argc)
if err != nil {
return nil, err
}
fieldNames = nil
case "--any-matching":
havingFieldsCriterion = havingAnyFieldsMatching
regexString, err = cli.VerbGetStringArg(verb, opt, args, &argi, argc)
if err != nil {
return nil, err
}
fieldNames = nil
case "--none-matching":
havingFieldsCriterion = havingNoFieldsMatching
regexString, err = cli.VerbGetStringArg(verb, opt, args, &argi, argc)
if err != nil {
return nil, err
}
fieldNames = nil
default:
return nil, cli.VerbErrorf(verb, "option \"%s\" not recognized", opt)
}
}
if havingFieldsCriterion == havingFieldsCriterionUnspecified {
return nil, cli.VerbErrorf(verb, "-a, -b, -n, or --any-matching/--all-matching/--none-matching is required")
}
if fieldNames == nil && regexString == "" {
return nil, cli.VerbErrorf(verb, "field names or regex required")
}
*pargi = argi
if !doConstruct { // All transformers must do this for main command-line parsing
return nil, nil
}
transformer, err := NewTransformerHavingFields(
havingFieldsCriterion,
fieldNames,
regexString,
)
if err != nil {
return nil, err
}
return transformer, nil
}
type TransformerHavingFields struct {
fieldNames []string
numFieldNames int64
fieldNameSet map[string]bool
regex *regexp.Regexp
recordTransformerFunc RecordTransformerFunc
}
func NewTransformerHavingFields(
havingFieldsCriterion tHavingFieldsCriterion,
fieldNames []string,
regexString string,
) (*TransformerHavingFields, error) {
tr := &TransformerHavingFields{}
if fieldNames != nil {
tr.fieldNames = fieldNames
tr.numFieldNames = int64(len(fieldNames))
tr.fieldNameSet = lib.StringListToSet(fieldNames)
switch havingFieldsCriterion {
case havingFieldsAtLeast:
tr.recordTransformerFunc = tr.transformHavingFieldsAtLeast
case havingFieldsWhichAre:
tr.recordTransformerFunc = tr.transformHavingFieldsWhichAre
case havingFieldsAtMost:
tr.recordTransformerFunc = tr.transformHavingFieldsAtMost
default:
lib.InternalCodingErrorIf(true)
}
} else {
// Let them type in a.*b if they want, or "a.*b", or "a.*b"i.
// Strip off the leading " and trailing " or "i.
regex, err := lib.CompileMillerRegex(regexString)
if err != nil {
return nil, cli.VerbErrorf(verbNameHavingFields, "cannot compile regex \"%s\"", regexString)
}
tr.regex = regex
switch havingFieldsCriterion {
case havingAllFieldsMatching:
tr.recordTransformerFunc = tr.transformHavingAllFieldsMatching
case havingAnyFieldsMatching:
tr.recordTransformerFunc = tr.transformHavingAnyFieldsMatching
case havingNoFieldsMatching:
tr.recordTransformerFunc = tr.transformHavingNoFieldsMatching
default:
lib.InternalCodingErrorIf(true)
}
}
return tr, nil
}
func (tr *TransformerHavingFields) Transform(
inrecAndContext *types.RecordAndContext,
outputRecordsAndContexts *[]*types.RecordAndContext, // list of *types.RecordAndContext
inputDownstreamDoneChannel <-chan bool,
outputDownstreamDoneChannel chan<- bool,
) {
HandleDefaultDownstreamDone(inputDownstreamDoneChannel, outputDownstreamDoneChannel)
tr.recordTransformerFunc(inrecAndContext, outputRecordsAndContexts, inputDownstreamDoneChannel, outputDownstreamDoneChannel)
}
func (tr *TransformerHavingFields) transformHavingFieldsAtLeast(
inrecAndContext *types.RecordAndContext,
outputRecordsAndContexts *[]*types.RecordAndContext, // list of *types.RecordAndContext
inputDownstreamDoneChannel <-chan bool,
outputDownstreamDoneChannel chan<- bool,
) {
if !inrecAndContext.EndOfStream {
inrec := inrecAndContext.Record
numFound := int64(0)
for pe := inrec.Head; pe != nil; pe = pe.Next {
if tr.fieldNameSet[pe.Key] {
numFound++
if numFound == tr.numFieldNames {
*outputRecordsAndContexts = append(*outputRecordsAndContexts, inrecAndContext)
return
}
}
}
} else {
*outputRecordsAndContexts = append(*outputRecordsAndContexts, inrecAndContext)
}
}
func (tr *TransformerHavingFields) transformHavingFieldsWhichAre(
inrecAndContext *types.RecordAndContext,
outputRecordsAndContexts *[]*types.RecordAndContext, // list of *types.RecordAndContext
inputDownstreamDoneChannel <-chan bool,
outputDownstreamDoneChannel chan<- bool,
) {
if !inrecAndContext.EndOfStream {
inrec := inrecAndContext.Record
if inrec.FieldCount != tr.numFieldNames {
return
}
for pe := inrec.Head; pe != nil; pe = pe.Next {
if !tr.fieldNameSet[pe.Key] {
return
}
}
*outputRecordsAndContexts = append(*outputRecordsAndContexts, inrecAndContext)
} else {
*outputRecordsAndContexts = append(*outputRecordsAndContexts, inrecAndContext)
}
}
func (tr *TransformerHavingFields) transformHavingFieldsAtMost(
inrecAndContext *types.RecordAndContext,
outputRecordsAndContexts *[]*types.RecordAndContext, // list of *types.RecordAndContext
inputDownstreamDoneChannel <-chan bool,
outputDownstreamDoneChannel chan<- bool,
) {
if !inrecAndContext.EndOfStream {
inrec := inrecAndContext.Record
for pe := inrec.Head; pe != nil; pe = pe.Next {
if !tr.fieldNameSet[pe.Key] {
return
}
}
*outputRecordsAndContexts = append(*outputRecordsAndContexts, inrecAndContext)
} else {
*outputRecordsAndContexts = append(*outputRecordsAndContexts, inrecAndContext)
}
}
func (tr *TransformerHavingFields) transformHavingAllFieldsMatching(
inrecAndContext *types.RecordAndContext,
outputRecordsAndContexts *[]*types.RecordAndContext, // list of *types.RecordAndContext
inputDownstreamDoneChannel <-chan bool,
outputDownstreamDoneChannel chan<- bool,
) {
if !inrecAndContext.EndOfStream {
inrec := inrecAndContext.Record
for pe := inrec.Head; pe != nil; pe = pe.Next {
if !tr.regex.MatchString(pe.Key) {
return
}
}
*outputRecordsAndContexts = append(*outputRecordsAndContexts, inrecAndContext)
} else {
*outputRecordsAndContexts = append(*outputRecordsAndContexts, inrecAndContext)
}
}
func (tr *TransformerHavingFields) transformHavingAnyFieldsMatching(
inrecAndContext *types.RecordAndContext,
outputRecordsAndContexts *[]*types.RecordAndContext, // list of *types.RecordAndContext
inputDownstreamDoneChannel <-chan bool,
outputDownstreamDoneChannel chan<- bool,
) {
if !inrecAndContext.EndOfStream {
inrec := inrecAndContext.Record
for pe := inrec.Head; pe != nil; pe = pe.Next {
if tr.regex.MatchString(pe.Key) {
*outputRecordsAndContexts = append(*outputRecordsAndContexts, inrecAndContext)
return
}
}
} else {
*outputRecordsAndContexts = append(*outputRecordsAndContexts, inrecAndContext)
}
}
func (tr *TransformerHavingFields) transformHavingNoFieldsMatching(
inrecAndContext *types.RecordAndContext,
outputRecordsAndContexts *[]*types.RecordAndContext, // list of *types.RecordAndContext
inputDownstreamDoneChannel <-chan bool,
outputDownstreamDoneChannel chan<- bool,
) {
if !inrecAndContext.EndOfStream {
inrec := inrecAndContext.Record
for pe := inrec.Head; pe != nil; pe = pe.Next {
if tr.regex.MatchString(pe.Key) {
return
}
}
*outputRecordsAndContexts = append(*outputRecordsAndContexts, inrecAndContext)
} else {
*outputRecordsAndContexts = append(*outputRecordsAndContexts, inrecAndContext)
}
}