mirror of
https://github.com/johnkerl/miller.git
synced 2026-07-30 03:01:39 +00:00
Phase 3 of plans/exit.md: the streaming-interface change, the load-bearing piece for #341 (DSL exit statement) and #440 (strict mode). - RecordTransformer.Transform and RecordTransformerFunc now return error. All 69 Transform implementations and their dispatch helpers updated (mechanical rewrite, compiler- and errcheck-verified). - runSingleTransformerBatch, on a Transform error, forwards any output produced before the failure plus an end-of-stream marker downstream, so the rest of the chain and the record-writer drain and finish cleanly; runSingleTransformer then surfaces the error to stream.Stream's select loop (non-blocking send; first error wins) and signals upstream-done so the record-reader stops. This is exactly the flush-then-exit sequencing a future DSL 'exit N' needs. - dataProcessingErrorChannel and FileOutputHandler.recordErroredChannel are now chan error instead of chan bool; ChannelWriter still prints write- error details at the site and sends the 'exiting due to data error' sentinel, preserving the exact stderr shape pinned by regression cases. - Mid-stream os.Exit sites converted to returned errors: put/filter DSL begin/main/end-block errors and the non-boolean filter-expression case, tee write/close failures, split write/open/close failures, join left-file ingest failures (both half-streaming and sorted paths, with full error plumbing through JoinBucketKeeper), histogram/stats2 ingest errors, surv fit errors, and step stepper allocation (tStepperAllocator now returns (tStepper, error); bad EWMA coefficients propagate; negative slwin parameters are reported by the CLI parser via the existing bad-stepper-name pattern). - The two genuinely internal join-bucket-keeper states now use lib.InternalCodingErrorWithMessageIf instead of hand-rolled print+exit. - pkg/transformers is now os.Exit-free. Behavior notes: the non-boolean filter message gains the standard 'mlr: ' prefix and a newline (it previously printed with neither); tee errors now include the underlying cause. All 4779 regression cases pass unchanged; mlr head early-out latency is unaffected (0.02s over 50M records). Co-authored-by: Claude Fable 5 <noreply@anthropic.com>
112 lines
4.3 KiB
Go
112 lines
4.3 KiB
Go
package stream
|
|
|
|
import (
|
|
"bufio"
|
|
"io"
|
|
|
|
"github.com/johnkerl/miller/v6/pkg/cli"
|
|
"github.com/johnkerl/miller/v6/pkg/input"
|
|
"github.com/johnkerl/miller/v6/pkg/output"
|
|
"github.com/johnkerl/miller/v6/pkg/transformers"
|
|
"github.com/johnkerl/miller/v6/pkg/types"
|
|
)
|
|
|
|
// Since Go is concurrent, the context struct (AWK-like variables such as
|
|
// FILENAME, NF, NF, FNR, etc.) needs to be duplicated and passed through the
|
|
// channels along with each record.
|
|
//
|
|
// * Record-readers update FILENAME, FILENUM, NF, NR, FNR within context structs.
|
|
//
|
|
// * Record-transformers can read these from the context structs.
|
|
//
|
|
// * Record-writers don't need them (OPS et al. are already in the
|
|
// writer-options struct). However, we have chained transformers using the
|
|
// 'then' command-line syntax. This means a given transformer might be piping
|
|
// its output to a record-writer, or another transformer. So, the
|
|
// record-and-context pair goes to the record-writers even though they don't
|
|
// need the contexts.
|
|
|
|
// Stream is the high-level sketch of Miller. It coordinates instantiating
|
|
// format-specific record-reader and record-writer objects, using flags from
|
|
// the command line; setting up I/O channels; running the record stream from
|
|
// the record-reader object, through the specified chain of transformers
|
|
// (verbs), to the record-writer object.
|
|
func Stream(
|
|
// fileNames argument is separate from options.FileNames for in-place mode,
|
|
// which sends along only one file name per call to Stream():
|
|
fileNames []string,
|
|
options *cli.TOptions,
|
|
recordTransformers []transformers.RecordTransformer,
|
|
outputStream io.WriteCloser,
|
|
outputIsStdout bool,
|
|
) error {
|
|
|
|
// Since Go is concurrent, the context struct needs to be duplicated and
|
|
// passed through the channels along with each record.
|
|
initialContext := types.NewContext()
|
|
|
|
// Instantiate the record-reader.
|
|
// RecordsPerBatch is tracked separately from ReaderOptions since join/repl
|
|
// may use batch size of 1.
|
|
recordReader, err := input.Create(&options.ReaderOptions, options.ReaderOptions.RecordsPerBatch)
|
|
if err != nil {
|
|
return err
|
|
}
|
|
|
|
// Instantiate the record-writer
|
|
recordWriter, err := output.Create(&options.WriterOptions)
|
|
if err != nil {
|
|
return err
|
|
}
|
|
|
|
// Set up the reader-to-transformer and transformer-to-writer channels.
|
|
readerChannel := make(chan []*types.RecordAndContext, 2) // list of *types.RecordAndContext
|
|
writerChannel := make(chan []*types.RecordAndContext, 1) // list of *types.RecordAndContext
|
|
|
|
// We're done when a fatal error is registered on input (file not found,
|
|
// etc) or when the record-writer has written all its output. We use
|
|
// channels to communicate both of these conditions. The
|
|
// dataProcessingErrorChannel carries mid-stream errors from the
|
|
// transformer chain (e.g. DSL runtime errors, tee/split write failures)
|
|
// and from the record-writer; senders use a non-blocking send, so the
|
|
// first error wins and is returned once the writer finishes draining.
|
|
inputErrorChannel := make(chan error, 1)
|
|
doneWritingChannel := make(chan bool, 1)
|
|
dataProcessingErrorChannel := make(chan error, 1)
|
|
|
|
// For mlr head, so a transformer can communicate it will disregard all
|
|
// further input. It writes this back upstream, and that is passed back to
|
|
// the record-reader which then stops reading input. This is necessary to
|
|
// get quick response from, for example, mlr head -n 10 on input files with
|
|
// millions or billions of records.
|
|
readerDownstreamDoneChannel := make(chan bool, 1)
|
|
|
|
// Start the reader, transformer, and writer. Let them run until fatal input
|
|
// error or end-of-processing happens.
|
|
bufferedOutputStream := bufio.NewWriter(outputStream)
|
|
|
|
go recordReader.Read(fileNames, *initialContext, readerChannel, inputErrorChannel, readerDownstreamDoneChannel)
|
|
go transformers.ChainTransformer(readerChannel, readerDownstreamDoneChannel, recordTransformers,
|
|
writerChannel, dataProcessingErrorChannel, options)
|
|
go output.ChannelWriter(writerChannel, recordWriter, &options.WriterOptions, doneWritingChannel,
|
|
dataProcessingErrorChannel, bufferedOutputStream, outputIsStdout)
|
|
|
|
var retval error
|
|
done := false
|
|
for !done {
|
|
select {
|
|
case ierr := <-inputErrorChannel:
|
|
retval = ierr
|
|
case derr := <-dataProcessingErrorChannel:
|
|
retval = derr
|
|
case <-doneWritingChannel:
|
|
done = true
|
|
}
|
|
}
|
|
|
|
if err := bufferedOutputStream.Flush(); err != nil && retval == nil {
|
|
retval = err
|
|
}
|
|
|
|
return retval
|
|
}
|