mirror of
https://github.com/johnkerl/miller.git
synced 2026-07-20 18:10:07 +00:00
* Rename inputChannel,outputChannel to readerChannel,writerChannel * Rename inputChannel,outputChannel to readerChannel,writerChannel (#772) * Start batched-reader API mods * Singleton-list step for reader-batching at input * CLI options for records-per-batch and hash-records * Push channelized-reader logic into DKVP reader * Push batching logic into chain-transformer, transformers, and channel-writer * foo * cmd/mprof and cmd/mprof2 * cmd/mprof3 and cmd/mprof4 * narrowed in on regexp-splitting on IFS/IPS as perf-hit * neaten * channelize nidx * cmd/mprof5 * channelize CSV reader * channelize NIDX reader * Dedupe DKVP-reader and NIDX-reader source files * channelize CSV-lite reader * channelize XTAB reader * batchify JSON reader * channelize GEN pseudo-reader * scripts for perf-testing on larger files * merge with main for #776 * Fix record-batching for join and repl * Fix comment-handling in channelized XTAB reader * Fix bug found in positional-rename
201 lines
5.2 KiB
Go
201 lines
5.2 KiB
Go
package transformers
|
|
|
|
import (
|
|
"container/list"
|
|
"fmt"
|
|
"os"
|
|
"strings"
|
|
|
|
"github.com/johnkerl/miller/internal/pkg/cli"
|
|
"github.com/johnkerl/miller/internal/pkg/types"
|
|
)
|
|
|
|
// ----------------------------------------------------------------
|
|
const verbNameHead = "head"
|
|
|
|
var HeadSetup = TransformerSetup{
|
|
Verb: verbNameHead,
|
|
UsageFunc: transformerHeadUsage,
|
|
ParseCLIFunc: transformerHeadParseCLI,
|
|
IgnoresInput: false,
|
|
}
|
|
|
|
func transformerHeadUsage(
|
|
o *os.File,
|
|
doExit bool,
|
|
exitCode int,
|
|
) {
|
|
fmt.Fprintf(o, "Usage: %s %s [options]\n", "mlr", verbNameHead)
|
|
fmt.Fprintf(o, "Passes through the first n records, optionally by category.\n")
|
|
fmt.Fprintf(o, "Without -g, ceases consuming more input (i.e. is fast) when n records have been read.\n")
|
|
|
|
fmt.Fprintf(o, "Options:\n")
|
|
fmt.Fprintf(o, "-g {a,b,c} Optional group-by-field names for head counts, e.g. a,b,c.\n")
|
|
fmt.Fprintf(o, "-n {n} Head-count to print. Default 10.\n")
|
|
fmt.Fprintf(o, "-h|--help Show this message.\n")
|
|
|
|
if doExit {
|
|
os.Exit(exitCode)
|
|
}
|
|
}
|
|
|
|
func transformerHeadParseCLI(
|
|
pargi *int,
|
|
argc int,
|
|
args []string,
|
|
_ *cli.TOptions,
|
|
doConstruct bool, // false for first pass of CLI-parse, true for second pass
|
|
) IRecordTransformer {
|
|
|
|
// Skip the verb name from the current spot in the mlr command line
|
|
argi := *pargi
|
|
verb := args[argi]
|
|
argi++
|
|
|
|
headCount := 10
|
|
var groupByFieldNames []string = nil
|
|
|
|
for argi < argc /* variable increment: 1 or 2 depending on flag */ {
|
|
opt := args[argi]
|
|
if !strings.HasPrefix(opt, "-") {
|
|
break // No more flag options to process
|
|
}
|
|
if args[argi] == "--" {
|
|
break // All transformers must do this so main-flags can follow verb-flags
|
|
}
|
|
argi++
|
|
|
|
if opt == "-h" || opt == "--help" {
|
|
transformerHeadUsage(os.Stdout, true, 0)
|
|
|
|
} else if opt == "-n" {
|
|
headCount = cli.VerbGetIntArgOrDie(verb, opt, args, &argi, argc)
|
|
|
|
} else if opt == "-g" {
|
|
groupByFieldNames = cli.VerbGetStringArrayArgOrDie(verb, opt, args, &argi, argc)
|
|
|
|
} else {
|
|
transformerHeadUsage(os.Stderr, true, 1)
|
|
}
|
|
}
|
|
|
|
*pargi = argi
|
|
if !doConstruct { // All transformers must do this for main command-line parsing
|
|
return nil
|
|
}
|
|
|
|
transformer, err := NewTransformerHead(
|
|
headCount,
|
|
groupByFieldNames,
|
|
)
|
|
if err != nil {
|
|
fmt.Fprintln(os.Stderr, err)
|
|
os.Exit(1)
|
|
}
|
|
|
|
return transformer
|
|
}
|
|
|
|
// ----------------------------------------------------------------
|
|
type TransformerHead struct {
|
|
// input
|
|
headCount int
|
|
groupByFieldNames []string
|
|
|
|
// state
|
|
recordTransformerFunc RecordTransformerFunc
|
|
unkeyedRecordCount int
|
|
keyedRecordCounts map[string]int
|
|
|
|
// See ChainTransformer
|
|
wroteDownstreamDone bool
|
|
}
|
|
|
|
func NewTransformerHead(
|
|
headCount int,
|
|
groupByFieldNames []string,
|
|
) (*TransformerHead, error) {
|
|
|
|
tr := &TransformerHead{
|
|
headCount: headCount,
|
|
groupByFieldNames: groupByFieldNames,
|
|
unkeyedRecordCount: 0,
|
|
keyedRecordCounts: make(map[string]int),
|
|
wroteDownstreamDone: false,
|
|
}
|
|
|
|
if groupByFieldNames == nil {
|
|
tr.recordTransformerFunc = tr.transformUnkeyed
|
|
} else {
|
|
tr.recordTransformerFunc = tr.transformKeyed
|
|
}
|
|
|
|
return tr, nil
|
|
}
|
|
|
|
// ----------------------------------------------------------------
|
|
|
|
func (tr *TransformerHead) Transform(
|
|
inrecAndContext *types.RecordAndContext,
|
|
outputRecordsAndContexts *list.List, // list of *types.RecordAndContext
|
|
inputDownstreamDoneChannel <-chan bool,
|
|
outputDownstreamDoneChannel chan<- bool,
|
|
) {
|
|
HandleDefaultDownstreamDone(inputDownstreamDoneChannel, outputDownstreamDoneChannel)
|
|
tr.recordTransformerFunc(inrecAndContext, outputRecordsAndContexts, inputDownstreamDoneChannel, outputDownstreamDoneChannel)
|
|
}
|
|
|
|
func (tr *TransformerHead) transformUnkeyed(
|
|
inrecAndContext *types.RecordAndContext,
|
|
outputRecordsAndContexts *list.List, // list of *types.RecordAndContext
|
|
inputDownstreamDoneChannel <-chan bool,
|
|
outputDownstreamDoneChannel chan<- bool,
|
|
) {
|
|
if !inrecAndContext.EndOfStream {
|
|
tr.unkeyedRecordCount++
|
|
if tr.unkeyedRecordCount <= tr.headCount {
|
|
outputRecordsAndContexts.PushBack(inrecAndContext)
|
|
} else if !tr.wroteDownstreamDone {
|
|
// Signify to data producers upstream that we'll ignore further
|
|
// data, so as far as we're concerned they can stop sending it. See
|
|
// ChainTransformer.
|
|
//TODO: maybe remove: outputRecordsAndContexts.PushBack(types.NewEndOfStreamMarker(&inrecAndContext.Context))
|
|
outputDownstreamDoneChannel <- true
|
|
tr.wroteDownstreamDone = true
|
|
}
|
|
} else {
|
|
outputRecordsAndContexts.PushBack(inrecAndContext)
|
|
}
|
|
}
|
|
|
|
func (tr *TransformerHead) transformKeyed(
|
|
inrecAndContext *types.RecordAndContext,
|
|
outputRecordsAndContexts *list.List, // list of *types.RecordAndContext
|
|
inputDownstreamDoneChannel <-chan bool,
|
|
outputDownstreamDoneChannel chan<- bool,
|
|
) {
|
|
if !inrecAndContext.EndOfStream {
|
|
inrec := inrecAndContext.Record
|
|
|
|
groupingKey, ok := inrec.GetSelectedValuesJoined(tr.groupByFieldNames)
|
|
if !ok {
|
|
return
|
|
}
|
|
|
|
count, present := tr.keyedRecordCounts[groupingKey]
|
|
if !present { // first time
|
|
tr.keyedRecordCounts[groupingKey] = 1
|
|
count = 1
|
|
} else {
|
|
tr.keyedRecordCounts[groupingKey] += 1
|
|
count += 1
|
|
}
|
|
|
|
if count <= tr.headCount {
|
|
outputRecordsAndContexts.PushBack(inrecAndContext)
|
|
}
|
|
|
|
} else {
|
|
outputRecordsAndContexts.PushBack(inrecAndContext)
|
|
}
|
|
}
|