@@ -35,17 +35,16 @@ import (
3535const (
3636 generatorVersion = "1"
3737
38- header = "id,name,email,amount,created,description\n "
3938 emailDomain = "@example.com"
4039 createdDate = "2026-08-01"
4140
4241 // amountWidth is six digits, a dot and two more. The range the amount is
4342 // drawn from below guarantees it.
4443 amountWidth = 9
4544
46- // rowTail is what follows the description: the closing quote and the
47- // newline .
48- rowTail = `"` + " \n "
45+ // closingQuote ends the description. What follows it is the row ending,
46+ // which the dialect decides, so the two are no longer one constant .
47+ closingQuote = `"`
4948
5049 // maxRowDigits bounds the width of the row number. A row is at least one
5150 // byte, so a file can never hold more rows than it has bytes, and a size is
@@ -54,13 +53,24 @@ const (
5453 // length whatever row number it lands on.
5554 maxRowDigits = 19
5655
57- // fixedWidth is every byte of a row except the row number, the name (which
58- // also forms the address) and the description. A constant expression, so it
59- // cannot drift away from the template above.
60- fixedWidth = 5 /* separators */ + len (emailDomain ) + amountWidth +
61- len (createdDate ) + 1 /* the opening quote */ + len (rowTail )
56+ // fixedBeforeEnding is every byte of a row except the row number, the name
57+ // (which also forms the address), the description and the row ending. A
58+ // constant expression, so it cannot drift away from the template above.
59+ //
60+ // The five separators count one byte each, which is a fact about the
61+ // separators offered rather than an assumption: every one of them is a
62+ // single byte, and dialect.go says so where they are declared.
63+ fixedBeforeEnding = 5 /* separators */ + len (emailDomain ) + amountWidth +
64+ len (createdDate ) + 1 /* the opening quote */ + len (closingQuote )
6265)
6366
67+ // fixedWidth is fixedBeforeEnding plus the row ending, which the dialect
68+ // decides. A CRLF row costs one byte more than an LF one, on every row, which
69+ // is why the minimum moves with this setting.
70+ func fixedWidth (d dialect ) int64 {
71+ return int64 (fixedBeforeEnding + len (d .eol ))
72+ }
73+
6474func init () {
6575 format .Register (format.Descriptor {
6676 ID : "csv" ,
@@ -70,9 +80,14 @@ func init() {
7080
7181 // A file with a header and no rows is legal CSV, and it is not something
7282 // anybody orders by naming a byte count - that is a shape request, and
73- // it arrives with the row count property. The minimum here is the header
74- // and one whole row.
75- MinBytes : minimumBytes (),
83+ // it would arrive with a row count setting, which this format does not
84+ // offer yet.
85+ //
86+ // The floor announced here is for the settings left alone. The dialect
87+ // moves the real one - a CRLF row costs a byte more and a file with no
88+ // header has one fewer line to pay for - so Plan works that one out and
89+ // names it. The log format is arranged the same way.
90+ MinBytes : minimumBytes (defaultDialect ()),
7691
7792 Padding : format.PaddingChannel {
7893 Name : "the description field of the last row" ,
@@ -85,26 +100,34 @@ func init() {
85100 // name and the manifest carry it instead.
86101 Label : format .LabelExternalOnly ,
87102 Oracle : "python-csv" ,
88- // Separator, quoting, column count and column types come later.
89- // Declaring none now makes a recipe asking for them fail loudly.
90- Properties : nil ,
103+ // Quoting, column count and column types come later. Declaring none of
104+ // them now makes a recipe asking for one fail loudly.
105+ Properties : properties () ,
91106 GeneratorVersion : generatorVersion ,
92107 Generator : generator {},
93108 })
94109}
95110
96111type generator struct {}
97112
98- type memo struct { seed uint64 }
113+ type memo struct {
114+ seed uint64
115+ dia dialect
116+ }
99117
100118func (generator ) Plan (r format.Request ) (format.Plan , error ) {
101- min := minimumBytes ()
119+ d , err := parseDialect (r .Properties )
120+ if err != nil {
121+ return format.Plan {}, err
122+ }
123+
124+ min := minimumBytes (d )
102125 if r .Bytes < min {
103126 return format.Plan {}, & format.BelowMinimumError {
104127 Format : "CSV" ,
105128 Requested : r .Bytes ,
106129 Minimum : min ,
107- Reason : "a table holds a header and whole rows, and one of each needs that much" ,
130+ Reason : reasonForMinimum ( d ) ,
108131 Hint : fmt .Sprintf ("Ask for %d B or more." , min ),
109132 }
110133 }
@@ -114,42 +137,62 @@ func (generator) Plan(r format.Request) (format.Plan, error) {
114137 Exact : true ,
115138 Determinism : format .DeterminismByte ,
116139 Properties : map [string ]any {
117- "encoding" : "utf-8" ,
118- "line_ending" : "lf" ,
119- "separator" : "," ,
120- "header" : true ,
121- "columns" : 6 ,
140+ "encoding" : "utf-8" ,
141+ // The manifest carries the separator as the CHARACTER, where the
142+ // recipe names it as a word. That difference is deliberate and is
143+ // the same one the contract already draws between size, which is an
144+ // intention, and bytes, which is a fact. Changing it would break
145+ // every script reading this field.
146+ "line_ending" : d .lineEndingID ,
147+ "separator" : string (d .sep ),
148+ "header" : d .header ,
149+ "columns" : len (columnNames ),
122150 // Stated even though it is always false here, so a test can assert
123151 // on it without knowing which formats carry a label internally.
124152 format .PropertyLabelEmbedded : false ,
125153 },
126- Memo : memo {seed : r .Seed },
154+ Memo : memo {seed : r .Seed , dia : d },
127155 }, nil
128156}
129157
158+ // reasonForMinimum says what the floor is made of, which changes with the
159+ // dialect. A file with no header pays for rows alone, and saying "a header and
160+ // whole rows" there would name something the file does not have.
161+ func reasonForMinimum (d dialect ) string {
162+ if ! d .header {
163+ return "a table holds whole rows, and one of them needs that much"
164+ }
165+ return "a table holds a header and whole rows, and one of each needs that much"
166+ }
167+
130168func (generator ) Write (ctx context.Context , w io.Writer , p format.Plan ) error {
131169 m , ok := p .Memo .(memo )
132170 if ! ok {
133171 return fmt .Errorf ("csv: the plan was not produced by this generator" )
134172 }
135173
136- if err := core .WriteAll (w , []byte (header )); err != nil {
137- return err
174+ if m .dia .header {
175+ if err := core .WriteAll (w , []byte (m .dia .headerLine ())); err != nil {
176+ return err
177+ }
138178 }
139179
140180 rng := core .NewRand (m .seed )
141- return core .FillRecords (ctx , w , rng , p .Bytes - int64 ( len ( header )) , & rows {})
181+ return core .FillRecords (ctx , w , rng , p .Bytes - m . dia . headerBytes () , & rows {dia : m . dia })
142182}
143183
144184// rows builds the data rows. It carries the row number, so the id column counts
145185// up the way a real export does.
146- type rows struct { next int64 }
186+ type rows struct {
187+ next int64
188+ dia dialect
189+ }
147190
148191// Shortest is the smallest row this builder can close a file with: the widest
149192// row number, the longest name in both the name and the address, and an empty
150193// description. It has to hold for every draw rather than for the lucky one.
151194func (r * rows ) Shortest () int64 {
152- return int64 (maxRowDigits + 2 * longestWord + fixedWidth )
195+ return int64 (maxRowDigits + 2 * longestWord ) + fixedWidth ( r . dia )
153196}
154197
155198func (r * rows ) Append (dst []byte , rng * rand.Rand ) []byte {
@@ -185,42 +228,51 @@ func (r *rows) append(dst []byte, rng *rand.Rand, want int64) []byte {
185228 whole := 100000 + rng .IntN (899999 )
186229 cents := rng .IntN (100 )
187230
231+ sep := r .dia .sep
232+
188233 dst = strconv .AppendInt (dst , r .next , 10 )
189- dst = append (dst , ',' )
234+ dst = append (dst , sep )
190235 dst = append (dst , name ... )
191- dst = append (dst , ',' )
236+ dst = append (dst , sep )
192237 dst = append (dst , name ... )
193238 dst = append (dst , emailDomain ... )
194- dst = append (dst , ',' )
239+ dst = append (dst , sep )
195240 dst = strconv .AppendInt (dst , int64 (whole ), 10 )
196241 dst = append (dst , '.' )
197242 if cents < 10 {
198243 dst = append (dst , '0' )
199244 }
200245 dst = strconv .AppendInt (dst , int64 (cents ), 10 )
201- dst = append (dst , ',' )
246+ dst = append (dst , sep )
202247 dst = append (dst , createdDate ... )
203- dst = append (dst , ',' , '"' )
248+ dst = append (dst , sep , '"' )
204249
205250 if want < 0 {
206- dst = appendPhrase (dst , rng , 3 + rng .IntN (5 ))
251+ dst = appendPhrase (dst , rng , 3 + rng .IntN (5 ), sep )
207252 } else {
208253 // Everything written so far, plus what still has to follow.
209- used := int64 (len (dst )- start ) + int64 (len (rowTail ))
210- dst = appendFiller (dst , want - used )
254+ used := int64 (len (dst )- start ) + int64 (len (closingQuote )) + int64 ( len ( r . dia . eol ))
255+ dst = appendFiller (dst , want - used , sep )
211256 }
212257
213- return append (dst , rowTail ... )
258+ dst = append (dst , closingQuote ... )
259+ return append (dst , r .dia .eol ... )
214260}
215261
216- // appendPhrase writes a readable description. Every few words it drops a comma,
217- // which is the case a CSV reader has to get right and the reason the column is
218- // quoted at all.
219- func appendPhrase (dst []byte , rng * rand.Rand , n int ) []byte {
262+ // appendPhrase writes a readable description. Every few words it drops the
263+ // SEPARATOR, which is the case a CSV reader has to get right and the reason the
264+ // column is quoted at all.
265+ //
266+ // The separator rather than always a comma, and that is the point of the
267+ // setting rather than a detail of it. A comma inside a semicolon separated file
268+ // needs no quoting, so a description that kept dropping commas would leave a
269+ // semicolon file never exercising the quoted path at all - the file would be
270+ // the right size, parse everywhere, and quietly test less than the comma one.
271+ func appendPhrase (dst []byte , rng * rand.Rand , n int , sep byte ) []byte {
220272 for i := 0 ; i < n ; i ++ {
221273 if i > 0 {
222274 if i % 3 == 0 {
223- dst = append (dst , ',' )
275+ dst = append (dst , sep )
224276 }
225277 dst = append (dst , ' ' )
226278 }
@@ -236,24 +288,32 @@ func appendPhrase(dst []byte, rng *rand.Rand, n int) []byte {
236288// field early.
237289// appendFiller stretches the description to the byte.
238290//
239- // A comma every fourth word, unlike every other format here, and on purpose:
240- // the description is a quoted field, so the padding is what makes a long file
241- // keep exercising the quoting rather than turning into plain words.
242- func appendFiller (dst []byte , n int64 ) []byte {
291+ // A separator every fourth word, unlike every other format here, and on
292+ // purpose: the description is a quoted field, so the padding is what makes a
293+ // long file keep exercising the quoting rather than turning into plain words.
294+ // It follows the dialect for the reason appendPhrase gives.
295+ func appendFiller (dst []byte , n int64 , sep byte ) []byte {
296+ both := string (sep ) + " "
243297 return core .AppendFiller (dst , words , n , func (i int ) string {
244298 if i % 4 == 0 {
245- return ", "
299+ return both
246300 }
247301 return " "
248302 })
249303}
250304
251- // minimumBytes is the header and one whole row, computed rather than written
252- // down so it cannot drift away from the template the way a number in a document
253- // would.
254- func minimumBytes () int64 {
255- var r rows
256- return int64 (len (header )) + r .Shortest ()
305+ // minimumBytes is the header, when there is one, and one whole row. Computed
306+ // rather than written down so it cannot drift away from the template the way a
307+ // number in a document would.
308+ //
309+ // It takes the dialect because the floor moves with it: a CRLF row costs a byte
310+ // more, and a file with no header has one fewer line to pay for. The registry
311+ // announces the floor for the settings left alone, and Plan works out the real
312+ // one for the settings that arrived - the same arrangement the log format uses,
313+ // where the entry shape moves the floor too.
314+ func minimumBytes (d dialect ) int64 {
315+ r := rows {dia : d }
316+ return d .headerBytes () + r .Shortest ()
257317}
258318
259319// longestWord is the widest draw, because the minimum has to hold for every
0 commit comments