Merge pull request 'Markdownテーブル形式に対応' (#8) from feature/markdown-table-format into main
Reviewed-on: #8
This commit was merged in pull request #8.
This commit is contained in:
@@ -15,10 +15,11 @@ TSVへ変換したり、Excelからコピーした表を再び構造化データ
|
||||
- CSV
|
||||
- TSV
|
||||
- XLSX
|
||||
- Markdown table (`markdown` / `md`)
|
||||
|
||||
## インストール
|
||||
|
||||
Go 1.25以上が必要です。
|
||||
Go 1.25.13以上が必要です。
|
||||
|
||||
```sh
|
||||
go install git.rumginger.org/agent/dataxl/cmd/dataxl@latest
|
||||
@@ -55,6 +56,12 @@ JSONからExcel workbookを作成:
|
||||
dataxl -from json -to xlsx -i people.json -o people.xlsx
|
||||
```
|
||||
|
||||
Markdownのテーブルを作成:
|
||||
|
||||
```sh
|
||||
dataxl -from yaml -to markdown -i examples/people.yaml -o people.md
|
||||
```
|
||||
|
||||
Excel workbookをYAMLへ変換:
|
||||
|
||||
```sh
|
||||
@@ -86,6 +93,15 @@ Excelから戻す場合は、シート上の範囲をコピーしてstdinへ渡
|
||||
dataxl -from tsv -to yaml > restored.yaml
|
||||
```
|
||||
|
||||
## Markdownテーブル
|
||||
|
||||
GitHub Flavored Markdown互換のヘッダー・区切り行を持つテーブルを読み書きできます。
|
||||
外側の `|` は省略可能で、区切り行の `:` による左寄せ・中央寄せ・右寄せ指定も
|
||||
入力時に受け付けます(配置指定自体は保持しません)。
|
||||
|
||||
セル内の `|` とバックスラッシュはMarkdownのbackslash escapeを使用します。
|
||||
改行や前後の空白・タブはHTML文字参照として表し、表形式間の変換で保持します。
|
||||
|
||||
## 入れ子構造の表現
|
||||
|
||||
構造化データの入れ子は、Excelで編集しやすい列名へ展開されます。
|
||||
@@ -154,8 +170,8 @@ CSV、TSV、XLSXから構造化形式へ戻す際は、セル文字列を次の
|
||||
|
||||
- `-i`: 入力ファイル。省略時はstdin。
|
||||
- `-o`: 出力ファイル。省略時はstdout。
|
||||
- `-from`: 入力形式。`json`, `yaml`, `toml`, `csv`, `tsv`, `xlsx`。
|
||||
- `-to`: 出力形式。`json`, `yaml`, `toml`, `csv`, `tsv`, `xlsx`。
|
||||
- `-from`: 入力形式。`json`, `yaml`, `toml`, `csv`, `tsv`, `xlsx`, `markdown`。
|
||||
- `-to`: 出力形式。`json`, `yaml`, `toml`, `csv`, `tsv`, `xlsx`, `markdown`。
|
||||
- `-sheet`: XLSXの読み書きに使うシート名。既定値は `Sheet1`。
|
||||
- `-pretty`: JSONなどの構造化出力を整形するか。既定値は `true`。
|
||||
- `-version`: バージョンを表示して終了します。
|
||||
|
||||
@@ -51,6 +51,8 @@ func normalizeFormat(format string) string {
|
||||
switch format {
|
||||
case "yml":
|
||||
return "yaml"
|
||||
case "md", "mdown", "mkd":
|
||||
return "markdown"
|
||||
case "xlsm", "xls":
|
||||
return "xlsx"
|
||||
default:
|
||||
@@ -67,7 +69,7 @@ func inferFormat(path string) string {
|
||||
|
||||
func validateFormat(format string) error {
|
||||
switch format {
|
||||
case "json", "yaml", "toml", "csv", "tsv", "xlsx":
|
||||
case "json", "yaml", "toml", "csv", "tsv", "xlsx", "markdown":
|
||||
return nil
|
||||
default:
|
||||
return fmt.Errorf("unsupported format %q", format)
|
||||
@@ -75,5 +77,5 @@ func validateFormat(format string) error {
|
||||
}
|
||||
|
||||
func isTabular(format string) bool {
|
||||
return format == "csv" || format == "tsv" || format == "xlsx"
|
||||
return format == "csv" || format == "tsv" || format == "xlsx" || format == "markdown"
|
||||
}
|
||||
|
||||
@@ -6,7 +6,7 @@ import (
|
||||
)
|
||||
|
||||
func TestConversionMatrixPreservesRepresentativeTable(t *testing.T) {
|
||||
formats := []string{"json", "yaml", "toml", "csv", "tsv", "xlsx"}
|
||||
formats := []string{"json", "yaml", "toml", "csv", "tsv", "xlsx", "markdown"}
|
||||
records := []any{
|
||||
map[string]any{
|
||||
"active": true,
|
||||
|
||||
@@ -220,7 +220,7 @@ func TestDelimitedSingleEmptyCellRowRoundTrip(t *testing.T) {
|
||||
|
||||
func TestStructuredRecordsWithoutScalarFieldsAreRejectedForTables(t *testing.T) {
|
||||
for _, input := range []string{`{}`, `[{}]`} {
|
||||
for _, format := range []string{"csv", "tsv", "xlsx"} {
|
||||
for _, format := range []string{"csv", "tsv", "xlsx", "markdown"} {
|
||||
t.Run(format+"_"+input, func(t *testing.T) {
|
||||
_, err := convert([]byte(input), "json", format, "Sheet1", false)
|
||||
if err == nil || !strings.Contains(err.Error(), "no scalar fields") {
|
||||
@@ -232,7 +232,7 @@ func TestStructuredRecordsWithoutScalarFieldsAreRejectedForTables(t *testing.T)
|
||||
}
|
||||
|
||||
func TestEmptyRecordListCanRoundTripThroughDelimitedFormats(t *testing.T) {
|
||||
for _, format := range []string{"csv", "tsv"} {
|
||||
for _, format := range []string{"csv", "tsv", "markdown"} {
|
||||
t.Run(format, func(t *testing.T) {
|
||||
tabular, err := convert([]byte(`[]`), "json", format, "Sheet1", false)
|
||||
if err != nil {
|
||||
|
||||
+1
-1
@@ -1,7 +1,7 @@
|
||||
// Package main implements the dataxl command-line converter.
|
||||
//
|
||||
// Conversion uses two internal representations: structured Go values for
|
||||
// JSON/YAML/TOML and table for CSV/TSV/XLSX. Nested structured values cross the
|
||||
// JSON/YAML/TOML and table for CSV/TSV/XLSX/Markdown. Nested structured values cross the
|
||||
// boundary through dotted and indexed column paths such as user.name and
|
||||
// items[0].sku.
|
||||
package main
|
||||
|
||||
+3
-3
@@ -60,8 +60,8 @@ func parseOptions(args []string, stderr io.Writer) (options, error) {
|
||||
fs.SetOutput(stderr)
|
||||
fs.StringVar(&opt.inFile, "i", "", "input file, defaults to stdin")
|
||||
fs.StringVar(&opt.outFile, "o", "", "output file, defaults to stdout")
|
||||
fs.StringVar(&opt.from, "from", "", "input format: json, yaml, toml, csv, tsv, xlsx")
|
||||
fs.StringVar(&opt.to, "to", "", "output format: json, yaml, toml, csv, tsv, xlsx")
|
||||
fs.StringVar(&opt.from, "from", "", "input format: json, yaml, toml, csv, tsv, xlsx, markdown")
|
||||
fs.StringVar(&opt.to, "to", "", "output format: json, yaml, toml, csv, tsv, xlsx, markdown")
|
||||
fs.StringVar(&opt.sheet, "sheet", "Sheet1", "worksheet name for xlsx input/output")
|
||||
fs.BoolVar(&opt.pretty, "pretty", true, "pretty-print structured output")
|
||||
fs.BoolVar(&opt.version, "version", false, "print version and exit")
|
||||
@@ -72,7 +72,7 @@ func parseOptions(args []string, stderr io.Writer) (options, error) {
|
||||
dataxl -from json -to xlsx -i data.json -o data.xlsx
|
||||
|
||||
Formats:
|
||||
json, yaml/yml, toml, csv, tsv, xlsx
|
||||
json, yaml/yml, toml, csv, tsv, xlsx, markdown/md
|
||||
|
||||
Notes:
|
||||
Structured records are flattened into spreadsheet columns such as user.name
|
||||
|
||||
@@ -129,6 +129,49 @@ func TestResolveFormatsFromFileExtensions(t *testing.T) {
|
||||
}
|
||||
}
|
||||
|
||||
func TestResolveMarkdownFormatFromFileExtension(t *testing.T) {
|
||||
opt := options{inFile: "input.json", outFile: "output.md"}
|
||||
if err := opt.resolveFormats(); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if opt.from != "json" || opt.to != "markdown" {
|
||||
t.Fatalf("resolved formats = %q -> %q, want json -> markdown", opt.from, opt.to)
|
||||
}
|
||||
}
|
||||
|
||||
func TestMarkdownTableRoundTrip(t *testing.T) {
|
||||
want := table{
|
||||
Header: []string{"id", "name", "note"},
|
||||
Rows: [][]string{{"1", "Alice | Bob", "line 1\nline 2\\n"}},
|
||||
}
|
||||
encoded, err := encodeTable(want, "markdown", "")
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
got, err := parseTable(encoded, "markdown", "")
|
||||
if err != nil {
|
||||
t.Fatalf("parse generated Markdown: %v\n%s", err, encoded)
|
||||
}
|
||||
if !reflect.DeepEqual(got, want) {
|
||||
t.Fatalf("table = %#v, want %#v\n%s", got, want, encoded)
|
||||
}
|
||||
}
|
||||
|
||||
func TestMarkdownTableAcceptsAlignmentAndOptionalOuterPipes(t *testing.T) {
|
||||
input := []byte("id | name | note\n---: | :--- | :---:\n1 | Alice \\| Bob | ok\n")
|
||||
got, err := parseTable(input, "markdown", "")
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
want := table{
|
||||
Header: []string{"id", "name", "note"},
|
||||
Rows: [][]string{{"1", "Alice | Bob", "ok"}},
|
||||
}
|
||||
if !reflect.DeepEqual(got, want) {
|
||||
t.Fatalf("table = %#v, want %#v", got, want)
|
||||
}
|
||||
}
|
||||
|
||||
func TestVersionFlagDoesNotRequireFormats(t *testing.T) {
|
||||
oldVersion := buildVersion
|
||||
buildVersion = "v9.8.7-test"
|
||||
|
||||
+200
-1
@@ -5,14 +5,16 @@ import (
|
||||
"encoding/csv"
|
||||
"encoding/json"
|
||||
"fmt"
|
||||
"html"
|
||||
"sort"
|
||||
"strconv"
|
||||
"strings"
|
||||
"unicode/utf8"
|
||||
|
||||
"github.com/xuri/excelize/v2"
|
||||
)
|
||||
|
||||
// table is the common representation for CSV, TSV and XLSX. Rows are padded
|
||||
// table is the common representation for CSV, TSV, XLSX and Markdown. Rows are padded
|
||||
// to Header width when read, which keeps subsequent conversions rectangular.
|
||||
type table struct {
|
||||
Header []string
|
||||
@@ -36,6 +38,8 @@ func parseTable(input []byte, format, sheet string) (table, error) {
|
||||
return table{}, err
|
||||
}
|
||||
return rowsToTable(rows)
|
||||
case "markdown":
|
||||
return readMarkdown(input)
|
||||
default:
|
||||
return table{}, fmt.Errorf("format %q is not tabular", format)
|
||||
}
|
||||
@@ -93,11 +97,206 @@ func encodeTable(t table, format, sheet string) ([]byte, error) {
|
||||
return writeDelimited(t, '\t')
|
||||
case "xlsx":
|
||||
return writeXLSX(t, sheet)
|
||||
case "markdown":
|
||||
return writeMarkdown(t)
|
||||
default:
|
||||
return nil, fmt.Errorf("format %q is not tabular", format)
|
||||
}
|
||||
}
|
||||
|
||||
func readMarkdown(input []byte) (table, error) {
|
||||
input = bytes.TrimPrefix(input, []byte{0xEF, 0xBB, 0xBF})
|
||||
if !utf8.Valid(input) {
|
||||
return table{}, fmt.Errorf("markdown input is not valid UTF-8")
|
||||
}
|
||||
text := strings.ReplaceAll(string(input), "\r\n", "\n")
|
||||
text = strings.TrimSpace(text)
|
||||
if text == "" {
|
||||
return table{}, nil
|
||||
}
|
||||
lines := strings.Split(text, "\n")
|
||||
if len(lines) < 2 {
|
||||
return table{}, fmt.Errorf("markdown table requires a header and separator row")
|
||||
}
|
||||
header, err := parseMarkdownRow(lines[0])
|
||||
if err != nil {
|
||||
return table{}, fmt.Errorf("markdown header: %w", err)
|
||||
}
|
||||
separator, err := parseMarkdownRow(lines[1])
|
||||
if err != nil {
|
||||
return table{}, fmt.Errorf("markdown separator: %w", err)
|
||||
}
|
||||
if len(separator) != len(header) {
|
||||
return table{}, fmt.Errorf("markdown separator has %d fields but the header has %d", len(separator), len(header))
|
||||
}
|
||||
for i, cell := range separator {
|
||||
value := strings.TrimSpace(cell)
|
||||
value = strings.TrimPrefix(value, ":")
|
||||
value = strings.TrimSuffix(value, ":")
|
||||
if len(value) < 3 || strings.Trim(value, "-") != "" {
|
||||
return table{}, fmt.Errorf("markdown separator field %d is invalid", i+1)
|
||||
}
|
||||
}
|
||||
rows := make([][]string, 0, len(lines)-2)
|
||||
for i, line := range lines[2:] {
|
||||
if strings.TrimSpace(line) == "" {
|
||||
continue
|
||||
}
|
||||
row, err := parseMarkdownRow(line)
|
||||
if err != nil {
|
||||
return table{}, fmt.Errorf("markdown row %d: %w", i+3, err)
|
||||
}
|
||||
if len(row) > len(header) {
|
||||
return table{}, fmt.Errorf("row %d has %d fields but the header has %d", i+3, len(row), len(header))
|
||||
}
|
||||
rows = append(rows, padRow(row, len(header)))
|
||||
}
|
||||
return table{Header: header, Rows: rows}, nil
|
||||
}
|
||||
|
||||
func parseMarkdownRow(line string) ([]string, error) {
|
||||
line = strings.TrimSpace(strings.TrimSuffix(line, "\r"))
|
||||
if strings.HasPrefix(line, "|") {
|
||||
line = line[1:]
|
||||
}
|
||||
if hasUnescapedTrailingPipe(line) {
|
||||
line = strings.TrimSpace(line[:len(line)-1])
|
||||
}
|
||||
var cells []string
|
||||
var cell strings.Builder
|
||||
escaped := false
|
||||
for _, r := range line {
|
||||
if escaped {
|
||||
cell.WriteRune('\\')
|
||||
cell.WriteRune(r)
|
||||
escaped = false
|
||||
continue
|
||||
}
|
||||
if r == '\\' {
|
||||
escaped = true
|
||||
continue
|
||||
}
|
||||
if r == '|' {
|
||||
cells = append(cells, decodeMarkdownCell(strings.Trim(cell.String(), " \t")))
|
||||
cell.Reset()
|
||||
continue
|
||||
}
|
||||
cell.WriteRune(r)
|
||||
}
|
||||
if escaped {
|
||||
return nil, fmt.Errorf("row ends with an incomplete escape")
|
||||
}
|
||||
cells = append(cells, decodeMarkdownCell(strings.Trim(cell.String(), " \t")))
|
||||
return cells, nil
|
||||
}
|
||||
|
||||
func decodeMarkdownCell(value string) string {
|
||||
var b strings.Builder
|
||||
escaped := false
|
||||
for _, r := range value {
|
||||
if escaped {
|
||||
if r == '\\' || r == '|' {
|
||||
b.WriteRune(r)
|
||||
} else {
|
||||
b.WriteRune('\\')
|
||||
b.WriteRune(r)
|
||||
}
|
||||
escaped = false
|
||||
continue
|
||||
}
|
||||
if r == '\\' {
|
||||
escaped = true
|
||||
continue
|
||||
}
|
||||
b.WriteRune(r)
|
||||
}
|
||||
return html.UnescapeString(b.String())
|
||||
}
|
||||
|
||||
func hasUnescapedTrailingPipe(line string) bool {
|
||||
line = strings.TrimSpace(line)
|
||||
if !strings.HasSuffix(line, "|") {
|
||||
return false
|
||||
}
|
||||
backslashes := 0
|
||||
for i := len(line) - 2; i >= 0 && line[i] == '\\'; i-- {
|
||||
backslashes++
|
||||
}
|
||||
return backslashes%2 == 0
|
||||
}
|
||||
|
||||
func writeMarkdown(t table) ([]byte, error) {
|
||||
if len(t.Header) == 0 && len(t.Rows) == 0 {
|
||||
return nil, nil
|
||||
}
|
||||
if len(t.Header) == 0 {
|
||||
return nil, fmt.Errorf("markdown table has rows but no header")
|
||||
}
|
||||
var b strings.Builder
|
||||
writeRow := func(row []string) {
|
||||
b.WriteString("| ")
|
||||
for i, value := range row {
|
||||
if i > 0 {
|
||||
b.WriteString(" | ")
|
||||
}
|
||||
b.WriteString(escapeMarkdownCell(value))
|
||||
}
|
||||
b.WriteString(" |\n")
|
||||
}
|
||||
writeRow(t.Header)
|
||||
b.WriteString("|")
|
||||
for range t.Header {
|
||||
b.WriteString(" --- |")
|
||||
}
|
||||
b.WriteByte('\n')
|
||||
for _, row := range t.Rows {
|
||||
writeRow(padRow(row, len(t.Header)))
|
||||
}
|
||||
return []byte(b.String()), nil
|
||||
}
|
||||
|
||||
func escapeMarkdownCell(value string) string {
|
||||
leadingEnd := len(value) - len(strings.TrimLeft(value, " \t"))
|
||||
trailingStart := len(strings.TrimRight(value, " \t"))
|
||||
var b strings.Builder
|
||||
for i, r := range value {
|
||||
switch r {
|
||||
case '\\':
|
||||
b.WriteString("\\\\")
|
||||
case '|':
|
||||
b.WriteString("\\|")
|
||||
case '\r':
|
||||
b.WriteString(" ")
|
||||
case '\n':
|
||||
b.WriteString(" ")
|
||||
case ' ':
|
||||
if i < leadingEnd || i >= trailingStart {
|
||||
b.WriteString(" ")
|
||||
} else {
|
||||
b.WriteByte(' ')
|
||||
}
|
||||
case '\t':
|
||||
if i < leadingEnd || i >= trailingStart {
|
||||
b.WriteString("	")
|
||||
} else {
|
||||
b.WriteByte('\t')
|
||||
}
|
||||
default:
|
||||
switch r {
|
||||
case '&':
|
||||
b.WriteString("&")
|
||||
case '<':
|
||||
b.WriteString("<")
|
||||
case '>':
|
||||
b.WriteString(">")
|
||||
default:
|
||||
b.WriteRune(r)
|
||||
}
|
||||
}
|
||||
}
|
||||
return b.String()
|
||||
}
|
||||
|
||||
func writeDelimited(t table, comma rune) ([]byte, error) {
|
||||
var b bytes.Buffer
|
||||
w := csv.NewWriter(&b)
|
||||
|
||||
@@ -3,7 +3,7 @@
|
||||
`dataxl` は、ファイル形式ごとの差を小さくするために、内部表現を2つに分けています。
|
||||
|
||||
- structured value: JSON/YAML/TOMLから読める `map[string]any`, `[]any`, scalar
|
||||
- table: Excel/CSV/TSVに近い `Header []string` と `Rows [][]string`
|
||||
- table: Excel/CSV/TSV/Markdownに近い `Header []string` と `Rows [][]string`
|
||||
|
||||
変換は原則として次のどれかです。
|
||||
|
||||
@@ -19,7 +19,7 @@
|
||||
- `main.go`: CLI option、stdin/stdout、ファイル入出力
|
||||
- `conversion.go`: 変換経路の選択、形式名の正規化・推定
|
||||
- `structured.go`: JSON/YAML/TOML adapter
|
||||
- `table.go`: CSV/TSV/XLSX adapter、table model、flatten
|
||||
- `table.go`: CSV/TSV/XLSX/Markdown adapter、table model、flatten
|
||||
- `path.go`: セル値の型推定、列パスのparse、unflatten
|
||||
|
||||
`convert` はファイル入出力から独立しているため、CLIを経由せず変換matrixをテストできます。
|
||||
@@ -40,9 +40,12 @@ type table struct {
|
||||
}
|
||||
```
|
||||
|
||||
CSV/TSV/XLSXの読み込みでは、短い行を空文字で埋めて列数を揃えます。ヘッダーより
|
||||
CSV/TSV/XLSX/Markdownの読み込みでは、短い行を空文字で埋めて列数を揃えます。ヘッダーより
|
||||
長い行は、名前のない値を破棄しないようエラーにします。
|
||||
XLSXの書き出しではヘッダーを太字にし、1行目を固定します。
|
||||
Markdownでは区切り行のalignment markerを受け付けますが、table modelには保持しません。
|
||||
セル内のdelimiterとバックスラッシュはbackslash escape、改行と前後空白はHTML文字参照で
|
||||
可逆化します。
|
||||
|
||||
## Flattening
|
||||
|
||||
@@ -129,11 +132,11 @@ TOMLはトップレベル配列を直接表せないため、表からTOMLへ出
|
||||
- `gopkg.in/yaml.v3`: YAML読み書き
|
||||
- `github.com/BurntSushi/toml`: TOML読み書き
|
||||
|
||||
Go 1.25以上を前提にしています。
|
||||
Go 1.25.13以上を前提にしています。
|
||||
|
||||
## Error handling
|
||||
|
||||
- 未対応形式、decode失敗、workbook/sheet操作失敗は呼び出し元へerrorを返します。
|
||||
- CSV/TSVのinvalid UTF-8、headerより長い行、値を持つ空header列を拒否します。
|
||||
- CSV/TSV/Markdownのinvalid UTF-8、headerより長い行、値を持つ空header列を拒否します。
|
||||
- path復元中の重複header、構文エラー、型競合は変換全体をerrorにします。
|
||||
- XLSXのstyle・pane設定も通常の変換errorとして扱い、不完全なworkbookを成功扱いしません。
|
||||
|
||||
+2
-1
@@ -2,7 +2,7 @@
|
||||
|
||||
## Requirements
|
||||
|
||||
- Go 1.25 or later
|
||||
- Go 1.25.13 or later
|
||||
|
||||
## Setup
|
||||
|
||||
@@ -58,6 +58,7 @@ The current tests cover:
|
||||
- YAML -> TSV flattening
|
||||
- TSV -> JSON path restoration
|
||||
- JSON -> XLSX -> JSON round trip
|
||||
- Markdown table parsing, escaping, and all-format conversion round trips
|
||||
- structured -> structured conversion without CLI/file I/O
|
||||
- extension normalization and format inference
|
||||
- short table row padding and wider-row rejection
|
||||
|
||||
Reference in New Issue
Block a user