Merge pull request 'Markdownテーブル形式に対応' (#8) from feature/markdown-table-format into main
CI / test (push) Successful in 11s
Release / release (push) Successful in 1m54s

Reviewed-on: #8
This commit was merged in pull request #8.
This commit is contained in:
2026-09-05 16:59:36 +09:00
11 changed files with 284 additions and 20 deletions
+19 -3
View File
@@ -15,10 +15,11 @@ TSVへ変換したり、Excelからコピーした表を再び構造化データ
- CSV - CSV
- TSV - TSV
- XLSX - XLSX
- Markdown table (`markdown` / `md`)
## インストール ## インストール
Go 1.25以上が必要です。 Go 1.25.13以上が必要です。
```sh ```sh
go install git.rumginger.org/agent/dataxl/cmd/dataxl@latest go install git.rumginger.org/agent/dataxl/cmd/dataxl@latest
@@ -55,6 +56,12 @@ JSONからExcel workbookを作成:
dataxl -from json -to xlsx -i people.json -o people.xlsx dataxl -from json -to xlsx -i people.json -o people.xlsx
``` ```
Markdownのテーブルを作成:
```sh
dataxl -from yaml -to markdown -i examples/people.yaml -o people.md
```
Excel workbookをYAMLへ変換: Excel workbookをYAMLへ変換:
```sh ```sh
@@ -86,6 +93,15 @@ Excelから戻す場合は、シート上の範囲をコピーしてstdinへ渡
dataxl -from tsv -to yaml > restored.yaml dataxl -from tsv -to yaml > restored.yaml
``` ```
## Markdownテーブル
GitHub Flavored Markdown互換のヘッダー・区切り行を持つテーブルを読み書きできます。
外側の `|` は省略可能で、区切り行の `:` による左寄せ・中央寄せ・右寄せ指定も
入力時に受け付けます(配置指定自体は保持しません)。
セル内の `|` とバックスラッシュはMarkdownのbackslash escapeを使用します。
改行や前後の空白・タブはHTML文字参照として表し、表形式間の変換で保持します。
## 入れ子構造の表現 ## 入れ子構造の表現
構造化データの入れ子は、Excelで編集しやすい列名へ展開されます。 構造化データの入れ子は、Excelで編集しやすい列名へ展開されます。
@@ -154,8 +170,8 @@ CSV、TSV、XLSXから構造化形式へ戻す際は、セル文字列を次の
- `-i`: 入力ファイル。省略時はstdin。 - `-i`: 入力ファイル。省略時はstdin。
- `-o`: 出力ファイル。省略時はstdout。 - `-o`: 出力ファイル。省略時はstdout。
- `-from`: 入力形式。`json`, `yaml`, `toml`, `csv`, `tsv`, `xlsx`。 - `-from`: 入力形式。`json`, `yaml`, `toml`, `csv`, `tsv`, `xlsx`, `markdown`。
- `-to`: 出力形式。`json`, `yaml`, `toml`, `csv`, `tsv`, `xlsx`。 - `-to`: 出力形式。`json`, `yaml`, `toml`, `csv`, `tsv`, `xlsx`, `markdown`。
- `-sheet`: XLSXの読み書きに使うシート名。既定値は `Sheet1`。 - `-sheet`: XLSXの読み書きに使うシート名。既定値は `Sheet1`。
- `-pretty`: JSONなどの構造化出力を整形するか。既定値は `true`。 - `-pretty`: JSONなどの構造化出力を整形するか。既定値は `true`。
- `-version`: バージョンを表示して終了します。 - `-version`: バージョンを表示して終了します。
+4 -2
View File
@@ -51,6 +51,8 @@ func normalizeFormat(format string) string {
switch format { switch format {
case "yml": case "yml":
return "yaml" return "yaml"
case "md", "mdown", "mkd":
return "markdown"
case "xlsm", "xls": case "xlsm", "xls":
return "xlsx" return "xlsx"
default: default:
@@ -67,7 +69,7 @@ func inferFormat(path string) string {
func validateFormat(format string) error { func validateFormat(format string) error {
switch format { switch format {
case "json", "yaml", "toml", "csv", "tsv", "xlsx": case "json", "yaml", "toml", "csv", "tsv", "xlsx", "markdown":
return nil return nil
default: default:
return fmt.Errorf("unsupported format %q", format) return fmt.Errorf("unsupported format %q", format)
@@ -75,5 +77,5 @@ func validateFormat(format string) error {
} }
func isTabular(format string) bool { func isTabular(format string) bool {
return format == "csv" || format == "tsv" || format == "xlsx" return format == "csv" || format == "tsv" || format == "xlsx" || format == "markdown"
} }
+1 -1
View File
@@ -6,7 +6,7 @@ import (
) )
func TestConversionMatrixPreservesRepresentativeTable(t *testing.T) { func TestConversionMatrixPreservesRepresentativeTable(t *testing.T) {
formats := []string{"json", "yaml", "toml", "csv", "tsv", "xlsx"} formats := []string{"json", "yaml", "toml", "csv", "tsv", "xlsx", "markdown"}
records := []any{ records := []any{
map[string]any{ map[string]any{
"active": true, "active": true,
+2 -2
View File
@@ -220,7 +220,7 @@ func TestDelimitedSingleEmptyCellRowRoundTrip(t *testing.T) {
func TestStructuredRecordsWithoutScalarFieldsAreRejectedForTables(t *testing.T) { func TestStructuredRecordsWithoutScalarFieldsAreRejectedForTables(t *testing.T) {
for _, input := range []string{`{}`, `[{}]`} { for _, input := range []string{`{}`, `[{}]`} {
for _, format := range []string{"csv", "tsv", "xlsx"} { for _, format := range []string{"csv", "tsv", "xlsx", "markdown"} {
t.Run(format+"_"+input, func(t *testing.T) { t.Run(format+"_"+input, func(t *testing.T) {
_, err := convert([]byte(input), "json", format, "Sheet1", false) _, err := convert([]byte(input), "json", format, "Sheet1", false)
if err == nil || !strings.Contains(err.Error(), "no scalar fields") { if err == nil || !strings.Contains(err.Error(), "no scalar fields") {
@@ -232,7 +232,7 @@ func TestStructuredRecordsWithoutScalarFieldsAreRejectedForTables(t *testing.T)
} }
func TestEmptyRecordListCanRoundTripThroughDelimitedFormats(t *testing.T) { func TestEmptyRecordListCanRoundTripThroughDelimitedFormats(t *testing.T) {
for _, format := range []string{"csv", "tsv"} { for _, format := range []string{"csv", "tsv", "markdown"} {
t.Run(format, func(t *testing.T) { t.Run(format, func(t *testing.T) {
tabular, err := convert([]byte(`[]`), "json", format, "Sheet1", false) tabular, err := convert([]byte(`[]`), "json", format, "Sheet1", false)
if err != nil { if err != nil {
+1 -1
View File
@@ -1,7 +1,7 @@
// Package main implements the dataxl command-line converter. // Package main implements the dataxl command-line converter.
// //
// Conversion uses two internal representations: structured Go values for // Conversion uses two internal representations: structured Go values for
// JSON/YAML/TOML and table for CSV/TSV/XLSX. Nested structured values cross the // JSON/YAML/TOML and table for CSV/TSV/XLSX/Markdown. Nested structured values cross the
// boundary through dotted and indexed column paths such as user.name and // boundary through dotted and indexed column paths such as user.name and
// items[0].sku. // items[0].sku.
package main package main
+3 -3
View File
@@ -60,8 +60,8 @@ func parseOptions(args []string, stderr io.Writer) (options, error) {
fs.SetOutput(stderr) fs.SetOutput(stderr)
fs.StringVar(&opt.inFile, "i", "", "input file, defaults to stdin") fs.StringVar(&opt.inFile, "i", "", "input file, defaults to stdin")
fs.StringVar(&opt.outFile, "o", "", "output file, defaults to stdout") fs.StringVar(&opt.outFile, "o", "", "output file, defaults to stdout")
fs.StringVar(&opt.from, "from", "", "input format: json, yaml, toml, csv, tsv, xlsx") fs.StringVar(&opt.from, "from", "", "input format: json, yaml, toml, csv, tsv, xlsx, markdown")
fs.StringVar(&opt.to, "to", "", "output format: json, yaml, toml, csv, tsv, xlsx") fs.StringVar(&opt.to, "to", "", "output format: json, yaml, toml, csv, tsv, xlsx, markdown")
fs.StringVar(&opt.sheet, "sheet", "Sheet1", "worksheet name for xlsx input/output") fs.StringVar(&opt.sheet, "sheet", "Sheet1", "worksheet name for xlsx input/output")
fs.BoolVar(&opt.pretty, "pretty", true, "pretty-print structured output") fs.BoolVar(&opt.pretty, "pretty", true, "pretty-print structured output")
fs.BoolVar(&opt.version, "version", false, "print version and exit") fs.BoolVar(&opt.version, "version", false, "print version and exit")
@@ -72,7 +72,7 @@ func parseOptions(args []string, stderr io.Writer) (options, error) {
dataxl -from json -to xlsx -i data.json -o data.xlsx dataxl -from json -to xlsx -i data.json -o data.xlsx
Formats: Formats:
json, yaml/yml, toml, csv, tsv, xlsx json, yaml/yml, toml, csv, tsv, xlsx, markdown/md
Notes: Notes:
Structured records are flattened into spreadsheet columns such as user.name Structured records are flattened into spreadsheet columns such as user.name
+43
View File
@@ -129,6 +129,49 @@ func TestResolveFormatsFromFileExtensions(t *testing.T) {
} }
} }
func TestResolveMarkdownFormatFromFileExtension(t *testing.T) {
opt := options{inFile: "input.json", outFile: "output.md"}
if err := opt.resolveFormats(); err != nil {
t.Fatal(err)
}
if opt.from != "json" || opt.to != "markdown" {
t.Fatalf("resolved formats = %q -> %q, want json -> markdown", opt.from, opt.to)
}
}
func TestMarkdownTableRoundTrip(t *testing.T) {
want := table{
Header: []string{"id", "name", "note"},
Rows: [][]string{{"1", "Alice | Bob", "line 1\nline 2\\n"}},
}
encoded, err := encodeTable(want, "markdown", "")
if err != nil {
t.Fatal(err)
}
got, err := parseTable(encoded, "markdown", "")
if err != nil {
t.Fatalf("parse generated Markdown: %v\n%s", err, encoded)
}
if !reflect.DeepEqual(got, want) {
t.Fatalf("table = %#v, want %#v\n%s", got, want, encoded)
}
}
func TestMarkdownTableAcceptsAlignmentAndOptionalOuterPipes(t *testing.T) {
input := []byte("id | name | note\n---: | :--- | :---:\n1 | Alice \\| Bob | ok\n")
got, err := parseTable(input, "markdown", "")
if err != nil {
t.Fatal(err)
}
want := table{
Header: []string{"id", "name", "note"},
Rows: [][]string{{"1", "Alice | Bob", "ok"}},
}
if !reflect.DeepEqual(got, want) {
t.Fatalf("table = %#v, want %#v", got, want)
}
}
func TestVersionFlagDoesNotRequireFormats(t *testing.T) { func TestVersionFlagDoesNotRequireFormats(t *testing.T) {
oldVersion := buildVersion oldVersion := buildVersion
buildVersion = "v9.8.7-test" buildVersion = "v9.8.7-test"
+200 -1
View File
@@ -5,14 +5,16 @@ import (
"encoding/csv" "encoding/csv"
"encoding/json" "encoding/json"
"fmt" "fmt"
"html"
"sort" "sort"
"strconv" "strconv"
"strings"
"unicode/utf8" "unicode/utf8"
"github.com/xuri/excelize/v2" "github.com/xuri/excelize/v2"
) )
// table is the common representation for CSV, TSV and XLSX. Rows are padded // table is the common representation for CSV, TSV, XLSX and Markdown. Rows are padded
// to Header width when read, which keeps subsequent conversions rectangular. // to Header width when read, which keeps subsequent conversions rectangular.
type table struct { type table struct {
Header []string Header []string
@@ -36,6 +38,8 @@ func parseTable(input []byte, format, sheet string) (table, error) {
return table{}, err return table{}, err
} }
return rowsToTable(rows) return rowsToTable(rows)
case "markdown":
return readMarkdown(input)
default: default:
return table{}, fmt.Errorf("format %q is not tabular", format) return table{}, fmt.Errorf("format %q is not tabular", format)
} }
@@ -93,11 +97,206 @@ func encodeTable(t table, format, sheet string) ([]byte, error) {
return writeDelimited(t, '\t') return writeDelimited(t, '\t')
case "xlsx": case "xlsx":
return writeXLSX(t, sheet) return writeXLSX(t, sheet)
case "markdown":
return writeMarkdown(t)
default: default:
return nil, fmt.Errorf("format %q is not tabular", format) return nil, fmt.Errorf("format %q is not tabular", format)
} }
} }
func readMarkdown(input []byte) (table, error) {
input = bytes.TrimPrefix(input, []byte{0xEF, 0xBB, 0xBF})
if !utf8.Valid(input) {
return table{}, fmt.Errorf("markdown input is not valid UTF-8")
}
text := strings.ReplaceAll(string(input), "\r\n", "\n")
text = strings.TrimSpace(text)
if text == "" {
return table{}, nil
}
lines := strings.Split(text, "\n")
if len(lines) < 2 {
return table{}, fmt.Errorf("markdown table requires a header and separator row")
}
header, err := parseMarkdownRow(lines[0])
if err != nil {
return table{}, fmt.Errorf("markdown header: %w", err)
}
separator, err := parseMarkdownRow(lines[1])
if err != nil {
return table{}, fmt.Errorf("markdown separator: %w", err)
}
if len(separator) != len(header) {
return table{}, fmt.Errorf("markdown separator has %d fields but the header has %d", len(separator), len(header))
}
for i, cell := range separator {
value := strings.TrimSpace(cell)
value = strings.TrimPrefix(value, ":")
value = strings.TrimSuffix(value, ":")
if len(value) < 3 || strings.Trim(value, "-") != "" {
return table{}, fmt.Errorf("markdown separator field %d is invalid", i+1)
}
}
rows := make([][]string, 0, len(lines)-2)
for i, line := range lines[2:] {
if strings.TrimSpace(line) == "" {
continue
}
row, err := parseMarkdownRow(line)
if err != nil {
return table{}, fmt.Errorf("markdown row %d: %w", i+3, err)
}
if len(row) > len(header) {
return table{}, fmt.Errorf("row %d has %d fields but the header has %d", i+3, len(row), len(header))
}
rows = append(rows, padRow(row, len(header)))
}
return table{Header: header, Rows: rows}, nil
}
func parseMarkdownRow(line string) ([]string, error) {
line = strings.TrimSpace(strings.TrimSuffix(line, "\r"))
if strings.HasPrefix(line, "|") {
line = line[1:]
}
if hasUnescapedTrailingPipe(line) {
line = strings.TrimSpace(line[:len(line)-1])
}
var cells []string
var cell strings.Builder
escaped := false
for _, r := range line {
if escaped {
cell.WriteRune('\\')
cell.WriteRune(r)
escaped = false
continue
}
if r == '\\' {
escaped = true
continue
}
if r == '|' {
cells = append(cells, decodeMarkdownCell(strings.Trim(cell.String(), " \t")))
cell.Reset()
continue
}
cell.WriteRune(r)
}
if escaped {
return nil, fmt.Errorf("row ends with an incomplete escape")
}
cells = append(cells, decodeMarkdownCell(strings.Trim(cell.String(), " \t")))
return cells, nil
}
func decodeMarkdownCell(value string) string {
var b strings.Builder
escaped := false
for _, r := range value {
if escaped {
if r == '\\' || r == '|' {
b.WriteRune(r)
} else {
b.WriteRune('\\')
b.WriteRune(r)
}
escaped = false
continue
}
if r == '\\' {
escaped = true
continue
}
b.WriteRune(r)
}
return html.UnescapeString(b.String())
}
func hasUnescapedTrailingPipe(line string) bool {
line = strings.TrimSpace(line)
if !strings.HasSuffix(line, "|") {
return false
}
backslashes := 0
for i := len(line) - 2; i >= 0 && line[i] == '\\'; i-- {
backslashes++
}
return backslashes%2 == 0
}
func writeMarkdown(t table) ([]byte, error) {
if len(t.Header) == 0 && len(t.Rows) == 0 {
return nil, nil
}
if len(t.Header) == 0 {
return nil, fmt.Errorf("markdown table has rows but no header")
}
var b strings.Builder
writeRow := func(row []string) {
b.WriteString("| ")
for i, value := range row {
if i > 0 {
b.WriteString(" | ")
}
b.WriteString(escapeMarkdownCell(value))
}
b.WriteString(" |\n")
}
writeRow(t.Header)
b.WriteString("|")
for range t.Header {
b.WriteString(" --- |")
}
b.WriteByte('\n')
for _, row := range t.Rows {
writeRow(padRow(row, len(t.Header)))
}
return []byte(b.String()), nil
}
func escapeMarkdownCell(value string) string {
leadingEnd := len(value) - len(strings.TrimLeft(value, " \t"))
trailingStart := len(strings.TrimRight(value, " \t"))
var b strings.Builder
for i, r := range value {
switch r {
case '\\':
b.WriteString("\\\\")
case '|':
b.WriteString("\\|")
case '\r':
b.WriteString("&#13;")
case '\n':
b.WriteString("&#10;")
case ' ':
if i < leadingEnd || i >= trailingStart {
b.WriteString("&#32;")
} else {
b.WriteByte(' ')
}
case '\t':
if i < leadingEnd || i >= trailingStart {
b.WriteString("&#9;")
} else {
b.WriteByte('\t')
}
default:
switch r {
case '&':
b.WriteString("&amp;")
case '<':
b.WriteString("&lt;")
case '>':
b.WriteString("&gt;")
default:
b.WriteRune(r)
}
}
}
return b.String()
}
func writeDelimited(t table, comma rune) ([]byte, error) { func writeDelimited(t table, comma rune) ([]byte, error) {
var b bytes.Buffer var b bytes.Buffer
w := csv.NewWriter(&b) w := csv.NewWriter(&b)
+8 -5
View File
@@ -3,7 +3,7 @@
`dataxl` は、ファイル形式ごとの差を小さくするために、内部表現を2つに分けています。 `dataxl` は、ファイル形式ごとの差を小さくするために、内部表現を2つに分けています。
- structured value: JSON/YAML/TOMLから読める `map[string]any`, `[]any`, scalar - structured value: JSON/YAML/TOMLから読める `map[string]any`, `[]any`, scalar
- table: Excel/CSV/TSVに近い `Header []string` と `Rows [][]string` - table: Excel/CSV/TSV/Markdownに近い `Header []string` と `Rows [][]string`
変換は原則として次のどれかです。 変換は原則として次のどれかです。
@@ -19,7 +19,7 @@
- `main.go`: CLI option、stdin/stdout、ファイル入出力 - `main.go`: CLI option、stdin/stdout、ファイル入出力
- `conversion.go`: 変換経路の選択、形式名の正規化・推定 - `conversion.go`: 変換経路の選択、形式名の正規化・推定
- `structured.go`: JSON/YAML/TOML adapter - `structured.go`: JSON/YAML/TOML adapter
- `table.go`: CSV/TSV/XLSX adapter、table model、flatten - `table.go`: CSV/TSV/XLSX/Markdown adapter、table model、flatten
- `path.go`: セル値の型推定、列パスのparse、unflatten - `path.go`: セル値の型推定、列パスのparse、unflatten
`convert` はファイル入出力から独立しているため、CLIを経由せず変換matrixをテストできます。 `convert` はファイル入出力から独立しているため、CLIを経由せず変換matrixをテストできます。
@@ -40,9 +40,12 @@ type table struct {
} }
``` ```
CSV/TSV/XLSXの読み込みでは、短い行を空文字で埋めて列数を揃えます。ヘッダーより CSV/TSV/XLSX/Markdownの読み込みでは、短い行を空文字で埋めて列数を揃えます。ヘッダーより
長い行は、名前のない値を破棄しないようエラーにします。 長い行は、名前のない値を破棄しないようエラーにします。
XLSXの書き出しではヘッダーを太字にし、1行目を固定します。 XLSXの書き出しではヘッダーを太字にし、1行目を固定します。
Markdownでは区切り行のalignment markerを受け付けますが、table modelには保持しません。
セル内のdelimiterとバックスラッシュはbackslash escape、改行と前後空白はHTML文字参照で
可逆化します。
## Flattening ## Flattening
@@ -129,11 +132,11 @@ TOMLはトップレベル配列を直接表せないため、表からTOMLへ出
- `gopkg.in/yaml.v3`: YAML読み書き - `gopkg.in/yaml.v3`: YAML読み書き
- `github.com/BurntSushi/toml`: TOML読み書き - `github.com/BurntSushi/toml`: TOML読み書き
Go 1.25以上を前提にしています。 Go 1.25.13以上を前提にしています。
## Error handling ## Error handling
- 未対応形式、decode失敗、workbook/sheet操作失敗は呼び出し元へerrorを返します。 - 未対応形式、decode失敗、workbook/sheet操作失敗は呼び出し元へerrorを返します。
- CSV/TSVのinvalid UTF-8、headerより長い行、値を持つ空header列を拒否します。 - CSV/TSV/Markdownのinvalid UTF-8、headerより長い行、値を持つ空header列を拒否します。
- path復元中の重複header、構文エラー、型競合は変換全体をerrorにします。 - path復元中の重複header、構文エラー、型競合は変換全体をerrorにします。
- XLSXのstyle・pane設定も通常の変換errorとして扱い、不完全なworkbookを成功扱いしません。 - XLSXのstyle・pane設定も通常の変換errorとして扱い、不完全なworkbookを成功扱いしません。
+2 -1
View File
@@ -2,7 +2,7 @@
## Requirements ## Requirements
- Go 1.25 or later - Go 1.25.13 or later
## Setup ## Setup
@@ -58,6 +58,7 @@ The current tests cover:
- YAML -> TSV flattening - YAML -> TSV flattening
- TSV -> JSON path restoration - TSV -> JSON path restoration
- JSON -> XLSX -> JSON round trip - JSON -> XLSX -> JSON round trip
- Markdown table parsing, escaping, and all-format conversion round trips
- structured -> structured conversion without CLI/file I/O - structured -> structured conversion without CLI/file I/O
- extension normalization and format inference - extension normalization and format inference
- short table row padding and wider-row rejection - short table row padding and wider-row rejection
+1 -1
View File
@@ -1,6 +1,6 @@
module git.rumginger.org/agent/dataxl module git.rumginger.org/agent/dataxl
go 1.25.0 go 1.25.13
require ( require (
github.com/BurntSushi/toml v1.6.0 github.com/BurntSushi/toml v1.6.0