Class: Embulk::Guess::CsvGuessPlugin

Inherits:
LineGuessPlugin show all
Defined in:
lib/embulk/guess/csv.rb

Defined Under Namespace

Classes: TimestampMatch

Constant Summary collapse

DELIMITER_CANDIDATES =
[
  ",", "\t", "|"
]
QUOTE_CANDIDATES =
[
  "\"", "'"
]
ESCAPE_CANDIDATES =
[
  "\\"
]
NULL_STRING_CANDIDATES =
[
  "null",
  "NULL",
  "#N/A",
  "\\N",  # MySQL LOAD, Hive STORED AS TEXTFILE
]
TRUE_STRINGS =

CsvParserPlugin.TRUE_STRINGS

Instance Method Summary collapse

Methods inherited from LineGuessPlugin

#guess

Methods inherited from Embulk::GuessPlugin

from_java, #guess, new_java

Instance Method Details

#guess_lines(config, sample_lines) ⇒ Object



36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
# File 'lib/embulk/guess/csv.rb', line 36

def guess_lines(config, sample_lines)
  delim = guess_delimiter(sample_lines)
  unless delim
    # not CSV file
    return {}
  end

  parser_config = config["parser"] || {}
  parser_guessed = {"type" => "csv", "delimiter" => delim}

  quote = guess_quote(sample_lines, delim)
  parser_guessed["quote"] = quote ? quote : ''

  escape = guess_escape(sample_lines, delim, quote)
  parser_guessed["escape"] = escape ? escape : ''

  null_string = guess_null_string(sample_lines, delim)
  parser_guessed["null_string"] = null_string if null_string
  # don't even set null_string to avoid confusion of null and 'null' in YAML format

  sample_records = sample_lines.map {|line| line.split(delim) }  # TODO use CsvTokenizer
  first_types = guess_field_types(sample_records[0, 1])
  other_types = guess_field_types(sample_records[1..-1])

  if first_types.size <= 1 || other_types.size <= 1
    # guess failed
    return {}
  end

  unless parser_config.has_key?("header_line")
    parser_guessed["header_line"] = (first_types != other_types && !first_types.any? {|t| t != ["string"] })
  end

  unless parser_config.has_key?("columns")
    if parser_guessed["header_line"] || parser_config["header_line"]
      column_names = sample_records.first
    else
      column_names = (0..other_types.size).to_a.map {|i| "c#{i}" }
    end
    schema = []
    column_names.zip(other_types).each do |name,(type,format)|
      if name && type
        if format
          schema << {"name" => name, "type" => type, "format" => format}
        else
          schema << {"name" => name, "type" => type}
        end
      end
    end
    parser_guessed["columns"] = schema
  end

  return {"parser" => parser_guessed}
end