Class: SplitPgDump::Table

Inherits:
Object
  • Object
show all
Includes:
NativeComputeName, ComputeName
Defined in:
lib/split_pgdump.rb

Defined Under Namespace

Modules: ComputeName Classes: NoColumn, OneFile

Constant Summary collapse

ONE_FILE_CACHE_SIZE =
3 * 128 * 1024
TOTAL_CACHE_SIZE =
4 * 128 * 1024

Instance Attribute Summary collapse

Instance Method Summary collapse

Methods included from ComputeName

#compute_name, #native_compute_name?

Methods included from NativeComputeName

#compute_name, #native_compute_name?

Constructor Details

#initialize(dir, schema, name, columns, rule) ⇒ Table

Returns a new instance of Table.



297
298
299
300
301
302
303
304
305
306
307
# File 'lib/split_pgdump.rb', line 297

def initialize(dir, schema, name, columns, rule)
  @dir = dir
  @table = name
  @schema = schema
  @columns = columns.map{|c| c.sub(/^"(.+)"$/, '\\1')}
  @file_name = "#{table_schema}.dat"
  apply_rule rule
  @files = {}
  @files_to_flush = {}
  @total_cache_size = 0
end

Instance Attribute Details

#columns ⇒ Object (readonly)

Returns the value of attribute columns.



296
297
298
# File 'lib/split_pgdump.rb', line 296

def columns
  @columns
end

#files ⇒ Object (readonly)

Returns the value of attribute files.



296
297
298
# File 'lib/split_pgdump.rb', line 296

def files
  @files
end

#sort_args ⇒ Object (readonly)

Returns the value of attribute sort_args.



296
297
298
# File 'lib/split_pgdump.rb', line 296

def sort_args
  @sort_args
end

#sort_line ⇒ Object (readonly)

Returns the value of attribute sort_line.



296
297
298
# File 'lib/split_pgdump.rb', line 296

def sort_line
  @sort_line
end

#table ⇒ Object (readonly)

Returns the value of attribute table.



296
297
298
# File 'lib/split_pgdump.rb', line 296

def table
  @table
end

Instance Method Details

#_mod(s, format, mod) ⇒ Object



309
310
311
# File 'lib/split_pgdump.rb', line 309

def _mod(s, format, mod)
  format % (s.to_i / mod * mod)
end

#add_line(line) ⇒ Object



389
390
391
392
393
394
395
396
397
398
399
400
401
402
# File 'lib/split_pgdump.rb', line 389

def add_line(line)
  fname = @split_rule ? file_name(line) : @file_name
  one_file = @files[fname] ||= OneFile.new(@dir, fname)

  @files_to_flush[one_file] = true  if one_file.cache_size == 0

  one_file.add_line(line)
  @total_cache_size += line.size
  if one_file.cache_size > ONE_FILE_CACHE_SIZE
    @total_cache_size -= one_file.cache_size
    one_file.flush
  end
  flush_all if @total_cache_size > TOTAL_CACHE_SIZE
end

#apply_rule(rule) ⇒ Object



313
314
315
316
317
318
319
320
321
322
323
324
325
326
327
328
329
330
331
332
333
334
335
336
337
338
339
340
341
342
343
344
345
346
347
348
349
350
351
352
353
354
355
356
357
358
359
360
361
362
363
364
365
366
367
368
369
370
371
372
373
# File 'lib/split_pgdump.rb', line 313

def apply_rule(rule)
  if rule
    unless rule.split_parts.empty?
      if native_compute_name?
        @split_rule = rule.split_parts.map do |part|
          case part
          when Array # field manipulations
            unless i = @columns.index(part[0])
              raise NoColumn, "Table #{@schema}.#{@table} has no column #{part[0]} for use in split"
            end
            [i, part[1..-1]]
          else
            part
          end
        end
      else
        split_string = ''
        split_rule = []
        rule.split_parts.map do |part|
          case part
          when Array #field manipulation
            unless i = @columns.index(part[0])
              raise NoColumn, "Table #{@schema}.#{@table} has no column #{part[0]} for use in split"
            end
            field = "values[#{i}]"
            part[1..-1].each do |action|
              ssize = split_rule.size
              case action
              when Range
                field << "[split_rule[#{ssize}]]"
                split_rule << action
              when Array # take module
                field = "_mod(#{field}, split_rule[#{ssize}], split_rule[#{ssize+1}])"
                split_rule.concat action
              end
            end
            split_string << "\#{#{field}}"
          when String
            split_string << part
          end
        end
        @split_rule = split_rule
        eval <<-"EOF"
          def self.compute_name(split_rule, values)
            %{#{split_string}}
          end
        EOF
      end
      @file_name = {}
    end

    @sort_args = rule.sort_keys.map do |key|
      i = @columns.find_index(key[:field])
      raise NoColumn, "Table #{@schema}.#{@table} has no column #{key[:field]} for use in sort"  unless i
      i += 1
      "--key=#{i},#{i}#{key[:flags]}"
    end
  else
    @sort_args = []
  end
end

#copy_lines ⇒ Object



410
411
412
413
414
415
416
417
418
# File 'lib/split_pgdump.rb', line 410

def copy_lines
  if block_given?
    @files.map{|n, one_file| one_file.file_name}.sort.each do |file_name|
      yield "\\copy #{@table} (#{@columns.join(', ')}) from #{file_name}"
    end
  else
    to_enum(:copy_lines)
  end
end

#file_name(line) ⇒ Object



379
380
381
382
383
384
385
386
387
# File 'lib/split_pgdump.rb', line 379

def file_name(line)
  values = line.split("\t")
  values.last.chomp!
  name = compute_name(@split_rule, values)
  @file_name[name] ||= begin
    name_strip = name.gsub(/\.\.|\s|\?|\*|'|"/, '_')
    "#{table_schema}/#{name_strip}.dat"
  end
end

#flush_all ⇒ Object



404
405
406
407
408
# File 'lib/split_pgdump.rb', line 404

def flush_all
  @files_to_flush.each{|one_file, _| one_file.flush }
  @files_to_flush.clear
  @total_cache_size = 0
end

#table_schema ⇒ Object



375
376
377
# File 'lib/split_pgdump.rb', line 375

def table_schema
  @schema == 'public' ? @table : "#@schema/#@table"
end