Class: Canopus::Snippet::Transform

Inherits:
Object
  • Object
show all
Defined in:
lib/canopus/snippet/transform.rb,
sig/snippet.rbs,
sig/workspace_services.rbs

Overview

Deliberately uses bounded Ruby Regexp, never eval or an external JS runtime. Supported syntax and JS portability boundaries are in docs/snippets.md.

Constant Summary collapse

TIMEOUT =

Returns:

  • (Float)
0.05
SPACE =
'\t\n\v\f\r \u00a0\u1680\u2000-\u200a\u2028\u2029\u202f\u205f\u3000\ufeff'
CASES =
%w[upcase downcase capitalize camelcase pascalcase snakecase kebabcase].freeze

Instance Attribute Summary collapse

Class Method Summary collapse

Instance Method Summary collapse

Constructor Details

#initialize(pattern, replacement, flags) ⇒ Transform

Returns a new instance of Transform.

Parameters:

  • pattern (String)
  • replacement (format_nodes)
  • flags (String)


91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
# File 'lib/canopus/snippet/transform.rb', line 91

def initialize(pattern, replacement, flags)
  raise Canopus::Error, "unsupported snippet transform flags: #{flags}" unless flags.match?(/\A[gimsu]*\z/) && flags.chars.uniq.length == flags.length
  raise Canopus::Error, "snippet transform pattern is too large" if pattern.bytesize > 65_536
  @pattern, @flags, @replacement = pattern.freeze, flags.freeze, replacement
  # Reject Ruby-only syntax and constructs with incompatible JS semantics.
  translated, in_class, scanner, groups = +"", false, StringScanner.new(pattern), []
  until scanner.eos?
    character = scanner.getch
    if character == "\\"
      following = scanner.getch
      raise Canopus::Error, "incomplete snippet regular expression escape" unless following
      raise Canopus::Error, "unsupported snippet regular expression escape: \\#{following}" if following.match?(/[AGKRXZzghkpPc1-9]/) || (following == "u" && flags.include?("i"))
      translated << case following
      when "s" then in_class ? SPACE : "[#{SPACE}]"
      when "S" then "[^#{SPACE}]"
      when "b" then in_class ? "\\b" : '(?:(?<=\w)(?!\w)|(?<!\w)(?=\w))'
      when "B"
        @nonword_boundary = true unless in_class
        in_class ? "B" : '(?:(?<=\w)(?=\w)|(?<!\w)(?!\w))'
      else character + following
      end
    else
      if !in_class && ((character == "(" && scanner.peek(1) == "?" && !scanner.rest.match?(/\A\?(?:[:=!]|<[=!])/)) || (character.match?(/[*+?}]/) && scanner.peek(1) == "+"))
        raise Canopus::Error, "unsupported snippet regular expression construct"
      end
      raise Canopus::Error, "unsupported snippet character-class intersection" if in_class && character == "&" && scanner.peek(1) == "&"
      if !in_class && character == "("
        capture = scanner.peek(1) != "?"
        groups.map! { |parent| parent || capture }
        groups << capture
        raise Canopus::Error, "snippet regular expression nesting exceeds #{Canopus::Snippet::MAX_DEPTH}" if groups.length > Canopus::Snippet::MAX_DEPTH
      elsif !in_class && character == ")"
        captures = groups.pop
        raise Canopus::Error, "captures inside repeated groups are not portable" if captures && scanner.peek(1).match?(/[*+{]/)
      end
      translated << case character
      when "[" then if in_class then "\\[" else in_class = true; character end
      when "]" then in_class = false; character
      when "^" then in_class ? character : flags.include?("m") ? '(?:\A|(?<=[\r\n\u2028\u2029]))' : '\A'
      when "$" then in_class ? character : flags.include?("m") ? '(?=\z|[\r\n\u2028\u2029])' : '\z'
      when "." then !in_class && !flags.include?("s") ? "[^\\n\\r\\u2028\\u2029]" : character
      else character
      end
    end
  end
  options = (flags.include?("i") ? Regexp::IGNORECASE : 0) | (flags.include?("s") ? Regexp::MULTILINE : 0)
  @regexp = Regexp.new(translated, options, timeout: TIMEOUT)
  freeze
rescue RegexpError => error
  raise Canopus::Error, "invalid snippet regular expression: #{error.message}"
end

Instance Attribute Details

#flags ⇒ String (readonly)

Returns the value of attribute flags.

Returns:

  • (String)


11
12
13
# File 'lib/canopus/snippet/transform.rb', line 11

def flags
  @flags
end

#pattern ⇒ String (readonly)

Returns the value of attribute pattern.

Returns:

  • (String)


11
12
13
# File 'lib/canopus/snippet/transform.rb', line 11

def pattern
  @pattern
end

Class Method Details

.parse(scanner, depth) ⇒ instance

Parameters:

  • scanner (StringScanner)
  • depth (Integer)

Returns:

  • (instance)

Raises:



13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
# File 'lib/canopus/snippet/transform.rb', line 13

def self.parse(scanner, depth)
  pattern = +""
  closed = false
  until scanner.eos?
    character = scanner.getch
    if character == "\\"
      following = scanner.getch
      raise Canopus::Error, "unclosed snippet transform" unless following
      pattern << (following == "/" ? "/" : "\\#{following}")
    elsif character == "/"
      closed = true
      break
    else
      pattern << character
    end
  end
  raise Canopus::Error, "unclosed snippet transform pattern" unless closed
  replacement = parse_format(scanner, ["/"], depth)
  raise Canopus::Error, "unclosed snippet transform replacement" unless scanner.getch == "/"
  flags = scanner.scan(/[A-Za-z]*/)
  raise Canopus::Error, "unclosed snippet transform" unless scanner.getch == "}"
  new(pattern, replacement, flags)
end

.parse_format(scanner, endings, depth, budget = [0]) ⇒ format_nodes

Parameters:

  • scanner (StringScanner)
  • endings (Array[String])
  • depth (Integer)
  • budget (Array[Integer]) (defaults to: [0])

Returns:

  • (format_nodes)

Raises:



37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
# File 'lib/canopus/snippet/transform.rb', line 37

def self.parse_format(scanner, endings, depth, budget = [0])
  raise Canopus::Error, "snippet format nesting exceeds #{Canopus::Snippet::MAX_DEPTH}" if depth > Canopus::Snippet::MAX_DEPTH
  nodes, literal = [], +""
  until scanner.eos? || endings.include?(scanner.peek(1))
    character = scanner.getch
    if character == "\\" && scanner.peek(1).match?(/[\\\/$}:]/)
      literal << scanner.getch
    elsif character == "$"
      braced = !!scanner.scan(/\{/)
      number = scanner.scan(/\d+/)
      unless number
        raise Canopus::Error, "invalid snippet transform capture" if braced
        literal << "$"
        next
      end
      raise Canopus::Error, "snippet capture index is too large" if number.length > 7 || number.to_i > 1_000_000
      nodes << literal.freeze unless literal.empty?
      literal = +""
      operation, positive, negative = :capture, [], []
      budget[0] += 1
      raise Canopus::Error, "too many snippet format captures" if budget[0] > Canopus::Snippet::MAX_NODES
      if braced && scanner.scan(/:/)
        operation_start = scanner.pos
        case scanner.getch
        when "/"
          operation = scanner.scan(/[a-z]+/)
          raise Canopus::Error, "unsupported snippet case conversion" unless CASES.include?(operation)
        when "+"
          operation = :if
          positive = parse_format(scanner, ["}"], depth + 1, budget)
        when "?"
          operation = :conditional
          positive = parse_format(scanner, [":", "}"], depth + 1, budget)
          raise Canopus::Error, "snippet conditional requires ':'" unless scanner.getch == ":"
          negative = parse_format(scanner, ["}"], depth + 1, budget)
        when "-"
          operation = :else
          negative = parse_format(scanner, ["}"], depth + 1, budget)
        else
          scanner.pos = operation_start
          operation = :else
          negative = parse_format(scanner, ["}"], depth + 1, budget)
        end
      end
      raise Canopus::Error, "unclosed snippet transform format" if braced && scanner.getch != "}"
      nodes << Format.new(number.to_i, operation, positive.freeze, negative.freeze)
    else
      literal << character
    end
  end
  nodes << literal.freeze unless literal.empty?
  nodes.freeze
end

Instance Method Details

#apply(value) ⇒ String

Parameters:

  • value (String)

Returns:

  • (String)


143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
# File 'lib/canopus/snippet/transform.rb', line 143

def apply(value)
  raise Canopus::Error, "snippet transform input exceeds output limit" if value.bytesize > Canopus::Snippet::MAX_OUTPUT
  astral = value.match?(/[\u{10000}-\u{10ffff}]/) || @pattern.match?(/[\u{10000}-\u{10ffff}]/)
  raise Canopus::Error, "astral Unicode snippet transforms require the 'u' flag" if astral && !@flags.include?("u")
  raise Canopus::Error, "\\B with astral Unicode is not portable between JavaScript and Ruby" if astral && @nonword_boundary
  if @flags.include?("i") && [value, @pattern].any? { |source| source.each_char.any? { |character| !character.ascii_only? && ((folded = character.downcase(:fold)).length != 1 || folded.ascii_only?) } }
    raise Canopus::Error, "Unicode case folding into ASCII or multiple characters is not portable"
  end
  output, cursor, count, matched = +"", 0, 0, false
  deadline = Process.clock_gettime(Process::CLOCK_MONOTONIC) + TIMEOUT
  Canopus.with_regexp_timeout(@regexp) do
    value.to_enum(:scan, @regexp).each do
      match = Regexp.last_match
      matched = true
      count += 1
      raise Canopus::Error, "snippet transform exceeded execution limit" if count > Canopus::Snippet::MAX_NODES || Process.clock_gettime(Process::CLOCK_MONOTONIC) > deadline
      first, last = match.byteoffset(0)
      append(output, value.byteslice(cursor...first))
      append(output, format(@replacement, match))
      cursor = last
      break unless @flags.include?("g")
    end
  end
  return format(@replacement, nil) if !matched && fallback?(@replacement)
  append(output, value.byteslice(cursor..))
  output
rescue Regexp::TimeoutError
  raise Canopus::Error, "snippet regular expression timed out"
end