Class: Jekyll::AgentAudit::Extractor
- Inherits:
-
Object
- Object
- Jekyll::AgentAudit::Extractor
- Defined in:
- lib/jekyll/agent_audit/extractor.rb
Constant Summary collapse
- MAX_NAME_BYTES =
512- MAX_EXCERPT_BYTES =
4_096
Instance Method Summary collapse
- #extract(entry) ⇒ Object
-
#initialize(configuration) ⇒ Extractor
constructor
A new instance of Extractor.
Constructor Details
#initialize(configuration) ⇒ Extractor
Returns a new instance of Extractor.
14 15 16 |
# File 'lib/jekyll/agent_audit/extractor.rb', line 14 def initialize(configuration) @configuration = configuration end |
Instance Method Details
#extract(entry) ⇒ Object
18 19 20 21 22 23 24 25 26 27 28 29 30 31 32 33 34 35 36 37 38 39 40 41 42 43 44 45 46 47 48 49 50 51 52 53 54 55 56 57 58 59 60 61 62 63 64 65 66 67 68 69 70 71 72 73 74 75 |
# File 'lib/jekyll/agent_audit/extractor.rb', line 18 def extract(entry) html = entry[:html] || read_artifact(entry[:absolute_path], entry[:destination_root]) raise InputError, "HTML exceeds max_html_bytes" if html.bytesize > limit(:max_html_bytes, 5 * 1024 * 1024) doc = Nokogiri::HTML5.parse(html, nil, "UTF-8") content = content_node(doc, entry) id_nodes = doc.css("[id]") id_index = id_nodes.each_with_object({}) { |node, index| index[node["id"].to_s] ||= node unless node["id"].to_s.empty? } result = entry.dup.merge( title: compact_text(node_text(doc.at_css("title"))), ids: Set.new(id_index.keys), duplicate_ids: duplicate_ids(id_nodes), duplicate_id_locations: duplicate_id_locations(id_nodes), links: [], canonicals: [], headings: [], base_href: doc.at_css("head base[href]")&.[]("href"), content_basis: content[:basis], content_confident: content[:confident], redirect: redirect?(doc), jsonld_scripts: [], metadata_nodes: Hash.new { |h, k| h[k] = [] }, diagnostics: content[:diagnostics] ) doc.css("a[href]").each do |node| result[:links] << occurrence(node, accessible_name(node, id_index), :navigation, content[:basis]) end doc.css("[cite]").each do |node| href = node["cite"].to_s.strip next if href.empty? result[:links] << {href: href, line: node.line, name: compact_text(node_text(node)), name_supported: true, hidden: hidden?(node), kind: :citation, region: :other} end doc.css("head link[href]").each do |node| rels = node["rel"].to_s.split.map(&:downcase) next if rels.include?("canonical") kind = rels.include?("citation") ? :citation : (rels.include?("alternate") ? :alternate : nil) next unless kind result[:links] << {href: node["href"].to_s, line: node.line, name: compact_text(node["title"]), name_supported: true, hidden: false, kind: kind, region: :other} end content[:node]&.css("h1,h2,h3,h4,h5,h6,[role='heading']")&.each do |node| next if excluded?(node) || decorative?(node) level = node.name =~ /h([1-6])/ ? Regexp.last_match(1).to_i : node["aria-level"].to_i next unless (1..6).cover?(level) name = accessible_name(node, id_index) result[:headings] << {level: level, name: name[:name], name_supported: name[:supported], hidden: hidden?(node), line: node.line} end doc.css("head link[rel]").each do |node| rels = node["rel"].to_s.split.map(&:downcase) result[:canonicals] << {href: node["href"], line: node.line} if rels.include?("canonical") end doc.css("script[type]").each do |node| next unless node["type"].to_s.downcase == "application/ld+json" raise InputError, "JSON-LD exceeds max_jsonld_bytes" if node.text.bytesize > limit(:max_jsonld_bytes, 1024 * 1024) result[:jsonld_scripts] << {text: node.text, line: node.line} end result[:content_text] = compact_text(content_text(content[:node]), MAX_EXCERPT_BYTES) (doc, result) result rescue Errno::ENOENT, Errno::EACCES => e raise InputError, e. rescue Nokogiri::XML::SyntaxError, ArgumentError => e raise InputError, "HTML could not be parsed: #{e.}" end |