Class: Retriever::Fetch

Inherits:
Object
  • Object
show all
Defined in:
lib/retriever/fetch.rb

Direct Known Subclasses

FetchFiles, FetchSEO, FetchSitemap

Instance Attribute Summary collapse

Instance Method Summary collapse

Constructor Details

#initialize(url, options) ⇒ Fetch

given target URL and RR options, creates a fetch object. There is no direct output, this is a parent class that the other fetch classes build off of.



13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
# File 'lib/retriever/fetch.rb', line 13

def initialize(url,options) #given target URL and RR options, creates a fetch object. There is no direct output, this is a parent class that the other fetch classes build off of.
	@connection_tally = {
		:success => 0,
		:error => 0,
		:error_client => 0,
		:error_server => 0
	}
	#OPTIONS
	@prgrss = options[:progress] ? options[:progress] : false
	@maxPages = options[:maxpages] ? options[:maxpages].to_i : 100
	@v= options[:verbose] ? true : false
	@output=options[:filename] ? options[:filename] : false
	@fh = options[:fileharvest] ? options[:fileharvest] : false
	@file_ext = @fh.to_s
	@s = options[:sitemap] ? options[:sitemap] : false
	@seo = options[:seo] ? true : false
	@autodown = options[:autodown] ? true : false
	#
	if @fh
		tempExtStr = "."+@file_ext+'\z'
		@file_re = Regexp.new(tempExtStr).freeze
	else
		errlog("Cannot AUTODOWNLOAD when not in FILEHARVEST MODE") if @autodown #when FH is not true, and autodown is true
	end
	if @prgrss
		errlog("CANNOT RUN VERBOSE & PROGRESSBAR AT SAME TIME, CHOOSE ONE, -v or -p") if @v #verbose & progressbar conflict
		prgressVars = {
			:title => "Pages Crawled",
			:starting_at => 1,
			:total => @maxPages,
			:format => '%a |%b>%i| %c/%C %t',
		}
		@progressbar = ProgressBar.create(prgressVars)
	end
	@t = Retriever::Target.new(url,@file_re)
	@already_crawled = BloomFilter::Native.new(:size => 1000000, :hashes => 5, :seed => 1, :bucket => 8, :raise => false)
	@already_crawled.insert(@t.target)
	if (@fh && !@output)
		@output = "rr-#{@t.host.split('.')[1]}"
	end
	fail "bad page source on target -- try HTTPS?" if !@t.source
end

Instance Attribute Details

#maxPages ⇒ Object (readonly)

Returns the value of attribute maxPages.



11
12
13
# File 'lib/retriever/fetch.rb', line 11

def maxPages
  @maxPages
end

#t ⇒ Object (readonly)

Returns the value of attribute t.



11
12
13
# File 'lib/retriever/fetch.rb', line 11

def t
  @t
end

Instance Method Details

#async_crawl_and_collect ⇒ Object

iterates over the excisting @linkStack, running asyncGetWave on it until we reach the @maxPages value.



107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
# File 'lib/retriever/fetch.rb', line 107

def async_crawl_and_collect() #iterates over the excisting @linkStack, running asyncGetWave on it until we reach the @maxPages value.
	while (@already_crawled.size < @maxPages)
		if @linkStack.empty?
			if @prgrss
				@progressbar.log("Can't find any more links. Site might be completely mapped.")
			else
				lg("Can't find any more links. Site might be completely mapped.")
			end
			break;
		end
		new_links_arr = self.asyncGetWave()
		next if (new_links_arr.nil? || new_links_arr.empty?)
		new_link_arr = new_links_arr-@linkStack #set operations to see are these in our previous visited pages arr?
		@linkStack.concat(new_links_arr).uniq!
		@data.concat(new_links_arr) if @s
	end
	@progressbar.finish if @prgrss #if we are done, let's make sure progress bar says we are done
end

#asyncGetWave ⇒ Object

send a new wave of GET requests, using current @linkStack



155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
# File 'lib/retriever/fetch.rb', line 155

def asyncGetWave() #send a new wave of GET requests, using current @linkStack
	new_stuff = []
	EM.synchrony do
		lenny = 0
	    concurrency = 10
	    EM::Synchrony::FiberIterator.new(@linkStack, concurrency).each do |url|
	    	next if (@already_crawled.size >= @maxPages)
	    	if @already_crawled.include?(url)
	    		@linkStack.delete(url)
	    		next
	    	end
	    	resp = EventMachine::HttpRequest.new(url).get
	    	next if !good_response?(resp,url)
	    	new_page = Retriever::Page.new(resp.response,@t)
	    	lg("Page Fetched: #{url}")
	    	@already_crawled.insert(url)
			if @prgrss
				@progressbar.increment if @already_crawled.size < @maxPages
			end
			if @seo
				seos = [url]
				seos.concat(new_page.parseSEO)
				@data.push(seos)
				lg("--page SEO scraped")
			end
			if new_page.links
				lg("--#{new_page.links.size} links found")
				internal_links_arr = new_page.parseInternalVisitable
				new_stuff.push(internal_links_arr)
				if @fh
					filez = new_page.parseFiles
					@data.concat(filez) if !filez.empty?
					lg("--#{filez.size} files found")
				end
			end
	    end
	    new_stuff = new_stuff.flatten # all completed requests
	    EventMachine.stop
	end
	new_stuff.uniq!
end

#dump ⇒ Object

prints current data collection to STDOUT, meant for CLI use.



61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
# File 'lib/retriever/fetch.rb', line 61

def dump #prints current data collection to STDOUT, meant for CLI use.
	puts "###############################"
	if @v
		puts "Connection Tally:"
		puts @connection_tally.to_s
		puts "###############################"
	end
	if @s
		puts "#{@t.target} Sitemap"
		puts "Page Count: #{@data.size}"
	elsif @fh
		puts "Target URL: #{@t.target}"
		puts "Filetype: #{@file_ext}"
		puts "File Count: #{@data.size}"
	elsif @seo
		puts "#{@t.target} SEO Metrics"
		puts "Page Count: #{@data.size}"
	else
		fail "ERROR - Cannot dump - Mode Not Found"
	end
	puts "###############################"
	@data.each do |line|
		puts line
	end
	puts "###############################"
	puts
end

#errlog(msg) ⇒ Object



55
56
57
# File 'lib/retriever/fetch.rb', line 55

def errlog(msg)
	raise "ERROR: #{msg}"
end

#good_response?(resp, url) ⇒ Boolean

returns true is resp is ok to continue process, false is we need to 'next' it

Returns:

  • (Boolean)


125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
# File 'lib/retriever/fetch.rb', line 125

def good_response?(resp, url) #returns true is resp is ok to continue process, false is we need to 'next' it
	return false if !resp
	if resp.response_header.redirection? #we got redirected
		loc = resp.response_header.location
		lg("#{url} Redirected to #{loc}")
		if t.host_re =~ loc #if being redirected to same host, let's add to linkstack
	    	@linkStack.push(loc) if !@already_crawled.include?(loc) #but only if we haven't already crawled it
	    	lg("--Added to linkStack for later")
	    	return false
	    end
	    lg("Redirection outside of target host. No - go. #{loc}")
	    return false
	end
	if (!resp.response_header.successful?) #if webpage is not text/html, let's not continue and lets also make sure we dont re-queue it
		lg("UNSUCCESSFUL CONNECTION -- #{url}")
		@connection_tally[:error] += 1
		@connection_tally[:error_server] += 1 if resp.response_header.server_error?
		@connection_tally[:error_client] += 1 if resp.response_header.client_error?
		return false
	end
	if (!(resp.response_header['CONTENT_TYPE'].include?("text/html"))) #if webpage is not text/html, let's not continue and lets also make sure we dont re-queue it
		@already_crawled.insert(url)
		@linkStack.delete(url)
		lg("Page Not text/html -- #{url}")
		return false
	end
	@connection_tally[:success] += 1
	return true
end

#lg(msg) ⇒ Object



58
59
60
# File 'lib/retriever/fetch.rb', line 58

def lg(msg)
	puts "### #{msg}" if @v
end

#write ⇒ Object

writes current data collection out to CSV in current directory



88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
# File 'lib/retriever/fetch.rb', line 88

def write #writes current data collection out to CSV in current directory
	if @output
		i = 0
		CSV.open("#{@output}.csv", "w") do |csv|
			if ((i == 0) && @seo)
				csv << ['URL','Page Title','Meta Description','H1','H2']
				i +=1
			end
			@data.each do |entry|
				csv << entry
			end
		end
		puts "###############################"
		puts "File Created: #{@output}.csv"
		puts "Object Count: #{@data.size}"
		puts "###############################"
		puts
	end
end