Last active 1 month ago

story_parser.rb Raw
1# Parse stories from other websites and uploaded files, looking for metadata to harvest
2# and put into the archive.
3#
4class StoryParser
5 require 'timeout'
6 require 'nokogiri'
7 require 'mechanize'
8 require 'open-uri'
9 include HtmlCleaner
10
11 OPTIONAL_META = {notes: 'Note',
12 freeform_string: 'Tag',
13 fandom_string: 'Fandom',
14 rating_string: 'Rating',
15 archive_warning_string: 'Warning',
16 relationship_string: 'Relationship|Pairing',
17 character_string: 'Character' }.freeze
18 REQUIRED_META = { title: 'Title',
19 summary: 'Summary',
20 revised_at: 'Date|Posted|Posted on|Posted at',
21 chapter_title: 'Chapter Title' }.freeze
22
23 # Use this for raising custom error messages
24 # (so that we can distinguish them from unexpected exceptions due to
25 # faulty code)
26 class Error < StandardError
27 end
28
29 # These attributes need to be moved from the work to the chapter
30 # format: {work_attribute_name: :chapter_attribute_name} (can be the same)
31 CHAPTER_ATTRIBUTES_ONLY = {}
32
33 # These attributes need to be copied from the work to the chapter
34 CHAPTER_ATTRIBUTES_ALSO = { revised_at: :published_at }.freeze
35
36 ### NOTE ON KNOWN SOURCES
37 # These lists will stop with the first one it matches, so put more-specific matches
38 # towards the front of the list.
39
40 # places for which we have a custom parse_story_from_[source] method
41 # for getting information out of the downloaded text
42 KNOWN_STORY_PARSERS = %w[ao3 deviantart dw lj].freeze
43
44 # places for which we have a custom parse_author_from_[source] method
45 # which returns an external_author object including an email address
46 KNOWN_AUTHOR_PARSERS = %w[lj].freeze
47
48 # places for which we have a download_story_from_[source]
49 # used to customize the downloading process
50 KNOWN_STORY_LOCATIONS = %w[lj].freeze
51
52 # places for which we have a download_chaptered_from
53 # to get a set of chapters all together
54 CHAPTERED_STORY_LOCATIONS = %w[ffnet thearchive_net efiction quotev].freeze
55
56 # regular expressions to match against the URLS
57 SOURCE_AO3 = '(archiveofourown\.org|ao3\.org|superlove\.sayitditto\.net|sunset\.femslash\.club)'.freeze
58 SOURCE_LJ = '((live|dead|insane)journal\.com)|journalfen(\.net|\.com)|dreamwidth\.org'.freeze
59 SOURCE_DW = 'dreamwidth\.org'.freeze
60 SOURCE_FFNET = '(^|[^A-Za-z0-9-])fanfiction\.net'.freeze
61 SOURCE_DEVIANTART = 'deviantart\.com'.freeze
62 SOURCE_THEARCHIVE_NET = 'the\-archive\.net'.freeze
63 SOURCE_EFICTION = 'viewstory\.php'.freeze
64 SOURCE_QUOTEV = 'quotev\.com'.freeze
65
66 # time out if we can't download fast enough
67 STORY_DOWNLOAD_TIMEOUT = 60
68 MAX_CHAPTER_COUNT = 200
69
70 # To check for duplicate chapters, take a slice this long out of the story
71 # (in characters)
72 DUPLICATE_CHAPTER_LENGTH = 10_000
73
74
75 # Import many stories
76 def import_many(urls, options = {})
77 # Try to get the works
78 works = []
79 failed_urls = []
80 errors = []
81 @options = options
82 urls.each do |url|
83 begin
84 response = download_and_parse_work(url, options)
85 work = response[:work]
86 if response[:status] == :created
87 if work && work.save
88 work.chapters.each(&:save)
89 works << work
90 else
91 failed_urls << url
92 errors << work.errors.values.join(", ")
93 work.delete if work
94 end
95 elsif response[:status] == :already_imported
96 raise StoryParser::Error, response[:message]
97 end
98 rescue Timeout::Error
99 failed_urls << url
100 errors << "Import has timed out. This may be due to connectivity problems with the source site. Please try again in a few minutes, or check Known Issues to see if there are import problems with this site."
101 work.delete if work
102 rescue Error => exception
103 failed_urls << url
104 errors << "We couldn't successfully import that work, sorry: #{exception.message}"
105 work.delete if work
106 end
107 end
108 [works, failed_urls, errors]
109 end
110
111 # Downloads a story and passes it on to the parser.
112 # If the URL of the story is from a site for which we have special rules
113 # (eg, downloading from a livejournal clone, you want to use ?format=light
114 # to get a nice and consistent post format), it will pre-process the url
115 # according to the rules for that site.
116 def download_and_parse_work(location, options = {})
117 status = :created
118 message = ""
119 work = Work.find_by_url(location)
120 if work.nil?
121 @options = options
122 source = get_source_if_known(CHAPTERED_STORY_LOCATIONS, location)
123 if source.nil?
124 story = download_text(location)
125 work = parse_story(story, location, options)
126 else
127 work = download_and_parse_chaptered_story(source, location, options)
128 end
129 else
130 status = :already_imported
131 message = "A work has already been imported from #{location}."
132 end
133 {
134 status: status,
135 message: message,
136 work: work
137 }
138 end
139
140 # Given an array of urls for chapters of a single story,
141 # download them all and combine into a single work
142 def import_chapters_into_story(locations, options = {})
143 status = :created
144 work = Work.find_by_url(locations.first)
145 if work.nil?
146 chapter_contents = []
147 @options = options
148 locations.each do |location|
149 chapter_contents << download_text(location)
150 end
151 work = parse_chapters_into_story(locations.first, chapter_contents, options)
152 message = "Successfully created work \"" + work.title + "\"."
153 else
154 status = :already_imported
155 message = "A work has already been imported from #{locations.first}."
156 end
157 {
158 status: status,
159 message: message,
160 work: work
161 }
162 end
163
164
165 ### OLD PARSING METHODS
166
167 # Import many stories
168 def import_from_urls(urls, options = {})
169 # Try to get the works
170 works = []
171 failed_urls = []
172 errors = []
173 @options = options
174 urls.each do |url|
175 begin
176 work = download_and_parse_story(url, options)
177 if work && work.save
178 work.chapters.each(&:save)
179 works << work
180 else
181 failed_urls << url
182 errors << work.errors.values.join(", ")
183 work.delete if work
184 end
185 rescue Timeout::Error
186 failed_urls << url
187 errors << "Import has timed out. This may be due to connectivity problems with the source site. Please try again in a few minutes, or check Known Issues to see if there are import problems with this site."
188 work.delete if work
189 rescue Error => exception
190 failed_urls << url
191 errors << "We couldn't successfully import that work, sorry: #{exception.message}"
192 work.delete if work
193 end
194 end
195 [works, failed_urls, errors]
196 end
197
198 # Downloads a story and passes it on to the parser.
199 # If the URL of the story is from a site for which we have special rules
200 # (eg, downloading from a livejournal clone, you want to use ?format=light
201 # to get a nice and consistent post format), it will pre-process the url
202 # according to the rules for that site.
203 def download_and_parse_story(location, options = {})
204 check_for_previous_import(location)
205 @options = options
206 source = get_source_if_known(CHAPTERED_STORY_LOCATIONS, location)
207 if source.nil?
208 story = download_text(location)
209 work = parse_story(story, location, options)
210 else
211 work = download_and_parse_chaptered_story(source, location, options)
212 end
213 work
214 end
215
216 # Given an array of urls for chapters of a single story,
217 # download them all and combine into a single work
218 def download_and_parse_chapters_into_story(locations, options = {})
219 check_for_previous_import(locations.first)
220 chapter_contents = []
221 @options = options
222 locations.each do |location|
223 chapter_contents << download_text(location)
224 end
225 parse_chapters_into_story(locations.first, chapter_contents, options)
226 end
227
228 ### PARSING METHODS
229
230 # Parses the text of a story, optionally from a given location.
231 def parse_story(story, location, options = {})
232 work_params = parse_common(story, location, options[:encoding], options[:detect_tags])
233
234 # move any attributes from work to chapter if necessary
235 set_work_attributes(Work.new(work_params), location, options)
236 end
237
238 # parses and adds a new chapter to the end of the work
239 def parse_chapter_of_work(work, chapter_content, location, options = {})
240 tmp_work_params = parse_common(chapter_content, location, options[:encoding], options[:detect_tags])
241 chapter = get_chapter_from_work_params(tmp_work_params)
242 work.chapters << set_chapter_attributes(work, chapter)
243 work
244 end
245
246 def parse_chapters_into_story(location, chapter_contents, options = {})
247 work = nil
248 chapter_contents.each do |content|
249 work_params = parse_common(content, location, options[:encoding], options[:detect_tags])
250 if work.nil?
251 # create the new work
252 work = Work.new(work_params)
253 else
254 new_chapter = get_chapter_from_work_params(work_params)
255 work.chapters << set_chapter_attributes(work, new_chapter)
256 end
257 end
258 set_work_attributes(work, location, options)
259 end
260
261 # Everything below here is protected and should not be touched by outside
262 # code -- please use the above functions to parse external works.
263
264 protected
265
266 # tries to create an external author for a given url
267 def parse_author(location, ext_author_name, ext_author_email)
268 if location.present? && ext_author_name.blank? && ext_author_email.blank?
269 source = get_source_if_known(KNOWN_AUTHOR_PARSERS, location)
270 if source.nil?
271 raise Error, "No external author name or email specified"
272 else
273 send("parse_author_from_#{source.downcase}", location)
274 end
275 else
276 parse_author_common(ext_author_email, ext_author_name)
277 end
278 end
279
280 # download an entire story from an archive type where we know how to parse multi-chaptered works
281 # this should only be called from download_and_parse_story
282 def download_and_parse_chaptered_story(source, location, options = {})
283 chapter_contents = send("download_chaptered_from_#{source.downcase}", location)
284 parse_chapters_into_story(location, chapter_contents, options)
285 end
286
287 # our custom url finder checks for previously imported URL in almost any format it may have been presented
288 def check_for_previous_import(location)
289 if Work.find_by_url(location).present?
290 raise Error, "A work has already been imported from #{location}."
291 end
292 end
293
294 def set_chapter_attributes(work, chapter)
295 chapter.position = work.chapters.length + 1
296 chapter.posted = true
297 chapter
298 end
299
300 def set_work_attributes(work, location = "", options = {})
301 raise Error, "Work could not be downloaded" if work.nil?
302
303 @options = options
304 work.imported_from_url = location # @todo remove this as part of AO3-6979
305 work.imported_url = ImportedUrl.new(original: work.imported_from_url)
306
307 work.ip_address = options[:ip_address]
308 work.expected_number_of_chapters = work.chapters.length
309 work.revised_at = work.chapters.last.published_at
310 if work.revised_at && work.revised_at.to_date < Date.current
311 work.backdate = true
312 end
313
314 # set authors for the works
315 pseuds = []
316 pseuds << User.current_user.default_pseud unless options[:do_not_set_current_author] || User.current_user.nil?
317 pseuds << options[:archivist].default_pseud if options[:archivist]
318 pseuds << options[:pseuds] if options[:pseuds]
319 pseuds = pseuds.flatten.compact.uniq
320 raise Error, "A work must have at least one author specified" if pseuds.empty?
321 pseuds.each do |pseud|
322 work.creatorships.build(pseud: pseud, enable_notifications: true)
323 work.chapters.each do |chapter|
324 chapter.creatorships.build(pseud: pseud)
325 end
326 end
327
328 # handle importing works for others
329 # build an external creatorship for each author
330 if options[:importing_for_others]
331 external_author_names = options[:external_author_names] || parse_author(location, options[:external_author_name], options[:external_author_email])
332 # convert to an array if not already one
333 external_author_names = [external_author_names] if external_author_names.is_a?(ExternalAuthorName)
334 if options[:external_coauthor_name].present?
335 external_author_names << parse_author(location, options[:external_coauthor_name], options[:external_coauthor_email])
336 end
337 external_author_names.each do |external_author_name|
338 next if !external_author_name || external_author_name.external_author.blank?
339 if external_author_name.external_author.do_not_import
340 # we're not allowed to import works from this address
341 raise Error, "Author #{external_author_name.name} at #{external_author_name.external_author.email} does not allow importing their work to this archive."
342 end
343 work.external_creatorships.build(external_author_name: external_author_name, archivist: (options[:archivist] || User.current_user))
344 end
345 end
346
347 # lock to registered users if specified or importing for others
348 work.restricted = options[:restricted] || options[:importing_for_others] || false
349
350 # set comment permissions
351 work.comment_permissions = options[:comment_permissions] || "enable_all"
352 work.moderated_commenting_enabled = options[:moderated_commenting_enabled] || false
353
354 # set default values for required tags
355 work.fandom_string = meta_or_default(work.fandom_string, options[:fandom], ArchiveConfig.FANDOM_NO_TAG_NAME)
356 work.rating_string = meta_or_default(work.rating_string, options[:rating], ArchiveConfig.RATING_DEFAULT_TAG_NAME)
357 work.archive_warning_strings = meta_or_default(work.archive_warning_strings, options[:archive_warning], ArchiveConfig.WARNING_DEFAULT_TAG_NAME)
358 work.category_string = meta_or_default(work.category_string, options[:category], [])
359 work.character_string = meta_or_default(work.character_string, options[:character], [])
360 work.relationship_string = meta_or_default(work.relationship_string, options[:relationship], [])
361 work.freeform_string = meta_or_default(work.freeform_string, options[:freeform], [])
362
363 # set default value for title
364 work.title = meta_or_default(work.title, options[:title], "Untitled Imported Work")
365 work.summary = meta_or_default(work.summary, options[:summary], '')
366 work.notes = meta_or_default(work.notes, options[:notes], '')
367
368 # set collection name if present
369 work.collection_names = get_collection_names(options[:collection_names]) if options[:collection_names].present?
370
371 # set default language (English)
372 work.language_id = options[:language_id] || Language.default.id
373
374 work.posted = true if options[:post_without_preview]
375 work.chapters.each do |chapter|
376 if chapter.content.length > ArchiveConfig.CONTENT_MAX
377 # TODO: eventually: insert a new chapter
378 chapter.content.truncate(ArchiveConfig.CONTENT_MAX, omission: "<strong>WARNING: import truncated automatically because chapter was too long! Please add a new chapter for remaining content.</strong>", separator: "</p>")
379 elsif chapter.content.empty?
380 raise Error, "Chapter #{chapter.position} of \"#{work.title}\" is blank."
381 end
382
383 chapter.posted = true # do not save - causes the chapters to exist even if work doesn't get created!
384 end
385 work
386 end
387
388 def parse_author_from_lj(location)
389 return if location !~ %r{^(?:http:\/\/)?(?<lj_name>[^.]*).(?<site_name>livejournal\.com|dreamwidth\.org|insanejournal\.com|journalfen.net)}
390 email = ""
391 lj_name = Regexp.last_match[:lj_name]
392 site_name = Regexp.last_match[:site_name]
393 if lj_name == "community"
394 # whups
395 post_text = download_text(location)
396 doc = Nokogiri.parse(post_text)
397 lj_name = doc.xpath("/html/body/div[2]/div/div/div/table/tbody/tr/td[2]/span/a[2]/b").content
398 end
399 profile_url = "http://#{lj_name}.#{site_name}/profile"
400 lj_profile = download_text(profile_url)
401 doc = Nokogiri.parse(lj_profile)
402 contact = doc.css('div.contact').inner_html
403 if contact.present?
404 contact.gsub! '<p class="section_body_title">Contact:</p>', ""
405 contact.gsub! /<\/?(span|i)>/, ""
406 contact.delete! "\n"
407 contact.gsub! "<br/>", ""
408 if contact =~ /(.*@.*\..*)/
409 email = Regexp.last_match[1]
410 end
411 end
412 email = "#{lj_name}@#{site_name}" if email.blank?
413 parse_author_common(email, lj_name)
414 end
415
416 def parse_author_from_unknown(_location)
417 # for now, nothing
418 nil
419 end
420
421 def parse_author_common(email, name)
422 errors = []
423
424 errors << "No author name specified" if name.blank?
425
426 if email.present?
427 external_author = ExternalAuthor.find_or_create_by(email: email)
428 errors += external_author.errors.full_messages
429 else
430 errors << "No author email specified"
431 end
432
433 raise Error, errors.join("\n") if errors.present?
434
435 # convert to ASCII and strip out invalid characters (everything except alphanumeric characters, _, @ and -)
436 redacted_name = name.to_ascii.gsub(/[^\w[ \-@.]]/u, "")
437 if redacted_name.present?
438 external_author.names.find_or_create_by(name: redacted_name)
439 else
440 external_author.default_name
441 end
442 end
443
444 def get_chapter_from_work_params(work_params)
445 @chapter = Chapter.new(work_params[:chapter_attributes])
446 # don't override specific chapter params (eg title) with work params
447 chapter_params = work_params.delete_if do |name, _param|
448 !@chapter.attribute_names.include?(name.to_s) || !@chapter.send(name.to_s).blank?
449 end
450 @chapter.update(chapter_params)
451 @chapter
452 end
453
454 def download_text(location)
455 source = get_source_if_known(KNOWN_STORY_LOCATIONS, location)
456 if source.nil?
457 download_with_timeout(location)
458 else
459 send("download_from_#{source.downcase}", location)
460 end
461 end
462
463 # canonicalize the url for downloading from lj or clones
464 def download_from_lj(location)
465 url = location
466 url.gsub!(/\#(.*)$/, "") # strip off any anchor information
467 url.gsub!(/\?(.*)$/, "") # strip off any existing params at the end
468 url.gsub!('_', '-') # convert underscores in usernames to hyphens
469 url += "?format=light" # go to light format
470 text = download_with_timeout(url)
471
472 if text.match(/adult_check/)
473 Timeout::timeout(STORY_DOWNLOAD_TIMEOUT) {
474 begin
475 agent = Mechanize.new
476 url.include?("dreamwidth") ? form = agent.get(url).forms.first : form = agent.get(url).forms.third
477 page = agent.submit(form, form.buttons.first) # submits the adult concepts form
478 text = page.body.force_encoding(agent.page.encoding)
479 rescue
480 text = ""
481 end
482 }
483 end
484 text
485 end
486
487 # grab all the chapters of the story from ff.net
488 def download_chaptered_from_ffnet(_location)
489 raise Error, "Sorry, Fanfiction.net does not allow imports from their site."
490 end
491
492 def download_chaptered_from_quotev(_location)
493 raise Error, "Sorry, Quotev.com does not allow imports from their site."
494 end
495
496 # this is an efiction archive but it doesn't handle chapters normally
497 # best way to handle is to get the full story printable version
498 # We have to make it a download-chaptered because otherwise it gets sent to the
499 # generic efiction version since chaptered sources are checked first
500 def download_chaptered_from_thearchive_net(location)
501 if location.match(/^(.*)\/.*viewstory\.php.*[^p]sid=(\d+)($|&)/i)
502 location = "#{$1}/viewstory.php?action=printable&psid=#{$2}"
503 end
504 text = download_with_timeout(location)
505 text.sub!('</style>', '</style></head>') unless text.match('</head>')
506 [text]
507 end
508
509 # grab all the chapters of a story from an efiction-based site
510 def download_chaptered_from_efiction(location)
511 chapter_contents = []
512 if location.match(/^(?<site>.*)\/.*viewstory\.php.*sid=(?<storyid>\d+)($|&)/i)
513 site = Regexp.last_match[:site]
514 storyid = Regexp.last_match[:storyid]
515 chapnum = 1
516 last_body = ""
517 Timeout::timeout(STORY_DOWNLOAD_TIMEOUT) do
518 loop do
519 url = "#{site}/viewstory.php?action=printable&sid=#{storyid}&chapter=#{chapnum}"
520 body = download_with_timeout(url)
521 # get a section to check that this isn't a duplicate of previous chapter
522 body_to_check = body.slice(10, DUPLICATE_CHAPTER_LENGTH)
523 if body.nil? || body_to_check == last_body || chapnum > MAX_CHAPTER_COUNT || body.match(/<div class='chaptertitle'> by <\/div>/) || body.match(/Access denied./) || body.match(/Chapter : /)
524 break
525 end
526 # save the value to check for duplicate chapter
527 last_body = body_to_check
528
529 # clean up the broken head in many efiction printable sites
530 body.sub!('</style>', '</style></head>') unless body.match('</head>')
531 chapter_contents << body
532 chapnum += 1
533 end
534 end
535 end
536 chapter_contents
537 end
538
539
540 # This is the heavy lifter, invoked by all the story and chapter parsers.
541 # It takes a single string containing the raw contents of a story, parses it with
542 # Nokogiri into the @doc object, and then and calls a subparser.
543 #
544 # If the story source can be identified as one of the sources we know how to parse in some custom/
545 # special way, parse_common calls the customized parse_story_from_[source] method.
546 # Otherwise, it falls back to parse_story_from_unknown.
547 #
548 # This produces a hash equivalent to the params hash that is normally created by the standard work
549 # upload form.
550 #
551 # parse_common then calls sanitize_params (which would also be called on the standard work upload
552 # form results) and returns the final sanitized hash.
553 #
554 def parse_common(story, location = nil, encoding = nil, detect_tags = true)
555 work_params = { title: "Untitled Imported Work", chapter_attributes: { content: "" } }
556
557 # Encode as HTML - the dummy "foo" tag will be stripped out by the sanitizer but forces Nokogiri to
558 # preserve line breaks in plain text documents
559 # Rescue all errors as Nokogiri complains about things the sanitizer will fix later
560 story.prepend("<foo></foo>")
561 @doc =
562 begin
563 Nokogiri::HTML5.parse(story, encoding: encoding)
564 rescue StandardError
565 Nokogiri::HTML5.parse("")
566 end
567
568 # Try to convert all relative links to absolute
569 base = @doc.at_css("base") ? @doc.css("base")[0]["href"] : location.split("?").first
570 if base.present?
571 @doc.css("a").each do |link|
572 next if link["href"].blank? || link["href"].start_with?("#")
573 begin
574 query = link["href"].match(/(\?.*)$/) ? $1 : ""
575 link["href"] = URI.join(base, link["href"].gsub(/(\?.*)$/, "")).to_s + query
576 rescue
577# ignored
578 end
579 end
580 end
581
582 # Extract metadata (unless detect_tags is false)
583 if location && (source = get_source_if_known(KNOWN_STORY_PARSERS, location))
584 params = send("parse_story_from_#{source.downcase}", story, detect_tags)
585 work_params.merge!(params)
586 else
587 work_params.merge!(parse_story_from_unknown(story, detect_tags))
588 end
589
590 shift_chapter_attributes(sanitize_params(work_params))
591 end
592
593 # our fallback: parse a story from an unknown source, so we have no special
594 # rules.
595 def parse_story_from_unknown(story, detect_tags = true)
596 work_params = { chapter_attributes: {} }
597 story_head = ""
598 story_head = @doc.css("head").inner_html if @doc.css("head")
599
600 # Story content - Look for progressively less specific containers or grab everything
601 element = @doc.at_css('.chapter-content') || @doc.at_css('body') || @doc.at_css('html') || @doc
602 storytext = element ? element.inner_html : story
603
604 meta = {}
605 meta.merge!(scan_text_for_meta(story_head, detect_tags)) unless story_head.blank?
606 meta.merge!(scan_text_for_meta(story, detect_tags))
607 meta[:title] ||= @doc.css('title').inner_html
608 work_params[:chapter_attributes][:title] = meta.delete(:chapter_title)
609 work_params[:chapter_attributes][:content] = clean_storytext(storytext)
610 work_params.merge!(meta)
611 end
612
613 # Parses a story from livejournal or a livejournal equivalent (eg, dreamwidth, insanejournal)
614 # Assumes that we have downloaded the story from one of those equivalents (ie, we've downloaded
615 # it in format=light which is a stripped-down plaintext version.)
616 #
617 def parse_story_from_lj(_story, detect_tags = true)
618 work_params = { chapter_attributes: {} }
619
620 # in LJ "light" format, the story contents are in the second div
621 # inside the body.
622 body = @doc.css("body")
623 storytext = body.css("article.b-singlepost-body").inner_html
624 storytext = body.css("div.aentry-post__text").inner_html if storytext.empty?
625 storytext = body.inner_html if storytext.empty?
626
627 # cleanup the text
628 # storytext.gsub!(/<br\s*\/?>/i, "\n") # replace the breaks with newlines
629 storytext = clean_storytext(storytext)
630
631 work_params[:chapter_attributes][:content] = storytext
632 work_params[:title] = @doc.css("title").inner_html
633 work_params[:title].gsub! /^[^:]+: /, ""
634 work_params.merge!(scan_text_for_meta(storytext, detect_tags))
635
636 date = @doc.css("time.b-singlepost-author-date")
637 date = @doc.css("p.aentry-head__date/time") if date.empty?
638 work_params[:revised_at] = convert_revised_at(date.first.inner_text) unless date.empty?
639
640 work_params
641 end
642
643 def parse_story_from_dw(_story, detect_tags = true)
644 work_params = { chapter_attributes: {} }
645
646 body = @doc.css("body")
647 content_divs = body.css("div.contents")
648
649 if content_divs[0].present?
650 # Get rid of the DW metadata table
651 content_divs[0].css("div.currents, ul.entry-management-links, div.header.inner, span.restrictions, h3.entry-title").each(&:remove)
652 storytext = content_divs[0].inner_html
653 else
654 storytext = body.inner_html
655 end
656
657 # cleanup the text
658 storytext = clean_storytext(storytext)
659
660 work_params[:chapter_attributes][:content] = storytext
661 work_params[:title] = @doc.css("title").inner_html
662 work_params[:title].gsub! /^[^:]+: /, ""
663 work_params.merge!(scan_text_for_meta(storytext, detect_tags))
664
665 font_blocks = @doc.xpath('//font')
666 unless font_blocks.empty?
667 date = font_blocks.first.inner_text
668 work_params[:revised_at] = convert_revised_at(date)
669 end
670
671 # get the date
672 date = @doc.css("span.date").inner_text
673 work_params[:revised_at] = convert_revised_at(date)
674
675 work_params
676 end
677
678 def parse_story_from_deviantart(_story, detect_tags = true)
679 work_params = { chapter_attributes: {} }
680 storytext = ""
681 notes = ""
682
683 body = @doc.css("body")
684 title = @doc.css("title").inner_html.gsub /\s*on deviantart$/i, ""
685
686 # Find the image (original size) if it's art
687 image_full = body.css("div.dev-view-deviation img.dev-content-full")
688 unless image_full[0].nil?
689 storytext = "<center><img src=\"#{image_full[0]["src"]}\"></center>"
690 end
691
692 # Find the fic text if it's fic (needs the id for disambiguation, the "deviantART loves you" bit in the footer has the same class path)
693 text_table = body.css(".grf-indent > div:nth-child(1)")[0]
694 unless text_table.nil?
695 # Try to remove some metadata (title and author) from the work's text, if possible
696 # Try to remove the title: if it exists, and if it's the same as the browser title
697 if text_table.css("h1")[0].present? && title && title.match(text_table.css("h1")[0].text)
698 text_table.css("h1")[0].remove
699 end
700
701 # Try to remove the author: if it exists, and if it follows a certain pattern
702 if text_table.css("small")[0].present? && text_table.css("small")[0].inner_html.match(/by ~.*?<a class="u" href=/m)
703 text_table.css("small")[0].remove
704 end
705 storytext = text_table.inner_html
706 end
707
708 # cleanup the text
709 storytext.gsub!(%r{<br\s*\/?>}i, "\n") # replace the breaks with newlines
710 storytext = clean_storytext(storytext)
711 work_params[:chapter_attributes][:content] = storytext
712
713 # Find the notes
714 content_divs = body.css("div.text-ctrl div.text")
715 notes = content_divs[0].inner_html unless content_divs[0].nil?
716
717 # cleanup the notes
718 notes.gsub!(%r{<br\s*\/?>}i, "\n") # replace the breaks with newlines
719 notes = clean_storytext(notes, "notes")
720 work_params[:notes] = notes
721
722 work_params.merge!(scan_text_for_meta(notes, detect_tags))
723 work_params[:title] = title
724
725 body.css("div.dev-title-container h1 a").each do |node|
726 if node["class"] != "u"
727 work_params[:title] = node.inner_html
728 end
729 end
730
731 tags = []
732 @doc.css("div.dev-about-cat-cc a.h").each { |node| tags << node.inner_html }
733 work_params[:freeform_string] = clean_tags(tags.join(ArchiveConfig.DELIMITER_FOR_OUTPUT))
734
735 details = @doc.css("div.dev-right-bar-content span[title]")
736 unless details[0].nil?
737 work_params[:revised_at] = convert_revised_at(details[0].inner_text)
738 end
739
740 work_params
741 end
742
743 def parse_story_from_ao3(_story, detect_tags = true)
744 work_params = { chapter_attributes: {} }
745
746 # Title: use the work heading, not the browser tab title
747 title_node = @doc.at_css('h2.title.heading')
748 work_params[:title] = if title_node
749 title_node.inner_text.strip
750 else
751 @doc.at_css('title')&.inner_text&.sub(/\s*\[Archive of Our Own\]\s*$/i, '')&.strip.to_s
752 end
753
754 # Summary
755 summary_node = @doc.at_css('.summary.module blockquote.userstuff')
756 work_params[:summary] = clean_storytext(summary_node.inner_html) if summary_node
757
758 # Author beginning notes (inside .preface.group, before chapter content)
759 preface = @doc.at_css('.preface.group')
760 if preface
761 notes_node = preface.at_css('.notes.module blockquote.userstuff')
762 work_params[:notes] = clean_storytext(notes_node.inner_html) if notes_node
763 end
764
765 # Story text: extract only from #chapters .userstuff, not the whole page body
766 chapters_div = @doc.at_css('#chapters')
767 if chapters_div
768 userstuff = chapters_div.at_css('.userstuff')
769 storytext = userstuff ? userstuff.inner_html : chapters_div.inner_html
770 else
771 storytext = @doc.at_css('body')&.inner_html || _story
772 end
773 work_params[:chapter_attributes][:content] = clean_storytext(storytext)
774
775 if detect_tags
776 meta_group = @doc.at_css('dl.work.meta.group')
777 if meta_group
778 rating = meta_group.css('dd.rating.tags li a.tag').map { |a| a.inner_text.strip }
779 work_params[:rating_string] = convert_rating_string(rating.first) if rating.any?
780
781 warnings = meta_group.css('dd.warning.tags li a.tag').map { |a| a.inner_text.strip }
782 work_params[:archive_warning_string] = warnings.join(', ') if warnings.any?
783
784 fandoms = meta_group.css('dd.fandom.tags li a.tag').map { |a| a.inner_text.strip }
785 work_params[:fandom_string] = clean_tags(fandoms.join(ArchiveConfig.DELIMITER_FOR_OUTPUT)) if fandoms.any?
786
787 relationships = meta_group.css('dd.relationship.tags li a.tag').map { |a| a.inner_text.strip }
788 work_params[:relationship_string] = clean_tags(relationships.join(ArchiveConfig.DELIMITER_FOR_OUTPUT)) if relationships.any?
789
790 characters = meta_group.css('dd.character.tags li a.tag').map { |a| a.inner_text.strip }
791 work_params[:character_string] = clean_tags(characters.join(ArchiveConfig.DELIMITER_FOR_OUTPUT)) if characters.any?
792
793 freeforms = meta_group.css('dd.freeform.tags li a.tag').map { |a| a.inner_text.strip }
794 work_params[:freeform_string] = clean_tags(freeforms.join(ArchiveConfig.DELIMITER_FOR_OUTPUT)) if freeforms.any?
795
796 published = meta_group.at_css('dd.published')
797 work_params[:revised_at] = convert_revised_at(published.inner_text.strip) if published
798 end
799 end
800
801 post_process_meta(work_params)
802 end
803
804 # Move and/or copy any meta attributes that need to be on the chapter rather
805 # than on the work itself
806 def shift_chapter_attributes(work_params)
807 CHAPTER_ATTRIBUTES_ONLY.each_pair do |work_attrib, chapter_attrib|
808 if work_params[work_attrib] && !work_params[:chapter_attributes][chapter_attrib]
809 work_params[:chapter_attributes][chapter_attrib] = work_params[work_attrib]
810 work_params.delete(work_attrib)
811 end
812 end
813
814 # copy any attributes from work to chapter as necessary
815 CHAPTER_ATTRIBUTES_ALSO.each_pair do |work_attrib, chapter_attrib|
816 if work_params[work_attrib] && !work_params[:chapter_attributes][chapter_attrib]
817 work_params[:chapter_attributes][chapter_attrib] = work_params[work_attrib]
818 end
819 end
820
821 work_params
822 end
823
824 # Find any cases of the given pieces of meta in the given text
825 # and return a hash of meta values
826 def scan_text_for_meta(text, detect_tags = true)
827 # break up the text with some extra newlines to make matching more likely
828 # and strip out some tags
829 text = text.gsub(/<br/, "\n<br")
830 text.gsub!(/<p/, "\n<p")
831 text.gsub!(/<\/?(label|span|div|b)(.*?)?>/, '')
832
833 meta = {}
834 metapatterns = detect_tags ? REQUIRED_META.merge(OPTIONAL_META) : REQUIRED_META
835 is_tag = {}.tap do |h|
836 %w[fandom_string relationship_string freeform_string rating_string archive_warning_string].each do |c|
837 h[c.to_sym] = true
838 end
839 end
840 handler = {}.tap do |h|
841 %w[rating_string revised_at].each do |c|
842 h[c.to_sym] = "convert_#{c.to_s.downcase}"
843 end
844 end
845
846 # 1. Look for Pattern: (whatever), optionally followed by a closing p or div tag
847 # 2. Set meta[:metaname] = whatever
848 # eg, if it finds Fandom: Stargate SG-1 it will set meta[:fandom] = Stargate SG-1
849 # 3. convert_<metaname> for cleanup if such a function is defined (eg convert_rating_string)
850 metapatterns.each do |metaname, pattern|
851 metapattern = Regexp.new("(?:#{pattern}|#{pattern.pluralize})\s*:\s*(.*?)(?:</(?:p|div)>)?$", Regexp::IGNORECASE)
852 if text.match(metapattern)
853 value = Regexp.last_match[1]
854 value = clean_tags(value) if is_tag[metaname]
855 value = clean_close_html_tags(value)
856 value.strip! # lose leading/trailing whitespace
857 value = send(handler[metaname], value) if handler[metaname]
858
859 meta[metaname] = value
860 end
861 end
862 post_process_meta meta
863 end
864
865 def download_with_timeout(location, limit = 10)
866 story = +""
867 Timeout.timeout(STORY_DOWNLOAD_TIMEOUT) do
868 begin
869 # we do a little cleanup here in case the user hasn't included the 'http://'
870 # or if they've used capital letters or an underscore in the hostname
871 uri = UrlFormatter.new(location).standardized
872 raise Error, I18n.t("story_parser.on_archive") if ArchiveConfig.PERMITTED_HOSTS.include?(uri.host)
873
874 env_proxy = ENV["http_proxy"]
875 http = if env_proxy
876 proxy = URI(env_proxy)
877 Net::HTTP.new(uri.hostname, uri.port, proxy.hostname, proxy.port)
878 else
879 Net::HTTP.new(uri.hostname, uri.port)
880 end
881 http.use_ssl = true if uri.scheme == "https"
882 response = http.start { |h| h.request_get(uri.path.presence || "/") }
883
884 case response
885 when Net::HTTPSuccess
886 story = response.body
887 when Net::HTTPRedirection
888 if limit.positive?
889 new_uri = URI.parse(response["location"])
890 new_uri = URI.join(uri, new_uri) if new_uri.relative?
891 story = download_with_timeout(new_uri.to_s, limit - 1)
892 end
893 else
894 Rails.logger.error("------- STORY PARSER: download_with_timeout: response is not success or redirection ------")
895 nil
896 end
897 rescue Errno::ECONNREFUSED, SocketError, EOFError => e
898 Rails.logger.error("------- STORY PARSER: download_with_timeout: error rescue: \n#{e.inspect} ------")
899 nil
900 end
901 end
902 if story.blank?
903 raise Error, "We couldn't download anything from #{location}. Please make sure that the URL is correct and complete, and try again."
904 end
905
906 # clean up any erroneously included string terminator (AO3-2251)
907 story.delete("\000")
908 end
909
910 def get_last_modified(location)
911 Timeout.timeout(STORY_DOWNLOAD_TIMEOUT) do
912 resp = open(location)
913 resp.last_modified
914 end
915 end
916
917 def get_source_if_known(known_sources, location)
918 known_sources.each do |source|
919 pattern = Regexp.new(eval("SOURCE_#{source.upcase}"), Regexp::IGNORECASE)
920 return source if location.match(pattern)
921 end
922 nil
923 end
924
925 def clean_close_html_tags(value)
926 # if there are any closing html tags at the start of the value let's ditch them
927 value.gsub(/^(\s*<\/[^>]+>)+/, '')
928 end
929
930 # We clean the text as if it had been submitted as the content of a chapter
931 def clean_storytext(storytext, field = "content")
932 storytext = storytext.encode("UTF-8", invalid: :replace, undef: :replace, replace: "") unless storytext.encoding.name == "UTF-8"
933 sanitize_value(field, storytext)
934 end
935
936 # works conservatively -- doesn't split on
937 # spaces and truncates instead.
938 def clean_tags(tags)
939 tags = Sanitize.clean(tags.force_encoding("UTF-8")) # no html allowed in tags
940 tags_list = tags =~ /,/ ? tags.split(/,/) : [tags]
941 new_list = []
942 tags_list.each do |tag|
943 tag.gsub!(/[*<>]/, '')
944 tag = truncate_on_word_boundary(tag, ArchiveConfig.TAG_MAX)
945 new_list << tag unless tag.blank?
946 end
947 new_list.join(ArchiveConfig.DELIMITER_FOR_OUTPUT)
948 end
949
950 def truncate_on_word_boundary(text, max_length)
951 return if text.blank?
952 words = text.split
953 truncated = words.first
954 if words.length > 1
955 words[1..words.length].each do |word|
956 truncated += " " + word if truncated.length + word.length + 1 <= max_length
957 end
958 end
959 truncated[0..max_length - 1]
960 end
961
962 # convert space-separated tags to comma-separated
963 def clean_and_split_tags(tags)
964 tags = tags.split(/\s+/).join(',') if !tags.match(/,/) && tags.match(/\s/)
965 clean_tags(tags)
966 end
967
968 # Convert the common ratings into whatever ratings we're
969 # using on this archive.
970 def convert_rating_string(rating)
971 rating = rating.downcase
972 if rating =~ /^(nc-?1[78]|x|ma|explicit)/
973 ArchiveConfig.RATING_EXPLICIT_TAG_NAME
974 elsif rating =~ /^(r|m|mature)/
975 ArchiveConfig.RATING_MATURE_TAG_NAME
976 elsif rating =~ /^(pg-?1[35]|t|teen)/
977 ArchiveConfig.RATING_TEEN_TAG_NAME
978 elsif rating =~ /^(pg|g|k+|k|general audiences)/
979 ArchiveConfig.RATING_GENERAL_TAG_NAME
980 else
981 ArchiveConfig.RATING_DEFAULT_TAG_NAME
982 end
983 end
984
985 def convert_revised_at(date_string)
986 begin
987 date = nil
988 if date_string =~ /^(\d+)$/
989 # probably seconds since the epoch
990 date = Time.at(Regex.last_match[1].to_i)
991 end
992 date ||= Date.parse(date_string)
993 return '' if date > Date.current
994 return date
995 rescue ArgumentError, TypeError
996 return ''
997 end
998 end
999
1000 # Additional processing for meta - currently to make sure warnings
1001 # that aren't Archive warnings become additional tags instead
1002 def post_process_meta(meta)
1003 if meta[:archive_warning_string]
1004 result = process_warnings(meta[:archive_warning_string], meta[:freeform_string])
1005 meta[:archive_warning_string] = result[:archive_warning_string]
1006 meta[:freeform_string] = result[:freeform_string]
1007 end
1008 meta
1009 end
1010
1011 def process_warnings(warning_string, freeform_string)
1012 result = {
1013 archive_warning_string: warning_string,
1014 freeform_string: freeform_string
1015 }
1016 new_warning = ''
1017 result[:archive_warning_string].split(/\s?,\s?/).each do |warning|
1018 if ArchiveWarning.warning? warning
1019 new_warning += ', ' unless new_warning.blank?
1020 new_warning += warning
1021 else
1022 result[:freeform_string] = (result[:freeform_string] || '') + ", #{warning}"
1023 end
1024 end
1025 result[:archive_warning_string] = new_warning
1026 result
1027 end
1028
1029 # tries to find appropriate existing collections and converts them to comma-separated collection names only
1030 def get_collection_names(collection_string)
1031 collections = ""
1032 collection_string.split(',').map(&:squish).each do |collection_name|
1033 collection = Collection.find_by(name: collection_name) || Collection.find_by(title: collection_name)
1034 if collection
1035 collections += ", " unless collections.blank?
1036 collections += collection.name
1037 end
1038 end
1039 collections
1040 end
1041
1042 # determine which value to use for a metadata field
1043 def meta_or_default(detected_field, provided_field, default = nil)
1044 if @options[:override_tags] || detected_field.blank?
1045 if provided_field.blank?
1046 detected_field.blank? ? default : detected_field
1047 else
1048 provided_field
1049 end
1050 else
1051 detected_field
1052 end
1053 end
1054end
1055