story_parser.rb
· 40 KiB · Ruby
Raw
# Parse stories from other websites and uploaded files, looking for metadata to harvest
# and put into the archive.
#
class StoryParser
require 'timeout'
require 'nokogiri'
require 'mechanize'
require 'open-uri'
include HtmlCleaner
OPTIONAL_META = {notes: 'Note',
freeform_string: 'Tag',
fandom_string: 'Fandom',
rating_string: 'Rating',
archive_warning_string: 'Warning',
relationship_string: 'Relationship|Pairing',
character_string: 'Character' }.freeze
REQUIRED_META = { title: 'Title',
summary: 'Summary',
revised_at: 'Date|Posted|Posted on|Posted at',
chapter_title: 'Chapter Title' }.freeze
# Use this for raising custom error messages
# (so that we can distinguish them from unexpected exceptions due to
# faulty code)
class Error < StandardError
end
# These attributes need to be moved from the work to the chapter
# format: {work_attribute_name: :chapter_attribute_name} (can be the same)
CHAPTER_ATTRIBUTES_ONLY = {}
# These attributes need to be copied from the work to the chapter
CHAPTER_ATTRIBUTES_ALSO = { revised_at: :published_at }.freeze
### NOTE ON KNOWN SOURCES
# These lists will stop with the first one it matches, so put more-specific matches
# towards the front of the list.
# places for which we have a custom parse_story_from_[source] method
# for getting information out of the downloaded text
KNOWN_STORY_PARSERS = %w[ao3 deviantart dw lj].freeze
# places for which we have a custom parse_author_from_[source] method
# which returns an external_author object including an email address
KNOWN_AUTHOR_PARSERS = %w[lj].freeze
# places for which we have a download_story_from_[source]
# used to customize the downloading process
KNOWN_STORY_LOCATIONS = %w[lj].freeze
# places for which we have a download_chaptered_from
# to get a set of chapters all together
CHAPTERED_STORY_LOCATIONS = %w[ffnet thearchive_net efiction quotev].freeze
# regular expressions to match against the URLS
SOURCE_AO3 = '(archiveofourown\.org|ao3\.org|superlove\.sayitditto\.net|sunset\.femslash\.club)'.freeze
SOURCE_LJ = '((live|dead|insane)journal\.com)|journalfen(\.net|\.com)|dreamwidth\.org'.freeze
SOURCE_DW = 'dreamwidth\.org'.freeze
SOURCE_FFNET = '(^|[^A-Za-z0-9-])fanfiction\.net'.freeze
SOURCE_DEVIANTART = 'deviantart\.com'.freeze
SOURCE_THEARCHIVE_NET = 'the\-archive\.net'.freeze
SOURCE_EFICTION = 'viewstory\.php'.freeze
SOURCE_QUOTEV = 'quotev\.com'.freeze
# time out if we can't download fast enough
STORY_DOWNLOAD_TIMEOUT = 60
MAX_CHAPTER_COUNT = 200
# To check for duplicate chapters, take a slice this long out of the story
# (in characters)
DUPLICATE_CHAPTER_LENGTH = 10_000
# Import many stories
def import_many(urls, options = {})
# Try to get the works
works = []
failed_urls = []
errors = []
@options = options
urls.each do |url|
begin
response = download_and_parse_work(url, options)
work = response[:work]
if response[:status] == :created
if work && work.save
work.chapters.each(&:save)
works << work
else
failed_urls << url
errors << work.errors.values.join(", ")
work.delete if work
end
elsif response[:status] == :already_imported
raise StoryParser::Error, response[:message]
end
rescue Timeout::Error
failed_urls << url
errors << "Import has timed out. This may be due to connectivity problems with the source site. Please try again in a few minutes, or check Known Issues to see if there are import problems with this site."
work.delete if work
rescue Error => exception
failed_urls << url
errors << "We couldn't successfully import that work, sorry: #{exception.message}"
work.delete if work
end
end
[works, failed_urls, errors]
end
# Downloads a story and passes it on to the parser.
# If the URL of the story is from a site for which we have special rules
# (eg, downloading from a livejournal clone, you want to use ?format=light
# to get a nice and consistent post format), it will pre-process the url
# according to the rules for that site.
def download_and_parse_work(location, options = {})
status = :created
message = ""
work = Work.find_by_url(location)
if work.nil?
@options = options
source = get_source_if_known(CHAPTERED_STORY_LOCATIONS, location)
if source.nil?
story = download_text(location)
work = parse_story(story, location, options)
else
work = download_and_parse_chaptered_story(source, location, options)
end
else
status = :already_imported
message = "A work has already been imported from #{location}."
end
{
status: status,
message: message,
work: work
}
end
# Given an array of urls for chapters of a single story,
# download them all and combine into a single work
def import_chapters_into_story(locations, options = {})
status = :created
work = Work.find_by_url(locations.first)
if work.nil?
chapter_contents = []
@options = options
locations.each do |location|
chapter_contents << download_text(location)
end
work = parse_chapters_into_story(locations.first, chapter_contents, options)
message = "Successfully created work \"" + work.title + "\"."
else
status = :already_imported
message = "A work has already been imported from #{locations.first}."
end
{
status: status,
message: message,
work: work
}
end
### OLD PARSING METHODS
# Import many stories
def import_from_urls(urls, options = {})
# Try to get the works
works = []
failed_urls = []
errors = []
@options = options
urls.each do |url|
begin
work = download_and_parse_story(url, options)
if work && work.save
work.chapters.each(&:save)
works << work
else
failed_urls << url
errors << work.errors.values.join(", ")
work.delete if work
end
rescue Timeout::Error
failed_urls << url
errors << "Import has timed out. This may be due to connectivity problems with the source site. Please try again in a few minutes, or check Known Issues to see if there are import problems with this site."
work.delete if work
rescue Error => exception
failed_urls << url
errors << "We couldn't successfully import that work, sorry: #{exception.message}"
work.delete if work
end
end
[works, failed_urls, errors]
end
# Downloads a story and passes it on to the parser.
# If the URL of the story is from a site for which we have special rules
# (eg, downloading from a livejournal clone, you want to use ?format=light
# to get a nice and consistent post format), it will pre-process the url
# according to the rules for that site.
def download_and_parse_story(location, options = {})
check_for_previous_import(location)
@options = options
source = get_source_if_known(CHAPTERED_STORY_LOCATIONS, location)
if source.nil?
story = download_text(location)
work = parse_story(story, location, options)
else
work = download_and_parse_chaptered_story(source, location, options)
end
work
end
# Given an array of urls for chapters of a single story,
# download them all and combine into a single work
def download_and_parse_chapters_into_story(locations, options = {})
check_for_previous_import(locations.first)
chapter_contents = []
@options = options
locations.each do |location|
chapter_contents << download_text(location)
end
parse_chapters_into_story(locations.first, chapter_contents, options)
end
### PARSING METHODS
# Parses the text of a story, optionally from a given location.
def parse_story(story, location, options = {})
work_params = parse_common(story, location, options[:encoding], options[:detect_tags])
# move any attributes from work to chapter if necessary
set_work_attributes(Work.new(work_params), location, options)
end
# parses and adds a new chapter to the end of the work
def parse_chapter_of_work(work, chapter_content, location, options = {})
tmp_work_params = parse_common(chapter_content, location, options[:encoding], options[:detect_tags])
chapter = get_chapter_from_work_params(tmp_work_params)
work.chapters << set_chapter_attributes(work, chapter)
work
end
def parse_chapters_into_story(location, chapter_contents, options = {})
work = nil
chapter_contents.each do |content|
work_params = parse_common(content, location, options[:encoding], options[:detect_tags])
if work.nil?
# create the new work
work = Work.new(work_params)
else
new_chapter = get_chapter_from_work_params(work_params)
work.chapters << set_chapter_attributes(work, new_chapter)
end
end
set_work_attributes(work, location, options)
end
# Everything below here is protected and should not be touched by outside
# code -- please use the above functions to parse external works.
protected
# tries to create an external author for a given url
def parse_author(location, ext_author_name, ext_author_email)
if location.present? && ext_author_name.blank? && ext_author_email.blank?
source = get_source_if_known(KNOWN_AUTHOR_PARSERS, location)
if source.nil?
raise Error, "No external author name or email specified"
else
send("parse_author_from_#{source.downcase}", location)
end
else
parse_author_common(ext_author_email, ext_author_name)
end
end
# download an entire story from an archive type where we know how to parse multi-chaptered works
# this should only be called from download_and_parse_story
def download_and_parse_chaptered_story(source, location, options = {})
chapter_contents = send("download_chaptered_from_#{source.downcase}", location)
parse_chapters_into_story(location, chapter_contents, options)
end
# our custom url finder checks for previously imported URL in almost any format it may have been presented
def check_for_previous_import(location)
if Work.find_by_url(location).present?
raise Error, "A work has already been imported from #{location}."
end
end
def set_chapter_attributes(work, chapter)
chapter.position = work.chapters.length + 1
chapter.posted = true
chapter
end
def set_work_attributes(work, location = "", options = {})
raise Error, "Work could not be downloaded" if work.nil?
@options = options
work.imported_from_url = location # @todo remove this as part of AO3-6979
work.imported_url = ImportedUrl.new(original: work.imported_from_url)
work.ip_address = options[:ip_address]
work.expected_number_of_chapters = work.chapters.length
work.revised_at = work.chapters.last.published_at
if work.revised_at && work.revised_at.to_date < Date.current
work.backdate = true
end
# set authors for the works
pseuds = []
pseuds << User.current_user.default_pseud unless options[:do_not_set_current_author] || User.current_user.nil?
pseuds << options[:archivist].default_pseud if options[:archivist]
pseuds << options[:pseuds] if options[:pseuds]
pseuds = pseuds.flatten.compact.uniq
raise Error, "A work must have at least one author specified" if pseuds.empty?
pseuds.each do |pseud|
work.creatorships.build(pseud: pseud, enable_notifications: true)
work.chapters.each do |chapter|
chapter.creatorships.build(pseud: pseud)
end
end
# handle importing works for others
# build an external creatorship for each author
if options[:importing_for_others]
external_author_names = options[:external_author_names] || parse_author(location, options[:external_author_name], options[:external_author_email])
# convert to an array if not already one
external_author_names = [external_author_names] if external_author_names.is_a?(ExternalAuthorName)
if options[:external_coauthor_name].present?
external_author_names << parse_author(location, options[:external_coauthor_name], options[:external_coauthor_email])
end
external_author_names.each do |external_author_name|
next if !external_author_name || external_author_name.external_author.blank?
if external_author_name.external_author.do_not_import
# we're not allowed to import works from this address
raise Error, "Author #{external_author_name.name} at #{external_author_name.external_author.email} does not allow importing their work to this archive."
end
work.external_creatorships.build(external_author_name: external_author_name, archivist: (options[:archivist] || User.current_user))
end
end
# lock to registered users if specified or importing for others
work.restricted = options[:restricted] || options[:importing_for_others] || false
# set comment permissions
work.comment_permissions = options[:comment_permissions] || "enable_all"
work.moderated_commenting_enabled = options[:moderated_commenting_enabled] || false
# set default values for required tags
work.fandom_string = meta_or_default(work.fandom_string, options[:fandom], ArchiveConfig.FANDOM_NO_TAG_NAME)
work.rating_string = meta_or_default(work.rating_string, options[:rating], ArchiveConfig.RATING_DEFAULT_TAG_NAME)
work.archive_warning_strings = meta_or_default(work.archive_warning_strings, options[:archive_warning], ArchiveConfig.WARNING_DEFAULT_TAG_NAME)
work.category_string = meta_or_default(work.category_string, options[:category], [])
work.character_string = meta_or_default(work.character_string, options[:character], [])
work.relationship_string = meta_or_default(work.relationship_string, options[:relationship], [])
work.freeform_string = meta_or_default(work.freeform_string, options[:freeform], [])
# set default value for title
work.title = meta_or_default(work.title, options[:title], "Untitled Imported Work")
work.summary = meta_or_default(work.summary, options[:summary], '')
work.notes = meta_or_default(work.notes, options[:notes], '')
# set collection name if present
work.collection_names = get_collection_names(options[:collection_names]) if options[:collection_names].present?
# set default language (English)
work.language_id = options[:language_id] || Language.default.id
work.posted = true if options[:post_without_preview]
work.chapters.each do |chapter|
if chapter.content.length > ArchiveConfig.CONTENT_MAX
# TODO: eventually: insert a new chapter
chapter.content.truncate(ArchiveConfig.CONTENT_MAX, omission: "<strong>WARNING: import truncated automatically because chapter was too long! Please add a new chapter for remaining content.</strong>", separator: "</p>")
elsif chapter.content.empty?
raise Error, "Chapter #{chapter.position} of \"#{work.title}\" is blank."
end
chapter.posted = true # do not save - causes the chapters to exist even if work doesn't get created!
end
work
end
def parse_author_from_lj(location)
return if location !~ %r{^(?:http:\/\/)?(?<lj_name>[^.]*).(?<site_name>livejournal\.com|dreamwidth\.org|insanejournal\.com|journalfen.net)}
email = ""
lj_name = Regexp.last_match[:lj_name]
site_name = Regexp.last_match[:site_name]
if lj_name == "community"
# whups
post_text = download_text(location)
doc = Nokogiri.parse(post_text)
lj_name = doc.xpath("/html/body/div[2]/div/div/div/table/tbody/tr/td[2]/span/a[2]/b").content
end
profile_url = "http://#{lj_name}.#{site_name}/profile"
lj_profile = download_text(profile_url)
doc = Nokogiri.parse(lj_profile)
contact = doc.css('div.contact').inner_html
if contact.present?
contact.gsub! '<p class="section_body_title">Contact:</p>', ""
contact.gsub! /<\/?(span|i)>/, ""
contact.delete! "\n"
contact.gsub! "<br/>", ""
if contact =~ /(.*@.*\..*)/
email = Regexp.last_match[1]
end
end
email = "#{lj_name}@#{site_name}" if email.blank?
parse_author_common(email, lj_name)
end
def parse_author_from_unknown(_location)
# for now, nothing
nil
end
def parse_author_common(email, name)
errors = []
errors << "No author name specified" if name.blank?
if email.present?
external_author = ExternalAuthor.find_or_create_by(email: email)
errors += external_author.errors.full_messages
else
errors << "No author email specified"
end
raise Error, errors.join("\n") if errors.present?
# convert to ASCII and strip out invalid characters (everything except alphanumeric characters, _, @ and -)
redacted_name = name.to_ascii.gsub(/[^\w[ \-@.]]/u, "")
if redacted_name.present?
external_author.names.find_or_create_by(name: redacted_name)
else
external_author.default_name
end
end
def get_chapter_from_work_params(work_params)
@chapter = Chapter.new(work_params[:chapter_attributes])
# don't override specific chapter params (eg title) with work params
chapter_params = work_params.delete_if do |name, _param|
!@chapter.attribute_names.include?(name.to_s) || !@chapter.send(name.to_s).blank?
end
@chapter.update(chapter_params)
@chapter
end
def download_text(location)
source = get_source_if_known(KNOWN_STORY_LOCATIONS, location)
if source.nil?
download_with_timeout(location)
else
send("download_from_#{source.downcase}", location)
end
end
# canonicalize the url for downloading from lj or clones
def download_from_lj(location)
url = location
url.gsub!(/\#(.*)$/, "") # strip off any anchor information
url.gsub!(/\?(.*)$/, "") # strip off any existing params at the end
url.gsub!('_', '-') # convert underscores in usernames to hyphens
url += "?format=light" # go to light format
text = download_with_timeout(url)
if text.match(/adult_check/)
Timeout::timeout(STORY_DOWNLOAD_TIMEOUT) {
begin
agent = Mechanize.new
url.include?("dreamwidth") ? form = agent.get(url).forms.first : form = agent.get(url).forms.third
page = agent.submit(form, form.buttons.first) # submits the adult concepts form
text = page.body.force_encoding(agent.page.encoding)
rescue
text = ""
end
}
end
text
end
# grab all the chapters of the story from ff.net
def download_chaptered_from_ffnet(_location)
raise Error, "Sorry, Fanfiction.net does not allow imports from their site."
end
def download_chaptered_from_quotev(_location)
raise Error, "Sorry, Quotev.com does not allow imports from their site."
end
# this is an efiction archive but it doesn't handle chapters normally
# best way to handle is to get the full story printable version
# We have to make it a download-chaptered because otherwise it gets sent to the
# generic efiction version since chaptered sources are checked first
def download_chaptered_from_thearchive_net(location)
if location.match(/^(.*)\/.*viewstory\.php.*[^p]sid=(\d+)($|&)/i)
location = "#{$1}/viewstory.php?action=printable&psid=#{$2}"
end
text = download_with_timeout(location)
text.sub!('</style>', '</style></head>') unless text.match('</head>')
[text]
end
# grab all the chapters of a story from an efiction-based site
def download_chaptered_from_efiction(location)
chapter_contents = []
if location.match(/^(?<site>.*)\/.*viewstory\.php.*sid=(?<storyid>\d+)($|&)/i)
site = Regexp.last_match[:site]
storyid = Regexp.last_match[:storyid]
chapnum = 1
last_body = ""
Timeout::timeout(STORY_DOWNLOAD_TIMEOUT) do
loop do
url = "#{site}/viewstory.php?action=printable&sid=#{storyid}&chapter=#{chapnum}"
body = download_with_timeout(url)
# get a section to check that this isn't a duplicate of previous chapter
body_to_check = body.slice(10, DUPLICATE_CHAPTER_LENGTH)
if body.nil? || body_to_check == last_body || chapnum > MAX_CHAPTER_COUNT || body.match(/<div class='chaptertitle'> by <\/div>/) || body.match(/Access denied./) || body.match(/Chapter : /)
break
end
# save the value to check for duplicate chapter
last_body = body_to_check
# clean up the broken head in many efiction printable sites
body.sub!('</style>', '</style></head>') unless body.match('</head>')
chapter_contents << body
chapnum += 1
end
end
end
chapter_contents
end
# This is the heavy lifter, invoked by all the story and chapter parsers.
# It takes a single string containing the raw contents of a story, parses it with
# Nokogiri into the @doc object, and then and calls a subparser.
#
# If the story source can be identified as one of the sources we know how to parse in some custom/
# special way, parse_common calls the customized parse_story_from_[source] method.
# Otherwise, it falls back to parse_story_from_unknown.
#
# This produces a hash equivalent to the params hash that is normally created by the standard work
# upload form.
#
# parse_common then calls sanitize_params (which would also be called on the standard work upload
# form results) and returns the final sanitized hash.
#
def parse_common(story, location = nil, encoding = nil, detect_tags = true)
work_params = { title: "Untitled Imported Work", chapter_attributes: { content: "" } }
# Encode as HTML - the dummy "foo" tag will be stripped out by the sanitizer but forces Nokogiri to
# preserve line breaks in plain text documents
# Rescue all errors as Nokogiri complains about things the sanitizer will fix later
story.prepend("<foo></foo>")
@doc =
begin
Nokogiri::HTML5.parse(story, encoding: encoding)
rescue StandardError
Nokogiri::HTML5.parse("")
end
# Try to convert all relative links to absolute
base = @doc.at_css("base") ? @doc.css("base")[0]["href"] : location.split("?").first
if base.present?
@doc.css("a").each do |link|
next if link["href"].blank? || link["href"].start_with?("#")
begin
query = link["href"].match(/(\?.*)$/) ? $1 : ""
link["href"] = URI.join(base, link["href"].gsub(/(\?.*)$/, "")).to_s + query
rescue
# ignored
end
end
end
# Extract metadata (unless detect_tags is false)
if location && (source = get_source_if_known(KNOWN_STORY_PARSERS, location))
params = send("parse_story_from_#{source.downcase}", story, detect_tags)
work_params.merge!(params)
else
work_params.merge!(parse_story_from_unknown(story, detect_tags))
end
shift_chapter_attributes(sanitize_params(work_params))
end
# our fallback: parse a story from an unknown source, so we have no special
# rules.
def parse_story_from_unknown(story, detect_tags = true)
work_params = { chapter_attributes: {} }
story_head = ""
story_head = @doc.css("head").inner_html if @doc.css("head")
# Story content - Look for progressively less specific containers or grab everything
element = @doc.at_css('.chapter-content') || @doc.at_css('body') || @doc.at_css('html') || @doc
storytext = element ? element.inner_html : story
meta = {}
meta.merge!(scan_text_for_meta(story_head, detect_tags)) unless story_head.blank?
meta.merge!(scan_text_for_meta(story, detect_tags))
meta[:title] ||= @doc.css('title').inner_html
work_params[:chapter_attributes][:title] = meta.delete(:chapter_title)
work_params[:chapter_attributes][:content] = clean_storytext(storytext)
work_params.merge!(meta)
end
# Parses a story from livejournal or a livejournal equivalent (eg, dreamwidth, insanejournal)
# Assumes that we have downloaded the story from one of those equivalents (ie, we've downloaded
# it in format=light which is a stripped-down plaintext version.)
#
def parse_story_from_lj(_story, detect_tags = true)
work_params = { chapter_attributes: {} }
# in LJ "light" format, the story contents are in the second div
# inside the body.
body = @doc.css("body")
storytext = body.css("article.b-singlepost-body").inner_html
storytext = body.css("div.aentry-post__text").inner_html if storytext.empty?
storytext = body.inner_html if storytext.empty?
# cleanup the text
# storytext.gsub!(/<br\s*\/?>/i, "\n") # replace the breaks with newlines
storytext = clean_storytext(storytext)
work_params[:chapter_attributes][:content] = storytext
work_params[:title] = @doc.css("title").inner_html
work_params[:title].gsub! /^[^:]+: /, ""
work_params.merge!(scan_text_for_meta(storytext, detect_tags))
date = @doc.css("time.b-singlepost-author-date")
date = @doc.css("p.aentry-head__date/time") if date.empty?
work_params[:revised_at] = convert_revised_at(date.first.inner_text) unless date.empty?
work_params
end
def parse_story_from_dw(_story, detect_tags = true)
work_params = { chapter_attributes: {} }
body = @doc.css("body")
content_divs = body.css("div.contents")
if content_divs[0].present?
# Get rid of the DW metadata table
content_divs[0].css("div.currents, ul.entry-management-links, div.header.inner, span.restrictions, h3.entry-title").each(&:remove)
storytext = content_divs[0].inner_html
else
storytext = body.inner_html
end
# cleanup the text
storytext = clean_storytext(storytext)
work_params[:chapter_attributes][:content] = storytext
work_params[:title] = @doc.css("title").inner_html
work_params[:title].gsub! /^[^:]+: /, ""
work_params.merge!(scan_text_for_meta(storytext, detect_tags))
font_blocks = @doc.xpath('//font')
unless font_blocks.empty?
date = font_blocks.first.inner_text
work_params[:revised_at] = convert_revised_at(date)
end
# get the date
date = @doc.css("span.date").inner_text
work_params[:revised_at] = convert_revised_at(date)
work_params
end
def parse_story_from_deviantart(_story, detect_tags = true)
work_params = { chapter_attributes: {} }
storytext = ""
notes = ""
body = @doc.css("body")
title = @doc.css("title").inner_html.gsub /\s*on deviantart$/i, ""
# Find the image (original size) if it's art
image_full = body.css("div.dev-view-deviation img.dev-content-full")
unless image_full[0].nil?
storytext = "<center><img src=\"#{image_full[0]["src"]}\"></center>"
end
# Find the fic text if it's fic (needs the id for disambiguation, the "deviantART loves you" bit in the footer has the same class path)
text_table = body.css(".grf-indent > div:nth-child(1)")[0]
unless text_table.nil?
# Try to remove some metadata (title and author) from the work's text, if possible
# Try to remove the title: if it exists, and if it's the same as the browser title
if text_table.css("h1")[0].present? && title && title.match(text_table.css("h1")[0].text)
text_table.css("h1")[0].remove
end
# Try to remove the author: if it exists, and if it follows a certain pattern
if text_table.css("small")[0].present? && text_table.css("small")[0].inner_html.match(/by ~.*?<a class="u" href=/m)
text_table.css("small")[0].remove
end
storytext = text_table.inner_html
end
# cleanup the text
storytext.gsub!(%r{<br\s*\/?>}i, "\n") # replace the breaks with newlines
storytext = clean_storytext(storytext)
work_params[:chapter_attributes][:content] = storytext
# Find the notes
content_divs = body.css("div.text-ctrl div.text")
notes = content_divs[0].inner_html unless content_divs[0].nil?
# cleanup the notes
notes.gsub!(%r{<br\s*\/?>}i, "\n") # replace the breaks with newlines
notes = clean_storytext(notes, "notes")
work_params[:notes] = notes
work_params.merge!(scan_text_for_meta(notes, detect_tags))
work_params[:title] = title
body.css("div.dev-title-container h1 a").each do |node|
if node["class"] != "u"
work_params[:title] = node.inner_html
end
end
tags = []
@doc.css("div.dev-about-cat-cc a.h").each { |node| tags << node.inner_html }
work_params[:freeform_string] = clean_tags(tags.join(ArchiveConfig.DELIMITER_FOR_OUTPUT))
details = @doc.css("div.dev-right-bar-content span[title]")
unless details[0].nil?
work_params[:revised_at] = convert_revised_at(details[0].inner_text)
end
work_params
end
def parse_story_from_ao3(_story, detect_tags = true)
work_params = { chapter_attributes: {} }
# Title: use the work heading, not the browser tab title
title_node = @doc.at_css('h2.title.heading')
work_params[:title] = if title_node
title_node.inner_text.strip
else
@doc.at_css('title')&.inner_text&.sub(/\s*\[Archive of Our Own\]\s*$/i, '')&.strip.to_s
end
# Summary
summary_node = @doc.at_css('.summary.module blockquote.userstuff')
work_params[:summary] = clean_storytext(summary_node.inner_html) if summary_node
# Author beginning notes (inside .preface.group, before chapter content)
preface = @doc.at_css('.preface.group')
if preface
notes_node = preface.at_css('.notes.module blockquote.userstuff')
work_params[:notes] = clean_storytext(notes_node.inner_html) if notes_node
end
# Story text: extract only from #chapters .userstuff, not the whole page body
chapters_div = @doc.at_css('#chapters')
if chapters_div
userstuff = chapters_div.at_css('.userstuff')
storytext = userstuff ? userstuff.inner_html : chapters_div.inner_html
else
storytext = @doc.at_css('body')&.inner_html || _story
end
work_params[:chapter_attributes][:content] = clean_storytext(storytext)
if detect_tags
meta_group = @doc.at_css('dl.work.meta.group')
if meta_group
rating = meta_group.css('dd.rating.tags li a.tag').map { |a| a.inner_text.strip }
work_params[:rating_string] = convert_rating_string(rating.first) if rating.any?
warnings = meta_group.css('dd.warning.tags li a.tag').map { |a| a.inner_text.strip }
work_params[:archive_warning_string] = warnings.join(', ') if warnings.any?
fandoms = meta_group.css('dd.fandom.tags li a.tag').map { |a| a.inner_text.strip }
work_params[:fandom_string] = clean_tags(fandoms.join(ArchiveConfig.DELIMITER_FOR_OUTPUT)) if fandoms.any?
relationships = meta_group.css('dd.relationship.tags li a.tag').map { |a| a.inner_text.strip }
work_params[:relationship_string] = clean_tags(relationships.join(ArchiveConfig.DELIMITER_FOR_OUTPUT)) if relationships.any?
characters = meta_group.css('dd.character.tags li a.tag').map { |a| a.inner_text.strip }
work_params[:character_string] = clean_tags(characters.join(ArchiveConfig.DELIMITER_FOR_OUTPUT)) if characters.any?
freeforms = meta_group.css('dd.freeform.tags li a.tag').map { |a| a.inner_text.strip }
work_params[:freeform_string] = clean_tags(freeforms.join(ArchiveConfig.DELIMITER_FOR_OUTPUT)) if freeforms.any?
published = meta_group.at_css('dd.published')
work_params[:revised_at] = convert_revised_at(published.inner_text.strip) if published
end
end
post_process_meta(work_params)
end
# Move and/or copy any meta attributes that need to be on the chapter rather
# than on the work itself
def shift_chapter_attributes(work_params)
CHAPTER_ATTRIBUTES_ONLY.each_pair do |work_attrib, chapter_attrib|
if work_params[work_attrib] && !work_params[:chapter_attributes][chapter_attrib]
work_params[:chapter_attributes][chapter_attrib] = work_params[work_attrib]
work_params.delete(work_attrib)
end
end
# copy any attributes from work to chapter as necessary
CHAPTER_ATTRIBUTES_ALSO.each_pair do |work_attrib, chapter_attrib|
if work_params[work_attrib] && !work_params[:chapter_attributes][chapter_attrib]
work_params[:chapter_attributes][chapter_attrib] = work_params[work_attrib]
end
end
work_params
end
# Find any cases of the given pieces of meta in the given text
# and return a hash of meta values
def scan_text_for_meta(text, detect_tags = true)
# break up the text with some extra newlines to make matching more likely
# and strip out some tags
text = text.gsub(/<br/, "\n<br")
text.gsub!(/<p/, "\n<p")
text.gsub!(/<\/?(label|span|div|b)(.*?)?>/, '')
meta = {}
metapatterns = detect_tags ? REQUIRED_META.merge(OPTIONAL_META) : REQUIRED_META
is_tag = {}.tap do |h|
%w[fandom_string relationship_string freeform_string rating_string archive_warning_string].each do |c|
h[c.to_sym] = true
end
end
handler = {}.tap do |h|
%w[rating_string revised_at].each do |c|
h[c.to_sym] = "convert_#{c.to_s.downcase}"
end
end
# 1. Look for Pattern: (whatever), optionally followed by a closing p or div tag
# 2. Set meta[:metaname] = whatever
# eg, if it finds Fandom: Stargate SG-1 it will set meta[:fandom] = Stargate SG-1
# 3. convert_<metaname> for cleanup if such a function is defined (eg convert_rating_string)
metapatterns.each do |metaname, pattern|
metapattern = Regexp.new("(?:#{pattern}|#{pattern.pluralize})\s*:\s*(.*?)(?:</(?:p|div)>)?$", Regexp::IGNORECASE)
if text.match(metapattern)
value = Regexp.last_match[1]
value = clean_tags(value) if is_tag[metaname]
value = clean_close_html_tags(value)
value.strip! # lose leading/trailing whitespace
value = send(handler[metaname], value) if handler[metaname]
meta[metaname] = value
end
end
post_process_meta meta
end
def download_with_timeout(location, limit = 10)
story = +""
Timeout.timeout(STORY_DOWNLOAD_TIMEOUT) do
begin
# we do a little cleanup here in case the user hasn't included the 'http://'
# or if they've used capital letters or an underscore in the hostname
uri = UrlFormatter.new(location).standardized
raise Error, I18n.t("story_parser.on_archive") if ArchiveConfig.PERMITTED_HOSTS.include?(uri.host)
env_proxy = ENV["http_proxy"]
http = if env_proxy
proxy = URI(env_proxy)
Net::HTTP.new(uri.hostname, uri.port, proxy.hostname, proxy.port)
else
Net::HTTP.new(uri.hostname, uri.port)
end
http.use_ssl = true if uri.scheme == "https"
response = http.start { |h| h.request_get(uri.path.presence || "/") }
case response
when Net::HTTPSuccess
story = response.body
when Net::HTTPRedirection
if limit.positive?
new_uri = URI.parse(response["location"])
new_uri = URI.join(uri, new_uri) if new_uri.relative?
story = download_with_timeout(new_uri.to_s, limit - 1)
end
else
Rails.logger.error("------- STORY PARSER: download_with_timeout: response is not success or redirection ------")
nil
end
rescue Errno::ECONNREFUSED, SocketError, EOFError => e
Rails.logger.error("------- STORY PARSER: download_with_timeout: error rescue: \n#{e.inspect} ------")
nil
end
end
if story.blank?
raise Error, "We couldn't download anything from #{location}. Please make sure that the URL is correct and complete, and try again."
end
# clean up any erroneously included string terminator (AO3-2251)
story.delete("\000")
end
def get_last_modified(location)
Timeout.timeout(STORY_DOWNLOAD_TIMEOUT) do
resp = open(location)
resp.last_modified
end
end
def get_source_if_known(known_sources, location)
known_sources.each do |source|
pattern = Regexp.new(eval("SOURCE_#{source.upcase}"), Regexp::IGNORECASE)
return source if location.match(pattern)
end
nil
end
def clean_close_html_tags(value)
# if there are any closing html tags at the start of the value let's ditch them
value.gsub(/^(\s*<\/[^>]+>)+/, '')
end
# We clean the text as if it had been submitted as the content of a chapter
def clean_storytext(storytext, field = "content")
storytext = storytext.encode("UTF-8", invalid: :replace, undef: :replace, replace: "") unless storytext.encoding.name == "UTF-8"
sanitize_value(field, storytext)
end
# works conservatively -- doesn't split on
# spaces and truncates instead.
def clean_tags(tags)
tags = Sanitize.clean(tags.force_encoding("UTF-8")) # no html allowed in tags
tags_list = tags =~ /,/ ? tags.split(/,/) : [tags]
new_list = []
tags_list.each do |tag|
tag.gsub!(/[*<>]/, '')
tag = truncate_on_word_boundary(tag, ArchiveConfig.TAG_MAX)
new_list << tag unless tag.blank?
end
new_list.join(ArchiveConfig.DELIMITER_FOR_OUTPUT)
end
def truncate_on_word_boundary(text, max_length)
return if text.blank?
words = text.split
truncated = words.first
if words.length > 1
words[1..words.length].each do |word|
truncated += " " + word if truncated.length + word.length + 1 <= max_length
end
end
truncated[0..max_length - 1]
end
# convert space-separated tags to comma-separated
def clean_and_split_tags(tags)
tags = tags.split(/\s+/).join(',') if !tags.match(/,/) && tags.match(/\s/)
clean_tags(tags)
end
# Convert the common ratings into whatever ratings we're
# using on this archive.
def convert_rating_string(rating)
rating = rating.downcase
if rating =~ /^(nc-?1[78]|x|ma|explicit)/
ArchiveConfig.RATING_EXPLICIT_TAG_NAME
elsif rating =~ /^(r|m|mature)/
ArchiveConfig.RATING_MATURE_TAG_NAME
elsif rating =~ /^(pg-?1[35]|t|teen)/
ArchiveConfig.RATING_TEEN_TAG_NAME
elsif rating =~ /^(pg|g|k+|k|general audiences)/
ArchiveConfig.RATING_GENERAL_TAG_NAME
else
ArchiveConfig.RATING_DEFAULT_TAG_NAME
end
end
def convert_revised_at(date_string)
begin
date = nil
if date_string =~ /^(\d+)$/
# probably seconds since the epoch
date = Time.at(Regex.last_match[1].to_i)
end
date ||= Date.parse(date_string)
return '' if date > Date.current
return date
rescue ArgumentError, TypeError
return ''
end
end
# Additional processing for meta - currently to make sure warnings
# that aren't Archive warnings become additional tags instead
def post_process_meta(meta)
if meta[:archive_warning_string]
result = process_warnings(meta[:archive_warning_string], meta[:freeform_string])
meta[:archive_warning_string] = result[:archive_warning_string]
meta[:freeform_string] = result[:freeform_string]
end
meta
end
def process_warnings(warning_string, freeform_string)
result = {
archive_warning_string: warning_string,
freeform_string: freeform_string
}
new_warning = ''
result[:archive_warning_string].split(/\s?,\s?/).each do |warning|
if ArchiveWarning.warning? warning
new_warning += ', ' unless new_warning.blank?
new_warning += warning
else
result[:freeform_string] = (result[:freeform_string] || '') + ", #{warning}"
end
end
result[:archive_warning_string] = new_warning
result
end
# tries to find appropriate existing collections and converts them to comma-separated collection names only
def get_collection_names(collection_string)
collections = ""
collection_string.split(',').map(&:squish).each do |collection_name|
collection = Collection.find_by(name: collection_name) || Collection.find_by(title: collection_name)
if collection
collections += ", " unless collections.blank?
collections += collection.name
end
end
collections
end
# determine which value to use for a metadata field
def meta_or_default(detected_field, provided_field, default = nil)
if @options[:override_tags] || detected_field.blank?
if provided_field.blank?
detected_field.blank? ? default : detected_field
else
provided_field
end
else
detected_field
end
end
end
| 1 | # Parse stories from other websites and uploaded files, looking for metadata to harvest |
| 2 | # and put into the archive. |
| 3 | # |
| 4 | class StoryParser |
| 5 | require 'timeout' |
| 6 | require 'nokogiri' |
| 7 | require 'mechanize' |
| 8 | require 'open-uri' |
| 9 | include HtmlCleaner |
| 10 | |
| 11 | OPTIONAL_META = {notes: 'Note', |
| 12 | freeform_string: 'Tag', |
| 13 | fandom_string: 'Fandom', |
| 14 | rating_string: 'Rating', |
| 15 | archive_warning_string: 'Warning', |
| 16 | relationship_string: 'Relationship|Pairing', |
| 17 | character_string: 'Character' }.freeze |
| 18 | REQUIRED_META = { title: 'Title', |
| 19 | summary: 'Summary', |
| 20 | revised_at: 'Date|Posted|Posted on|Posted at', |
| 21 | chapter_title: 'Chapter Title' }.freeze |
| 22 | |
| 23 | # Use this for raising custom error messages |
| 24 | # (so that we can distinguish them from unexpected exceptions due to |
| 25 | # faulty code) |
| 26 | class Error < StandardError |
| 27 | end |
| 28 | |
| 29 | # These attributes need to be moved from the work to the chapter |
| 30 | # format: {work_attribute_name: :chapter_attribute_name} (can be the same) |
| 31 | CHAPTER_ATTRIBUTES_ONLY = {} |
| 32 | |
| 33 | # These attributes need to be copied from the work to the chapter |
| 34 | CHAPTER_ATTRIBUTES_ALSO = { revised_at: :published_at }.freeze |
| 35 | |
| 36 | ### NOTE ON KNOWN SOURCES |
| 37 | # These lists will stop with the first one it matches, so put more-specific matches |
| 38 | # towards the front of the list. |
| 39 | |
| 40 | # places for which we have a custom parse_story_from_[source] method |
| 41 | # for getting information out of the downloaded text |
| 42 | KNOWN_STORY_PARSERS = %w[ao3 deviantart dw lj].freeze |
| 43 | |
| 44 | # places for which we have a custom parse_author_from_[source] method |
| 45 | # which returns an external_author object including an email address |
| 46 | KNOWN_AUTHOR_PARSERS = %w[lj].freeze |
| 47 | |
| 48 | # places for which we have a download_story_from_[source] |
| 49 | # used to customize the downloading process |
| 50 | KNOWN_STORY_LOCATIONS = %w[lj].freeze |
| 51 | |
| 52 | # places for which we have a download_chaptered_from |
| 53 | # to get a set of chapters all together |
| 54 | CHAPTERED_STORY_LOCATIONS = %w[ffnet thearchive_net efiction quotev].freeze |
| 55 | |
| 56 | # regular expressions to match against the URLS |
| 57 | SOURCE_AO3 = '(archiveofourown\.org|ao3\.org|superlove\.sayitditto\.net|sunset\.femslash\.club)'.freeze |
| 58 | SOURCE_LJ = '((live|dead|insane)journal\.com)|journalfen(\.net|\.com)|dreamwidth\.org'.freeze |
| 59 | SOURCE_DW = 'dreamwidth\.org'.freeze |
| 60 | SOURCE_FFNET = '(^|[^A-Za-z0-9-])fanfiction\.net'.freeze |
| 61 | SOURCE_DEVIANTART = 'deviantart\.com'.freeze |
| 62 | SOURCE_THEARCHIVE_NET = 'the\-archive\.net'.freeze |
| 63 | SOURCE_EFICTION = 'viewstory\.php'.freeze |
| 64 | SOURCE_QUOTEV = 'quotev\.com'.freeze |
| 65 | |
| 66 | # time out if we can't download fast enough |
| 67 | STORY_DOWNLOAD_TIMEOUT = 60 |
| 68 | MAX_CHAPTER_COUNT = 200 |
| 69 | |
| 70 | # To check for duplicate chapters, take a slice this long out of the story |
| 71 | # (in characters) |
| 72 | DUPLICATE_CHAPTER_LENGTH = 10_000 |
| 73 | |
| 74 | |
| 75 | # Import many stories |
| 76 | def import_many(urls, options = {}) |
| 77 | # Try to get the works |
| 78 | works = [] |
| 79 | failed_urls = [] |
| 80 | errors = [] |
| 81 | @options = options |
| 82 | urls.each do |url| |
| 83 | begin |
| 84 | response = download_and_parse_work(url, options) |
| 85 | work = response[:work] |
| 86 | if response[:status] == :created |
| 87 | if work && work.save |
| 88 | work.chapters.each(&:save) |
| 89 | works << work |
| 90 | else |
| 91 | failed_urls << url |
| 92 | errors << work.errors.values.join(", ") |
| 93 | work.delete if work |
| 94 | end |
| 95 | elsif response[:status] == :already_imported |
| 96 | raise StoryParser::Error, response[:message] |
| 97 | end |
| 98 | rescue Timeout::Error |
| 99 | failed_urls << url |
| 100 | errors << "Import has timed out. This may be due to connectivity problems with the source site. Please try again in a few minutes, or check Known Issues to see if there are import problems with this site." |
| 101 | work.delete if work |
| 102 | rescue Error => exception |
| 103 | failed_urls << url |
| 104 | errors << "We couldn't successfully import that work, sorry: #{exception.message}" |
| 105 | work.delete if work |
| 106 | end |
| 107 | end |
| 108 | [works, failed_urls, errors] |
| 109 | end |
| 110 | |
| 111 | # Downloads a story and passes it on to the parser. |
| 112 | # If the URL of the story is from a site for which we have special rules |
| 113 | # (eg, downloading from a livejournal clone, you want to use ?format=light |
| 114 | # to get a nice and consistent post format), it will pre-process the url |
| 115 | # according to the rules for that site. |
| 116 | def download_and_parse_work(location, options = {}) |
| 117 | status = :created |
| 118 | message = "" |
| 119 | work = Work.find_by_url(location) |
| 120 | if work.nil? |
| 121 | @options = options |
| 122 | source = get_source_if_known(CHAPTERED_STORY_LOCATIONS, location) |
| 123 | if source.nil? |
| 124 | story = download_text(location) |
| 125 | work = parse_story(story, location, options) |
| 126 | else |
| 127 | work = download_and_parse_chaptered_story(source, location, options) |
| 128 | end |
| 129 | else |
| 130 | status = :already_imported |
| 131 | message = "A work has already been imported from #{location}." |
| 132 | end |
| 133 | { |
| 134 | status: status, |
| 135 | message: message, |
| 136 | work: work |
| 137 | } |
| 138 | end |
| 139 | |
| 140 | # Given an array of urls for chapters of a single story, |
| 141 | # download them all and combine into a single work |
| 142 | def import_chapters_into_story(locations, options = {}) |
| 143 | status = :created |
| 144 | work = Work.find_by_url(locations.first) |
| 145 | if work.nil? |
| 146 | chapter_contents = [] |
| 147 | @options = options |
| 148 | locations.each do |location| |
| 149 | chapter_contents << download_text(location) |
| 150 | end |
| 151 | work = parse_chapters_into_story(locations.first, chapter_contents, options) |
| 152 | message = "Successfully created work \"" + work.title + "\"." |
| 153 | else |
| 154 | status = :already_imported |
| 155 | message = "A work has already been imported from #{locations.first}." |
| 156 | end |
| 157 | { |
| 158 | status: status, |
| 159 | message: message, |
| 160 | work: work |
| 161 | } |
| 162 | end |
| 163 | |
| 164 | |
| 165 | ### OLD PARSING METHODS |
| 166 | |
| 167 | # Import many stories |
| 168 | def import_from_urls(urls, options = {}) |
| 169 | # Try to get the works |
| 170 | works = [] |
| 171 | failed_urls = [] |
| 172 | errors = [] |
| 173 | @options = options |
| 174 | urls.each do |url| |
| 175 | begin |
| 176 | work = download_and_parse_story(url, options) |
| 177 | if work && work.save |
| 178 | work.chapters.each(&:save) |
| 179 | works << work |
| 180 | else |
| 181 | failed_urls << url |
| 182 | errors << work.errors.values.join(", ") |
| 183 | work.delete if work |
| 184 | end |
| 185 | rescue Timeout::Error |
| 186 | failed_urls << url |
| 187 | errors << "Import has timed out. This may be due to connectivity problems with the source site. Please try again in a few minutes, or check Known Issues to see if there are import problems with this site." |
| 188 | work.delete if work |
| 189 | rescue Error => exception |
| 190 | failed_urls << url |
| 191 | errors << "We couldn't successfully import that work, sorry: #{exception.message}" |
| 192 | work.delete if work |
| 193 | end |
| 194 | end |
| 195 | [works, failed_urls, errors] |
| 196 | end |
| 197 | |
| 198 | # Downloads a story and passes it on to the parser. |
| 199 | # If the URL of the story is from a site for which we have special rules |
| 200 | # (eg, downloading from a livejournal clone, you want to use ?format=light |
| 201 | # to get a nice and consistent post format), it will pre-process the url |
| 202 | # according to the rules for that site. |
| 203 | def download_and_parse_story(location, options = {}) |
| 204 | check_for_previous_import(location) |
| 205 | @options = options |
| 206 | source = get_source_if_known(CHAPTERED_STORY_LOCATIONS, location) |
| 207 | if source.nil? |
| 208 | story = download_text(location) |
| 209 | work = parse_story(story, location, options) |
| 210 | else |
| 211 | work = download_and_parse_chaptered_story(source, location, options) |
| 212 | end |
| 213 | work |
| 214 | end |
| 215 | |
| 216 | # Given an array of urls for chapters of a single story, |
| 217 | # download them all and combine into a single work |
| 218 | def download_and_parse_chapters_into_story(locations, options = {}) |
| 219 | check_for_previous_import(locations.first) |
| 220 | chapter_contents = [] |
| 221 | @options = options |
| 222 | locations.each do |location| |
| 223 | chapter_contents << download_text(location) |
| 224 | end |
| 225 | parse_chapters_into_story(locations.first, chapter_contents, options) |
| 226 | end |
| 227 | |
| 228 | ### PARSING METHODS |
| 229 | |
| 230 | # Parses the text of a story, optionally from a given location. |
| 231 | def parse_story(story, location, options = {}) |
| 232 | work_params = parse_common(story, location, options[:encoding], options[:detect_tags]) |
| 233 | |
| 234 | # move any attributes from work to chapter if necessary |
| 235 | set_work_attributes(Work.new(work_params), location, options) |
| 236 | end |
| 237 | |
| 238 | # parses and adds a new chapter to the end of the work |
| 239 | def parse_chapter_of_work(work, chapter_content, location, options = {}) |
| 240 | tmp_work_params = parse_common(chapter_content, location, options[:encoding], options[:detect_tags]) |
| 241 | chapter = get_chapter_from_work_params(tmp_work_params) |
| 242 | work.chapters << set_chapter_attributes(work, chapter) |
| 243 | work |
| 244 | end |
| 245 | |
| 246 | def parse_chapters_into_story(location, chapter_contents, options = {}) |
| 247 | work = nil |
| 248 | chapter_contents.each do |content| |
| 249 | work_params = parse_common(content, location, options[:encoding], options[:detect_tags]) |
| 250 | if work.nil? |
| 251 | # create the new work |
| 252 | work = Work.new(work_params) |
| 253 | else |
| 254 | new_chapter = get_chapter_from_work_params(work_params) |
| 255 | work.chapters << set_chapter_attributes(work, new_chapter) |
| 256 | end |
| 257 | end |
| 258 | set_work_attributes(work, location, options) |
| 259 | end |
| 260 | |
| 261 | # Everything below here is protected and should not be touched by outside |
| 262 | # code -- please use the above functions to parse external works. |
| 263 | |
| 264 | protected |
| 265 | |
| 266 | # tries to create an external author for a given url |
| 267 | def parse_author(location, ext_author_name, ext_author_email) |
| 268 | if location.present? && ext_author_name.blank? && ext_author_email.blank? |
| 269 | source = get_source_if_known(KNOWN_AUTHOR_PARSERS, location) |
| 270 | if source.nil? |
| 271 | raise Error, "No external author name or email specified" |
| 272 | else |
| 273 | send("parse_author_from_#{source.downcase}", location) |
| 274 | end |
| 275 | else |
| 276 | parse_author_common(ext_author_email, ext_author_name) |
| 277 | end |
| 278 | end |
| 279 | |
| 280 | # download an entire story from an archive type where we know how to parse multi-chaptered works |
| 281 | # this should only be called from download_and_parse_story |
| 282 | def download_and_parse_chaptered_story(source, location, options = {}) |
| 283 | chapter_contents = send("download_chaptered_from_#{source.downcase}", location) |
| 284 | parse_chapters_into_story(location, chapter_contents, options) |
| 285 | end |
| 286 | |
| 287 | # our custom url finder checks for previously imported URL in almost any format it may have been presented |
| 288 | def check_for_previous_import(location) |
| 289 | if Work.find_by_url(location).present? |
| 290 | raise Error, "A work has already been imported from #{location}." |
| 291 | end |
| 292 | end |
| 293 | |
| 294 | def set_chapter_attributes(work, chapter) |
| 295 | chapter.position = work.chapters.length + 1 |
| 296 | chapter.posted = true |
| 297 | chapter |
| 298 | end |
| 299 | |
| 300 | def set_work_attributes(work, location = "", options = {}) |
| 301 | raise Error, "Work could not be downloaded" if work.nil? |
| 302 | |
| 303 | @options = options |
| 304 | work.imported_from_url = location # @todo remove this as part of AO3-6979 |
| 305 | work.imported_url = ImportedUrl.new(original: work.imported_from_url) |
| 306 | |
| 307 | work.ip_address = options[:ip_address] |
| 308 | work.expected_number_of_chapters = work.chapters.length |
| 309 | work.revised_at = work.chapters.last.published_at |
| 310 | if work.revised_at && work.revised_at.to_date < Date.current |
| 311 | work.backdate = true |
| 312 | end |
| 313 | |
| 314 | # set authors for the works |
| 315 | pseuds = [] |
| 316 | pseuds << User.current_user.default_pseud unless options[:do_not_set_current_author] || User.current_user.nil? |
| 317 | pseuds << options[:archivist].default_pseud if options[:archivist] |
| 318 | pseuds << options[:pseuds] if options[:pseuds] |
| 319 | pseuds = pseuds.flatten.compact.uniq |
| 320 | raise Error, "A work must have at least one author specified" if pseuds.empty? |
| 321 | pseuds.each do |pseud| |
| 322 | work.creatorships.build(pseud: pseud, enable_notifications: true) |
| 323 | work.chapters.each do |chapter| |
| 324 | chapter.creatorships.build(pseud: pseud) |
| 325 | end |
| 326 | end |
| 327 | |
| 328 | # handle importing works for others |
| 329 | # build an external creatorship for each author |
| 330 | if options[:importing_for_others] |
| 331 | external_author_names = options[:external_author_names] || parse_author(location, options[:external_author_name], options[:external_author_email]) |
| 332 | # convert to an array if not already one |
| 333 | external_author_names = [external_author_names] if external_author_names.is_a?(ExternalAuthorName) |
| 334 | if options[:external_coauthor_name].present? |
| 335 | external_author_names << parse_author(location, options[:external_coauthor_name], options[:external_coauthor_email]) |
| 336 | end |
| 337 | external_author_names.each do |external_author_name| |
| 338 | next if !external_author_name || external_author_name.external_author.blank? |
| 339 | if external_author_name.external_author.do_not_import |
| 340 | # we're not allowed to import works from this address |
| 341 | raise Error, "Author #{external_author_name.name} at #{external_author_name.external_author.email} does not allow importing their work to this archive." |
| 342 | end |
| 343 | work.external_creatorships.build(external_author_name: external_author_name, archivist: (options[:archivist] || User.current_user)) |
| 344 | end |
| 345 | end |
| 346 | |
| 347 | # lock to registered users if specified or importing for others |
| 348 | work.restricted = options[:restricted] || options[:importing_for_others] || false |
| 349 | |
| 350 | # set comment permissions |
| 351 | work.comment_permissions = options[:comment_permissions] || "enable_all" |
| 352 | work.moderated_commenting_enabled = options[:moderated_commenting_enabled] || false |
| 353 | |
| 354 | # set default values for required tags |
| 355 | work.fandom_string = meta_or_default(work.fandom_string, options[:fandom], ArchiveConfig.FANDOM_NO_TAG_NAME) |
| 356 | work.rating_string = meta_or_default(work.rating_string, options[:rating], ArchiveConfig.RATING_DEFAULT_TAG_NAME) |
| 357 | work.archive_warning_strings = meta_or_default(work.archive_warning_strings, options[:archive_warning], ArchiveConfig.WARNING_DEFAULT_TAG_NAME) |
| 358 | work.category_string = meta_or_default(work.category_string, options[:category], []) |
| 359 | work.character_string = meta_or_default(work.character_string, options[:character], []) |
| 360 | work.relationship_string = meta_or_default(work.relationship_string, options[:relationship], []) |
| 361 | work.freeform_string = meta_or_default(work.freeform_string, options[:freeform], []) |
| 362 | |
| 363 | # set default value for title |
| 364 | work.title = meta_or_default(work.title, options[:title], "Untitled Imported Work") |
| 365 | work.summary = meta_or_default(work.summary, options[:summary], '') |
| 366 | work.notes = meta_or_default(work.notes, options[:notes], '') |
| 367 | |
| 368 | # set collection name if present |
| 369 | work.collection_names = get_collection_names(options[:collection_names]) if options[:collection_names].present? |
| 370 | |
| 371 | # set default language (English) |
| 372 | work.language_id = options[:language_id] || Language.default.id |
| 373 | |
| 374 | work.posted = true if options[:post_without_preview] |
| 375 | work.chapters.each do |chapter| |
| 376 | if chapter.content.length > ArchiveConfig.CONTENT_MAX |
| 377 | # TODO: eventually: insert a new chapter |
| 378 | chapter.content.truncate(ArchiveConfig.CONTENT_MAX, omission: "<strong>WARNING: import truncated automatically because chapter was too long! Please add a new chapter for remaining content.</strong>", separator: "</p>") |
| 379 | elsif chapter.content.empty? |
| 380 | raise Error, "Chapter #{chapter.position} of \"#{work.title}\" is blank." |
| 381 | end |
| 382 | |
| 383 | chapter.posted = true # do not save - causes the chapters to exist even if work doesn't get created! |
| 384 | end |
| 385 | work |
| 386 | end |
| 387 | |
| 388 | def parse_author_from_lj(location) |
| 389 | return if location !~ %r{^(?:http:\/\/)?(?<lj_name>[^.]*).(?<site_name>livejournal\.com|dreamwidth\.org|insanejournal\.com|journalfen.net)} |
| 390 | email = "" |
| 391 | lj_name = Regexp.last_match[:lj_name] |
| 392 | site_name = Regexp.last_match[:site_name] |
| 393 | if lj_name == "community" |
| 394 | # whups |
| 395 | post_text = download_text(location) |
| 396 | doc = Nokogiri.parse(post_text) |
| 397 | lj_name = doc.xpath("/html/body/div[2]/div/div/div/table/tbody/tr/td[2]/span/a[2]/b").content |
| 398 | end |
| 399 | profile_url = "http://#{lj_name}.#{site_name}/profile" |
| 400 | lj_profile = download_text(profile_url) |
| 401 | doc = Nokogiri.parse(lj_profile) |
| 402 | contact = doc.css('div.contact').inner_html |
| 403 | if contact.present? |
| 404 | contact.gsub! '<p class="section_body_title">Contact:</p>', "" |
| 405 | contact.gsub! /<\/?(span|i)>/, "" |
| 406 | contact.delete! "\n" |
| 407 | contact.gsub! "<br/>", "" |
| 408 | if contact =~ /(.*@.*\..*)/ |
| 409 | email = Regexp.last_match[1] |
| 410 | end |
| 411 | end |
| 412 | email = "#{lj_name}@#{site_name}" if email.blank? |
| 413 | parse_author_common(email, lj_name) |
| 414 | end |
| 415 | |
| 416 | def parse_author_from_unknown(_location) |
| 417 | # for now, nothing |
| 418 | nil |
| 419 | end |
| 420 | |
| 421 | def parse_author_common(email, name) |
| 422 | errors = [] |
| 423 | |
| 424 | errors << "No author name specified" if name.blank? |
| 425 | |
| 426 | if email.present? |
| 427 | external_author = ExternalAuthor.find_or_create_by(email: email) |
| 428 | errors += external_author.errors.full_messages |
| 429 | else |
| 430 | errors << "No author email specified" |
| 431 | end |
| 432 | |
| 433 | raise Error, errors.join("\n") if errors.present? |
| 434 | |
| 435 | # convert to ASCII and strip out invalid characters (everything except alphanumeric characters, _, @ and -) |
| 436 | redacted_name = name.to_ascii.gsub(/[^\w[ \-@.]]/u, "") |
| 437 | if redacted_name.present? |
| 438 | external_author.names.find_or_create_by(name: redacted_name) |
| 439 | else |
| 440 | external_author.default_name |
| 441 | end |
| 442 | end |
| 443 | |
| 444 | def get_chapter_from_work_params(work_params) |
| 445 | @chapter = Chapter.new(work_params[:chapter_attributes]) |
| 446 | # don't override specific chapter params (eg title) with work params |
| 447 | chapter_params = work_params.delete_if do |name, _param| |
| 448 | !@chapter.attribute_names.include?(name.to_s) || !@chapter.send(name.to_s).blank? |
| 449 | end |
| 450 | @chapter.update(chapter_params) |
| 451 | @chapter |
| 452 | end |
| 453 | |
| 454 | def download_text(location) |
| 455 | source = get_source_if_known(KNOWN_STORY_LOCATIONS, location) |
| 456 | if source.nil? |
| 457 | download_with_timeout(location) |
| 458 | else |
| 459 | send("download_from_#{source.downcase}", location) |
| 460 | end |
| 461 | end |
| 462 | |
| 463 | # canonicalize the url for downloading from lj or clones |
| 464 | def download_from_lj(location) |
| 465 | url = location |
| 466 | url.gsub!(/\#(.*)$/, "") # strip off any anchor information |
| 467 | url.gsub!(/\?(.*)$/, "") # strip off any existing params at the end |
| 468 | url.gsub!('_', '-') # convert underscores in usernames to hyphens |
| 469 | url += "?format=light" # go to light format |
| 470 | text = download_with_timeout(url) |
| 471 | |
| 472 | if text.match(/adult_check/) |
| 473 | Timeout::timeout(STORY_DOWNLOAD_TIMEOUT) { |
| 474 | begin |
| 475 | agent = Mechanize.new |
| 476 | url.include?("dreamwidth") ? form = agent.get(url).forms.first : form = agent.get(url).forms.third |
| 477 | page = agent.submit(form, form.buttons.first) # submits the adult concepts form |
| 478 | text = page.body.force_encoding(agent.page.encoding) |
| 479 | rescue |
| 480 | text = "" |
| 481 | end |
| 482 | } |
| 483 | end |
| 484 | text |
| 485 | end |
| 486 | |
| 487 | # grab all the chapters of the story from ff.net |
| 488 | def download_chaptered_from_ffnet(_location) |
| 489 | raise Error, "Sorry, Fanfiction.net does not allow imports from their site." |
| 490 | end |
| 491 | |
| 492 | def download_chaptered_from_quotev(_location) |
| 493 | raise Error, "Sorry, Quotev.com does not allow imports from their site." |
| 494 | end |
| 495 | |
| 496 | # this is an efiction archive but it doesn't handle chapters normally |
| 497 | # best way to handle is to get the full story printable version |
| 498 | # We have to make it a download-chaptered because otherwise it gets sent to the |
| 499 | # generic efiction version since chaptered sources are checked first |
| 500 | def download_chaptered_from_thearchive_net(location) |
| 501 | if location.match(/^(.*)\/.*viewstory\.php.*[^p]sid=(\d+)($|&)/i) |
| 502 | location = "#{$1}/viewstory.php?action=printable&psid=#{$2}" |
| 503 | end |
| 504 | text = download_with_timeout(location) |
| 505 | text.sub!('</style>', '</style></head>') unless text.match('</head>') |
| 506 | [text] |
| 507 | end |
| 508 | |
| 509 | # grab all the chapters of a story from an efiction-based site |
| 510 | def download_chaptered_from_efiction(location) |
| 511 | chapter_contents = [] |
| 512 | if location.match(/^(?<site>.*)\/.*viewstory\.php.*sid=(?<storyid>\d+)($|&)/i) |
| 513 | site = Regexp.last_match[:site] |
| 514 | storyid = Regexp.last_match[:storyid] |
| 515 | chapnum = 1 |
| 516 | last_body = "" |
| 517 | Timeout::timeout(STORY_DOWNLOAD_TIMEOUT) do |
| 518 | loop do |
| 519 | url = "#{site}/viewstory.php?action=printable&sid=#{storyid}&chapter=#{chapnum}" |
| 520 | body = download_with_timeout(url) |
| 521 | # get a section to check that this isn't a duplicate of previous chapter |
| 522 | body_to_check = body.slice(10, DUPLICATE_CHAPTER_LENGTH) |
| 523 | if body.nil? || body_to_check == last_body || chapnum > MAX_CHAPTER_COUNT || body.match(/<div class='chaptertitle'> by <\/div>/) || body.match(/Access denied./) || body.match(/Chapter : /) |
| 524 | break |
| 525 | end |
| 526 | # save the value to check for duplicate chapter |
| 527 | last_body = body_to_check |
| 528 | |
| 529 | # clean up the broken head in many efiction printable sites |
| 530 | body.sub!('</style>', '</style></head>') unless body.match('</head>') |
| 531 | chapter_contents << body |
| 532 | chapnum += 1 |
| 533 | end |
| 534 | end |
| 535 | end |
| 536 | chapter_contents |
| 537 | end |
| 538 | |
| 539 | |
| 540 | # This is the heavy lifter, invoked by all the story and chapter parsers. |
| 541 | # It takes a single string containing the raw contents of a story, parses it with |
| 542 | # Nokogiri into the @doc object, and then and calls a subparser. |
| 543 | # |
| 544 | # If the story source can be identified as one of the sources we know how to parse in some custom/ |
| 545 | # special way, parse_common calls the customized parse_story_from_[source] method. |
| 546 | # Otherwise, it falls back to parse_story_from_unknown. |
| 547 | # |
| 548 | # This produces a hash equivalent to the params hash that is normally created by the standard work |
| 549 | # upload form. |
| 550 | # |
| 551 | # parse_common then calls sanitize_params (which would also be called on the standard work upload |
| 552 | # form results) and returns the final sanitized hash. |
| 553 | # |
| 554 | def parse_common(story, location = nil, encoding = nil, detect_tags = true) |
| 555 | work_params = { title: "Untitled Imported Work", chapter_attributes: { content: "" } } |
| 556 | |
| 557 | # Encode as HTML - the dummy "foo" tag will be stripped out by the sanitizer but forces Nokogiri to |
| 558 | # preserve line breaks in plain text documents |
| 559 | # Rescue all errors as Nokogiri complains about things the sanitizer will fix later |
| 560 | story.prepend("<foo></foo>") |
| 561 | @doc = |
| 562 | begin |
| 563 | Nokogiri::HTML5.parse(story, encoding: encoding) |
| 564 | rescue StandardError |
| 565 | Nokogiri::HTML5.parse("") |
| 566 | end |
| 567 | |
| 568 | # Try to convert all relative links to absolute |
| 569 | base = @doc.at_css("base") ? @doc.css("base")[0]["href"] : location.split("?").first |
| 570 | if base.present? |
| 571 | @doc.css("a").each do |link| |
| 572 | next if link["href"].blank? || link["href"].start_with?("#") |
| 573 | begin |
| 574 | query = link["href"].match(/(\?.*)$/) ? $1 : "" |
| 575 | link["href"] = URI.join(base, link["href"].gsub(/(\?.*)$/, "")).to_s + query |
| 576 | rescue |
| 577 | # ignored |
| 578 | end |
| 579 | end |
| 580 | end |
| 581 | |
| 582 | # Extract metadata (unless detect_tags is false) |
| 583 | if location && (source = get_source_if_known(KNOWN_STORY_PARSERS, location)) |
| 584 | params = send("parse_story_from_#{source.downcase}", story, detect_tags) |
| 585 | work_params.merge!(params) |
| 586 | else |
| 587 | work_params.merge!(parse_story_from_unknown(story, detect_tags)) |
| 588 | end |
| 589 | |
| 590 | shift_chapter_attributes(sanitize_params(work_params)) |
| 591 | end |
| 592 | |
| 593 | # our fallback: parse a story from an unknown source, so we have no special |
| 594 | # rules. |
| 595 | def parse_story_from_unknown(story, detect_tags = true) |
| 596 | work_params = { chapter_attributes: {} } |
| 597 | story_head = "" |
| 598 | story_head = @doc.css("head").inner_html if @doc.css("head") |
| 599 | |
| 600 | # Story content - Look for progressively less specific containers or grab everything |
| 601 | element = @doc.at_css('.chapter-content') || @doc.at_css('body') || @doc.at_css('html') || @doc |
| 602 | storytext = element ? element.inner_html : story |
| 603 | |
| 604 | meta = {} |
| 605 | meta.merge!(scan_text_for_meta(story_head, detect_tags)) unless story_head.blank? |
| 606 | meta.merge!(scan_text_for_meta(story, detect_tags)) |
| 607 | meta[:title] ||= @doc.css('title').inner_html |
| 608 | work_params[:chapter_attributes][:title] = meta.delete(:chapter_title) |
| 609 | work_params[:chapter_attributes][:content] = clean_storytext(storytext) |
| 610 | work_params.merge!(meta) |
| 611 | end |
| 612 | |
| 613 | # Parses a story from livejournal or a livejournal equivalent (eg, dreamwidth, insanejournal) |
| 614 | # Assumes that we have downloaded the story from one of those equivalents (ie, we've downloaded |
| 615 | # it in format=light which is a stripped-down plaintext version.) |
| 616 | # |
| 617 | def parse_story_from_lj(_story, detect_tags = true) |
| 618 | work_params = { chapter_attributes: {} } |
| 619 | |
| 620 | # in LJ "light" format, the story contents are in the second div |
| 621 | # inside the body. |
| 622 | body = @doc.css("body") |
| 623 | storytext = body.css("article.b-singlepost-body").inner_html |
| 624 | storytext = body.css("div.aentry-post__text").inner_html if storytext.empty? |
| 625 | storytext = body.inner_html if storytext.empty? |
| 626 | |
| 627 | # cleanup the text |
| 628 | # storytext.gsub!(/<br\s*\/?>/i, "\n") # replace the breaks with newlines |
| 629 | storytext = clean_storytext(storytext) |
| 630 | |
| 631 | work_params[:chapter_attributes][:content] = storytext |
| 632 | work_params[:title] = @doc.css("title").inner_html |
| 633 | work_params[:title].gsub! /^[^:]+: /, "" |
| 634 | work_params.merge!(scan_text_for_meta(storytext, detect_tags)) |
| 635 | |
| 636 | date = @doc.css("time.b-singlepost-author-date") |
| 637 | date = @doc.css("p.aentry-head__date/time") if date.empty? |
| 638 | work_params[:revised_at] = convert_revised_at(date.first.inner_text) unless date.empty? |
| 639 | |
| 640 | work_params |
| 641 | end |
| 642 | |
| 643 | def parse_story_from_dw(_story, detect_tags = true) |
| 644 | work_params = { chapter_attributes: {} } |
| 645 | |
| 646 | body = @doc.css("body") |
| 647 | content_divs = body.css("div.contents") |
| 648 | |
| 649 | if content_divs[0].present? |
| 650 | # Get rid of the DW metadata table |
| 651 | content_divs[0].css("div.currents, ul.entry-management-links, div.header.inner, span.restrictions, h3.entry-title").each(&:remove) |
| 652 | storytext = content_divs[0].inner_html |
| 653 | else |
| 654 | storytext = body.inner_html |
| 655 | end |
| 656 | |
| 657 | # cleanup the text |
| 658 | storytext = clean_storytext(storytext) |
| 659 | |
| 660 | work_params[:chapter_attributes][:content] = storytext |
| 661 | work_params[:title] = @doc.css("title").inner_html |
| 662 | work_params[:title].gsub! /^[^:]+: /, "" |
| 663 | work_params.merge!(scan_text_for_meta(storytext, detect_tags)) |
| 664 | |
| 665 | font_blocks = @doc.xpath('//font') |
| 666 | unless font_blocks.empty? |
| 667 | date = font_blocks.first.inner_text |
| 668 | work_params[:revised_at] = convert_revised_at(date) |
| 669 | end |
| 670 | |
| 671 | # get the date |
| 672 | date = @doc.css("span.date").inner_text |
| 673 | work_params[:revised_at] = convert_revised_at(date) |
| 674 | |
| 675 | work_params |
| 676 | end |
| 677 | |
| 678 | def parse_story_from_deviantart(_story, detect_tags = true) |
| 679 | work_params = { chapter_attributes: {} } |
| 680 | storytext = "" |
| 681 | notes = "" |
| 682 | |
| 683 | body = @doc.css("body") |
| 684 | title = @doc.css("title").inner_html.gsub /\s*on deviantart$/i, "" |
| 685 | |
| 686 | # Find the image (original size) if it's art |
| 687 | image_full = body.css("div.dev-view-deviation img.dev-content-full") |
| 688 | unless image_full[0].nil? |
| 689 | storytext = "<center><img src=\"#{image_full[0]["src"]}\"></center>" |
| 690 | end |
| 691 | |
| 692 | # Find the fic text if it's fic (needs the id for disambiguation, the "deviantART loves you" bit in the footer has the same class path) |
| 693 | text_table = body.css(".grf-indent > div:nth-child(1)")[0] |
| 694 | unless text_table.nil? |
| 695 | # Try to remove some metadata (title and author) from the work's text, if possible |
| 696 | # Try to remove the title: if it exists, and if it's the same as the browser title |
| 697 | if text_table.css("h1")[0].present? && title && title.match(text_table.css("h1")[0].text) |
| 698 | text_table.css("h1")[0].remove |
| 699 | end |
| 700 | |
| 701 | # Try to remove the author: if it exists, and if it follows a certain pattern |
| 702 | if text_table.css("small")[0].present? && text_table.css("small")[0].inner_html.match(/by ~.*?<a class="u" href=/m) |
| 703 | text_table.css("small")[0].remove |
| 704 | end |
| 705 | storytext = text_table.inner_html |
| 706 | end |
| 707 | |
| 708 | # cleanup the text |
| 709 | storytext.gsub!(%r{<br\s*\/?>}i, "\n") # replace the breaks with newlines |
| 710 | storytext = clean_storytext(storytext) |
| 711 | work_params[:chapter_attributes][:content] = storytext |
| 712 | |
| 713 | # Find the notes |
| 714 | content_divs = body.css("div.text-ctrl div.text") |
| 715 | notes = content_divs[0].inner_html unless content_divs[0].nil? |
| 716 | |
| 717 | # cleanup the notes |
| 718 | notes.gsub!(%r{<br\s*\/?>}i, "\n") # replace the breaks with newlines |
| 719 | notes = clean_storytext(notes, "notes") |
| 720 | work_params[:notes] = notes |
| 721 | |
| 722 | work_params.merge!(scan_text_for_meta(notes, detect_tags)) |
| 723 | work_params[:title] = title |
| 724 | |
| 725 | body.css("div.dev-title-container h1 a").each do |node| |
| 726 | if node["class"] != "u" |
| 727 | work_params[:title] = node.inner_html |
| 728 | end |
| 729 | end |
| 730 | |
| 731 | tags = [] |
| 732 | @doc.css("div.dev-about-cat-cc a.h").each { |node| tags << node.inner_html } |
| 733 | work_params[:freeform_string] = clean_tags(tags.join(ArchiveConfig.DELIMITER_FOR_OUTPUT)) |
| 734 | |
| 735 | details = @doc.css("div.dev-right-bar-content span[title]") |
| 736 | unless details[0].nil? |
| 737 | work_params[:revised_at] = convert_revised_at(details[0].inner_text) |
| 738 | end |
| 739 | |
| 740 | work_params |
| 741 | end |
| 742 | |
| 743 | def parse_story_from_ao3(_story, detect_tags = true) |
| 744 | work_params = { chapter_attributes: {} } |
| 745 | |
| 746 | # Title: use the work heading, not the browser tab title |
| 747 | title_node = @doc.at_css('h2.title.heading') |
| 748 | work_params[:title] = if title_node |
| 749 | title_node.inner_text.strip |
| 750 | else |
| 751 | @doc.at_css('title')&.inner_text&.sub(/\s*\[Archive of Our Own\]\s*$/i, '')&.strip.to_s |
| 752 | end |
| 753 | |
| 754 | # Summary |
| 755 | summary_node = @doc.at_css('.summary.module blockquote.userstuff') |
| 756 | work_params[:summary] = clean_storytext(summary_node.inner_html) if summary_node |
| 757 | |
| 758 | # Author beginning notes (inside .preface.group, before chapter content) |
| 759 | preface = @doc.at_css('.preface.group') |
| 760 | if preface |
| 761 | notes_node = preface.at_css('.notes.module blockquote.userstuff') |
| 762 | work_params[:notes] = clean_storytext(notes_node.inner_html) if notes_node |
| 763 | end |
| 764 | |
| 765 | # Story text: extract only from #chapters .userstuff, not the whole page body |
| 766 | chapters_div = @doc.at_css('#chapters') |
| 767 | if chapters_div |
| 768 | userstuff = chapters_div.at_css('.userstuff') |
| 769 | storytext = userstuff ? userstuff.inner_html : chapters_div.inner_html |
| 770 | else |
| 771 | storytext = @doc.at_css('body')&.inner_html || _story |
| 772 | end |
| 773 | work_params[:chapter_attributes][:content] = clean_storytext(storytext) |
| 774 | |
| 775 | if detect_tags |
| 776 | meta_group = @doc.at_css('dl.work.meta.group') |
| 777 | if meta_group |
| 778 | rating = meta_group.css('dd.rating.tags li a.tag').map { |a| a.inner_text.strip } |
| 779 | work_params[:rating_string] = convert_rating_string(rating.first) if rating.any? |
| 780 | |
| 781 | warnings = meta_group.css('dd.warning.tags li a.tag').map { |a| a.inner_text.strip } |
| 782 | work_params[:archive_warning_string] = warnings.join(', ') if warnings.any? |
| 783 | |
| 784 | fandoms = meta_group.css('dd.fandom.tags li a.tag').map { |a| a.inner_text.strip } |
| 785 | work_params[:fandom_string] = clean_tags(fandoms.join(ArchiveConfig.DELIMITER_FOR_OUTPUT)) if fandoms.any? |
| 786 | |
| 787 | relationships = meta_group.css('dd.relationship.tags li a.tag').map { |a| a.inner_text.strip } |
| 788 | work_params[:relationship_string] = clean_tags(relationships.join(ArchiveConfig.DELIMITER_FOR_OUTPUT)) if relationships.any? |
| 789 | |
| 790 | characters = meta_group.css('dd.character.tags li a.tag').map { |a| a.inner_text.strip } |
| 791 | work_params[:character_string] = clean_tags(characters.join(ArchiveConfig.DELIMITER_FOR_OUTPUT)) if characters.any? |
| 792 | |
| 793 | freeforms = meta_group.css('dd.freeform.tags li a.tag').map { |a| a.inner_text.strip } |
| 794 | work_params[:freeform_string] = clean_tags(freeforms.join(ArchiveConfig.DELIMITER_FOR_OUTPUT)) if freeforms.any? |
| 795 | |
| 796 | published = meta_group.at_css('dd.published') |
| 797 | work_params[:revised_at] = convert_revised_at(published.inner_text.strip) if published |
| 798 | end |
| 799 | end |
| 800 | |
| 801 | post_process_meta(work_params) |
| 802 | end |
| 803 | |
| 804 | # Move and/or copy any meta attributes that need to be on the chapter rather |
| 805 | # than on the work itself |
| 806 | def shift_chapter_attributes(work_params) |
| 807 | CHAPTER_ATTRIBUTES_ONLY.each_pair do |work_attrib, chapter_attrib| |
| 808 | if work_params[work_attrib] && !work_params[:chapter_attributes][chapter_attrib] |
| 809 | work_params[:chapter_attributes][chapter_attrib] = work_params[work_attrib] |
| 810 | work_params.delete(work_attrib) |
| 811 | end |
| 812 | end |
| 813 | |
| 814 | # copy any attributes from work to chapter as necessary |
| 815 | CHAPTER_ATTRIBUTES_ALSO.each_pair do |work_attrib, chapter_attrib| |
| 816 | if work_params[work_attrib] && !work_params[:chapter_attributes][chapter_attrib] |
| 817 | work_params[:chapter_attributes][chapter_attrib] = work_params[work_attrib] |
| 818 | end |
| 819 | end |
| 820 | |
| 821 | work_params |
| 822 | end |
| 823 | |
| 824 | # Find any cases of the given pieces of meta in the given text |
| 825 | # and return a hash of meta values |
| 826 | def scan_text_for_meta(text, detect_tags = true) |
| 827 | # break up the text with some extra newlines to make matching more likely |
| 828 | # and strip out some tags |
| 829 | text = text.gsub(/<br/, "\n<br") |
| 830 | text.gsub!(/<p/, "\n<p") |
| 831 | text.gsub!(/<\/?(label|span|div|b)(.*?)?>/, '') |
| 832 | |
| 833 | meta = {} |
| 834 | metapatterns = detect_tags ? REQUIRED_META.merge(OPTIONAL_META) : REQUIRED_META |
| 835 | is_tag = {}.tap do |h| |
| 836 | %w[fandom_string relationship_string freeform_string rating_string archive_warning_string].each do |c| |
| 837 | h[c.to_sym] = true |
| 838 | end |
| 839 | end |
| 840 | handler = {}.tap do |h| |
| 841 | %w[rating_string revised_at].each do |c| |
| 842 | h[c.to_sym] = "convert_#{c.to_s.downcase}" |
| 843 | end |
| 844 | end |
| 845 | |
| 846 | # 1. Look for Pattern: (whatever), optionally followed by a closing p or div tag |
| 847 | # 2. Set meta[:metaname] = whatever |
| 848 | # eg, if it finds Fandom: Stargate SG-1 it will set meta[:fandom] = Stargate SG-1 |
| 849 | # 3. convert_<metaname> for cleanup if such a function is defined (eg convert_rating_string) |
| 850 | metapatterns.each do |metaname, pattern| |
| 851 | metapattern = Regexp.new("(?:#{pattern}|#{pattern.pluralize})\s*:\s*(.*?)(?:</(?:p|div)>)?$", Regexp::IGNORECASE) |
| 852 | if text.match(metapattern) |
| 853 | value = Regexp.last_match[1] |
| 854 | value = clean_tags(value) if is_tag[metaname] |
| 855 | value = clean_close_html_tags(value) |
| 856 | value.strip! # lose leading/trailing whitespace |
| 857 | value = send(handler[metaname], value) if handler[metaname] |
| 858 | |
| 859 | meta[metaname] = value |
| 860 | end |
| 861 | end |
| 862 | post_process_meta meta |
| 863 | end |
| 864 | |
| 865 | def download_with_timeout(location, limit = 10) |
| 866 | story = +"" |
| 867 | Timeout.timeout(STORY_DOWNLOAD_TIMEOUT) do |
| 868 | begin |
| 869 | # we do a little cleanup here in case the user hasn't included the 'http://' |
| 870 | # or if they've used capital letters or an underscore in the hostname |
| 871 | uri = UrlFormatter.new(location).standardized |
| 872 | raise Error, I18n.t("story_parser.on_archive") if ArchiveConfig.PERMITTED_HOSTS.include?(uri.host) |
| 873 | |
| 874 | env_proxy = ENV["http_proxy"] |
| 875 | http = if env_proxy |
| 876 | proxy = URI(env_proxy) |
| 877 | Net::HTTP.new(uri.hostname, uri.port, proxy.hostname, proxy.port) |
| 878 | else |
| 879 | Net::HTTP.new(uri.hostname, uri.port) |
| 880 | end |
| 881 | http.use_ssl = true if uri.scheme == "https" |
| 882 | response = http.start { |h| h.request_get(uri.path.presence || "/") } |
| 883 | |
| 884 | case response |
| 885 | when Net::HTTPSuccess |
| 886 | story = response.body |
| 887 | when Net::HTTPRedirection |
| 888 | if limit.positive? |
| 889 | new_uri = URI.parse(response["location"]) |
| 890 | new_uri = URI.join(uri, new_uri) if new_uri.relative? |
| 891 | story = download_with_timeout(new_uri.to_s, limit - 1) |
| 892 | end |
| 893 | else |
| 894 | Rails.logger.error("------- STORY PARSER: download_with_timeout: response is not success or redirection ------") |
| 895 | nil |
| 896 | end |
| 897 | rescue Errno::ECONNREFUSED, SocketError, EOFError => e |
| 898 | Rails.logger.error("------- STORY PARSER: download_with_timeout: error rescue: \n#{e.inspect} ------") |
| 899 | nil |
| 900 | end |
| 901 | end |
| 902 | if story.blank? |
| 903 | raise Error, "We couldn't download anything from #{location}. Please make sure that the URL is correct and complete, and try again." |
| 904 | end |
| 905 | |
| 906 | # clean up any erroneously included string terminator (AO3-2251) |
| 907 | story.delete("\000") |
| 908 | end |
| 909 | |
| 910 | def get_last_modified(location) |
| 911 | Timeout.timeout(STORY_DOWNLOAD_TIMEOUT) do |
| 912 | resp = open(location) |
| 913 | resp.last_modified |
| 914 | end |
| 915 | end |
| 916 | |
| 917 | def get_source_if_known(known_sources, location) |
| 918 | known_sources.each do |source| |
| 919 | pattern = Regexp.new(eval("SOURCE_#{source.upcase}"), Regexp::IGNORECASE) |
| 920 | return source if location.match(pattern) |
| 921 | end |
| 922 | nil |
| 923 | end |
| 924 | |
| 925 | def clean_close_html_tags(value) |
| 926 | # if there are any closing html tags at the start of the value let's ditch them |
| 927 | value.gsub(/^(\s*<\/[^>]+>)+/, '') |
| 928 | end |
| 929 | |
| 930 | # We clean the text as if it had been submitted as the content of a chapter |
| 931 | def clean_storytext(storytext, field = "content") |
| 932 | storytext = storytext.encode("UTF-8", invalid: :replace, undef: :replace, replace: "") unless storytext.encoding.name == "UTF-8" |
| 933 | sanitize_value(field, storytext) |
| 934 | end |
| 935 | |
| 936 | # works conservatively -- doesn't split on |
| 937 | # spaces and truncates instead. |
| 938 | def clean_tags(tags) |
| 939 | tags = Sanitize.clean(tags.force_encoding("UTF-8")) # no html allowed in tags |
| 940 | tags_list = tags =~ /,/ ? tags.split(/,/) : [tags] |
| 941 | new_list = [] |
| 942 | tags_list.each do |tag| |
| 943 | tag.gsub!(/[*<>]/, '') |
| 944 | tag = truncate_on_word_boundary(tag, ArchiveConfig.TAG_MAX) |
| 945 | new_list << tag unless tag.blank? |
| 946 | end |
| 947 | new_list.join(ArchiveConfig.DELIMITER_FOR_OUTPUT) |
| 948 | end |
| 949 | |
| 950 | def truncate_on_word_boundary(text, max_length) |
| 951 | return if text.blank? |
| 952 | words = text.split |
| 953 | truncated = words.first |
| 954 | if words.length > 1 |
| 955 | words[1..words.length].each do |word| |
| 956 | truncated += " " + word if truncated.length + word.length + 1 <= max_length |
| 957 | end |
| 958 | end |
| 959 | truncated[0..max_length - 1] |
| 960 | end |
| 961 | |
| 962 | # convert space-separated tags to comma-separated |
| 963 | def clean_and_split_tags(tags) |
| 964 | tags = tags.split(/\s+/).join(',') if !tags.match(/,/) && tags.match(/\s/) |
| 965 | clean_tags(tags) |
| 966 | end |
| 967 | |
| 968 | # Convert the common ratings into whatever ratings we're |
| 969 | # using on this archive. |
| 970 | def convert_rating_string(rating) |
| 971 | rating = rating.downcase |
| 972 | if rating =~ /^(nc-?1[78]|x|ma|explicit)/ |
| 973 | ArchiveConfig.RATING_EXPLICIT_TAG_NAME |
| 974 | elsif rating =~ /^(r|m|mature)/ |
| 975 | ArchiveConfig.RATING_MATURE_TAG_NAME |
| 976 | elsif rating =~ /^(pg-?1[35]|t|teen)/ |
| 977 | ArchiveConfig.RATING_TEEN_TAG_NAME |
| 978 | elsif rating =~ /^(pg|g|k+|k|general audiences)/ |
| 979 | ArchiveConfig.RATING_GENERAL_TAG_NAME |
| 980 | else |
| 981 | ArchiveConfig.RATING_DEFAULT_TAG_NAME |
| 982 | end |
| 983 | end |
| 984 | |
| 985 | def convert_revised_at(date_string) |
| 986 | begin |
| 987 | date = nil |
| 988 | if date_string =~ /^(\d+)$/ |
| 989 | # probably seconds since the epoch |
| 990 | date = Time.at(Regex.last_match[1].to_i) |
| 991 | end |
| 992 | date ||= Date.parse(date_string) |
| 993 | return '' if date > Date.current |
| 994 | return date |
| 995 | rescue ArgumentError, TypeError |
| 996 | return '' |
| 997 | end |
| 998 | end |
| 999 | |
| 1000 | # Additional processing for meta - currently to make sure warnings |
| 1001 | # that aren't Archive warnings become additional tags instead |
| 1002 | def post_process_meta(meta) |
| 1003 | if meta[:archive_warning_string] |
| 1004 | result = process_warnings(meta[:archive_warning_string], meta[:freeform_string]) |
| 1005 | meta[:archive_warning_string] = result[:archive_warning_string] |
| 1006 | meta[:freeform_string] = result[:freeform_string] |
| 1007 | end |
| 1008 | meta |
| 1009 | end |
| 1010 | |
| 1011 | def process_warnings(warning_string, freeform_string) |
| 1012 | result = { |
| 1013 | archive_warning_string: warning_string, |
| 1014 | freeform_string: freeform_string |
| 1015 | } |
| 1016 | new_warning = '' |
| 1017 | result[:archive_warning_string].split(/\s?,\s?/).each do |warning| |
| 1018 | if ArchiveWarning.warning? warning |
| 1019 | new_warning += ', ' unless new_warning.blank? |
| 1020 | new_warning += warning |
| 1021 | else |
| 1022 | result[:freeform_string] = (result[:freeform_string] || '') + ", #{warning}" |
| 1023 | end |
| 1024 | end |
| 1025 | result[:archive_warning_string] = new_warning |
| 1026 | result |
| 1027 | end |
| 1028 | |
| 1029 | # tries to find appropriate existing collections and converts them to comma-separated collection names only |
| 1030 | def get_collection_names(collection_string) |
| 1031 | collections = "" |
| 1032 | collection_string.split(',').map(&:squish).each do |collection_name| |
| 1033 | collection = Collection.find_by(name: collection_name) || Collection.find_by(title: collection_name) |
| 1034 | if collection |
| 1035 | collections += ", " unless collections.blank? |
| 1036 | collections += collection.name |
| 1037 | end |
| 1038 | end |
| 1039 | collections |
| 1040 | end |
| 1041 | |
| 1042 | # determine which value to use for a metadata field |
| 1043 | def meta_or_default(detected_field, provided_field, default = nil) |
| 1044 | if @options[:override_tags] || detected_field.blank? |
| 1045 | if provided_field.blank? |
| 1046 | detected_field.blank? ? default : detected_field |
| 1047 | else |
| 1048 | provided_field |
| 1049 | end |
| 1050 | else |
| 1051 | detected_field |
| 1052 | end |
| 1053 | end |
| 1054 | end |
| 1055 |