Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
26 changes: 13 additions & 13 deletions apis_v1/views/views_extension.py
Original file line number Diff line number Diff line change
Expand Up @@ -80,7 +80,7 @@ def process_pdf_to_html(pdf_url, return_version):
output_from_subprocess = 'exception occurred before output was captured'
status = ''
success = False
# logger.error('entry to process_pdf_to_html:' + pdf_url + ' ' + str(return_version))
logger.error('entry to process_pdf_to_html: ' + pdf_url + ' ' + str(return_version))

# Version report, only used to debug connectivity to the Tika server
if return_version:
Expand All @@ -101,7 +101,7 @@ def process_pdf_to_html(pdf_url, return_version):
}
return json_data

# logger.error('immediately after return_version: ' + str(return_version))
logger.error('immediately after return_version: ' + str(return_version))
pdf_file_name = os.path.basename(pdf_url)
absolute_html_file = build_absolute_path_for_tempfile(pdf_file_name).replace('.pdf', '.html')
try:
Expand All @@ -119,26 +119,26 @@ def process_pdf_to_html(pdf_url, return_version):
try:
raw = scraper.get(pdf_url)
pdf_text_text = raw.content # in bytes, not using str(raw.content)
# logger.error('cloudscraper attempt with base PDF url : ' + pdf_url +
# ' returned bytes: ' + str(len(pdf_text_text)))
logger.error('cloudscraper attempt with base PDF url : ' + pdf_url +
' returned bytes: ' + str(len(pdf_text_text)))
success = True

# Probably got a http 403 forbidden, due to cloudscraper unsuccessfully handling a Cloudflare challenge
# Now try to use Google's (hopefully) cached version of the page
except Exception as scraper_or_tempfile_error:
status = "First pass with base url failed with a " + str(scraper_or_tempfile_error)
# logger.error('cloudscraper with base PDF url or tempfile write exception: ' +
# str(scraper_or_tempfile_error))
logger.error('cloudscraper with base PDF url or tempfile write exception: ' +
str(scraper_or_tempfile_error))

if not success:
logger.error('first pass === not success')
is_pdf = False
try:
# logger.error('first pass === not success, pdf_url: ' + pdf_url)
logger.error('first pass === not success, pdf_url: ' + pdf_url)
encoded = quote(pdf_url, safe='')
# logger.error('encoded success: ' + encoded)
logger.error('encoded success: ' + encoded)
google_cached_pdf_url = 'https://webcache.googleusercontent.com/search?q=cache:' + encoded
# logger.error('cloudscraper attempt with google cached PDF url: ' + google_cached_pdf_url)
logger.error('cloudscraper attempt with google cached PDF url: ' + google_cached_pdf_url)

headers = {
'User-Agent': 'Mozilla/5.0 (Windows NT 6.1; WOW64) AppleWebKit/537.36 (KHTML, like Gecko) '
Expand All @@ -149,14 +149,14 @@ def process_pdf_to_html(pdf_url, return_version):
'Accept-Language': 'en-US,en;q=0.8',
'Connection': 'keep-alive'}
r = requests.get(google_cached_pdf_url, headers)
# logger.error('after requests.get: ' + google_cached_pdf_url)
logger.error('after requests.get: ' + google_cached_pdf_url)
# skip saving the pdf file (since we don't have one), and write the final html file to the temp dir
html_text_text = r.text
out_file = open(absolute_html_file, 'w')
out_file.write(html_text_text)

# logger.error('requests was successful with google cached PDF url : ' + google_cached_pdf_url +
# ' returned bytes: ' + str(len(pdf_text_text)))
logger.error('requests was successful with google cached PDF url : ' + google_cached_pdf_url +
' returned bytes: ' + str(len(pdf_text_text)))
success = True
except Exception as scraper_or_tempfile_error2: # Out of luck
status += ", Second pass with google cached PDF url failed with a: " + str(scraper_or_tempfile_error2)
Expand All @@ -175,7 +175,7 @@ def process_pdf_to_html(pdf_url, return_version):
output_from_subprocess = 'Tika status: ' + str(tika_response.status_code)
with open(absolute_html_file, 'w') as out_file:
out_file.write(tika_response.text)
# logger.error('Tika PUT output: ' + output_from_subprocess)
logger.error('Tika PUT output: ' + output_from_subprocess)
except Exception as tika_error:
status += ', ' + str(tika_error)
logger.error('Tika PUT request exception: ' + str(tika_error))
Expand Down
4 changes: 3 additions & 1 deletion candidate/controllers.py
Original file line number Diff line number Diff line change
Expand Up @@ -3384,7 +3384,9 @@ def find_organization_endorsements_of_candidates_on_one_web_page(site_url, endor

if site_url.lower().endswith(".pdf"):
print("PDF Detected ", site_url)
response = process_pdf_to_html(site_url)

return_version = False
response = process_pdf_to_html(site_url, return_version)
if positive_value_exists(response['s3_url_for_html']):
# Overwrite the site_url parameter, with a url to an html representation of the PDF file
site_url = response['s3_url_for_html']
Expand Down
Loading