Sitelet https://github.com/documentcloud/docsplit/pull/134/commits/a6d5f4e7074ac60a0f418312f7b916d6dc3e922c
Skip to content
Open
Changes from 1 commit
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
Prev Previous commit
Next Next commit
Raise upon failed image conversion
  • Loading branch information
tbk303 committed Feb 6, 2019
commit a6d5f4e7074ac60a0f418312f7b916d6dc3e922c
2 changes: 2 additions & 0 deletions lib/docsplit/text_extractor.rb
Original file line number Diff line number Diff line change
Expand Up @@ -69,6 +69,7 @@ def extract_from_ocr(pdf, pages)
escaped_tiff = ESCAPE[tiff]
file = "#{base_path}_#{page}"
run "MAGICK_TMPDIR=#{tempdir} OMP_NUM_THREADS=2 timeout #{TIMEOUT} gm convert -despeckle +adjoin #{MEMORY_ARGS} #{OCR_FLAGS} #{escaped_pdf}[#{page - 1}] #{escaped_tiff} 2>&1"
raise Docsplit::ExtractionFailed unless File.exists? escaped_tiff
run "tesseract #{escaped_tiff} #{ESCAPE[file]} -l #{@language} #{psm} 2>&1"
clean_text(file + '.txt') if @clean_ocr
FileUtils.remove_entry_secure tiff
Expand All @@ -78,6 +79,7 @@ def extract_from_ocr(pdf, pages)
escaped_tiff = ESCAPE[tiff]
run "MAGICK_TMPDIR=#{tempdir} OMP_NUM_THREADS=2 timeout #{TIMEOUT} gm convert -despeckle #{MEMORY_ARGS} #{OCR_FLAGS} #{escaped_pdf} #{escaped_tiff} 2>&1"
#if the user says don't do orientation detection or the plugin is not installed, set psm to 0
raise Docsplit::ExtractionFailed unless File.exists? escaped_tiff
run "tesseract #{escaped_tiff} #{base_path} -l #{@language} #{psm} 2>&1"
clean_text(base_path + '.txt') if @clean_ocr
end
Expand Down