OCR: Save the OCR result to a DOCX document
In This Topic
VintaSoft Imaging .NET SDK allows to save the OCR result to a DOCX document. For doing this VintaSoft Imaging .NET SDK saves the OCR result to a PDF document and converts created PDF document to a DOCX document.
If OCR result contains information about text and tables, created DOCX document will contain text and tables.
Here is C#/VB.NET code that shows how to convert a raster image file to a DOCX document (with OCR and table detection):
using Vintasoft.Imaging;
using Vintasoft.Imaging.ImageProcessing.Document;
using Vintasoft.Imaging.ImageProcessing.Info.TableDetection;
using Vintasoft.Imaging.Ocr;
using Vintasoft.Imaging.Ocr.Results;
using Vintasoft.Imaging.Ocr.Tesseract;
using Vintasoft.Imaging.Pdf;
using Vintasoft.Imaging.Pdf.Ocr;
using Vintasoft.Imaging.Pdf.Office;
class ConvertRasterImagesToDocxDocumentExample
{
// Required assemblies to run this code:
// Vintasoft.Shared.dll, Vintasoft.Imaging.dll,
// Vintasoft.Imaging.DocCleanup.dll,
// Vintasoft.Imaging.Office.OpenXml.dll,
// Vintasoft.Imaging.Ocr.dll, Vintasoft.Imaging.Ocr.Tesseract.dll,
// Vintasoft.Imaging.Pdf.dll, Vintasoft.Imaging.Pdf.Office.dll, Vintasoft.Imaging.Pdf.Ocr.dll
/// <summary>
/// Converts a raster image file to a DOCX document (with OCR and table detection).
/// </summary>
/// <param name="ocrLanguage">An OCR language.</param>
/// <param name="imageFilename">A filename of source raster image (TIFF, PNG, ...) file.</param>
/// <param name="docxFilename">A filename of destination DOCX file.</param>
public static void ConvertRasterImagesToDocxDocument(
OcrLanguage ocrLanguage, string imageFilename, string docxFilename)
{
// create an image collection
using (ImageCollection images = new ImageCollection())
{
// add images from image file into image collection
images.Add(imageFilename);
// create a searchable PDF document
using (PdfDocument document = new PdfDocument())
{
// create a PDF document builder
PdfDocumentBuilder documentBuilder = new PdfDocumentBuilder(document);
// specify that text must be placed over image
documentBuilder.PageCreationMode = PdfPageCreationMode.TextOverImage;
// create the Tesseract OCR engine
using (TesseractOcr tesseractOcr = new TesseractOcr(@".\TesseractOCR"))
{
// create OCR settings
OcrEngineSettings ocrSettings = new OcrEngineSettings(ocrLanguage);
// create the OCR engine manager
OcrEngineManager engineManager = new OcrEngineManager(tesseractOcr);
// OCR Preprocessing:
// AutoInvert, HalftoneRemoval, BorderClear, Deskew, HolePunchRemoval,
// Despeckle, AutoTextOrientation, Segmentation, TableWithBordersDetection
OcrPreprocessingCommand ocrPreprocessing = new OcrPreprocessingCommand();
// disable binarization
ocrPreprocessing.Binarization = null;
// for each image in image collection
foreach (VintasoftImage image in images)
{
// execute preprocessing
ocrPreprocessing.ExecuteInPlace(image);
// recognize text on image
OcrPage page = engineManager.Recognize(image, ocrSettings, ocrPreprocessing.DetectedRegions);
// add recognized OCR page to the PDF document
documentBuilder.AddPage(image, page);
}
// shutdown OCR engine
tesseractOcr.Shutdown();
// clear and dispose images in image collection
images.ClearAndDisposeItems();
// create converter that allows to convert PDF document to a DOCX document
using (PdfToDocxConverter converter = new PdfToDocxConverter())
{
// set the destination DOCX file
converter.OutputFilename = docxFilename;
// set converter settings
converter.ConvertGraphics = true;
converter.DetectHeaderFooter = true;
// enable table detection
converter.DetectTables = true;
converter.TableDetectionCommand = new TableWithBordersDetectionCommand();
// convert searchable PDF document to a DOCX document
converter.Execute(document);
}
}
}
}
}
}
Imports Vintasoft.Imaging
Imports Vintasoft.Imaging.ImageProcessing.Document
Imports Vintasoft.Imaging.ImageProcessing.Info.TableDetection
Imports Vintasoft.Imaging.Ocr
Imports Vintasoft.Imaging.Ocr.Results
Imports Vintasoft.Imaging.Ocr.Tesseract
Imports Vintasoft.Imaging.Pdf
Imports Vintasoft.Imaging.Pdf.Ocr
Imports Vintasoft.Imaging.Pdf.Office
Class ConvertRasterImagesToDocxDocumentExample
' Required assemblies to run this code:
' Vintasoft.Shared.dll, Vintasoft.Imaging.dll,
' Vintasoft.Imaging.DocCleanup.dll,
' Vintasoft.Imaging.Office.OpenXml.dll,
' Vintasoft.Imaging.Ocr.dll, Vintasoft.Imaging.Ocr.Tesseract.dll,
' Vintasoft.Imaging.Pdf.dll, Vintasoft.Imaging.Pdf.Office.dll, Vintasoft.Imaging.Pdf.Ocr.dll
''' <summary>
''' Converts a raster image file to a DOCX document (with OCR and table detection).
''' </summary>
''' <param name="ocrLanguage">An OCR language.</param>
''' <param name="imageFilename">A filename of source raster image (TIFF, PNG, ...) file.</param>
''' <param name="docxFilename">A filename of destination DOCX file.</param>
Public Shared Sub ConvertRasterImagesToDocxDocument(ocrLanguage As OcrLanguage, imageFilename As String, docxFilename As String)
' create an image collection
Using images As New ImageCollection()
' add images from image file into image collection
images.Add(imageFilename)
' create a searchable PDF document
Using document As New PdfDocument()
' create a PDF document builder
Dim documentBuilder As New PdfDocumentBuilder(document)
' specify that text must be placed over image
documentBuilder.PageCreationMode = PdfPageCreationMode.TextOverImage
' create the Tesseract OCR engine
Using tesseractOcr As New TesseractOcr(".\TesseractOCR")
' create OCR settings
Dim ocrSettings As New OcrEngineSettings(ocrLanguage)
' create the OCR engine manager
Dim engineManager As New OcrEngineManager(tesseractOcr)
' OCR Preprocessing:
' AutoInvert, HalftoneRemoval, BorderClear, Deskew, HolePunchRemoval,
' Despeckle, AutoTextOrientation, Segmentation, TableWithBordersDetection
Dim ocrPreprocessing As New OcrPreprocessingCommand()
' disable binarization
ocrPreprocessing.Binarization = Nothing
' for each image in image collection
For Each image As VintasoftImage In images
' execute preprocessing
ocrPreprocessing.ExecuteInPlace(image)
' recognize text on image
Dim page As OcrPage = engineManager.Recognize(image, ocrSettings, ocrPreprocessing.DetectedRegions)
' add recognized OCR page to the PDF document
documentBuilder.AddPage(image, page)
Next
' shutdown OCR engine
tesseractOcr.Shutdown()
' clear and dispose images in image collection
images.ClearAndDisposeItems()
' create converter that allows to convert PDF document to a DOCX document
Using converter As New PdfToDocxConverter()
' set the destination DOCX file
converter.OutputFilename = docxFilename
' set converter settings
converter.ConvertGraphics = True
converter.DetectHeaderFooter = True
' enable table detection
converter.DetectTables = True
converter.TableDetectionCommand = New TableWithBordersDetectionCommand()
' convert searchable PDF document to a DOCX document
converter.Execute(document)
End Using
End Using
End Using
End Using
End Sub
End Class