VintaSoft Imaging .NET SDK 15.2: Documentation for .NET developer
In This Topic
OCR: Save the OCR result to a DOCX document
In This Topic
VintaSoft Imaging .NET SDK allows to save the OCR result to a DOCX document. For doing this VintaSoft Imaging .NET SDK saves the OCR result to a PDF document and converts created PDF document to a DOCX document.

If OCR result contains information about text and tables, created DOCX document will contain text and tables.
Here is C#/VB.NET code that shows how to convert a raster image file to a DOCX document (with OCR and table detection):
using Vintasoft.Imaging;
using Vintasoft.Imaging.ImageProcessing.Document;
using Vintasoft.Imaging.ImageProcessing.Info.TableDetection;
using Vintasoft.Imaging.Ocr;
using Vintasoft.Imaging.Ocr.Results;
using Vintasoft.Imaging.Ocr.Tesseract;
using Vintasoft.Imaging.Pdf;
using Vintasoft.Imaging.Pdf.Ocr;
using Vintasoft.Imaging.Pdf.Office;

class ConvertRasterImagesToDocxDocumentExample
{
    // Required assemblies to run this code:
    // Vintasoft.Shared.dll, Vintasoft.Imaging.dll,
    // Vintasoft.Imaging.DocCleanup.dll,
    // Vintasoft.Imaging.Office.OpenXml.dll,
    // Vintasoft.Imaging.Ocr.dll, Vintasoft.Imaging.Ocr.Tesseract.dll,
    // Vintasoft.Imaging.Pdf.dll, Vintasoft.Imaging.Pdf.Office.dll, Vintasoft.Imaging.Pdf.Ocr.dll

    /// <summary>
    /// Converts a raster image file to a DOCX document (with OCR and table detection).
    /// </summary>
    /// <param name="ocrLanguage">An OCR language.</param>
    /// <param name="imageFilename">A filename of source raster image (TIFF, PNG, ...) file.</param>
    /// <param name="docxFilename">A filename of destination DOCX file.</param>
    public static void ConvertRasterImagesToDocxDocument(
        OcrLanguage ocrLanguage, string imageFilename, string docxFilename)
    {
        // create an image collection
        using (ImageCollection images = new ImageCollection())
        {
            // add images from image file into image collection
            images.Add(imageFilename);

            // create a searchable PDF document
            using (PdfDocument document = new PdfDocument())
            {
                // create a PDF document builder
                PdfDocumentBuilder documentBuilder = new PdfDocumentBuilder(document);

                // specify that text must be placed over image
                documentBuilder.PageCreationMode = PdfPageCreationMode.TextOverImage;

                // create the Tesseract OCR engine
                using (TesseractOcr tesseractOcr = new TesseractOcr(@".\TesseractOCR"))
                {
                    // create OCR settings
                    OcrEngineSettings ocrSettings = new OcrEngineSettings(ocrLanguage);

                    // create the OCR engine manager
                    OcrEngineManager engineManager = new OcrEngineManager(tesseractOcr);

                    // OCR Preprocessing:
                    // AutoInvert, HalftoneRemoval, BorderClear, Deskew, HolePunchRemoval,
                    // Despeckle, AutoTextOrientation, Segmentation, TableWithBordersDetection
                    OcrPreprocessingCommand ocrPreprocessing = new OcrPreprocessingCommand();
                    // disable binarization
                    ocrPreprocessing.Binarization = null;

                    // for each image in image collection
                    foreach (VintasoftImage image in images)
                    {
                        // execute preprocessing
                        ocrPreprocessing.ExecuteInPlace(image);

                        // recognize text on image
                        OcrPage page = engineManager.Recognize(image, ocrSettings, ocrPreprocessing.DetectedRegions);

                        // add recognized OCR page to the PDF document
                        documentBuilder.AddPage(image, page);
                    }

                    // shutdown OCR engine
                    tesseractOcr.Shutdown();

                    // clear and dispose images in image collection
                    images.ClearAndDisposeItems();

                    // create converter that allows to convert PDF document to a DOCX document
                    using (PdfToDocxConverter converter = new PdfToDocxConverter())
                    {
                        // set the destination DOCX file
                        converter.OutputFilename = docxFilename;

                        // set converter settings
                        converter.ConvertGraphics = true;
                        converter.DetectHeaderFooter = true;

                        // enable table detection
                        converter.DetectTables = true;
                        converter.TableDetectionCommand = new TableWithBordersDetectionCommand();

                        // convert searchable PDF document to a DOCX document
                        converter.Execute(document);
                    }
                }
            }
        }
    }
}

Imports Vintasoft.Imaging
Imports Vintasoft.Imaging.ImageProcessing.Document
Imports Vintasoft.Imaging.ImageProcessing.Info.TableDetection
Imports Vintasoft.Imaging.Ocr
Imports Vintasoft.Imaging.Ocr.Results
Imports Vintasoft.Imaging.Ocr.Tesseract
Imports Vintasoft.Imaging.Pdf
Imports Vintasoft.Imaging.Pdf.Ocr
Imports Vintasoft.Imaging.Pdf.Office

Class ConvertRasterImagesToDocxDocumentExample
    ' Required assemblies to run this code:
    ' Vintasoft.Shared.dll, Vintasoft.Imaging.dll,
    ' Vintasoft.Imaging.DocCleanup.dll,
    ' Vintasoft.Imaging.Office.OpenXml.dll,
    ' Vintasoft.Imaging.Ocr.dll, Vintasoft.Imaging.Ocr.Tesseract.dll,
    ' Vintasoft.Imaging.Pdf.dll, Vintasoft.Imaging.Pdf.Office.dll, Vintasoft.Imaging.Pdf.Ocr.dll

    ''' <summary>
    ''' Converts a raster image file to a DOCX document (with OCR and table detection).
    ''' </summary>
    ''' <param name="ocrLanguage">An OCR language.</param>
    ''' <param name="imageFilename">A filename of source raster image (TIFF, PNG, ...) file.</param>
    ''' <param name="docxFilename">A filename of destination DOCX file.</param>
    Public Shared Sub ConvertRasterImagesToDocxDocument(ocrLanguage As OcrLanguage, imageFilename As String, docxFilename As String)
        ' create an image collection
        Using images As New ImageCollection()
            ' add images from image file into image collection
            images.Add(imageFilename)

            ' create a searchable PDF document
            Using document As New PdfDocument()
                ' create a PDF document builder
                Dim documentBuilder As New PdfDocumentBuilder(document)

                ' specify that text must be placed over image
                documentBuilder.PageCreationMode = PdfPageCreationMode.TextOverImage

                ' create the Tesseract OCR engine
                Using tesseractOcr As New TesseractOcr(".\TesseractOCR")
                    ' create OCR settings
                    Dim ocrSettings As New OcrEngineSettings(ocrLanguage)

                    ' create the OCR engine manager
                    Dim engineManager As New OcrEngineManager(tesseractOcr)

                    ' OCR Preprocessing:
                    ' AutoInvert, HalftoneRemoval, BorderClear, Deskew, HolePunchRemoval,
                    ' Despeckle, AutoTextOrientation, Segmentation, TableWithBordersDetection
                    Dim ocrPreprocessing As New OcrPreprocessingCommand()
                    ' disable binarization
                    ocrPreprocessing.Binarization = Nothing

                    ' for each image in image collection
                    For Each image As VintasoftImage In images
                        ' execute preprocessing
                        ocrPreprocessing.ExecuteInPlace(image)

                        ' recognize text on image
                        Dim page As OcrPage = engineManager.Recognize(image, ocrSettings, ocrPreprocessing.DetectedRegions)

                        ' add recognized OCR page to the PDF document
                        documentBuilder.AddPage(image, page)
                    Next

                    ' shutdown OCR engine
                    tesseractOcr.Shutdown()

                    ' clear and dispose images in image collection
                    images.ClearAndDisposeItems()

                    ' create converter that allows to convert PDF document to a DOCX document
                    Using converter As New PdfToDocxConverter()
                        ' set the destination DOCX file
                        converter.OutputFilename = docxFilename

                        ' set converter settings
                        converter.ConvertGraphics = True
                        converter.DetectHeaderFooter = True

                        ' enable table detection
                        converter.DetectTables = True
                        converter.TableDetectionCommand = New TableWithBordersDetectionCommand()

                        ' convert searchable PDF document to a DOCX document
                        converter.Execute(document)
                    End Using
                End Using
            End Using
        End Using
    End Sub
End Class