VintaSoft Imaging .NET SDK 15.1: Documentation for .NET developer
In This Topic
OCR: How to tune OCR to recognize only digits?
In This Topic
Here is C#/VB.NET code that shows how to configure "white" list of OCR engine to recognize only digits:
/// <summary>
/// Specifies that text contains only the limited set of characters and
/// recognizes the text in image.
/// </summary>
/// <param name="filename">The name of file, which stores images with text.</param>
public static void OcrImageWithDigits(string filename)
{
    // create an image collection
    using (Vintasoft.Imaging.ImageCollection images = 
        new Vintasoft.Imaging.ImageCollection())
    {
        // add images from file to the image collection
        images.Add(filename);

        System.Console.WriteLine("Create Tesseract OCR engine...");
        // create the Tesseract OCR engine
        using (Vintasoft.Imaging.Ocr.Tesseract.TesseractOcr tesseractOcr = 
            new Vintasoft.Imaging.Ocr.Tesseract.TesseractOcr())
        {
            System.Console.WriteLine("Initialize OCR engine...");
            // init the Tesseract OCR engine
            tesseractOcr.Init(new Vintasoft.Imaging.Ocr.OcrEngineSettings(
                Vintasoft.Imaging.Ocr.OcrLanguage.English));

            // set the "white list" of recognizing characters
            tesseractOcr.SetVariable(
                "tessedit_char_whitelist", "01234567890");

            // for each image
            foreach (Vintasoft.Imaging.VintasoftImage image in images)
            {
                System.Console.WriteLine("Recognize the image...");

                // recognize text in image
                Vintasoft.Imaging.Ocr.Results.OcrPage ocrResult = tesseractOcr.Recognize(image);

                // output the recognized text

                System.Console.WriteLine("Page Text:");
                System.Console.WriteLine(ocrResult.GetText());
                System.Console.WriteLine();
            }

            // shutdown the Tesseract OCR engine
            tesseractOcr.Shutdown();
        }

        // free images
        images.ClearAndDisposeItems();
    }
}
''' <summary>
''' Specifies that text contains only the limited set of characters and
''' recognizes the text in image.
''' </summary>
''' <param name="filename">The name of file, which stores images with text.</param>
Public Shared Sub OcrImageWithDigits(filename As String)
    ' create an image collection
    Using images As New Vintasoft.Imaging.ImageCollection()
        ' add images from file to the image collection
        images.Add(filename)

        System.Console.WriteLine("Create Tesseract OCR engine...")
        ' create the Tesseract OCR engine
        Using tesseractOcr As New Vintasoft.Imaging.Ocr.Tesseract.TesseractOcr()
            System.Console.WriteLine("Initialize OCR engine...")
            ' init the Tesseract OCR engine
            tesseractOcr.Init(New Vintasoft.Imaging.Ocr.OcrEngineSettings(Vintasoft.Imaging.Ocr.OcrLanguage.English))

            ' set the "white list" of recognizing characters
            tesseractOcr.SetVariable("tessedit_char_whitelist", "01234567890")

            ' for each image
            For Each image As Vintasoft.Imaging.VintasoftImage In images
                System.Console.WriteLine("Recognize the image...")

                ' recognize text in image
                Dim ocrResult As Vintasoft.Imaging.Ocr.Results.OcrPage = tesseractOcr.Recognize(image)

                ' output the recognized text

                System.Console.WriteLine("Page Text:")
                System.Console.WriteLine(ocrResult.GetText())
                System.Console.WriteLine()
            Next

            ' shutdown the Tesseract OCR engine
            tesseractOcr.Shutdown()
        End Using

        ' free images
        images.ClearAndDisposeItems()
    End Using
End Sub