new IronOcr.OcrInput().LoadPdf("document.pdf").HighlightTextAndSaveAsImages(new IronOcr.IronTesseract(), "highlight_page_", IronOcr.ResultHighlightType.Word);
using IronOcr;IronTesseract ocrTesseract = new IronTesseract();using var ocrInput = new OcrInput();ocrInput.LoadPdf("document.pdf");ocrInput.HighlightTextAndSaveAsImages(ocrTesseract, "highlight_page_", ResultHighlightType.Paragraph);
using IronOcr;
IronTesseract ocrTesseract = new IronTesseract();
using var ocrInput = new OcrInput();
ocrInput.LoadPdf("document.pdf");
ocrInput.HighlightTextAndSaveAsImages(ocrTesseract, "highlight_page_", ResultHighlightType.Paragraph);
ImportsIronOcrDim ocrTesseract As New IronTesseract()Using ocrInput As New OcrInput() ocrInput.LoadPdf("document.pdf") ocrInput.HighlightTextAndSaveAsImages(ocrTesseract, "highlight_page_", ResultHighlightType.Paragraph)EndUsing
Imports IronOcr
Dim ocrTesseract As New IronTesseract()
Using ocrInput As New OcrInput()
ocrInput.LoadPdf("document.pdf")
ocrInput.HighlightTextAndSaveAsImages(ocrTesseract, "highlight_page_", ResultHighlightType.Paragraph)
End Using
using IronOcr;using System;// Initialize the OCR engine with custom configurationIronTesseract ocrTesseract = new IronTesseract();// Configure for better accuracy if neededocrTesseract.Configuration.ReadBarCodes = false; // Disable if not needed for performanceocrTesseract.Configuration.PageSegmentationMode = TesseractPageSegmentationMode.AutoOsd;// Load the PDF documentusing var ocrInput = new OcrInput();ocrInput.LoadPdf("document.pdf");// Generate highlights for each typeConsole.WriteLine("Generating character-level highlights...");ocrInput.HighlightTextAndSaveAsImages(ocrTesseract, "highlight_character_", ResultHighlightType.Character);Console.WriteLine("Generating word-level highlights...");ocrInput.HighlightTextAndSaveAsImages(ocrTesseract, "highlight_word_", ResultHighlightType.Word);Console.WriteLine("Generating line-level highlights...");ocrInput.HighlightTextAndSaveAsImages(ocrTesseract, "highlight_line_", ResultHighlightType.Line);Console.WriteLine("Generating paragraph-level highlights...");ocrInput.HighlightTextAndSaveAsImages(ocrTesseract, "highlight_paragraph_", ResultHighlightType.Paragraph);Console.WriteLine("All highlight images have been generated successfully!");
using IronOcr;
using System;
// Initialize the OCR engine with custom configuration
IronTesseract ocrTesseract = new IronTesseract();
// Configure for better accuracy if needed
ocrTesseract.Configuration.ReadBarCodes = false; // Disable if not needed for performance
ocrTesseract.Configuration.PageSegmentationMode = TesseractPageSegmentationMode.AutoOsd;
// Load the PDF document
using var ocrInput = new OcrInput();
ocrInput.LoadPdf("document.pdf");
// Generate highlights for each type
Console.WriteLine("Generating character-level highlights...");
ocrInput.HighlightTextAndSaveAsImages(ocrTesseract, "highlight_character_", ResultHighlightType.Character);
Console.WriteLine("Generating word-level highlights...");
ocrInput.HighlightTextAndSaveAsImages(ocrTesseract, "highlight_word_", ResultHighlightType.Word);
Console.WriteLine("Generating line-level highlights...");
ocrInput.HighlightTextAndSaveAsImages(ocrTesseract, "highlight_line_", ResultHighlightType.Line);
Console.WriteLine("Generating paragraph-level highlights...");
ocrInput.HighlightTextAndSaveAsImages(ocrTesseract, "highlight_paragraph_", ResultHighlightType.Paragraph);
Console.WriteLine("All highlight images have been generated successfully!");
ImportsIronOcrImportsSystem' Initialize the OCR engine with custom configurationDim ocrTesseract As New IronTesseract()' Configure for better accuracy if neededocrTesseract.Configuration.ReadBarCodes = False ' Disable if not needed for performanceocrTesseract.Configuration.PageSegmentationMode = TesseractPageSegmentationMode.AutoOsd' Load the PDF documentUsing ocrInput As New OcrInput() ocrInput.LoadPdf("document.pdf") ' Generate highlights for each typeConsole.WriteLine("Generating character-level highlights...") ocrInput.HighlightTextAndSaveAsImages(ocrTesseract, "highlight_character_", ResultHighlightType.Character)Console.WriteLine("Generating word-level highlights...") ocrInput.HighlightTextAndSaveAsImages(ocrTesseract, "highlight_word_", ResultHighlightType.Word)Console.WriteLine("Generating line-level highlights...") ocrInput.HighlightTextAndSaveAsImages(ocrTesseract, "highlight_line_", ResultHighlightType.Line)Console.WriteLine("Generating paragraph-level highlights...") ocrInput.HighlightTextAndSaveAsImages(ocrTesseract, "highlight_paragraph_", ResultHighlightType.Paragraph)EndUsingConsole.WriteLine("All highlight images have been generated successfully!")
Imports IronOcr
Imports System
' Initialize the OCR engine with custom configuration
Dim ocrTesseract As New IronTesseract()
' Configure for better accuracy if needed
ocrTesseract.Configuration.ReadBarCodes = False ' Disable if not needed for performance
ocrTesseract.Configuration.PageSegmentationMode = TesseractPageSegmentationMode.AutoOsd
' Load the PDF document
Using ocrInput As New OcrInput()
ocrInput.LoadPdf("document.pdf")
' Generate highlights for each type
Console.WriteLine("Generating character-level highlights...")
ocrInput.HighlightTextAndSaveAsImages(ocrTesseract, "highlight_character_", ResultHighlightType.Character)
Console.WriteLine("Generating word-level highlights...")
ocrInput.HighlightTextAndSaveAsImages(ocrTesseract, "highlight_word_", ResultHighlightType.Word)
Console.WriteLine("Generating line-level highlights...")
ocrInput.HighlightTextAndSaveAsImages(ocrTesseract, "highlight_line_", ResultHighlightType.Line)
Console.WriteLine("Generating paragraph-level highlights...")
ocrInput.HighlightTextAndSaveAsImages(ocrTesseract, "highlight_paragraph_", ResultHighlightType.Paragraph)
End Using
Console.WriteLine("All highlight images have been generated successfully!")
using IronOcr;using System.IO;IronTesseract ocrTesseract = new IronTesseract();// Load a multi-page documentusing var ocrInput = new OcrInput();ocrInput.LoadPdf("multi-page-document.pdf");// Create output directory if it doesn't existstring outputDir = "highlighted_pages";Directory.CreateDirectory(outputDir);// Generate highlights for each page// Files will be named: highlighted_pages/page_0.png, page_1.png, etc.ocrInput.HighlightTextAndSaveAsImages(ocrTesseract, Path.Combine(outputDir, "page_"), ResultHighlightType.Word);// Count generated files for verificationint pageCount = Directory.GetFiles(outputDir, "page_*.png").Length;Console.WriteLine($"Generated {pageCount} highlighted page images");
using IronOcr;
using System.IO;
IronTesseract ocrTesseract = new IronTesseract();
// Load a multi-page document
using var ocrInput = new OcrInput();
ocrInput.LoadPdf("multi-page-document.pdf");
// Create output directory if it doesn't exist
string outputDir = "highlighted_pages";
Directory.CreateDirectory(outputDir);
// Generate highlights for each page
// Files will be named: highlighted_pages/page_0.png, page_1.png, etc.
ocrInput.HighlightTextAndSaveAsImages(ocrTesseract,
Path.Combine(outputDir, "page_"),
ResultHighlightType.Word);
// Count generated files for verification
int pageCount = Directory.GetFiles(outputDir, "page_*.png").Length;
Console.WriteLine($"Generated {pageCount} highlighted page images");
ImportsIronOcrImportsSystem.IODim ocrTesseract As New IronTesseract()' Load a multi-page documentUsing ocrInput As New OcrInput() ocrInput.LoadPdf("multi-page-document.pdf") ' Create output directory if it doesn't exist Dim outputDir AsString = "highlighted_pages"Directory.CreateDirectory(outputDir) ' Generate highlights for each page ' Files will be named: highlighted_pages/page_0.png, page_1.png, etc. ocrInput.HighlightTextAndSaveAsImages(ocrTesseract, Path.Combine(outputDir, "page_"), ResultHighlightType.Word) ' Count generated files for verification Dim pageCount AsInteger = Directory.GetFiles(outputDir, "page_*.png").LengthConsole.WriteLine($"Generated {pageCount} highlighted page images")EndUsing
Imports IronOcr
Imports System.IO
Dim ocrTesseract As New IronTesseract()
' Load a multi-page document
Using ocrInput As New OcrInput()
ocrInput.LoadPdf("multi-page-document.pdf")
' Create output directory if it doesn't exist
Dim outputDir As String = "highlighted_pages"
Directory.CreateDirectory(outputDir)
' Generate highlights for each page
' Files will be named: highlighted_pages/page_0.png, page_1.png, etc.
ocrInput.HighlightTextAndSaveAsImages(ocrTesseract,
Path.Combine(outputDir, "page_"),
ResultHighlightType.Word)
' Count generated files for verification
Dim pageCount As Integer = Directory.GetFiles(outputDir, "page_*.png").Length
Console.WriteLine($"Generated {pageCount} highlighted page images")
End Using
try{ using var ocrInput = new OcrInput(); ocrInput.LoadPdf("document.pdf"); // Apply image filters if needed for better recognition ocrInput.Deskew(); // Correct slight rotations ocrInput.DeNoise(); // Remove background noise ocrInput.HighlightTextAndSaveAsImages(ocrTesseract, "highlight_", ResultHighlightType.Word);}catch (Exception ex){Console.WriteLine($"Error during highlighting: {ex.Message}"); // Log error details for debugging}
try
{
using var ocrInput = new OcrInput();
ocrInput.LoadPdf("document.pdf");
// Apply image filters if needed for better recognition
ocrInput.Deskew(); // Correct slight rotations
ocrInput.DeNoise(); // Remove background noise
ocrInput.HighlightTextAndSaveAsImages(ocrTesseract, "highlight_", ResultHighlightType.Word);
}
catch (Exception ex)
{
Console.WriteLine($"Error during highlighting: {ex.Message}");
// Log error details for debugging
}
ImportsSystemTryUsing ocrInput As New OcrInput() ocrInput.LoadPdf("document.pdf") ' Apply image filters if needed for better recognition ocrInput.Deskew() ' Correct slight rotations ocrInput.DeNoise() ' Remove background noise ocrInput.HighlightTextAndSaveAsImages(ocrTesseract, "highlight_", ResultHighlightType.Word)EndUsingCatch ex AsExceptionConsole.WriteLine($"Error during highlighting: {ex.Message}") ' Log error details for debuggingEndTry
Imports System
Try
Using ocrInput As New OcrInput()
ocrInput.LoadPdf("document.pdf")
' Apply image filters if needed for better recognition
ocrInput.Deskew() ' Correct slight rotations
ocrInput.DeNoise() ' Remove background noise
ocrInput.HighlightTextAndSaveAsImages(ocrTesseract, "highlight_", ResultHighlightType.Word)
End Using
Catch ex As Exception
Console.WriteLine($"Error during highlighting: {ex.Message}")
' Log error details for debugging
End Try
public class OcrDebugger{ private readonly IronTesseract _tesseract; private readonly bool _debugMode; publicOcrDebugger(bool enableDebugMode = false) { _tesseract = new IronTesseract(); _debugMode = enableDebugMode; } public OcrResultProcessDocument(string filePath) { using var input = new OcrInput(); input.LoadPdf(filePath); // Apply preprocessing input.Deskew(); input.DeNoise(); // Generate debug highlights if in debug mode if (_debugMode) { string debugPath = $"debug_{Path.GetFileNameWithoutExtension(filePath)}_"; input.HighlightTextAndSaveAsImages(_tesseract, debugPath, ResultHighlightType.Word); } // Perform actual OCR return _tesseract.Read(input); }}
public class OcrDebugger
{
private readonly IronTesseract _tesseract;
private readonly bool _debugMode;
public OcrDebugger(bool enableDebugMode = false)
{
_tesseract = new IronTesseract();
_debugMode = enableDebugMode;
}
public OcrResult ProcessDocument(string filePath)
{
using var input = new OcrInput();
input.LoadPdf(filePath);
// Apply preprocessing
input.Deskew();
input.DeNoise();
// Generate debug highlights if in debug mode
if (_debugMode)
{
string debugPath = $"debug_{Path.GetFileNameWithoutExtension(filePath)}_";
input.HighlightTextAndSaveAsImages(_tesseract, debugPath, ResultHighlightType.Word);
}
// Perform actual OCR
return _tesseract.Read(input);
}
}
Public Class OcrDebugger PrivateReadOnly _tesseract AsIronTesseract PrivateReadOnly _debugMode AsBoolean Public Sub New(Optional enableDebugMode AsBoolean = False) _tesseract = New IronTesseract() _debugMode = enableDebugMode End Sub Public Function ProcessDocument(filePath AsString) AsOcrResultUsing input As New OcrInput() input.LoadPdf(filePath) ' Apply preprocessing input.Deskew() input.DeNoise() ' Generate debug highlights if in debug mode If _debugMode Then Dim debugPath AsString = $"debug_{Path.GetFileNameWithoutExtension(filePath)}_" input.HighlightTextAndSaveAsImages(_tesseract, debugPath, ResultHighlightType.Word) End If ' Perform actual OCR Return _tesseract.Read(input)EndUsing End FunctionEnd Class
Public Class OcrDebugger
Private ReadOnly _tesseract As IronTesseract
Private ReadOnly _debugMode As Boolean
Public Sub New(Optional enableDebugMode As Boolean = False)
_tesseract = New IronTesseract()
_debugMode = enableDebugMode
End Sub
Public Function ProcessDocument(filePath As String) As OcrResult
Using input As New OcrInput()
input.LoadPdf(filePath)
' Apply preprocessing
input.Deskew()
input.DeNoise()
' Generate debug highlights if in debug mode
If _debugMode Then
Dim debugPath As String = $"debug_{Path.GetFileNameWithoutExtension(filePath)}_"
input.HighlightTextAndSaveAsImages(_tesseract, debugPath, ResultHighlightType.Word)
End If
' Perform actual OCR
Return _tesseract.Read(input)
End Using
End Function
End Class
using IronOCR,您只需一行代码就能高亮 PDF 中的单词:new IronOcr.OcrInput().LoadPdf("document.pdf").HighlightTextAndSaveAsImages(new IronOcr.IronTesseract(), "highlight_page_", IronOcr.ResultHighlightType.Word).这将加载 PDF 并创建带有高亮单词的图像。
How can I address issues with blank output images in IronOCR?
Ensure the input document contains readable text, and that the OCR engine is configured correctly. You may also need image optimization or orientation adjustments to improve results.
What performance best practices should I follow when using IronOCR highlights?
Manage large highlighted images by ensuring enough available space and consider implementing multithreading for improved speed. Proper error handling should also be in place for file operations.