mirror of
https://github.com/ajisaacs/OpenNest.git
synced 2026-10-10 16:44:13 -04:00
fix(test): read PDF text by visual rows across platforms
This commit is contained in:
1 parent
5fc563bbca
commit
3ddfff242d
2 files changed
+25
-10
No files matched your search
@@ -28,22 +28,36 @@ public static class ReportPdf
|
||||
}
|
||||
}
|
||||
|
||||
/// <summary>Layout text of one page; skipped where poppler-utils is not installed.</summary>
|
||||
/// <summary>Visual-row text of one page; bbox coordinates avoid platform-specific -layout order.</summary>
|
||||
public static string Text(string path, int page) =>
|
||||
Regex.Replace(Run("pdftotext", "-layout", "-f", Page(page), "-l", Page(page), path, "-"), "[ \t]+\n", "\n");
|
||||
LayoutPage(Run("pdftotext", "-bbox", "-f", Page(page), "-l", Page(page), path, "-"));
|
||||
|
||||
/// <summary>Every page's layout text, in order.</summary>
|
||||
/// <summary>Every page's visual-row text, in order.</summary>
|
||||
public static string[] Pages(string path)
|
||||
{
|
||||
var pages = Run("pdftotext", "-layout", path, "-").Split('\f');
|
||||
// pdftotext terminates the last page with a form feed.
|
||||
return pages.Take(pages.Length - 1).Select(page => Regex.Replace(page, "[ \t]+\n", "\n")).ToArray();
|
||||
var html = Run("pdftotext", "-bbox", path, "-");
|
||||
return Regex.Matches(html, @"<page\b[^>]*>.*?</page>", RegexOptions.Singleline)
|
||||
.Select(match => LayoutPage(match.Value)).ToArray();
|
||||
}
|
||||
|
||||
private static string LayoutPage(string html)
|
||||
{
|
||||
var rows = new List<(double Top, List<(double Left, string Text)> Words)>();
|
||||
foreach (var word in ParseWords(html).OrderBy(word => word.Top).ThenBy(word => word.Left))
|
||||
{
|
||||
if (rows.Count == 0 || System.Math.Abs(word.Top - rows[^1].Top) > 1.5)
|
||||
rows.Add((word.Top, []));
|
||||
rows[^1].Words.Add((word.Left, word.Text));
|
||||
}
|
||||
return string.Join("\n", rows.Select(row => string.Join(" ", row.Words.OrderBy(word => word.Left).Select(word => word.Text)))) + "\n";
|
||||
}
|
||||
|
||||
/// <summary>Words with page-space boxes (points, Y down).</summary>
|
||||
public static List<(double Left, double Top, double Right, double Bottom, string Text)> Words(string path, int page)
|
||||
public static List<(double Left, double Top, double Right, double Bottom, string Text)> Words(string path, int page) =>
|
||||
ParseWords(Run("pdftotext", "-bbox", "-f", Page(page), "-l", Page(page), path, "-"));
|
||||
|
||||
private static List<(double Left, double Top, double Right, double Bottom, string Text)> ParseWords(string xml)
|
||||
{
|
||||
var xml = Run("pdftotext", "-bbox", "-f", Page(page), "-l", Page(page), path, "-");
|
||||
return Regex.Matches(xml, "<word xMin=\"([\\d.]+)\" yMin=\"([\\d.]+)\" xMax=\"([\\d.]+)\" yMax=\"([\\d.]+)\">([^<]*)</word>")
|
||||
.Select(match => (Number(match.Groups[1].Value), Number(match.Groups[2].Value), Number(match.Groups[3].Value),
|
||||
Number(match.Groups[4].Value), System.Net.WebUtility.HtmlDecode(match.Groups[5].Value)))
|
||||
|
||||
@@ -131,8 +131,9 @@ dotnet test OpenNest.Tests/OpenNest.Tests.csproj --filter FullyQualifiedName~Rep
|
||||
```
|
||||
|
||||
The writer tests read page sizes and content streams with PDFsharp. Text
|
||||
assertions extract with Poppler's `pdftotext -layout` and are skipped, not
|
||||
passed, when poppler-utils is absent. For manual inspection also use `pdfinfo`,
|
||||
assertions use Poppler's `pdftotext -bbox` coordinates to rebuild visual rows;
|
||||
`-layout` can return columns in a different order on Windows. They are skipped,
|
||||
not passed, when poppler-utils is absent. For manual inspection also use `pdfinfo`,
|
||||
`pdffonts` and `pdftoppm -png` (Linux verification baseline: poppler-utils
|
||||
24.02.0). Confirm page sizes, quantity rows, page X of Y, embedded fonts, native
|
||||
vector curves, unfilled holes and visible tab gaps with no connecting stroke.
|
||||
|
||||
Reference in new issue
Block a user