mirror of
https://github.com/UglyToad/PdfPig.git
synced 2026-03-10 00:23:29 +08:00
Some checks failed
Build, test and publish draft / build (push) Has been cancelled
Build and test [MacOS] / build (push) Has been cancelled
Run Common Crawl Tests / build (0000-0001) (push) Has been cancelled
Run Common Crawl Tests / build (0002-0003) (push) Has been cancelled
Run Common Crawl Tests / build (0004-0005) (push) Has been cancelled
Run Common Crawl Tests / build (0006-0007) (push) Has been cancelled
Run Integration Tests / build (push) Has been cancelled
Nightly Release / Check if this commit has already been published (push) Has been cancelled
Nightly Release / tests (push) Has been cancelled
Nightly Release / build_and_publish_nightly (push) Has been cancelled
142 lines
5.4 KiB
C#
142 lines
5.4 KiB
C#
namespace UglyToad.PdfPig.Tests.Integration
|
|
{
|
|
using PdfPig.Geometry;
|
|
|
|
public class IntegrationDocumentTests
|
|
{
|
|
private static readonly Lazy<string> DocumentFolder = new Lazy<string>(() => Path.GetFullPath(Path.Combine(AppDomain.CurrentDomain.BaseDirectory, "..", "..", "..", "Integration", "Documents")));
|
|
private static readonly HashSet<string> _documentsToIgnore =
|
|
[
|
|
"issue_671.pdf",
|
|
"GHOSTSCRIPT-698363-0.pdf",
|
|
"ErcotFacts.pdf",
|
|
"cmap-parsing-exception.pdf"
|
|
];
|
|
|
|
|
|
[Theory]
|
|
[MemberData(nameof(GetAllDocuments))]
|
|
public void CheckGlyphLooseBoundingBoxes(string documentName)
|
|
{
|
|
// Add the full path back on, we removed it so we could see it in the test explorer.
|
|
documentName = Path.Combine(DocumentFolder.Value, documentName);
|
|
|
|
using (var document = PdfDocument.Open(documentName, new ParsingOptions { UseLenientParsing = true }))
|
|
{
|
|
for (var i = 0; i < document.NumberOfPages; i++)
|
|
{
|
|
var page = document.GetPage(i + 1);
|
|
foreach (var letter in page.Letters)
|
|
{
|
|
var bbox = letter.GlyphRectangle;
|
|
if (bbox.Height > 0)
|
|
{
|
|
if (letter.GlyphRectangleLoose.Height <= 0)
|
|
{
|
|
_ = letter.GetFont().GetAscent();
|
|
}
|
|
|
|
Assert.True(letter.GlyphRectangleLoose.Height > 0, $"Page {i + 1}");
|
|
}
|
|
}
|
|
}
|
|
}
|
|
}
|
|
|
|
[Theory]
|
|
[MemberData(nameof(GetAllDocuments))]
|
|
public void CanReadAllPages(string documentName)
|
|
{
|
|
// Add the full path back on, we removed it so we could see it in the test explorer.
|
|
documentName = Path.Combine(DocumentFolder.Value, documentName);
|
|
|
|
using (var document = PdfDocument.Open(documentName, new ParsingOptions { UseLenientParsing = false }))
|
|
{
|
|
for (var i = 0; i < document.NumberOfPages; i++)
|
|
{
|
|
var page = document.GetPage(i + 1);
|
|
|
|
Assert.NotNull(page.GetAnnotations().ToArray());
|
|
}
|
|
}
|
|
}
|
|
|
|
[Theory]
|
|
[MemberData(nameof(GetAllDocuments))]
|
|
public void CanUseStreamForFirstPage(string documentName)
|
|
{
|
|
// Add the full path back on, we removed it so we could see it in the test explorer.
|
|
documentName = Path.Combine(DocumentFolder.Value, documentName);
|
|
|
|
var bytes = File.ReadAllBytes(documentName);
|
|
|
|
using (var memoryStream = new MemoryStream(bytes))
|
|
using (var document = PdfDocument.Open(memoryStream, new ParsingOptions { UseLenientParsing = false }))
|
|
{
|
|
for (var i = 0; i < document.NumberOfPages; i++)
|
|
{
|
|
var page = document.GetPage(i + 1);
|
|
|
|
Assert.NotNull(page.GetAnnotations().ToArray());
|
|
}
|
|
}
|
|
}
|
|
|
|
[Theory]
|
|
[MemberData(nameof(GetAllDocuments))]
|
|
public void CanTokenizeAllAccessibleObjects(string documentName)
|
|
{
|
|
documentName = Path.Combine(DocumentFolder.Value, documentName);
|
|
|
|
using (var document = PdfDocument.Open(documentName, new ParsingOptions { UseLenientParsing = false }))
|
|
{
|
|
Assert.NotNull(document.Structure.Catalog);
|
|
|
|
//Assert.True(document.Structure.CrossReferenceTable.ObjectOffsets.Count > 0, "Cross reference table was empty.");
|
|
//foreach (var objectOffset in document.Structure.CrossReferenceTable.ObjectOffsets)
|
|
//{
|
|
// var token = document.Structure.GetObject(objectOffset.Key);
|
|
|
|
// Assert.NotNull(token);
|
|
//}
|
|
}
|
|
}
|
|
|
|
[Theory]
|
|
[MemberData(nameof(GetAllDocuments))]
|
|
public void CanAccessImagesOnEveryPage(string documentName)
|
|
{
|
|
documentName = Path.Combine(DocumentFolder.Value, documentName);
|
|
|
|
using (var document = PdfDocument.Open(documentName, new ParsingOptions { UseLenientParsing = false }))
|
|
{
|
|
for (var i = 0; i < document.NumberOfPages; i++)
|
|
{
|
|
var page = document.GetPage(i + 1);
|
|
|
|
var images = page.GetImages();
|
|
|
|
Assert.NotNull(images);
|
|
|
|
foreach (var image in images)
|
|
{
|
|
Assert.True(image.WidthInSamples > 0, $"Image had width of zero on page {i + 1}.");
|
|
Assert.True(image.HeightInSamples > 0, $"Image had height of zero on page {i + 1}.");
|
|
}
|
|
}
|
|
}
|
|
}
|
|
|
|
public static IEnumerable<object[]> GetAllDocuments
|
|
{
|
|
get
|
|
{
|
|
var files = Directory.GetFiles(DocumentFolder.Value, "*.pdf");
|
|
|
|
// Return the shortname so we can see it in the test explorer.
|
|
return files.Where(x => !_documentsToIgnore.Any(i => x.EndsWith(i))).Select(x => new object[] { Path.GetFileName(x) });
|
|
}
|
|
}
|
|
}
|
|
}
|