<?php
/**
* SeekQuarry/Yioop --
* Open Source Pure PHP Search Engine, Crawler, and Indexer
*
* Copyright (C) 2009 - 2026 Chris Pollett chris@pollett.org
*
* LICENSE:
*
* This program is free software: you can redistribute it and/or modify
* it under the terms of the GNU General Public License as published by
* the Free Software Foundation, either version 3 of the License, or
* (at your option) any later version.
*
* This program is distributed in the hope that it will be useful,
* but WITHOUT ANY WARRANTY; without even the implied warranty of
* MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
* GNU General Public License for more details.
*
* You should have received a copy of the GNU General Public License
* along with this program. If not, see <https://www.gnu.org/licenses/>.
*
* END LICENSE
*
* @author Chris Pollett chris@pollett.org
* @license https://www.gnu.org/licenses/ GPL3
* @link https://www.seekquarry.com/
* @copyright 2009 - 2026
* @filesource
*/
namespace seekquarry\yioop\tests;
use seekquarry\yioop\configs as C;
use seekquarry\yioop\library as L;
use seekquarry\yioop\library\ComputerVision;
use seekquarry\yioop\library\CrawlConstants;
use seekquarry\yioop\library\processors\JpgProcessor;
use seekquarry\yioop\library\processors\PdfProcessor;
use seekquarry\yioop\library\UnitTest;
/**
* UnitTest for the PdfProcessor class. A PdfProcessor is used to process
* a .pdf file and extract summary from it. This
* class tests the processing of an .pdf file.
*
* @author Chris Pollett
*/
class PdfProcessorTest extends UnitTest implements CrawlConstants
{
/**
* Creates a new PdfProcessor object using the test.pdf
* file made from the seekquarry landing page
*/
public function setUp()
{
}
/**
* Delete any files associated with our test on PdfProcessor (in this case
* none)
*/
public function tearDown()
{
}
/**
* Test case to check whether words known to be in the PDF were extracted
* is retrieved correctly.
*/
public function wordExtractionTestCase()
{
$pdf_object = new PdfProcessor();
$url = "http://www.yioop.com/test.pdf";
$filename = C\PARENT_DIR . "/tests/test_files/test.pdf";
$page = file_get_contents($filename);
$summary = $pdf_object->process($page, $url);
$words = explode(" ", $summary[self::DESCRIPTION]);
$this->assertTrue(in_array("Documentation", $words),
"Word Extraction 1");
$this->assertTrue(in_array("Yioop", $words),
"Word Extraction 2");
$this->assertTrue(in_array("Open", $words),
"Word Extraction 3");
}
/**
* Tests Tessaract text extraction from Images
*/
public function textFromImageTestCase()
{
if (ComputerVision::ocrEnabled()) {
$pdf_object = new PdfProcessor();
$url = "http://www.yioop.com/test2.pdf";
$filename = C\PARENT_DIR . "/tests/test_files/test2.pdf";
$page = file_get_contents($filename);
$summary = $pdf_object->process($page, $url);
$words = explode(" ", $summary[self::DESCRIPTION]);
$this->assertTrue(in_array("Maureen", $words),
"Word From Image Extraction 1");
$this->assertTrue(in_array("Phantom", $words),
"Word From Image Extraction 2");
$this->assertTrue(in_array("playing", $words),
"Word From Image Extraction 3");
}
}
/**
* Where a case that writes files puts them.
* @var string
*/
public $where = "";
/**
* The jpeg reader written here reads an ordinary jpeg the same as the
* one PHP comes with, which is known right. That is what says the
* reading is sound; the four-color case then differs from it only in
* how the values are turned into colors.
*/
public function readsAnOrdinaryJpegTheSameWayTestCase()
{
$made = imagecreatetruecolor(64, 48);
for ($down = 0; $down < 48; $down++) {
for ($across = 0; $across < 64; $across++) {
imagesetpixel($made, $across, $down,
imagecolorallocate($made, $across * 4, $down * 5, 128));
}
}
ob_start();
imagejpeg($made, null, 100);
$jpeg = ob_get_clean();
$read = JpgProcessor::readJpeg($jpeg);
$this->assertTrue($read !== false, "the jpeg is read");
$this->assertEqual($read["width"], 64, "at the right width");
$this->assertEqual(count($read["planes"]), 3, "with three colors");
$theirs = imagecreatefromstring($jpeg);
$worst = 0;
for ($down = 0; $down < 48; $down += 3) {
for ($across = 0; $across < 64; $across += 3) {
$said = imagecolorat($theirs, $across, $down);
$at = $down * 64 + $across;
$light = $read["planes"][0][$at];
$blueness = $read["planes"][1][$at] - 128;
$redness = $read["planes"][2][$at] - 128;
$mine = [max(0, min(255,
(int)round($light + 1.402 * $redness))),
max(0, min(255, (int)round($light -
0.344136 * $blueness - 0.714136 * $redness))),
max(0, min(255,
(int)round($light + 1.772 * $blueness)))];
$theirs_now = [($said >> 16) & 255, ($said >> 8) & 255,
$said & 255];
for ($which = 0; $which < 3; $which++) {
$worst = max($worst,
abs($mine[$which] - $theirs_now[$which]));
}
}
}
$this->assertTrue($worst <= 3,
"every color within rounding of what PHP's reader gives, " .
"worst was $worst");
}
/**
* Makes a picture of a given size to test with.
*
* @param int $width how wide
* @param int $height how tall
* @return string the picture's bytes as a jpeg
*/
public function aPicture($width, $height)
{
$made = imagecreatetruecolor($width, $height);
imagefill($made, 0, 0, imagecolorallocate($made, 30, 90, 200));
ob_start();
imagejpeg($made);
$bytes = ob_get_clean();
return $bytes;
}
/**
* The first picture kept whole inside a portable document is found,
* which is what a scanned page is, and a document holding none says
* so rather than handing back something that is not a picture.
*/
public function pictureInsideADocumentIsFoundTestCase()
{
$picture = $this->aPicture(300, 200);
$document = "%PDF-1.4\n5 0 obj<</Filter/DCTDecode>>stream\n" .
$picture . "\nendstream endobj";
$found = PdfProcessor::pictureInDocument($document);
$this->assertTrue($found !== false, "the picture is found");
$this->assertEqual(substr($found, 0, 2), "\xFF\xD8",
"and it is the picture rather than the words around it");
$this->assertTrue(PdfProcessor::pictureInDocument(
"%PDF-1.4\n5 0 obj<</Filter/FlateDecode>>stream\nxx\nendstream"
) === false, "a document with no such picture says so");
}
}