MINI
C:
/
xampp
/
htdocs
/
.library
/
vendor
/
smalot
/
pdfparser
/
src
/
Smalot
/
PdfParser
/
RawData
/
Upload File:
files >> C:/xampp/htdocs/.library/vendor/smalot/pdfparser/src/Smalot/PdfParser/RawData/FilterHelper.php
<?php /** * This file is based on code of tecnickcom/TCPDF PDF library. * * Original author Nicola Asuni (info@tecnick.com) and * contributors (https://github.com/tecnickcom/TCPDF/graphs/contributors). * * @see https://github.com/tecnickcom/TCPDF * * Original code was licensed on the terms of the LGPL v3. * * ------------------------------------------------------------------------------ * * @file This file is part of the PdfParser library. * * @author Konrad Abicht <k.abicht@gmail.com> * @date 2020-01-06 * * @license LGPLv3 * @url <https://github.com/smalot/pdfparser> * * PdfParser is a pdf library written in PHP, extraction oriented. * Copyright (C) 2017 - Sébastien MALOT <sebastien@malot.fr> * * This program is free software: you can redistribute it and/or modify * it under the terms of the GNU Lesser General Public License as published by * the Free Software Foundation, either version 3 of the License, or * (at your option) any later version. * * This program is distributed in the hope that it will be useful, * but WITHOUT ANY WARRANTY; without even the implied warranty of * MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the * GNU Lesser General Public License for more details. * * You should have received a copy of the GNU Lesser General Public License * along with this program. * If not, see <http://www.pdfparser.org/sites/default/LICENSE.txt>. */ namespace Smalot\PdfParser\RawData; use Exception; class FilterHelper { protected $availableFilters = ['ASCIIHexDecode', 'ASCII85Decode', 'LZWDecode', 'FlateDecode', 'RunLengthDecode']; /** * Decode data using the specified filter type. * * @param string $filter Filter name * @param string $data Data to decode * * @return string Decoded data string * * @throws Exception if a certain decode function is not implemented yet */ public function decodeFilter($filter, $data) { switch ($filter) { case 'ASCIIHexDecode': return $this->decodeFilterASCIIHexDecode($data); case 'ASCII85Decode': return $this->decodeFilterASCII85Decode($data); case 'LZWDecode': return $this->decodeFilterLZWDecode($data); case 'FlateDecode': return $this->decodeFilterFlateDecode($data); case 'RunLengthDecode': return $this->decodeFilterRunLengthDecode($data); case 'CCITTFaxDecode': throw new Exception('Decode CCITTFaxDecode not implemented yet.'); case 'JBIG2Decode': throw new Exception('Decode JBIG2Decode not implemented yet.'); case 'DCTDecode': throw new Exception('Decode DCTDecode not implemented yet.'); case 'JPXDecode': throw new Exception('Decode JPXDecode not implemented yet.'); case 'Crypt': throw new Exception('Decode Crypt not implemented yet.'); default: return $data; } } /** * ASCIIHexDecode * * Decodes data encoded in an ASCII hexadecimal representation, reproducing the original binary data. * * @param string $data Data to decode * * @return string data string */ protected function decodeFilterASCIIHexDecode($data) { // all white-space characters shall be ignored $data = preg_replace('/[\s]/', '', $data); // check for EOD character: GREATER-THAN SIGN (3Eh) $eod = strpos($data, '>'); if (false !== $eod) { // remove EOD and extra data (if any) $data = substr($data, 0, $eod); $eod = true; } // get data length $data_length = \strlen($data); if (0 != ($data_length % 2)) { // odd number of hexadecimal digits if ($eod) { // EOD shall behave as if a 0 (zero) followed the last digit $data = substr($data, 0, -1).'0'.substr($data, -1); } else { throw new Exception('decodeFilterASCIIHexDecode: invalid code'); } } // check for invalid characters if (preg_match('/[^a-fA-F\d]/', $data) > 0) { throw new Exception('decodeFilterASCIIHexDecode: invalid code'); } // get one byte of binary data for each pair of ASCII hexadecimal digits $decoded = pack('H*', $data); return $decoded; } /** * ASCII85Decode * * Decodes data encoded in an ASCII base-85 representation, reproducing the original binary data. * * @param string $data Data to decode * * @return string data string */ protected function decodeFilterASCII85Decode($data) { // initialize string to return $decoded = ''; // all white-space characters shall be ignored $data = preg_replace('/[\s]/', '', $data); // remove start sequence 2-character sequence <~ (3Ch)(7Eh) if (false !== strpos($data, '<~')) { // remove EOD and extra data (if any) $data = substr($data, 2); } // check for EOD: 2-character sequence ~> (7Eh)(3Eh) $eod = strpos($data, '~>'); if (false !== $eod) { // remove EOD and extra data (if any) $data = substr($data, 0, $eod); } // data length $data_length = \strlen($data); // check for invalid characters if (preg_match('/[^\x21-\x75,\x74]/', $data) > 0) { throw new Exception('decodeFilterASCII85Decode: invalid code'); } // z sequence $zseq = \chr(0).\chr(0).\chr(0).\chr(0); // position inside a group of 4 bytes (0-3) $group_pos = 0; $tuple = 0; $pow85 = [(85 * 85 * 85 * 85), (85 * 85 * 85), (85 * 85), 85, 1]; // for each byte for ($i = 0; $i < $data_length; ++$i) { // get char value $char = \ord($data[$i]); if (122 == $char) { // 'z' if (0 == $group_pos) { $decoded .= $zseq; } else { throw new Exception('decodeFilterASCII85Decode: invalid code'); } } else { // the value represented by a group of 5 characters should never be greater than 2^32 - 1 $tuple += (($char - 33) * $pow85[$group_pos]); if (4 == $group_pos) { $decoded .= \chr($tuple >> 24).\chr($tuple >> 16).\chr($tuple >> 8).\chr($tuple); $tuple = 0; $group_pos = 0; } else { ++$group_pos; } } } if ($group_pos > 1) { $tuple += $pow85[($group_pos - 1)]; } // last tuple (if any) switch ($group_pos) { case 4: $decoded .= \chr($tuple >> 24).\chr($tuple >> 16).\chr($tuple >> 8); break; case 3: $decoded .= \chr($tuple >> 24).\chr($tuple >> 16); break; case 2: $decoded .= \chr($tuple >> 24); break; case 1: throw new Exception('decodeFilterASCII85Decode: invalid code'); } return $decoded; } /** * FlateDecode * * Decompresses data encoded using the zlib/deflate compression method, reproducing the original text or binary data. * * @param string $data Data to decode * * @return string data string */ protected function decodeFilterFlateDecode($data) { /* * gzuncompress may throw a not catchable E_WARNING in case of an error (like $data is empty) * the following set_error_handler changes an E_WARNING to an E_ERROR, which is catchable. */ set_error_handler(function ($errNo, $errStr) { if (E_WARNING === $errNo) { throw new Exception($errStr); } else { // fallback to default php error handler return false; } }); // initialize string to return try { $decoded = gzuncompress($data); if (false === $decoded) { throw new Exception('decodeFilterFlateDecode: invalid code'); } } catch (Exception $e) { throw $e; } finally { // Restore old handler just in case it was customized outside of PDFParser. restore_error_handler(); } return $decoded; } /** * LZWDecode * * Decompresses data encoded using the LZW (Lempel-Ziv-Welch) adaptive compression method, reproducing the original text or binary data. * * @param string $data Data to decode * * @return string Data string */ protected function decodeFilterLZWDecode($data) { // initialize string to return $decoded = ''; // data length $data_length = \strlen($data); // convert string to binary string $bitstring = ''; for ($i = 0; $i < $data_length; ++$i) { $bitstring .= sprintf('%08b', \ord($data[$i])); } // get the number of bits $data_length = \strlen($bitstring); // initialize code length in bits $bitlen = 9; // initialize dictionary index $dix = 258; // initialize the dictionary (with the first 256 entries). $dictionary = []; for ($i = 0; $i < 256; ++$i) { $dictionary[$i] = \chr($i); } // previous val $prev_index = 0; // while we encounter EOD marker (257), read code_length bits while (($data_length > 0) and (257 != ($index = bindec(substr($bitstring, 0, $bitlen))))) { // remove read bits from string $bitstring = substr($bitstring, $bitlen); // update number of bits $data_length -= $bitlen; if (256 == $index) { // clear-table marker // reset code length in bits $bitlen = 9; // reset dictionary index $dix = 258; $prev_index = 256; // reset the dictionary (with the first 256 entries). $dictionary = []; for ($i = 0; $i < 256; ++$i) { $dictionary[$i] = \chr($i); } } elseif (256 == $prev_index) { // first entry $decoded .= $dictionary[$index]; $prev_index = $index; } else { // check if index exist in the dictionary if ($index < $dix) { // index exist on dictionary $decoded .= $dictionary[$index]; $dic_val = $dictionary[$prev_index].$dictionary[$index][0]; // store current index $prev_index = $index; } else { // index do not exist on dictionary $dic_val = $dictionary[$prev_index].$dictionary[$prev_index][0]; $decoded .= $dic_val; } // update dictionary $dictionary[$dix] = $dic_val; ++$dix; // change bit length by case if (2047 == $dix) { $bitlen = 12; } elseif (1023 == $dix) { $bitlen = 11; } elseif (511 == $dix) { $bitlen = 10; } } } return $decoded; } /** * RunLengthDecode * * Decompresses data encoded using a byte-oriented run-length encoding algorithm. * * @param string $data Data to decode * * @return string */ protected function decodeFilterRunLengthDecode($data) { // initialize string to return $decoded = ''; // data length $data_length = \strlen($data); $i = 0; while ($i < $data_length) { // get current byte value $byte = \ord($data[$i]); if (128 == $byte) { // a length value of 128 denote EOD break; } elseif ($byte < 128) { // if the length byte is in the range 0 to 127 // the following length + 1 (1 to 128) bytes shall be copied literally during decompression $decoded .= substr($data, ($i + 1), ($byte + 1)); // move to next block $i += ($byte + 2); } else { // if length is in the range 129 to 255, // the following single byte shall be copied 257 - length (2 to 128) times during decompression $decoded .= str_repeat($data[($i + 1)], (257 - $byte)); // move to next block $i += 2; } } return $decoded; } /** * @return array list of available filters */ public function getAvailableFilters() { return $this->availableFilters; } }
Copyright ©2021 || Defacer Indonesia