Initial import
This commit is contained in:
Vendored
+16
@@ -0,0 +1,16 @@
|
||||
<?php
|
||||
$type = 'Core';
|
||||
$name = 'Courier';
|
||||
$up = -100;
|
||||
$ut = 50;
|
||||
for($i=0;$i<=255;$i++)
|
||||
$cw[chr($i)] = 600;
|
||||
$enc = 'cp1252';
|
||||
$uv = array(0=>array(0,128),128=>8364,130=>8218,131=>402,132=>8222,133=>8230,134=>array(8224,2),136=>710,137=>8240,138=>352,139=>8249,140=>338,142=>381,145=>array(8216,2),147=>array(8220,2),149=>8226,150=>array(8211,2),152=>732,153=>8482,154=>353,155=>8250,156=>339,158=>382,159=>376,160=>array(160,96));
|
||||
$desc = '';
|
||||
$diff = '';
|
||||
$file = '';
|
||||
$size1 = 0;
|
||||
$size2 = 0;
|
||||
$originalsize = 0;
|
||||
?>
|
||||
Vendored
+16
@@ -0,0 +1,16 @@
|
||||
<?php
|
||||
$type = 'Core';
|
||||
$name = 'Courier-Bold';
|
||||
$up = -100;
|
||||
$ut = 50;
|
||||
for($i=0;$i<=255;$i++)
|
||||
$cw[chr($i)] = 600;
|
||||
$enc = 'cp1252';
|
||||
$uv = array(0=>array(0,128),128=>8364,130=>8218,131=>402,132=>8222,133=>8230,134=>array(8224,2),136=>710,137=>8240,138=>352,139=>8249,140=>338,142=>381,145=>array(8216,2),147=>array(8220,2),149=>8226,150=>array(8211,2),152=>732,153=>8482,154=>353,155=>8250,156=>339,158=>382,159=>376,160=>array(160,96));
|
||||
$desc = '';
|
||||
$diff = '';
|
||||
$file = '';
|
||||
$size1 = 0;
|
||||
$size2 = 0;
|
||||
$originalsize = 0;
|
||||
?>
|
||||
Vendored
+16
@@ -0,0 +1,16 @@
|
||||
<?php
|
||||
$type = 'Core';
|
||||
$name = 'Courier-BoldOblique';
|
||||
$up = -100;
|
||||
$ut = 50;
|
||||
for($i=0;$i<=255;$i++)
|
||||
$cw[chr($i)] = 600;
|
||||
$enc = 'cp1252';
|
||||
$uv = array(0=>array(0,128),128=>8364,130=>8218,131=>402,132=>8222,133=>8230,134=>array(8224,2),136=>710,137=>8240,138=>352,139=>8249,140=>338,142=>381,145=>array(8216,2),147=>array(8220,2),149=>8226,150=>array(8211,2),152=>732,153=>8482,154=>353,155=>8250,156=>339,158=>382,159=>376,160=>array(160,96));
|
||||
$desc = '';
|
||||
$diff = '';
|
||||
$file = '';
|
||||
$size1 = 0;
|
||||
$size2 = 0;
|
||||
$originalsize = 0;
|
||||
?>
|
||||
Vendored
+16
@@ -0,0 +1,16 @@
|
||||
<?php
|
||||
$type = 'Core';
|
||||
$name = 'Courier-Oblique';
|
||||
$up = -100;
|
||||
$ut = 50;
|
||||
for($i=0;$i<=255;$i++)
|
||||
$cw[chr($i)] = 600;
|
||||
$enc = 'cp1252';
|
||||
$uv = array(0=>array(0,128),128=>8364,130=>8218,131=>402,132=>8222,133=>8230,134=>array(8224,2),136=>710,137=>8240,138=>352,139=>8249,140=>338,142=>381,145=>array(8216,2),147=>array(8220,2),149=>8226,150=>array(8211,2),152=>732,153=>8482,154=>353,155=>8250,156=>339,158=>382,159=>376,160=>array(160,96));
|
||||
$desc = '';
|
||||
$diff = '';
|
||||
$file = '';
|
||||
$size1 = 0;
|
||||
$size2 = 0;
|
||||
$originalsize = 0;
|
||||
?>
|
||||
Vendored
+27
@@ -0,0 +1,27 @@
|
||||
<?php
|
||||
$type = 'Core';
|
||||
$name = 'Helvetica';
|
||||
$up = -100;
|
||||
$ut = 50;
|
||||
$cw = array(
|
||||
chr(0)=>278,chr(1)=>278,chr(2)=>278,chr(3)=>278,chr(4)=>278,chr(5)=>278,chr(6)=>278,chr(7)=>278,chr(8)=>278,chr(9)=>278,chr(10)=>278,chr(11)=>278,chr(12)=>278,chr(13)=>278,chr(14)=>278,chr(15)=>278,chr(16)=>278,chr(17)=>278,chr(18)=>278,chr(19)=>278,chr(20)=>278,chr(21)=>278,
|
||||
chr(22)=>278,chr(23)=>278,chr(24)=>278,chr(25)=>278,chr(26)=>278,chr(27)=>278,chr(28)=>278,chr(29)=>278,chr(30)=>278,chr(31)=>278,' '=>278,'!'=>278,'"'=>355,'#'=>556,'$'=>556,'%'=>889,'&'=>667,'\''=>191,'('=>333,')'=>333,'*'=>389,'+'=>584,
|
||||
','=>278,'-'=>333,'.'=>278,'/'=>278,'0'=>556,'1'=>556,'2'=>556,'3'=>556,'4'=>556,'5'=>556,'6'=>556,'7'=>556,'8'=>556,'9'=>556,':'=>278,';'=>278,'<'=>584,'='=>584,'>'=>584,'?'=>556,'@'=>1015,'A'=>667,
|
||||
'B'=>667,'C'=>722,'D'=>722,'E'=>667,'F'=>611,'G'=>778,'H'=>722,'I'=>278,'J'=>500,'K'=>667,'L'=>556,'M'=>833,'N'=>722,'O'=>778,'P'=>667,'Q'=>778,'R'=>722,'S'=>667,'T'=>611,'U'=>722,'V'=>667,'W'=>944,
|
||||
'X'=>667,'Y'=>667,'Z'=>611,'['=>278,'\\'=>278,']'=>278,'^'=>469,'_'=>556,'`'=>333,'a'=>556,'b'=>556,'c'=>500,'d'=>556,'e'=>556,'f'=>278,'g'=>556,'h'=>556,'i'=>222,'j'=>222,'k'=>500,'l'=>222,'m'=>833,
|
||||
'n'=>556,'o'=>556,'p'=>556,'q'=>556,'r'=>333,'s'=>500,'t'=>278,'u'=>556,'v'=>500,'w'=>722,'x'=>500,'y'=>500,'z'=>500,'{'=>334,'|'=>260,'}'=>334,'~'=>584,chr(127)=>350,chr(128)=>556,chr(129)=>350,chr(130)=>222,chr(131)=>556,
|
||||
chr(132)=>333,chr(133)=>1000,chr(134)=>556,chr(135)=>556,chr(136)=>333,chr(137)=>1000,chr(138)=>667,chr(139)=>333,chr(140)=>1000,chr(141)=>350,chr(142)=>611,chr(143)=>350,chr(144)=>350,chr(145)=>222,chr(146)=>222,chr(147)=>333,chr(148)=>333,chr(149)=>350,chr(150)=>556,chr(151)=>1000,chr(152)=>333,chr(153)=>1000,
|
||||
chr(154)=>500,chr(155)=>333,chr(156)=>944,chr(157)=>350,chr(158)=>500,chr(159)=>667,chr(160)=>278,chr(161)=>333,chr(162)=>556,chr(163)=>556,chr(164)=>556,chr(165)=>556,chr(166)=>260,chr(167)=>556,chr(168)=>333,chr(169)=>737,chr(170)=>370,chr(171)=>556,chr(172)=>584,chr(173)=>333,chr(174)=>737,chr(175)=>333,
|
||||
chr(176)=>400,chr(177)=>584,chr(178)=>333,chr(179)=>333,chr(180)=>333,chr(181)=>556,chr(182)=>537,chr(183)=>278,chr(184)=>333,chr(185)=>333,chr(186)=>365,chr(187)=>556,chr(188)=>834,chr(189)=>834,chr(190)=>834,chr(191)=>611,chr(192)=>667,chr(193)=>667,chr(194)=>667,chr(195)=>667,chr(196)=>667,chr(197)=>667,
|
||||
chr(198)=>1000,chr(199)=>722,chr(200)=>667,chr(201)=>667,chr(202)=>667,chr(203)=>667,chr(204)=>278,chr(205)=>278,chr(206)=>278,chr(207)=>278,chr(208)=>722,chr(209)=>722,chr(210)=>778,chr(211)=>778,chr(212)=>778,chr(213)=>778,chr(214)=>778,chr(215)=>584,chr(216)=>778,chr(217)=>722,chr(218)=>722,chr(219)=>722,
|
||||
chr(220)=>722,chr(221)=>667,chr(222)=>667,chr(223)=>611,chr(224)=>556,chr(225)=>556,chr(226)=>556,chr(227)=>556,chr(228)=>556,chr(229)=>556,chr(230)=>889,chr(231)=>500,chr(232)=>556,chr(233)=>556,chr(234)=>556,chr(235)=>556,chr(236)=>278,chr(237)=>278,chr(238)=>278,chr(239)=>278,chr(240)=>556,chr(241)=>556,
|
||||
chr(242)=>556,chr(243)=>556,chr(244)=>556,chr(245)=>556,chr(246)=>556,chr(247)=>584,chr(248)=>611,chr(249)=>556,chr(250)=>556,chr(251)=>556,chr(252)=>556,chr(253)=>500,chr(254)=>556,chr(255)=>500);
|
||||
$enc = 'cp1252';
|
||||
$uv = array(0=>array(0,128),128=>8364,130=>8218,131=>402,132=>8222,133=>8230,134=>array(8224,2),136=>710,137=>8240,138=>352,139=>8249,140=>338,142=>381,145=>array(8216,2),147=>array(8220,2),149=>8226,150=>array(8211,2),152=>732,153=>8482,154=>353,155=>8250,156=>339,158=>382,159=>376,160=>array(160,96));
|
||||
$desc = '';
|
||||
$diff = '';
|
||||
$file = '';
|
||||
$size1 = 0;
|
||||
$size2 = 0;
|
||||
$originalsize = 0;
|
||||
?>
|
||||
Vendored
+27
@@ -0,0 +1,27 @@
|
||||
<?php
|
||||
$type = 'Core';
|
||||
$name = 'Helvetica-Bold';
|
||||
$up = -100;
|
||||
$ut = 50;
|
||||
$cw = array(
|
||||
chr(0)=>278,chr(1)=>278,chr(2)=>278,chr(3)=>278,chr(4)=>278,chr(5)=>278,chr(6)=>278,chr(7)=>278,chr(8)=>278,chr(9)=>278,chr(10)=>278,chr(11)=>278,chr(12)=>278,chr(13)=>278,chr(14)=>278,chr(15)=>278,chr(16)=>278,chr(17)=>278,chr(18)=>278,chr(19)=>278,chr(20)=>278,chr(21)=>278,
|
||||
chr(22)=>278,chr(23)=>278,chr(24)=>278,chr(25)=>278,chr(26)=>278,chr(27)=>278,chr(28)=>278,chr(29)=>278,chr(30)=>278,chr(31)=>278,' '=>278,'!'=>333,'"'=>474,'#'=>556,'$'=>556,'%'=>889,'&'=>722,'\''=>238,'('=>333,')'=>333,'*'=>389,'+'=>584,
|
||||
','=>278,'-'=>333,'.'=>278,'/'=>278,'0'=>556,'1'=>556,'2'=>556,'3'=>556,'4'=>556,'5'=>556,'6'=>556,'7'=>556,'8'=>556,'9'=>556,':'=>333,';'=>333,'<'=>584,'='=>584,'>'=>584,'?'=>611,'@'=>975,'A'=>722,
|
||||
'B'=>722,'C'=>722,'D'=>722,'E'=>667,'F'=>611,'G'=>778,'H'=>722,'I'=>278,'J'=>556,'K'=>722,'L'=>611,'M'=>833,'N'=>722,'O'=>778,'P'=>667,'Q'=>778,'R'=>722,'S'=>667,'T'=>611,'U'=>722,'V'=>667,'W'=>944,
|
||||
'X'=>667,'Y'=>667,'Z'=>611,'['=>333,'\\'=>278,']'=>333,'^'=>584,'_'=>556,'`'=>333,'a'=>556,'b'=>611,'c'=>556,'d'=>611,'e'=>556,'f'=>333,'g'=>611,'h'=>611,'i'=>278,'j'=>278,'k'=>556,'l'=>278,'m'=>889,
|
||||
'n'=>611,'o'=>611,'p'=>611,'q'=>611,'r'=>389,'s'=>556,'t'=>333,'u'=>611,'v'=>556,'w'=>778,'x'=>556,'y'=>556,'z'=>500,'{'=>389,'|'=>280,'}'=>389,'~'=>584,chr(127)=>350,chr(128)=>556,chr(129)=>350,chr(130)=>278,chr(131)=>556,
|
||||
chr(132)=>500,chr(133)=>1000,chr(134)=>556,chr(135)=>556,chr(136)=>333,chr(137)=>1000,chr(138)=>667,chr(139)=>333,chr(140)=>1000,chr(141)=>350,chr(142)=>611,chr(143)=>350,chr(144)=>350,chr(145)=>278,chr(146)=>278,chr(147)=>500,chr(148)=>500,chr(149)=>350,chr(150)=>556,chr(151)=>1000,chr(152)=>333,chr(153)=>1000,
|
||||
chr(154)=>556,chr(155)=>333,chr(156)=>944,chr(157)=>350,chr(158)=>500,chr(159)=>667,chr(160)=>278,chr(161)=>333,chr(162)=>556,chr(163)=>556,chr(164)=>556,chr(165)=>556,chr(166)=>280,chr(167)=>556,chr(168)=>333,chr(169)=>737,chr(170)=>370,chr(171)=>556,chr(172)=>584,chr(173)=>333,chr(174)=>737,chr(175)=>333,
|
||||
chr(176)=>400,chr(177)=>584,chr(178)=>333,chr(179)=>333,chr(180)=>333,chr(181)=>611,chr(182)=>556,chr(183)=>278,chr(184)=>333,chr(185)=>333,chr(186)=>365,chr(187)=>556,chr(188)=>834,chr(189)=>834,chr(190)=>834,chr(191)=>611,chr(192)=>722,chr(193)=>722,chr(194)=>722,chr(195)=>722,chr(196)=>722,chr(197)=>722,
|
||||
chr(198)=>1000,chr(199)=>722,chr(200)=>667,chr(201)=>667,chr(202)=>667,chr(203)=>667,chr(204)=>278,chr(205)=>278,chr(206)=>278,chr(207)=>278,chr(208)=>722,chr(209)=>722,chr(210)=>778,chr(211)=>778,chr(212)=>778,chr(213)=>778,chr(214)=>778,chr(215)=>584,chr(216)=>778,chr(217)=>722,chr(218)=>722,chr(219)=>722,
|
||||
chr(220)=>722,chr(221)=>667,chr(222)=>667,chr(223)=>611,chr(224)=>556,chr(225)=>556,chr(226)=>556,chr(227)=>556,chr(228)=>556,chr(229)=>556,chr(230)=>889,chr(231)=>556,chr(232)=>556,chr(233)=>556,chr(234)=>556,chr(235)=>556,chr(236)=>278,chr(237)=>278,chr(238)=>278,chr(239)=>278,chr(240)=>611,chr(241)=>611,
|
||||
chr(242)=>611,chr(243)=>611,chr(244)=>611,chr(245)=>611,chr(246)=>611,chr(247)=>584,chr(248)=>611,chr(249)=>611,chr(250)=>611,chr(251)=>611,chr(252)=>611,chr(253)=>556,chr(254)=>611,chr(255)=>556);
|
||||
$enc = 'cp1252';
|
||||
$uv = array(0=>array(0,128),128=>8364,130=>8218,131=>402,132=>8222,133=>8230,134=>array(8224,2),136=>710,137=>8240,138=>352,139=>8249,140=>338,142=>381,145=>array(8216,2),147=>array(8220,2),149=>8226,150=>array(8211,2),152=>732,153=>8482,154=>353,155=>8250,156=>339,158=>382,159=>376,160=>array(160,96));
|
||||
$desc = '';
|
||||
$diff = '';
|
||||
$file = '';
|
||||
$size1 = 0;
|
||||
$size2 = 0;
|
||||
$originalsize = 0;
|
||||
?>
|
||||
Vendored
+27
@@ -0,0 +1,27 @@
|
||||
<?php
|
||||
$type = 'Core';
|
||||
$name = 'Helvetica-BoldOblique';
|
||||
$up = -100;
|
||||
$ut = 50;
|
||||
$cw = array(
|
||||
chr(0)=>278,chr(1)=>278,chr(2)=>278,chr(3)=>278,chr(4)=>278,chr(5)=>278,chr(6)=>278,chr(7)=>278,chr(8)=>278,chr(9)=>278,chr(10)=>278,chr(11)=>278,chr(12)=>278,chr(13)=>278,chr(14)=>278,chr(15)=>278,chr(16)=>278,chr(17)=>278,chr(18)=>278,chr(19)=>278,chr(20)=>278,chr(21)=>278,
|
||||
chr(22)=>278,chr(23)=>278,chr(24)=>278,chr(25)=>278,chr(26)=>278,chr(27)=>278,chr(28)=>278,chr(29)=>278,chr(30)=>278,chr(31)=>278,' '=>278,'!'=>333,'"'=>474,'#'=>556,'$'=>556,'%'=>889,'&'=>722,'\''=>238,'('=>333,')'=>333,'*'=>389,'+'=>584,
|
||||
','=>278,'-'=>333,'.'=>278,'/'=>278,'0'=>556,'1'=>556,'2'=>556,'3'=>556,'4'=>556,'5'=>556,'6'=>556,'7'=>556,'8'=>556,'9'=>556,':'=>333,';'=>333,'<'=>584,'='=>584,'>'=>584,'?'=>611,'@'=>975,'A'=>722,
|
||||
'B'=>722,'C'=>722,'D'=>722,'E'=>667,'F'=>611,'G'=>778,'H'=>722,'I'=>278,'J'=>556,'K'=>722,'L'=>611,'M'=>833,'N'=>722,'O'=>778,'P'=>667,'Q'=>778,'R'=>722,'S'=>667,'T'=>611,'U'=>722,'V'=>667,'W'=>944,
|
||||
'X'=>667,'Y'=>667,'Z'=>611,'['=>333,'\\'=>278,']'=>333,'^'=>584,'_'=>556,'`'=>333,'a'=>556,'b'=>611,'c'=>556,'d'=>611,'e'=>556,'f'=>333,'g'=>611,'h'=>611,'i'=>278,'j'=>278,'k'=>556,'l'=>278,'m'=>889,
|
||||
'n'=>611,'o'=>611,'p'=>611,'q'=>611,'r'=>389,'s'=>556,'t'=>333,'u'=>611,'v'=>556,'w'=>778,'x'=>556,'y'=>556,'z'=>500,'{'=>389,'|'=>280,'}'=>389,'~'=>584,chr(127)=>350,chr(128)=>556,chr(129)=>350,chr(130)=>278,chr(131)=>556,
|
||||
chr(132)=>500,chr(133)=>1000,chr(134)=>556,chr(135)=>556,chr(136)=>333,chr(137)=>1000,chr(138)=>667,chr(139)=>333,chr(140)=>1000,chr(141)=>350,chr(142)=>611,chr(143)=>350,chr(144)=>350,chr(145)=>278,chr(146)=>278,chr(147)=>500,chr(148)=>500,chr(149)=>350,chr(150)=>556,chr(151)=>1000,chr(152)=>333,chr(153)=>1000,
|
||||
chr(154)=>556,chr(155)=>333,chr(156)=>944,chr(157)=>350,chr(158)=>500,chr(159)=>667,chr(160)=>278,chr(161)=>333,chr(162)=>556,chr(163)=>556,chr(164)=>556,chr(165)=>556,chr(166)=>280,chr(167)=>556,chr(168)=>333,chr(169)=>737,chr(170)=>370,chr(171)=>556,chr(172)=>584,chr(173)=>333,chr(174)=>737,chr(175)=>333,
|
||||
chr(176)=>400,chr(177)=>584,chr(178)=>333,chr(179)=>333,chr(180)=>333,chr(181)=>611,chr(182)=>556,chr(183)=>278,chr(184)=>333,chr(185)=>333,chr(186)=>365,chr(187)=>556,chr(188)=>834,chr(189)=>834,chr(190)=>834,chr(191)=>611,chr(192)=>722,chr(193)=>722,chr(194)=>722,chr(195)=>722,chr(196)=>722,chr(197)=>722,
|
||||
chr(198)=>1000,chr(199)=>722,chr(200)=>667,chr(201)=>667,chr(202)=>667,chr(203)=>667,chr(204)=>278,chr(205)=>278,chr(206)=>278,chr(207)=>278,chr(208)=>722,chr(209)=>722,chr(210)=>778,chr(211)=>778,chr(212)=>778,chr(213)=>778,chr(214)=>778,chr(215)=>584,chr(216)=>778,chr(217)=>722,chr(218)=>722,chr(219)=>722,
|
||||
chr(220)=>722,chr(221)=>667,chr(222)=>667,chr(223)=>611,chr(224)=>556,chr(225)=>556,chr(226)=>556,chr(227)=>556,chr(228)=>556,chr(229)=>556,chr(230)=>889,chr(231)=>556,chr(232)=>556,chr(233)=>556,chr(234)=>556,chr(235)=>556,chr(236)=>278,chr(237)=>278,chr(238)=>278,chr(239)=>278,chr(240)=>611,chr(241)=>611,
|
||||
chr(242)=>611,chr(243)=>611,chr(244)=>611,chr(245)=>611,chr(246)=>611,chr(247)=>584,chr(248)=>611,chr(249)=>611,chr(250)=>611,chr(251)=>611,chr(252)=>611,chr(253)=>556,chr(254)=>611,chr(255)=>556);
|
||||
$enc = 'cp1252';
|
||||
$uv = array(0=>array(0,128),128=>8364,130=>8218,131=>402,132=>8222,133=>8230,134=>array(8224,2),136=>710,137=>8240,138=>352,139=>8249,140=>338,142=>381,145=>array(8216,2),147=>array(8220,2),149=>8226,150=>array(8211,2),152=>732,153=>8482,154=>353,155=>8250,156=>339,158=>382,159=>376,160=>array(160,96));
|
||||
$desc = '';
|
||||
$diff = '';
|
||||
$file = '';
|
||||
$size1 = 0;
|
||||
$size2 = 0;
|
||||
$originalsize = 0;
|
||||
?>
|
||||
Vendored
+27
@@ -0,0 +1,27 @@
|
||||
<?php
|
||||
$type = 'Core';
|
||||
$name = 'Helvetica-Oblique';
|
||||
$up = -100;
|
||||
$ut = 50;
|
||||
$cw = array(
|
||||
chr(0)=>278,chr(1)=>278,chr(2)=>278,chr(3)=>278,chr(4)=>278,chr(5)=>278,chr(6)=>278,chr(7)=>278,chr(8)=>278,chr(9)=>278,chr(10)=>278,chr(11)=>278,chr(12)=>278,chr(13)=>278,chr(14)=>278,chr(15)=>278,chr(16)=>278,chr(17)=>278,chr(18)=>278,chr(19)=>278,chr(20)=>278,chr(21)=>278,
|
||||
chr(22)=>278,chr(23)=>278,chr(24)=>278,chr(25)=>278,chr(26)=>278,chr(27)=>278,chr(28)=>278,chr(29)=>278,chr(30)=>278,chr(31)=>278,' '=>278,'!'=>278,'"'=>355,'#'=>556,'$'=>556,'%'=>889,'&'=>667,'\''=>191,'('=>333,')'=>333,'*'=>389,'+'=>584,
|
||||
','=>278,'-'=>333,'.'=>278,'/'=>278,'0'=>556,'1'=>556,'2'=>556,'3'=>556,'4'=>556,'5'=>556,'6'=>556,'7'=>556,'8'=>556,'9'=>556,':'=>278,';'=>278,'<'=>584,'='=>584,'>'=>584,'?'=>556,'@'=>1015,'A'=>667,
|
||||
'B'=>667,'C'=>722,'D'=>722,'E'=>667,'F'=>611,'G'=>778,'H'=>722,'I'=>278,'J'=>500,'K'=>667,'L'=>556,'M'=>833,'N'=>722,'O'=>778,'P'=>667,'Q'=>778,'R'=>722,'S'=>667,'T'=>611,'U'=>722,'V'=>667,'W'=>944,
|
||||
'X'=>667,'Y'=>667,'Z'=>611,'['=>278,'\\'=>278,']'=>278,'^'=>469,'_'=>556,'`'=>333,'a'=>556,'b'=>556,'c'=>500,'d'=>556,'e'=>556,'f'=>278,'g'=>556,'h'=>556,'i'=>222,'j'=>222,'k'=>500,'l'=>222,'m'=>833,
|
||||
'n'=>556,'o'=>556,'p'=>556,'q'=>556,'r'=>333,'s'=>500,'t'=>278,'u'=>556,'v'=>500,'w'=>722,'x'=>500,'y'=>500,'z'=>500,'{'=>334,'|'=>260,'}'=>334,'~'=>584,chr(127)=>350,chr(128)=>556,chr(129)=>350,chr(130)=>222,chr(131)=>556,
|
||||
chr(132)=>333,chr(133)=>1000,chr(134)=>556,chr(135)=>556,chr(136)=>333,chr(137)=>1000,chr(138)=>667,chr(139)=>333,chr(140)=>1000,chr(141)=>350,chr(142)=>611,chr(143)=>350,chr(144)=>350,chr(145)=>222,chr(146)=>222,chr(147)=>333,chr(148)=>333,chr(149)=>350,chr(150)=>556,chr(151)=>1000,chr(152)=>333,chr(153)=>1000,
|
||||
chr(154)=>500,chr(155)=>333,chr(156)=>944,chr(157)=>350,chr(158)=>500,chr(159)=>667,chr(160)=>278,chr(161)=>333,chr(162)=>556,chr(163)=>556,chr(164)=>556,chr(165)=>556,chr(166)=>260,chr(167)=>556,chr(168)=>333,chr(169)=>737,chr(170)=>370,chr(171)=>556,chr(172)=>584,chr(173)=>333,chr(174)=>737,chr(175)=>333,
|
||||
chr(176)=>400,chr(177)=>584,chr(178)=>333,chr(179)=>333,chr(180)=>333,chr(181)=>556,chr(182)=>537,chr(183)=>278,chr(184)=>333,chr(185)=>333,chr(186)=>365,chr(187)=>556,chr(188)=>834,chr(189)=>834,chr(190)=>834,chr(191)=>611,chr(192)=>667,chr(193)=>667,chr(194)=>667,chr(195)=>667,chr(196)=>667,chr(197)=>667,
|
||||
chr(198)=>1000,chr(199)=>722,chr(200)=>667,chr(201)=>667,chr(202)=>667,chr(203)=>667,chr(204)=>278,chr(205)=>278,chr(206)=>278,chr(207)=>278,chr(208)=>722,chr(209)=>722,chr(210)=>778,chr(211)=>778,chr(212)=>778,chr(213)=>778,chr(214)=>778,chr(215)=>584,chr(216)=>778,chr(217)=>722,chr(218)=>722,chr(219)=>722,
|
||||
chr(220)=>722,chr(221)=>667,chr(222)=>667,chr(223)=>611,chr(224)=>556,chr(225)=>556,chr(226)=>556,chr(227)=>556,chr(228)=>556,chr(229)=>556,chr(230)=>889,chr(231)=>500,chr(232)=>556,chr(233)=>556,chr(234)=>556,chr(235)=>556,chr(236)=>278,chr(237)=>278,chr(238)=>278,chr(239)=>278,chr(240)=>556,chr(241)=>556,
|
||||
chr(242)=>556,chr(243)=>556,chr(244)=>556,chr(245)=>556,chr(246)=>556,chr(247)=>584,chr(248)=>611,chr(249)=>556,chr(250)=>556,chr(251)=>556,chr(252)=>556,chr(253)=>500,chr(254)=>556,chr(255)=>500);
|
||||
$enc = 'cp1252';
|
||||
$uv = array(0=>array(0,128),128=>8364,130=>8218,131=>402,132=>8222,133=>8230,134=>array(8224,2),136=>710,137=>8240,138=>352,139=>8249,140=>338,142=>381,145=>array(8216,2),147=>array(8220,2),149=>8226,150=>array(8211,2),152=>732,153=>8482,154=>353,155=>8250,156=>339,158=>382,159=>376,160=>array(160,96));
|
||||
$desc = '';
|
||||
$diff = '';
|
||||
$file = '';
|
||||
$size1 = 0;
|
||||
$size2 = 0;
|
||||
$originalsize = 0;
|
||||
?>
|
||||
Vendored
+26
@@ -0,0 +1,26 @@
|
||||
<?php
|
||||
$type = 'Core';
|
||||
$name = 'Symbol';
|
||||
$up = -100;
|
||||
$ut = 50;
|
||||
$cw = array(
|
||||
chr(0)=>250,chr(1)=>250,chr(2)=>250,chr(3)=>250,chr(4)=>250,chr(5)=>250,chr(6)=>250,chr(7)=>250,chr(8)=>250,chr(9)=>250,chr(10)=>250,chr(11)=>250,chr(12)=>250,chr(13)=>250,chr(14)=>250,chr(15)=>250,chr(16)=>250,chr(17)=>250,chr(18)=>250,chr(19)=>250,chr(20)=>250,chr(21)=>250,
|
||||
chr(22)=>250,chr(23)=>250,chr(24)=>250,chr(25)=>250,chr(26)=>250,chr(27)=>250,chr(28)=>250,chr(29)=>250,chr(30)=>250,chr(31)=>250,' '=>250,'!'=>333,'"'=>713,'#'=>500,'$'=>549,'%'=>833,'&'=>778,'\''=>439,'('=>333,')'=>333,'*'=>500,'+'=>549,
|
||||
','=>250,'-'=>549,'.'=>250,'/'=>278,'0'=>500,'1'=>500,'2'=>500,'3'=>500,'4'=>500,'5'=>500,'6'=>500,'7'=>500,'8'=>500,'9'=>500,':'=>278,';'=>278,'<'=>549,'='=>549,'>'=>549,'?'=>444,'@'=>549,'A'=>722,
|
||||
'B'=>667,'C'=>722,'D'=>612,'E'=>611,'F'=>763,'G'=>603,'H'=>722,'I'=>333,'J'=>631,'K'=>722,'L'=>686,'M'=>889,'N'=>722,'O'=>722,'P'=>768,'Q'=>741,'R'=>556,'S'=>592,'T'=>611,'U'=>690,'V'=>439,'W'=>768,
|
||||
'X'=>645,'Y'=>795,'Z'=>611,'['=>333,'\\'=>863,']'=>333,'^'=>658,'_'=>500,'`'=>500,'a'=>631,'b'=>549,'c'=>549,'d'=>494,'e'=>439,'f'=>521,'g'=>411,'h'=>603,'i'=>329,'j'=>603,'k'=>549,'l'=>549,'m'=>576,
|
||||
'n'=>521,'o'=>549,'p'=>549,'q'=>521,'r'=>549,'s'=>603,'t'=>439,'u'=>576,'v'=>713,'w'=>686,'x'=>493,'y'=>686,'z'=>494,'{'=>480,'|'=>200,'}'=>480,'~'=>549,chr(127)=>0,chr(128)=>0,chr(129)=>0,chr(130)=>0,chr(131)=>0,
|
||||
chr(132)=>0,chr(133)=>0,chr(134)=>0,chr(135)=>0,chr(136)=>0,chr(137)=>0,chr(138)=>0,chr(139)=>0,chr(140)=>0,chr(141)=>0,chr(142)=>0,chr(143)=>0,chr(144)=>0,chr(145)=>0,chr(146)=>0,chr(147)=>0,chr(148)=>0,chr(149)=>0,chr(150)=>0,chr(151)=>0,chr(152)=>0,chr(153)=>0,
|
||||
chr(154)=>0,chr(155)=>0,chr(156)=>0,chr(157)=>0,chr(158)=>0,chr(159)=>0,chr(160)=>750,chr(161)=>620,chr(162)=>247,chr(163)=>549,chr(164)=>167,chr(165)=>713,chr(166)=>500,chr(167)=>753,chr(168)=>753,chr(169)=>753,chr(170)=>753,chr(171)=>1042,chr(172)=>987,chr(173)=>603,chr(174)=>987,chr(175)=>603,
|
||||
chr(176)=>400,chr(177)=>549,chr(178)=>411,chr(179)=>549,chr(180)=>549,chr(181)=>713,chr(182)=>494,chr(183)=>460,chr(184)=>549,chr(185)=>549,chr(186)=>549,chr(187)=>549,chr(188)=>1000,chr(189)=>603,chr(190)=>1000,chr(191)=>658,chr(192)=>823,chr(193)=>686,chr(194)=>795,chr(195)=>987,chr(196)=>768,chr(197)=>768,
|
||||
chr(198)=>823,chr(199)=>768,chr(200)=>768,chr(201)=>713,chr(202)=>713,chr(203)=>713,chr(204)=>713,chr(205)=>713,chr(206)=>713,chr(207)=>713,chr(208)=>768,chr(209)=>713,chr(210)=>790,chr(211)=>790,chr(212)=>890,chr(213)=>823,chr(214)=>549,chr(215)=>250,chr(216)=>713,chr(217)=>603,chr(218)=>603,chr(219)=>1042,
|
||||
chr(220)=>987,chr(221)=>603,chr(222)=>987,chr(223)=>603,chr(224)=>494,chr(225)=>329,chr(226)=>790,chr(227)=>790,chr(228)=>786,chr(229)=>713,chr(230)=>384,chr(231)=>384,chr(232)=>384,chr(233)=>384,chr(234)=>384,chr(235)=>384,chr(236)=>494,chr(237)=>494,chr(238)=>494,chr(239)=>494,chr(240)=>0,chr(241)=>329,
|
||||
chr(242)=>274,chr(243)=>686,chr(244)=>686,chr(245)=>686,chr(246)=>384,chr(247)=>384,chr(248)=>384,chr(249)=>384,chr(250)=>384,chr(251)=>384,chr(252)=>494,chr(253)=>494,chr(254)=>494,chr(255)=>0);
|
||||
$uv = array(32=>160,33=>33,34=>8704,35=>35,36=>8707,37=>array(37,2),39=>8715,40=>array(40,2),42=>8727,43=>array(43,2),45=>8722,46=>array(46,18),64=>8773,65=>array(913,2),67=>935,68=>array(916,2),70=>934,71=>915,72=>919,73=>921,74=>977,75=>array(922,4),79=>array(927,2),81=>920,82=>929,83=>array(931,3),86=>962,87=>937,88=>926,89=>936,90=>918,91=>91,92=>8756,93=>93,94=>8869,95=>95,96=>63717,97=>array(945,2),99=>967,100=>array(948,2),102=>966,103=>947,104=>951,105=>953,106=>981,107=>array(954,4),111=>array(959,2),113=>952,114=>961,115=>array(963,3),118=>982,119=>969,120=>958,121=>968,122=>950,123=>array(123,3),126=>8764,160=>8364,161=>978,162=>8242,163=>8804,164=>8725,165=>8734,166=>402,167=>9827,168=>9830,169=>9829,170=>9824,171=>8596,172=>array(8592,4),176=>array(176,2),178=>8243,179=>8805,180=>215,181=>8733,182=>8706,183=>8226,184=>247,185=>array(8800,2),187=>8776,188=>8230,189=>array(63718,2),191=>8629,192=>8501,193=>8465,194=>8476,195=>8472,196=>8855,197=>8853,198=>8709,199=>array(8745,2),201=>8835,202=>8839,203=>8836,204=>8834,205=>8838,206=>array(8712,2),208=>8736,209=>8711,210=>63194,211=>63193,212=>63195,213=>8719,214=>8730,215=>8901,216=>172,217=>array(8743,2),219=>8660,220=>array(8656,4),224=>9674,225=>9001,226=>array(63720,3),229=>8721,230=>array(63723,10),241=>9002,242=>8747,243=>8992,244=>63733,245=>8993,246=>array(63734,9));
|
||||
$desc = '';
|
||||
$diff = '';
|
||||
$file = '';
|
||||
$size1 = 0;
|
||||
$size2 = 0;
|
||||
$originalsize = 0;
|
||||
?>
|
||||
Vendored
+27
@@ -0,0 +1,27 @@
|
||||
<?php
|
||||
$type = 'Core';
|
||||
$name = 'Times-Roman';
|
||||
$up = -100;
|
||||
$ut = 50;
|
||||
$cw = array(
|
||||
chr(0)=>250,chr(1)=>250,chr(2)=>250,chr(3)=>250,chr(4)=>250,chr(5)=>250,chr(6)=>250,chr(7)=>250,chr(8)=>250,chr(9)=>250,chr(10)=>250,chr(11)=>250,chr(12)=>250,chr(13)=>250,chr(14)=>250,chr(15)=>250,chr(16)=>250,chr(17)=>250,chr(18)=>250,chr(19)=>250,chr(20)=>250,chr(21)=>250,
|
||||
chr(22)=>250,chr(23)=>250,chr(24)=>250,chr(25)=>250,chr(26)=>250,chr(27)=>250,chr(28)=>250,chr(29)=>250,chr(30)=>250,chr(31)=>250,' '=>250,'!'=>333,'"'=>408,'#'=>500,'$'=>500,'%'=>833,'&'=>778,'\''=>180,'('=>333,')'=>333,'*'=>500,'+'=>564,
|
||||
','=>250,'-'=>333,'.'=>250,'/'=>278,'0'=>500,'1'=>500,'2'=>500,'3'=>500,'4'=>500,'5'=>500,'6'=>500,'7'=>500,'8'=>500,'9'=>500,':'=>278,';'=>278,'<'=>564,'='=>564,'>'=>564,'?'=>444,'@'=>921,'A'=>722,
|
||||
'B'=>667,'C'=>667,'D'=>722,'E'=>611,'F'=>556,'G'=>722,'H'=>722,'I'=>333,'J'=>389,'K'=>722,'L'=>611,'M'=>889,'N'=>722,'O'=>722,'P'=>556,'Q'=>722,'R'=>667,'S'=>556,'T'=>611,'U'=>722,'V'=>722,'W'=>944,
|
||||
'X'=>722,'Y'=>722,'Z'=>611,'['=>333,'\\'=>278,']'=>333,'^'=>469,'_'=>500,'`'=>333,'a'=>444,'b'=>500,'c'=>444,'d'=>500,'e'=>444,'f'=>333,'g'=>500,'h'=>500,'i'=>278,'j'=>278,'k'=>500,'l'=>278,'m'=>778,
|
||||
'n'=>500,'o'=>500,'p'=>500,'q'=>500,'r'=>333,'s'=>389,'t'=>278,'u'=>500,'v'=>500,'w'=>722,'x'=>500,'y'=>500,'z'=>444,'{'=>480,'|'=>200,'}'=>480,'~'=>541,chr(127)=>350,chr(128)=>500,chr(129)=>350,chr(130)=>333,chr(131)=>500,
|
||||
chr(132)=>444,chr(133)=>1000,chr(134)=>500,chr(135)=>500,chr(136)=>333,chr(137)=>1000,chr(138)=>556,chr(139)=>333,chr(140)=>889,chr(141)=>350,chr(142)=>611,chr(143)=>350,chr(144)=>350,chr(145)=>333,chr(146)=>333,chr(147)=>444,chr(148)=>444,chr(149)=>350,chr(150)=>500,chr(151)=>1000,chr(152)=>333,chr(153)=>980,
|
||||
chr(154)=>389,chr(155)=>333,chr(156)=>722,chr(157)=>350,chr(158)=>444,chr(159)=>722,chr(160)=>250,chr(161)=>333,chr(162)=>500,chr(163)=>500,chr(164)=>500,chr(165)=>500,chr(166)=>200,chr(167)=>500,chr(168)=>333,chr(169)=>760,chr(170)=>276,chr(171)=>500,chr(172)=>564,chr(173)=>333,chr(174)=>760,chr(175)=>333,
|
||||
chr(176)=>400,chr(177)=>564,chr(178)=>300,chr(179)=>300,chr(180)=>333,chr(181)=>500,chr(182)=>453,chr(183)=>250,chr(184)=>333,chr(185)=>300,chr(186)=>310,chr(187)=>500,chr(188)=>750,chr(189)=>750,chr(190)=>750,chr(191)=>444,chr(192)=>722,chr(193)=>722,chr(194)=>722,chr(195)=>722,chr(196)=>722,chr(197)=>722,
|
||||
chr(198)=>889,chr(199)=>667,chr(200)=>611,chr(201)=>611,chr(202)=>611,chr(203)=>611,chr(204)=>333,chr(205)=>333,chr(206)=>333,chr(207)=>333,chr(208)=>722,chr(209)=>722,chr(210)=>722,chr(211)=>722,chr(212)=>722,chr(213)=>722,chr(214)=>722,chr(215)=>564,chr(216)=>722,chr(217)=>722,chr(218)=>722,chr(219)=>722,
|
||||
chr(220)=>722,chr(221)=>722,chr(222)=>556,chr(223)=>500,chr(224)=>444,chr(225)=>444,chr(226)=>444,chr(227)=>444,chr(228)=>444,chr(229)=>444,chr(230)=>667,chr(231)=>444,chr(232)=>444,chr(233)=>444,chr(234)=>444,chr(235)=>444,chr(236)=>278,chr(237)=>278,chr(238)=>278,chr(239)=>278,chr(240)=>500,chr(241)=>500,
|
||||
chr(242)=>500,chr(243)=>500,chr(244)=>500,chr(245)=>500,chr(246)=>500,chr(247)=>564,chr(248)=>500,chr(249)=>500,chr(250)=>500,chr(251)=>500,chr(252)=>500,chr(253)=>500,chr(254)=>500,chr(255)=>500);
|
||||
$enc = 'cp1252';
|
||||
$uv = array(0=>array(0,128),128=>8364,130=>8218,131=>402,132=>8222,133=>8230,134=>array(8224,2),136=>710,137=>8240,138=>352,139=>8249,140=>338,142=>381,145=>array(8216,2),147=>array(8220,2),149=>8226,150=>array(8211,2),152=>732,153=>8482,154=>353,155=>8250,156=>339,158=>382,159=>376,160=>array(160,96));
|
||||
$desc = '';
|
||||
$diff = '';
|
||||
$file = '';
|
||||
$size1 = 0;
|
||||
$size2 = 0;
|
||||
$originalsize = 0;
|
||||
?>
|
||||
Vendored
+27
@@ -0,0 +1,27 @@
|
||||
<?php
|
||||
$type = 'Core';
|
||||
$name = 'Times-Bold';
|
||||
$up = -100;
|
||||
$ut = 50;
|
||||
$cw = array(
|
||||
chr(0)=>250,chr(1)=>250,chr(2)=>250,chr(3)=>250,chr(4)=>250,chr(5)=>250,chr(6)=>250,chr(7)=>250,chr(8)=>250,chr(9)=>250,chr(10)=>250,chr(11)=>250,chr(12)=>250,chr(13)=>250,chr(14)=>250,chr(15)=>250,chr(16)=>250,chr(17)=>250,chr(18)=>250,chr(19)=>250,chr(20)=>250,chr(21)=>250,
|
||||
chr(22)=>250,chr(23)=>250,chr(24)=>250,chr(25)=>250,chr(26)=>250,chr(27)=>250,chr(28)=>250,chr(29)=>250,chr(30)=>250,chr(31)=>250,' '=>250,'!'=>333,'"'=>555,'#'=>500,'$'=>500,'%'=>1000,'&'=>833,'\''=>278,'('=>333,')'=>333,'*'=>500,'+'=>570,
|
||||
','=>250,'-'=>333,'.'=>250,'/'=>278,'0'=>500,'1'=>500,'2'=>500,'3'=>500,'4'=>500,'5'=>500,'6'=>500,'7'=>500,'8'=>500,'9'=>500,':'=>333,';'=>333,'<'=>570,'='=>570,'>'=>570,'?'=>500,'@'=>930,'A'=>722,
|
||||
'B'=>667,'C'=>722,'D'=>722,'E'=>667,'F'=>611,'G'=>778,'H'=>778,'I'=>389,'J'=>500,'K'=>778,'L'=>667,'M'=>944,'N'=>722,'O'=>778,'P'=>611,'Q'=>778,'R'=>722,'S'=>556,'T'=>667,'U'=>722,'V'=>722,'W'=>1000,
|
||||
'X'=>722,'Y'=>722,'Z'=>667,'['=>333,'\\'=>278,']'=>333,'^'=>581,'_'=>500,'`'=>333,'a'=>500,'b'=>556,'c'=>444,'d'=>556,'e'=>444,'f'=>333,'g'=>500,'h'=>556,'i'=>278,'j'=>333,'k'=>556,'l'=>278,'m'=>833,
|
||||
'n'=>556,'o'=>500,'p'=>556,'q'=>556,'r'=>444,'s'=>389,'t'=>333,'u'=>556,'v'=>500,'w'=>722,'x'=>500,'y'=>500,'z'=>444,'{'=>394,'|'=>220,'}'=>394,'~'=>520,chr(127)=>350,chr(128)=>500,chr(129)=>350,chr(130)=>333,chr(131)=>500,
|
||||
chr(132)=>500,chr(133)=>1000,chr(134)=>500,chr(135)=>500,chr(136)=>333,chr(137)=>1000,chr(138)=>556,chr(139)=>333,chr(140)=>1000,chr(141)=>350,chr(142)=>667,chr(143)=>350,chr(144)=>350,chr(145)=>333,chr(146)=>333,chr(147)=>500,chr(148)=>500,chr(149)=>350,chr(150)=>500,chr(151)=>1000,chr(152)=>333,chr(153)=>1000,
|
||||
chr(154)=>389,chr(155)=>333,chr(156)=>722,chr(157)=>350,chr(158)=>444,chr(159)=>722,chr(160)=>250,chr(161)=>333,chr(162)=>500,chr(163)=>500,chr(164)=>500,chr(165)=>500,chr(166)=>220,chr(167)=>500,chr(168)=>333,chr(169)=>747,chr(170)=>300,chr(171)=>500,chr(172)=>570,chr(173)=>333,chr(174)=>747,chr(175)=>333,
|
||||
chr(176)=>400,chr(177)=>570,chr(178)=>300,chr(179)=>300,chr(180)=>333,chr(181)=>556,chr(182)=>540,chr(183)=>250,chr(184)=>333,chr(185)=>300,chr(186)=>330,chr(187)=>500,chr(188)=>750,chr(189)=>750,chr(190)=>750,chr(191)=>500,chr(192)=>722,chr(193)=>722,chr(194)=>722,chr(195)=>722,chr(196)=>722,chr(197)=>722,
|
||||
chr(198)=>1000,chr(199)=>722,chr(200)=>667,chr(201)=>667,chr(202)=>667,chr(203)=>667,chr(204)=>389,chr(205)=>389,chr(206)=>389,chr(207)=>389,chr(208)=>722,chr(209)=>722,chr(210)=>778,chr(211)=>778,chr(212)=>778,chr(213)=>778,chr(214)=>778,chr(215)=>570,chr(216)=>778,chr(217)=>722,chr(218)=>722,chr(219)=>722,
|
||||
chr(220)=>722,chr(221)=>722,chr(222)=>611,chr(223)=>556,chr(224)=>500,chr(225)=>500,chr(226)=>500,chr(227)=>500,chr(228)=>500,chr(229)=>500,chr(230)=>722,chr(231)=>444,chr(232)=>444,chr(233)=>444,chr(234)=>444,chr(235)=>444,chr(236)=>278,chr(237)=>278,chr(238)=>278,chr(239)=>278,chr(240)=>500,chr(241)=>556,
|
||||
chr(242)=>500,chr(243)=>500,chr(244)=>500,chr(245)=>500,chr(246)=>500,chr(247)=>570,chr(248)=>500,chr(249)=>556,chr(250)=>556,chr(251)=>556,chr(252)=>556,chr(253)=>500,chr(254)=>556,chr(255)=>500);
|
||||
$enc = 'cp1252';
|
||||
$uv = array(0=>array(0,128),128=>8364,130=>8218,131=>402,132=>8222,133=>8230,134=>array(8224,2),136=>710,137=>8240,138=>352,139=>8249,140=>338,142=>381,145=>array(8216,2),147=>array(8220,2),149=>8226,150=>array(8211,2),152=>732,153=>8482,154=>353,155=>8250,156=>339,158=>382,159=>376,160=>array(160,96));
|
||||
$desc = '';
|
||||
$diff = '';
|
||||
$file = '';
|
||||
$size1 = 0;
|
||||
$size2 = 0;
|
||||
$originalsize = 0;
|
||||
?>
|
||||
Vendored
+27
@@ -0,0 +1,27 @@
|
||||
<?php
|
||||
$type = 'Core';
|
||||
$name = 'Times-BoldItalic';
|
||||
$up = -100;
|
||||
$ut = 50;
|
||||
$cw = array(
|
||||
chr(0)=>250,chr(1)=>250,chr(2)=>250,chr(3)=>250,chr(4)=>250,chr(5)=>250,chr(6)=>250,chr(7)=>250,chr(8)=>250,chr(9)=>250,chr(10)=>250,chr(11)=>250,chr(12)=>250,chr(13)=>250,chr(14)=>250,chr(15)=>250,chr(16)=>250,chr(17)=>250,chr(18)=>250,chr(19)=>250,chr(20)=>250,chr(21)=>250,
|
||||
chr(22)=>250,chr(23)=>250,chr(24)=>250,chr(25)=>250,chr(26)=>250,chr(27)=>250,chr(28)=>250,chr(29)=>250,chr(30)=>250,chr(31)=>250,' '=>250,'!'=>389,'"'=>555,'#'=>500,'$'=>500,'%'=>833,'&'=>778,'\''=>278,'('=>333,')'=>333,'*'=>500,'+'=>570,
|
||||
','=>250,'-'=>333,'.'=>250,'/'=>278,'0'=>500,'1'=>500,'2'=>500,'3'=>500,'4'=>500,'5'=>500,'6'=>500,'7'=>500,'8'=>500,'9'=>500,':'=>333,';'=>333,'<'=>570,'='=>570,'>'=>570,'?'=>500,'@'=>832,'A'=>667,
|
||||
'B'=>667,'C'=>667,'D'=>722,'E'=>667,'F'=>667,'G'=>722,'H'=>778,'I'=>389,'J'=>500,'K'=>667,'L'=>611,'M'=>889,'N'=>722,'O'=>722,'P'=>611,'Q'=>722,'R'=>667,'S'=>556,'T'=>611,'U'=>722,'V'=>667,'W'=>889,
|
||||
'X'=>667,'Y'=>611,'Z'=>611,'['=>333,'\\'=>278,']'=>333,'^'=>570,'_'=>500,'`'=>333,'a'=>500,'b'=>500,'c'=>444,'d'=>500,'e'=>444,'f'=>333,'g'=>500,'h'=>556,'i'=>278,'j'=>278,'k'=>500,'l'=>278,'m'=>778,
|
||||
'n'=>556,'o'=>500,'p'=>500,'q'=>500,'r'=>389,'s'=>389,'t'=>278,'u'=>556,'v'=>444,'w'=>667,'x'=>500,'y'=>444,'z'=>389,'{'=>348,'|'=>220,'}'=>348,'~'=>570,chr(127)=>350,chr(128)=>500,chr(129)=>350,chr(130)=>333,chr(131)=>500,
|
||||
chr(132)=>500,chr(133)=>1000,chr(134)=>500,chr(135)=>500,chr(136)=>333,chr(137)=>1000,chr(138)=>556,chr(139)=>333,chr(140)=>944,chr(141)=>350,chr(142)=>611,chr(143)=>350,chr(144)=>350,chr(145)=>333,chr(146)=>333,chr(147)=>500,chr(148)=>500,chr(149)=>350,chr(150)=>500,chr(151)=>1000,chr(152)=>333,chr(153)=>1000,
|
||||
chr(154)=>389,chr(155)=>333,chr(156)=>722,chr(157)=>350,chr(158)=>389,chr(159)=>611,chr(160)=>250,chr(161)=>389,chr(162)=>500,chr(163)=>500,chr(164)=>500,chr(165)=>500,chr(166)=>220,chr(167)=>500,chr(168)=>333,chr(169)=>747,chr(170)=>266,chr(171)=>500,chr(172)=>606,chr(173)=>333,chr(174)=>747,chr(175)=>333,
|
||||
chr(176)=>400,chr(177)=>570,chr(178)=>300,chr(179)=>300,chr(180)=>333,chr(181)=>576,chr(182)=>500,chr(183)=>250,chr(184)=>333,chr(185)=>300,chr(186)=>300,chr(187)=>500,chr(188)=>750,chr(189)=>750,chr(190)=>750,chr(191)=>500,chr(192)=>667,chr(193)=>667,chr(194)=>667,chr(195)=>667,chr(196)=>667,chr(197)=>667,
|
||||
chr(198)=>944,chr(199)=>667,chr(200)=>667,chr(201)=>667,chr(202)=>667,chr(203)=>667,chr(204)=>389,chr(205)=>389,chr(206)=>389,chr(207)=>389,chr(208)=>722,chr(209)=>722,chr(210)=>722,chr(211)=>722,chr(212)=>722,chr(213)=>722,chr(214)=>722,chr(215)=>570,chr(216)=>722,chr(217)=>722,chr(218)=>722,chr(219)=>722,
|
||||
chr(220)=>722,chr(221)=>611,chr(222)=>611,chr(223)=>500,chr(224)=>500,chr(225)=>500,chr(226)=>500,chr(227)=>500,chr(228)=>500,chr(229)=>500,chr(230)=>722,chr(231)=>444,chr(232)=>444,chr(233)=>444,chr(234)=>444,chr(235)=>444,chr(236)=>278,chr(237)=>278,chr(238)=>278,chr(239)=>278,chr(240)=>500,chr(241)=>556,
|
||||
chr(242)=>500,chr(243)=>500,chr(244)=>500,chr(245)=>500,chr(246)=>500,chr(247)=>570,chr(248)=>500,chr(249)=>556,chr(250)=>556,chr(251)=>556,chr(252)=>556,chr(253)=>444,chr(254)=>500,chr(255)=>444);
|
||||
$enc = 'cp1252';
|
||||
$uv = array(0=>array(0,128),128=>8364,130=>8218,131=>402,132=>8222,133=>8230,134=>array(8224,2),136=>710,137=>8240,138=>352,139=>8249,140=>338,142=>381,145=>array(8216,2),147=>array(8220,2),149=>8226,150=>array(8211,2),152=>732,153=>8482,154=>353,155=>8250,156=>339,158=>382,159=>376,160=>array(160,96));
|
||||
$desc = '';
|
||||
$diff = '';
|
||||
$file = '';
|
||||
$size1 = 0;
|
||||
$size2 = 0;
|
||||
$originalsize = 0;
|
||||
?>
|
||||
Vendored
+27
@@ -0,0 +1,27 @@
|
||||
<?php
|
||||
$type = 'Core';
|
||||
$name = 'Times-Italic';
|
||||
$up = -100;
|
||||
$ut = 50;
|
||||
$cw = array(
|
||||
chr(0)=>250,chr(1)=>250,chr(2)=>250,chr(3)=>250,chr(4)=>250,chr(5)=>250,chr(6)=>250,chr(7)=>250,chr(8)=>250,chr(9)=>250,chr(10)=>250,chr(11)=>250,chr(12)=>250,chr(13)=>250,chr(14)=>250,chr(15)=>250,chr(16)=>250,chr(17)=>250,chr(18)=>250,chr(19)=>250,chr(20)=>250,chr(21)=>250,
|
||||
chr(22)=>250,chr(23)=>250,chr(24)=>250,chr(25)=>250,chr(26)=>250,chr(27)=>250,chr(28)=>250,chr(29)=>250,chr(30)=>250,chr(31)=>250,' '=>250,'!'=>333,'"'=>420,'#'=>500,'$'=>500,'%'=>833,'&'=>778,'\''=>214,'('=>333,')'=>333,'*'=>500,'+'=>675,
|
||||
','=>250,'-'=>333,'.'=>250,'/'=>278,'0'=>500,'1'=>500,'2'=>500,'3'=>500,'4'=>500,'5'=>500,'6'=>500,'7'=>500,'8'=>500,'9'=>500,':'=>333,';'=>333,'<'=>675,'='=>675,'>'=>675,'?'=>500,'@'=>920,'A'=>611,
|
||||
'B'=>611,'C'=>667,'D'=>722,'E'=>611,'F'=>611,'G'=>722,'H'=>722,'I'=>333,'J'=>444,'K'=>667,'L'=>556,'M'=>833,'N'=>667,'O'=>722,'P'=>611,'Q'=>722,'R'=>611,'S'=>500,'T'=>556,'U'=>722,'V'=>611,'W'=>833,
|
||||
'X'=>611,'Y'=>556,'Z'=>556,'['=>389,'\\'=>278,']'=>389,'^'=>422,'_'=>500,'`'=>333,'a'=>500,'b'=>500,'c'=>444,'d'=>500,'e'=>444,'f'=>278,'g'=>500,'h'=>500,'i'=>278,'j'=>278,'k'=>444,'l'=>278,'m'=>722,
|
||||
'n'=>500,'o'=>500,'p'=>500,'q'=>500,'r'=>389,'s'=>389,'t'=>278,'u'=>500,'v'=>444,'w'=>667,'x'=>444,'y'=>444,'z'=>389,'{'=>400,'|'=>275,'}'=>400,'~'=>541,chr(127)=>350,chr(128)=>500,chr(129)=>350,chr(130)=>333,chr(131)=>500,
|
||||
chr(132)=>556,chr(133)=>889,chr(134)=>500,chr(135)=>500,chr(136)=>333,chr(137)=>1000,chr(138)=>500,chr(139)=>333,chr(140)=>944,chr(141)=>350,chr(142)=>556,chr(143)=>350,chr(144)=>350,chr(145)=>333,chr(146)=>333,chr(147)=>556,chr(148)=>556,chr(149)=>350,chr(150)=>500,chr(151)=>889,chr(152)=>333,chr(153)=>980,
|
||||
chr(154)=>389,chr(155)=>333,chr(156)=>667,chr(157)=>350,chr(158)=>389,chr(159)=>556,chr(160)=>250,chr(161)=>389,chr(162)=>500,chr(163)=>500,chr(164)=>500,chr(165)=>500,chr(166)=>275,chr(167)=>500,chr(168)=>333,chr(169)=>760,chr(170)=>276,chr(171)=>500,chr(172)=>675,chr(173)=>333,chr(174)=>760,chr(175)=>333,
|
||||
chr(176)=>400,chr(177)=>675,chr(178)=>300,chr(179)=>300,chr(180)=>333,chr(181)=>500,chr(182)=>523,chr(183)=>250,chr(184)=>333,chr(185)=>300,chr(186)=>310,chr(187)=>500,chr(188)=>750,chr(189)=>750,chr(190)=>750,chr(191)=>500,chr(192)=>611,chr(193)=>611,chr(194)=>611,chr(195)=>611,chr(196)=>611,chr(197)=>611,
|
||||
chr(198)=>889,chr(199)=>667,chr(200)=>611,chr(201)=>611,chr(202)=>611,chr(203)=>611,chr(204)=>333,chr(205)=>333,chr(206)=>333,chr(207)=>333,chr(208)=>722,chr(209)=>667,chr(210)=>722,chr(211)=>722,chr(212)=>722,chr(213)=>722,chr(214)=>722,chr(215)=>675,chr(216)=>722,chr(217)=>722,chr(218)=>722,chr(219)=>722,
|
||||
chr(220)=>722,chr(221)=>556,chr(222)=>611,chr(223)=>500,chr(224)=>500,chr(225)=>500,chr(226)=>500,chr(227)=>500,chr(228)=>500,chr(229)=>500,chr(230)=>667,chr(231)=>444,chr(232)=>444,chr(233)=>444,chr(234)=>444,chr(235)=>444,chr(236)=>278,chr(237)=>278,chr(238)=>278,chr(239)=>278,chr(240)=>500,chr(241)=>500,
|
||||
chr(242)=>500,chr(243)=>500,chr(244)=>500,chr(245)=>500,chr(246)=>500,chr(247)=>675,chr(248)=>500,chr(249)=>500,chr(250)=>500,chr(251)=>500,chr(252)=>500,chr(253)=>444,chr(254)=>500,chr(255)=>444);
|
||||
$enc = 'cp1252';
|
||||
$uv = array(0=>array(0,128),128=>8364,130=>8218,131=>402,132=>8222,133=>8230,134=>array(8224,2),136=>710,137=>8240,138=>352,139=>8249,140=>338,142=>381,145=>array(8216,2),147=>array(8220,2),149=>8226,150=>array(8211,2),152=>732,153=>8482,154=>353,155=>8250,156=>339,158=>382,159=>376,160=>array(160,96));
|
||||
$desc = '';
|
||||
$diff = '';
|
||||
$file = '';
|
||||
$size1 = 0;
|
||||
$size2 = 0;
|
||||
$originalsize = 0;
|
||||
?>
|
||||
Vendored
+26
@@ -0,0 +1,26 @@
|
||||
<?php
|
||||
$type = 'Core';
|
||||
$name = 'ZapfDingbats';
|
||||
$up = -100;
|
||||
$ut = 50;
|
||||
$cw = array(
|
||||
chr(0)=>0,chr(1)=>0,chr(2)=>0,chr(3)=>0,chr(4)=>0,chr(5)=>0,chr(6)=>0,chr(7)=>0,chr(8)=>0,chr(9)=>0,chr(10)=>0,chr(11)=>0,chr(12)=>0,chr(13)=>0,chr(14)=>0,chr(15)=>0,chr(16)=>0,chr(17)=>0,chr(18)=>0,chr(19)=>0,chr(20)=>0,chr(21)=>0,
|
||||
chr(22)=>0,chr(23)=>0,chr(24)=>0,chr(25)=>0,chr(26)=>0,chr(27)=>0,chr(28)=>0,chr(29)=>0,chr(30)=>0,chr(31)=>0,' '=>278,'!'=>974,'"'=>961,'#'=>974,'$'=>980,'%'=>719,'&'=>789,'\''=>790,'('=>791,')'=>690,'*'=>960,'+'=>939,
|
||||
','=>549,'-'=>855,'.'=>911,'/'=>933,'0'=>911,'1'=>945,'2'=>974,'3'=>755,'4'=>846,'5'=>762,'6'=>761,'7'=>571,'8'=>677,'9'=>763,':'=>760,';'=>759,'<'=>754,'='=>494,'>'=>552,'?'=>537,'@'=>577,'A'=>692,
|
||||
'B'=>786,'C'=>788,'D'=>788,'E'=>790,'F'=>793,'G'=>794,'H'=>816,'I'=>823,'J'=>789,'K'=>841,'L'=>823,'M'=>833,'N'=>816,'O'=>831,'P'=>923,'Q'=>744,'R'=>723,'S'=>749,'T'=>790,'U'=>792,'V'=>695,'W'=>776,
|
||||
'X'=>768,'Y'=>792,'Z'=>759,'['=>707,'\\'=>708,']'=>682,'^'=>701,'_'=>826,'`'=>815,'a'=>789,'b'=>789,'c'=>707,'d'=>687,'e'=>696,'f'=>689,'g'=>786,'h'=>787,'i'=>713,'j'=>791,'k'=>785,'l'=>791,'m'=>873,
|
||||
'n'=>761,'o'=>762,'p'=>762,'q'=>759,'r'=>759,'s'=>892,'t'=>892,'u'=>788,'v'=>784,'w'=>438,'x'=>138,'y'=>277,'z'=>415,'{'=>392,'|'=>392,'}'=>668,'~'=>668,chr(127)=>0,chr(128)=>390,chr(129)=>390,chr(130)=>317,chr(131)=>317,
|
||||
chr(132)=>276,chr(133)=>276,chr(134)=>509,chr(135)=>509,chr(136)=>410,chr(137)=>410,chr(138)=>234,chr(139)=>234,chr(140)=>334,chr(141)=>334,chr(142)=>0,chr(143)=>0,chr(144)=>0,chr(145)=>0,chr(146)=>0,chr(147)=>0,chr(148)=>0,chr(149)=>0,chr(150)=>0,chr(151)=>0,chr(152)=>0,chr(153)=>0,
|
||||
chr(154)=>0,chr(155)=>0,chr(156)=>0,chr(157)=>0,chr(158)=>0,chr(159)=>0,chr(160)=>0,chr(161)=>732,chr(162)=>544,chr(163)=>544,chr(164)=>910,chr(165)=>667,chr(166)=>760,chr(167)=>760,chr(168)=>776,chr(169)=>595,chr(170)=>694,chr(171)=>626,chr(172)=>788,chr(173)=>788,chr(174)=>788,chr(175)=>788,
|
||||
chr(176)=>788,chr(177)=>788,chr(178)=>788,chr(179)=>788,chr(180)=>788,chr(181)=>788,chr(182)=>788,chr(183)=>788,chr(184)=>788,chr(185)=>788,chr(186)=>788,chr(187)=>788,chr(188)=>788,chr(189)=>788,chr(190)=>788,chr(191)=>788,chr(192)=>788,chr(193)=>788,chr(194)=>788,chr(195)=>788,chr(196)=>788,chr(197)=>788,
|
||||
chr(198)=>788,chr(199)=>788,chr(200)=>788,chr(201)=>788,chr(202)=>788,chr(203)=>788,chr(204)=>788,chr(205)=>788,chr(206)=>788,chr(207)=>788,chr(208)=>788,chr(209)=>788,chr(210)=>788,chr(211)=>788,chr(212)=>894,chr(213)=>838,chr(214)=>1016,chr(215)=>458,chr(216)=>748,chr(217)=>924,chr(218)=>748,chr(219)=>918,
|
||||
chr(220)=>927,chr(221)=>928,chr(222)=>928,chr(223)=>834,chr(224)=>873,chr(225)=>828,chr(226)=>924,chr(227)=>924,chr(228)=>917,chr(229)=>930,chr(230)=>931,chr(231)=>463,chr(232)=>883,chr(233)=>836,chr(234)=>836,chr(235)=>867,chr(236)=>867,chr(237)=>696,chr(238)=>696,chr(239)=>874,chr(240)=>0,chr(241)=>874,
|
||||
chr(242)=>760,chr(243)=>946,chr(244)=>771,chr(245)=>865,chr(246)=>771,chr(247)=>888,chr(248)=>967,chr(249)=>888,chr(250)=>831,chr(251)=>873,chr(252)=>927,chr(253)=>970,chr(254)=>918,chr(255)=>0);
|
||||
$uv = array(32=>32,33=>array(9985,4),37=>9742,38=>array(9990,4),42=>9755,43=>9758,44=>array(9996,28),72=>9733,73=>array(10025,35),108=>9679,109=>10061,110=>9632,111=>array(10063,4),115=>9650,116=>9660,117=>9670,118=>10070,119=>9687,120=>array(10072,7),128=>array(10088,14),161=>array(10081,7),168=>9827,169=>9830,170=>9829,171=>9824,172=>array(9312,10),182=>array(10102,31),213=>8594,214=>array(8596,2),216=>array(10136,24),241=>array(10161,14));
|
||||
$desc = '';
|
||||
$diff = '';
|
||||
$file = '';
|
||||
$size1 = 0;
|
||||
$size2 = 0;
|
||||
$originalsize = 0;
|
||||
?>
|
||||
Vendored
+1551
File diff suppressed because it is too large
Load Diff
Vendored
+165
@@ -0,0 +1,165 @@
|
||||
GNU LESSER GENERAL PUBLIC LICENSE
|
||||
Version 3, 29 June 2007
|
||||
|
||||
Copyright (C) 2007 Free Software Foundation, Inc. <http://fsf.org/>
|
||||
Everyone is permitted to copy and distribute verbatim copies
|
||||
of this license document, but changing it is not allowed.
|
||||
|
||||
|
||||
This version of the GNU Lesser General Public License incorporates
|
||||
the terms and conditions of version 3 of the GNU General Public
|
||||
License, supplemented by the additional permissions listed below.
|
||||
|
||||
0. Additional Definitions.
|
||||
|
||||
As used herein, "this License" refers to version 3 of the GNU Lesser
|
||||
General Public License, and the "GNU GPL" refers to version 3 of the GNU
|
||||
General Public License.
|
||||
|
||||
"The Library" refers to a covered work governed by this License,
|
||||
other than an Application or a Combined Work as defined below.
|
||||
|
||||
An "Application" is any work that makes use of an interface provided
|
||||
by the Library, but which is not otherwise based on the Library.
|
||||
Defining a subclass of a class defined by the Library is deemed a mode
|
||||
of using an interface provided by the Library.
|
||||
|
||||
A "Combined Work" is a work produced by combining or linking an
|
||||
Application with the Library. The particular version of the Library
|
||||
with which the Combined Work was made is also called the "Linked
|
||||
Version".
|
||||
|
||||
The "Minimal Corresponding Source" for a Combined Work means the
|
||||
Corresponding Source for the Combined Work, excluding any source code
|
||||
for portions of the Combined Work that, considered in isolation, are
|
||||
based on the Application, and not on the Linked Version.
|
||||
|
||||
The "Corresponding Application Code" for a Combined Work means the
|
||||
object code and/or source code for the Application, including any data
|
||||
and utility programs needed for reproducing the Combined Work from the
|
||||
Application, but excluding the System Libraries of the Combined Work.
|
||||
|
||||
1. Exception to Section 3 of the GNU GPL.
|
||||
|
||||
You may convey a covered work under sections 3 and 4 of this License
|
||||
without being bound by section 3 of the GNU GPL.
|
||||
|
||||
2. Conveying Modified Versions.
|
||||
|
||||
If you modify a copy of the Library, and, in your modifications, a
|
||||
facility refers to a function or data to be supplied by an Application
|
||||
that uses the facility (other than as an argument passed when the
|
||||
facility is invoked), then you may convey a copy of the modified
|
||||
version:
|
||||
|
||||
a) under this License, provided that you make a good faith effort to
|
||||
ensure that, in the event an Application does not supply the
|
||||
function or data, the facility still operates, and performs
|
||||
whatever part of its purpose remains meaningful, or
|
||||
|
||||
b) under the GNU GPL, with none of the additional permissions of
|
||||
this License applicable to that copy.
|
||||
|
||||
3. Object Code Incorporating Material from Library Header Files.
|
||||
|
||||
The object code form of an Application may incorporate material from
|
||||
a header file that is part of the Library. You may convey such object
|
||||
code under terms of your choice, provided that, if the incorporated
|
||||
material is not limited to numerical parameters, data structure
|
||||
layouts and accessors, or small macros, inline functions and templates
|
||||
(ten or fewer lines in length), you do both of the following:
|
||||
|
||||
a) Give prominent notice with each copy of the object code that the
|
||||
Library is used in it and that the Library and its use are
|
||||
covered by this License.
|
||||
|
||||
b) Accompany the object code with a copy of the GNU GPL and this license
|
||||
document.
|
||||
|
||||
4. Combined Works.
|
||||
|
||||
You may convey a Combined Work under terms of your choice that,
|
||||
taken together, effectively do not restrict modification of the
|
||||
portions of the Library contained in the Combined Work and reverse
|
||||
engineering for debugging such modifications, if you also do each of
|
||||
the following:
|
||||
|
||||
a) Give prominent notice with each copy of the Combined Work that
|
||||
the Library is used in it and that the Library and its use are
|
||||
covered by this License.
|
||||
|
||||
b) Accompany the Combined Work with a copy of the GNU GPL and this license
|
||||
document.
|
||||
|
||||
c) For a Combined Work that displays copyright notices during
|
||||
execution, include the copyright notice for the Library among
|
||||
these notices, as well as a reference directing the user to the
|
||||
copies of the GNU GPL and this license document.
|
||||
|
||||
d) Do one of the following:
|
||||
|
||||
0) Convey the Minimal Corresponding Source under the terms of this
|
||||
License, and the Corresponding Application Code in a form
|
||||
suitable for, and under terms that permit, the user to
|
||||
recombine or relink the Application with a modified version of
|
||||
the Linked Version to produce a modified Combined Work, in the
|
||||
manner specified by section 6 of the GNU GPL for conveying
|
||||
Corresponding Source.
|
||||
|
||||
1) Use a suitable shared library mechanism for linking with the
|
||||
Library. A suitable mechanism is one that (a) uses at run time
|
||||
a copy of the Library already present on the user's computer
|
||||
system, and (b) will operate properly with a modified version
|
||||
of the Library that is interface-compatible with the Linked
|
||||
Version.
|
||||
|
||||
e) Provide Installation Information, but only if you would otherwise
|
||||
be required to provide such information under section 6 of the
|
||||
GNU GPL, and only to the extent that such information is
|
||||
necessary to install and execute a modified version of the
|
||||
Combined Work produced by recombining or relinking the
|
||||
Application with a modified version of the Linked Version. (If
|
||||
you use option 4d0, the Installation Information must accompany
|
||||
the Minimal Corresponding Source and Corresponding Application
|
||||
Code. If you use option 4d1, you must provide the Installation
|
||||
Information in the manner specified by section 6 of the GNU GPL
|
||||
for conveying Corresponding Source.)
|
||||
|
||||
5. Combined Libraries.
|
||||
|
||||
You may place library facilities that are a work based on the
|
||||
Library side by side in a single library together with other library
|
||||
facilities that are not Applications and are not covered by this
|
||||
License, and convey such a combined library under terms of your
|
||||
choice, if you do both of the following:
|
||||
|
||||
a) Accompany the combined library with a copy of the same work based
|
||||
on the Library, uncombined with any other library facilities,
|
||||
conveyed under the terms of this License.
|
||||
|
||||
b) Give prominent notice with the combined library that part of it
|
||||
is a work based on the Library, and explaining where to find the
|
||||
accompanying uncombined form of the same work.
|
||||
|
||||
6. Revised Versions of the GNU Lesser General Public License.
|
||||
|
||||
The Free Software Foundation may publish revised and/or new versions
|
||||
of the GNU Lesser General Public License from time to time. Such new
|
||||
versions will be similar in spirit to the present version, but may
|
||||
differ in detail to address new problems or concerns.
|
||||
|
||||
Each version is given a distinguishing version number. If the
|
||||
Library as you received it specifies that a certain numbered version
|
||||
of the GNU Lesser General Public License "or any later version"
|
||||
applies to it, you have the option of following the terms and
|
||||
conditions either of that published version or of any later version
|
||||
published by the Free Software Foundation. If the Library as you
|
||||
received it does not specify a version number of the GNU Lesser
|
||||
General Public License, you may choose any version of the GNU Lesser
|
||||
General Public License ever published by the Free Software Foundation.
|
||||
|
||||
If the Library as you received it specifies that a proxy can decide
|
||||
whether future versions of the GNU Lesser General Public License shall
|
||||
apply, that proxy's public statement of acceptance of any version is
|
||||
permanent authorization for you to choose that version for the
|
||||
Library.
|
||||
Vendored
+59
@@ -0,0 +1,59 @@
|
||||
# PDF parser
|
||||
|
||||
[](//packagist.org/packages/smalot/pdfparser)
|
||||

|
||||

|
||||
[](https://scrutinizer-ci.com/g/smalot/pdfparser/?branch=master)
|
||||
[](//packagist.org/packages/smalot/pdfparser)
|
||||
|
||||
The `smalot/pdfparser` is a standalone PHP package that provides various tools to extract data from PDF files.
|
||||
|
||||
This library is under **active maintenance**.
|
||||
There is no active development by the author of this library (at the moment), but we welcome any pull request adding/extending functionality!
|
||||
See [CONTRIBUTING.md](./CONTRIBUTING.md) for further information about how to contribute.
|
||||
|
||||
## Features
|
||||
|
||||
- Load/parse objects and headers
|
||||
- Extract metadata (author, description, ...)
|
||||
- Extract text from ordered pages
|
||||
- Support of compressed PDFs
|
||||
- Support of MAC OS Roman charset encoding
|
||||
- Handling of hexa and octal encoding in text sections
|
||||
- Create custom configurations (see [CustomConfig.md](/doc/CustomConfig.md)).
|
||||
|
||||
Currently, secured documents and extracting form data are not supported.
|
||||
|
||||
## License
|
||||
|
||||
This library is under the [LGPLv3 license](https://github.com/smalot/pdfparser/blob/master/LICENSE.txt).
|
||||
|
||||
## Install
|
||||
|
||||
This library requires PHP 7.1+ since [v1](https://github.com/smalot/pdfparser/releases/tag/v1.0.0).
|
||||
You can install it via [Composer](https://getcomposer.org/):
|
||||
|
||||
```bash
|
||||
composer require smalot/pdfparser
|
||||
```
|
||||
|
||||
In case you can't use Composer, you can include `alt_autoload.php-dist`. It will include all required files automatically.
|
||||
|
||||
## Quick example
|
||||
|
||||
```php
|
||||
<?php
|
||||
|
||||
// Parse PDF file and build necessary objects.
|
||||
$parser = new \Smalot\PdfParser\Parser();
|
||||
$pdf = $parser->parseFile('/path/to/document.pdf');
|
||||
|
||||
$text = $pdf->getText();
|
||||
echo $text;
|
||||
```
|
||||
|
||||
Further usage information can be found [here](/doc/Usage.md).
|
||||
|
||||
## Documentation
|
||||
|
||||
Documentation can be found in the [doc](/doc) folder.
|
||||
+37
@@ -0,0 +1,37 @@
|
||||
{
|
||||
"name": "smalot/pdfparser",
|
||||
"description": "Pdf parser library. Can read and extract information from pdf file.",
|
||||
"keywords": ["PDF", "text", "parser", "parse", "extract"],
|
||||
"type": "library",
|
||||
"license": "LGPL-3.0",
|
||||
"authors": [
|
||||
{
|
||||
"name": "Sebastien MALOT",
|
||||
"email": "sebastien@malot.fr"
|
||||
}
|
||||
],
|
||||
"support": {
|
||||
"issues": "https://github.com/smalot/pdfparser/issues"
|
||||
},
|
||||
"homepage": "https://www.pdfparser.org",
|
||||
"require": {
|
||||
"php": ">=7.1",
|
||||
"symfony/polyfill-mbstring": "^1.18",
|
||||
"ext-zlib": "*",
|
||||
"ext-iconv": "*"
|
||||
},
|
||||
"autoload": {
|
||||
"psr-0": {
|
||||
"Smalot\\PdfParser\\": "src/"
|
||||
}
|
||||
},
|
||||
"autoload-dev": {
|
||||
"psr-4": {
|
||||
"PerformanceTests\\": "tests/Performance/",
|
||||
"PHPUnitTests\\": "tests/PHPUnit/"
|
||||
}
|
||||
},
|
||||
"config": {
|
||||
"process-timeout": 1200
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,175 @@
|
||||
<?php
|
||||
|
||||
/**
|
||||
* @file
|
||||
* This file is part of the PdfParser library.
|
||||
*
|
||||
* @author Konrad Abicht <hi@inspirito.de>
|
||||
*
|
||||
* @date 2020-11-22
|
||||
*
|
||||
* @license LGPLv3
|
||||
*
|
||||
* @url <https://github.com/smalot/pdfparser>
|
||||
*
|
||||
* PdfParser is a pdf library written in PHP, extraction oriented.
|
||||
* Copyright (C) 2017 - Sébastien MALOT <sebastien@malot.fr>
|
||||
*
|
||||
* This program is free software: you can redistribute it and/or modify
|
||||
* it under the terms of the GNU Lesser General Public License as published by
|
||||
* the Free Software Foundation, either version 3 of the License, or
|
||||
* (at your option) any later version.
|
||||
*
|
||||
* This program is distributed in the hope that it will be useful,
|
||||
* but WITHOUT ANY WARRANTY; without even the implied warranty of
|
||||
* MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
|
||||
* GNU Lesser General Public License for more details.
|
||||
*
|
||||
* You should have received a copy of the GNU Lesser General Public License
|
||||
* along with this program.
|
||||
* If not, see <http://www.pdfparser.org/sites/default/LICENSE.txt>.
|
||||
*/
|
||||
|
||||
namespace Smalot\PdfParser;
|
||||
|
||||
/**
|
||||
* This class contains configurations used in various classes. You can override them
|
||||
* manually, in case default values aren't working.
|
||||
*
|
||||
* @see https://github.com/smalot/pdfparser/issues/305
|
||||
*/
|
||||
class Config
|
||||
{
|
||||
private $fontSpaceLimit = -50;
|
||||
|
||||
/**
|
||||
* @var string
|
||||
*/
|
||||
private $horizontalOffset = ' ';
|
||||
|
||||
/**
|
||||
* Represents: (NUL, HT, LF, FF, CR, SP)
|
||||
*
|
||||
* @var string
|
||||
*/
|
||||
private $pdfWhitespaces = "\0\t\n\f\r ";
|
||||
|
||||
/**
|
||||
* Represents: (NUL, HT, LF, FF, CR, SP)
|
||||
*
|
||||
* @var string
|
||||
*/
|
||||
private $pdfWhitespacesRegex = '[\0\t\n\f\r ]';
|
||||
|
||||
/**
|
||||
* Whether to retain raw image data as content or discard it to save memory
|
||||
*
|
||||
* @var bool
|
||||
*/
|
||||
private $retainImageContent = true;
|
||||
|
||||
/**
|
||||
* Memory limit to use when de-compressing files, in bytes.
|
||||
*
|
||||
* @var int
|
||||
*/
|
||||
private $decodeMemoryLimit = 0;
|
||||
|
||||
/**
|
||||
* Whether to include font id and size in dataTm array
|
||||
*
|
||||
* @var bool
|
||||
*/
|
||||
private $dataTmFontInfoHasToBeIncluded = false;
|
||||
|
||||
/**
|
||||
* Whether to attempt to read PDFs even if they are marked as encrypted.
|
||||
*
|
||||
* @var bool
|
||||
*/
|
||||
private $ignoreEncryption = false;
|
||||
|
||||
public function getFontSpaceLimit()
|
||||
{
|
||||
return $this->fontSpaceLimit;
|
||||
}
|
||||
|
||||
public function setFontSpaceLimit($value)
|
||||
{
|
||||
$this->fontSpaceLimit = $value;
|
||||
}
|
||||
|
||||
public function getHorizontalOffset(): string
|
||||
{
|
||||
return $this->horizontalOffset;
|
||||
}
|
||||
|
||||
public function setHorizontalOffset($value): void
|
||||
{
|
||||
$this->horizontalOffset = $value;
|
||||
}
|
||||
|
||||
public function getPdfWhitespaces(): string
|
||||
{
|
||||
return $this->pdfWhitespaces;
|
||||
}
|
||||
|
||||
public function setPdfWhitespaces(string $pdfWhitespaces): void
|
||||
{
|
||||
$this->pdfWhitespaces = $pdfWhitespaces;
|
||||
}
|
||||
|
||||
public function getPdfWhitespacesRegex(): string
|
||||
{
|
||||
return $this->pdfWhitespacesRegex;
|
||||
}
|
||||
|
||||
public function setPdfWhitespacesRegex(string $pdfWhitespacesRegex): void
|
||||
{
|
||||
$this->pdfWhitespacesRegex = $pdfWhitespacesRegex;
|
||||
}
|
||||
|
||||
public function getRetainImageContent(): bool
|
||||
{
|
||||
return $this->retainImageContent;
|
||||
}
|
||||
|
||||
public function setRetainImageContent(bool $retainImageContent): void
|
||||
{
|
||||
$this->retainImageContent = $retainImageContent;
|
||||
}
|
||||
|
||||
public function getDecodeMemoryLimit(): int
|
||||
{
|
||||
return $this->decodeMemoryLimit;
|
||||
}
|
||||
|
||||
public function setDecodeMemoryLimit(int $decodeMemoryLimit): void
|
||||
{
|
||||
$this->decodeMemoryLimit = $decodeMemoryLimit;
|
||||
}
|
||||
|
||||
public function getDataTmFontInfoHasToBeIncluded(): bool
|
||||
{
|
||||
return $this->dataTmFontInfoHasToBeIncluded;
|
||||
}
|
||||
|
||||
public function setDataTmFontInfoHasToBeIncluded(bool $dataTmFontInfoHasToBeIncluded): void
|
||||
{
|
||||
$this->dataTmFontInfoHasToBeIncluded = $dataTmFontInfoHasToBeIncluded;
|
||||
}
|
||||
|
||||
public function getIgnoreEncryption(): bool
|
||||
{
|
||||
return $this->ignoreEncryption;
|
||||
}
|
||||
|
||||
/**
|
||||
* @deprecated this is a temporary workaround, don't rely on it
|
||||
* @see https://github.com/smalot/pdfparser/pull/653
|
||||
*/
|
||||
public function setIgnoreEncryption(bool $ignoreEncryption): void
|
||||
{
|
||||
$this->ignoreEncryption = $ignoreEncryption;
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,470 @@
|
||||
<?php
|
||||
|
||||
/**
|
||||
* @file
|
||||
* This file is part of the PdfParser library.
|
||||
*
|
||||
* @author Sébastien MALOT <sebastien@malot.fr>
|
||||
*
|
||||
* @date 2017-01-03
|
||||
*
|
||||
* @license LGPLv3
|
||||
*
|
||||
* @url <https://github.com/smalot/pdfparser>
|
||||
*
|
||||
* PdfParser is a pdf library written in PHP, extraction oriented.
|
||||
* Copyright (C) 2017 - Sébastien MALOT <sebastien@malot.fr>
|
||||
*
|
||||
* This program is free software: you can redistribute it and/or modify
|
||||
* it under the terms of the GNU Lesser General Public License as published by
|
||||
* the Free Software Foundation, either version 3 of the License, or
|
||||
* (at your option) any later version.
|
||||
*
|
||||
* This program is distributed in the hope that it will be useful,
|
||||
* but WITHOUT ANY WARRANTY; without even the implied warranty of
|
||||
* MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
|
||||
* GNU Lesser General Public License for more details.
|
||||
*
|
||||
* You should have received a copy of the GNU Lesser General Public License
|
||||
* along with this program.
|
||||
* If not, see <http://www.pdfparser.org/sites/default/LICENSE.txt>.
|
||||
*/
|
||||
|
||||
namespace Smalot\PdfParser;
|
||||
|
||||
use Smalot\PdfParser\Encoding\PDFDocEncoding;
|
||||
use Smalot\PdfParser\Exception\MissingCatalogException;
|
||||
|
||||
/**
|
||||
* Technical references :
|
||||
* - http://www.mactech.com/articles/mactech/Vol.15/15.09/PDFIntro/index.html
|
||||
* - http://framework.zend.com/issues/secure/attachment/12512/Pdf.php
|
||||
* - http://www.php.net/manual/en/ref.pdf.php#74211
|
||||
* - http://cpansearch.perl.org/src/JV/PostScript-Font-1.10.02/lib/PostScript/ISOLatin1Encoding.pm
|
||||
* - http://cpansearch.perl.org/src/JV/PostScript-Font-1.10.02/lib/PostScript/ISOLatin9Encoding.pm
|
||||
* - http://cpansearch.perl.org/src/JV/PostScript-Font-1.10.02/lib/PostScript/StandardEncoding.pm
|
||||
* - http://cpansearch.perl.org/src/JV/PostScript-Font-1.10.02/lib/PostScript/WinAnsiEncoding.pm
|
||||
*
|
||||
* Class Document
|
||||
*/
|
||||
class Document
|
||||
{
|
||||
/**
|
||||
* @var PDFObject[]
|
||||
*/
|
||||
protected $objects = [];
|
||||
|
||||
/**
|
||||
* @var array
|
||||
*/
|
||||
protected $dictionary = [];
|
||||
|
||||
/**
|
||||
* @var Header
|
||||
*/
|
||||
protected $trailer;
|
||||
|
||||
/**
|
||||
* @var array<mixed>
|
||||
*/
|
||||
protected $metadata = [];
|
||||
|
||||
/**
|
||||
* @var array
|
||||
*/
|
||||
protected $details;
|
||||
|
||||
public function __construct()
|
||||
{
|
||||
$this->trailer = new Header([], $this);
|
||||
}
|
||||
|
||||
public function init()
|
||||
{
|
||||
$this->buildDictionary();
|
||||
|
||||
$this->buildDetails();
|
||||
|
||||
// Propagate init to objects.
|
||||
foreach ($this->objects as $object) {
|
||||
$object->getHeader()->init();
|
||||
$object->init();
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Build dictionary based on type header field.
|
||||
*/
|
||||
protected function buildDictionary()
|
||||
{
|
||||
// Build dictionary.
|
||||
$this->dictionary = [];
|
||||
|
||||
foreach ($this->objects as $id => $object) {
|
||||
// Cache objects by type and subtype
|
||||
$type = $object->getHeader()->get('Type')->getContent();
|
||||
|
||||
if (null != $type) {
|
||||
if (!isset($this->dictionary[$type])) {
|
||||
$this->dictionary[$type] = [
|
||||
'all' => [],
|
||||
'subtype' => [],
|
||||
];
|
||||
}
|
||||
|
||||
$this->dictionary[$type]['all'][$id] = $object;
|
||||
|
||||
$subtype = $object->getHeader()->get('Subtype')->getContent();
|
||||
if (null != $subtype) {
|
||||
if (!isset($this->dictionary[$type]['subtype'][$subtype])) {
|
||||
$this->dictionary[$type]['subtype'][$subtype] = [];
|
||||
}
|
||||
$this->dictionary[$type]['subtype'][$subtype][$id] = $object;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Build details array.
|
||||
*/
|
||||
protected function buildDetails()
|
||||
{
|
||||
// Build details array.
|
||||
$details = [];
|
||||
|
||||
// Extract document info
|
||||
if ($this->trailer->has('Info')) {
|
||||
/** @var PDFObject $info */
|
||||
$info = $this->trailer->get('Info');
|
||||
// This could be an ElementMissing object, so we need to check for
|
||||
// the getHeader method first.
|
||||
if (null !== $info && method_exists($info, 'getHeader')) {
|
||||
$details = $info->getHeader()->getDetails();
|
||||
}
|
||||
}
|
||||
|
||||
// Retrieve the page count
|
||||
try {
|
||||
$pages = $this->getPages();
|
||||
$details['Pages'] = \count($pages);
|
||||
} catch (\Exception $e) {
|
||||
$details['Pages'] = 0;
|
||||
}
|
||||
|
||||
// Decode and repair encoded document properties
|
||||
foreach ($details as $key => $value) {
|
||||
if (\is_string($value)) {
|
||||
// If the string is already UTF-8 encoded, that means we only
|
||||
// need to repair Adobe's ham-fisted insertion of line-feeds
|
||||
// every ~127 characters, which doesn't seem to be multi-byte
|
||||
// safe
|
||||
if (mb_check_encoding($value, 'UTF-8')) {
|
||||
// Remove literal backslash + line-feed "\\r"
|
||||
$value = str_replace("\x5c\x0d", '', $value);
|
||||
|
||||
// Remove backslash plus bytes written into high part of
|
||||
// multibyte unicode character
|
||||
while (preg_match("/\x5c\x5c\xe0([\xb4-\xb8])(.)/", $value, $match)) {
|
||||
$diff = (\ord($match[1]) - 182) * 64;
|
||||
$newbyte = PDFDocEncoding::convertPDFDoc2UTF8(\chr(\ord($match[2]) + $diff));
|
||||
$value = preg_replace("/\x5c\x5c\xe0".$match[1].$match[2].'/', $newbyte, $value);
|
||||
}
|
||||
|
||||
// Remove bytes written into low part of multibyte unicode
|
||||
// character
|
||||
while (preg_match("/(.)\x9c\xe0([\xb3-\xb7])/", $value, $match)) {
|
||||
$diff = \ord($match[2]) - 181;
|
||||
$newbyte = \chr(\ord($match[1]) + $diff);
|
||||
$value = preg_replace('/'.$match[1]."\x9c\xe0".$match[2].'/', $newbyte, $value);
|
||||
}
|
||||
|
||||
// Remove this byte string that Adobe occasionally adds
|
||||
// between two single byte characters in a unicode string
|
||||
$value = str_replace("\xe5\xb0\x8d", '', $value);
|
||||
|
||||
$details[$key] = $value;
|
||||
} else {
|
||||
// If the string is just PDFDocEncoding, remove any line-feeds
|
||||
// and decode the whole thing.
|
||||
$value = str_replace("\\\r", '', $value);
|
||||
$details[$key] = PDFDocEncoding::convertPDFDoc2UTF8($value);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
$details = array_merge($details, $this->metadata);
|
||||
|
||||
$this->details = $details;
|
||||
}
|
||||
|
||||
/**
|
||||
* Extract XMP Metadata
|
||||
*/
|
||||
public function extractXMPMetadata(string $content): void
|
||||
{
|
||||
$xml = xml_parser_create();
|
||||
xml_parser_set_option($xml, \XML_OPTION_SKIP_WHITE, 1);
|
||||
|
||||
if (1 === xml_parse_into_struct($xml, $content, $values, $index)) {
|
||||
/*
|
||||
* short overview about the following code parts:
|
||||
*
|
||||
* The output of xml_parse_into_struct is a single dimensional array (= $values), and the $stack is a last-on,
|
||||
* first-off array of pointers to positions in $metadata, while iterating through it, that potentially turn the
|
||||
* results into a more intuitive multi-dimensional array. When an "open" XML tag is encountered,
|
||||
* we save the current $metadata context in the $stack, then create a child array of $metadata and
|
||||
* make that the current $metadata context. When a "close" XML tag is encountered, the operations are
|
||||
* reversed: the most recently added $metadata context from $stack (IOW, the parent of the current
|
||||
* element) is set as the current $metadata context.
|
||||
*/
|
||||
$metadata = [];
|
||||
$stack = [];
|
||||
foreach ($values as $val) {
|
||||
// Standardize to lowercase
|
||||
$val['tag'] = strtolower($val['tag']);
|
||||
|
||||
// Ignore structural x: and rdf: XML elements
|
||||
if (0 === strpos($val['tag'], 'x:')) {
|
||||
continue;
|
||||
} elseif (0 === strpos($val['tag'], 'rdf:') && 'rdf:li' != $val['tag']) {
|
||||
continue;
|
||||
}
|
||||
|
||||
switch ($val['type']) {
|
||||
case 'open':
|
||||
// Create an array of list items
|
||||
if ('rdf:li' == $val['tag']) {
|
||||
$metadata[] = [];
|
||||
|
||||
// Move up one level in the stack
|
||||
$stack[\count($stack)] = &$metadata;
|
||||
$metadata = &$metadata[\count($metadata) - 1];
|
||||
} else {
|
||||
// Else create an array of named values
|
||||
$metadata[$val['tag']] = [];
|
||||
|
||||
// Move up one level in the stack
|
||||
$stack[\count($stack)] = &$metadata;
|
||||
$metadata = &$metadata[$val['tag']];
|
||||
}
|
||||
break;
|
||||
|
||||
case 'complete':
|
||||
if (isset($val['value'])) {
|
||||
// Assign a value to this list item
|
||||
if ('rdf:li' == $val['tag']) {
|
||||
$metadata[] = $val['value'];
|
||||
|
||||
// Else assign a value to this property
|
||||
} else {
|
||||
$metadata[$val['tag']] = $val['value'];
|
||||
}
|
||||
}
|
||||
break;
|
||||
|
||||
case 'close':
|
||||
// If the value of this property is an array
|
||||
if (\is_array($metadata)) {
|
||||
// If the value is a single element array
|
||||
// where the element is of type string, use
|
||||
// the value of the first list item as the
|
||||
// value for this property
|
||||
if (1 == \count($metadata) && isset($metadata[0]) && \is_string($metadata[0])) {
|
||||
$metadata = $metadata[0];
|
||||
} elseif (0 == \count($metadata)) {
|
||||
// if the value is an empty array, set
|
||||
// the value of this property to the empty
|
||||
// string
|
||||
$metadata = '';
|
||||
}
|
||||
}
|
||||
|
||||
// Move down one level in the stack
|
||||
$metadata = &$stack[\count($stack) - 1];
|
||||
unset($stack[\count($stack) - 1]);
|
||||
break;
|
||||
}
|
||||
}
|
||||
|
||||
// Only use this metadata if it's referring to a PDF
|
||||
if (!isset($metadata['dc:format']) || 'application/pdf' == $metadata['dc:format']) {
|
||||
// According to the XMP specifications: 'Conflict resolution
|
||||
// for separate packets that describe the same resource is
|
||||
// beyond the scope of this document.' - Section 6.1
|
||||
// Source: https://www.adobe.com/devnet/xmp.html
|
||||
// Source: https://github.com/adobe/XMP-Toolkit-SDK/blob/main/docs/XMPSpecificationPart1.pdf
|
||||
// So if there are multiple XMP blocks, just merge the values
|
||||
// of each found block over top of the existing values
|
||||
$this->metadata = array_merge($this->metadata, $metadata);
|
||||
}
|
||||
}
|
||||
|
||||
// TODO: remove this if-clause and its content when dropping PHP 7 support
|
||||
if (version_compare(PHP_VERSION, '8.0.0', '<')) {
|
||||
// ref: https://www.php.net/manual/en/function.xml-parser-free.php
|
||||
xml_parser_free($xml);
|
||||
|
||||
// to avoid memory leaks; documentation said:
|
||||
// > it was necessary to also explicitly unset the reference to parser to avoid memory leaks
|
||||
unset($xml);
|
||||
}
|
||||
}
|
||||
|
||||
public function getDictionary(): array
|
||||
{
|
||||
return $this->dictionary;
|
||||
}
|
||||
|
||||
/**
|
||||
* @param PDFObject[] $objects
|
||||
*/
|
||||
public function setObjects($objects = [])
|
||||
{
|
||||
$this->objects = (array) $objects;
|
||||
|
||||
$this->init();
|
||||
}
|
||||
|
||||
/**
|
||||
* @return PDFObject[]
|
||||
*/
|
||||
public function getObjects()
|
||||
{
|
||||
return $this->objects;
|
||||
}
|
||||
|
||||
/**
|
||||
* @return PDFObject|Font|Page|Element|null
|
||||
*/
|
||||
public function getObjectById(string $id)
|
||||
{
|
||||
if (isset($this->objects[$id])) {
|
||||
return $this->objects[$id];
|
||||
}
|
||||
|
||||
return null;
|
||||
}
|
||||
|
||||
public function hasObjectsByType(string $type, ?string $subtype = null): bool
|
||||
{
|
||||
return 0 < \count($this->getObjectsByType($type, $subtype));
|
||||
}
|
||||
|
||||
public function getObjectsByType(string $type, ?string $subtype = null): array
|
||||
{
|
||||
if (!isset($this->dictionary[$type])) {
|
||||
return [];
|
||||
}
|
||||
|
||||
if (null != $subtype) {
|
||||
if (!isset($this->dictionary[$type]['subtype'][$subtype])) {
|
||||
return [];
|
||||
}
|
||||
|
||||
return $this->dictionary[$type]['subtype'][$subtype];
|
||||
}
|
||||
|
||||
return $this->dictionary[$type]['all'];
|
||||
}
|
||||
|
||||
/**
|
||||
* @return Font[]
|
||||
*/
|
||||
public function getFonts()
|
||||
{
|
||||
return $this->getObjectsByType('Font');
|
||||
}
|
||||
|
||||
public function getFirstFont(): ?Font
|
||||
{
|
||||
$fonts = $this->getFonts();
|
||||
if ([] === $fonts) {
|
||||
return null;
|
||||
}
|
||||
|
||||
return reset($fonts);
|
||||
}
|
||||
|
||||
/**
|
||||
* @return Page[]
|
||||
*
|
||||
* @throws MissingCatalogException
|
||||
*/
|
||||
public function getPages()
|
||||
{
|
||||
if ($this->hasObjectsByType('Catalog')) {
|
||||
// Search for catalog to list pages.
|
||||
$catalogues = $this->getObjectsByType('Catalog');
|
||||
$catalogue = reset($catalogues);
|
||||
|
||||
/** @var Pages $object */
|
||||
$object = $catalogue->get('Pages');
|
||||
if (method_exists($object, 'getPages')) {
|
||||
return $object->getPages(true);
|
||||
}
|
||||
}
|
||||
|
||||
if ($this->hasObjectsByType('Pages')) {
|
||||
// Search for pages to list kids.
|
||||
$pages = [];
|
||||
|
||||
/** @var Pages[] $objects */
|
||||
$objects = $this->getObjectsByType('Pages');
|
||||
foreach ($objects as $object) {
|
||||
$pages = array_merge($pages, $object->getPages(true));
|
||||
}
|
||||
|
||||
return $pages;
|
||||
}
|
||||
|
||||
if ($this->hasObjectsByType('Page')) {
|
||||
// Search for 'page' (unordered pages).
|
||||
$pages = $this->getObjectsByType('Page');
|
||||
|
||||
return array_values($pages);
|
||||
}
|
||||
|
||||
throw new MissingCatalogException('Missing catalog.');
|
||||
}
|
||||
|
||||
public function getText(?int $pageLimit = null): string
|
||||
{
|
||||
$texts = [];
|
||||
$pages = $this->getPages();
|
||||
|
||||
// Only use the first X number of pages if $pageLimit is set and numeric.
|
||||
if (\is_int($pageLimit) && 0 < $pageLimit) {
|
||||
$pages = \array_slice($pages, 0, $pageLimit);
|
||||
}
|
||||
|
||||
foreach ($pages as $index => $page) {
|
||||
/**
|
||||
* In some cases, the $page variable may be null.
|
||||
*/
|
||||
if (null === $page) {
|
||||
continue;
|
||||
}
|
||||
if ($text = trim($page->getText())) {
|
||||
$texts[] = $text;
|
||||
}
|
||||
}
|
||||
|
||||
return implode("\n\n", $texts);
|
||||
}
|
||||
|
||||
public function getTrailer(): Header
|
||||
{
|
||||
return $this->trailer;
|
||||
}
|
||||
|
||||
public function setTrailer(Header $trailer)
|
||||
{
|
||||
$this->trailer = $trailer;
|
||||
}
|
||||
|
||||
public function getDetails(): array
|
||||
{
|
||||
return $this->details;
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,156 @@
|
||||
<?php
|
||||
|
||||
/**
|
||||
* @file
|
||||
* This file is part of the PdfParser library.
|
||||
*
|
||||
* @author Sébastien MALOT <sebastien@malot.fr>
|
||||
*
|
||||
* @date 2017-01-03
|
||||
*
|
||||
* @license LGPLv3
|
||||
*
|
||||
* @url <https://github.com/smalot/pdfparser>
|
||||
*
|
||||
* PdfParser is a pdf library written in PHP, extraction oriented.
|
||||
* Copyright (C) 2017 - Sébastien MALOT <sebastien@malot.fr>
|
||||
*
|
||||
* This program is free software: you can redistribute it and/or modify
|
||||
* it under the terms of the GNU Lesser General Public License as published by
|
||||
* the Free Software Foundation, either version 3 of the License, or
|
||||
* (at your option) any later version.
|
||||
*
|
||||
* This program is distributed in the hope that it will be useful,
|
||||
* but WITHOUT ANY WARRANTY; without even the implied warranty of
|
||||
* MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
|
||||
* GNU Lesser General Public License for more details.
|
||||
*
|
||||
* You should have received a copy of the GNU Lesser General Public License
|
||||
* along with this program.
|
||||
* If not, see <http://www.pdfparser.org/sites/default/LICENSE.txt>.
|
||||
*/
|
||||
|
||||
namespace Smalot\PdfParser;
|
||||
|
||||
use Smalot\PdfParser\Element\ElementArray;
|
||||
use Smalot\PdfParser\Element\ElementBoolean;
|
||||
use Smalot\PdfParser\Element\ElementDate;
|
||||
use Smalot\PdfParser\Element\ElementHexa;
|
||||
use Smalot\PdfParser\Element\ElementName;
|
||||
use Smalot\PdfParser\Element\ElementNull;
|
||||
use Smalot\PdfParser\Element\ElementNumeric;
|
||||
use Smalot\PdfParser\Element\ElementString;
|
||||
use Smalot\PdfParser\Element\ElementStruct;
|
||||
use Smalot\PdfParser\Element\ElementXRef;
|
||||
|
||||
/**
|
||||
* Class Element
|
||||
*/
|
||||
class Element
|
||||
{
|
||||
/**
|
||||
* @var Document|null
|
||||
*/
|
||||
protected $document;
|
||||
|
||||
protected $value;
|
||||
|
||||
public function __construct($value, ?Document $document = null)
|
||||
{
|
||||
$this->value = $value;
|
||||
$this->document = $document;
|
||||
}
|
||||
|
||||
public function init()
|
||||
{
|
||||
}
|
||||
|
||||
public function equals($value): bool
|
||||
{
|
||||
return $value == $this->value;
|
||||
}
|
||||
|
||||
public function contains($value): bool
|
||||
{
|
||||
if (\is_array($this->value)) {
|
||||
/** @var Element $val */
|
||||
foreach ($this->value as $val) {
|
||||
if ($val->equals($value)) {
|
||||
return true;
|
||||
}
|
||||
}
|
||||
|
||||
return false;
|
||||
}
|
||||
|
||||
return $this->equals($value);
|
||||
}
|
||||
|
||||
public function getContent()
|
||||
{
|
||||
return $this->value;
|
||||
}
|
||||
|
||||
public function __toString(): string
|
||||
{
|
||||
return (string) $this->value;
|
||||
}
|
||||
|
||||
public static function parse(string $content, ?Document $document = null, int &$position = 0)
|
||||
{
|
||||
$args = \func_get_args();
|
||||
$only_values = isset($args[3]) ? $args[3] : false;
|
||||
$content = trim($content);
|
||||
$values = [];
|
||||
|
||||
do {
|
||||
$old_position = $position;
|
||||
|
||||
if (!$only_values) {
|
||||
if (!preg_match('/\G\s*(?P<name>\/[A-Z#0-9\._]+)(?P<value>.*)/si', $content, $match, 0, $position)) {
|
||||
break;
|
||||
} else {
|
||||
$name = preg_replace_callback(
|
||||
'/#([0-9a-f]{2})/i',
|
||||
function ($m) {
|
||||
return \chr(base_convert($m[1], 16, 10));
|
||||
},
|
||||
ltrim($match['name'], '/')
|
||||
);
|
||||
$value = $match['value'];
|
||||
$position = strpos($content, $value, $position + \strlen($match['name']));
|
||||
}
|
||||
} else {
|
||||
$name = \count($values);
|
||||
$value = substr($content, $position);
|
||||
}
|
||||
|
||||
if ($element = ElementName::parse($value, $document, $position)) {
|
||||
$values[$name] = $element;
|
||||
} elseif ($element = ElementXRef::parse($value, $document, $position)) {
|
||||
$values[$name] = $element;
|
||||
} elseif ($element = ElementNumeric::parse($value, $document, $position)) {
|
||||
$values[$name] = $element;
|
||||
} elseif ($element = ElementStruct::parse($value, $document, $position)) {
|
||||
$values[$name] = $element;
|
||||
} elseif ($element = ElementBoolean::parse($value, $document, $position)) {
|
||||
$values[$name] = $element;
|
||||
} elseif ($element = ElementNull::parse($value, $document, $position)) {
|
||||
$values[$name] = $element;
|
||||
} elseif ($element = ElementDate::parse($value, $document, $position)) {
|
||||
$values[$name] = $element;
|
||||
} elseif ($element = ElementString::parse($value, $document, $position)) {
|
||||
$values[$name] = $element;
|
||||
} elseif ($element = ElementHexa::parse($value, $document, $position)) {
|
||||
$values[$name] = $element;
|
||||
} elseif ($element = ElementArray::parse($value, $document, $position)) {
|
||||
$values[$name] = $element;
|
||||
} else {
|
||||
$position = $old_position;
|
||||
break;
|
||||
}
|
||||
} while ($position < \strlen($content));
|
||||
|
||||
return $values;
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,139 @@
|
||||
<?php
|
||||
|
||||
/**
|
||||
* @file
|
||||
* This file is part of the PdfParser library.
|
||||
*
|
||||
* @author Sébastien MALOT <sebastien@malot.fr>
|
||||
*
|
||||
* @date 2017-01-03
|
||||
*
|
||||
* @license LGPLv3
|
||||
*
|
||||
* @url <https://github.com/smalot/pdfparser>
|
||||
*
|
||||
* PdfParser is a pdf library written in PHP, extraction oriented.
|
||||
* Copyright (C) 2017 - Sébastien MALOT <sebastien@malot.fr>
|
||||
*
|
||||
* This program is free software: you can redistribute it and/or modify
|
||||
* it under the terms of the GNU Lesser General Public License as published by
|
||||
* the Free Software Foundation, either version 3 of the License, or
|
||||
* (at your option) any later version.
|
||||
*
|
||||
* This program is distributed in the hope that it will be useful,
|
||||
* but WITHOUT ANY WARRANTY; without even the implied warranty of
|
||||
* MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
|
||||
* GNU Lesser General Public License for more details.
|
||||
*
|
||||
* You should have received a copy of the GNU Lesser General Public License
|
||||
* along with this program.
|
||||
* If not, see <http://www.pdfparser.org/sites/default/LICENSE.txt>.
|
||||
*/
|
||||
|
||||
namespace Smalot\PdfParser\Element;
|
||||
|
||||
use Smalot\PdfParser\Document;
|
||||
use Smalot\PdfParser\Element;
|
||||
use Smalot\PdfParser\Header;
|
||||
use Smalot\PdfParser\PDFObject;
|
||||
|
||||
/**
|
||||
* Class ElementArray
|
||||
*/
|
||||
class ElementArray extends Element
|
||||
{
|
||||
public function __construct($value, ?Document $document = null)
|
||||
{
|
||||
parent::__construct($value, $document);
|
||||
}
|
||||
|
||||
public function getContent()
|
||||
{
|
||||
foreach ($this->value as $name => $element) {
|
||||
$this->resolveXRef($name);
|
||||
}
|
||||
|
||||
return parent::getContent();
|
||||
}
|
||||
|
||||
public function getRawContent(): array
|
||||
{
|
||||
return $this->value;
|
||||
}
|
||||
|
||||
public function getDetails(bool $deep = true): array
|
||||
{
|
||||
$values = [];
|
||||
$elements = $this->getContent();
|
||||
|
||||
foreach ($elements as $key => $element) {
|
||||
if ($element instanceof Header && $deep) {
|
||||
$values[$key] = $element->getDetails($deep);
|
||||
} elseif ($element instanceof PDFObject && $deep) {
|
||||
$values[$key] = $element->getDetails(false);
|
||||
} elseif ($element instanceof self) {
|
||||
if ($deep) {
|
||||
$values[$key] = $element->getDetails();
|
||||
}
|
||||
} elseif ($element instanceof Element && !($element instanceof self)) {
|
||||
$values[$key] = $element->getContent();
|
||||
}
|
||||
}
|
||||
|
||||
return $values;
|
||||
}
|
||||
|
||||
public function __toString(): string
|
||||
{
|
||||
return implode(',', $this->value);
|
||||
}
|
||||
|
||||
/**
|
||||
* @return Element|PDFObject
|
||||
*/
|
||||
protected function resolveXRef(string $name)
|
||||
{
|
||||
if (($obj = $this->value[$name]) instanceof ElementXRef) {
|
||||
/** @var ElementXRef $obj */
|
||||
$obj = $this->document->getObjectById($obj->getId());
|
||||
$this->value[$name] = $obj;
|
||||
}
|
||||
|
||||
return $this->value[$name];
|
||||
}
|
||||
|
||||
/**
|
||||
* @todo: These methods return mixed and mismatched types throughout the hierarchy
|
||||
*
|
||||
* @return bool|ElementArray
|
||||
*/
|
||||
public static function parse(string $content, ?Document $document = null, int &$offset = 0)
|
||||
{
|
||||
if (preg_match('/^\s*\[(?P<array>.*)/is', $content, $match)) {
|
||||
preg_match_all('/(.*?)(\[|\])/s', trim($content), $matches);
|
||||
|
||||
$level = 0;
|
||||
$sub = '';
|
||||
foreach ($matches[0] as $part) {
|
||||
$sub .= $part;
|
||||
$level += (false !== strpos($part, '[') ? 1 : -1);
|
||||
if ($level <= 0) {
|
||||
break;
|
||||
}
|
||||
}
|
||||
|
||||
// Removes 1 level [ and ].
|
||||
$sub = substr(trim($sub), 1, -1);
|
||||
$sub_offset = 0;
|
||||
$values = Element::parse($sub, $document, $sub_offset, true);
|
||||
|
||||
$offset += strpos($content, '[') + 1;
|
||||
// Find next ']' position
|
||||
$offset += \strlen($sub) + 1;
|
||||
|
||||
return new self($values, $document);
|
||||
}
|
||||
|
||||
return false;
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,75 @@
|
||||
<?php
|
||||
|
||||
/**
|
||||
* @file
|
||||
* This file is part of the PdfParser library.
|
||||
*
|
||||
* @author Sébastien MALOT <sebastien@malot.fr>
|
||||
*
|
||||
* @date 2017-01-03
|
||||
*
|
||||
* @license LGPLv3
|
||||
*
|
||||
* @url <https://github.com/smalot/pdfparser>
|
||||
*
|
||||
* PdfParser is a pdf library written in PHP, extraction oriented.
|
||||
* Copyright (C) 2017 - Sébastien MALOT <sebastien@malot.fr>
|
||||
*
|
||||
* This program is free software: you can redistribute it and/or modify
|
||||
* it under the terms of the GNU Lesser General Public License as published by
|
||||
* the Free Software Foundation, either version 3 of the License, or
|
||||
* (at your option) any later version.
|
||||
*
|
||||
* This program is distributed in the hope that it will be useful,
|
||||
* but WITHOUT ANY WARRANTY; without even the implied warranty of
|
||||
* MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
|
||||
* GNU Lesser General Public License for more details.
|
||||
*
|
||||
* You should have received a copy of the GNU Lesser General Public License
|
||||
* along with this program.
|
||||
* If not, see <http://www.pdfparser.org/sites/default/LICENSE.txt>.
|
||||
*/
|
||||
|
||||
namespace Smalot\PdfParser\Element;
|
||||
|
||||
use Smalot\PdfParser\Document;
|
||||
use Smalot\PdfParser\Element;
|
||||
|
||||
/**
|
||||
* Class ElementBoolean
|
||||
*/
|
||||
class ElementBoolean extends Element
|
||||
{
|
||||
/**
|
||||
* @param string|bool $value
|
||||
*/
|
||||
public function __construct($value)
|
||||
{
|
||||
parent::__construct('true' == strtolower($value) || true === $value, null);
|
||||
}
|
||||
|
||||
public function __toString(): string
|
||||
{
|
||||
return $this->value ? 'true' : 'false';
|
||||
}
|
||||
|
||||
public function equals($value): bool
|
||||
{
|
||||
return $this->getContent() === $value;
|
||||
}
|
||||
|
||||
/**
|
||||
* @return bool|ElementBoolean
|
||||
*/
|
||||
public static function parse(string $content, ?Document $document = null, int &$offset = 0)
|
||||
{
|
||||
if (preg_match('/^\s*(?P<value>true|false)/is', $content, $match)) {
|
||||
$value = $match['value'];
|
||||
$offset += strpos($content, $value) + \strlen($value);
|
||||
|
||||
return new self($value);
|
||||
}
|
||||
|
||||
return false;
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,139 @@
|
||||
<?php
|
||||
|
||||
/**
|
||||
* @file
|
||||
* This file is part of the PdfParser library.
|
||||
*
|
||||
* @author Sébastien MALOT <sebastien@malot.fr>
|
||||
*
|
||||
* @date 2017-01-03
|
||||
*
|
||||
* @license LGPLv3
|
||||
*
|
||||
* @url <https://github.com/smalot/pdfparser>
|
||||
*
|
||||
* PdfParser is a pdf library written in PHPi, extraction oriented.
|
||||
* Copyright (C) 2017 - Sébastien MALOT <sebastien@malot.fr>
|
||||
*
|
||||
* This program is free software: you can redistribute it and/or modify
|
||||
* it under the terms of the GNU Lesser General Public License as published by
|
||||
* the Free Software Foundation, either version 3 of the License, or
|
||||
* (at your option) any later version.
|
||||
*
|
||||
* This program is distributed in the hope that it will be useful,
|
||||
* but WITHOUT ANY WARRANTY; without even the implied warranty of
|
||||
* MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
|
||||
* GNU Lesser General Public License for more details.
|
||||
*
|
||||
* You should have received a copy of the GNU Lesser General Public License
|
||||
* along with this program.
|
||||
* If not, see <http://www.pdfparser.org/sites/default/LICENSE.txt>.
|
||||
*/
|
||||
|
||||
namespace Smalot\PdfParser\Element;
|
||||
|
||||
use Smalot\PdfParser\Document;
|
||||
|
||||
/**
|
||||
* Class ElementDate
|
||||
*/
|
||||
class ElementDate extends ElementString
|
||||
{
|
||||
/**
|
||||
* @var array<int,string>
|
||||
*/
|
||||
protected static $formats = [
|
||||
4 => 'Y',
|
||||
6 => 'Ym',
|
||||
8 => 'Ymd',
|
||||
10 => 'YmdH',
|
||||
12 => 'YmdHi',
|
||||
14 => 'YmdHis',
|
||||
15 => 'YmdHise',
|
||||
17 => 'YmdHisO',
|
||||
18 => 'YmdHisO',
|
||||
19 => 'YmdHisO',
|
||||
];
|
||||
|
||||
/**
|
||||
* @var string
|
||||
*/
|
||||
protected $format = 'c';
|
||||
|
||||
/**
|
||||
* @var \DateTime
|
||||
*/
|
||||
protected $value;
|
||||
|
||||
public function __construct($value)
|
||||
{
|
||||
if (!($value instanceof \DateTime)) {
|
||||
throw new \Exception('DateTime required.'); // FIXME: Sometimes strings are passed to this function
|
||||
}
|
||||
|
||||
parent::__construct($value);
|
||||
}
|
||||
|
||||
public function setFormat(string $format)
|
||||
{
|
||||
$this->format = $format;
|
||||
}
|
||||
|
||||
public function equals($value): bool
|
||||
{
|
||||
if ($value instanceof \DateTime) {
|
||||
$timestamp = $value->getTimeStamp();
|
||||
} else {
|
||||
$timestamp = strtotime($value);
|
||||
}
|
||||
|
||||
return $timestamp == $this->value->getTimeStamp();
|
||||
}
|
||||
|
||||
public function __toString(): string
|
||||
{
|
||||
return (string) $this->value->format($this->format);
|
||||
}
|
||||
|
||||
/**
|
||||
* @return bool|ElementDate
|
||||
*/
|
||||
public static function parse(string $content, ?Document $document = null, int &$offset = 0)
|
||||
{
|
||||
if (preg_match('/^\s*\(D\:(?P<name>.*?)\)/s', $content, $match)) {
|
||||
$name = $match['name'];
|
||||
$name = str_replace("'", '', $name);
|
||||
$date = false;
|
||||
|
||||
// Smallest format : Y
|
||||
// Full format : YmdHisP
|
||||
if (preg_match('/^\d{4}(\d{2}(\d{2}(\d{2}(\d{2}(\d{2}(Z(\d{2,4})?|[\+-]?\d{2}(\d{2})?)?)?)?)?)?)?$/', $name)) {
|
||||
if ($pos = strpos($name, 'Z')) {
|
||||
$name = substr($name, 0, $pos + 1);
|
||||
} elseif (18 == \strlen($name) && preg_match('/[^\+-]0000$/', $name)) {
|
||||
$name = substr($name, 0, -4).'+0000';
|
||||
}
|
||||
|
||||
$format = self::$formats[\strlen($name)];
|
||||
$date = \DateTime::createFromFormat($format, $name, new \DateTimeZone('UTC'));
|
||||
} else {
|
||||
// special cases
|
||||
if (preg_match('/^\d{1,2}-\d{1,2}-\d{4},?\s+\d{2}:\d{2}:\d{2}[\+-]\d{4}$/', $name)) {
|
||||
$name = str_replace(',', '', $name);
|
||||
$format = 'n-j-Y H:i:sO';
|
||||
$date = \DateTime::createFromFormat($format, $name, new \DateTimeZone('UTC'));
|
||||
}
|
||||
}
|
||||
|
||||
if (!$date) {
|
||||
return false;
|
||||
}
|
||||
|
||||
$offset += strpos($content, '(D:') + \strlen($match['name']) + 4; // 1 for '(D:' and ')'
|
||||
|
||||
return new self($date);
|
||||
}
|
||||
|
||||
return false;
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,91 @@
|
||||
<?php
|
||||
|
||||
/**
|
||||
* @file
|
||||
* This file is part of the PdfParser library.
|
||||
*
|
||||
* @author Sébastien MALOT <sebastien@malot.fr>
|
||||
*
|
||||
* @date 2017-01-03
|
||||
*
|
||||
* @license LGPLv3
|
||||
*
|
||||
* @url <https://github.com/smalot/pdfparser>
|
||||
*
|
||||
* PdfParser is a pdf library written in PHP, extraction oriented.
|
||||
* Copyright (C) 2017 - Sébastien MALOT <sebastien@malot.fr>
|
||||
*
|
||||
* This program is free software: you can redistribute it and/or modify
|
||||
* it under the terms of the GNU Lesser General Public License as published by
|
||||
* the Free Software Foundation, either version 3 of the License, or
|
||||
* (at your option) any later version.
|
||||
*
|
||||
* This program is distributed in the hope that it will be useful,
|
||||
* but WITHOUT ANY WARRANTY; without even the implied warranty of
|
||||
* MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
|
||||
* GNU Lesser General Public License for more details.
|
||||
*
|
||||
* You should have received a copy of the GNU Lesser General Public License
|
||||
* along with this program.
|
||||
* If not, see <http://www.pdfparser.org/sites/default/LICENSE.txt>.
|
||||
*/
|
||||
|
||||
namespace Smalot\PdfParser\Element;
|
||||
|
||||
use Smalot\PdfParser\Document;
|
||||
|
||||
/**
|
||||
* Class ElementHexa
|
||||
*/
|
||||
class ElementHexa extends ElementString
|
||||
{
|
||||
/**
|
||||
* @return bool|ElementHexa|ElementDate
|
||||
*/
|
||||
public static function parse(string $content, ?Document $document = null, int &$offset = 0)
|
||||
{
|
||||
if (preg_match('/^\s*\<(?P<name>[A-F0-9]+)\>/is', $content, $match)) {
|
||||
$name = $match['name'];
|
||||
$offset += strpos($content, '<'.$name) + \strlen($name) + 2; // 1 for '>'
|
||||
// repackage string as standard
|
||||
$name = '('.self::decode($name).')';
|
||||
$element = ElementDate::parse($name, $document);
|
||||
|
||||
if (!$element) {
|
||||
$element = ElementString::parse($name, $document);
|
||||
}
|
||||
|
||||
return $element;
|
||||
}
|
||||
|
||||
return false;
|
||||
}
|
||||
|
||||
public static function decode(string $value): string
|
||||
{
|
||||
$text = '';
|
||||
|
||||
// Filter $value of non-hexadecimal characters
|
||||
$value = (string) preg_replace('/[^0-9a-f]/i', '', $value);
|
||||
|
||||
// Check for leading zeros (4-byte hexadecimal indicator), or
|
||||
// the BE BOM
|
||||
if ('00' === substr($value, 0, 2) || 'feff' === strtolower(substr($value, 0, 4))) {
|
||||
$value = (string) preg_replace('/^feff/i', '', $value);
|
||||
for ($i = 0, $length = \strlen($value); $i < $length; $i += 4) {
|
||||
$hex = substr($value, $i, 4);
|
||||
$text .= '&#'.str_pad(hexdec($hex), 4, '0', \STR_PAD_LEFT).';';
|
||||
}
|
||||
} else {
|
||||
// Otherwise decode this as 2-byte hexadecimal
|
||||
for ($i = 0, $length = \strlen($value); $i < $length; $i += 2) {
|
||||
$hex = substr($value, $i, 2);
|
||||
$text .= \chr(hexdec($hex));
|
||||
}
|
||||
}
|
||||
|
||||
$text = html_entity_decode($text, \ENT_NOQUOTES, 'UTF-8');
|
||||
|
||||
return $text;
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,66 @@
|
||||
<?php
|
||||
|
||||
/**
|
||||
* @file
|
||||
* This file is part of the PdfParser library.
|
||||
*
|
||||
* @author Sébastien MALOT <sebastien@malot.fr>
|
||||
*
|
||||
* @date 2017-01-03
|
||||
*
|
||||
* @license LGPLv3
|
||||
*
|
||||
* @url <https://github.com/smalot/pdfparser>
|
||||
*
|
||||
* PdfParser is a pdf library written in PHP, extraction oriented.
|
||||
* Copyright (C) 2017 - Sébastien MALOT <sebastien@malot.fr>
|
||||
*
|
||||
* This program is free software: you can redistribute it and/or modify
|
||||
* it under the terms of the GNU Lesser General Public License as published by
|
||||
* the Free Software Foundation, either version 3 of the License, or
|
||||
* (at your option) any later version.
|
||||
*
|
||||
* This program is distributed in the hope that it will be useful,
|
||||
* but WITHOUT ANY WARRANTY; without even the implied warranty of
|
||||
* MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
|
||||
* GNU Lesser General Public License for more details.
|
||||
*
|
||||
* You should have received a copy of the GNU Lesser General Public License
|
||||
* along with this program.
|
||||
* If not, see <http://www.pdfparser.org/sites/default/LICENSE.txt>.
|
||||
*/
|
||||
|
||||
namespace Smalot\PdfParser\Element;
|
||||
|
||||
use Smalot\PdfParser\Element;
|
||||
|
||||
/**
|
||||
* Class ElementMissing
|
||||
*/
|
||||
class ElementMissing extends Element
|
||||
{
|
||||
public function __construct()
|
||||
{
|
||||
parent::__construct(null, null);
|
||||
}
|
||||
|
||||
public function equals($value): bool
|
||||
{
|
||||
return false;
|
||||
}
|
||||
|
||||
public function contains($value): bool
|
||||
{
|
||||
return false;
|
||||
}
|
||||
|
||||
public function getContent(): bool
|
||||
{
|
||||
return false;
|
||||
}
|
||||
|
||||
public function __toString(): string
|
||||
{
|
||||
return '';
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,69 @@
|
||||
<?php
|
||||
|
||||
/**
|
||||
* @file
|
||||
* This file is part of the PdfParser library.
|
||||
*
|
||||
* @author Sébastien MALOT <sebastien@malot.fr>
|
||||
*
|
||||
* @date 2017-01-03
|
||||
*
|
||||
* @license LGPLv3
|
||||
*
|
||||
* @url <https://github.com/smalot/pdfparser>
|
||||
*
|
||||
* PdfParser is a pdf library written in PHP, extraction oriented.
|
||||
* Copyright (C) 2017 - Sébastien MALOT <sebastien@malot.fr>
|
||||
*
|
||||
* This program is free software: you can redistribute it and/or modify
|
||||
* it under the terms of the GNU Lesser General Public License as published by
|
||||
* the Free Software Foundation, either version 3 of the License, or
|
||||
* (at your option) any later version.
|
||||
*
|
||||
* This program is distributed in the hope that it will be useful,
|
||||
* but WITHOUT ANY WARRANTY; without even the implied warranty of
|
||||
* MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
|
||||
* GNU Lesser General Public License for more details.
|
||||
*
|
||||
* You should have received a copy of the GNU Lesser General Public License
|
||||
* along with this program.
|
||||
* If not, see <http://www.pdfparser.org/sites/default/LICENSE.txt>.
|
||||
*/
|
||||
|
||||
namespace Smalot\PdfParser\Element;
|
||||
|
||||
use Smalot\PdfParser\Document;
|
||||
use Smalot\PdfParser\Element;
|
||||
use Smalot\PdfParser\Font;
|
||||
|
||||
/**
|
||||
* Class ElementName
|
||||
*/
|
||||
class ElementName extends Element
|
||||
{
|
||||
public function __construct(string $value)
|
||||
{
|
||||
parent::__construct($value, null);
|
||||
}
|
||||
|
||||
public function equals($value): bool
|
||||
{
|
||||
return $value == $this->value;
|
||||
}
|
||||
|
||||
/**
|
||||
* @return bool|ElementName
|
||||
*/
|
||||
public static function parse(string $content, ?Document $document = null, int &$offset = 0)
|
||||
{
|
||||
if (preg_match('/^\s*\/([A-Z0-9\-\+,#\.]+)/is', $content, $match)) {
|
||||
$name = $match[1];
|
||||
$offset += strpos($content, $name) + \strlen($name);
|
||||
$name = Font::decodeEntities($name);
|
||||
|
||||
return new self($name);
|
||||
}
|
||||
|
||||
return false;
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,71 @@
|
||||
<?php
|
||||
|
||||
/**
|
||||
* @file
|
||||
* This file is part of the PdfParser library.
|
||||
*
|
||||
* @author Sébastien MALOT <sebastien@malot.fr>
|
||||
*
|
||||
* @date 2017-01-03
|
||||
*
|
||||
* @license LGPLv3
|
||||
*
|
||||
* @url <https://github.com/smalot/pdfparser>
|
||||
*
|
||||
* PdfParser is a pdf library written in PHP, extraction oriented.
|
||||
* Copyright (C) 2017 - Sébastien MALOT <sebastien@malot.fr>
|
||||
*
|
||||
* This program is free software: you can redistribute it and/or modify
|
||||
* it under the terms of the GNU Lesser General Public License as published by
|
||||
* the Free Software Foundation, either version 3 of the License, or
|
||||
* (at your option) any later version.
|
||||
*
|
||||
* This program is distributed in the hope that it will be useful,
|
||||
* but WITHOUT ANY WARRANTY; without even the implied warranty of
|
||||
* MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
|
||||
* GNU Lesser General Public License for more details.
|
||||
*
|
||||
* You should have received a copy of the GNU Lesser General Public License
|
||||
* along with this program.
|
||||
* If not, see <http://www.pdfparser.org/sites/default/LICENSE.txt>.
|
||||
*/
|
||||
|
||||
namespace Smalot\PdfParser\Element;
|
||||
|
||||
use Smalot\PdfParser\Document;
|
||||
use Smalot\PdfParser\Element;
|
||||
|
||||
/**
|
||||
* Class ElementNull
|
||||
*/
|
||||
class ElementNull extends Element
|
||||
{
|
||||
public function __construct()
|
||||
{
|
||||
parent::__construct(null, null);
|
||||
}
|
||||
|
||||
public function __toString(): string
|
||||
{
|
||||
return 'null';
|
||||
}
|
||||
|
||||
public function equals($value): bool
|
||||
{
|
||||
return $this->getContent() === $value;
|
||||
}
|
||||
|
||||
/**
|
||||
* @return bool|ElementNull
|
||||
*/
|
||||
public static function parse(string $content, ?Document $document = null, int &$offset = 0)
|
||||
{
|
||||
if (preg_match('/^\s*(null)/s', $content, $match)) {
|
||||
$offset += strpos($content, 'null') + \strlen('null');
|
||||
|
||||
return new self();
|
||||
}
|
||||
|
||||
return false;
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,62 @@
|
||||
<?php
|
||||
|
||||
/**
|
||||
* @file
|
||||
* This file is part of the PdfParser library.
|
||||
*
|
||||
* @author Sébastien MALOT <sebastien@malot.fr>
|
||||
*
|
||||
* @date 2017-01-03
|
||||
*
|
||||
* @license LGPLv3
|
||||
*
|
||||
* @url <https://github.com/smalot/pdfparser>
|
||||
*
|
||||
* PdfParser is a pdf library written in PHP, extraction oriented.
|
||||
* Copyright (C) 2017 - Sébastien MALOT <sebastien@malot.fr>
|
||||
*
|
||||
* This program is free software: you can redistribute it and/or modify
|
||||
* it under the terms of the GNU Lesser General Public License as published by
|
||||
* the Free Software Foundation, either version 3 of the License, or
|
||||
* (at your option) any later version.
|
||||
*
|
||||
* This program is distributed in the hope that it will be useful,
|
||||
* but WITHOUT ANY WARRANTY; without even the implied warranty of
|
||||
* MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
|
||||
* GNU Lesser General Public License for more details.
|
||||
*
|
||||
* You should have received a copy of the GNU Lesser General Public License
|
||||
* along with this program.
|
||||
* If not, see <http://www.pdfparser.org/sites/default/LICENSE.txt>.
|
||||
*/
|
||||
|
||||
namespace Smalot\PdfParser\Element;
|
||||
|
||||
use Smalot\PdfParser\Document;
|
||||
use Smalot\PdfParser\Element;
|
||||
|
||||
/**
|
||||
* Class ElementNumeric
|
||||
*/
|
||||
class ElementNumeric extends Element
|
||||
{
|
||||
public function __construct(string $value)
|
||||
{
|
||||
parent::__construct((float) $value, null);
|
||||
}
|
||||
|
||||
/**
|
||||
* @return bool|ElementNumeric
|
||||
*/
|
||||
public static function parse(string $content, ?Document $document = null, int &$offset = 0)
|
||||
{
|
||||
if (preg_match('/^\s*(?P<value>\-?[0-9\.]+)/s', $content, $match)) {
|
||||
$value = $match['value'];
|
||||
$offset += strpos($content, $value) + \strlen($value);
|
||||
|
||||
return new self($value);
|
||||
}
|
||||
|
||||
return false;
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,93 @@
|
||||
<?php
|
||||
|
||||
/**
|
||||
* @file
|
||||
* This file is part of the PdfParser library.
|
||||
*
|
||||
* @author Sébastien MALOT <sebastien@malot.fr>
|
||||
*
|
||||
* @date 2017-01-03
|
||||
*
|
||||
* @license LGPLv3
|
||||
*
|
||||
* @url <https://github.com/smalot/pdfparser>
|
||||
*
|
||||
* PdfParser is a pdf library written in PHP, extraction oriented.
|
||||
* Copyright (C) 2017 - Sébastien MALOT <sebastien@malot.fr>
|
||||
*
|
||||
* This program is free software: you can redistribute it and/or modify
|
||||
* it under the terms of the GNU Lesser General Public License as published by
|
||||
* the Free Software Foundation, either version 3 of the License, or
|
||||
* (at your option) any later version.
|
||||
*
|
||||
* This program is distributed in the hope that it will be useful,
|
||||
* but WITHOUT ANY WARRANTY; without even the implied warranty of
|
||||
* MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
|
||||
* GNU Lesser General Public License for more details.
|
||||
*
|
||||
* You should have received a copy of the GNU Lesser General Public License
|
||||
* along with this program.
|
||||
* If not, see <http://www.pdfparser.org/sites/default/LICENSE.txt>.
|
||||
*/
|
||||
|
||||
namespace Smalot\PdfParser\Element;
|
||||
|
||||
use Smalot\PdfParser\Document;
|
||||
use Smalot\PdfParser\Element;
|
||||
use Smalot\PdfParser\Font;
|
||||
|
||||
/**
|
||||
* Class ElementString
|
||||
*/
|
||||
class ElementString extends Element
|
||||
{
|
||||
public function __construct($value)
|
||||
{
|
||||
parent::__construct($value, null);
|
||||
}
|
||||
|
||||
public function equals($value): bool
|
||||
{
|
||||
return $value == $this->value;
|
||||
}
|
||||
|
||||
/**
|
||||
* @return bool|ElementString
|
||||
*/
|
||||
public static function parse(string $content, ?Document $document = null, int &$offset = 0)
|
||||
{
|
||||
if (preg_match('/^\s*\((?P<name>.*)/s', $content, $match)) {
|
||||
$name = $match['name'];
|
||||
|
||||
// Find next ')' not escaped.
|
||||
$cur_start_text = $start_search_end = 0;
|
||||
while (false !== ($cur_start_pos = strpos($name, ')', $start_search_end))) {
|
||||
$cur_extract = substr($name, $cur_start_text, $cur_start_pos - $cur_start_text);
|
||||
preg_match('/(?P<escape>[\\\]*)$/s', $cur_extract, $match);
|
||||
if (!(\strlen($match['escape']) % 2)) {
|
||||
break;
|
||||
}
|
||||
$start_search_end = $cur_start_pos + 1;
|
||||
}
|
||||
|
||||
// Extract string.
|
||||
$name = substr($name, 0, (int) $cur_start_pos);
|
||||
$offset += strpos($content, '(') + $cur_start_pos + 2; // 2 for '(' and ')'
|
||||
$name = str_replace(
|
||||
['\\\\', '\\ ', '\\/', '\(', '\)', '\n', '\r', '\t'],
|
||||
['\\', ' ', '/', '(', ')', "\n", "\r", "\t"],
|
||||
$name
|
||||
);
|
||||
|
||||
// Decode string.
|
||||
$name = Font::decodeOctal($name);
|
||||
$name = Font::decodeEntities($name);
|
||||
$name = Font::decodeHexadecimal($name, false);
|
||||
$name = Font::decodeUnicode($name);
|
||||
|
||||
return new self($name);
|
||||
}
|
||||
|
||||
return false;
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,75 @@
|
||||
<?php
|
||||
|
||||
/**
|
||||
* @file
|
||||
* This file is part of the PdfParser library.
|
||||
*
|
||||
* @author Sébastien MALOT <sebastien@malot.fr>
|
||||
*
|
||||
* @date 2017-01-03
|
||||
*
|
||||
* @license LGPLv3
|
||||
*
|
||||
* @url <https://github.com/smalot/pdfparser>
|
||||
*
|
||||
* PdfParser is a pdf library written in PHP, extraction oriented.
|
||||
* Copyright (C) 2017 - Sébastien MALOT <sebastien@malot.fr>
|
||||
*
|
||||
* This program is free software: you can redistribute it and/or modify
|
||||
* it under the terms of the GNU Lesser General Public License as published by
|
||||
* the Free Software Foundation, either version 3 of the License, or
|
||||
* (at your option) any later version.
|
||||
*
|
||||
* This program is distributed in the hope that it will be useful,
|
||||
* but WITHOUT ANY WARRANTY; without even the implied warranty of
|
||||
* MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
|
||||
* GNU Lesser General Public License for more details.
|
||||
*
|
||||
* You should have received a copy of the GNU Lesser General Public License
|
||||
* along with this program.
|
||||
* If not, see <http://www.pdfparser.org/sites/default/LICENSE.txt>.
|
||||
*/
|
||||
|
||||
namespace Smalot\PdfParser\Element;
|
||||
|
||||
use Smalot\PdfParser\Document;
|
||||
use Smalot\PdfParser\Element;
|
||||
use Smalot\PdfParser\Header;
|
||||
|
||||
/**
|
||||
* Class ElementStruct
|
||||
*/
|
||||
class ElementStruct extends Element
|
||||
{
|
||||
/**
|
||||
* @return false|Header
|
||||
*/
|
||||
public static function parse(string $content, ?Document $document = null, int &$offset = 0)
|
||||
{
|
||||
if (preg_match('/^\s*<<(?P<struct>.*)/is', $content)) {
|
||||
preg_match_all('/(.*?)(<<|>>)/s', trim($content), $matches);
|
||||
|
||||
$level = 0;
|
||||
$sub = '';
|
||||
foreach ($matches[0] as $part) {
|
||||
$sub .= $part;
|
||||
$level += (false !== strpos($part, '<<') ? 1 : -1);
|
||||
if ($level <= 0) {
|
||||
break;
|
||||
}
|
||||
}
|
||||
|
||||
$offset += strpos($content, '<<') + \strlen(rtrim($sub));
|
||||
|
||||
// Removes '<<' and '>>'.
|
||||
$sub = trim((string) preg_replace('/^\s*<<(.*)>>\s*$/s', '\\1', $sub));
|
||||
|
||||
$position = 0;
|
||||
$elements = Element::parse($sub, $document, $position);
|
||||
|
||||
return new Header($elements, $document);
|
||||
}
|
||||
|
||||
return false;
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,98 @@
|
||||
<?php
|
||||
|
||||
/**
|
||||
* @file
|
||||
* This file is part of the PdfParser library.
|
||||
*
|
||||
* @author Sébastien MALOT <sebastien@malot.fr>
|
||||
*
|
||||
* @date 2017-01-03
|
||||
*
|
||||
* @license LGPLv3
|
||||
*
|
||||
* @url <https://github.com/smalot/pdfparser>
|
||||
*
|
||||
* PdfParser is a pdf library written in PHP, extraction oriented.
|
||||
* Copyright (C) 2017 - Sébastien MALOT <sebastien@malot.fr>
|
||||
*
|
||||
* This program is free software: you can redistribute it and/or modify
|
||||
* it under the terms of the GNU Lesser General Public License as published by
|
||||
* the Free Software Foundation, either version 3 of the License, or
|
||||
* (at your option) any later version.
|
||||
*
|
||||
* This program is distributed in the hope that it will be useful,
|
||||
* but WITHOUT ANY WARRANTY; without even the implied warranty of
|
||||
* MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
|
||||
* GNU Lesser General Public License for more details.
|
||||
*
|
||||
* You should have received a copy of the GNU Lesser General Public License
|
||||
* along with this program.
|
||||
* If not, see <http://www.pdfparser.org/sites/default/LICENSE.txt>.
|
||||
*/
|
||||
|
||||
namespace Smalot\PdfParser\Element;
|
||||
|
||||
use Smalot\PdfParser\Document;
|
||||
use Smalot\PdfParser\Element;
|
||||
|
||||
/**
|
||||
* Class ElementXRef
|
||||
*/
|
||||
class ElementXRef extends Element
|
||||
{
|
||||
public function getId(): string
|
||||
{
|
||||
return $this->getContent();
|
||||
}
|
||||
|
||||
public function getObject()
|
||||
{
|
||||
return $this->document->getObjectById($this->getId());
|
||||
}
|
||||
|
||||
public function equals($value): bool
|
||||
{
|
||||
/**
|
||||
* In case $value is a number and $this->value is a string like 5_0
|
||||
*
|
||||
* Without this if-clause code like:
|
||||
*
|
||||
* $element = new ElementXRef('5_0');
|
||||
* $this->assertTrue($element->equals(5));
|
||||
*
|
||||
* would fail (= 5_0 and 5 are not equal in PHP 8.0+).
|
||||
*/
|
||||
if (
|
||||
true === is_numeric($value)
|
||||
&& true === \is_string($this->getContent())
|
||||
&& 1 === preg_match('/[0-9]+\_[0-9]+/', $this->getContent(), $matches)
|
||||
) {
|
||||
return (float) $this->getContent() == $value;
|
||||
}
|
||||
|
||||
$id = ($value instanceof self) ? $value->getId() : $value;
|
||||
|
||||
return $this->getId() == $id;
|
||||
}
|
||||
|
||||
public function __toString(): string
|
||||
{
|
||||
return '#Obj#'.$this->getId();
|
||||
}
|
||||
|
||||
/**
|
||||
* @return bool|ElementXRef
|
||||
*/
|
||||
public static function parse(string $content, ?Document $document = null, int &$offset = 0)
|
||||
{
|
||||
if (preg_match('/^\s*(?P<id>[0-9]+\s+[0-9]+\s+R)/s', $content, $match)) {
|
||||
$id = $match['id'];
|
||||
$offset += strpos($content, $id) + \strlen($id);
|
||||
$id = str_replace(' ', '_', rtrim($id, ' R'));
|
||||
|
||||
return new self($id, $document);
|
||||
}
|
||||
|
||||
return false;
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,162 @@
|
||||
<?php
|
||||
|
||||
/**
|
||||
* @file
|
||||
* This file is part of the PdfParser library.
|
||||
*
|
||||
* @author Sébastien MALOT <sebastien@malot.fr>
|
||||
*
|
||||
* @date 2017-01-03
|
||||
*
|
||||
* @license LGPLv3
|
||||
*
|
||||
* @url <https://github.com/smalot/pdfparser>
|
||||
*
|
||||
* PdfParser is a pdf library written in PHP, extraction oriented.
|
||||
* Copyright (C) 2017 - Sébastien MALOT <sebastien@malot.fr>
|
||||
*
|
||||
* This program is free software: you can redistribute it and/or modify
|
||||
* it under the terms of the GNU Lesser General Public License as published by
|
||||
* the Free Software Foundation, either version 3 of the License, or
|
||||
* (at your option) any later version.
|
||||
*
|
||||
* This program is distributed in the hope that it will be useful,
|
||||
* but WITHOUT ANY WARRANTY; without even the implied warranty of
|
||||
* MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
|
||||
* GNU Lesser General Public License for more details.
|
||||
*
|
||||
* You should have received a copy of the GNU Lesser General Public License
|
||||
* along with this program.
|
||||
* If not, see <http://www.pdfparser.org/sites/default/LICENSE.txt>.
|
||||
*/
|
||||
|
||||
namespace Smalot\PdfParser;
|
||||
|
||||
use Smalot\PdfParser\Element\ElementNumeric;
|
||||
use Smalot\PdfParser\Encoding\EncodingLocator;
|
||||
use Smalot\PdfParser\Encoding\PostScriptGlyphs;
|
||||
use Smalot\PdfParser\Exception\EncodingNotFoundException;
|
||||
|
||||
/**
|
||||
* Class Encoding
|
||||
*/
|
||||
class Encoding extends PDFObject
|
||||
{
|
||||
/**
|
||||
* @var array
|
||||
*/
|
||||
protected $encoding;
|
||||
|
||||
/**
|
||||
* @var array
|
||||
*/
|
||||
protected $differences;
|
||||
|
||||
/**
|
||||
* @var array
|
||||
*/
|
||||
protected $mapping;
|
||||
|
||||
public function init()
|
||||
{
|
||||
$this->mapping = [];
|
||||
$this->differences = [];
|
||||
$this->encoding = [];
|
||||
|
||||
if ($this->has('BaseEncoding')) {
|
||||
$this->encoding = EncodingLocator::getEncoding($this->getEncodingClass())->getTranslations();
|
||||
|
||||
// Build table including differences.
|
||||
$differences = $this->get('Differences')->getContent();
|
||||
$code = 0;
|
||||
|
||||
if (!\is_array($differences)) {
|
||||
return;
|
||||
}
|
||||
|
||||
foreach ($differences as $difference) {
|
||||
/** @var ElementNumeric $difference */
|
||||
if ($difference instanceof ElementNumeric) {
|
||||
$code = $difference->getContent();
|
||||
continue;
|
||||
}
|
||||
|
||||
// ElementName
|
||||
$this->differences[$code] = $difference;
|
||||
if (\is_object($difference)) {
|
||||
$this->differences[$code] = $difference->getContent();
|
||||
}
|
||||
|
||||
// For the next char.
|
||||
++$code;
|
||||
}
|
||||
|
||||
$this->mapping = $this->encoding;
|
||||
foreach ($this->differences as $code => $difference) {
|
||||
/* @var string $difference */
|
||||
$this->mapping[$code] = $difference;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
public function getDetails(bool $deep = true): array
|
||||
{
|
||||
$details = [];
|
||||
|
||||
$details['BaseEncoding'] = ($this->has('BaseEncoding') ? (string) $this->get('BaseEncoding') : 'Ansi');
|
||||
$details['Differences'] = ($this->has('Differences') ? (string) $this->get('Differences') : '');
|
||||
|
||||
$details += parent::getDetails($deep);
|
||||
|
||||
return $details;
|
||||
}
|
||||
|
||||
public function translateChar($dec): ?int
|
||||
{
|
||||
if (isset($this->mapping[$dec])) {
|
||||
$dec = $this->mapping[$dec];
|
||||
}
|
||||
|
||||
return PostScriptGlyphs::getCodePoint($dec);
|
||||
}
|
||||
|
||||
/**
|
||||
* Returns encoding class name if available or empty string (only prior PHP 7.4).
|
||||
*
|
||||
* @throws \Exception On PHP 7.4+ an exception is thrown if encoding class doesn't exist.
|
||||
*/
|
||||
public function __toString(): string
|
||||
{
|
||||
try {
|
||||
return $this->getEncodingClass();
|
||||
} catch (\Exception $e) {
|
||||
// prior to PHP 7.4 toString has to return an empty string.
|
||||
if (version_compare(\PHP_VERSION, '7.4.0', '<')) {
|
||||
return '';
|
||||
}
|
||||
throw $e;
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* @throws EncodingNotFoundException
|
||||
*/
|
||||
protected function getEncodingClass(): string
|
||||
{
|
||||
// Load reference table charset.
|
||||
$baseEncoding = preg_replace('/[^A-Z0-9]/is', '', $this->get('BaseEncoding')->getContent());
|
||||
|
||||
// Check for empty BaseEncoding field value
|
||||
if (!\is_string($baseEncoding) || 0 == \strlen($baseEncoding)) {
|
||||
$baseEncoding = 'StandardEncoding';
|
||||
}
|
||||
|
||||
$className = '\\Smalot\\PdfParser\\Encoding\\'.$baseEncoding;
|
||||
|
||||
if (!class_exists($className)) {
|
||||
throw new EncodingNotFoundException('Missing encoding data for: "'.$baseEncoding.'".');
|
||||
}
|
||||
|
||||
return $className;
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,8 @@
|
||||
<?php
|
||||
|
||||
namespace Smalot\PdfParser\Encoding;
|
||||
|
||||
abstract class AbstractEncoding
|
||||
{
|
||||
abstract public function getTranslations(): array;
|
||||
}
|
||||
@@ -0,0 +1,17 @@
|
||||
<?php
|
||||
|
||||
namespace Smalot\PdfParser\Encoding;
|
||||
|
||||
class EncodingLocator
|
||||
{
|
||||
protected static $encodings;
|
||||
|
||||
public static function getEncoding(string $encodingClassName): AbstractEncoding
|
||||
{
|
||||
if (!isset(self::$encodings[$encodingClassName])) {
|
||||
self::$encodings[$encodingClassName] = new $encodingClassName();
|
||||
}
|
||||
|
||||
return self::$encodings[$encodingClassName];
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,76 @@
|
||||
<?php
|
||||
|
||||
/**
|
||||
* @file
|
||||
* This file is part of the PdfParser library.
|
||||
*
|
||||
* @author Sébastien MALOT <sebastien@malot.fr>
|
||||
*
|
||||
* @date 2017-01-03
|
||||
*
|
||||
* @license LGPLv3
|
||||
*
|
||||
* @url <https://github.com/smalot/pdfparser>
|
||||
*
|
||||
* PdfParser is a pdf library written in PHP, extraction oriented.
|
||||
* Copyright (C) 2017 - Sébastien MALOT <sebastien@malot.fr>
|
||||
*
|
||||
* This program is free software: you can redistribute it and/or modify
|
||||
* it under the terms of the GNU Lesser General Public License as published by
|
||||
* the Free Software Foundation, either version 3 of the License, or
|
||||
* (at your option) any later version.
|
||||
*
|
||||
* This program is distributed in the hope that it will be useful,
|
||||
* but WITHOUT ANY WARRANTY; without even the implied warranty of
|
||||
* MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
|
||||
* GNU Lesser General Public License for more details.
|
||||
*
|
||||
* You should have received a copy of the GNU Lesser General Public License
|
||||
* along with this program.
|
||||
* If not, see <http://www.pdfparser.org/sites/default/LICENSE.txt>.
|
||||
*/
|
||||
|
||||
// Source : http://cpansearch.perl.org/src/JV/PostScript-Font-1.10.02/lib/PostScript/ISOLatin1Encoding.pm
|
||||
|
||||
namespace Smalot\PdfParser\Encoding;
|
||||
|
||||
/**
|
||||
* Class ISOLatin1Encoding
|
||||
*/
|
||||
class ISOLatin1Encoding extends AbstractEncoding
|
||||
{
|
||||
public function getTranslations(): array
|
||||
{
|
||||
$encoding =
|
||||
'.notdef .notdef .notdef .notdef .notdef .notdef .notdef .notdef '.
|
||||
'.notdef .notdef .notdef .notdef .notdef .notdef .notdef .notdef '.
|
||||
'.notdef .notdef .notdef .notdef .notdef .notdef .notdef .notdef '.
|
||||
'.notdef .notdef .notdef .notdef .notdef .notdef .notdef .notdef '.
|
||||
'space exclam quotedbl numbersign dollar percent ampersand quoteright '.
|
||||
'parenleft parenright asterisk plus comma minus period slash zero one '.
|
||||
'two three four five six seven eight nine colon semicolon less equal '.
|
||||
'greater question at A B C D E F G H I J K L M N O P Q R S T U V W X '.
|
||||
'Y Z bracketleft backslash bracketright asciicircum underscore '.
|
||||
'quoteleft a b c d e f g h i j k l m n o p q r s t u v w x y z '.
|
||||
'braceleft bar braceright asciitilde .notdef .notdef .notdef .notdef '.
|
||||
'.notdef .notdef .notdef .notdef .notdef .notdef .notdef .notdef '.
|
||||
'.notdef .notdef .notdef .notdef .notdef dotlessi grave acute '.
|
||||
'circumflex tilde macron breve dotaccent dieresis .notdef ring '.
|
||||
'cedilla .notdef hungarumlaut ogonek caron space exclamdown cent '.
|
||||
'sterling currency yen brokenbar section dieresis copyright '.
|
||||
'ordfeminine guillemotleft logicalnot hyphen registered macron degree '.
|
||||
'plusminus twosuperior threesuperior acute mu paragraph '.
|
||||
'periodcentered cedilla onesuperior ordmasculine guillemotright '.
|
||||
'onequarter onehalf threequarters questiondown Agrave Aacute '.
|
||||
'Acircumflex Atilde Adieresis Aring AE Ccedilla Egrave Eacute '.
|
||||
'Ecircumflex Edieresis Igrave Iacute Icircumflex Idieresis Eth Ntilde '.
|
||||
'Ograve Oacute Ocircumflex Otilde Odieresis multiply Oslash Ugrave '.
|
||||
'Uacute Ucircumflex Udieresis Yacute Thorn germandbls agrave aacute '.
|
||||
'acircumflex atilde adieresis aring ae ccedilla egrave eacute '.
|
||||
'ecircumflex edieresis igrave iacute icircumflex idieresis eth ntilde '.
|
||||
'ograve oacute ocircumflex otilde odieresis divide oslash ugrave '.
|
||||
'uacute ucircumflex udieresis yacute thorn ydieresis';
|
||||
|
||||
return explode(' ', $encoding);
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,76 @@
|
||||
<?php
|
||||
|
||||
/**
|
||||
* @file
|
||||
* This file is part of the PdfParser library.
|
||||
*
|
||||
* @author Sébastien MALOT <sebastien@malot.fr>
|
||||
*
|
||||
* @date 2017-01-03
|
||||
*
|
||||
* @license LGPLv3
|
||||
*
|
||||
* @url <https://github.com/smalot/pdfparser>
|
||||
*
|
||||
* PdfParser is a pdf library written in PHP, extraction oriented.
|
||||
* Copyright (C) 2017 - Sébastien MALOT <sebastien@malot.fr>
|
||||
*
|
||||
* This program is free software: you can redistribute it and/or modify
|
||||
* it under the terms of the GNU Lesser General Public License as published by
|
||||
* the Free Software Foundation, either version 3 of the License, or
|
||||
* (at your option) any later version.
|
||||
*
|
||||
* This program is distributed in the hope that it will be useful,
|
||||
* but WITHOUT ANY WARRANTY; without even the implied warranty of
|
||||
* MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
|
||||
* GNU Lesser General Public License for more details.
|
||||
*
|
||||
* You should have received a copy of the GNU Lesser General Public License
|
||||
* along with this program.
|
||||
* If not, see <http://www.pdfparser.org/sites/default/LICENSE.txt>.
|
||||
*/
|
||||
|
||||
// Source : http://cpansearch.perl.org/src/JV/PostScript-Font-1.10.02/lib/PostScript/ISOLatin9Encoding.pm
|
||||
|
||||
namespace Smalot\PdfParser\Encoding;
|
||||
|
||||
/**
|
||||
* Class ISOLatin9Encoding
|
||||
*/
|
||||
class ISOLatin9Encoding extends AbstractEncoding
|
||||
{
|
||||
public function getTranslations(): array
|
||||
{
|
||||
$encoding =
|
||||
'.notdef .notdef .notdef .notdef .notdef .notdef .notdef .notdef '.
|
||||
'.notdef .notdef .notdef .notdef .notdef .notdef .notdef .notdef '.
|
||||
'.notdef .notdef .notdef .notdef .notdef .notdef .notdef .notdef '.
|
||||
'.notdef .notdef .notdef .notdef .notdef .notdef .notdef .notdef '.
|
||||
'space exclam quotedbl numbersign dollar percent ampersand quoteright '.
|
||||
'parenleft parenright asterisk plus comma minus period slash zero one '.
|
||||
'two three four five six seven eight nine colon semicolon less equal '.
|
||||
'greater question at A B C D E F G H I J K L M N O P Q R S T U V W X '.
|
||||
'Y Z bracketleft backslash bracketright asciicircum underscore '.
|
||||
'quoteleft a b c d e f g h i j k l m n o p q r s t u v w x y z '.
|
||||
'braceleft bar braceright asciitilde .notdef .notdef .notdef .notdef '.
|
||||
'.notdef .notdef .notdef .notdef .notdef .notdef .notdef .notdef '.
|
||||
'.notdef .notdef .notdef .notdef .notdef dotlessi grave acute '.
|
||||
'circumflex tilde macron breve dotaccent dieresis .notdef ring '.
|
||||
'cedilla .notdef hungarumlaut ogonek caron space exclamdown cent '.
|
||||
'sterling Euro yen Scaron section scaron copyright '.
|
||||
'ordfeminine guillemotleft logicalnot hyphen registered macron degree '.
|
||||
'plusminus twosuperior threesuperior Zcaron mu paragraph '.
|
||||
'periodcentered zcaron onesuperior ordmasculine guillemotright '.
|
||||
'OE oe Ydieresis questiondown Agrave Aacute '.
|
||||
'Acircumflex Atilde Adieresis Aring AE Ccedilla Egrave Eacute '.
|
||||
'Ecircumflex Edieresis Igrave Iacute Icircumflex Idieresis Eth Ntilde '.
|
||||
'Ograve Oacute Ocircumflex Otilde Odieresis multiply Oslash Ugrave '.
|
||||
'Uacute Ucircumflex Udieresis Yacute Thorn germandbls agrave aacute '.
|
||||
'acircumflex atilde adieresis aring ae ccedilla egrave eacute '.
|
||||
'ecircumflex edieresis igrave iacute icircumflex idieresis eth ntilde '.
|
||||
'ograve oacute ocircumflex otilde odieresis divide oslash ugrave '.
|
||||
'uacute ucircumflex udieresis yacute thorn ydieresis';
|
||||
|
||||
return explode(' ', $encoding);
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,80 @@
|
||||
<?php
|
||||
|
||||
/**
|
||||
* @file
|
||||
* This file is part of the PdfParser library.
|
||||
*
|
||||
* @author Sébastien MALOT <sebastien@malot.fr>
|
||||
*
|
||||
* @date 2017-01-03
|
||||
*
|
||||
* @license LGPLv3
|
||||
*
|
||||
* @url <https://github.com/smalot/pdfparser>
|
||||
*
|
||||
* PdfParser is a pdf library written in PHP, extraction oriented.
|
||||
* Copyright (C) 2017 - Sébastien MALOT <sebastien@malot.fr>
|
||||
*
|
||||
* This program is free software: you can redistribute it and/or modify
|
||||
* it under the terms of the GNU Lesser General Public License as published by
|
||||
* the Free Software Foundation, either version 3 of the License, or
|
||||
* (at your option) any later version.
|
||||
*
|
||||
* This program is distributed in the hope that it will be useful,
|
||||
* but WITHOUT ANY WARRANTY; without even the implied warranty of
|
||||
* MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
|
||||
* GNU Lesser General Public License for more details.
|
||||
*
|
||||
* You should have received a copy of the GNU Lesser General Public License
|
||||
* along with this program.
|
||||
* If not, see <http://www.pdfparser.org/sites/default/LICENSE.txt>.
|
||||
*/
|
||||
|
||||
// Source : http://www.opensource.apple.com/source/vim/vim-34/vim/runtime/print/mac-roman.ps
|
||||
|
||||
namespace Smalot\PdfParser\Encoding;
|
||||
|
||||
/**
|
||||
* Class MacRomanEncoding
|
||||
*/
|
||||
class MacRomanEncoding extends AbstractEncoding
|
||||
{
|
||||
public function getTranslations(): array
|
||||
{
|
||||
$encoding =
|
||||
'.notdef .notdef .notdef .notdef .notdef .notdef .notdef .notdef '.
|
||||
'.notdef .notdef .notdef .notdef .notdef .notdef .notdef .notdef '.
|
||||
'.notdef .notdef .notdef .notdef .notdef .notdef .notdef .notdef '.
|
||||
'.notdef .notdef .notdef .notdef .notdef .notdef .notdef .notdef '.
|
||||
'space exclam quotedbl numbersign dollar percent ampersand quotesingle '.
|
||||
'parenleft parenright asterisk plus comma minus period slash '.
|
||||
'zero one two three four five six seven '.
|
||||
'eight nine colon semicolon less equal greater question '.
|
||||
'at A B C D E F G '.
|
||||
'H I J K L M N O '.
|
||||
'P Q R S T U V W '.
|
||||
'X Y Z bracketleft backslash bracketright asciicircum underscore '.
|
||||
'grave a b c d e f g '.
|
||||
'h i j k l m n o '.
|
||||
'p q r s t u v w '.
|
||||
'x y z braceleft bar braceright asciitilde .notdef '.
|
||||
'Adieresis Aring Ccedilla Eacute Ntilde Odieresis Udieresis aacute '.
|
||||
'agrave acircumflex adieresis atilde aring ccedilla eacute egrave '.
|
||||
'ecircumflex edieresis iacute igrave icircumflex idieresis ntilde oacute '.
|
||||
'ograve ocircumflex odieresis otilde uacute ugrave ucircumflex udieresis '.
|
||||
'dagger degree cent sterling section bullet paragraph germandbls '.
|
||||
'registered copyright trademark acute dieresis notequal AE Oslash '.
|
||||
'infinity plusminus lessequal greaterequal yen mu partialdiff summation '.
|
||||
'Pi pi integral ordfeminine ordmasculine Omega ae oslash '.
|
||||
'questiondown exclamdown logicalnot radical florin approxequal delta guillemotleft '.
|
||||
'guillemotright ellipsis space Agrave Atilde Otilde OE oe '.
|
||||
'endash emdash quotedblleft quotedblright quoteleft quoteright divide lozenge '.
|
||||
'ydieresis Ydieresis fraction currency guilsinglleft guilsinglright fi fl '.
|
||||
'daggerdbl periodcentered quotesinglbase quotedblbase perthousand Acircumflex Ecircumflex Aacute '.
|
||||
'Edieresis Egrave Iacute Icircumflex Idieresis Igrave Oacute Ocircumflex '.
|
||||
'heart Ograve Uacute Ucircumflex Ugrave dotlessi circumflex tilde '.
|
||||
'macron breve dotaccent ring cedilla hungarumlaut ogonek caron';
|
||||
|
||||
return explode(' ', $encoding);
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,189 @@
|
||||
<?php
|
||||
|
||||
/**
|
||||
* @file This file is part of the PdfParser library.
|
||||
*
|
||||
* @author Brian Huisman <bhuisman@greywyvern.com>
|
||||
*
|
||||
* @date 2023-06-28
|
||||
*
|
||||
* @license LGPLv3
|
||||
*
|
||||
* @url <https://github.com/smalot/pdfparser>
|
||||
*
|
||||
* PdfParser is a pdf library written in PHP, extraction oriented.
|
||||
* Copyright (C) 2017 - Sébastien MALOT <sebastien@malot.fr>
|
||||
*
|
||||
* This program is free software: you can redistribute it and/or modify
|
||||
* it under the terms of the GNU Lesser General Public License as published by
|
||||
* the Free Software Foundation, either version 3 of the License, or
|
||||
* (at your option) any later version.
|
||||
*
|
||||
* This program is distributed in the hope that it will be useful,
|
||||
* but WITHOUT ANY WARRANTY; without even the implied warranty of
|
||||
* MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
|
||||
* GNU Lesser General Public License for more details.
|
||||
*
|
||||
* You should have received a copy of the GNU Lesser General Public License
|
||||
* along with this program.
|
||||
* If not, see <http://www.pdfparser.org/sites/default/LICENSE.txt>.
|
||||
*/
|
||||
|
||||
// Source : https://opensource.adobe.com/dc-acrobat-sdk-docs/pdfstandards/pdfreference1.2.pdf
|
||||
// Source : https://ia801001.us.archive.org/1/items/pdf1.7/pdf_reference_1-7.pdf
|
||||
|
||||
namespace Smalot\PdfParser\Encoding;
|
||||
|
||||
/**
|
||||
* Class PDFDocEncoding
|
||||
*/
|
||||
class PDFDocEncoding
|
||||
{
|
||||
public static function getCodePage(): array
|
||||
{
|
||||
return [
|
||||
"\x18" => "\u{02d8}", // breve
|
||||
"\x19" => "\u{02c7}", // caron
|
||||
"\x1a" => "\u{02c6}", // circumflex
|
||||
"\x1b" => "\u{02d9}", // dotaccent
|
||||
"\x1c" => "\u{02dd}", // hungarumlaut
|
||||
"\x1d" => "\u{02db}", // ogonek
|
||||
"\x1e" => "\u{02de}", // ring
|
||||
"\x1f" => "\u{02dc}", // tilde
|
||||
"\x7f" => '',
|
||||
"\x80" => "\u{2022}", // bullet
|
||||
"\x81" => "\u{2020}", // dagger
|
||||
"\x82" => "\u{2021}", // daggerdbl
|
||||
"\x83" => "\u{2026}", // ellipsis
|
||||
"\x84" => "\u{2014}", // emdash
|
||||
"\x85" => "\u{2013}", // endash
|
||||
"\x86" => "\u{0192}", // florin
|
||||
"\x87" => "\u{2044}", // fraction
|
||||
"\x88" => "\u{2039}", // guilsinglleft
|
||||
"\x89" => "\u{203a}", // guilsinglright
|
||||
"\x8a" => "\u{2212}", // minus
|
||||
"\x8b" => "\u{2030}", // perthousand
|
||||
"\x8c" => "\u{201e}", // quotedblbase
|
||||
"\x8d" => "\u{201c}", // quotedblleft
|
||||
"\x8e" => "\u{201d}", // quotedblright
|
||||
"\x8f" => "\u{2018}", // quoteleft
|
||||
"\x90" => "\u{2019}", // quoteright
|
||||
"\x91" => "\u{201a}", // quotesinglbase
|
||||
"\x92" => "\u{2122}", // trademark
|
||||
"\x93" => "\u{fb01}", // fi
|
||||
"\x94" => "\u{fb02}", // fl
|
||||
"\x95" => "\u{0141}", // Lslash
|
||||
"\x96" => "\u{0152}", // OE
|
||||
"\x97" => "\u{0160}", // Scaron
|
||||
"\x98" => "\u{0178}", // Ydieresis
|
||||
"\x99" => "\u{017d}", // Zcaron
|
||||
"\x9a" => "\u{0131}", // dotlessi
|
||||
"\x9b" => "\u{0142}", // lslash
|
||||
"\x9c" => "\u{0153}", // oe
|
||||
"\x9d" => "\u{0161}", // scaron
|
||||
"\x9e" => "\u{017e}", // zcaron
|
||||
"\x9f" => '',
|
||||
"\xa0" => "\u{20ac}", // Euro
|
||||
"\xa1" => "\u{00a1}", // exclamdown
|
||||
"\xa2" => "\u{00a2}", // cent
|
||||
"\xa3" => "\u{00a3}", // sterling
|
||||
"\xa4" => "\u{00a4}", // currency
|
||||
"\xa5" => "\u{00a5}", // yen
|
||||
"\xa6" => "\u{00a6}", // brokenbar
|
||||
"\xa7" => "\u{00a7}", // section
|
||||
"\xa8" => "\u{00a8}", // dieresis
|
||||
"\xa9" => "\u{00a9}", // copyright
|
||||
"\xaa" => "\u{00aa}", // ordfeminine
|
||||
"\xab" => "\u{00ab}", // guillemotleft
|
||||
"\xac" => "\u{00ac}", // logicalnot
|
||||
"\xad" => '',
|
||||
"\xae" => "\u{00ae}", // registered
|
||||
"\xaf" => "\u{00af}", // macron
|
||||
"\xb0" => "\u{00b0}", // degree
|
||||
"\xb1" => "\u{00b1}", // plusminus
|
||||
"\xb2" => "\u{00b2}", // twosuperior
|
||||
"\xb3" => "\u{00b3}", // threesuperior
|
||||
"\xb4" => "\u{00b4}", // acute
|
||||
"\xb5" => "\u{00b5}", // mu
|
||||
"\xb6" => "\u{00b6}", // paragraph
|
||||
"\xb7" => "\u{00b7}", // periodcentered
|
||||
"\xb8" => "\u{00b8}", // cedilla
|
||||
"\xb9" => "\u{00b9}", // onesuperior
|
||||
"\xba" => "\u{00ba}", // ordmasculine
|
||||
"\xbb" => "\u{00bb}", // guillemotright
|
||||
"\xbc" => "\u{00bc}", // onequarter
|
||||
"\xbd" => "\u{00bd}", // onehalf
|
||||
"\xbe" => "\u{00be}", // threequarters
|
||||
"\xbf" => "\u{00bf}", // questiondown
|
||||
"\xc0" => "\u{00c0}", // Agrave
|
||||
"\xc1" => "\u{00c1}", // Aacute
|
||||
"\xc2" => "\u{00c2}", // Acircumflex
|
||||
"\xc3" => "\u{00c3}", // Atilde
|
||||
"\xc4" => "\u{00c4}", // Adieresis
|
||||
"\xc5" => "\u{00c5}", // Aring
|
||||
"\xc6" => "\u{00c6}", // AE
|
||||
"\xc7" => "\u{00c7}", // Ccedill
|
||||
"\xc8" => "\u{00c8}", // Egrave
|
||||
"\xc9" => "\u{00c9}", // Eacute
|
||||
"\xca" => "\u{00ca}", // Ecircumflex
|
||||
"\xcb" => "\u{00cb}", // Edieresis
|
||||
"\xcc" => "\u{00cc}", // Igrave
|
||||
"\xcd" => "\u{00cd}", // Iacute
|
||||
"\xce" => "\u{00ce}", // Icircumflex
|
||||
"\xcf" => "\u{00cf}", // Idieresis
|
||||
"\xd0" => "\u{00d0}", // Eth
|
||||
"\xd1" => "\u{00d1}", // Ntilde
|
||||
"\xd2" => "\u{00d2}", // Ograve
|
||||
"\xd3" => "\u{00d3}", // Oacute
|
||||
"\xd4" => "\u{00d4}", // Ocircumflex
|
||||
"\xd5" => "\u{00d5}", // Otilde
|
||||
"\xd6" => "\u{00d6}", // Odieresis
|
||||
"\xd7" => "\u{00d7}", // multiply
|
||||
"\xd8" => "\u{00d8}", // Oslash
|
||||
"\xd9" => "\u{00d9}", // Ugrave
|
||||
"\xda" => "\u{00da}", // Uacute
|
||||
"\xdb" => "\u{00db}", // Ucircumflex
|
||||
"\xdc" => "\u{00dc}", // Udieresis
|
||||
"\xdd" => "\u{00dd}", // Yacute
|
||||
"\xde" => "\u{00de}", // Thorn
|
||||
"\xdf" => "\u{00df}", // germandbls
|
||||
"\xe0" => "\u{00e0}", // agrave
|
||||
"\xe1" => "\u{00e1}", // aacute
|
||||
"\xe2" => "\u{00e2}", // acircumflex
|
||||
"\xe3" => "\u{00e3}", // atilde
|
||||
"\xe4" => "\u{00e4}", // adieresis
|
||||
"\xe5" => "\u{00e5}", // aring
|
||||
"\xe6" => "\u{00e6}", // ae
|
||||
"\xe7" => "\u{00e7}", // ccedilla
|
||||
"\xe8" => "\u{00e8}", // egrave
|
||||
"\xe9" => "\u{00e9}", // eacute
|
||||
"\xea" => "\u{00ea}", // ecircumflex
|
||||
"\xeb" => "\u{00eb}", // edieresis
|
||||
"\xec" => "\u{00ec}", // igrave
|
||||
"\xed" => "\u{00ed}", // iacute
|
||||
"\xee" => "\u{00ee}", // icircumflex
|
||||
"\xef" => "\u{00ef}", // idieresis
|
||||
"\xf0" => "\u{00f0}", // eth
|
||||
"\xf1" => "\u{00f1}", // ntilde
|
||||
"\xf2" => "\u{00f2}", // ograve
|
||||
"\xf3" => "\u{00f3}", // oacute
|
||||
"\xf4" => "\u{00f4}", // ocircumflex
|
||||
"\xf5" => "\u{00f5}", // otilde
|
||||
"\xf6" => "\u{00f6}", // odieresis
|
||||
"\xf7" => "\u{00f7}", // divide
|
||||
"\xf8" => "\u{00f8}", // oslash
|
||||
"\xf9" => "\u{00f9}", // ugrave
|
||||
"\xfa" => "\u{00fa}", // uacute
|
||||
"\xfb" => "\u{00fb}", // ucircumflex
|
||||
"\xfc" => "\u{00fc}", // udieresis
|
||||
"\xfd" => "\u{00fd}", // yacute
|
||||
"\xfe" => "\u{00fe}", // thorn
|
||||
"\xff" => "\u{00ff}", // ydieresis
|
||||
];
|
||||
}
|
||||
|
||||
public static function convertPDFDoc2UTF8(string $content): string
|
||||
{
|
||||
return strtr($content, static::getCodePage());
|
||||
}
|
||||
}
|
||||
File diff suppressed because it is too large
Load Diff
@@ -0,0 +1,76 @@
|
||||
<?php
|
||||
|
||||
/**
|
||||
* @file
|
||||
* This file is part of the PdfParser library.
|
||||
*
|
||||
* @author Sébastien MALOT <sebastien@malot.fr>
|
||||
*
|
||||
* @date 2017-01-03
|
||||
*
|
||||
* @license LGPLv3
|
||||
*
|
||||
* @url <https://github.com/smalot/pdfparser>
|
||||
*
|
||||
* PdfParser is a pdf library written in PHP, extraction oriented.
|
||||
* Copyright (C) 2017 - Sébastien MALOT <sebastien@malot.fr>
|
||||
*
|
||||
* This program is free software: you can redistribute it and/or modify
|
||||
* it under the terms of the GNU Lesser General Public License as published by
|
||||
* the Free Software Foundation, either version 3 of the License, or
|
||||
* (at your option) any later version.
|
||||
*
|
||||
* This program is distributed in the hope that it will be useful,
|
||||
* but WITHOUT ANY WARRANTY; without even the implied warranty of
|
||||
* MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
|
||||
* GNU Lesser General Public License for more details.
|
||||
*
|
||||
* You should have received a copy of the GNU Lesser General Public License
|
||||
* along with this program.
|
||||
* If not, see <http://www.pdfparser.org/sites/default/LICENSE.txt>.
|
||||
*/
|
||||
|
||||
// Source : http://cpansearch.perl.org/src/JV/PostScript-Font-1.10.02/lib/PostScript/StandardEncoding.pm
|
||||
|
||||
namespace Smalot\PdfParser\Encoding;
|
||||
|
||||
/**
|
||||
* Class StandardEncoding
|
||||
*/
|
||||
class StandardEncoding extends AbstractEncoding
|
||||
{
|
||||
public function getTranslations(): array
|
||||
{
|
||||
$encoding =
|
||||
'.notdef .notdef .notdef .notdef .notdef .notdef .notdef .notdef '.
|
||||
'.notdef .notdef .notdef .notdef .notdef .notdef .notdef .notdef '.
|
||||
'.notdef .notdef .notdef .notdef .notdef .notdef .notdef .notdef '.
|
||||
'.notdef .notdef .notdef .notdef .notdef .notdef .notdef .notdef '.
|
||||
'space exclam quotedbl numbersign dollar percent ampersand quoteright '.
|
||||
'parenleft parenright asterisk plus comma hyphen period slash zero '.
|
||||
'one two three four five six seven eight nine colon semicolon less '.
|
||||
'equal greater question at A B C D E F G H I J K L M N O P Q R S T U '.
|
||||
'V W X Y Z bracketleft backslash bracketright asciicircum underscore '.
|
||||
'quoteleft a b c d e f g h i j k l m n o p q r s t u v w x y z '.
|
||||
'braceleft bar braceright asciitilde .notdef .notdef .notdef .notdef '.
|
||||
'.notdef .notdef .notdef .notdef .notdef .notdef .notdef .notdef '.
|
||||
'.notdef .notdef .notdef .notdef .notdef .notdef .notdef .notdef '.
|
||||
'.notdef .notdef .notdef .notdef .notdef .notdef .notdef .notdef '.
|
||||
'.notdef .notdef .notdef .notdef .notdef .notdef exclamdown cent '.
|
||||
'sterling fraction yen florin section currency quotesingle '.
|
||||
'quotedblleft guillemotleft guilsinglleft guilsinglright fi fl '.
|
||||
'.notdef endash dagger daggerdbl periodcentered .notdef paragraph '.
|
||||
'bullet quotesinglbase quotedblbase quotedblright guillemotright '.
|
||||
'ellipsis perthousand .notdef questiondown .notdef grave acute '.
|
||||
'circumflex tilde macron breve dotaccent dieresis .notdef ring '.
|
||||
'cedilla .notdef hungarumlaut ogonek caron emdash .notdef .notdef '.
|
||||
'.notdef .notdef .notdef .notdef .notdef .notdef .notdef .notdef '.
|
||||
'.notdef .notdef .notdef .notdef .notdef .notdef AE .notdef '.
|
||||
'ordfeminine .notdef .notdef .notdef .notdef Lslash Oslash OE '.
|
||||
'ordmasculine .notdef .notdef .notdef .notdef .notdef ae .notdef '.
|
||||
'.notdef .notdef dotlessi .notdef .notdef lslash oslash oe germandbls '.
|
||||
'.notdef .notdef .notdef .notdef';
|
||||
|
||||
return explode(' ', $encoding);
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,76 @@
|
||||
<?php
|
||||
|
||||
/**
|
||||
* @file
|
||||
* This file is part of the PdfParser library.
|
||||
*
|
||||
* @author Sébastien MALOT <sebastien@malot.fr>
|
||||
*
|
||||
* @date 2017-01-03
|
||||
*
|
||||
* @license LGPLv3
|
||||
*
|
||||
* @url <https://github.com/smalot/pdfparser>
|
||||
*
|
||||
* PdfParser is a pdf library written in PHP, extraction oriented.
|
||||
* Copyright (C) 2017 - Sébastien MALOT <sebastien@malot.fr>
|
||||
*
|
||||
* This program is free software: you can redistribute it and/or modify
|
||||
* it under the terms of the GNU Lesser General Public License as published by
|
||||
* the Free Software Foundation, either version 3 of the License, or
|
||||
* (at your option) any later version.
|
||||
*
|
||||
* This program is distributed in the hope that it will be useful,
|
||||
* but WITHOUT ANY WARRANTY; without even the implied warranty of
|
||||
* MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
|
||||
* GNU Lesser General Public License for more details.
|
||||
*
|
||||
* You should have received a copy of the GNU Lesser General Public License
|
||||
* along with this program.
|
||||
* If not, see <http://www.pdfparser.org/sites/default/LICENSE.txt>.
|
||||
*/
|
||||
|
||||
// Source : http://cpansearch.perl.org/src/JV/PostScript-Font-1.10.02/lib/PostScript/WinANSIEncoding.pm
|
||||
|
||||
namespace Smalot\PdfParser\Encoding;
|
||||
|
||||
/**
|
||||
* Class WinAnsiEncoding
|
||||
*/
|
||||
class WinAnsiEncoding extends AbstractEncoding
|
||||
{
|
||||
public function getTranslations(): array
|
||||
{
|
||||
$encoding =
|
||||
'.notdef .notdef .notdef .notdef .notdef .notdef .notdef .notdef '.
|
||||
'.notdef .notdef .notdef .notdef .notdef .notdef .notdef .notdef '.
|
||||
'.notdef .notdef .notdef .notdef .notdef .notdef .notdef .notdef '.
|
||||
'.notdef .notdef .notdef .notdef .notdef .notdef .notdef .notdef '.
|
||||
'space exclam quotedbl numbersign dollar percent ampersand quotesingle '.
|
||||
'parenleft parenright asterisk plus comma hyphen period slash zero one '.
|
||||
'two three four five six seven eight nine colon semicolon less equal '.
|
||||
'greater question at A B C D E F G H I J K L M N O P Q R S T U V W X '.
|
||||
'Y Z bracketleft backslash bracketright asciicircum underscore '.
|
||||
'grave a b c d e f g h i j k l m n o p q r s t u v w x y z '.
|
||||
'braceleft bar braceright asciitilde bullet Euro bullet quotesinglbase '.
|
||||
'florin quotedblbase ellipsis dagger daggerdbl circumflex perthousand '.
|
||||
'Scaron guilsinglleft OE bullet Zcaron bullet bullet quoteleft quoteright '.
|
||||
'quotedblleft quotedblright bullet endash emdash tilde trademark scaron '.
|
||||
'guilsinglright oe bullet zcaron Ydieresis space exclamdown cent '.
|
||||
'sterling currency yen brokenbar section dieresis copyright '.
|
||||
'ordfeminine guillemotleft logicalnot hyphen registered macron degree '.
|
||||
'plusminus twosuperior threesuperior acute mu paragraph '.
|
||||
'periodcentered cedilla onesuperior ordmasculine guillemotright '.
|
||||
'onequarter onehalf threequarters questiondown Agrave Aacute '.
|
||||
'Acircumflex Atilde Adieresis Aring AE Ccedilla Egrave Eacute '.
|
||||
'Ecircumflex Edieresis Igrave Iacute Icircumflex Idieresis Eth Ntilde '.
|
||||
'Ograve Oacute Ocircumflex Otilde Odieresis multiply Oslash Ugrave '.
|
||||
'Uacute Ucircumflex Udieresis Yacute Thorn germandbls agrave aacute '.
|
||||
'acircumflex atilde adieresis aring ae ccedilla egrave eacute '.
|
||||
'ecircumflex edieresis igrave iacute icircumflex idieresis eth ntilde '.
|
||||
'ograve oacute ocircumflex otilde odieresis divide oslash ugrave '.
|
||||
'uacute ucircumflex udieresis yacute thorn ydieresis';
|
||||
|
||||
return explode(' ', $encoding);
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,12 @@
|
||||
<?php
|
||||
|
||||
declare(strict_types=1);
|
||||
|
||||
namespace Smalot\PdfParser\Exception;
|
||||
|
||||
/**
|
||||
* This Exception is thrown when no PDF data was given.
|
||||
*/
|
||||
class EmptyPdfException extends \Exception
|
||||
{
|
||||
}
|
||||
+7
@@ -0,0 +1,7 @@
|
||||
<?php
|
||||
|
||||
namespace Smalot\PdfParser\Exception;
|
||||
|
||||
class EncodingNotFoundException extends \Exception
|
||||
{
|
||||
}
|
||||
Vendored
+12
@@ -0,0 +1,12 @@
|
||||
<?php
|
||||
|
||||
declare(strict_types=1);
|
||||
|
||||
namespace Smalot\PdfParser\Exception;
|
||||
|
||||
/**
|
||||
* This exception is thrown when an invalid dictionary object is encountered.
|
||||
*/
|
||||
class InvalidDictionaryObjectException extends \Exception
|
||||
{
|
||||
}
|
||||
+12
@@ -0,0 +1,12 @@
|
||||
<?php
|
||||
|
||||
declare(strict_types=1);
|
||||
|
||||
namespace Smalot\PdfParser\Exception;
|
||||
|
||||
/**
|
||||
* This exception is thrown when the catalog is missing.
|
||||
*/
|
||||
class MissingCatalogException extends \Exception
|
||||
{
|
||||
}
|
||||
+12
@@ -0,0 +1,12 @@
|
||||
<?php
|
||||
|
||||
declare(strict_types=1);
|
||||
|
||||
namespace Smalot\PdfParser\Exception;
|
||||
|
||||
/**
|
||||
* This Exception is thrown when the %PDF- header is missing.
|
||||
*/
|
||||
class MissingPdfHeaderException extends \Exception
|
||||
{
|
||||
}
|
||||
+12
@@ -0,0 +1,12 @@
|
||||
<?php
|
||||
|
||||
declare(strict_types=1);
|
||||
|
||||
namespace Smalot\PdfParser\Exception;
|
||||
|
||||
/**
|
||||
* This Exception is thrown when a functionality has not yet been implemented.
|
||||
*/
|
||||
class NotImplementedException extends \Exception
|
||||
{
|
||||
}
|
||||
@@ -0,0 +1,709 @@
|
||||
<?php
|
||||
|
||||
/**
|
||||
* @file
|
||||
* This file is part of the PdfParser library.
|
||||
*
|
||||
* @author Sébastien MALOT <sebastien@malot.fr>
|
||||
*
|
||||
* @date 2017-01-03
|
||||
*
|
||||
* @license LGPLv3
|
||||
*
|
||||
* @url <https://github.com/smalot/pdfparser>
|
||||
*
|
||||
* PdfParser is a pdf library written in PHP, extraction oriented.
|
||||
* Copyright (C) 2017 - Sébastien MALOT <sebastien@malot.fr>
|
||||
*
|
||||
* This program is free software: you can redistribute it and/or modify
|
||||
* it under the terms of the GNU Lesser General Public License as published by
|
||||
* the Free Software Foundation, either version 3 of the License, or
|
||||
* (at your option) any later version.
|
||||
*
|
||||
* This program is distributed in the hope that it will be useful,
|
||||
* but WITHOUT ANY WARRANTY; without even the implied warranty of
|
||||
* MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
|
||||
* GNU Lesser General Public License for more details.
|
||||
*
|
||||
* You should have received a copy of the GNU Lesser General Public License
|
||||
* along with this program.
|
||||
* If not, see <http://www.pdfparser.org/sites/default/LICENSE.txt>.
|
||||
*/
|
||||
|
||||
namespace Smalot\PdfParser;
|
||||
|
||||
use Smalot\PdfParser\Encoding\WinAnsiEncoding;
|
||||
use Smalot\PdfParser\Exception\EncodingNotFoundException;
|
||||
|
||||
/**
|
||||
* Class Font
|
||||
*/
|
||||
class Font extends PDFObject
|
||||
{
|
||||
public const MISSING = '?';
|
||||
|
||||
/**
|
||||
* @var array
|
||||
*/
|
||||
protected $table;
|
||||
|
||||
/**
|
||||
* @var array
|
||||
*/
|
||||
protected $tableSizes;
|
||||
|
||||
/**
|
||||
* Caches results from uchr.
|
||||
*
|
||||
* @var array
|
||||
*/
|
||||
private static $uchrCache = [];
|
||||
|
||||
/**
|
||||
* In some PDF-files encoding could be referenced by object id but object itself does not contain
|
||||
* `/Type /Encoding` in its dictionary. These objects wouldn't be initialized as Encoding in
|
||||
* \Smalot\PdfParser\PDFObject::factory() during file parsing (they would be just PDFObject).
|
||||
*
|
||||
* Therefore, we create an instance of Encoding from them during decoding and cache this value in this property.
|
||||
*
|
||||
* @var Encoding
|
||||
*
|
||||
* @see https://github.com/smalot/pdfparser/pull/500
|
||||
*/
|
||||
private $initializedEncodingByPdfObject;
|
||||
|
||||
public function init()
|
||||
{
|
||||
// Load translate table.
|
||||
$this->loadTranslateTable();
|
||||
}
|
||||
|
||||
public function getName(): string
|
||||
{
|
||||
return $this->has('BaseFont') ? (string) $this->get('BaseFont') : '[Unknown]';
|
||||
}
|
||||
|
||||
public function getType(): string
|
||||
{
|
||||
return (string) $this->header->get('Subtype');
|
||||
}
|
||||
|
||||
public function getDetails(bool $deep = true): array
|
||||
{
|
||||
$details = [];
|
||||
|
||||
$details['Name'] = $this->getName();
|
||||
$details['Type'] = $this->getType();
|
||||
$details['Encoding'] = ($this->has('Encoding') ? (string) $this->get('Encoding') : 'Ansi');
|
||||
|
||||
$details += parent::getDetails($deep);
|
||||
|
||||
return $details;
|
||||
}
|
||||
|
||||
/**
|
||||
* @return string|bool
|
||||
*/
|
||||
public function translateChar(string $char, bool $use_default = true)
|
||||
{
|
||||
$dec = hexdec(bin2hex($char));
|
||||
|
||||
if (\array_key_exists($dec, $this->table)) {
|
||||
return $this->table[$dec];
|
||||
}
|
||||
|
||||
// fallback for decoding single-byte ANSI characters that are not in the lookup table
|
||||
$fallbackDecoded = $char;
|
||||
if (
|
||||
\strlen($char) < 2
|
||||
&& $this->has('Encoding')
|
||||
&& $this->get('Encoding') instanceof Encoding
|
||||
) {
|
||||
try {
|
||||
if (WinAnsiEncoding::class === $this->get('Encoding')->__toString()) {
|
||||
$fallbackDecoded = self::uchr($dec);
|
||||
}
|
||||
} catch (EncodingNotFoundException $e) {
|
||||
// Encoding->getEncodingClass() throws EncodingNotFoundException when BaseEncoding doesn't exists
|
||||
// See table 5.11 on PDF 1.5 specs for more info
|
||||
}
|
||||
}
|
||||
|
||||
return $use_default ? self::MISSING : $fallbackDecoded;
|
||||
}
|
||||
|
||||
/**
|
||||
* Convert unicode character code to "utf-8" encoded string.
|
||||
*
|
||||
* @param int|float $code Unicode character code. Will be casted to int internally!
|
||||
*/
|
||||
public static function uchr($code): string
|
||||
{
|
||||
// note:
|
||||
// $code was typed as int before, but changed in https://github.com/smalot/pdfparser/pull/623
|
||||
// because in some cases uchr was called with a float instead of an integer.
|
||||
$code = (int) $code;
|
||||
|
||||
if (!isset(self::$uchrCache[$code])) {
|
||||
// html_entity_decode() will not work with UTF-16 or UTF-32 char entities,
|
||||
// therefore, we use mb_convert_encoding() instead
|
||||
self::$uchrCache[$code] = mb_convert_encoding("&#{$code};", 'UTF-8', 'HTML-ENTITIES');
|
||||
}
|
||||
|
||||
return self::$uchrCache[$code];
|
||||
}
|
||||
|
||||
/**
|
||||
* Init internal chars translation table by ToUnicode CMap.
|
||||
*/
|
||||
public function loadTranslateTable(): array
|
||||
{
|
||||
if (null !== $this->table) {
|
||||
return $this->table;
|
||||
}
|
||||
|
||||
$this->table = [];
|
||||
$this->tableSizes = [
|
||||
'from' => 1,
|
||||
'to' => 1,
|
||||
];
|
||||
|
||||
if ($this->has('ToUnicode')) {
|
||||
$content = $this->get('ToUnicode')->getContent();
|
||||
$matches = [];
|
||||
|
||||
// Support for multiple spacerange sections
|
||||
if (preg_match_all('/begincodespacerange(?P<sections>.*?)endcodespacerange/s', $content, $matches)) {
|
||||
foreach ($matches['sections'] as $section) {
|
||||
$regexp = '/<(?P<from>[0-9A-F]+)> *<(?P<to>[0-9A-F]+)>[ \r\n]+/is';
|
||||
|
||||
preg_match_all($regexp, $section, $matches);
|
||||
|
||||
$this->tableSizes = [
|
||||
'from' => max(1, \strlen(current($matches['from'])) / 2),
|
||||
'to' => max(1, \strlen(current($matches['to'])) / 2),
|
||||
];
|
||||
|
||||
break;
|
||||
}
|
||||
}
|
||||
|
||||
// Support for multiple bfchar sections
|
||||
if (preg_match_all('/beginbfchar(?P<sections>.*?)endbfchar/s', $content, $matches)) {
|
||||
foreach ($matches['sections'] as $section) {
|
||||
$regexp = '/<(?P<from>[0-9A-F]+)> *<(?P<to>[0-9A-F]+)>[ \r\n]+/is';
|
||||
|
||||
preg_match_all($regexp, $section, $matches);
|
||||
|
||||
$this->tableSizes['from'] = max(1, \strlen(current($matches['from'])) / 2);
|
||||
|
||||
foreach ($matches['from'] as $key => $from) {
|
||||
$parts = preg_split(
|
||||
'/([0-9A-F]{4})/i',
|
||||
$matches['to'][$key],
|
||||
0,
|
||||
\PREG_SPLIT_NO_EMPTY | \PREG_SPLIT_DELIM_CAPTURE
|
||||
);
|
||||
$text = '';
|
||||
foreach ($parts as $part) {
|
||||
$text .= self::uchr(hexdec($part));
|
||||
}
|
||||
$this->table[hexdec($from)] = $text;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// Support for multiple bfrange sections
|
||||
if (preg_match_all('/beginbfrange(?P<sections>.*?)endbfrange/s', $content, $matches)) {
|
||||
foreach ($matches['sections'] as $section) {
|
||||
/**
|
||||
* Regexp to capture <from>, <to>, and either <offset> or [...] items.
|
||||
* - (?P<from>...) Source range's start
|
||||
* - (?P<to>...) Source range's end
|
||||
* - (?P<dest>...) Destination range's offset or each char code
|
||||
* Some PDF file has 2-byte Unicode values on new lines > added \r\n
|
||||
*/
|
||||
$regexp = '/<(?P<from>[0-9A-F]+)> *<(?P<to>[0-9A-F]+)> *(?P<dest><[0-9A-F]+>|\[[\r\n<>0-9A-F ]+\])[ \r\n]+/is';
|
||||
|
||||
preg_match_all($regexp, $section, $matches);
|
||||
|
||||
foreach ($matches['from'] as $key => $from) {
|
||||
$char_from = hexdec($from);
|
||||
$char_to = hexdec($matches['to'][$key]);
|
||||
$dest = $matches['dest'][$key];
|
||||
|
||||
if (1 === preg_match('/^<(?P<offset>[0-9A-F]+)>$/i', $dest, $offset_matches)) {
|
||||
// Support for : <srcCode1> <srcCode2> <dstString>
|
||||
$offset = hexdec($offset_matches['offset']);
|
||||
|
||||
for ($char = $char_from; $char <= $char_to; ++$char) {
|
||||
$this->table[$char] = self::uchr($char - $char_from + $offset);
|
||||
}
|
||||
} else {
|
||||
// Support for : <srcCode1> <srcCodeN> [<dstString1> <dstString2> ... <dstStringN>]
|
||||
$strings = [];
|
||||
$matched = preg_match_all('/<(?P<string>[0-9A-F]+)> */is', $dest, $strings);
|
||||
if (false === $matched || 0 === $matched) {
|
||||
continue;
|
||||
}
|
||||
|
||||
foreach ($strings['string'] as $position => $string) {
|
||||
$parts = preg_split(
|
||||
'/([0-9A-F]{4})/i',
|
||||
$string,
|
||||
0,
|
||||
\PREG_SPLIT_NO_EMPTY | \PREG_SPLIT_DELIM_CAPTURE
|
||||
);
|
||||
if (false === $parts) {
|
||||
continue;
|
||||
}
|
||||
$text = '';
|
||||
foreach ($parts as $part) {
|
||||
$text .= self::uchr(hexdec($part));
|
||||
}
|
||||
$this->table[$char_from + $position] = $text;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
return $this->table;
|
||||
}
|
||||
|
||||
/**
|
||||
* Set custom char translation table where:
|
||||
* - key - integer character code;
|
||||
* - value - "utf-8" encoded value;
|
||||
*
|
||||
* @return void
|
||||
*/
|
||||
public function setTable(array $table)
|
||||
{
|
||||
$this->table = $table;
|
||||
}
|
||||
|
||||
/**
|
||||
* Calculate text width with data from header 'Widths'. If width of character is not found then character is added to missing array.
|
||||
*/
|
||||
public function calculateTextWidth(string $text, ?array &$missing = null): ?float
|
||||
{
|
||||
$index_map = array_flip($this->table);
|
||||
$details = $this->getDetails();
|
||||
|
||||
// Usually, Widths key is set in $details array, but if it isn't use an empty array instead.
|
||||
$widths = $details['Widths'] ?? [];
|
||||
|
||||
/*
|
||||
* Widths array is zero indexed but table is not. We must map them based on FirstChar and LastChar
|
||||
*
|
||||
* Note: Without the change you would see warnings in PHP 8.4 because the values of FirstChar or LastChar
|
||||
* can be null sometimes.
|
||||
*/
|
||||
$width_map = array_flip(range((int) $details['FirstChar'], (int) $details['LastChar']));
|
||||
|
||||
$width = null;
|
||||
$missing = [];
|
||||
$textLength = mb_strlen($text);
|
||||
for ($i = 0; $i < $textLength; ++$i) {
|
||||
$char = mb_substr($text, $i, 1);
|
||||
if (
|
||||
!\array_key_exists($char, $index_map)
|
||||
|| !\array_key_exists($index_map[$char], $width_map)
|
||||
|| !\array_key_exists($width_map[$index_map[$char]], $widths)
|
||||
) {
|
||||
$missing[] = $char;
|
||||
continue;
|
||||
}
|
||||
$width_index = $width_map[$index_map[$char]];
|
||||
$width += $widths[$width_index];
|
||||
}
|
||||
|
||||
return $width;
|
||||
}
|
||||
|
||||
/**
|
||||
* Decode hexadecimal encoded string. If $add_braces is true result value would be wrapped by parentheses.
|
||||
*/
|
||||
public static function decodeHexadecimal(string $hexa, bool $add_braces = false): string
|
||||
{
|
||||
// Special shortcut for XML content.
|
||||
if (false !== stripos($hexa, '<?xml')) {
|
||||
return $hexa;
|
||||
}
|
||||
|
||||
$text = '';
|
||||
$parts = preg_split('/(<[a-f0-9\s]+>)/si', $hexa, -1, \PREG_SPLIT_NO_EMPTY | \PREG_SPLIT_DELIM_CAPTURE);
|
||||
|
||||
foreach ($parts as $part) {
|
||||
if (preg_match('/^<[a-f0-9\s]+>$/si', $part)) {
|
||||
// strip whitespace
|
||||
$part = preg_replace("/\s/", '', $part);
|
||||
$part = trim($part, '<>');
|
||||
if ($add_braces) {
|
||||
$text .= '(';
|
||||
}
|
||||
|
||||
$part = pack('H*', $part);
|
||||
$text .= ($add_braces ? preg_replace('/\\\/s', '\\\\\\', $part) : $part);
|
||||
|
||||
if ($add_braces) {
|
||||
$text .= ')';
|
||||
}
|
||||
} else {
|
||||
$text .= $part;
|
||||
}
|
||||
}
|
||||
|
||||
return $text;
|
||||
}
|
||||
|
||||
/**
|
||||
* Decode string with octal-decoded chunks.
|
||||
*/
|
||||
public static function decodeOctal(string $text): string
|
||||
{
|
||||
// Replace all double backslashes \\ with a special string
|
||||
$text = strtr($text, ['\\\\' => '[**pdfparserdblslsh**]']);
|
||||
|
||||
// Now we can replace all octal codes without worrying about
|
||||
// escaped backslashes
|
||||
$text = preg_replace_callback('/\\\\([0-7]{1,3})/', function ($m) {
|
||||
return \chr(octdec($m[1]));
|
||||
}, $text);
|
||||
|
||||
// Unescape any parentheses
|
||||
$text = str_replace(['\\(', '\\)'], ['(', ')'], $text);
|
||||
|
||||
// Replace instances of the special string with a single backslash
|
||||
return str_replace('[**pdfparserdblslsh**]', '\\', $text);
|
||||
}
|
||||
|
||||
/**
|
||||
* Decode string with html entity encoded chars.
|
||||
*/
|
||||
public static function decodeEntities(string $text): string
|
||||
{
|
||||
return preg_replace_callback('/#([0-9a-f]{2})/i', function ($m) {
|
||||
return \chr(hexdec($m[1]));
|
||||
}, $text);
|
||||
}
|
||||
|
||||
/**
|
||||
* Check if given string is Unicode text (by BOM);
|
||||
* If true - decode to "utf-8" encoded string.
|
||||
* Otherwise - return text as is.
|
||||
*
|
||||
* @todo Rename in next major release to make the name correspond to reality (for ex. decodeIfUnicode())
|
||||
*/
|
||||
public static function decodeUnicode(string $text): string
|
||||
{
|
||||
if ("\xFE\xFF" === substr($text, 0, 2)) {
|
||||
// Strip U+FEFF byte order marker.
|
||||
$decode = substr($text, 2);
|
||||
$text = '';
|
||||
$length = \strlen($decode);
|
||||
|
||||
for ($i = 0; $i < $length; $i += 2) {
|
||||
$text .= self::uchr(hexdec(bin2hex(substr($decode, $i, 2))));
|
||||
}
|
||||
}
|
||||
|
||||
return $text;
|
||||
}
|
||||
|
||||
/**
|
||||
* @todo Deprecated, use $this->config->getFontSpaceLimit() instead.
|
||||
*/
|
||||
protected function getFontSpaceLimit(): int
|
||||
{
|
||||
return $this->config->getFontSpaceLimit();
|
||||
}
|
||||
|
||||
/**
|
||||
* Decode text by commands array.
|
||||
*/
|
||||
public function decodeText(array $commands, float $fontFactor = 4): string
|
||||
{
|
||||
$word_position = 0;
|
||||
$words = [];
|
||||
$font_space = $this->getFontSpaceLimit() * abs($fontFactor) / 4;
|
||||
|
||||
foreach ($commands as $command) {
|
||||
switch ($command[PDFObject::TYPE]) {
|
||||
case 'n':
|
||||
$offset = (float) trim($command[PDFObject::COMMAND]);
|
||||
if ($offset - (float) $font_space < 0) {
|
||||
$word_position = \count($words);
|
||||
}
|
||||
continue 2;
|
||||
case '<':
|
||||
// Decode hexadecimal.
|
||||
$text = self::decodeHexadecimal('<'.$command[PDFObject::COMMAND].'>');
|
||||
break;
|
||||
|
||||
default:
|
||||
// Decode octal (if necessary).
|
||||
$text = self::decodeOctal($command[PDFObject::COMMAND]);
|
||||
}
|
||||
|
||||
// replace escaped chars
|
||||
$text = str_replace(
|
||||
['\\\\', '\(', '\)', '\n', '\r', '\t', '\f', '\ ', '\b'],
|
||||
[\chr(92), \chr(40), \chr(41), \chr(10), \chr(13), \chr(9), \chr(12), \chr(32), \chr(8)],
|
||||
$text
|
||||
);
|
||||
|
||||
// add content to result string
|
||||
if (isset($words[$word_position])) {
|
||||
$words[$word_position] .= $text;
|
||||
} else {
|
||||
$words[$word_position] = $text;
|
||||
}
|
||||
}
|
||||
|
||||
foreach ($words as &$word) {
|
||||
$word = $this->decodeContent($word);
|
||||
$word = str_replace("\t", ' ', $word);
|
||||
}
|
||||
|
||||
// Remove internal "words" that are just spaces, but leave them
|
||||
// if they are at either end of the array of words. This fixes,
|
||||
// for example, lines that are justified to fill
|
||||
// a whole row.
|
||||
for ($x = \count($words) - 2; $x >= 1; --$x) {
|
||||
if ('' === trim($words[$x], ' ')) {
|
||||
unset($words[$x]);
|
||||
}
|
||||
}
|
||||
$words = array_values($words);
|
||||
|
||||
// Cut down on the number of unnecessary internal spaces by
|
||||
// imploding the string on the null byte, and checking if the
|
||||
// text includes extra spaces on either side. If so, merge
|
||||
// where appropriate.
|
||||
$words = implode("\x00\x00", $words);
|
||||
$words = str_replace(
|
||||
[" \x00\x00 ", "\x00\x00 ", " \x00\x00", "\x00\x00"],
|
||||
[' ', ' ', ' ', ' '],
|
||||
$words
|
||||
);
|
||||
|
||||
return $words;
|
||||
}
|
||||
|
||||
/**
|
||||
* Decode given $text to "utf-8" encoded string.
|
||||
*
|
||||
* @param bool $unicode This parameter is deprecated and might be removed in a future release
|
||||
*/
|
||||
public function decodeContent(string $text, ?bool &$unicode = null): string
|
||||
{
|
||||
// If this string begins with a UTF-16BE BOM, then decode it
|
||||
// directly as Unicode
|
||||
if ("\xFE\xFF" === substr($text, 0, 2)) {
|
||||
return $this->decodeUnicode($text);
|
||||
}
|
||||
|
||||
if ($this->has('ToUnicode')) {
|
||||
return $this->decodeContentByToUnicodeCMapOrDescendantFonts($text);
|
||||
}
|
||||
|
||||
if ($this->has('Encoding')) {
|
||||
$result = $this->decodeContentByEncoding($text);
|
||||
|
||||
if (null !== $result) {
|
||||
return $result;
|
||||
}
|
||||
}
|
||||
|
||||
return $this->decodeContentByAutodetectIfNecessary($text);
|
||||
}
|
||||
|
||||
/**
|
||||
* First try to decode $text by ToUnicode CMap.
|
||||
* If char translation not found in ToUnicode CMap tries:
|
||||
* - If DescendantFonts exists tries to decode char by one of that fonts.
|
||||
* - If have no success to decode by DescendantFonts interpret $text as a string with "Windows-1252" encoding.
|
||||
* - If DescendantFonts does not exist just return "?" as decoded char.
|
||||
*
|
||||
* @todo Seems this is invalid algorithm that do not follow pdf-format specification. Must be rewritten.
|
||||
*/
|
||||
private function decodeContentByToUnicodeCMapOrDescendantFonts(string $text): string
|
||||
{
|
||||
$bytes = $this->tableSizes['from'];
|
||||
|
||||
if ($bytes) {
|
||||
$result = '';
|
||||
$length = \strlen($text);
|
||||
|
||||
for ($i = 0; $i < $length; $i += $bytes) {
|
||||
$char = substr($text, $i, $bytes);
|
||||
|
||||
if (false !== ($decoded = $this->translateChar($char, false))) {
|
||||
$char = $decoded;
|
||||
} elseif ($this->has('DescendantFonts')) {
|
||||
if ($this->get('DescendantFonts') instanceof PDFObject) {
|
||||
$fonts = $this->get('DescendantFonts')->getHeader()->getElements();
|
||||
} else {
|
||||
$fonts = $this->get('DescendantFonts')->getContent();
|
||||
}
|
||||
$decoded = false;
|
||||
|
||||
foreach ($fonts as $font) {
|
||||
if ($font instanceof self) {
|
||||
if (false !== ($decoded = $font->translateChar($char, false))) {
|
||||
$decoded = mb_convert_encoding($decoded, 'UTF-8', 'Windows-1252');
|
||||
break;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
if (false !== $decoded) {
|
||||
$char = $decoded;
|
||||
} else {
|
||||
$char = mb_convert_encoding($char, 'UTF-8', 'Windows-1252');
|
||||
}
|
||||
} else {
|
||||
$char = self::MISSING;
|
||||
}
|
||||
|
||||
$result .= $char;
|
||||
}
|
||||
|
||||
$text = $result;
|
||||
}
|
||||
|
||||
return $text;
|
||||
}
|
||||
|
||||
/**
|
||||
* Decode content by any type of Encoding (dictionary's item) instance.
|
||||
*/
|
||||
private function decodeContentByEncoding(string $text): ?string
|
||||
{
|
||||
$encoding = $this->get('Encoding');
|
||||
|
||||
// When Encoding referenced by object id (/Encoding 520 0 R) but object itself does not contain `/Type /Encoding` in it's dictionary.
|
||||
if ($encoding instanceof PDFObject) {
|
||||
$encoding = $this->getInitializedEncodingByPdfObject($encoding);
|
||||
}
|
||||
|
||||
// When Encoding referenced by object id (/Encoding 520 0 R) but object itself contains `/Type /Encoding` in it's dictionary.
|
||||
if ($encoding instanceof Encoding) {
|
||||
return $this->decodeContentByEncodingEncoding($text, $encoding);
|
||||
}
|
||||
|
||||
// When Encoding is just string (/Encoding /WinAnsiEncoding)
|
||||
if ($encoding instanceof Element) { // todo: ElementString class must by used?
|
||||
return $this->decodeContentByEncodingElement($text, $encoding);
|
||||
}
|
||||
|
||||
// don't double-encode strings already in UTF-8
|
||||
if (!mb_check_encoding($text, 'UTF-8')) {
|
||||
return mb_convert_encoding($text, 'UTF-8', 'Windows-1252');
|
||||
}
|
||||
|
||||
return $text;
|
||||
}
|
||||
|
||||
/**
|
||||
* Returns already created or create a new one if not created before Encoding instance by PDFObject instance.
|
||||
*/
|
||||
private function getInitializedEncodingByPdfObject(PDFObject $PDFObject): Encoding
|
||||
{
|
||||
if (!$this->initializedEncodingByPdfObject) {
|
||||
$this->initializedEncodingByPdfObject = $this->createInitializedEncodingByPdfObject($PDFObject);
|
||||
}
|
||||
|
||||
return $this->initializedEncodingByPdfObject;
|
||||
}
|
||||
|
||||
/**
|
||||
* Decode content when $encoding (given by $this->get('Encoding')) is instance of Encoding.
|
||||
*/
|
||||
private function decodeContentByEncodingEncoding(string $text, Encoding $encoding): string
|
||||
{
|
||||
$result = '';
|
||||
$length = \strlen($text);
|
||||
|
||||
for ($i = 0; $i < $length; ++$i) {
|
||||
$dec_av = hexdec(bin2hex($text[$i]));
|
||||
$dec_ap = $encoding->translateChar($dec_av);
|
||||
$result .= self::uchr($dec_ap ?? $dec_av);
|
||||
}
|
||||
|
||||
return $result;
|
||||
}
|
||||
|
||||
/**
|
||||
* Decode content when $encoding (given by $this->get('Encoding')) is instance of Element.
|
||||
*/
|
||||
private function decodeContentByEncodingElement(string $text, Element $encoding): ?string
|
||||
{
|
||||
$pdfEncodingName = $encoding->getContent();
|
||||
|
||||
// mb_convert_encoding does not support MacRoman/macintosh,
|
||||
// so we use iconv() here
|
||||
$iconvEncodingName = $this->getIconvEncodingNameOrNullByPdfEncodingName($pdfEncodingName);
|
||||
|
||||
return $iconvEncodingName ? iconv($iconvEncodingName, 'UTF-8//TRANSLIT//IGNORE', $text) : null;
|
||||
}
|
||||
|
||||
/**
|
||||
* Convert PDF encoding name to iconv-known encoding name.
|
||||
*/
|
||||
private function getIconvEncodingNameOrNullByPdfEncodingName(string $pdfEncodingName): ?string
|
||||
{
|
||||
$pdfToIconvEncodingNameMap = [
|
||||
'StandardEncoding' => 'ISO-8859-1',
|
||||
'MacRomanEncoding' => 'MACINTOSH',
|
||||
'WinAnsiEncoding' => 'CP1252',
|
||||
];
|
||||
|
||||
return \array_key_exists($pdfEncodingName, $pdfToIconvEncodingNameMap)
|
||||
? $pdfToIconvEncodingNameMap[$pdfEncodingName]
|
||||
: null;
|
||||
}
|
||||
|
||||
/**
|
||||
* If string seems like "utf-8" encoded string do nothing and just return given string as is.
|
||||
* Otherwise, interpret string as "Window-1252" encoded string.
|
||||
*
|
||||
* @return string|false
|
||||
*/
|
||||
private function decodeContentByAutodetectIfNecessary(string $text)
|
||||
{
|
||||
if (mb_check_encoding($text, 'UTF-8')) {
|
||||
return $text;
|
||||
}
|
||||
|
||||
return mb_convert_encoding($text, 'UTF-8', 'Windows-1252');
|
||||
// todo: Why exactly `Windows-1252` used?
|
||||
}
|
||||
|
||||
/**
|
||||
* Create Encoding instance by PDFObject instance and init it.
|
||||
*/
|
||||
private function createInitializedEncodingByPdfObject(PDFObject $PDFObject): Encoding
|
||||
{
|
||||
$encoding = $this->createEncodingByPdfObject($PDFObject);
|
||||
$encoding->init();
|
||||
|
||||
return $encoding;
|
||||
}
|
||||
|
||||
/**
|
||||
* Create Encoding instance by PDFObject instance (without init).
|
||||
*/
|
||||
private function createEncodingByPdfObject(PDFObject $PDFObject): Encoding
|
||||
{
|
||||
$document = $PDFObject->getDocument();
|
||||
$header = $PDFObject->getHeader();
|
||||
$content = $PDFObject->getContent();
|
||||
$config = $PDFObject->getConfig();
|
||||
|
||||
return new Encoding($document, $header, $content, $config);
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,42 @@
|
||||
<?php
|
||||
|
||||
/**
|
||||
* @file
|
||||
* This file is part of the PdfParser library.
|
||||
*
|
||||
* @author Sébastien MALOT <sebastien@malot.fr>
|
||||
*
|
||||
* @date 2017-01-03
|
||||
*
|
||||
* @license LGPLv3
|
||||
*
|
||||
* @url <https://github.com/smalot/pdfparser>
|
||||
*
|
||||
* PdfParser is a pdf library written in PHP, extraction oriented.
|
||||
* Copyright (C) 2017 - Sébastien MALOT <sebastien@malot.fr>
|
||||
*
|
||||
* This program is free software: you can redistribute it and/or modify
|
||||
* it under the terms of the GNU Lesser General Public License as published by
|
||||
* the Free Software Foundation, either version 3 of the License, or
|
||||
* (at your option) any later version.
|
||||
*
|
||||
* This program is distributed in the hope that it will be useful,
|
||||
* but WITHOUT ANY WARRANTY; without even the implied warranty of
|
||||
* MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
|
||||
* GNU Lesser General Public License for more details.
|
||||
*
|
||||
* You should have received a copy of the GNU Lesser General Public License
|
||||
* along with this program.
|
||||
* If not, see <http://www.pdfparser.org/sites/default/LICENSE.txt>.
|
||||
*/
|
||||
|
||||
namespace Smalot\PdfParser\Font;
|
||||
|
||||
use Smalot\PdfParser\Font;
|
||||
|
||||
/**
|
||||
* Class FontCIDFontType0
|
||||
*/
|
||||
class FontCIDFontType0 extends Font
|
||||
{
|
||||
}
|
||||
@@ -0,0 +1,42 @@
|
||||
<?php
|
||||
|
||||
/**
|
||||
* @file
|
||||
* This file is part of the PdfParser library.
|
||||
*
|
||||
* @author Sébastien MALOT <sebastien@malot.fr>
|
||||
*
|
||||
* @date 2017-01-03
|
||||
*
|
||||
* @license LGPLv3
|
||||
*
|
||||
* @url <https://github.com/smalot/pdfparser>
|
||||
*
|
||||
* PdfParser is a pdf library written in PHP, extraction oriented.
|
||||
* Copyright (C) 2017 - Sébastien MALOT <sebastien@malot.fr>
|
||||
*
|
||||
* This program is free software: you can redistribute it and/or modify
|
||||
* it under the terms of the GNU Lesser General Public License as published by
|
||||
* the Free Software Foundation, either version 3 of the License, or
|
||||
* (at your option) any later version.
|
||||
*
|
||||
* This program is distributed in the hope that it will be useful,
|
||||
* but WITHOUT ANY WARRANTY; without even the implied warranty of
|
||||
* MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
|
||||
* GNU Lesser General Public License for more details.
|
||||
*
|
||||
* You should have received a copy of the GNU Lesser General Public License
|
||||
* along with this program.
|
||||
* If not, see <http://www.pdfparser.org/sites/default/LICENSE.txt>.
|
||||
*/
|
||||
|
||||
namespace Smalot\PdfParser\Font;
|
||||
|
||||
use Smalot\PdfParser\Font;
|
||||
|
||||
/**
|
||||
* Class FontCIDFontType2
|
||||
*/
|
||||
class FontCIDFontType2 extends Font
|
||||
{
|
||||
}
|
||||
@@ -0,0 +1,42 @@
|
||||
<?php
|
||||
|
||||
/**
|
||||
* @file
|
||||
* This file is part of the PdfParser library.
|
||||
*
|
||||
* @author Sébastien MALOT <sebastien@malot.fr>
|
||||
*
|
||||
* @date 2017-01-03
|
||||
*
|
||||
* @license LGPLv3
|
||||
*
|
||||
* @url <https://github.com/smalot/pdfparser>
|
||||
*
|
||||
* PdfParser is a pdf library written in PHP, extraction oriented.
|
||||
* Copyright (C) 2017 - Sébastien MALOT <sebastien@malot.fr>
|
||||
*
|
||||
* This program is free software: you can redistribute it and/or modify
|
||||
* it under the terms of the GNU Lesser General Public License as published by
|
||||
* the Free Software Foundation, either version 3 of the License, or
|
||||
* (at your option) any later version.
|
||||
*
|
||||
* This program is distributed in the hope that it will be useful,
|
||||
* but WITHOUT ANY WARRANTY; without even the implied warranty of
|
||||
* MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
|
||||
* GNU Lesser General Public License for more details.
|
||||
*
|
||||
* You should have received a copy of the GNU Lesser General Public License
|
||||
* along with this program.
|
||||
* If not, see <http://www.pdfparser.org/sites/default/LICENSE.txt>.
|
||||
*/
|
||||
|
||||
namespace Smalot\PdfParser\Font;
|
||||
|
||||
use Smalot\PdfParser\Font;
|
||||
|
||||
/**
|
||||
* Class FontTrueType
|
||||
*/
|
||||
class FontTrueType extends Font
|
||||
{
|
||||
}
|
||||
@@ -0,0 +1,42 @@
|
||||
<?php
|
||||
|
||||
/**
|
||||
* @file
|
||||
* This file is part of the PdfParser library.
|
||||
*
|
||||
* @author Sébastien MALOT <sebastien@malot.fr>
|
||||
*
|
||||
* @date 2017-01-03
|
||||
*
|
||||
* @license LGPLv3
|
||||
*
|
||||
* @url <https://github.com/smalot/pdfparser>
|
||||
*
|
||||
* PdfParser is a pdf library written in PHP, extraction oriented.
|
||||
* Copyright (C) 2017 - Sébastien MALOT <sebastien@malot.fr>
|
||||
*
|
||||
* This program is free software: you can redistribute it and/or modify
|
||||
* it under the terms of the GNU Lesser General Public License as published by
|
||||
* the Free Software Foundation, either version 3 of the License, or
|
||||
* (at your option) any later version.
|
||||
*
|
||||
* This program is distributed in the hope that it will be useful,
|
||||
* but WITHOUT ANY WARRANTY; without even the implied warranty of
|
||||
* MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
|
||||
* GNU Lesser General Public License for more details.
|
||||
*
|
||||
* You should have received a copy of the GNU Lesser General Public License
|
||||
* along with this program.
|
||||
* If not, see <http://www.pdfparser.org/sites/default/LICENSE.txt>.
|
||||
*/
|
||||
|
||||
namespace Smalot\PdfParser\Font;
|
||||
|
||||
use Smalot\PdfParser\Font;
|
||||
|
||||
/**
|
||||
* Class FontType0
|
||||
*/
|
||||
class FontType0 extends Font
|
||||
{
|
||||
}
|
||||
@@ -0,0 +1,42 @@
|
||||
<?php
|
||||
|
||||
/**
|
||||
* @file
|
||||
* This file is part of the PdfParser library.
|
||||
*
|
||||
* @author Sébastien MALOT <sebastien@malot.fr>
|
||||
*
|
||||
* @date 2017-01-03
|
||||
*
|
||||
* @license LGPLv3
|
||||
*
|
||||
* @url <https://github.com/smalot/pdfparser>
|
||||
*
|
||||
* PdfParser is a pdf library written in PHP, extraction oriented.
|
||||
* Copyright (C) 2017 - Sébastien MALOT <sebastien@malot.fr>
|
||||
*
|
||||
* This program is free software: you can redistribute it and/or modify
|
||||
* it under the terms of the GNU Lesser General Public License as published by
|
||||
* the Free Software Foundation, either version 3 of the License, or
|
||||
* (at your option) any later version.
|
||||
*
|
||||
* This program is distributed in the hope that it will be useful,
|
||||
* but WITHOUT ANY WARRANTY; without even the implied warranty of
|
||||
* MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
|
||||
* GNU Lesser General Public License for more details.
|
||||
*
|
||||
* You should have received a copy of the GNU Lesser General Public License
|
||||
* along with this program.
|
||||
* If not, see <http://www.pdfparser.org/sites/default/LICENSE.txt>.
|
||||
*/
|
||||
|
||||
namespace Smalot\PdfParser\Font;
|
||||
|
||||
use Smalot\PdfParser\Font;
|
||||
|
||||
/**
|
||||
* Class FontType1
|
||||
*/
|
||||
class FontType1 extends Font
|
||||
{
|
||||
}
|
||||
@@ -0,0 +1,42 @@
|
||||
<?php
|
||||
|
||||
/**
|
||||
* @file
|
||||
* This file is part of the PdfParser library.
|
||||
*
|
||||
* @author Sébastien MALOT <sebastien@malot.fr>
|
||||
*
|
||||
* @date 2017-01-03
|
||||
*
|
||||
* @license LGPLv3
|
||||
*
|
||||
* @url <https://github.com/smalot/pdfparser>
|
||||
*
|
||||
* PdfParser is a pdf library written in PHP, extraction oriented.
|
||||
* Copyright (C) 2017 - Sébastien MALOT <sebastien@malot.fr>
|
||||
*
|
||||
* This program is free software: you can redistribute it and/or modify
|
||||
* it under the terms of the GNU Lesser General Public License as published by
|
||||
* the Free Software Foundation, either version 3 of the License, or
|
||||
* (at your option) any later version.
|
||||
*
|
||||
* This program is distributed in the hope that it will be useful,
|
||||
* but WITHOUT ANY WARRANTY; without even the implied warranty of
|
||||
* MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
|
||||
* GNU Lesser General Public License for more details.
|
||||
*
|
||||
* You should have received a copy of the GNU Lesser General Public License
|
||||
* along with this program.
|
||||
* If not, see <http://www.pdfparser.org/sites/default/LICENSE.txt>.
|
||||
*/
|
||||
|
||||
namespace Smalot\PdfParser\Font;
|
||||
|
||||
use Smalot\PdfParser\Font;
|
||||
|
||||
/**
|
||||
* Class FontType3
|
||||
*/
|
||||
class FontType3 extends Font
|
||||
{
|
||||
}
|
||||
@@ -0,0 +1,194 @@
|
||||
<?php
|
||||
|
||||
/**
|
||||
* @file
|
||||
* This file is part of the PdfParser library.
|
||||
*
|
||||
* @author Sébastien MALOT <sebastien@malot.fr>
|
||||
*
|
||||
* @date 2017-01-03
|
||||
*
|
||||
* @license LGPLv3
|
||||
*
|
||||
* @url <https://github.com/smalot/pdfparser>
|
||||
*
|
||||
* PdfParser is a pdf library written in PHP, extraction oriented.
|
||||
* Copyright (C) 2017 - Sébastien MALOT <sebastien@malot.fr>
|
||||
*
|
||||
* This program is free software: you can redistribute it and/or modify
|
||||
* it under the terms of the GNU Lesser General Public License as published by
|
||||
* the Free Software Foundation, either version 3 of the License, or
|
||||
* (at your option) any later version.
|
||||
*
|
||||
* This program is distributed in the hope that it will be useful,
|
||||
* but WITHOUT ANY WARRANTY; without even the implied warranty of
|
||||
* MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
|
||||
* GNU Lesser General Public License for more details.
|
||||
*
|
||||
* You should have received a copy of the GNU Lesser General Public License
|
||||
* along with this program.
|
||||
* If not, see <http://www.pdfparser.org/sites/default/LICENSE.txt>.
|
||||
*/
|
||||
|
||||
namespace Smalot\PdfParser;
|
||||
|
||||
use Smalot\PdfParser\Element\ElementArray;
|
||||
use Smalot\PdfParser\Element\ElementMissing;
|
||||
use Smalot\PdfParser\Element\ElementStruct;
|
||||
use Smalot\PdfParser\Element\ElementXRef;
|
||||
|
||||
/**
|
||||
* Class Header
|
||||
*/
|
||||
class Header
|
||||
{
|
||||
/**
|
||||
* @var Document|null
|
||||
*/
|
||||
protected $document;
|
||||
|
||||
/**
|
||||
* @var Element[]
|
||||
*/
|
||||
protected $elements;
|
||||
|
||||
/**
|
||||
* @param Element[] $elements list of elements
|
||||
* @param Document $document document
|
||||
*/
|
||||
public function __construct(array $elements = [], ?Document $document = null)
|
||||
{
|
||||
$this->elements = $elements;
|
||||
$this->document = $document;
|
||||
}
|
||||
|
||||
public function init()
|
||||
{
|
||||
foreach ($this->elements as $element) {
|
||||
if ($element instanceof Element) {
|
||||
$element->init();
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Returns all elements.
|
||||
*/
|
||||
public function getElements()
|
||||
{
|
||||
foreach ($this->elements as $name => $element) {
|
||||
$this->resolveXRef($name);
|
||||
}
|
||||
|
||||
return $this->elements;
|
||||
}
|
||||
|
||||
/**
|
||||
* Used only for debug.
|
||||
*/
|
||||
public function getElementTypes(): array
|
||||
{
|
||||
$types = [];
|
||||
|
||||
foreach ($this->elements as $key => $element) {
|
||||
$types[$key] = \get_class($element);
|
||||
}
|
||||
|
||||
return $types;
|
||||
}
|
||||
|
||||
public function getDetails(bool $deep = true): array
|
||||
{
|
||||
$values = [];
|
||||
$elements = $this->getElements();
|
||||
|
||||
foreach ($elements as $key => $element) {
|
||||
if ($element instanceof self && $deep) {
|
||||
$values[$key] = $element->getDetails($deep);
|
||||
} elseif ($element instanceof PDFObject && $deep) {
|
||||
$values[$key] = $element->getDetails(false);
|
||||
} elseif ($element instanceof ElementArray) {
|
||||
if ($deep) {
|
||||
$values[$key] = $element->getDetails();
|
||||
}
|
||||
} elseif ($element instanceof Element) {
|
||||
$values[$key] = (string) $element;
|
||||
}
|
||||
}
|
||||
|
||||
return $values;
|
||||
}
|
||||
|
||||
/**
|
||||
* Indicate if an element name is available in header.
|
||||
*
|
||||
* @param string $name the name of the element
|
||||
*/
|
||||
public function has(string $name): bool
|
||||
{
|
||||
return \array_key_exists($name, $this->elements);
|
||||
}
|
||||
|
||||
/**
|
||||
* @return Element|PDFObject
|
||||
*/
|
||||
public function get(string $name)
|
||||
{
|
||||
if (\array_key_exists($name, $this->elements) && $element = $this->resolveXRef($name)) {
|
||||
return $element;
|
||||
}
|
||||
|
||||
return new ElementMissing();
|
||||
}
|
||||
|
||||
/**
|
||||
* Resolve XRef to object.
|
||||
*
|
||||
* @return Element|PDFObject
|
||||
*
|
||||
* @throws \Exception
|
||||
*/
|
||||
protected function resolveXRef(string $name)
|
||||
{
|
||||
if (($obj = $this->elements[$name]) instanceof ElementXRef && null !== $this->document) {
|
||||
/** @var ElementXRef $obj */
|
||||
$object = $this->document->getObjectById($obj->getId());
|
||||
|
||||
if (null === $object) {
|
||||
return new ElementMissing();
|
||||
}
|
||||
|
||||
// Update elements list for future calls.
|
||||
$this->elements[$name] = $object;
|
||||
}
|
||||
|
||||
return $this->elements[$name];
|
||||
}
|
||||
|
||||
/**
|
||||
* @param string $content The content to parse
|
||||
* @param Document $document The document
|
||||
* @param int $position The new position of the cursor after parsing
|
||||
*/
|
||||
public static function parse(string $content, Document $document, int &$position = 0): self
|
||||
{
|
||||
/* @var Header $header */
|
||||
if ('<<' == substr(trim($content), 0, 2)) {
|
||||
$header = ElementStruct::parse($content, $document, $position);
|
||||
} else {
|
||||
$elements = ElementArray::parse($content, $document, $position);
|
||||
$header = new self([], $document);
|
||||
|
||||
if ($elements) {
|
||||
$header = new self($elements->getRawContent(), null);
|
||||
}
|
||||
}
|
||||
|
||||
if ($header) {
|
||||
return $header;
|
||||
}
|
||||
|
||||
// Build an empty header.
|
||||
return new self([], $document);
|
||||
}
|
||||
}
|
||||
File diff suppressed because it is too large
Load Diff
+1014
File diff suppressed because it is too large
Load Diff
@@ -0,0 +1,131 @@
|
||||
<?php
|
||||
|
||||
/**
|
||||
* @file
|
||||
* This file is part of the PdfParser library.
|
||||
*
|
||||
* @author Sébastien MALOT <sebastien@malot.fr>
|
||||
*
|
||||
* @date 2017-01-03
|
||||
*
|
||||
* @license LGPLv3
|
||||
*
|
||||
* @url <https://github.com/smalot/pdfparser>
|
||||
*
|
||||
* PdfParser is a pdf library written in PHP, extraction oriented.
|
||||
* Copyright (C) 2017 - Sébastien MALOT <sebastien@malot.fr>
|
||||
*
|
||||
* This program is free software: you can redistribute it and/or modify
|
||||
* it under the terms of the GNU Lesser General Public License as published by
|
||||
* the Free Software Foundation, either version 3 of the License, or
|
||||
* (at your option) any later version.
|
||||
*
|
||||
* This program is distributed in the hope that it will be useful,
|
||||
* but WITHOUT ANY WARRANTY; without even the implied warranty of
|
||||
* MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
|
||||
* GNU Lesser General Public License for more details.
|
||||
*
|
||||
* You should have received a copy of the GNU Lesser General Public License
|
||||
* along with this program.
|
||||
* If not, see <http://www.pdfparser.org/sites/default/LICENSE.txt>.
|
||||
*/
|
||||
|
||||
namespace Smalot\PdfParser;
|
||||
|
||||
use Smalot\PdfParser\Element\ElementArray;
|
||||
|
||||
/**
|
||||
* Class Pages
|
||||
*/
|
||||
class Pages extends PDFObject
|
||||
{
|
||||
/**
|
||||
* @var array<\Smalot\PdfParser\Font>|null
|
||||
*/
|
||||
protected $fonts;
|
||||
|
||||
/**
|
||||
* @todo Objects other than Pages or Page might need to be treated specifically
|
||||
* in order to get Page objects out of them.
|
||||
*
|
||||
* @see https://github.com/smalot/pdfparser/issues/331
|
||||
*/
|
||||
public function getPages(bool $deep = false): array
|
||||
{
|
||||
if (!$this->has('Kids')) {
|
||||
return [];
|
||||
}
|
||||
|
||||
/** @var ElementArray $kidsElement */
|
||||
$kidsElement = $this->get('Kids');
|
||||
|
||||
if (!$deep) {
|
||||
return $kidsElement->getContent();
|
||||
}
|
||||
|
||||
// Prepare to apply the Pages' object's fonts to each page
|
||||
if (false === \is_array($this->fonts)) {
|
||||
$this->setupFonts();
|
||||
}
|
||||
$fontsAvailable = 0 < \count($this->fonts);
|
||||
|
||||
$kids = $kidsElement->getContent();
|
||||
$pages = [];
|
||||
|
||||
foreach ($kids as $kid) {
|
||||
if ($kid instanceof self) {
|
||||
$pages = array_merge($pages, $kid->getPages(true));
|
||||
} elseif ($kid instanceof Page) {
|
||||
if ($fontsAvailable) {
|
||||
$kid->setFonts($this->fonts);
|
||||
}
|
||||
$pages[] = $kid;
|
||||
}
|
||||
}
|
||||
|
||||
return $pages;
|
||||
}
|
||||
|
||||
/**
|
||||
* Gathers information about fonts and collects them in a list.
|
||||
*
|
||||
* @return void
|
||||
*
|
||||
* @internal
|
||||
*/
|
||||
protected function setupFonts()
|
||||
{
|
||||
$resources = $this->get('Resources');
|
||||
|
||||
if (method_exists($resources, 'has') && $resources->has('Font')) {
|
||||
// no fonts available, therefore stop here
|
||||
if ($resources->get('Font') instanceof Element\ElementMissing) {
|
||||
return;
|
||||
}
|
||||
|
||||
if ($resources->get('Font') instanceof Header) {
|
||||
$fonts = $resources->get('Font')->getElements();
|
||||
} else {
|
||||
$fonts = $resources->get('Font')->getHeader()->getElements();
|
||||
}
|
||||
|
||||
$table = [];
|
||||
|
||||
foreach ($fonts as $id => $font) {
|
||||
if ($font instanceof Font) {
|
||||
$table[$id] = $font;
|
||||
|
||||
// Store too on cleaned id value (only numeric)
|
||||
$id = preg_replace('/[^0-9\.\-_]/', '', $id);
|
||||
if ('' != $id) {
|
||||
$table[$id] = $font;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
$this->fonts = $table;
|
||||
} else {
|
||||
$this->fonts = [];
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,331 @@
|
||||
<?php
|
||||
|
||||
/**
|
||||
* @file
|
||||
* This file is part of the PdfParser library.
|
||||
*
|
||||
* @author Sébastien MALOT <sebastien@malot.fr>
|
||||
*
|
||||
* @date 2017-01-03
|
||||
*
|
||||
* @license LGPLv3
|
||||
*
|
||||
* @url <https://github.com/smalot/pdfparser>
|
||||
*
|
||||
* PdfParser is a pdf library written in PHP, extraction oriented.
|
||||
* Copyright (C) 2017 - Sébastien MALOT <sebastien@malot.fr>
|
||||
*
|
||||
* This program is free software: you can redistribute it and/or modify
|
||||
* it under the terms of the GNU Lesser General Public License as published by
|
||||
* the Free Software Foundation, either version 3 of the License, or
|
||||
* (at your option) any later version.
|
||||
*
|
||||
* This program is distributed in the hope that it will be useful,
|
||||
* but WITHOUT ANY WARRANTY; without even the implied warranty of
|
||||
* MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
|
||||
* GNU Lesser General Public License for more details.
|
||||
*
|
||||
* You should have received a copy of the GNU Lesser General Public License
|
||||
* along with this program.
|
||||
* If not, see <http://www.pdfparser.org/sites/default/LICENSE.txt>.
|
||||
*/
|
||||
|
||||
namespace Smalot\PdfParser;
|
||||
|
||||
use Smalot\PdfParser\Element\ElementArray;
|
||||
use Smalot\PdfParser\Element\ElementBoolean;
|
||||
use Smalot\PdfParser\Element\ElementDate;
|
||||
use Smalot\PdfParser\Element\ElementHexa;
|
||||
use Smalot\PdfParser\Element\ElementName;
|
||||
use Smalot\PdfParser\Element\ElementNull;
|
||||
use Smalot\PdfParser\Element\ElementNumeric;
|
||||
use Smalot\PdfParser\Element\ElementString;
|
||||
use Smalot\PdfParser\Element\ElementXRef;
|
||||
use Smalot\PdfParser\RawData\RawDataParser;
|
||||
|
||||
/**
|
||||
* Class Parser
|
||||
*/
|
||||
class Parser
|
||||
{
|
||||
/**
|
||||
* @var Config
|
||||
*/
|
||||
private $config;
|
||||
|
||||
/**
|
||||
* @var PDFObject[]
|
||||
*/
|
||||
protected $objects = [];
|
||||
|
||||
protected $rawDataParser;
|
||||
|
||||
public function __construct($cfg = [], ?Config $config = null)
|
||||
{
|
||||
$this->config = $config ?: new Config();
|
||||
$this->rawDataParser = new RawDataParser($cfg, $this->config);
|
||||
}
|
||||
|
||||
public function getConfig(): Config
|
||||
{
|
||||
return $this->config;
|
||||
}
|
||||
|
||||
/**
|
||||
* @throws \Exception
|
||||
*/
|
||||
public function parseFile(string $filename): Document
|
||||
{
|
||||
$content = file_get_contents($filename);
|
||||
|
||||
/*
|
||||
* 2018/06/20 @doganoo as multiple times a
|
||||
* users have complained that the parseFile()
|
||||
* method dies silently, it is an better option
|
||||
* to remove the error control operator (@) and
|
||||
* let the users know that the method throws an exception
|
||||
* by adding @throws tag to PHPDoc.
|
||||
*
|
||||
* See here for an example: https://github.com/smalot/pdfparser/issues/204
|
||||
*/
|
||||
return $this->parseContent($content);
|
||||
}
|
||||
|
||||
/**
|
||||
* @param string $content PDF content to parse
|
||||
*
|
||||
* @throws \Exception if secured PDF file was detected
|
||||
* @throws \Exception if no object list was found
|
||||
*/
|
||||
public function parseContent(string $content): Document
|
||||
{
|
||||
// Create structure from raw data.
|
||||
list($xref, $data) = $this->rawDataParser->parseData($content);
|
||||
|
||||
if (isset($xref['trailer']['encrypt']) && false === $this->config->getIgnoreEncryption()) {
|
||||
throw new \Exception('Secured pdf file are currently not supported.');
|
||||
}
|
||||
|
||||
if (empty($data)) {
|
||||
throw new \Exception('Object list not found. Possible secured file.');
|
||||
}
|
||||
|
||||
// Create destination object.
|
||||
$document = new Document();
|
||||
$this->objects = [];
|
||||
|
||||
foreach ($data as $id => $structure) {
|
||||
$this->parseObject($id, $structure, $document);
|
||||
unset($data[$id]);
|
||||
}
|
||||
|
||||
$document->setTrailer($this->parseTrailer($xref['trailer'], $document));
|
||||
$document->setObjects($this->objects);
|
||||
|
||||
return $document;
|
||||
}
|
||||
|
||||
protected function parseTrailer(array $structure, ?Document $document)
|
||||
{
|
||||
$trailer = [];
|
||||
|
||||
foreach ($structure as $name => $values) {
|
||||
$name = ucfirst($name);
|
||||
|
||||
if (is_numeric($values)) {
|
||||
$trailer[$name] = new ElementNumeric($values);
|
||||
} elseif (\is_array($values)) {
|
||||
$value = $this->parseTrailer($values, null);
|
||||
$trailer[$name] = new ElementArray($value, null);
|
||||
} elseif (false !== strpos($values, '_')) {
|
||||
$trailer[$name] = new ElementXRef($values, $document);
|
||||
} else {
|
||||
$trailer[$name] = $this->parseHeaderElement('(', $values, $document);
|
||||
}
|
||||
}
|
||||
|
||||
return new Header($trailer, $document);
|
||||
}
|
||||
|
||||
protected function parseObject(string $id, array $structure, ?Document $document)
|
||||
{
|
||||
$header = new Header([], $document);
|
||||
$content = '';
|
||||
|
||||
foreach ($structure as $position => $part) {
|
||||
if (\is_int($part)) {
|
||||
$part = [null, null];
|
||||
}
|
||||
switch ($part[0]) {
|
||||
case '[':
|
||||
$elements = [];
|
||||
|
||||
foreach ($part[1] as $sub_element) {
|
||||
$sub_type = $sub_element[0];
|
||||
$sub_value = $sub_element[1];
|
||||
$elements[] = $this->parseHeaderElement($sub_type, $sub_value, $document);
|
||||
}
|
||||
|
||||
$header = new Header($elements, $document);
|
||||
break;
|
||||
|
||||
case '<<':
|
||||
$header = $this->parseHeader($part[1], $document);
|
||||
break;
|
||||
|
||||
case 'stream':
|
||||
$content = isset($part[3][0]) ? $part[3][0] : $part[1];
|
||||
|
||||
if ($header->get('Type')->equals('ObjStm')) {
|
||||
$match = [];
|
||||
|
||||
// Split xrefs and contents.
|
||||
preg_match('/^((\d+\s+\d+\s*)*)(.*)$/s', $content, $match);
|
||||
$content = $match[3];
|
||||
|
||||
// Extract xrefs.
|
||||
$xrefs = preg_split(
|
||||
'/(\d+\s+\d+\s*)/s',
|
||||
$match[1],
|
||||
-1,
|
||||
\PREG_SPLIT_NO_EMPTY | \PREG_SPLIT_DELIM_CAPTURE
|
||||
);
|
||||
$table = [];
|
||||
|
||||
foreach ($xrefs as $xref) {
|
||||
list($id, $position) = preg_split("/\s+/", trim($xref));
|
||||
$table[$position] = $id;
|
||||
}
|
||||
|
||||
ksort($table);
|
||||
|
||||
$ids = array_values($table);
|
||||
$positions = array_keys($table);
|
||||
|
||||
foreach ($positions as $index => $position) {
|
||||
$id = $ids[$index].'_0';
|
||||
$next_position = isset($positions[$index + 1]) ? $positions[$index + 1] : \strlen($content);
|
||||
$sub_content = substr($content, $position, (int) $next_position - (int) $position);
|
||||
|
||||
$sub_header = Header::parse($sub_content, $document);
|
||||
$object = PDFObject::factory($document, $sub_header, '', $this->config);
|
||||
$this->objects[$id] = $object;
|
||||
}
|
||||
|
||||
// It is not necessary to store this content.
|
||||
|
||||
return;
|
||||
} elseif ($header->get('Type')->equals('Metadata')) {
|
||||
// Attempt to parse XMP XML Metadata
|
||||
$document->extractXMPMetadata($content);
|
||||
}
|
||||
break;
|
||||
|
||||
default:
|
||||
if ('null' != $part) {
|
||||
$element = $this->parseHeaderElement($part[0], $part[1], $document);
|
||||
|
||||
if ($element) {
|
||||
$header = new Header([$element], $document);
|
||||
}
|
||||
}
|
||||
break;
|
||||
}
|
||||
}
|
||||
|
||||
if (!isset($this->objects[$id])) {
|
||||
$this->objects[$id] = PDFObject::factory($document, $header, $content, $this->config);
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* @throws \Exception
|
||||
*/
|
||||
protected function parseHeader(array $structure, ?Document $document): Header
|
||||
{
|
||||
$elements = [];
|
||||
$count = \count($structure);
|
||||
|
||||
for ($position = 0; $position < $count; $position += 2) {
|
||||
$name = $structure[$position][1];
|
||||
$type = $structure[$position + 1][0];
|
||||
$value = $structure[$position + 1][1];
|
||||
|
||||
$elements[$name] = $this->parseHeaderElement($type, $value, $document);
|
||||
}
|
||||
|
||||
return new Header($elements, $document);
|
||||
}
|
||||
|
||||
/**
|
||||
* @param string|array $value
|
||||
*
|
||||
* @return Element|Header|null
|
||||
*
|
||||
* @throws \Exception
|
||||
*/
|
||||
protected function parseHeaderElement(?string $type, $value, ?Document $document)
|
||||
{
|
||||
$valueIsEmpty = null == $value || '' == $value || false == $value;
|
||||
if (('<<' === $type || '>>' === $type) && $valueIsEmpty) {
|
||||
$value = [];
|
||||
}
|
||||
|
||||
switch ($type) {
|
||||
case '<<':
|
||||
case '>>':
|
||||
$header = $this->parseHeader($value, $document);
|
||||
PDFObject::factory($document, $header, null, $this->config);
|
||||
|
||||
return $header;
|
||||
|
||||
case 'numeric':
|
||||
return new ElementNumeric($value);
|
||||
|
||||
case 'boolean':
|
||||
return new ElementBoolean($value);
|
||||
|
||||
case 'null':
|
||||
return new ElementNull();
|
||||
|
||||
case '(':
|
||||
if ($date = ElementDate::parse('('.$value.')', $document)) {
|
||||
return $date;
|
||||
}
|
||||
|
||||
return ElementString::parse('('.$value.')', $document);
|
||||
|
||||
case '<':
|
||||
return $this->parseHeaderElement('(', ElementHexa::decode($value), $document);
|
||||
|
||||
case '/':
|
||||
return ElementName::parse('/'.$value, $document);
|
||||
|
||||
case 'ojbref': // old mistake in tcpdf parser
|
||||
case 'objref':
|
||||
return new ElementXRef($value, $document);
|
||||
|
||||
case '[':
|
||||
$values = [];
|
||||
|
||||
if (\is_array($value)) {
|
||||
foreach ($value as $sub_element) {
|
||||
$sub_type = $sub_element[0];
|
||||
$sub_value = $sub_element[1];
|
||||
$values[] = $this->parseHeaderElement($sub_type, $sub_value, $document);
|
||||
}
|
||||
}
|
||||
|
||||
return new ElementArray($values, $document);
|
||||
|
||||
case 'endstream':
|
||||
case 'obj': // I don't know what it means but got my project fixed.
|
||||
case '':
|
||||
// Nothing to do with.
|
||||
return null;
|
||||
|
||||
default:
|
||||
throw new \Exception('Invalid type: "'.$type.'".');
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,427 @@
|
||||
<?php
|
||||
|
||||
/**
|
||||
* This file is based on code of tecnickcom/TCPDF PDF library.
|
||||
*
|
||||
* Original author Nicola Asuni (info@tecnick.com) and
|
||||
* contributors (https://github.com/tecnickcom/TCPDF/graphs/contributors).
|
||||
*
|
||||
* @see https://github.com/tecnickcom/TCPDF
|
||||
*
|
||||
* Original code was licensed on the terms of the LGPL v3.
|
||||
*
|
||||
* ------------------------------------------------------------------------------
|
||||
*
|
||||
* @file This file is part of the PdfParser library.
|
||||
*
|
||||
* @author Konrad Abicht <k.abicht@gmail.com>
|
||||
*
|
||||
* @date 2020-01-06
|
||||
*
|
||||
* @license LGPLv3
|
||||
*
|
||||
* @url <https://github.com/smalot/pdfparser>
|
||||
*
|
||||
* PdfParser is a pdf library written in PHP, extraction oriented.
|
||||
* Copyright (C) 2017 - Sébastien MALOT <sebastien@malot.fr>
|
||||
*
|
||||
* This program is free software: you can redistribute it and/or modify
|
||||
* it under the terms of the GNU Lesser General Public License as published by
|
||||
* the Free Software Foundation, either version 3 of the License, or
|
||||
* (at your option) any later version.
|
||||
*
|
||||
* This program is distributed in the hope that it will be useful,
|
||||
* but WITHOUT ANY WARRANTY; without even the implied warranty of
|
||||
* MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
|
||||
* GNU Lesser General Public License for more details.
|
||||
*
|
||||
* You should have received a copy of the GNU Lesser General Public License
|
||||
* along with this program.
|
||||
* If not, see <http://www.pdfparser.org/sites/default/LICENSE.txt>.
|
||||
*/
|
||||
|
||||
namespace Smalot\PdfParser\RawData;
|
||||
|
||||
use Smalot\PdfParser\Exception\NotImplementedException;
|
||||
|
||||
class FilterHelper
|
||||
{
|
||||
protected $availableFilters = ['ASCIIHexDecode', 'ASCII85Decode', 'LZWDecode', 'FlateDecode', 'RunLengthDecode'];
|
||||
|
||||
/**
|
||||
* Decode data using the specified filter type.
|
||||
*
|
||||
* @param string $filter Filter name
|
||||
* @param string $data Data to decode
|
||||
*
|
||||
* @return string Decoded data string
|
||||
*
|
||||
* @throws \Exception
|
||||
* @throws \Smalot\PdfParser\Exception\NotImplementedException if a certain decode function is not implemented yet
|
||||
*/
|
||||
public function decodeFilter(string $filter, string $data, int $decodeMemoryLimit = 0): string
|
||||
{
|
||||
switch ($filter) {
|
||||
case 'ASCIIHexDecode':
|
||||
return $this->decodeFilterASCIIHexDecode($data);
|
||||
|
||||
case 'ASCII85Decode':
|
||||
return $this->decodeFilterASCII85Decode($data);
|
||||
|
||||
case 'LZWDecode':
|
||||
return $this->decodeFilterLZWDecode($data);
|
||||
|
||||
case 'FlateDecode':
|
||||
return $this->decodeFilterFlateDecode($data, $decodeMemoryLimit);
|
||||
|
||||
case 'RunLengthDecode':
|
||||
return $this->decodeFilterRunLengthDecode($data);
|
||||
|
||||
case 'CCITTFaxDecode':
|
||||
throw new NotImplementedException('Decode CCITTFaxDecode not implemented yet.');
|
||||
case 'JBIG2Decode':
|
||||
throw new NotImplementedException('Decode JBIG2Decode not implemented yet.');
|
||||
case 'DCTDecode':
|
||||
throw new NotImplementedException('Decode DCTDecode not implemented yet.');
|
||||
case 'JPXDecode':
|
||||
throw new NotImplementedException('Decode JPXDecode not implemented yet.');
|
||||
case 'Crypt':
|
||||
throw new NotImplementedException('Decode Crypt not implemented yet.');
|
||||
default:
|
||||
return $data;
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* ASCIIHexDecode
|
||||
*
|
||||
* Decodes data encoded in an ASCII hexadecimal representation, reproducing the original binary data.
|
||||
*
|
||||
* @param string $data Data to decode
|
||||
*
|
||||
* @return string data string
|
||||
*
|
||||
* @throws \Exception
|
||||
*/
|
||||
protected function decodeFilterASCIIHexDecode(string $data): string
|
||||
{
|
||||
// all white-space characters shall be ignored
|
||||
$data = preg_replace('/[\s]/', '', $data);
|
||||
// check for EOD character: GREATER-THAN SIGN (3Eh)
|
||||
$eod = strpos($data, '>');
|
||||
if (false !== $eod) {
|
||||
// remove EOD and extra data (if any)
|
||||
$data = substr($data, 0, $eod);
|
||||
$eod = true;
|
||||
}
|
||||
// get data length
|
||||
$data_length = \strlen($data);
|
||||
if (0 != ($data_length % 2)) {
|
||||
// odd number of hexadecimal digits
|
||||
if ($eod) {
|
||||
// EOD shall behave as if a 0 (zero) followed the last digit
|
||||
$data = substr($data, 0, -1).'0'.substr($data, -1);
|
||||
} else {
|
||||
throw new \Exception('decodeFilterASCIIHexDecode: invalid code');
|
||||
}
|
||||
}
|
||||
// check for invalid characters
|
||||
if (preg_match('/[^a-fA-F\d]/', $data) > 0) {
|
||||
throw new \Exception('decodeFilterASCIIHexDecode: invalid code');
|
||||
}
|
||||
// get one byte of binary data for each pair of ASCII hexadecimal digits
|
||||
$decoded = pack('H*', $data);
|
||||
|
||||
return $decoded;
|
||||
}
|
||||
|
||||
/**
|
||||
* ASCII85Decode
|
||||
*
|
||||
* Decodes data encoded in an ASCII base-85 representation, reproducing the original binary data.
|
||||
*
|
||||
* @param string $data Data to decode
|
||||
*
|
||||
* @return string data string
|
||||
*
|
||||
* @throws \Exception
|
||||
*/
|
||||
protected function decodeFilterASCII85Decode(string $data): string
|
||||
{
|
||||
// initialize string to return
|
||||
$decoded = '';
|
||||
// all white-space characters shall be ignored
|
||||
$data = preg_replace('/[\s]/', '', $data);
|
||||
// remove start sequence 2-character sequence <~ (3Ch)(7Eh)
|
||||
if (0 === strpos($data, '<~')) {
|
||||
// remove EOD and extra data (if any)
|
||||
$data = substr($data, 2);
|
||||
}
|
||||
// check for EOD: 2-character sequence ~> (7Eh)(3Eh)
|
||||
$eod = strpos($data, '~>');
|
||||
if (\strlen($data) - 2 === $eod) {
|
||||
// remove EOD and extra data (if any)
|
||||
$data = substr($data, 0, $eod);
|
||||
}
|
||||
// data length
|
||||
$data_length = \strlen($data);
|
||||
// check for invalid characters
|
||||
if (preg_match('/[^\x21-\x75,\x74]/', $data) > 0) {
|
||||
throw new \Exception('decodeFilterASCII85Decode: invalid code');
|
||||
}
|
||||
// z sequence
|
||||
$zseq = \chr(0).\chr(0).\chr(0).\chr(0);
|
||||
// position inside a group of 4 bytes (0-3)
|
||||
$group_pos = 0;
|
||||
$tuple = 0;
|
||||
$pow85 = [85 * 85 * 85 * 85, 85 * 85 * 85, 85 * 85, 85, 1];
|
||||
|
||||
// for each byte
|
||||
for ($i = 0; $i < $data_length; ++$i) {
|
||||
// get char value
|
||||
$char = \ord($data[$i]);
|
||||
if (122 == $char) { // 'z'
|
||||
if (0 == $group_pos) {
|
||||
$decoded .= $zseq;
|
||||
} else {
|
||||
throw new \Exception('decodeFilterASCII85Decode: invalid code');
|
||||
}
|
||||
} else {
|
||||
// the value represented by a group of 5 characters should never be greater than 2^32 - 1
|
||||
$tuple += (($char - 33) * $pow85[$group_pos]);
|
||||
if (4 == $group_pos) {
|
||||
// The following if-clauses are an attempt to fix/suppress the following deprecation warning:
|
||||
// chr(): Providing a value not in-between 0 and 255 is deprecated, this is because a byte value
|
||||
// must be in the [0, 255] interval. The value used will be constrained using % 256
|
||||
// I know this is ugly and there might be more fancier ways. If you know one, feel free to provide a pull request.
|
||||
if (255 < $tuple >> 8) {
|
||||
$chr8Part = \chr(($tuple >> 8) % 256);
|
||||
} else {
|
||||
$chr8Part = \chr($tuple >> 8);
|
||||
}
|
||||
|
||||
if (255 < $tuple >> 16) {
|
||||
$chr16Part = \chr(($tuple >> 16) % 256);
|
||||
} else {
|
||||
$chr16Part = \chr($tuple >> 16);
|
||||
}
|
||||
|
||||
if (255 < $tuple >> 24) {
|
||||
$chr24Part = \chr(($tuple >> 24) % 256);
|
||||
} else {
|
||||
$chr24Part = \chr(($tuple >> 24) & 0xFF);
|
||||
}
|
||||
|
||||
if (255 < $tuple) {
|
||||
$chrTuple = \chr($tuple % 256);
|
||||
} else {
|
||||
$chrTuple = \chr($tuple);
|
||||
}
|
||||
|
||||
$decoded .= $chr24Part . $chr16Part . $chr8Part . $chrTuple;
|
||||
$tuple = 0;
|
||||
$group_pos = 0;
|
||||
} else {
|
||||
++$group_pos;
|
||||
}
|
||||
}
|
||||
}
|
||||
if ($group_pos > 1) {
|
||||
$tuple += $pow85[$group_pos - 1];
|
||||
}
|
||||
// last tuple (if any)
|
||||
switch ($group_pos) {
|
||||
case 4:
|
||||
$decoded .= \chr(($tuple >> 24) & 0xFF).\chr(($tuple >> 16) & 0xFF).\chr(($tuple >> 8) & 0xFF);
|
||||
break;
|
||||
|
||||
case 3:
|
||||
$decoded .= \chr(($tuple >> 24) & 0xFF).\chr(($tuple >> 16) & 0xFF);
|
||||
break;
|
||||
|
||||
case 2:
|
||||
$decoded .= \chr(($tuple >> 24) & 0xFF);
|
||||
break;
|
||||
|
||||
case 1:
|
||||
throw new \Exception('decodeFilterASCII85Decode: invalid code');
|
||||
}
|
||||
|
||||
return $decoded;
|
||||
}
|
||||
|
||||
/**
|
||||
* FlateDecode
|
||||
*
|
||||
* Decompresses data encoded using the zlib/deflate compression method, reproducing the original text or binary data.
|
||||
*
|
||||
* @param string $data Data to decode
|
||||
* @param int $decodeMemoryLimit Memory limit on deflation
|
||||
*
|
||||
* @return string data string
|
||||
*
|
||||
* @throws \Exception
|
||||
*/
|
||||
protected function decodeFilterFlateDecode(string $data, int $decodeMemoryLimit): ?string
|
||||
{
|
||||
// Uncatchable E_WARNING for "data error" is @ suppressed
|
||||
// so execution may proceed with an alternate decompression
|
||||
// method.
|
||||
$decoded = @gzuncompress($data, $decodeMemoryLimit);
|
||||
|
||||
if (false === $decoded) {
|
||||
// If gzuncompress() failed, try again using the compress.zlib://
|
||||
// wrapper to decode it in a file-based context.
|
||||
// See: https://www.php.net/manual/en/function.gzuncompress.php#79042
|
||||
// Issue: https://github.com/smalot/pdfparser/issues/592
|
||||
$ztmp = tmpfile();
|
||||
if (false != $ztmp) {
|
||||
fwrite($ztmp, "\x1f\x8b\x08\x00\x00\x00\x00\x00".$data);
|
||||
$file = stream_get_meta_data($ztmp)['uri'];
|
||||
if (0 === $decodeMemoryLimit) {
|
||||
$decoded = file_get_contents('compress.zlib://'.$file);
|
||||
} else {
|
||||
$decoded = file_get_contents('compress.zlib://'.$file, false, null, 0, $decodeMemoryLimit);
|
||||
}
|
||||
fclose($ztmp);
|
||||
}
|
||||
}
|
||||
|
||||
if (false === \is_string($decoded) || '' === $decoded) {
|
||||
// If the decoded string is empty, that means decoding failed.
|
||||
throw new \Exception('decodeFilterFlateDecode: invalid data');
|
||||
}
|
||||
|
||||
return $decoded;
|
||||
}
|
||||
|
||||
/**
|
||||
* LZWDecode
|
||||
*
|
||||
* Decompresses data encoded using the LZW (Lempel-Ziv-Welch) adaptive compression method, reproducing the original text or binary data.
|
||||
*
|
||||
* @param string $data Data to decode
|
||||
*
|
||||
* @return string Data string
|
||||
*/
|
||||
protected function decodeFilterLZWDecode(string $data): string
|
||||
{
|
||||
// initialize string to return
|
||||
$decoded = '';
|
||||
// data length
|
||||
$data_length = \strlen($data);
|
||||
// convert string to binary string
|
||||
$bitstring = '';
|
||||
for ($i = 0; $i < $data_length; ++$i) {
|
||||
$bitstring .= \sprintf('%08b', \ord($data[$i]));
|
||||
}
|
||||
// get the number of bits
|
||||
$data_length = \strlen($bitstring);
|
||||
// initialize code length in bits
|
||||
$bitlen = 9;
|
||||
// initialize dictionary index
|
||||
$dix = 258;
|
||||
// initialize the dictionary (with the first 256 entries).
|
||||
$dictionary = [];
|
||||
for ($i = 0; $i < 256; ++$i) {
|
||||
$dictionary[$i] = \chr($i);
|
||||
}
|
||||
// previous val
|
||||
$prev_index = 0;
|
||||
// while we encounter EOD marker (257), read code_length bits
|
||||
while (($data_length > 0) && (257 != ($index = bindec(substr($bitstring, 0, $bitlen))))) {
|
||||
// remove read bits from string
|
||||
$bitstring = substr($bitstring, $bitlen);
|
||||
// update number of bits
|
||||
$data_length -= $bitlen;
|
||||
if (256 == $index) { // clear-table marker
|
||||
// reset code length in bits
|
||||
$bitlen = 9;
|
||||
// reset dictionary index
|
||||
$dix = 258;
|
||||
$prev_index = 256;
|
||||
// reset the dictionary (with the first 256 entries).
|
||||
$dictionary = [];
|
||||
for ($i = 0; $i < 256; ++$i) {
|
||||
$dictionary[$i] = \chr($i);
|
||||
}
|
||||
} elseif (256 == $prev_index) {
|
||||
// first entry
|
||||
$decoded .= $dictionary[$index];
|
||||
$prev_index = $index;
|
||||
} else {
|
||||
// check if index exist in the dictionary
|
||||
if ($index < $dix) {
|
||||
// index exist on dictionary
|
||||
$decoded .= $dictionary[$index];
|
||||
$dic_val = $dictionary[$prev_index].$dictionary[$index][0];
|
||||
// store current index
|
||||
$prev_index = $index;
|
||||
} else {
|
||||
// index do not exist on dictionary
|
||||
$dic_val = $dictionary[$prev_index].$dictionary[$prev_index][0];
|
||||
$decoded .= $dic_val;
|
||||
}
|
||||
// update dictionary
|
||||
$dictionary[$dix] = $dic_val;
|
||||
++$dix;
|
||||
// change bit length by case
|
||||
if (2047 == $dix) {
|
||||
$bitlen = 12;
|
||||
} elseif (1023 == $dix) {
|
||||
$bitlen = 11;
|
||||
} elseif (511 == $dix) {
|
||||
$bitlen = 10;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
return $decoded;
|
||||
}
|
||||
|
||||
/**
|
||||
* RunLengthDecode
|
||||
*
|
||||
* Decompresses data encoded using a byte-oriented run-length encoding algorithm.
|
||||
*
|
||||
* @param string $data Data to decode
|
||||
*/
|
||||
protected function decodeFilterRunLengthDecode(string $data): string
|
||||
{
|
||||
// initialize string to return
|
||||
$decoded = '';
|
||||
// data length
|
||||
$data_length = \strlen($data);
|
||||
$i = 0;
|
||||
while ($i < $data_length) {
|
||||
// get current byte value
|
||||
$byte = \ord($data[$i]);
|
||||
if (128 == $byte) {
|
||||
// a length value of 128 denote EOD
|
||||
break;
|
||||
} elseif ($byte < 128) {
|
||||
// if the length byte is in the range 0 to 127
|
||||
// the following length + 1 (1 to 128) bytes shall be copied literally during decompression
|
||||
$decoded .= substr($data, $i + 1, $byte + 1);
|
||||
// move to next block
|
||||
$i += ($byte + 2);
|
||||
} else {
|
||||
// if length is in the range 129 to 255,
|
||||
// the following single byte shall be copied 257 - length (2 to 128) times during decompression
|
||||
$decoded .= str_repeat($data[$i + 1], 257 - $byte);
|
||||
// move to next block
|
||||
$i += 2;
|
||||
}
|
||||
}
|
||||
|
||||
return $decoded;
|
||||
}
|
||||
|
||||
/**
|
||||
* @return array list of available filters
|
||||
*/
|
||||
public function getAvailableFilters(): array
|
||||
{
|
||||
return $this->availableFilters;
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,990 @@
|
||||
<?php
|
||||
|
||||
/**
|
||||
* This file is based on code of tecnickcom/TCPDF PDF library.
|
||||
*
|
||||
* Original author Nicola Asuni (info@tecnick.com) and
|
||||
* contributors (https://github.com/tecnickcom/TCPDF/graphs/contributors).
|
||||
*
|
||||
* @see https://github.com/tecnickcom/TCPDF
|
||||
*
|
||||
* Original code was licensed on the terms of the LGPL v3.
|
||||
*
|
||||
* ------------------------------------------------------------------------------
|
||||
*
|
||||
* @file This file is part of the PdfParser library.
|
||||
*
|
||||
* @author Konrad Abicht <k.abicht@gmail.com>
|
||||
*
|
||||
* @date 2020-01-06
|
||||
*
|
||||
* @license LGPLv3
|
||||
*
|
||||
* @url <https://github.com/smalot/pdfparser>
|
||||
*
|
||||
* PdfParser is a pdf library written in PHP, extraction oriented.
|
||||
* Copyright (C) 2017 - Sébastien MALOT <sebastien@malot.fr>
|
||||
*
|
||||
* This program is free software: you can redistribute it and/or modify
|
||||
* it under the terms of the GNU Lesser General Public License as published by
|
||||
* the Free Software Foundation, either version 3 of the License, or
|
||||
* (at your option) any later version.
|
||||
*
|
||||
* This program is distributed in the hope that it will be useful,
|
||||
* but WITHOUT ANY WARRANTY; without even the implied warranty of
|
||||
* MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
|
||||
* GNU Lesser General Public License for more details.
|
||||
*
|
||||
* You should have received a copy of the GNU Lesser General Public License
|
||||
* along with this program.
|
||||
* If not, see <http://www.pdfparser.org/sites/default/LICENSE.txt>.
|
||||
*/
|
||||
|
||||
namespace Smalot\PdfParser\RawData;
|
||||
|
||||
use Smalot\PdfParser\Config;
|
||||
use Smalot\PdfParser\Exception\EmptyPdfException;
|
||||
use Smalot\PdfParser\Exception\MissingPdfHeaderException;
|
||||
|
||||
class RawDataParser
|
||||
{
|
||||
/**
|
||||
* @var Config
|
||||
*/
|
||||
private $config;
|
||||
|
||||
/**
|
||||
* Configuration array.
|
||||
*
|
||||
* @var array<string,bool>
|
||||
*/
|
||||
protected $cfg = [
|
||||
// if `true` ignore filter decoding errors
|
||||
'ignore_filter_decoding_errors' => true,
|
||||
// if `true` ignore missing filter decoding errors
|
||||
'ignore_missing_filter_decoders' => true,
|
||||
];
|
||||
|
||||
protected $filterHelper;
|
||||
protected $objects;
|
||||
|
||||
/**
|
||||
* @param array $cfg Configuration array, default is []
|
||||
*/
|
||||
public function __construct($cfg = [], ?Config $config = null)
|
||||
{
|
||||
// merge given array with default values
|
||||
$this->cfg = array_merge($this->cfg, $cfg);
|
||||
|
||||
$this->filterHelper = new FilterHelper();
|
||||
$this->config = $config ?: new Config();
|
||||
}
|
||||
|
||||
/**
|
||||
* Decode the specified stream.
|
||||
*
|
||||
* @param string $pdfData PDF data
|
||||
* @param array $sdic Stream's dictionary array
|
||||
* @param string $stream Stream to decode
|
||||
*
|
||||
* @return array containing decoded stream data and remaining filters
|
||||
*
|
||||
* @throws \Exception
|
||||
*/
|
||||
protected function decodeStream(string $pdfData, array $xref, array $sdic, string $stream): array
|
||||
{
|
||||
// get stream length and filters
|
||||
$slength = \strlen($stream);
|
||||
if ($slength <= 0) {
|
||||
return ['', []];
|
||||
}
|
||||
$filters = [];
|
||||
foreach ($sdic as $k => $v) {
|
||||
if ('/' == $v[0]) {
|
||||
if (('Length' == $v[1]) && (isset($sdic[$k + 1])) && ('numeric' == $sdic[$k + 1][0])) {
|
||||
// get declared stream length
|
||||
$declength = (int) $sdic[$k + 1][1];
|
||||
if ($declength < $slength) {
|
||||
$stream = substr($stream, 0, $declength);
|
||||
$slength = $declength;
|
||||
}
|
||||
} elseif (('Filter' == $v[1]) && (isset($sdic[$k + 1]))) {
|
||||
// resolve indirect object
|
||||
$objval = $this->getObjectVal($pdfData, $xref, $sdic[$k + 1]);
|
||||
if ('/' == $objval[0]) {
|
||||
// single filter
|
||||
$filters[] = $objval[1];
|
||||
} elseif ('[' == $objval[0]) {
|
||||
// array of filters
|
||||
foreach ($objval[1] as $flt) {
|
||||
if ('/' == $flt[0]) {
|
||||
$filters[] = $flt[1];
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// decode the stream
|
||||
$remaining_filters = [];
|
||||
foreach ($filters as $filter) {
|
||||
if (\in_array($filter, $this->filterHelper->getAvailableFilters(), true)) {
|
||||
try {
|
||||
$stream = $this->filterHelper->decodeFilter($filter, $stream, $this->config->getDecodeMemoryLimit());
|
||||
} catch (\Exception $e) {
|
||||
$emsg = $e->getMessage();
|
||||
if ((('~' == $emsg[0]) && !$this->cfg['ignore_missing_filter_decoders'])
|
||||
|| (('~' != $emsg[0]) && !$this->cfg['ignore_filter_decoding_errors'])
|
||||
) {
|
||||
throw new \Exception($e->getMessage());
|
||||
}
|
||||
}
|
||||
} else {
|
||||
// add missing filter to array
|
||||
$remaining_filters[] = $filter;
|
||||
}
|
||||
}
|
||||
|
||||
return [$stream, $remaining_filters];
|
||||
}
|
||||
|
||||
/**
|
||||
* Decode the Cross-Reference section
|
||||
*
|
||||
* @param string $pdfData PDF data
|
||||
* @param int $startxref Offset at which the xref section starts (position of the 'xref' keyword)
|
||||
* @param array $xref Previous xref array (if any)
|
||||
* @param array<int> $visitedOffsets Array of visited offsets to prevent infinite loops
|
||||
*
|
||||
* @return array containing xref and trailer data
|
||||
*
|
||||
* @throws \Exception
|
||||
*/
|
||||
protected function decodeXref(string $pdfData, int $startxref, array $xref = [], array $visitedOffsets = []): array
|
||||
{
|
||||
$startxref += 4; // 4 is the length of the word 'xref'
|
||||
// skip initial white space chars
|
||||
$offset = $startxref + strspn($pdfData, $this->config->getPdfWhitespaces(), $startxref);
|
||||
// initialize object number
|
||||
$obj_num = 0;
|
||||
// search for cross-reference entries or subsection
|
||||
while (preg_match('/([0-9]+)[\x20]([0-9]+)[\x20]?([nf]?)(\r\n|[\x20]?[\r\n])/', $pdfData, $matches, \PREG_OFFSET_CAPTURE, $offset) > 0) {
|
||||
if ($matches[0][1] != $offset) {
|
||||
// we are on another section
|
||||
break;
|
||||
}
|
||||
$offset += \strlen($matches[0][0]);
|
||||
if ('n' == $matches[3][0]) {
|
||||
// create unique object index: [object number]_[generation number]
|
||||
$index = $obj_num.'_'.(int) $matches[2][0];
|
||||
// check if object already exist
|
||||
if (!isset($xref['xref'][$index])) {
|
||||
// store object offset position
|
||||
$xref['xref'][$index] = (int) $matches[1][0];
|
||||
}
|
||||
++$obj_num;
|
||||
} elseif ('f' == $matches[3][0]) {
|
||||
++$obj_num;
|
||||
} else {
|
||||
// object number (index)
|
||||
$obj_num = (int) $matches[1][0];
|
||||
}
|
||||
}
|
||||
// get trailer data
|
||||
if (preg_match('/trailer[\s]*<<(.*)>>/isU', $pdfData, $matches, \PREG_OFFSET_CAPTURE, $offset) > 0) {
|
||||
$trailer_data = $matches[1][0];
|
||||
if (!isset($xref['trailer']) || empty($xref['trailer'])) {
|
||||
// get only the last updated version
|
||||
$xref['trailer'] = [];
|
||||
// parse trailer_data
|
||||
if (preg_match('/Size[\s]+([0-9]+)/i', $trailer_data, $matches) > 0) {
|
||||
$xref['trailer']['size'] = (int) $matches[1];
|
||||
}
|
||||
if (preg_match('/Root[\s]+([0-9]+)[\s]+([0-9]+)[\s]+R/i', $trailer_data, $matches) > 0) {
|
||||
$xref['trailer']['root'] = (int) $matches[1].'_'.(int) $matches[2];
|
||||
}
|
||||
if (preg_match('/Encrypt[\s]+([0-9]+)[\s]+([0-9]+)[\s]+R/i', $trailer_data, $matches) > 0) {
|
||||
$xref['trailer']['encrypt'] = (int) $matches[1].'_'.(int) $matches[2];
|
||||
}
|
||||
if (preg_match('/Info[\s]+([0-9]+)[\s]+([0-9]+)[\s]+R/i', $trailer_data, $matches) > 0) {
|
||||
$xref['trailer']['info'] = (int) $matches[1].'_'.(int) $matches[2];
|
||||
}
|
||||
if (preg_match('/ID[\s]*[\[][\s]*[<]([^>]*)[>][\s]*[<]([^>]*)[>]/i', $trailer_data, $matches) > 0) {
|
||||
$xref['trailer']['id'] = [];
|
||||
$xref['trailer']['id'][0] = $matches[1];
|
||||
$xref['trailer']['id'][1] = $matches[2];
|
||||
}
|
||||
}
|
||||
if (preg_match('/Prev[\s]+([0-9]+)/i', $trailer_data, $matches) > 0) {
|
||||
$offset = (int) $matches[1];
|
||||
if (0 != $offset) {
|
||||
// get previous xref
|
||||
$xref = $this->getXrefData($pdfData, $offset, $xref, $visitedOffsets);
|
||||
}
|
||||
}
|
||||
} else {
|
||||
throw new \Exception('Unable to find trailer');
|
||||
}
|
||||
|
||||
return $xref;
|
||||
}
|
||||
|
||||
/**
|
||||
* Decode the Cross-Reference Stream section
|
||||
*
|
||||
* @param string $pdfData PDF data
|
||||
* @param int $startxref Offset at which the xref section starts
|
||||
* @param array $xref Previous xref array (if any)
|
||||
* @param array<int> $visitedOffsets Array of visited offsets to prevent infinite loops
|
||||
*
|
||||
* @return array containing xref and trailer data
|
||||
*
|
||||
* @throws \Exception if unknown PNG predictor detected
|
||||
*/
|
||||
protected function decodeXrefStream(string $pdfData, int $startxref, array $xref = [], array $visitedOffsets = []): array
|
||||
{
|
||||
// try to read Cross-Reference Stream
|
||||
$xrefobj = $this->getRawObject($pdfData, $startxref);
|
||||
$xrefcrs = $this->getIndirectObject($pdfData, $xref, $xrefobj[1], $startxref, true);
|
||||
if (!isset($xref['trailer']) || empty($xref['trailer'])) {
|
||||
// get only the last updated version
|
||||
$xref['trailer'] = [];
|
||||
$filltrailer = true;
|
||||
} else {
|
||||
$filltrailer = false;
|
||||
}
|
||||
if (!isset($xref['xref'])) {
|
||||
$xref['xref'] = [];
|
||||
}
|
||||
$valid_crs = false;
|
||||
$columns = 0;
|
||||
$predictor = null;
|
||||
$sarr = $xrefcrs[0][1];
|
||||
if (!\is_array($sarr)) {
|
||||
$sarr = [];
|
||||
}
|
||||
|
||||
$wb = [];
|
||||
|
||||
foreach ($sarr as $k => $v) {
|
||||
if (
|
||||
('/' == $v[0])
|
||||
&& ('Type' == $v[1])
|
||||
&& (
|
||||
isset($sarr[$k + 1])
|
||||
&& '/' == $sarr[$k + 1][0]
|
||||
&& 'XRef' == $sarr[$k + 1][1]
|
||||
)
|
||||
) {
|
||||
$valid_crs = true;
|
||||
} elseif (('/' == $v[0]) && ('Index' == $v[1]) && (isset($sarr[$k + 1]))) {
|
||||
// initialize list for: first object number in the subsection / number of objects
|
||||
$index_blocks = [];
|
||||
for ($m = 0; $m < \count($sarr[$k + 1][1]); $m += 2) {
|
||||
$index_blocks[] = [$sarr[$k + 1][1][$m][1], $sarr[$k + 1][1][$m + 1][1]];
|
||||
}
|
||||
} elseif (('/' == $v[0]) && ('Prev' == $v[1]) && (isset($sarr[$k + 1]) && ('numeric' == $sarr[$k + 1][0]))) {
|
||||
// get previous xref offset
|
||||
$prevxref = (int) $sarr[$k + 1][1];
|
||||
} elseif (('/' == $v[0]) && ('W' == $v[1]) && (isset($sarr[$k + 1]))) {
|
||||
// number of bytes (in the decoded stream) of the corresponding field
|
||||
$wb[0] = (int) $sarr[$k + 1][1][0][1];
|
||||
$wb[1] = (int) $sarr[$k + 1][1][1][1];
|
||||
$wb[2] = (int) $sarr[$k + 1][1][2][1];
|
||||
} elseif (('/' == $v[0]) && ('DecodeParms' == $v[1]) && (isset($sarr[$k + 1][1]))) {
|
||||
$decpar = $sarr[$k + 1][1];
|
||||
foreach ($decpar as $kdc => $vdc) {
|
||||
if (
|
||||
'/' == $vdc[0]
|
||||
&& 'Columns' == $vdc[1]
|
||||
&& (
|
||||
isset($decpar[$kdc + 1])
|
||||
&& 'numeric' == $decpar[$kdc + 1][0]
|
||||
)
|
||||
) {
|
||||
$columns = (int) $decpar[$kdc + 1][1];
|
||||
} elseif (
|
||||
'/' == $vdc[0]
|
||||
&& 'Predictor' == $vdc[1]
|
||||
&& (
|
||||
isset($decpar[$kdc + 1])
|
||||
&& 'numeric' == $decpar[$kdc + 1][0]
|
||||
)
|
||||
) {
|
||||
$predictor = (int) $decpar[$kdc + 1][1];
|
||||
}
|
||||
}
|
||||
} elseif ($filltrailer) {
|
||||
if (('/' == $v[0]) && ('Size' == $v[1]) && (isset($sarr[$k + 1]) && ('numeric' == $sarr[$k + 1][0]))) {
|
||||
$xref['trailer']['size'] = $sarr[$k + 1][1];
|
||||
} elseif (('/' == $v[0]) && ('Root' == $v[1]) && (isset($sarr[$k + 1]) && ('objref' == $sarr[$k + 1][0]))) {
|
||||
$xref['trailer']['root'] = $sarr[$k + 1][1];
|
||||
} elseif (('/' == $v[0]) && ('Info' == $v[1]) && (isset($sarr[$k + 1]) && ('objref' == $sarr[$k + 1][0]))) {
|
||||
$xref['trailer']['info'] = $sarr[$k + 1][1];
|
||||
} elseif (('/' == $v[0]) && ('Encrypt' == $v[1]) && (isset($sarr[$k + 1]) && ('objref' == $sarr[$k + 1][0]))) {
|
||||
$xref['trailer']['encrypt'] = $sarr[$k + 1][1];
|
||||
} elseif (('/' == $v[0]) && ('ID' == $v[1]) && (isset($sarr[$k + 1]))) {
|
||||
$xref['trailer']['id'] = [];
|
||||
$xref['trailer']['id'][0] = $sarr[$k + 1][1][0][1];
|
||||
$xref['trailer']['id'][1] = $sarr[$k + 1][1][1][1];
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// decode data
|
||||
if ($valid_crs && isset($xrefcrs[1][3][0])) {
|
||||
if (null !== $predictor) {
|
||||
// number of bytes in a row
|
||||
$rowlen = ($columns + 1);
|
||||
// convert the stream into an array of integers
|
||||
/** @var array<int> */
|
||||
$sdata = unpack('C*', $xrefcrs[1][3][0]);
|
||||
// TODO: Handle the case when unpack returns false
|
||||
|
||||
// split the rows
|
||||
$sdata = array_chunk($sdata, $rowlen);
|
||||
|
||||
// initialize decoded array
|
||||
$ddata = [];
|
||||
// initialize first row with zeros
|
||||
$prev_row = array_fill(0, $rowlen, 0);
|
||||
// for each row apply PNG unpredictor
|
||||
foreach ($sdata as $k => $row) {
|
||||
// initialize new row
|
||||
$ddata[$k] = [];
|
||||
// get PNG predictor value
|
||||
$predictor = (10 + $row[0]);
|
||||
// for each byte on the row
|
||||
for ($i = 1; $i <= $columns; ++$i) {
|
||||
// new index
|
||||
$j = ($i - 1);
|
||||
$row_up = $prev_row[$j];
|
||||
if (1 == $i) {
|
||||
$row_left = 0;
|
||||
$row_upleft = 0;
|
||||
} else {
|
||||
$row_left = $row[$i - 1];
|
||||
$row_upleft = $prev_row[$j - 1];
|
||||
}
|
||||
switch ($predictor) {
|
||||
case 10: // PNG prediction (on encoding, PNG None on all rows)
|
||||
$ddata[$k][$j] = $row[$i];
|
||||
break;
|
||||
|
||||
case 11: // PNG prediction (on encoding, PNG Sub on all rows)
|
||||
$ddata[$k][$j] = (($row[$i] + $row_left) & 0xFF);
|
||||
break;
|
||||
|
||||
case 12: // PNG prediction (on encoding, PNG Up on all rows)
|
||||
$ddata[$k][$j] = (($row[$i] + $row_up) & 0xFF);
|
||||
break;
|
||||
|
||||
case 13: // PNG prediction (on encoding, PNG Average on all rows)
|
||||
$ddata[$k][$j] = (($row[$i] + (($row_left + $row_up) / 2)) & 0xFF);
|
||||
break;
|
||||
|
||||
case 14: // PNG prediction (on encoding, PNG Paeth on all rows)
|
||||
// initial estimate
|
||||
$p = ($row_left + $row_up - $row_upleft);
|
||||
// distances
|
||||
$pa = abs($p - $row_left);
|
||||
$pb = abs($p - $row_up);
|
||||
$pc = abs($p - $row_upleft);
|
||||
$pmin = min($pa, $pb, $pc);
|
||||
// return minimum distance
|
||||
switch ($pmin) {
|
||||
case $pa:
|
||||
$ddata[$k][$j] = (($row[$i] + $row_left) & 0xFF);
|
||||
break;
|
||||
|
||||
case $pb:
|
||||
$ddata[$k][$j] = (($row[$i] + $row_up) & 0xFF);
|
||||
break;
|
||||
|
||||
case $pc:
|
||||
$ddata[$k][$j] = (($row[$i] + $row_upleft) & 0xFF);
|
||||
break;
|
||||
}
|
||||
break;
|
||||
|
||||
default: // PNG prediction (on encoding, PNG optimum)
|
||||
throw new \Exception('Unknown PNG predictor: '.$predictor);
|
||||
}
|
||||
}
|
||||
$prev_row = $ddata[$k];
|
||||
} // end for each row
|
||||
// complete decoding
|
||||
} else {
|
||||
// number of bytes in a row
|
||||
$rowlen = array_sum($wb);
|
||||
if (0 < $rowlen) {
|
||||
// convert the stream into an array of integers
|
||||
$sdata = unpack('C*', $xrefcrs[1][3][0]);
|
||||
// split the rows
|
||||
$ddata = array_chunk($sdata, $rowlen);
|
||||
} else {
|
||||
// if the row length is zero, $ddata should be an empty array as well
|
||||
$ddata = [];
|
||||
}
|
||||
}
|
||||
|
||||
$sdata = [];
|
||||
|
||||
// for every row
|
||||
foreach ($ddata as $k => $row) {
|
||||
// initialize new row
|
||||
$sdata[$k] = [0, 0, 0];
|
||||
if (0 == $wb[0]) {
|
||||
// default type field
|
||||
$sdata[$k][0] = 1;
|
||||
}
|
||||
$i = 0; // count bytes in the row
|
||||
// for every column
|
||||
for ($c = 0; $c < 3; ++$c) {
|
||||
// for every byte on the column
|
||||
for ($b = 0; $b < $wb[$c]; ++$b) {
|
||||
if (isset($row[$i])) {
|
||||
$sdata[$k][$c] += ($row[$i] << (($wb[$c] - 1 - $b) * 8));
|
||||
}
|
||||
++$i;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// fill xref
|
||||
if (isset($index_blocks)) {
|
||||
// load the first object number of the first /Index entry
|
||||
$obj_num = $index_blocks[0][0];
|
||||
} else {
|
||||
$obj_num = 0;
|
||||
}
|
||||
foreach ($sdata as $k => $row) {
|
||||
switch ($row[0]) {
|
||||
case 0: // (f) linked list of free objects
|
||||
break;
|
||||
|
||||
case 1: // (n) objects that are in use but are not compressed
|
||||
// create unique object index: [object number]_[generation number]
|
||||
$index = $obj_num.'_'.$row[2];
|
||||
// check if object already exist
|
||||
if (!isset($xref['xref'][$index])) {
|
||||
// store object offset position
|
||||
$xref['xref'][$index] = $row[1];
|
||||
}
|
||||
break;
|
||||
|
||||
case 2: // compressed objects
|
||||
// $row[1] = object number of the object stream in which this object is stored
|
||||
// $row[2] = index of this object within the object stream
|
||||
$index = $row[1].'_0_'.$row[2];
|
||||
$xref['xref'][$index] = -1;
|
||||
break;
|
||||
|
||||
default: // null objects
|
||||
break;
|
||||
}
|
||||
++$obj_num;
|
||||
if (isset($index_blocks)) {
|
||||
// reduce the number of remaining objects
|
||||
--$index_blocks[0][1];
|
||||
if (0 == $index_blocks[0][1]) {
|
||||
// remove the actual used /Index entry
|
||||
array_shift($index_blocks);
|
||||
if (0 < \count($index_blocks)) {
|
||||
// load the first object number of the following /Index entry
|
||||
$obj_num = $index_blocks[0][0];
|
||||
} else {
|
||||
// if there are no more entries, remove $index_blocks to avoid actions on an empty array
|
||||
unset($index_blocks);
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
} // end decoding data
|
||||
if (isset($prevxref)) {
|
||||
// get previous xref
|
||||
$xref = $this->getXrefData($pdfData, $prevxref, $xref, $visitedOffsets);
|
||||
}
|
||||
|
||||
return $xref;
|
||||
}
|
||||
|
||||
protected function getObjectHeaderPattern(array $objRefs): string
|
||||
{
|
||||
// consider all whitespace character (PDF specifications)
|
||||
return '/'.$objRefs[0].$this->config->getPdfWhitespacesRegex().$objRefs[1].$this->config->getPdfWhitespacesRegex().'obj/';
|
||||
}
|
||||
|
||||
protected function getObjectHeaderLen(array $objRefs): int
|
||||
{
|
||||
// "4 0 obj"
|
||||
// 2 whitespaces + strlen("obj") = 5
|
||||
return 5 + \strlen($objRefs[0]) + \strlen($objRefs[1]);
|
||||
}
|
||||
|
||||
/**
|
||||
* Get content of indirect object.
|
||||
*
|
||||
* @param string $pdfData PDF data
|
||||
* @param string $objRef Object number and generation number separated by underscore character
|
||||
* @param int $offset Object offset
|
||||
* @param bool $decoding If true decode streams
|
||||
*
|
||||
* @return array containing object data
|
||||
*
|
||||
* @throws \Exception if invalid object reference found
|
||||
*/
|
||||
protected function getIndirectObject(string $pdfData, array $xref, string $objRef, int $offset = 0, bool $decoding = true): array
|
||||
{
|
||||
/*
|
||||
* build indirect object header
|
||||
*/
|
||||
// $objHeader = "[object number] [generation number] obj"
|
||||
$objRefArr = explode('_', $objRef);
|
||||
if (2 !== \count($objRefArr)) {
|
||||
throw new \Exception('Invalid object reference for $obj.');
|
||||
}
|
||||
|
||||
$objHeaderLen = $this->getObjectHeaderLen($objRefArr);
|
||||
|
||||
/*
|
||||
* check if we are in position
|
||||
*/
|
||||
// ignore whitespace characters at offset
|
||||
$offset += strspn($pdfData, $this->config->getPdfWhitespaces(), $offset);
|
||||
// ignore leading zeros for object number
|
||||
$offset += strspn($pdfData, '0', $offset);
|
||||
if (0 == preg_match($this->getObjectHeaderPattern($objRefArr), substr($pdfData, $offset, $objHeaderLen))) {
|
||||
// an indirect reference to an undefined object shall be considered a reference to the null object
|
||||
return ['null', 'null', $offset];
|
||||
}
|
||||
|
||||
/*
|
||||
* get content
|
||||
*/
|
||||
// starting position of object content
|
||||
$offset += $objHeaderLen;
|
||||
$objContentArr = [];
|
||||
$i = 0; // object main index
|
||||
$header = null;
|
||||
do {
|
||||
$oldOffset = $offset;
|
||||
// get element
|
||||
$element = $this->getRawObject($pdfData, $offset, null != $header ? $header[1] : null);
|
||||
$offset = $element[2];
|
||||
// decode stream using stream's dictionary information
|
||||
if ($decoding && ('stream' === $element[0]) && null != $header) {
|
||||
$element[3] = $this->decodeStream($pdfData, $xref, $header[1], $element[1]);
|
||||
}
|
||||
$objContentArr[$i] = $element;
|
||||
$header = isset($element[0]) && '<<' === $element[0] ? $element : null;
|
||||
++$i;
|
||||
} while (('endobj' !== $element[0]) && ($offset !== $oldOffset));
|
||||
// remove closing delimiter
|
||||
array_pop($objContentArr);
|
||||
|
||||
/*
|
||||
* return raw object content
|
||||
*/
|
||||
return $objContentArr;
|
||||
}
|
||||
|
||||
/**
|
||||
* Get the content of object, resolving indirect object reference if necessary.
|
||||
*
|
||||
* @param string $pdfData PDF data
|
||||
* @param array $obj Object value
|
||||
*
|
||||
* @return array containing object data
|
||||
*
|
||||
* @throws \Exception
|
||||
*/
|
||||
protected function getObjectVal(string $pdfData, $xref, array $obj): array
|
||||
{
|
||||
if ('objref' == $obj[0]) {
|
||||
// reference to indirect object
|
||||
if (isset($this->objects[$obj[1]])) {
|
||||
// this object has been already parsed
|
||||
return $this->objects[$obj[1]];
|
||||
} elseif (isset($xref[$obj[1]])) {
|
||||
// parse new object
|
||||
$this->objects[$obj[1]] = $this->getIndirectObject($pdfData, $xref, $obj[1], $xref[$obj[1]], false);
|
||||
|
||||
return $this->objects[$obj[1]];
|
||||
}
|
||||
}
|
||||
|
||||
return $obj;
|
||||
}
|
||||
|
||||
/**
|
||||
* Get object type, raw value and offset to next object
|
||||
*
|
||||
* @param int $offset Object offset
|
||||
* @param array|null $headerDic obj header's dictionary, parsed by getRawObject. Used for stream parsing optimization
|
||||
*
|
||||
* @return array containing object type, raw value and offset to next object
|
||||
*/
|
||||
protected function getRawObject(string $pdfData, int $offset = 0, ?array $headerDic = null): array
|
||||
{
|
||||
$objtype = ''; // object type to be returned
|
||||
$objval = ''; // object value to be returned
|
||||
|
||||
// skip initial white space chars
|
||||
$offset += strspn($pdfData, $this->config->getPdfWhitespaces(), $offset);
|
||||
|
||||
// get first char
|
||||
$char = $pdfData[$offset];
|
||||
// get object type
|
||||
switch ($char) {
|
||||
case '%': // \x25 PERCENT SIGN
|
||||
// skip comment and search for next token
|
||||
$next = strcspn($pdfData, "\r\n", $offset);
|
||||
if ($next > 0) {
|
||||
$offset += $next;
|
||||
|
||||
return $this->getRawObject($pdfData, $offset);
|
||||
}
|
||||
break;
|
||||
|
||||
case '/': // \x2F SOLIDUS
|
||||
// name object
|
||||
$objtype = $char;
|
||||
++$offset;
|
||||
$span = strcspn($pdfData, "\x00\x09\x0a\x0c\x0d\x20\n\t\r\v\f\x28\x29\x3c\x3e\x5b\x5d\x7b\x7d\x2f\x25", $offset, 256);
|
||||
if ($span > 0) {
|
||||
$objval = substr($pdfData, $offset, $span); // unescaped value
|
||||
$offset += $span;
|
||||
}
|
||||
break;
|
||||
|
||||
case '(': // \x28 LEFT PARENTHESIS
|
||||
case ')': // \x29 RIGHT PARENTHESIS
|
||||
// literal string object
|
||||
$objtype = $char;
|
||||
++$offset;
|
||||
$strpos = $offset;
|
||||
if ('(' == $char) {
|
||||
$open_bracket = 1;
|
||||
while ($open_bracket > 0) {
|
||||
if (!isset($pdfData[$strpos])) {
|
||||
break;
|
||||
}
|
||||
$ch = $pdfData[$strpos];
|
||||
switch ($ch) {
|
||||
case '\\': // REVERSE SOLIDUS (5Ch) (Backslash)
|
||||
// skip next character
|
||||
++$strpos;
|
||||
break;
|
||||
|
||||
case '(': // LEFT PARENHESIS (28h)
|
||||
++$open_bracket;
|
||||
break;
|
||||
|
||||
case ')': // RIGHT PARENTHESIS (29h)
|
||||
--$open_bracket;
|
||||
break;
|
||||
}
|
||||
++$strpos;
|
||||
}
|
||||
$objval = substr($pdfData, $offset, $strpos - $offset - 1);
|
||||
$offset = $strpos;
|
||||
}
|
||||
break;
|
||||
|
||||
case '[': // \x5B LEFT SQUARE BRACKET
|
||||
case ']': // \x5D RIGHT SQUARE BRACKET
|
||||
// array object
|
||||
$objtype = $char;
|
||||
++$offset;
|
||||
if ('[' == $char) {
|
||||
// get array content
|
||||
$objval = [];
|
||||
do {
|
||||
$oldOffset = $offset;
|
||||
// get element
|
||||
$element = $this->getRawObject($pdfData, $offset);
|
||||
$offset = $element[2];
|
||||
$objval[] = $element;
|
||||
} while ((']' != $element[0]) && ($offset != $oldOffset));
|
||||
// remove closing delimiter
|
||||
array_pop($objval);
|
||||
}
|
||||
break;
|
||||
|
||||
case '<': // \x3C LESS-THAN SIGN
|
||||
case '>': // \x3E GREATER-THAN SIGN
|
||||
if (isset($pdfData[$offset + 1]) && ($pdfData[$offset + 1] == $char)) {
|
||||
// dictionary object
|
||||
$objtype = $char.$char;
|
||||
$offset += 2;
|
||||
if ('<' == $char) {
|
||||
// get array content
|
||||
$objval = [];
|
||||
do {
|
||||
$oldOffset = $offset;
|
||||
// get element
|
||||
$element = $this->getRawObject($pdfData, $offset);
|
||||
$offset = $element[2];
|
||||
$objval[] = $element;
|
||||
} while (('>>' != $element[0]) && ($offset != $oldOffset));
|
||||
// remove closing delimiter
|
||||
array_pop($objval);
|
||||
}
|
||||
} else {
|
||||
// hexadecimal string object
|
||||
$objtype = $char;
|
||||
++$offset;
|
||||
|
||||
$span = strspn($pdfData, "0123456789abcdefABCDEF\x09\x0a\x0c\x0d\x20", $offset);
|
||||
$dataToCheck = $pdfData[$offset + $span] ?? null;
|
||||
if ('<' == $char && $span > 0 && '>' == $dataToCheck) {
|
||||
// remove white space characters
|
||||
$objval = strtr(substr($pdfData, $offset, $span), $this->config->getPdfWhitespaces(), '');
|
||||
$offset += $span + 1;
|
||||
} elseif (false !== ($endpos = strpos($pdfData, '>', $offset))) {
|
||||
$offset = $endpos + 1;
|
||||
}
|
||||
}
|
||||
break;
|
||||
|
||||
default:
|
||||
if ('endobj' == substr($pdfData, $offset, 6)) {
|
||||
// indirect object
|
||||
$objtype = 'endobj';
|
||||
$offset += 6;
|
||||
} elseif ('null' == substr($pdfData, $offset, 4)) {
|
||||
// null object
|
||||
$objtype = 'null';
|
||||
$offset += 4;
|
||||
$objval = 'null';
|
||||
} elseif ('true' == substr($pdfData, $offset, 4)) {
|
||||
// boolean true object
|
||||
$objtype = 'boolean';
|
||||
$offset += 4;
|
||||
$objval = 'true';
|
||||
} elseif ('false' == substr($pdfData, $offset, 5)) {
|
||||
// boolean false object
|
||||
$objtype = 'boolean';
|
||||
$offset += 5;
|
||||
$objval = 'false';
|
||||
} elseif ('stream' == substr($pdfData, $offset, 6)) {
|
||||
// start stream object
|
||||
$objtype = 'stream';
|
||||
$offset += 6;
|
||||
if (1 == preg_match('/^( *[\r]?[\n])/isU', substr($pdfData, $offset, 4), $matches)) {
|
||||
$offset += \strlen($matches[0]);
|
||||
|
||||
// we get stream length here to later help preg_match test less data
|
||||
$streamLen = (int) $this->getHeaderValue($headerDic, 'Length', 'numeric', 0);
|
||||
$skip = false === $this->config->getRetainImageContent() && 'XObject' == $this->getHeaderValue($headerDic, 'Type', '/') && 'Image' == $this->getHeaderValue($headerDic, 'Subtype', '/');
|
||||
|
||||
$pregResult = preg_match(
|
||||
'/(endstream)[\x09\x0a\x0c\x0d\x20]/isU',
|
||||
$pdfData,
|
||||
$matches,
|
||||
\PREG_OFFSET_CAPTURE,
|
||||
$offset + $streamLen
|
||||
);
|
||||
|
||||
if (1 == $pregResult) {
|
||||
$objval = $skip ? '' : substr($pdfData, $offset, $matches[0][1] - $offset);
|
||||
$offset = $matches[1][1];
|
||||
}
|
||||
}
|
||||
} elseif ('endstream' == substr($pdfData, $offset, 9)) {
|
||||
// end stream object
|
||||
$objtype = 'endstream';
|
||||
$offset += 9;
|
||||
} elseif (1 == preg_match('/^([0-9]+)[\s]+([0-9]+)[\s]+R/iU', substr($pdfData, $offset, 33), $matches)) {
|
||||
// indirect object reference
|
||||
$objtype = 'objref';
|
||||
$offset += \strlen($matches[0]);
|
||||
$objval = (int) $matches[1].'_'.(int) $matches[2];
|
||||
} elseif (1 == preg_match('/^([0-9]+)[\s]+([0-9]+)[\s]+obj/iU', substr($pdfData, $offset, 33), $matches)) {
|
||||
// object start
|
||||
$objtype = 'obj';
|
||||
$objval = (int) $matches[1].'_'.(int) $matches[2];
|
||||
$offset += \strlen($matches[0]);
|
||||
} elseif (($numlen = strspn($pdfData, '+-.0123456789', $offset)) > 0) {
|
||||
// numeric object
|
||||
$objtype = 'numeric';
|
||||
$objval = substr($pdfData, $offset, $numlen);
|
||||
$offset += $numlen;
|
||||
}
|
||||
break;
|
||||
}
|
||||
|
||||
return [$objtype, $objval, $offset];
|
||||
}
|
||||
|
||||
/**
|
||||
* Get value of an object header's section (obj << YYY >> part ).
|
||||
*
|
||||
* It is similar to Header::get('...')->getContent(), the only difference is it can be used during the parsing process,
|
||||
* when no Smalot\PdfParser\Header objects are created yet.
|
||||
*
|
||||
* @param string $key header's section name
|
||||
* @param string $type type of the section (i.e. 'numeric', '/', '<<', etc.)
|
||||
* @param string|array|null $default default value for header's section
|
||||
*
|
||||
* @return string|array|null value of obj header's section, or default value if none found, or its type doesn't match $type param
|
||||
*/
|
||||
private function getHeaderValue(?array $headerDic, string $key, string $type, $default = '')
|
||||
{
|
||||
if (false === \is_array($headerDic)) {
|
||||
return $default;
|
||||
}
|
||||
|
||||
/*
|
||||
* It recieves dictionary of header fields, as it is returned by RawDataParser::getRawObject,
|
||||
* iterates over it, searching for section of type '/' whith requested key.
|
||||
* If such a section is found, it tries to receive it's value (next object in dictionary),
|
||||
* returning it, if it matches requested type, or default value otherwise.
|
||||
*/
|
||||
foreach ($headerDic as $i => $val) {
|
||||
$isSectionName = \is_array($val) && 3 == \count($val) && '/' == $val[0];
|
||||
if (
|
||||
$isSectionName
|
||||
&& $val[1] == $key
|
||||
&& isset($headerDic[$i + 1])
|
||||
) {
|
||||
$isSectionValue = \is_array($headerDic[$i + 1]) && 1 < \count($headerDic[$i + 1]);
|
||||
|
||||
return $isSectionValue && $type == $headerDic[$i + 1][0]
|
||||
? $headerDic[$i + 1][1]
|
||||
: $default;
|
||||
}
|
||||
}
|
||||
|
||||
return $default;
|
||||
}
|
||||
|
||||
/**
|
||||
* Get Cross-Reference (xref) table and trailer data from PDF document data.
|
||||
*
|
||||
* @param int $offset xref offset (if known)
|
||||
* @param array $xref previous xref array (if any)
|
||||
* @param array<int> $visitedOffsets array of visited offsets to prevent infinite loops
|
||||
*
|
||||
* @return array containing xref and trailer data
|
||||
*
|
||||
* @throws \Exception if it was unable to find startxref
|
||||
* @throws \Exception if it was unable to find xref
|
||||
*/
|
||||
protected function getXrefData(string $pdfData, int $offset = 0, array $xref = [], array $visitedOffsets = []): array
|
||||
{
|
||||
// Check for circular references to prevent infinite loops
|
||||
if (\in_array($offset, $visitedOffsets, true)) {
|
||||
// We've already processed this offset, skip to avoid infinite loop
|
||||
return $xref;
|
||||
}
|
||||
|
||||
// Track this offset as visited
|
||||
$visitedOffsets[] = $offset;
|
||||
// If the $offset is currently pointed at whitespace, bump it
|
||||
// forward until it isn't; affects loosely targetted offsets
|
||||
// for the 'xref' keyword
|
||||
// See: https://github.com/smalot/pdfparser/issues/673
|
||||
$bumpOffset = $offset;
|
||||
while (preg_match('/\s/', substr($pdfData, $bumpOffset, 1))) {
|
||||
++$bumpOffset;
|
||||
}
|
||||
|
||||
// Find all startxref tables from this $offset forward
|
||||
$startxrefPreg = preg_match_all(
|
||||
'/(?<=[\r\n])startxref[\s]*[\r\n]+([0-9]+)[\s]*[\r\n]+%%EOF/i',
|
||||
$pdfData,
|
||||
$startxrefMatches,
|
||||
\PREG_SET_ORDER,
|
||||
$offset
|
||||
);
|
||||
|
||||
if (0 == $startxrefPreg) {
|
||||
// No startxref tables were found
|
||||
throw new \Exception('Unable to find startxref');
|
||||
} elseif (0 == $offset) {
|
||||
// Use the last startxref in the document
|
||||
$startxref = (int) $startxrefMatches[\count($startxrefMatches) - 1][1];
|
||||
} elseif (strpos($pdfData, 'xref', $bumpOffset) == $bumpOffset) {
|
||||
// Already pointing at the xref table
|
||||
$startxref = $bumpOffset;
|
||||
} elseif (preg_match('/([0-9]+[\s][0-9]+[\s]obj)/i', $pdfData, $matches, 0, $bumpOffset)) {
|
||||
// Cross-Reference Stream object
|
||||
$startxref = $bumpOffset;
|
||||
} else {
|
||||
// Use the next startxref from this $offset
|
||||
$startxref = (int) $startxrefMatches[0][1];
|
||||
}
|
||||
|
||||
if ($startxref > \strlen($pdfData)) {
|
||||
throw new \Exception('Unable to find xref (PDF corrupted?)');
|
||||
}
|
||||
|
||||
// check xref position
|
||||
if (strpos($pdfData, 'xref', $startxref) == $startxref) {
|
||||
// Cross-Reference
|
||||
$xref = $this->decodeXref($pdfData, $startxref, $xref, $visitedOffsets);
|
||||
} else {
|
||||
// Check if the $pdfData might have the wrong line-endings
|
||||
$pdfDataUnix = str_replace("\r\n", "\n", $pdfData);
|
||||
if ($startxref < \strlen($pdfDataUnix) && strpos($pdfDataUnix, 'xref', $startxref) == $startxref) {
|
||||
// Return Unix-line-ending flag
|
||||
$xref = ['Unix' => true];
|
||||
} else {
|
||||
// Cross-Reference Stream
|
||||
$xref = $this->decodeXrefStream($pdfData, $startxref, $xref, $visitedOffsets);
|
||||
}
|
||||
}
|
||||
if (empty($xref)) {
|
||||
throw new \Exception('Unable to find xref');
|
||||
}
|
||||
|
||||
return $xref;
|
||||
}
|
||||
|
||||
/**
|
||||
* Parses PDF data and returns extracted data as array.
|
||||
*
|
||||
* @param string $data PDF data to parse
|
||||
*
|
||||
* @return array array of parsed PDF document objects
|
||||
*
|
||||
* @throws EmptyPdfException if empty PDF data given
|
||||
* @throws MissingPdfHeaderException if PDF data missing `%PDF-` header
|
||||
*/
|
||||
public function parseData(string $data): array
|
||||
{
|
||||
if (empty($data)) {
|
||||
throw new EmptyPdfException('Empty PDF data given.');
|
||||
}
|
||||
// find the pdf header starting position
|
||||
if (false === ($trimpos = strpos($data, '%PDF-'))) {
|
||||
throw new MissingPdfHeaderException('Invalid PDF data: Missing `%PDF-` header.');
|
||||
}
|
||||
|
||||
// get PDF content string
|
||||
$pdfData = $trimpos > 0 ? substr($data, $trimpos) : $data;
|
||||
|
||||
// get xref and trailer data
|
||||
$xref = $this->getXrefData($pdfData);
|
||||
|
||||
// If we found Unix line-endings
|
||||
if (isset($xref['Unix'])) {
|
||||
$pdfData = str_replace("\r\n", "\n", $pdfData);
|
||||
$xref = $this->getXrefData($pdfData);
|
||||
}
|
||||
|
||||
// parse all document objects
|
||||
$objects = [];
|
||||
foreach ($xref['xref'] as $obj => $offset) {
|
||||
if (!isset($objects[$obj]) && ($offset > 0)) {
|
||||
// decode objects with positive offset
|
||||
$objects[$obj] = $this->getIndirectObject($pdfData, $xref, $obj, $offset, true);
|
||||
}
|
||||
}
|
||||
|
||||
return [$xref, $objects];
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,51 @@
|
||||
<?php
|
||||
|
||||
/**
|
||||
* @file
|
||||
* This file is part of the PdfParser library.
|
||||
*
|
||||
* @author Sébastien MALOT <sebastien@malot.fr>
|
||||
*
|
||||
* @date 2017-01-03
|
||||
*
|
||||
* @license LGPLv3
|
||||
*
|
||||
* @url <https://github.com/smalot/pdfparser>
|
||||
*
|
||||
* PdfParser is a pdf library written in PHP, extraction oriented.
|
||||
* Copyright (C) 2017 - Sébastien MALOT <sebastien@malot.fr>
|
||||
*
|
||||
* This program is free software: you can redistribute it and/or modify
|
||||
* it under the terms of the GNU Lesser General Public License as published by
|
||||
* the Free Software Foundation, either version 3 of the License, or
|
||||
* (at your option) any later version.
|
||||
*
|
||||
* This program is distributed in the hope that it will be useful,
|
||||
* but WITHOUT ANY WARRANTY; without even the implied warranty of
|
||||
* MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
|
||||
* GNU Lesser General Public License for more details.
|
||||
*
|
||||
* You should have received a copy of the GNU Lesser General Public License
|
||||
* along with this program.
|
||||
* If not, see <http://www.pdfparser.org/sites/default/LICENSE.txt>.
|
||||
*/
|
||||
|
||||
namespace Smalot\PdfParser\XObject;
|
||||
|
||||
use Smalot\PdfParser\Header;
|
||||
use Smalot\PdfParser\Page;
|
||||
use Smalot\PdfParser\PDFObject;
|
||||
|
||||
/**
|
||||
* Class Form
|
||||
*/
|
||||
class Form extends Page
|
||||
{
|
||||
public function getText(?Page $page = null): string
|
||||
{
|
||||
$header = new Header([], $this->document);
|
||||
$contents = new PDFObject($this->document, $header, $this->content, $this->config);
|
||||
|
||||
return $contents->getText($this);
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,47 @@
|
||||
<?php
|
||||
|
||||
/**
|
||||
* @file
|
||||
* This file is part of the PdfParser library.
|
||||
*
|
||||
* @author Sébastien MALOT <sebastien@malot.fr>
|
||||
*
|
||||
* @date 2017-01-03
|
||||
*
|
||||
* @license LGPLv3
|
||||
*
|
||||
* @url <https://github.com/smalot/pdfparser>
|
||||
*
|
||||
* PdfParser is a pdf library written in PHP, extraction oriented.
|
||||
* Copyright (C) 2017 - Sébastien MALOT <sebastien@malot.fr>
|
||||
*
|
||||
* This program is free software: you can redistribute it and/or modify
|
||||
* it under the terms of the GNU Lesser General Public License as published by
|
||||
* the Free Software Foundation, either version 3 of the License, or
|
||||
* (at your option) any later version.
|
||||
*
|
||||
* This program is distributed in the hope that it will be useful,
|
||||
* but WITHOUT ANY WARRANTY; without even the implied warranty of
|
||||
* MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
|
||||
* GNU Lesser General Public License for more details.
|
||||
*
|
||||
* You should have received a copy of the GNU Lesser General Public License
|
||||
* along with this program.
|
||||
* If not, see <http://www.pdfparser.org/sites/default/LICENSE.txt>.
|
||||
*/
|
||||
|
||||
namespace Smalot\PdfParser\XObject;
|
||||
|
||||
use Smalot\PdfParser\Page;
|
||||
use Smalot\PdfParser\PDFObject;
|
||||
|
||||
/**
|
||||
* Class Image
|
||||
*/
|
||||
class Image extends PDFObject
|
||||
{
|
||||
public function getText(?Page $page = null): string
|
||||
{
|
||||
return '';
|
||||
}
|
||||
}
|
||||
Reference in New Issue
Block a user