MarkdownTestHelper.php 6.9 KB

123456789101112131415161718192021222324252627282930313233343536373839404142434445464748495051525354555657585960616263646566676869707172737475767778798081828384858687888990919293949596979899100101102103104105106107108109110111112113114115116117118119120121122123124125126127128129130131132133134135136137138139140141142143144145146147148149150151152153154155156157158159160161162163164165166167168169170171172173174175176177178179180181182183184185186187188189190191192193194195196197198199200201202203204205206207208209210211212213214215216217218219220221222223224225226227228229230231232233234235236237238239240241242243244245246247248249250251252253254255256257258259260261262263264265266267
  1. <?php
  2. use PHPUnit\Framework\TestCase;
  3. class MarkdownTestHelper
  4. {
  5. /**
  6. * Takes an input directory containing .text and .(x)html files, and returns an array
  7. * of .text files and the corresponding output xhtml or html file. Can be used in a unit test data provider.
  8. *
  9. * @param string $directory Input directory
  10. *
  11. * @return array
  12. */
  13. public static function getInputOutputPaths($directory) {
  14. $iterator = new RecursiveIteratorIterator(new RecursiveDirectoryIterator($directory));
  15. $regexIterator = new RegexIterator(
  16. $iterator,
  17. '/^.+\.text$/',
  18. RecursiveRegexIterator::GET_MATCH
  19. );
  20. $dataValues = array();
  21. /** @var SplFileInfo $inputFile */
  22. foreach ($regexIterator as $inputFiles) {
  23. foreach ($inputFiles as $inputMarkdownPath) {
  24. $xhtml = true;
  25. $expectedHtmlPath = substr($inputMarkdownPath, 0, -4) . 'xhtml';
  26. if (!file_exists($expectedHtmlPath)) {
  27. $expectedHtmlPath = substr($inputMarkdownPath, 0, -4) . 'html';
  28. $xhtml = false;
  29. }
  30. $dataValues[] = array($inputMarkdownPath, $expectedHtmlPath, $xhtml);
  31. }
  32. }
  33. return $dataValues;
  34. }
  35. /**
  36. * Applies PHPUnit's assertSame after normalizing both strings (e.g. ignoring whitespace differences).
  37. * Uses logic found originally in MDTest.
  38. *
  39. * @param string $string1
  40. * @param string $string2
  41. * @param string $message Positive message to print when test fails (e.g. "String1 matches String2")
  42. * @param bool $xhtml
  43. */
  44. public static function assertSameNormalized($string1, $string2, $message, $xhtml = true) {
  45. $t_result = $string1;
  46. $t_output = $string2;
  47. // DOMDocuments
  48. if ($xhtml) {
  49. $document = new DOMDocument();
  50. $doc_result = $document->loadXML('<!DOCTYPE html>' .
  51. "<html xmlns='http://www.w3.org/1999/xhtml'>" .
  52. "<body>$t_result</body></html>");
  53. $document2 = new DOMDocument();
  54. $doc_output = $document2->loadXML('<!DOCTYPE html>' .
  55. "<html xmlns='http://www.w3.org/1999/xhtml'>" .
  56. "<body>$t_output</body></html>");
  57. if ($doc_result) {
  58. static::normalizeElementContent($document->documentElement, false);
  59. $n_result = $document->saveXML();
  60. } else {
  61. $n_result = '--- Expected Result: XML Parse Error ---';
  62. }
  63. if ($doc_output) {
  64. static::normalizeElementContent($document2->documentElement, false);
  65. $n_output = $document2->saveXML();
  66. } else {
  67. $n_output = '--- Output: XML Parse Error ---';
  68. }
  69. } else {
  70. // '@' suppressors used because some tests have invalid HTML (multiple elements with the same id attribute)
  71. // Perhaps isolate to a separate test and remove this?
  72. $document = new DOMDocument();
  73. $doc_result = @$document->loadHTML($t_result);
  74. $document2 = new DOMDocument();
  75. $doc_output = @$document2->loadHTML($t_output);
  76. if ($doc_result) {
  77. static::normalizeElementContent($document->documentElement, false);
  78. $n_result = $document->saveHTML();
  79. } else {
  80. $n_result = '--- Expected Result: HTML Parse Error ---';
  81. }
  82. if ($doc_output) {
  83. static::normalizeElementContent($document2->documentElement, false);
  84. $n_output = $document2->saveHTML();
  85. } else {
  86. $n_output = '--- Output: HTML Parse Error ---';
  87. }
  88. }
  89. $n_result = preg_replace('{^.*?<body>|</body>.*?$}is', '', $n_result);
  90. $n_output = preg_replace('{^.*?<body>|</body>.*?$}is', '', $n_output);
  91. $c_result = $n_result;
  92. $c_output = $n_output;
  93. $c_result = trim($c_result) . "\n";
  94. $c_output = trim($c_output) . "\n";
  95. // This will throw a test exception if the strings don't exactly match
  96. TestCase::assertSame($c_result, $c_output, $message);
  97. }
  98. /**
  99. * @param DOMElement $element Modifies this element by reference
  100. * @param bool $whitespace_preserve Preserve Whitespace
  101. * @return void
  102. */
  103. protected static function normalizeElementContent($element, $whitespace_preserve) {
  104. #
  105. # Normalize content of HTML DOM $element. The $whitespace_preserve
  106. # argument indicates that whitespace is significant and shouldn't be
  107. # normalized; it should be used for the content of certain elements like
  108. # <pre> or <script>.
  109. #
  110. $node_list = $element->childNodes;
  111. switch (strtolower($element->nodeName)) {
  112. case 'body':
  113. case 'div':
  114. case 'blockquote':
  115. case 'ul':
  116. case 'ol':
  117. case 'dl':
  118. case 'h1':
  119. case 'h2':
  120. case 'h3':
  121. case 'h4':
  122. case 'h5':
  123. case 'h6':
  124. $whitespace = "\n\n";
  125. break;
  126. case 'table':
  127. $whitespace = "\n";
  128. break;
  129. case 'pre':
  130. case 'script':
  131. case 'style':
  132. case 'title':
  133. $whitespace_preserve = true;
  134. $whitespace = "";
  135. break;
  136. default:
  137. $whitespace = "";
  138. break;
  139. }
  140. foreach ($node_list as $node) {
  141. switch ($node->nodeType) {
  142. case XML_ELEMENT_NODE:
  143. static::normalizeElementContent($node, $whitespace_preserve);
  144. static::normalizeElementAttributes($node);
  145. switch (strtolower($node->nodeName)) {
  146. case 'p':
  147. case 'div':
  148. case 'hr':
  149. case 'blockquote':
  150. case 'ul':
  151. case 'ol':
  152. case 'dl':
  153. case 'li':
  154. case 'address':
  155. case 'table':
  156. case 'dd':
  157. case 'pre':
  158. case 'h1':
  159. case 'h2':
  160. case 'h3':
  161. case 'h4':
  162. case 'h5':
  163. case 'h6':
  164. $whitespace = "\n\n";
  165. break;
  166. case 'tr':
  167. case 'td':
  168. case 'dt':
  169. $whitespace = "\n";
  170. break;
  171. default:
  172. $whitespace = "";
  173. break;
  174. }
  175. if (($whitespace === "\n\n" || $whitespace === "\n") &&
  176. $node->nextSibling &&
  177. $node->nextSibling->nodeType != XML_TEXT_NODE) {
  178. $element->insertBefore(new DOMText($whitespace), $node->nextSibling);
  179. }
  180. break;
  181. case XML_TEXT_NODE:
  182. if (!$whitespace_preserve) {
  183. if (trim($node->data) === "") {
  184. $node->data = $whitespace;
  185. }
  186. else {
  187. $node->data = preg_replace('{\s+}', ' ', $node->data);
  188. }
  189. }
  190. break;
  191. }
  192. }
  193. if (!$whitespace_preserve &&
  194. ($whitespace === "\n\n" || $whitespace === "\n")) {
  195. if ($element->firstChild) {
  196. if ($element->firstChild->nodeType == XML_TEXT_NODE) {
  197. $element->firstChild->data =
  198. preg_replace('{^\s+}', "\n", $element->firstChild->data);
  199. }
  200. else {
  201. $element->insertBefore(new DOMText("\n"), $element->firstChild);
  202. }
  203. }
  204. if ($element->lastChild) {
  205. if ($element->lastChild->nodeType == XML_TEXT_NODE) {
  206. $element->lastChild->data =
  207. preg_replace('{\s+$}', "\n", $element->lastChild->data);
  208. }
  209. else {
  210. $element->insertBefore(new DOMText("\n"), null);
  211. }
  212. }
  213. }
  214. }
  215. /**
  216. * @param DOMElement $element Modifies this element by reference
  217. */
  218. protected static function normalizeElementAttributes (DOMElement $element)
  219. {
  220. #
  221. # Sort attributes by name.
  222. #
  223. // Gather the list of attributes as an array.
  224. $attr_list = array();
  225. foreach ($element->attributes as $attr_node) {
  226. $attr_list[$attr_node->name] = $attr_node;
  227. }
  228. // Sort attribute list by name.
  229. ksort($attr_list);
  230. // Remove then put back each attribute following sort order.
  231. foreach ($attr_list as $attr_node) {
  232. $element->removeAttributeNode($attr_node);
  233. $element->setAttributeNode($attr_node);
  234. }
  235. }
  236. }