Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
52 changes: 40 additions & 12 deletions src/Lexer/Lexer.php
Original file line number Diff line number Diff line change
Expand Up @@ -21,8 +21,6 @@ final class Lexer
private const PATTERN_LINK_DEFINITION = '/^\[([^\]\[]+)\]:\s+(?:<((?:[^<>\\\\\n]|\\\\.)*)>|(\S+))(?:\s+(?:"((?:[^"\\\\]|\\\\.)*)"|\'((?:[^\'\\\\]|\\\\.)*)\'|\(((?:[^()\\\\]|\\\\.)*)\)))?$/';
/** Matches a standalone title line (CommonMark §4.7 multiline link ref definition). */
private const PATTERN_STANDALONE_TITLE = '/^(?:"((?:[^"\\\\]|\\\\.)*)"|\'((?:[^\'\\\\]|\\\\.)*)\'|\(((?:[^()\\\\]|\\\\.)*)\))\s*$/';
private const PATTERN_TABLE_ROW = '/^\|?[^|]+(?:\|[^|]+)+\|?$/';
private const PATTERN_TABLE_SEPARATOR = '/^\|?[ \t:|-]+(?:\|[ \t:|-]+)+\|?$/';
private const PATTERN_SETEXT_H1 = '/^=+\s*$/';
private const PATTERN_SETEXT_H2 = '/^-+\s*$/';
private const PATTERN_COLUMNS_OPEN = '/^:::\s*columns\s*$/i';
Expand Down Expand Up @@ -103,7 +101,17 @@ public function tokenize(string $markdown): array
$footnoteBodyLabel = '';
$footnoteBodyLines = [];

foreach ($lines as $raw) {
// GFM table state: number of columns of the open table (0 = no table).
$tableColumns = 0;
$skipNextLine = false; // delimiter row already consumed with its header

foreach ($lines as $lineIdx => $raw) {
if ($skipNextLine) {
$skipNextLine = false;
continue;
}
$inTable = $tableColumns;
$tableColumns = 0;
$line = rtrim($raw, "\r");
$expanded = $this->expandTabs($line);

Expand Down Expand Up @@ -355,6 +363,35 @@ public function tokenize(string $markdown): array

$token = $this->matchLine($line);

// GFM tables: a header row is only a table when the next line is a delimiter
// row with the same number of cells; rows then continue until a blank line
// or the start of another block.
if ($token->type === TokenType::PARAGRAPH) {
if ($inTable > 0) {
$tokens[] = new Token(TokenType::TABLE_ROW, $line);
$tableColumns = $inTable;
continue;
}
$nextLine = isset($lines[$lineIdx + 1]) ? rtrim($lines[$lineIdx + 1], "\r") : null;
if ($nextLine !== null
&& TableCells::hasPipe($line)
&& !str_starts_with($this->expandTabs($nextLine), ' ')
) {
$aligns = TableCells::alignments($nextLine);
if ($aligns !== null && count($aligns) === count(TableCells::split($line))) {
foreach ($pendingLines as $pt) {
$tokens[] = $pt;
}
$pendingLines = [];
$tokens[] = new Token(TokenType::TABLE_ROW, $line);
$tokens[] = new Token(TokenType::TABLE_SEPARATOR, $nextLine);
$tableColumns = count($aligns);
$skipNextLine = true;
continue;
}
}
}

// A LINK_DEFINITION with no title may have its title on the next line (CommonMark §4.7).
if ($token->type === TokenType::LINK_DEFINITION && $token->meta['title'] === null) {
$pendingLinkDef = $token;
Expand Down Expand Up @@ -582,15 +619,6 @@ private function matchLine(string $line): Token
);
}

if (str_contains($line, '|')) {
if (preg_match(self::PATTERN_TABLE_SEPARATOR, $line)) {
return new Token(TokenType::TABLE_SEPARATOR, $line);
}
if (preg_match(self::PATTERN_TABLE_ROW, $line)) {
return new Token(TokenType::TABLE_ROW, $line);
}
}

// Footnote definition: [^label]: body — must run before LINK_DEFINITION
// because [^label]: body also matches PATTERN_LINK_DEFINITION (label=[^label], href=body).
if (preg_match(self::PATTERN_FOOTNOTE_DEF, $line, $m)) {
Expand Down
90 changes: 90 additions & 0 deletions src/Lexer/TableCells.php
Original file line number Diff line number Diff line change
@@ -0,0 +1,90 @@
<?php

declare(strict_types=1);

namespace PhpMarkdown\Lexer;

/**
* Splits a GFM table row into cells.
*
* Cells are separated by unescaped '|'; one leading and one trailing pipe are optional.
* An escaped pipe (\|) never separates cells — it is kept as '\|' in the cell text so the
* caller decides how to unescape it (GFM replaces it with '|' before inline parsing,
* including inside code spans).
*/
final class TableCells
{
/** @return list<string> trimmed raw cell contents */
public static function split(string $line): array
{
$line = trim($line, " \t");
if (str_starts_with($line, '|')) {
$line = substr($line, 1);
}
if (str_ends_with($line, '|') && !self::isEscaped($line, strlen($line) - 1)) {
$line = substr($line, 0, -1);
}

$cells = [];
$cell = '';
$len = strlen($line);
for ($i = 0; $i < $len; $i++) {
$ch = $line[$i];
if ($ch === '\\' && $i + 1 < $len) {
$cell .= $ch . $line[$i + 1];
$i++;
continue;
}
if ($ch === '|') {
$cells[] = trim($cell, " \t");
$cell = '';
continue;
}
$cell .= $ch;
}
$cells[] = trim($cell, " \t");

return $cells;
}

/** Whether $line contains at least one unescaped pipe. */
public static function hasPipe(string $line): bool
{
return count(self::split('x' . $line . 'x')) > 1;
}

/**
* Parse a delimiter row ("| :-- | --: |"). Returns one alignment per column
* ('left', 'right', 'center' or ''), or null if $line is not a delimiter row.
*
* @return list<string>|null
*/
public static function alignments(string $line): ?array
{
if (!self::hasPipe($line)) {
return null;
}
$aligns = [];
foreach (self::split($line) as $cell) {
if (!preg_match('/^(:?)-+(:?)$/', $cell, $m)) {
return null;
}
$aligns[] = match (true) {
$m[1] !== '' && $m[2] !== '' => 'center',
$m[2] !== '' => 'right',
$m[1] !== '' => 'left',
default => '',
};
}
return $aligns;
}

private static function isEscaped(string $line, int $pos): bool
{
$backslashes = 0;
for ($i = $pos - 1; $i >= 0 && $line[$i] === '\\'; $i--) {
$backslashes++;
}
return $backslashes % 2 === 1;
}
}
78 changes: 33 additions & 45 deletions src/Parser/Parser.php
Original file line number Diff line number Diff line change
Expand Up @@ -6,6 +6,7 @@

use PhpMarkdown\Exception\ParseException;
use PhpMarkdown\Lexer\Lexer;
use PhpMarkdown\Lexer\TableCells;
use PhpMarkdown\Lexer\Token;
use PhpMarkdown\Lexer\TokenType;
use PhpMarkdown\Node\Block\BlockquoteNode;
Expand Down Expand Up @@ -307,69 +308,56 @@ private function buildTable(array $tokens, int &$i): TableNode
$headerCells = $this->parseCells($tokens[$i]->content);
$i++;

// Separator row — extract alignment, advance
// Delimiter row — the Lexer only emits a table when it is present and valid.
$aligns = [];
if ($i < $count && $tokens[$i]->type === TokenType::TABLE_SEPARATOR) {
$aligns = $this->parseAlignments($tokens[$i]->content);
$aligns = TableCells::alignments($tokens[$i]->content) ?? [];
$i++;
}
$columns = count($headerCells);

$rows[] = new TableRowNode(
cells: array_map(
fn(string $cell, int $idx) => new TableCellNode(
children: $this->inlineParser->parse($cell, $this->linkRefs, $this->footnoteDefs),
align: $aligns[$idx] ?? '',
),
$headerCells,
array_keys($headerCells),
),
isHeader: true,
);
$rows[] = $this->buildTableRow($headerCells, $aligns, $columns, isHeader: true);

// Body rows
while ($i < $count && $tokens[$i]->type === TokenType::TABLE_ROW) {
$cells = $this->parseCells($tokens[$i]->content);
$rows[] = new TableRowNode(
cells: array_map(
fn(string $cell, int $idx) => new TableCellNode(
children: $this->inlineParser->parse($cell, $this->linkRefs, $this->footnoteDefs),
align: $aligns[$idx] ?? '',
),
$cells,
array_keys($cells),
),
isHeader: false,
);
$rows[] = $this->buildTableRow($this->parseCells($tokens[$i]->content), $aligns, $columns, isHeader: false);
$i++;
}

return new TableNode(rows: $rows);
}

/** @return string[] */
private function parseCells(string $line): array
/**
* GFM: every row has exactly as many cells as the header — missing cells are
* empty, excess cells are ignored.
*
* @param list<string> $cells
* @param list<string> $aligns
*/
private function buildTableRow(array $cells, array $aligns, int $columns, bool $isHeader): TableRowNode
{
$line = trim($line, ' |');
return array_map(trim(...), explode('|', $line));
$nodes = [];
for ($idx = 0; $idx < $columns; $idx++) {
$nodes[] = new TableCellNode(
children: $this->inlineParser->parse($cells[$idx] ?? '', $this->linkRefs, $this->footnoteDefs),
align: $aligns[$idx] ?? '',
);
}
return new TableRowNode(cells: $nodes, isHeader: $isHeader);
}

/** @return string[] */
private function parseAlignments(string $separator): array
/**
* Split a row on unescaped pipes; an escaped pipe becomes a literal '|' before
* inline parsing, including inside code spans (GFM §4.10).
*
* @return list<string>
*/
private function parseCells(string $line): array
{
$separator = trim($separator, ' |');
$aligns = [];
foreach (explode('|', $separator) as $col) {
$col = trim($col);
$left = str_starts_with($col, ':');
$right = str_ends_with($col, ':');
$aligns[] = match (true) {
$left && $right => 'center',
$right => 'right',
$left => 'left',
default => '',
};
}
return $aligns;
return array_map(
static fn(string $cell): string => str_replace('\\|', '|', $cell),
TableCells::split($line),
);
}

/**
Expand Down
56 changes: 56 additions & 0 deletions tests/Integration/TableTest.php
Original file line number Diff line number Diff line change
@@ -0,0 +1,56 @@
<?php

declare(strict_types=1);

namespace PhpMarkdown\Tests\Integration;

use PhpMarkdown\MarkdownParser;
use PHPUnit\Framework\Attributes\DataProvider;
use PHPUnit\Framework\TestCase;

/**
* GFM tables (§4.10): detection requires a valid delimiter row; escaped pipes; cell count.
*/
final class TableTest extends TestCase
{
/** @return array<string, array{string, string}> */
public static function cases(): array
{
return [
'pipes in prose are not a table' => [
'I like cats | dogs | birds',
"<p>I like cats | dogs | birds</p>\n",
],
'delimiter cell count must match header' => [
"| a | b |\n| --- |\n| c |",
"<p>| a | b |\n| --- |\n| c |</p>\n",
],
'single column table' => [
"| a |\n| :-: |\n| b |",
'<table><thead><tr><th align="center">a</th></tr></thead><tbody><tr><td align="center">b</td></tr></tbody></table>',
],
'escaped pipe stays in cell, also in code span' => [
"| a | b |\n|---|---|\n| `x\\|y` | 1 \\| 2 |",
'<table><thead><tr><th>a</th><th>b</th></tr></thead><tbody><tr><td><code>x|y</code></td><td>1 | 2</td></tr></tbody></table>',
],
'short rows padded, long rows truncated' => [
"| a | b |\n|---|---|\n| 1 |\n| 1 | 2 | 3 |",
'<table><thead><tr><th>a</th><th>b</th></tr></thead><tbody><tr><td>1</td><td></td></tr><tr><td>1</td><td>2</td></tr></tbody></table>',
],
'table ends at another block' => [
"a | b\n--- | ---\nc | d\n# h",
"<table><thead><tr><th>a</th><th>b</th></tr></thead><tbody><tr><td>c</td><td>d</td></tr></tbody></table><h1>h</h1>\n",
],
'table after a paragraph line' => [
"para\na | b\n-|-\n1|2",
"<p>para</p>\n<table><thead><tr><th>a</th><th>b</th></tr></thead><tbody><tr><td>1</td><td>2</td></tr></tbody></table>",
],
];
}

#[DataProvider('cases')]
public function testTable(string $markdown, string $expected): void
{
$this->assertSame($expected, (new MarkdownParser())->parse($markdown));
}
}
5 changes: 3 additions & 2 deletions tests/Unit/ColumnsLexerTest.php
Original file line number Diff line number Diff line change
Expand Up @@ -161,12 +161,13 @@ public function testPipeTripleOutsideColumnsBlockIsNotColumnsContainer(): void
}
}

public function testPipeTripleOutsideColumnsBlockProducesTableSeparator(): void
public function testPipeTripleOutsideColumnsBlockIsParagraph(): void
{
// Outside a columns block, "|||" is not a table either (no header/delimiter pair).
$tokens = $this->lexer->tokenize("|||");

$this->assertCount(1, $tokens);
$this->assertSame(TokenType::TABLE_SEPARATOR, $tokens[0]->type);
$this->assertSame(TokenType::PARAGRAPH, $tokens[0]->type);
}

// -------------------------------------------------------------------------
Expand Down
Loading
Loading