565 lines
18 KiB
PHP
565 lines
18 KiB
PHP
<?php
|
|
// +----------------------------------------------------------------------
|
|
// | YwxApp [ WE CAN DO IT JUST THINK ]
|
|
// +----------------------------------------------------------------------
|
|
// | Copyright (c) 2026-2036 http://ywxapp.cn All rights reserved.
|
|
// +----------------------------------------------------------------------
|
|
// | Author: ywxapp<admin@ywxapp.cn>
|
|
// +----------------------------------------------------------------------
|
|
declare (strict_types = 1);
|
|
|
|
namespace addon\docs\service;
|
|
|
|
/**
|
|
* 文档导入服务
|
|
*
|
|
* 支持三种来源:
|
|
* 1. .docx 优先使用 phpoffice/phpword,未安装时退回内置 ZIP + XML 解析
|
|
* 2. .md 内置轻量 Markdown 转 HTML
|
|
* 3. .html 直接清洗后入库
|
|
*
|
|
* @author ywxapp <admin@ywxapp.cn>
|
|
*/
|
|
class DocImportService
|
|
{
|
|
/**
|
|
* 允许导入的扩展名
|
|
*/
|
|
const ALLOW_EXT = ['docx', 'md', 'markdown', 'html', 'htm', 'txt'];
|
|
|
|
/**
|
|
* 解析上传文件为文档内容
|
|
*
|
|
* @param string $filePath 文件绝对路径
|
|
* @param string $ext 扩展名(小写)
|
|
* @return array{title: string, content: string, content_md: string}
|
|
* @throws \Exception 解析失败时抛出
|
|
*/
|
|
public static function parse(string $filePath, string $ext): array
|
|
{
|
|
if (!is_file($filePath)) {
|
|
throw new \Exception('导入文件不存在');
|
|
}
|
|
|
|
$ext = strtolower($ext);
|
|
if (!in_array($ext, self::ALLOW_EXT, true)) {
|
|
throw new \Exception('不支持的文件类型:' . $ext);
|
|
}
|
|
|
|
switch ($ext) {
|
|
case 'docx':
|
|
return self::parseDocx($filePath);
|
|
case 'md':
|
|
case 'markdown':
|
|
return self::parseMarkdown($filePath);
|
|
default:
|
|
return self::parseHtml($filePath);
|
|
}
|
|
}
|
|
|
|
/**
|
|
* 解析 docx
|
|
*
|
|
* @param string $filePath 文件路径
|
|
* @return array
|
|
* @throws \Exception
|
|
*/
|
|
protected static function parseDocx(string $filePath): array
|
|
{
|
|
if (class_exists('\PhpOffice\PhpWord\IOFactory')) {
|
|
try {
|
|
return self::parseDocxByPhpWord($filePath);
|
|
} catch (\Throwable $e) {
|
|
// PhpWord 解析异常时退回内置解析,保证功能可用
|
|
}
|
|
}
|
|
return self::parseDocxByZip($filePath);
|
|
}
|
|
|
|
/**
|
|
* 使用 phpoffice/phpword 解析 docx
|
|
*
|
|
* @param string $filePath 文件路径
|
|
* @return array
|
|
* @throws \Exception
|
|
*/
|
|
protected static function parseDocxByPhpWord(string $filePath): array
|
|
{
|
|
$reader = \PhpOffice\PhpWord\IOFactory::createReader('Word2007');
|
|
$document = $reader->load($filePath);
|
|
|
|
$writer = \PhpOffice\PhpWord\IOFactory::createWriter($document, 'HTML');
|
|
ob_start();
|
|
$writer->save('php://output');
|
|
$html = (string) ob_get_clean();
|
|
|
|
// PhpWord 输出整页 HTML,仅保留 body 内部
|
|
if (preg_match('#<body[^>]*>(.*?)</body>#is', $html, $m)) {
|
|
$html = $m[1];
|
|
}
|
|
|
|
$html = self::sanitize($html);
|
|
$title = self::guessTitle($html, $filePath);
|
|
|
|
return ['title' => $title, 'content' => $html, 'content_md' => ''];
|
|
}
|
|
|
|
/**
|
|
* 内置 docx 解析:解压 word/document.xml 后提取段落
|
|
*
|
|
* 不依赖任何扩展包,能还原段落、标题层级与加粗,适用于纯文字文档。
|
|
*
|
|
* @param string $filePath 文件路径
|
|
* @return array
|
|
* @throws \Exception
|
|
*/
|
|
protected static function parseDocxByZip(string $filePath): array
|
|
{
|
|
if (!class_exists('\ZipArchive')) {
|
|
throw new \Exception('解析 docx 需要 PHP 开启 zip 扩展,或安装 phpoffice/phpword');
|
|
}
|
|
|
|
$zip = new \ZipArchive();
|
|
if ($zip->open($filePath) !== true) {
|
|
throw new \Exception('无法打开 docx 文件,请确认文件未损坏');
|
|
}
|
|
|
|
$xml = $zip->getFromName('word/document.xml');
|
|
$zip->close();
|
|
|
|
if ($xml === false || $xml === '') {
|
|
throw new \Exception('docx 内容为空或格式不正确');
|
|
}
|
|
|
|
$prev = libxml_use_internal_errors(true);
|
|
$dom = new \DOMDocument();
|
|
$dom->loadXML($xml, LIBXML_NOCDATA | LIBXML_NONET);
|
|
libxml_clear_errors();
|
|
libxml_use_internal_errors($prev);
|
|
|
|
$xpath = new \DOMXPath($dom);
|
|
$xpath->registerNamespace('w', 'http://schemas.openxmlformats.org/wordprocessingml/2006/main');
|
|
|
|
$paragraphs = $xpath->query('//w:body/w:p');
|
|
$htmlParts = [];
|
|
|
|
if ($paragraphs !== false) {
|
|
foreach ($paragraphs as $p) {
|
|
$text = '';
|
|
$runs = $xpath->query('.//w:t', $p);
|
|
if ($runs !== false) {
|
|
foreach ($runs as $t) {
|
|
$text .= $t->textContent;
|
|
}
|
|
}
|
|
|
|
$text = trim($text);
|
|
if ($text === '') {
|
|
continue;
|
|
}
|
|
|
|
// 读取段落样式名判断标题层级,如 Heading1 / 标题 1
|
|
$level = 0;
|
|
$styleNode = $xpath->query('.//w:pStyle/@w:val', $p);
|
|
if ($styleNode !== false && $styleNode->length > 0) {
|
|
$style = (string) $styleNode->item(0)->nodeValue;
|
|
if (preg_match('/(?:heading|标题)\s*([1-6])/i', $style, $m)) {
|
|
$level = (int) $m[1];
|
|
}
|
|
}
|
|
|
|
$safe = htmlspecialchars($text, ENT_QUOTES | ENT_HTML5, 'UTF-8');
|
|
if ($level >= 1 && $level <= 6) {
|
|
$htmlParts[] = '<h' . $level . '>' . $safe . '</h' . $level . '>';
|
|
} else {
|
|
$htmlParts[] = '<p>' . $safe . '</p>';
|
|
}
|
|
}
|
|
}
|
|
|
|
if (empty($htmlParts)) {
|
|
throw new \Exception('未能从 docx 中提取到文本内容');
|
|
}
|
|
|
|
$html = implode("\n", $htmlParts);
|
|
$title = self::guessTitle($html, $filePath);
|
|
|
|
return ['title' => $title, 'content' => $html, 'content_md' => ''];
|
|
}
|
|
|
|
/**
|
|
* 解析 Markdown 文件
|
|
*
|
|
* @param string $filePath 文件路径
|
|
* @return array
|
|
*/
|
|
protected static function parseMarkdown(string $filePath): array
|
|
{
|
|
$md = (string) file_get_contents($filePath);
|
|
$md = self::normalizeEncoding($md);
|
|
|
|
$html = self::markdownToHtml($md);
|
|
$title = '';
|
|
|
|
// 优先取首个一级标题作为文档标题
|
|
if (preg_match('/^\s*#\s+(.+)$/m', $md, $m)) {
|
|
$title = trim($m[1]);
|
|
}
|
|
if ($title === '') {
|
|
$title = self::guessTitle($html, $filePath);
|
|
}
|
|
|
|
return ['title' => $title, 'content' => $html, 'content_md' => $md];
|
|
}
|
|
|
|
/**
|
|
* 解析 HTML / 纯文本文件
|
|
*
|
|
* @param string $filePath 文件路径
|
|
* @return array
|
|
*/
|
|
protected static function parseHtml(string $filePath): array
|
|
{
|
|
$raw = (string) file_get_contents($filePath);
|
|
$raw = self::normalizeEncoding($raw);
|
|
|
|
if (preg_match('#<body[^>]*>(.*?)</body>#is', $raw, $m)) {
|
|
$raw = $m[1];
|
|
}
|
|
|
|
// 纯文本按行包 p 标签
|
|
if (strip_tags($raw) === $raw) {
|
|
$lines = preg_split('/\r\n|\r|\n/', $raw) ?: [];
|
|
$parts = [];
|
|
foreach ($lines as $line) {
|
|
$line = trim($line);
|
|
if ($line !== '') {
|
|
$parts[] = '<p>' . htmlspecialchars($line, ENT_QUOTES | ENT_HTML5, 'UTF-8') . '</p>';
|
|
}
|
|
}
|
|
$raw = implode("\n", $parts);
|
|
}
|
|
|
|
$html = self::sanitize($raw);
|
|
$title = self::guessTitle($html, $filePath);
|
|
|
|
return ['title' => $title, 'content' => $html, 'content_md' => ''];
|
|
}
|
|
|
|
/**
|
|
* 轻量 Markdown 转 HTML
|
|
*
|
|
* 覆盖文档站常用语法:标题、围栏代码块、行内代码、粗斜体、
|
|
* 链接、图片、无序 / 有序列表、引用、分隔线、表格。
|
|
* 若项目已安装 erusev/parsedown 则优先使用。
|
|
*
|
|
* @param string $md Markdown 源码
|
|
* @return string
|
|
*/
|
|
public static function markdownToHtml(string $md): string
|
|
{
|
|
if (class_exists('\Parsedown')) {
|
|
try {
|
|
$parser = new \Parsedown();
|
|
if (method_exists($parser, 'setSafeMode')) {
|
|
$parser->setSafeMode(true);
|
|
}
|
|
return (string) $parser->text($md);
|
|
} catch (\Throwable $e) {
|
|
// 解析失败退回内置实现
|
|
}
|
|
}
|
|
|
|
$md = str_replace(["\r\n", "\r"], "\n", $md);
|
|
|
|
// 先抽出围栏代码块,避免其内部符号被后续规则误处理
|
|
$blocks = [];
|
|
$md = preg_replace_callback('/```([a-zA-Z0-9#+\-]*)\n(.*?)```/s', function ($m) use (&$blocks) {
|
|
$lang = $m[1] !== '' ? ' class="language-' . htmlspecialchars($m[1], ENT_QUOTES) . '"' : '';
|
|
$code = htmlspecialchars($m[2], ENT_QUOTES | ENT_HTML5, 'UTF-8');
|
|
$token = '@@DOCS_CODE_' . count($blocks) . '@@';
|
|
$blocks[$token] = '<pre><code' . $lang . '>' . $code . '</code></pre>';
|
|
return "\n" . $token . "\n";
|
|
}, $md) ?? $md;
|
|
|
|
$lines = explode("\n", $md);
|
|
$out = [];
|
|
$listType = ''; // ul / ol / 空
|
|
$inQuote = false;
|
|
$inTable = false;
|
|
|
|
$closeList = function () use (&$listType, &$out) {
|
|
if ($listType !== '') {
|
|
$out[] = '</' . $listType . '>';
|
|
$listType = '';
|
|
}
|
|
};
|
|
$closeQuote = function () use (&$inQuote, &$out) {
|
|
if ($inQuote) {
|
|
$out[] = '</blockquote>';
|
|
$inQuote = false;
|
|
}
|
|
};
|
|
$closeTable = function () use (&$inTable, &$out) {
|
|
if ($inTable) {
|
|
$out[] = '</tbody></table>';
|
|
$inTable = false;
|
|
}
|
|
};
|
|
|
|
foreach ($lines as $i => $line) {
|
|
$trim = trim($line);
|
|
|
|
// 代码块占位符原样输出
|
|
if (isset($blocks[$trim])) {
|
|
$closeList();
|
|
$closeQuote();
|
|
$closeTable();
|
|
$out[] = $trim;
|
|
continue;
|
|
}
|
|
|
|
if ($trim === '') {
|
|
$closeList();
|
|
$closeQuote();
|
|
$closeTable();
|
|
continue;
|
|
}
|
|
|
|
// 分隔线
|
|
if (preg_match('/^(\*{3,}|-{3,}|_{3,})$/', $trim)) {
|
|
$closeList();
|
|
$closeQuote();
|
|
$closeTable();
|
|
$out[] = '<hr>';
|
|
continue;
|
|
}
|
|
|
|
// 标题
|
|
if (preg_match('/^(#{1,6})\s+(.*)$/', $trim, $m)) {
|
|
$closeList();
|
|
$closeQuote();
|
|
$closeTable();
|
|
$level = strlen($m[1]);
|
|
$out[] = '<h' . $level . '>' . self::inline($m[2]) . '</h' . $level . '>';
|
|
continue;
|
|
}
|
|
|
|
// 表格:当前行是 | a | b |,下一行是分隔行
|
|
if (!$inTable && strpos($trim, '|') !== false
|
|
&& isset($lines[$i + 1]) && preg_match('/^\s*\|?[\s:\-\|]+\|[\s:\-\|]*$/', $lines[$i + 1])) {
|
|
$closeList();
|
|
$closeQuote();
|
|
$cells = self::tableCells($trim);
|
|
$out[] = '<table><thead><tr>';
|
|
foreach ($cells as $cell) {
|
|
$out[] = '<th>' . self::inline($cell) . '</th>';
|
|
}
|
|
$out[] = '</tr></thead><tbody>';
|
|
$inTable = true;
|
|
continue;
|
|
}
|
|
if ($inTable) {
|
|
// 跳过分隔行
|
|
if (preg_match('/^\s*\|?[\s:\-\|]+\|[\s:\-\|]*$/', $trim)) {
|
|
continue;
|
|
}
|
|
if (strpos($trim, '|') !== false) {
|
|
$cells = self::tableCells($trim);
|
|
$out[] = '<tr>';
|
|
foreach ($cells as $cell) {
|
|
$out[] = '<td>' . self::inline($cell) . '</td>';
|
|
}
|
|
$out[] = '</tr>';
|
|
continue;
|
|
}
|
|
$closeTable();
|
|
}
|
|
|
|
// 引用
|
|
if (preg_match('/^>\s?(.*)$/', $trim, $m)) {
|
|
$closeList();
|
|
if (!$inQuote) {
|
|
$out[] = '<blockquote>';
|
|
$inQuote = true;
|
|
}
|
|
$out[] = '<p>' . self::inline($m[1]) . '</p>';
|
|
continue;
|
|
}
|
|
$closeQuote();
|
|
|
|
// 无序列表
|
|
if (preg_match('/^[\*\-\+]\s+(.*)$/', $trim, $m)) {
|
|
if ($listType !== 'ul') {
|
|
$closeList();
|
|
$out[] = '<ul>';
|
|
$listType = 'ul';
|
|
}
|
|
$out[] = '<li>' . self::inline($m[1]) . '</li>';
|
|
continue;
|
|
}
|
|
|
|
// 有序列表
|
|
if (preg_match('/^\d+\.\s+(.*)$/', $trim, $m)) {
|
|
if ($listType !== 'ol') {
|
|
$closeList();
|
|
$out[] = '<ol>';
|
|
$listType = 'ol';
|
|
}
|
|
$out[] = '<li>' . self::inline($m[1]) . '</li>';
|
|
continue;
|
|
}
|
|
|
|
$closeList();
|
|
$out[] = '<p>' . self::inline($trim) . '</p>';
|
|
}
|
|
|
|
$closeList();
|
|
$closeQuote();
|
|
$closeTable();
|
|
|
|
$html = implode("\n", $out);
|
|
|
|
// 还原代码块
|
|
if (!empty($blocks)) {
|
|
$html = strtr($html, $blocks);
|
|
}
|
|
|
|
return $html;
|
|
}
|
|
|
|
/**
|
|
* 拆分表格行单元格
|
|
*
|
|
* @param string $line 表格行
|
|
* @return array
|
|
*/
|
|
protected static function tableCells(string $line): array
|
|
{
|
|
$line = trim($line, "| \t");
|
|
$cells = explode('|', $line);
|
|
return array_map('trim', $cells);
|
|
}
|
|
|
|
/**
|
|
* 处理行内 Markdown 语法
|
|
*
|
|
* @param string $text 行内文本
|
|
* @return string
|
|
*/
|
|
protected static function inline(string $text): string
|
|
{
|
|
$text = htmlspecialchars($text, ENT_QUOTES | ENT_HTML5, 'UTF-8');
|
|
|
|
// 行内代码优先,避免其中的星号被当作强调
|
|
$codes = [];
|
|
$text = preg_replace_callback('/`([^`]+)`/', function ($m) use (&$codes) {
|
|
$token = '@@DOCS_IC_' . count($codes) . '@@';
|
|
$codes[$token] = '<code>' . $m[1] . '</code>';
|
|
return $token;
|
|
}, $text) ?? $text;
|
|
|
|
// 图片
|
|
$text = preg_replace('/!\[([^\]]*)\]\(([^)\s]+)[^)]*\)/', '<img src="$2" alt="$1">', $text) ?? $text;
|
|
// 链接
|
|
$text = preg_replace('/\[([^\]]+)\]\(([^)\s]+)[^)]*\)/', '<a href="$2" target="_blank" rel="noopener">$1</a>', $text) ?? $text;
|
|
// 加粗
|
|
$text = preg_replace('/\*\*([^*]+)\*\*/', '<strong>$1</strong>', $text) ?? $text;
|
|
$text = preg_replace('/__([^_]+)__/', '<strong>$1</strong>', $text) ?? $text;
|
|
// 斜体
|
|
$text = preg_replace('/\*([^*]+)\*/', '<em>$1</em>', $text) ?? $text;
|
|
// 删除线
|
|
$text = preg_replace('/~~([^~]+)~~/', '<del>$1</del>', $text) ?? $text;
|
|
|
|
if (!empty($codes)) {
|
|
$text = strtr($text, $codes);
|
|
}
|
|
|
|
return $text;
|
|
}
|
|
|
|
/**
|
|
* 清洗 HTML,移除脚本与事件属性
|
|
*
|
|
* @param string $html 原始 HTML
|
|
* @return string
|
|
*/
|
|
public static function sanitize(string $html): string
|
|
{
|
|
if ($html === '') {
|
|
return '';
|
|
}
|
|
$html = preg_replace('#<(script|style|iframe|object|embed)\b[^>]*>.*?</\1>#is', '', $html) ?? $html;
|
|
$html = preg_replace('#<\?php.*?\?>#is', '', $html) ?? $html;
|
|
// 移除 onclick 等事件属性
|
|
$html = preg_replace('/\son[a-z]+\s*=\s*("[^"]*"|\'[^\']*\'|[^\s>]+)/i', '', $html) ?? $html;
|
|
// 移除 javascript: 协议
|
|
$html = preg_replace('/(href|src)\s*=\s*("|\')\s*javascript:[^"\']*\2/i', '$1="#"', $html) ?? $html;
|
|
return trim($html);
|
|
}
|
|
|
|
/**
|
|
* 推测标题:优先取首个标题标签,其次取文件名
|
|
*
|
|
* @param string $html 内容
|
|
* @param string $filePath 文件路径
|
|
* @return string
|
|
*/
|
|
protected static function guessTitle(string $html, string $filePath): string
|
|
{
|
|
if (preg_match('#<h[1-3][^>]*>(.*?)</h[1-3]>#is', $html, $m)) {
|
|
$title = trim(strip_tags($m[1]));
|
|
if ($title !== '') {
|
|
return mb_substr($title, 0, 200, 'UTF-8');
|
|
}
|
|
}
|
|
|
|
$base = pathinfo($filePath, PATHINFO_FILENAME);
|
|
return $base !== '' ? mb_substr($base, 0, 200, 'UTF-8') : '未命名文档';
|
|
}
|
|
|
|
/**
|
|
* 统一转 UTF-8,兼容 GBK 编码的文本文件
|
|
*
|
|
* @param string $text 原始文本
|
|
* @return string
|
|
*/
|
|
protected static function normalizeEncoding(string $text): string
|
|
{
|
|
if ($text === '' || !function_exists('mb_detect_encoding')) {
|
|
return $text;
|
|
}
|
|
// 去除 UTF-8 BOM
|
|
$text = preg_replace('/^\xEF\xBB\xBF/', '', $text) ?? $text;
|
|
|
|
$encoding = mb_detect_encoding($text, ['UTF-8', 'GB18030', 'GBK', 'BIG5'], true);
|
|
if ($encoding && strtoupper($encoding) !== 'UTF-8') {
|
|
$converted = @mb_convert_encoding($text, 'UTF-8', $encoding);
|
|
if ($converted !== false) {
|
|
return $converted;
|
|
}
|
|
}
|
|
return $text;
|
|
}
|
|
|
|
/**
|
|
* 由标题生成 URL 标识
|
|
*
|
|
* @param string $title 标题
|
|
* @return string
|
|
*/
|
|
public static function slugify(string $title): string
|
|
{
|
|
$slug = mb_strtolower(trim($title), 'UTF-8');
|
|
// 仅保留字母数字与中划线,中文标题回退为拼音无关的哈希短码
|
|
$ascii = preg_replace('/[^a-z0-9]+/u', '-', $slug) ?? '';
|
|
$ascii = trim($ascii, '-');
|
|
|
|
if ($ascii === '' || !preg_match('/[a-z0-9]/', $ascii)) {
|
|
return 'doc-' . substr(md5($title . microtime(true)), 0, 8);
|
|
}
|
|
|
|
return mb_substr($ascii, 0, 80, 'UTF-8');
|
|
}
|
|
}
|