llm = $llm ?: new LLMService(); } /** * 按 user_id 查 user_cv,取最新一条简历解析。 * * @param int $userId * @return array */ public function parseFromUserId($userId) { $userId = intval($userId); if ($userId <= 0) { throw new Exception('user_id 无效'); } $row = Db::name('user_cv') ->where('user_id', $userId) ->where('state', 0) ->order('ctime desc') ->find(); if (empty($row) || trim((string) $row['cv']) === '') { throw new Exception('未找到该用户的简历(user_cv)'); } $cvRel = $this->normalizeCvRelativePath($row['cv']); $cvUrl = $this->buildCvUrl($cvRel); $local = $this->resolveLocalCvPath($cvRel); $temp = false; $filePath = $local; if ($filePath === '' || !is_file($filePath)) { $filePath = $this->downloadCvToTemp($cvUrl, $cvRel); $temp = true; } try { $result = $this->parseFromFile($filePath); } finally { if ($temp && is_file($filePath)) { @unlink($filePath); } } $result['user_id'] = $userId; $result['user_cv_id'] = intval($row['user_cv_id'] ?? 0); $result['cv'] = $cvRel; $result['cv_url'] = $cvUrl; return $result; } /** * @param string $filePath 服务器本地绝对路径 * @return array */ public function parseFromFile($filePath) { $filePath = trim((string) $filePath); if ($filePath === '' || !is_file($filePath)) { throw new Exception('文件不存在'); } $text = trim($this->extractFileText($filePath)); if ($text === '') { throw new Exception('未能从文件中提取到文本'); } $text = $this->sanitizeUtf8($text); $maxChars = max(1000, (int) Env::get('promotion.promotion_llm_max_chars', 2500)); if (mb_strlen($text) > $maxChars) { $text = mb_substr($text, 0, $maxChars) . "\n...(truncated)"; } $llmResult = $this->parseUserInfoWithLlm($text); return [ 'file' => basename($filePath), 'text_length' => mb_strlen($text), 'text_preview' => mb_substr($text, 0, 600), 'user_info' => $llmResult['user_info'], 'llm_raw' => $llmResult['raw'], ]; } private function normalizeCvRelativePath($cv) { $cv = trim(str_replace('\\', '/', (string) $cv)); if ($cv === '') { return ''; } if (preg_match('#^https?://#i', $cv)) { return $cv; } return ltrim($cv, '/'); } private function buildCvUrl($cvRel) { if (preg_match('#^https?://#i', $cvRel)) { return $cvRel; } $base = rtrim(trim((string) Env::get('reviewer.cv_base_url', self::CV_BASE_URL_DEFAULT)), '/') . '/'; return $base . $cvRel; } private function resolveLocalCvPath($cvRel) { if (preg_match('#^https?://#i', $cvRel)) { return ''; } $path = ROOT_PATH . 'public' . DS . 'reviewer' . DS . str_replace('/', DS, $cvRel); return is_file($path) ? $path : ''; } private function downloadCvToTemp($url, $cvRel) { $ext = strtolower(pathinfo(parse_url($cvRel, PHP_URL_PATH) ?: $cvRel, PATHINFO_EXTENSION)); if ($ext === '') { $ext = 'dat'; } $saveDir = ROOT_PATH . 'runtime' . DS . 'user_file_parse'; if (!is_dir($saveDir)) { @mkdir($saveDir, 0755, true); } $path = $saveDir . DS . date('YmdHis') . '_' . uniqid('', true) . '.' . $ext; $ch = curl_init($url); curl_setopt_array($ch, [ CURLOPT_RETURNTRANSFER => true, CURLOPT_FOLLOWLOCATION => true, CURLOPT_CONNECTTIMEOUT => 15, CURLOPT_TIMEOUT => 90, CURLOPT_SSL_VERIFYPEER => false, CURLOPT_SSL_VERIFYHOST => 0, ]); $body = curl_exec($ch); $code = (int) curl_getinfo($ch, CURLINFO_HTTP_CODE); $err = curl_error($ch); curl_close($ch); if ($body === false || $code < 200 || $code >= 300) { throw new Exception('下载简历失败(HTTP ' . $code . '): ' . $url . ($err !== '' ? ' ' . $err : '')); } if (@file_put_contents($path, $body) === false) { throw new Exception('保存简历临时文件失败'); } return $path; } private function extractFileText($filePath) { $ext = strtolower(pathinfo($filePath, PATHINFO_EXTENSION)); if (in_array($ext, ['txt', 'text', 'md', 'csv', 'log'], true)) { $raw = @file_get_contents($filePath); return is_string($raw) ? $raw : ''; } if ($ext === 'pdf') { return $this->extractPdfText($filePath); } if (in_array($ext, ['doc', 'docx'], true)) { return $this->extractDocxText($filePath); } // 未识别扩展名:先尝试 Word,再尝试 PDF,最后按纯文本读 $text = $this->tryExtractDocxText($filePath); if ($text !== '') { return $text; } $text = $this->tryExtractPdfText($filePath); if ($text !== '') { return $text; } $raw = @file_get_contents($filePath); if (!is_string($raw) || $raw === '') { return ''; } if ($this->looksLikeBinary($raw)) { throw new Exception('不支持的简历文件格式: .' . ($ext !== '' ? $ext : 'unknown')); } return $raw; } private function extractPdfText($filePath) { $text = $this->tryExtractPdfText($filePath); if ($text === '') { throw new Exception('PDF 文本提取失败,请确认已安装 smalot/pdfparser:composer require smalot/pdfparser'); } return $text; } private function tryExtractPdfText($filePath) { if (!class_exists(PdfParser::class)) { \think\Log::warning('UserInfoFromFile: smalot/pdfparser not installed'); return ''; } try { $parser = new PdfParser(); $pdf = $parser->parseFile($filePath); $text = $pdf->getText(); return is_string($text) ? $text : ''; } catch (\Throwable $e) { \think\Log::warning('UserInfoFromFile PDF extract: ' . $e->getMessage()); return ''; } } private function extractDocxText($filePath) { $text = $this->tryExtractDocxText($filePath); if ($text === '') { throw new Exception('Word 文档文本提取失败'); } return $text; } private function tryExtractDocxText($filePath) { if (!function_exists('docxReader')) { return ''; } try { $text = docxReader($filePath); return is_string($text) ? $text : ''; } catch (\Throwable $e) { \think\Log::warning('UserInfoFromFile docxReader: ' . $e->getMessage()); return ''; } } private function looksLikeBinary($raw) { if (strpos($raw, "\0") !== false) { return true; } $sample = substr($raw, 0, 4096); if ($sample === '') { return false; } $nonPrintable = preg_match_all('/[\x00-\x08\x0B\x0C\x0E-\x1F]/', $sample); return $nonPrintable > 10; } private function parseUserInfoWithLlm($text) { $system = <<<'PROMPT' 你是学术人员信息抽取助手。根据用户上传的简历、简介或申请表文本,提取可用于期刊投稿/审稿人档案的基本信息。 只输出一个 JSON 对象,不要 markdown,不要解释。字段说明: - realname: 中文姓名;若只有拼音可填拼音;无法判断则空字符串 - email: 邮箱 - phone: 手机或电话 - orcid: ORCID 号(仅数字与连字符) - technical: 职称/职务(如教授、副主任护师、医学博士) - field: 主要研究领域,中文,1~3 个词用顿号分隔 - company: 当前主要任职机构/工作单位(必填优先提取;填完整机构名,如「中南大学湘雅二医院」「暨南大学」,不要只写科室;多机构取最主要或最近一条) - department: 所在科室/部门/学院(如感染科、药学院) - country: 国家,默认中国可填「中国」 - introduction: 100字以内中文简介(职务+机构+方向) 缺失字段用空字符串 "",不要编造。机构信息请从「工作单位」「任职」「所在单位」「affiliation」等段落提取。 PROMPT; \think\Log::info('UserInfoFromFile user head: ' . mb_substr($text, 0, 600)); $raw = $this->llm->requestChat([ ['role' => 'system', 'content' => $system], ['role' => 'user', 'content' => "请从以下文本提取用户信息:\n\n" . $text], ], 0.1); if ($raw === null) { throw new Exception('大模型请求失败,请检查 .env 中 [promotion] PROMOTION_LLM_URL / PROMOTION_LLM_MODEL(与参考文献校对相同)'); } $parsed = $this->llm->parseJsonResponse($raw); if ($parsed === null) { throw new Exception('大模型返回无法解析为 JSON:' . mb_substr($raw, 0, 200)); } return [ 'user_info' => $this->normalizeUserInfo($parsed), 'raw' => $raw, ]; } private function normalizeUserInfo(array $parsed) { $keys = ['realname', 'email', 'phone', 'orcid', 'technical', 'field', 'company', 'department', 'country', 'introduction']; $out = []; foreach ($keys as $k) { $out[$k] = trim((string) ($parsed[$k] ?? '')); } // 机构:兼容大模型可能返回的别名字段 if ($out['company'] === '') { foreach (['机构', '单位', 'institution', 'affiliation', 'organization', 'org', '工作单位', '任职单位'] as $alias) { $val = trim((string) ($parsed[$alias] ?? '')); if ($val !== '') { $out['company'] = $val; break; } } } if ($out['orcid'] !== '') { $out['orcid'] = preg_replace('/\s+/', '', $out['orcid']); } // 对外同时返回 institution(与 company 同值) $out['institution'] = $out['company']; return $out; } private function sanitizeUtf8($text) { $text = (string) $text; if (function_exists('mb_convert_encoding')) { $text = mb_convert_encoding($text, 'UTF-8', 'UTF-8'); } if (function_exists('iconv')) { $fixed = @iconv('UTF-8', 'UTF-8//IGNORE', $text); if (is_string($fixed)) { $text = $fixed; } } return $text; } }