diff --git a/par-api/src/main/java/com/par/api/controller/AdapterController.java b/par-api/src/main/java/com/par/api/controller/AdapterController.java new file mode 100644 index 0000000..34b3d09 --- /dev/null +++ b/par-api/src/main/java/com/par/api/controller/AdapterController.java @@ -0,0 +1,87 @@ +package com.par.api.controller; + +import com.fasterxml.jackson.databind.ObjectMapper; +import com.par.core.adapter.AdapterConfig; +import com.par.core.adapter.AdapterEngine; +import com.par.core.dto.ApiResponse; +import com.par.core.entity.AccessResult; +import com.par.core.entity.SiteConfig; +import com.par.core.mapper.AccessResultMapper; +import com.par.core.mapper.SiteConfigMapper; +import com.par.core.storage.CachedStorageService; +import lombok.RequiredArgsConstructor; +import lombok.extern.slf4j.Slf4j; +import org.springframework.web.bind.annotation.*; + +import java.util.Map; + +/** + * 适配器解析控制器 + */ +@Slf4j +@RestController +@RequiredArgsConstructor +public class AdapterController { + + private final AdapterEngine adapterEngine; + private final SiteConfigMapper siteConfigMapper; + private final AccessResultMapper accessResultMapper; + private final CachedStorageService storageService; + private static final ObjectMapper objectMapper = new ObjectMapper(); + + /** + * 用指定适配配置解析指定访问结果 + */ + @PostMapping("/api/v1/admin/adapter/parse") + public ApiResponse> parse( + @RequestBody Map payload) { + + Long configId = toLong(payload.get("configId")); + Long accessResultId = toLong(payload.get("accessResultId")); + + if (configId == null || accessResultId == null) { + return ApiResponse.error(400, "configId and accessResultId are required"); + } + + // 加载适配配置 + SiteConfig siteConfig = siteConfigMapper.selectById(configId); + if (siteConfig == null) return ApiResponse.error(404, "配置不存在"); + + String configJson; + try { configJson = storageService.read(siteConfig.getStoragePath()); } + catch (Exception e) { return ApiResponse.error(500, "读取配置失败: " + e.getMessage()); } + if (configJson == null) return ApiResponse.error(404, "配置内容为空"); + + AdapterConfig adapterConfig; + try { adapterConfig = objectMapper.readValue(configJson, AdapterConfig.class); } + catch (Exception e) { return ApiResponse.error(400, "配置格式无效: " + e.getMessage()); } + + // 加载访问结果 HTML + AccessResult accessResult = accessResultMapper.selectById(accessResultId); + if (accessResult == null) return ApiResponse.error(404, "访问结果不存在"); + + String html; + try { html = storageService.read(accessResult.getStoragePath()); } + catch (Exception e) { return ApiResponse.error(500, "读取访问结果失败: " + e.getMessage()); } + if (html == null) return ApiResponse.error(404, "访问结果内容为空"); + + // 执行解析 + Map result; + try { + result = adapterEngine.extract(adapterConfig, html, accessResult.getUrl()); + } catch (Exception e) { + log.error("Adapter parse error", e); + return ApiResponse.error(500, "解析异常: " + e.getMessage()); + } + + return ApiResponse.success(result); + } + + private Long toLong(Object val) { + if (val instanceof Number n) return n.longValue(); + if (val instanceof String s) { + try { return Long.parseLong(s); } catch (NumberFormatException ignored) {} + } + return null; + } +} diff --git a/par-core/pom.xml b/par-core/pom.xml index 4dba60e..17210c4 100644 --- a/par-core/pom.xml +++ b/par-core/pom.xml @@ -96,5 +96,11 @@ qiniu-java-sdk 7.15.0 + + + + org.jsoup + jsoup + diff --git a/par-core/src/main/java/com/par/core/adapter/AdapterConfig.java b/par-core/src/main/java/com/par/core/adapter/AdapterConfig.java new file mode 100644 index 0000000..ae6f6c5 --- /dev/null +++ b/par-core/src/main/java/com/par/core/adapter/AdapterConfig.java @@ -0,0 +1,22 @@ +package com.par.core.adapter; + +import com.fasterxml.jackson.annotation.JsonIgnoreProperties; +import lombok.Data; + +import java.util.Map; + +/** + * 适配器配置根对象 + */ +@Data +@JsonIgnoreProperties(ignoreUnknown = true) +public class AdapterConfig { + /** 站点 ID */ + private String siteId; + /** 站点类型 */ + private String siteType; + /** 域名列表 */ + private String[] domains; + /** 页面解析规则 */ + private Map pages; +} diff --git a/par-core/src/main/java/com/par/core/adapter/AdapterEngine.java b/par-core/src/main/java/com/par/core/adapter/AdapterEngine.java new file mode 100644 index 0000000..3e86518 --- /dev/null +++ b/par-core/src/main/java/com/par/core/adapter/AdapterEngine.java @@ -0,0 +1,211 @@ +package com.par.core.adapter; + +import com.fasterxml.jackson.databind.ObjectMapper; +import lombok.extern.slf4j.Slf4j; +import org.jsoup.Jsoup; +import org.jsoup.nodes.Document; +import org.jsoup.nodes.Element; +import org.jsoup.select.Elements; +import org.springframework.stereotype.Service; + +import java.util.*; +import java.util.regex.Matcher; +import java.util.regex.Pattern; +import java.util.stream.Collectors; + +/** + * 适配器解析引擎:根据配置文件的 CSS 选择器从 HTML 中提取数据 + */ +@Slf4j +@Service +public class AdapterEngine { + + private static final ObjectMapper objectMapper = new ObjectMapper(); + + /** + * 解析单个 HTML 页面,匹配配置中的页面规则并提取数据 + * @return { pageName, pageUrl, extracted: [{field: value, ...}, ...] } + */ + public Map extract(AdapterConfig config, String html, String accessUrl) { + Map result = new LinkedHashMap<>(); + Document doc = Jsoup.parse(html); + + // 匹配页面 + Map.Entry matched = matchPage(config.getPages(), accessUrl); + if (matched == null) { + result.put("matched", false); + result.put("message", "未匹配到页面规则"); + result.put("availablePages", config.getPages().keySet()); + return result; + } + + String pageName = matched.getKey(); + PageConfig page = matched.getValue(); + result.put("matched", true); + result.put("pageName", pageName); + result.put("pageUrl", accessUrl); + + // 提取数据 + List> extracted = extractRows(doc, page); + result.put("extracted", extracted); + result.put("rowCount", extracted.size()); + result.put("fieldCount", page.getFields() != null ? page.getFields().size() : 0); + + return result; + } + + /** + * 匹配 URL 对应的页面规则 + */ + private Map.Entry matchPage(Map pages, String accessUrl) { + if (pages == null || pages.isEmpty() || accessUrl == null) return null; + + for (Map.Entry entry : pages.entrySet()) { + PageConfig page = entry.getValue(); + + // 精确 URL 匹配 + if (page.getUrl() != null) { + String normalizedPageUrl = page.getUrl().replaceAll("\\?.+", ""); + String normalizedAccessUrl = accessUrl.replaceAll("\\?.+", ""); + if (normalizedAccessUrl.contains(normalizedPageUrl) + || normalizedPageUrl.contains(normalizedAccessUrl)) { + return entry; + } + } + + // URL 正则匹配 + if (page.getUrlPattern() != null) { + try { + if (Pattern.compile(page.getUrlPattern()).matcher(accessUrl).find()) { + return entry; + } + } catch (Exception ignored) {} + } + + // 模糊匹配:URL 路径中包含页面 key + if (accessUrl.contains("/" + entry.getKey())) { + return entry; + } + } + + return null; + } + + /** + * 从文档中提取所有行的数据 + */ + @SuppressWarnings("unchecked") + private List> extractRows(Document doc, PageConfig page) { + if (page.getFields() == null || page.getFields().isEmpty()) return List.of(); + + Elements rows; + if (page.getListContainer() != null && !page.getListContainer().isBlank()) { + rows = doc.select(page.getListContainer()); + } else { + // 无列表容器 → 整页提取一行 + rows = new Elements(); + rows.add(doc); + } + + List> results = new ArrayList<>(); + int skip = Math.max(0, page.getSkipRows()); + for (int i = skip; i < rows.size(); i++) { + Element row = rows.get(i); + Map rowData = new LinkedHashMap<>(); + boolean hasData = false; + + for (Map.Entry field : page.getFields().entrySet()) { + String fieldName = field.getKey(); + FieldExtractor fe = field.getValue(); + if (fe.getSelector() == null) continue; + + Element el = row.selectFirst(fe.getSelector()); + if (el == null) { + if (fe.getOptional() == null || !fe.getOptional()) { + rowData.put(fieldName, null); + } + continue; + } + + String raw; + String type = fe.getType() != null ? fe.getType() : "text"; + switch (type) { + case "attribute": + raw = fe.getAttribute() != null ? el.attr(fe.getAttribute()) : el.text(); + break; + case "html": + raw = el.html(); + break; + default: + raw = el.text(); + } + + Object value = applyTransform(raw, fe.getTransform()); + rowData.put(fieldName, value); + hasData = true; + } + + if (hasData) results.add(rowData); + } + + return results; + } + + /** + * 应用内置转换器 + */ + private Object applyTransform(String raw, String transform) { + if (raw == null || transform == null) return raw; + + try { + return switch (transform.toLowerCase()) { + case "int", "parseint" -> { + String clean = raw.replaceAll("[^\\d-]", ""); + yield clean.isEmpty() ? null : Integer.parseInt(clean); + } + case "long" -> { + String clean = raw.replaceAll("[^\\d-]", ""); + yield clean.isEmpty() ? null : Long.parseLong(clean); + } + case "double", "float" -> { + String clean = raw.replaceAll("[^\\d.-]", ""); + yield clean.isEmpty() ? null : Double.parseDouble(clean); + } + case "filesize" -> parseFileSize(raw); + case "boolean", "bool" -> parseBoolean(raw); + default -> raw; + }; + } catch (Exception e) { + return raw; + } + } + + private Long parseFileSize(String raw) { + if (raw == null) return null; + Matcher m = Pattern.compile("([\\d.]+)\\s*(GB|MB|KB|TB|B|GiB|MiB|KiB|TiB)", + Pattern.CASE_INSENSITIVE).matcher(raw); + if (!m.find()) { + // try just number + String num = raw.replaceAll("[^\\d.]", ""); + if (!num.isEmpty()) return (long) Double.parseDouble(num); + return null; + } + double val = Double.parseDouble(m.group(1)); + String unit = m.group(2).toUpperCase(); + return (long) (val * switch (unit) { + case "TB", "TIB" -> 1024L * 1024 * 1024 * 1024; + case "GB", "GIB" -> 1024L * 1024 * 1024; + case "MB", "MIB" -> 1024L * 1024; + case "KB", "KIB" -> 1024L; + default -> 1L; + }); + } + + private Boolean parseBoolean(String raw) { + if (raw == null) return null; + String s = raw.trim().toLowerCase(); + if (s.equals("true") || s.equals("yes") || s.equals("1") || s.equals("是")) return true; + if (s.equals("false") || s.equals("no") || s.equals("0") || s.equals("否")) return false; + return null; + } +} diff --git a/par-core/src/main/java/com/par/core/adapter/FieldExtractor.java b/par-core/src/main/java/com/par/core/adapter/FieldExtractor.java new file mode 100644 index 0000000..44481e4 --- /dev/null +++ b/par-core/src/main/java/com/par/core/adapter/FieldExtractor.java @@ -0,0 +1,22 @@ +package com.par.core.adapter; + +import com.fasterxml.jackson.annotation.JsonIgnoreProperties; +import lombok.Data; + +/** + * 字段提取器定义 + */ +@Data +@JsonIgnoreProperties(ignoreUnknown = true) +public class FieldExtractor { + /** CSS 选择器 */ + private String selector; + /** 提取类型: text | attribute | html */ + private String type = "text"; + /** type=attribute 时指定属性名 */ + private String attribute; + /** 转换器名称 */ + private String transform; + /** 是否可选 */ + private Boolean optional; +} diff --git a/par-core/src/main/java/com/par/core/adapter/PageConfig.java b/par-core/src/main/java/com/par/core/adapter/PageConfig.java new file mode 100644 index 0000000..82c7b45 --- /dev/null +++ b/par-core/src/main/java/com/par/core/adapter/PageConfig.java @@ -0,0 +1,26 @@ +package com.par.core.adapter; + +import com.fasterxml.jackson.annotation.JsonIgnoreProperties; +import lombok.Data; + +import java.util.Map; + +/** + * 页面解析配置 + */ +@Data +@JsonIgnoreProperties(ignoreUnknown = true) +public class PageConfig { + /** 页面 URL 或 URL 模板 */ + private String url; + /** URL 正则匹配模式 */ + private String urlPattern; + /** 请求方法 */ + private String method = "GET"; + /** 列表容器 CSS 选择器 */ + private String listContainer; + /** 跳过行数 */ + private int skipRows; + /** 字段定义 map */ + private Map fields; +} diff --git a/pom.xml b/pom.xml index 8d17d1f..5f267dc 100644 --- a/pom.xml +++ b/pom.xml @@ -117,6 +117,13 @@ springdoc-openapi-starter-webmvc-ui ${springdoc.version} + + + + org.jsoup + jsoup + 1.18.1 + diff --git a/web/src/api/index.js b/web/src/api/index.js index 6684e21..d9f1916 100644 --- a/web/src/api/index.js +++ b/web/src/api/index.js @@ -86,5 +86,8 @@ export default { bySite: (siteId) => api.get(`/admin/error-logs/by-site/${siteId}`), get: (id) => api.get(`/admin/error-logs/${id}`), }, + adapter: { + parse: (configId, accessResultId) => api.post('/admin/adapter/parse', { configId, accessResultId }), + }, health: () => api.get('/health', { baseURL: '' }) } diff --git a/web/src/views/AccessResultCheck.vue b/web/src/views/AccessResultCheck.vue index 7395b81..f9ef82c 100644 --- a/web/src/views/AccessResultCheck.vue +++ b/web/src/views/AccessResultCheck.vue @@ -113,6 +113,7 @@ 有效 无效 查看 + 解析
@@ -145,6 +146,36 @@ 无效 + + + +
解析中...
+
{{ parseError }}
+ + +
@@ -175,6 +206,11 @@ const viewingResult = ref({}) const viewingContentType = ref('html') const viewingContent = ref('') +const parseVisible = ref(false) +const parseLoading = ref(false) +const parseError = ref('') +const parseData = ref({}) + const expandedUrls = reactive(new Set()) const MAX_VISIBLE = 5 @@ -322,6 +358,25 @@ const singleValidate = async (id, isValid) => { } catch { ElMessage.error('校验失败') } } +const parseColumns = computed(() => { + if (!parseData.value.extracted?.length) return [] + return Object.keys(parseData.value.extracted[0]) +}) + +const parseResult = async (row) => { + if (!selectedConfig.value) { ElMessage.warning('请先选择适配配置'); return } + parseVisible.value = true + parseLoading.value = true + parseError.value = '' + parseData.value = {} + try { + const r = await api.adapter.parse(selectedConfig.value.id, row.id) + parseData.value = r.data || {} + } catch (e) { + parseError.value = e.response?.data?.message || '解析失败' + } finally { parseLoading.value = false } +} + const viewContent = async (row) => { viewingResult.value = { ...row } viewingContentType.value = row.contentType || 'html'