test(gov-policy): JSDOM-against-frozen-fixture tests for in-browser extractors (#1340)
* test(gov-policy): JSDOM-against-frozen-fixture tests for in-browser extractors Applies the pattern documented in skills/opencli-adapter-author/references/jsdom-fixture-pattern.md (introduced in #1319 alongside the dianping reference test in #1313) to the gov-policy adapter. Refactor: the inline IIFE inside `page.evaluate` template literal is hoisted to a top-level `extractSearchRows` / `extractRecentRows` function using bare `document` / `location`. Same code now runs identically in: - the live browser (injected via `${extractor.toString()}`) - JSDOM unit tests (with `globalThis.document` / `globalThis.location` swapped) Tests: - 6 new cases in clis/gov-policy/gov-policy.test.js (was commands.test.js). - 3 representative search result cards (1 with real article snippet, 2 with only publish-time in `.description`) and 5 recent listing rows in the fixtures. - ok:false fallback path covered for both extractors. - Lock-in: `要闻` type-tag prefix fusion in title and empty-source contract on recent listings (no `.source` / `.from` elements on that page) are asserted explicitly so a future selector tweak can't silently change them. Reverse-validated against two buggy variants per the reference doc: breaking the title selector and stripping the `要闻` prefix both fail the JSDOM assertions with helpful diffs. Fixture sanitization follows the reference doc step-by-step: scripts / styles / iframes / comments / preload links stripped, image srcs replaced with `placeholder.png`, trimmed to the minimum subtree that exercises the extractor (3 search items, 5 recent rows), all whitespace-only lines removed. * fix(gov-policy): use typed errors for touched commands --------- Co-authored-by: jackwener <jakevingoo@gmail.com>
This commit is contained in:
@@ -0,0 +1,16 @@
|
||||
<!DOCTYPE html>
|
||||
<html lang="zh">
|
||||
<head>
|
||||
<meta charset="UTF-8">
|
||||
<title>国务院最新政策文件</title>
|
||||
</head>
|
||||
<body>
|
||||
<div class="news_box">
|
||||
<div class="list list_1 list_2">
|
||||
<ul id="list-1-ajax-id">
|
||||
<li><h4><a href="https://www.gov.cn/zhengce/202604/content_7066998.htm" target="_blank">中共中央办公厅 国务院办公厅关于加强新就业群体服务管理的意见</a><span class="date">2026-04-26</span></h4></li><li><h4><a href="https://www.gov.cn/zhengce/202604/content_7066695.htm" target="_blank">中共中央办公厅 国务院办公厅印发《碳达峰碳中和综合评价考核办法》</a><span class="date">2026-04-23</span></h4></li><li><h4><a href="https://www.gov.cn/zhengce/202604/content_7066623.htm" target="_blank">中办国办关于更高水平更高质量做好节能降碳工作的意见</a><span class="date">2026-04-22</span></h4></li><li><h4><a href="https://www.gov.cn/zhengce/content/202604/content_7066483.htm" target="_blank">国务院关于推进服务业扩能提质的意见</a><span class="date">2026-04-21</span></h4></li><li><h4 class="line"><a href="https://www.gov.cn/zhengce/content/202604/content_7066113.htm" target="_blank">国务院办公厅转发海关总署《关于促进综合保税区扩能提质的若干措施》的通知</a><span class="date">2026-04-17</span></h4></li>
|
||||
</ul>
|
||||
</div>
|
||||
</div>
|
||||
</body>
|
||||
</html>
|
||||
@@ -0,0 +1,41 @@
|
||||
<!DOCTYPE html>
|
||||
<html lang="zh">
|
||||
<head>
|
||||
<meta charset="UTF-8">
|
||||
<title>中国政府网站内搜索 - 数字经济</title>
|
||||
</head>
|
||||
<body>
|
||||
<div class="basic_result_content js_basic_result_content clearfix">
|
||||
<div class="item hasImg is-news" data-documentid="thirdparty_code_107_t_17e00f1efbb_25992758">
|
||||
<div class="partitions"></div>
|
||||
<a class="title log-anchor" title="经济数据速览:10组数字看一季度中国经济" href="https://www.gov.cn/zhengce/jiedu/tujie/202604/content_7065945.htm" target="_blank">
|
||||
<span class="type_title">要闻</span><em>经济</em>数据速览:10组<em>数字</em>看一季度中国<em>经济</em>
|
||||
</a>
|
||||
<div class="description">
|
||||
<img src="placeholder.png" aria-label="经济数据速览:10组数字看一季度中国经济">
|
||||
<span class="sourceTime">发布时间:2026-4-16</span>
|
||||
</div>
|
||||
</div><div class="item hasImg is-news" data-documentid="thirdparty_code_107_t_17e00f1efbb_25839485">
|
||||
<div class="partitions"></div>
|
||||
<a class="title log-anchor" title="何立峰会见法国经济、财政和工业、能源与数字主权部部长莱斯屈尔" href="https://www.gov.cn/yaowen/liebiao/202603/content_7062985.htm" target="_blank">
|
||||
<span class="type_title">要闻</span>何立峰会见法国<em>经济</em>、财政和工业、能源与<em>数字</em>主权部部长莱斯屈尔
|
||||
</a>
|
||||
<div class="description">
|
||||
<div class="detail js_limit_text">
|
||||
<p class="js_text line_clamp_two">新华社巴黎3月16日电 当地时间3月16日,中法高级别经济财金对话中方牵头人、国务院副总理何立峰在中美经贸巴黎磋商之后应约会见对话法方牵头人,法国经济、财政和工业、能源与数字主权部部长莱斯屈尔,就近期中法经济财金合作和双方共同关心的问题深入交换意见。 何立峰表示,中方愿与法方一道,落实好习近平主席与马克龙总统重要共识,进一步深化中法经济财金各领域交流与合作,推动中法经济关系行稳致远。何立峰还应询介</p>
|
||||
</div>
|
||||
<span class="sourceTime">发布时间:2026-3-17</span>
|
||||
</div>
|
||||
</div><div class="item hasImg is-news" data-documentid="thirdparty_code_107_t_17e00f1efbb_25836739">
|
||||
<div class="partitions"></div>
|
||||
<a class="title log-anchor" title="经济数据速览:7组数字看1—2月份中国经济" href="https://www.gov.cn/zhengce/jiedu/tujie/202603/content_7062831.htm" target="_blank">
|
||||
<span class="type_title">要闻</span><em>经济</em>数据速览:7组<em>数字</em>看1—2月份中国<em>经济</em>
|
||||
</a>
|
||||
<div class="description">
|
||||
<img src="placeholder.png" aria-label="经济数据速览:7组数字看1—2月份中国经济">
|
||||
<span class="sourceTime">发布时间:2026-3-16</span>
|
||||
</div>
|
||||
</div>
|
||||
</div>
|
||||
</body>
|
||||
</html>
|
||||
@@ -1,27 +0,0 @@
|
||||
import { describe, expect, it, vi } from 'vitest';
|
||||
import { getRegistry } from '@jackwener/opencli/registry';
|
||||
import './search.js';
|
||||
import './recent.js';
|
||||
|
||||
describe('gov-policy commands', () => {
|
||||
const search = getRegistry().get('gov-policy/search');
|
||||
const recent = getRegistry().get('gov-policy/recent');
|
||||
|
||||
it('registers both commands as public browser commands', () => {
|
||||
expect(search).toBeDefined();
|
||||
expect(recent).toBeDefined();
|
||||
expect(search.browser).toBe(true);
|
||||
expect(recent.browser).toBe(true);
|
||||
expect(search.strategy).toBe('public');
|
||||
expect(recent.strategy).toBe('public');
|
||||
});
|
||||
|
||||
it('rejects empty search queries before browser navigation', async () => {
|
||||
const page = { goto: vi.fn() };
|
||||
await expect(search.func(page, { query: ' ' })).rejects.toMatchObject({
|
||||
name: 'ArgumentError',
|
||||
code: 'ARGUMENT',
|
||||
});
|
||||
expect(page.goto).not.toHaveBeenCalled();
|
||||
});
|
||||
});
|
||||
@@ -0,0 +1,224 @@
|
||||
import { readFileSync } from 'node:fs';
|
||||
import { dirname, join } from 'node:path';
|
||||
import { fileURLToPath } from 'node:url';
|
||||
import { JSDOM } from 'jsdom';
|
||||
import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest';
|
||||
import {
|
||||
ArgumentError,
|
||||
CommandExecutionError,
|
||||
EmptyResultError,
|
||||
} from '@jackwener/opencli/errors';
|
||||
import { getRegistry } from '@jackwener/opencli/registry';
|
||||
import { extractSearchRows } from './search.js';
|
||||
import { extractRecentRows } from './recent.js';
|
||||
|
||||
const __dirname = dirname(fileURLToPath(import.meta.url));
|
||||
const SEARCH_FIXTURE = readFileSync(join(__dirname, '__fixtures__/search.html'), 'utf8');
|
||||
const RECENT_FIXTURE = readFileSync(join(__dirname, '__fixtures__/recent.html'), 'utf8');
|
||||
|
||||
function createPageMock(evaluateResult, overrides = {}) {
|
||||
const evaluate = typeof evaluateResult === 'function'
|
||||
? vi.fn(evaluateResult)
|
||||
: vi.fn().mockResolvedValue(evaluateResult);
|
||||
return {
|
||||
goto: vi.fn().mockResolvedValue(undefined),
|
||||
wait: vi.fn().mockResolvedValue(undefined),
|
||||
evaluate,
|
||||
...overrides,
|
||||
};
|
||||
}
|
||||
|
||||
describe('gov-policy commands — registration', () => {
|
||||
it('registers search and recent as public browser commands', () => {
|
||||
const search = getRegistry().get('gov-policy/search');
|
||||
const recent = getRegistry().get('gov-policy/recent');
|
||||
|
||||
expect(search).toBeDefined();
|
||||
expect(recent).toBeDefined();
|
||||
expect(search.browser).toBe(true);
|
||||
expect(recent.browser).toBe(true);
|
||||
expect(search.strategy).toBe('public');
|
||||
expect(recent.strategy).toBe('public');
|
||||
expect(search.columns).toEqual(['rank', 'title', 'description', 'date', 'url']);
|
||||
expect(recent.columns).toEqual(['rank', 'title', 'date', 'source', 'url']);
|
||||
});
|
||||
|
||||
it('rejects empty search queries before browser navigation', async () => {
|
||||
const search = getRegistry().get('gov-policy/search');
|
||||
const page = { goto: vi.fn() };
|
||||
await expect(search.func(page, { query: ' ' })).rejects.toThrow(ArgumentError);
|
||||
expect(page.goto).not.toHaveBeenCalled();
|
||||
});
|
||||
|
||||
it('rejects invalid limits before browser navigation', async () => {
|
||||
const search = getRegistry().get('gov-policy/search');
|
||||
const recent = getRegistry().get('gov-policy/recent');
|
||||
const page = createPageMock({ ok: true, rows: [] });
|
||||
|
||||
await expect(search.func(page, { query: '数字经济', limit: '0' })).rejects.toThrow(ArgumentError);
|
||||
await expect(search.func(page, { query: '数字经济', limit: '1.5' })).rejects.toThrow(ArgumentError);
|
||||
await expect(search.func(page, { query: '数字经济', limit: '21' })).rejects.toThrow(ArgumentError);
|
||||
await expect(recent.func(page, { limit: 'abc' })).rejects.toThrow(ArgumentError);
|
||||
expect(page.goto).not.toHaveBeenCalled();
|
||||
expect(page.evaluate).not.toHaveBeenCalled();
|
||||
});
|
||||
|
||||
it('maps empty search pages, selector drift, and browser failures to typed errors', async () => {
|
||||
const search = getRegistry().get('gov-policy/search');
|
||||
const recent = getRegistry().get('gov-policy/recent');
|
||||
|
||||
await expect(search.func(createPageMock({
|
||||
ok: false,
|
||||
sample: '很抱歉,没有找到与 数字经济 相关的结果',
|
||||
url: 'https://sousuo.www.gov.cn/sousuo/search.shtml?searchWord=x',
|
||||
}), { query: '数字经济' })).rejects.toThrow(EmptyResultError);
|
||||
|
||||
await expect(recent.func(createPageMock({
|
||||
ok: false,
|
||||
sample: '<main>unexpected government page shell</main>',
|
||||
url: 'https://www.gov.cn/zhengce/zuixin/index.htm',
|
||||
}), {})).rejects.toThrow(CommandExecutionError);
|
||||
|
||||
await expect(search.func(createPageMock(
|
||||
{ ok: true, rows: [] },
|
||||
{ goto: vi.fn().mockRejectedValue(new Error('browser disconnected')) },
|
||||
), { query: '数字经济' })).rejects.toThrow(CommandExecutionError);
|
||||
});
|
||||
});
|
||||
|
||||
/**
|
||||
* In-browser DOM extractors against frozen sanitized HTML fixtures.
|
||||
*
|
||||
* Mocked-page.evaluate tests can't catch silent bugs that live inside the
|
||||
* extractor, since they feed pre-baked results to the func and the real
|
||||
* DOM walk never runs.
|
||||
*
|
||||
* These tests replay the real (sanitized) HTML through JSDOM so changes
|
||||
* to the extractor logic that re-introduce a silent regression fail in CI.
|
||||
*/
|
||||
describe('gov-policy adapter — extractors against frozen HTML fixtures', () => {
|
||||
let originalDocument;
|
||||
let originalLocation;
|
||||
|
||||
beforeEach(() => {
|
||||
originalDocument = globalThis.document;
|
||||
originalLocation = globalThis.location;
|
||||
});
|
||||
|
||||
afterEach(() => {
|
||||
globalThis.document = originalDocument;
|
||||
globalThis.location = originalLocation;
|
||||
});
|
||||
|
||||
function loadFixture(html, url) {
|
||||
const dom = new JSDOM(html, { url });
|
||||
globalThis.document = dom.window.document;
|
||||
globalThis.location = dom.window.location;
|
||||
return dom;
|
||||
}
|
||||
|
||||
it('extractSearchRows returns three result-shaped rows with title, description, date, url', () => {
|
||||
loadFixture(SEARCH_FIXTURE, 'https://sousuo.www.gov.cn/sousuo/search.shtml?searchWord=%E6%95%B0%E5%AD%97%E7%BB%8F%E6%B5%8E');
|
||||
|
||||
const result = extractSearchRows();
|
||||
|
||||
expect(result.ok).toBe(true);
|
||||
expect(result.rows).toHaveLength(3);
|
||||
|
||||
// Rank 1 + 3 are the homogeneous "type tag + emphasized title" cards
|
||||
// whose .description div carries only the date span (no real snippet).
|
||||
expect(result.rows[0]).toMatchObject({
|
||||
rank: 1,
|
||||
title: '要闻经济数据速览:10组数字看一季度中国经济',
|
||||
date: '2026-4-16',
|
||||
url: 'https://www.gov.cn/zhengce/jiedu/tujie/202604/content_7065945.htm',
|
||||
});
|
||||
// Description for these rows is just the publish-time line.
|
||||
expect(result.rows[0].description).toContain('发布时间');
|
||||
expect(result.rows[0].description).toContain('2026-4-16');
|
||||
|
||||
// Rank 2 has a real article snippet inside .description > .detail > p,
|
||||
// and the extractor must capture it (sliced to 120 chars).
|
||||
expect(result.rows[1]).toMatchObject({
|
||||
rank: 2,
|
||||
title: '要闻何立峰会见法国经济、财政和工业、能源与数字主权部部长莱斯屈尔',
|
||||
date: '2026-3-17',
|
||||
url: 'https://www.gov.cn/yaowen/liebiao/202603/content_7062985.htm',
|
||||
});
|
||||
expect(result.rows[1].description.length).toBeLessThanOrEqual(120);
|
||||
expect(result.rows[1].description).toContain('新华社巴黎');
|
||||
|
||||
expect(result.rows[2]).toMatchObject({
|
||||
rank: 3,
|
||||
title: '要闻经济数据速览:7组数字看1—2月份中国经济',
|
||||
date: '2026-3-16',
|
||||
url: 'https://www.gov.cn/zhengce/jiedu/tujie/202603/content_7062831.htm',
|
||||
});
|
||||
|
||||
// Lock the no-collapse contract on the title: the type_title prefix
|
||||
// ('要闻') is fused into the textContent because we read the whole <a>.
|
||||
// If a future refactor strips the prefix, this assertion catches it
|
||||
// before any field-level downstream surprises.
|
||||
for (const row of result.rows) {
|
||||
expect(row.title.startsWith('要闻')).toBe(true);
|
||||
}
|
||||
});
|
||||
|
||||
it('extractRecentRows returns five rows with title, date, url and empty source', () => {
|
||||
loadFixture(RECENT_FIXTURE, 'https://www.gov.cn/zhengce/zuixin/index.htm');
|
||||
|
||||
const result = extractRecentRows();
|
||||
|
||||
expect(result.ok).toBe(true);
|
||||
expect(result.rows).toHaveLength(5);
|
||||
|
||||
expect(result.rows[0]).toMatchObject({
|
||||
rank: 1,
|
||||
title: '中共中央办公厅 国务院办公厅关于加强新就业群体服务管理的意见',
|
||||
date: '2026-04-26',
|
||||
url: 'https://www.gov.cn/zhengce/202604/content_7066998.htm',
|
||||
});
|
||||
expect(result.rows[1]).toMatchObject({
|
||||
rank: 2,
|
||||
date: '2026-04-23',
|
||||
});
|
||||
expect(result.rows[4]).toMatchObject({
|
||||
rank: 5,
|
||||
date: '2026-04-17',
|
||||
});
|
||||
|
||||
// gov.cn/zhengce/zuixin layout has no .source / .from elements, so
|
||||
// the source field is always an empty string. Lock that contract:
|
||||
// a future selector change that picks up unrelated text would break
|
||||
// it and fail this assertion.
|
||||
for (const row of result.rows) {
|
||||
expect(row.source).toBe('');
|
||||
}
|
||||
});
|
||||
|
||||
it('extractSearchRows signals ok:false with a sample when result list is empty', () => {
|
||||
loadFixture(
|
||||
'<html><head><title>blocked</title></head><body><main>访问受限,请稍后再试</main></body></html>',
|
||||
'https://sousuo.www.gov.cn/sousuo/search.shtml?searchWord=zzz',
|
||||
);
|
||||
|
||||
const result = extractSearchRows();
|
||||
|
||||
expect(result.ok).toBe(false);
|
||||
expect(result.url).toBe('https://sousuo.www.gov.cn/sousuo/search.shtml?searchWord=zzz');
|
||||
expect(result.sample).toContain('访问受限');
|
||||
});
|
||||
|
||||
it('extractRecentRows signals ok:false with a sample when listing is empty', () => {
|
||||
loadFixture(
|
||||
'<html><head><title>not found</title></head><body><main>页面正在加载</main></body></html>',
|
||||
'https://www.gov.cn/zhengce/zuixin/index.htm',
|
||||
);
|
||||
|
||||
const result = extractRecentRows();
|
||||
|
||||
expect(result.ok).toBe(false);
|
||||
expect(result.url).toBe('https://www.gov.cn/zhengce/zuixin/index.htm');
|
||||
expect(result.sample).toContain('页面正在加载');
|
||||
});
|
||||
});
|
||||
+66
-23
@@ -1,5 +1,58 @@
|
||||
/**
|
||||
* gov-policy recent — latest State Council policy releases.
|
||||
*
|
||||
* Targets www.gov.cn/zhengce/zuixin/index.htm. The listing is rendered
|
||||
* server-side into one of `.news_box li`, `.list li`, `.list_item`,
|
||||
* `.news-list li` (the page rotates between layouts).
|
||||
*
|
||||
* The DOM extractor is defined as a top-level function and injected
|
||||
* into `page.evaluate` via `.toString()`, so the same code is exercised
|
||||
* by a JSDOM-against-frozen-fixture unit test (see gov-policy.test.js).
|
||||
*/
|
||||
|
||||
import { cli, Strategy } from '@jackwener/opencli/registry';
|
||||
import { clampInt } from '../_shared/common.js';
|
||||
import {
|
||||
classifyExtractorFailure,
|
||||
parseGovPolicyLimit,
|
||||
requireRows,
|
||||
wrapBrowserError,
|
||||
} from './utils.js';
|
||||
|
||||
/**
|
||||
* Pure DOM extractor for the gov-policy "latest policies" listing page.
|
||||
*
|
||||
* Uses bare `document` / `location` so it runs identically in:
|
||||
* - the live browser (injected via `${extractRecentRows.toString()}`)
|
||||
* - JSDOM unit tests (which swap `globalThis.document` / `globalThis.location`)
|
||||
*/
|
||||
export function extractRecentRows() {
|
||||
const normalize = (v) => (v || '').replace(/\s+/g, ' ').trim();
|
||||
const items = document.querySelectorAll('.news_box li, .list li, .list_item, .news-list li');
|
||||
if (items.length === 0) {
|
||||
const body = document.body;
|
||||
const sampleText = (body && (body.innerText || body.textContent)) || '';
|
||||
return {
|
||||
ok: false,
|
||||
sample: sampleText.slice(0, 800),
|
||||
url: location.href,
|
||||
};
|
||||
}
|
||||
const rows = [];
|
||||
for (const el of items) {
|
||||
const titleEl = el.querySelector('a');
|
||||
const title = normalize(titleEl?.textContent);
|
||||
if (!title || title.length < 4) continue;
|
||||
|
||||
let url = titleEl?.getAttribute('href') || '';
|
||||
if (url && !url.startsWith('http')) url = 'https://www.gov.cn' + url;
|
||||
|
||||
const date = (el.textContent || '').match(/(\d{4}[-./]\d{1,2}[-./]\d{1,2})/)?.[1] || '';
|
||||
const source = normalize(el.querySelector('.source, .from')?.textContent);
|
||||
|
||||
rows.push({ rank: rows.length + 1, title, date, source, url });
|
||||
}
|
||||
return { ok: true, rows };
|
||||
}
|
||||
|
||||
cli({
|
||||
site: 'gov-policy',
|
||||
@@ -14,34 +67,24 @@ cli({
|
||||
],
|
||||
columns: ['rank', 'title', 'date', 'source', 'url'],
|
||||
func: async (page, kwargs) => {
|
||||
const limit = clampInt(kwargs.limit, 10, 1, 20);
|
||||
await page.goto('https://www.gov.cn/zhengce/zuixin/index.htm');
|
||||
await page.wait(4);
|
||||
const data = await page.evaluate(`
|
||||
const limit = parseGovPolicyLimit(kwargs.limit, 'recent');
|
||||
try {
|
||||
await page.goto('https://www.gov.cn/zhengce/zuixin/index.htm');
|
||||
await page.wait(4);
|
||||
// Poll until the SSR listing mounts.
|
||||
await page.evaluate(`
|
||||
(async () => {
|
||||
const normalize = v => (v || '').replace(/\\s+/g, ' ').trim();
|
||||
for (let i = 0; i < 20; i++) {
|
||||
if (document.querySelector('.news_box li, .list li, .list_item, .news-list li')) break;
|
||||
await new Promise(r => setTimeout(r, 500));
|
||||
}
|
||||
const results = [];
|
||||
for (const el of document.querySelectorAll('.news_box li, .list li, .list_item, .news-list li')) {
|
||||
const titleEl = el.querySelector('a');
|
||||
const title = normalize(titleEl?.textContent);
|
||||
if (!title || title.length < 4) continue;
|
||||
|
||||
let url = titleEl?.getAttribute('href') || '';
|
||||
if (url && !url.startsWith('http')) url = 'https://www.gov.cn' + url;
|
||||
|
||||
const date = (el.textContent || '').match(/(\\d{4}[-./]\\d{1,2}[-./]\\d{1,2})/)?.[1] || '';
|
||||
const source = normalize(el.querySelector('.source, .from')?.textContent);
|
||||
|
||||
results.push({ rank: results.length + 1, title, date, source, url });
|
||||
if (results.length >= ${limit}) break;
|
||||
}
|
||||
return results;
|
||||
})()
|
||||
`);
|
||||
return Array.isArray(data) ? data : [];
|
||||
const result = await page.evaluate(`(${extractRecentRows.toString()})()`);
|
||||
if (!result || !result.ok) classifyExtractorFailure('recent', result);
|
||||
return requireRows('recent', result.rows).slice(0, limit);
|
||||
} catch (error) {
|
||||
wrapBrowserError('recent', error);
|
||||
}
|
||||
},
|
||||
});
|
||||
|
||||
+65
-22
@@ -1,5 +1,57 @@
|
||||
/**
|
||||
* gov-policy search — Chinese government policy full-text search.
|
||||
*
|
||||
* Targets sousuo.www.gov.cn. Results are server-rendered into
|
||||
* `.basic_result_content .item` cards.
|
||||
*
|
||||
* The DOM extractor is defined as a top-level function and injected
|
||||
* into `page.evaluate` via `.toString()`, so the same code is exercised
|
||||
* by a JSDOM-against-frozen-fixture unit test (see gov-policy.test.js).
|
||||
*/
|
||||
|
||||
import { cli, Strategy } from '@jackwener/opencli/registry';
|
||||
import { clampInt, requireNonEmptyQuery } from '../_shared/common.js';
|
||||
import { requireNonEmptyQuery } from '../_shared/common.js';
|
||||
import {
|
||||
classifyExtractorFailure,
|
||||
parseGovPolicyLimit,
|
||||
requireRows,
|
||||
wrapBrowserError,
|
||||
} from './utils.js';
|
||||
|
||||
/**
|
||||
* Pure DOM extractor for the gov-policy search-results page.
|
||||
*
|
||||
* Uses bare `document` / `location` so it runs identically in:
|
||||
* - the live browser (injected via `${extractSearchRows.toString()}`)
|
||||
* - JSDOM unit tests (which swap `globalThis.document` / `globalThis.location`)
|
||||
*/
|
||||
export function extractSearchRows() {
|
||||
const normalize = (v) => (v || '').replace(/\s+/g, ' ').trim();
|
||||
const items = document.querySelectorAll('.basic_result_content .item, .js_basic_result_content .item');
|
||||
if (items.length === 0) {
|
||||
const body = document.body;
|
||||
const sampleText = (body && (body.innerText || body.textContent)) || '';
|
||||
return {
|
||||
ok: false,
|
||||
sample: sampleText.slice(0, 800),
|
||||
url: location.href,
|
||||
};
|
||||
}
|
||||
const rows = [];
|
||||
for (const el of items) {
|
||||
const titleEl = el.querySelector('a.title, .title a, a.log-anchor');
|
||||
const title = normalize(titleEl?.textContent).replace(/<[^>]+>/g, '');
|
||||
if (!title || title.length < 4) continue;
|
||||
|
||||
let url = titleEl?.getAttribute('href') || '';
|
||||
if (url && !url.startsWith('http')) url = 'https://www.gov.cn' + url;
|
||||
|
||||
const description = normalize(el.querySelector('.description')?.textContent).slice(0, 120);
|
||||
const date = (el.textContent || '').match(/(\d{4}[-./]\d{1,2}[-./]\d{1,2})/)?.[1] || '';
|
||||
rows.push({ rank: rows.length + 1, title, description, date, url });
|
||||
}
|
||||
return { ok: true, rows };
|
||||
}
|
||||
|
||||
cli({
|
||||
site: 'gov-policy',
|
||||
@@ -15,34 +67,25 @@ cli({
|
||||
],
|
||||
columns: ['rank', 'title', 'description', 'date', 'url'],
|
||||
func: async (page, kwargs) => {
|
||||
const limit = clampInt(kwargs.limit, 10, 1, 20);
|
||||
const limit = parseGovPolicyLimit(kwargs.limit, 'search');
|
||||
const query = requireNonEmptyQuery(kwargs.query);
|
||||
await page.goto(`https://sousuo.www.gov.cn/sousuo/search.shtml?code=17da70961a7&dataTypeId=107&searchWord=${encodeURIComponent(query)}`);
|
||||
await page.wait(5);
|
||||
const data = await page.evaluate(`
|
||||
try {
|
||||
await page.goto(`https://sousuo.www.gov.cn/sousuo/search.shtml?code=17da70961a7&dataTypeId=107&searchWord=${encodeURIComponent(query)}`);
|
||||
await page.wait(5);
|
||||
// Poll until the SSR result list mounts.
|
||||
await page.evaluate(`
|
||||
(async () => {
|
||||
const normalize = v => (v || '').replace(/\\s+/g, ' ').trim();
|
||||
for (let i = 0; i < 30; i++) {
|
||||
if (document.querySelectorAll('.basic_result_content .item, .js_basic_result_content .item').length > 0) break;
|
||||
await new Promise(r => setTimeout(r, 500));
|
||||
}
|
||||
const results = [];
|
||||
for (const el of document.querySelectorAll('.basic_result_content .item, .js_basic_result_content .item')) {
|
||||
const titleEl = el.querySelector('a.title, .title a, a.log-anchor');
|
||||
const title = normalize(titleEl?.textContent).replace(/<[^>]+>/g, '');
|
||||
if (!title || title.length < 4) continue;
|
||||
|
||||
let url = titleEl?.getAttribute('href') || '';
|
||||
if (url && !url.startsWith('http')) url = 'https://www.gov.cn' + url;
|
||||
|
||||
const description = normalize(el.querySelector('.description')?.textContent).slice(0, 120);
|
||||
const date = (el.textContent || '').match(/(\\d{4}[-./]\\d{1,2}[-./]\\d{1,2})/)?.[1] || '';
|
||||
results.push({ rank: results.length + 1, title, description, date, url });
|
||||
if (results.length >= ${limit}) break;
|
||||
}
|
||||
return results;
|
||||
})()
|
||||
`);
|
||||
return Array.isArray(data) ? data : [];
|
||||
const result = await page.evaluate(`(${extractSearchRows.toString()})()`);
|
||||
if (!result || !result.ok) classifyExtractorFailure('search', result);
|
||||
return requireRows('search', result.rows).slice(0, limit);
|
||||
} catch (error) {
|
||||
wrapBrowserError('search', error);
|
||||
}
|
||||
},
|
||||
});
|
||||
|
||||
@@ -0,0 +1,54 @@
|
||||
import { ArgumentError, CommandExecutionError, EmptyResultError } from '@jackwener/opencli/errors';
|
||||
|
||||
const EMPTY_RESULT_PATTERNS = [
|
||||
/没有找到/,
|
||||
/暂无/,
|
||||
/无相关/,
|
||||
/未找到/,
|
||||
/搜索结果为\s*0/,
|
||||
/很抱歉/,
|
||||
];
|
||||
|
||||
export function parseGovPolicyLimit(raw, command) {
|
||||
const value = raw ?? 10;
|
||||
const limit = Number(value);
|
||||
if (!Number.isInteger(limit) || limit < 1) {
|
||||
throw new ArgumentError(`gov-policy ${command} --limit must be a positive integer`);
|
||||
}
|
||||
if (limit > 20) {
|
||||
throw new ArgumentError(`gov-policy ${command} --limit must be <= 20`);
|
||||
}
|
||||
return limit;
|
||||
}
|
||||
|
||||
export function classifyExtractorFailure(command, result) {
|
||||
const sample = String(result?.sample || '').replace(/\s+/g, ' ').trim();
|
||||
const url = String(result?.url || '').trim();
|
||||
if (command === 'search' && EMPTY_RESULT_PATTERNS.some((pattern) => pattern.test(sample))) {
|
||||
throw new EmptyResultError('gov-policy search', sample ? sample.slice(0, 160) : undefined);
|
||||
}
|
||||
const context = [url && `url=${url}`, sample && `sample=${sample.slice(0, 160)}`]
|
||||
.filter(Boolean)
|
||||
.join('; ');
|
||||
throw new CommandExecutionError(
|
||||
`gov-policy ${command} page did not expose readable result rows`,
|
||||
context || 'The page structure may have changed or the page did not finish rendering.',
|
||||
);
|
||||
}
|
||||
|
||||
export function requireRows(command, rows) {
|
||||
if (!Array.isArray(rows) || rows.length === 0) {
|
||||
throw new CommandExecutionError(
|
||||
`gov-policy ${command} extractor returned no result rows`,
|
||||
'The page structure may have changed or all result cards were missing required title fields.',
|
||||
);
|
||||
}
|
||||
return rows;
|
||||
}
|
||||
|
||||
export function wrapBrowserError(command, error) {
|
||||
if (error instanceof ArgumentError || error instanceof EmptyResultError || error instanceof CommandExecutionError) {
|
||||
throw error;
|
||||
}
|
||||
throw new CommandExecutionError(`gov-policy ${command} browser extraction failed: ${error?.message ?? error}`);
|
||||
}
|
||||
Reference in New Issue
Block a user