Compare commits
1 Commits
| Author | SHA1 | Date | |
|---|---|---|---|
| ce8bc3e73f |
@@ -9,11 +9,11 @@ outputs:
|
||||
runs:
|
||||
using: composite
|
||||
steps:
|
||||
- name: Install real Chrome for Testing
|
||||
uses: browser-actions/setup-chrome@v2
|
||||
- name: Install real Chrome (stable)
|
||||
uses: browser-actions/setup-chrome@v1
|
||||
id: setup-chrome
|
||||
with:
|
||||
chrome-version: latest
|
||||
chrome-version: stable
|
||||
|
||||
- name: Verify Chrome installation
|
||||
shell: bash
|
||||
|
||||
@@ -50,20 +50,6 @@ jobs:
|
||||
exit 1
|
||||
fi
|
||||
|
||||
# Guard: adapter rows must not silently emit keys omitted from `columns`.
|
||||
# Existing findings are tracked in scripts/silent-column-drop-baseline.json;
|
||||
# this gate rejects newly introduced drops while allowing incremental cleanup.
|
||||
- name: Check silent column drops
|
||||
if: runner.os == 'Linux'
|
||||
run: npm run check:silent-column-drop
|
||||
|
||||
# Guard: adapters should fail with typed errors instead of silently
|
||||
# returning empty arrays, clamping user input, or inventing sentinel data.
|
||||
# Existing findings are tracked in scripts/typed-error-lint-baseline.json.
|
||||
- name: Check typed-error lint baseline
|
||||
if: runner.os == 'Linux'
|
||||
run: npm run check:typed-error-lint
|
||||
|
||||
# ── Unit tests (vitest shard) ──
|
||||
# PR: ubuntu + Node 22 only (fast feedback, 2 jobs).
|
||||
# Push to main/dev: full matrix for cross-platform/cross-version coverage (12 jobs).
|
||||
@@ -110,12 +96,8 @@ jobs:
|
||||
- name: Run unit tests under Bun
|
||||
run: bun vitest run --project unit --reporter=verbose
|
||||
|
||||
# Adapter tests are pure unit tests — OS doesn't affect results. Gated off
|
||||
# `pull_request` to keep PR CI under ~2 minutes; adapter authors run focused
|
||||
# tests locally before pushing, and `push` to main / nightly cron / manual
|
||||
# dispatch still guard the merged state.
|
||||
# Adapter tests are pure unit tests — OS doesn't affect results.
|
||||
adapter-test:
|
||||
if: github.event_name == 'push' || github.event_name == 'schedule' || github.event_name == 'workflow_dispatch'
|
||||
runs-on: ubuntu-latest
|
||||
needs: build
|
||||
steps:
|
||||
|
||||
@@ -1,10 +1,6 @@
|
||||
name: E2E Headed Chrome
|
||||
|
||||
on:
|
||||
# E2E removed from `pull_request` to keep PR feedback under ~2 minutes; PR-time
|
||||
# protection is the CI workflow (typecheck / unit / lint / adapter / build).
|
||||
# E2E still guards `main` directly, runs nightly, and on release tag push so
|
||||
# protocol/CDP/extension contract regressions are caught before they ship.
|
||||
push:
|
||||
branches: [main, dev]
|
||||
paths:
|
||||
@@ -17,11 +13,18 @@ on:
|
||||
- 'tests/smoke/**'
|
||||
- '.github/actions/setup-chrome/**'
|
||||
- '.github/workflows/e2e-headed.yml'
|
||||
tags: ['v*']
|
||||
schedule:
|
||||
# Daily 08:00 UTC — catch flake / Chrome-version drift even when no commits
|
||||
# touched the watched paths recently.
|
||||
- cron: '0 8 * * *'
|
||||
pull_request:
|
||||
branches: [main, dev]
|
||||
paths:
|
||||
- 'extension/**'
|
||||
- 'src/browser/**'
|
||||
- 'src/daemon.ts'
|
||||
- 'src/execution.ts'
|
||||
- 'src/interceptor.ts'
|
||||
- 'tests/e2e/**'
|
||||
- 'tests/smoke/**'
|
||||
- '.github/actions/setup-chrome/**'
|
||||
- '.github/workflows/e2e-headed.yml'
|
||||
workflow_dispatch:
|
||||
|
||||
concurrency:
|
||||
@@ -56,35 +59,12 @@ jobs:
|
||||
- name: Build
|
||||
run: npm run build
|
||||
|
||||
- name: Build extension
|
||||
run: npm run build --prefix extension
|
||||
|
||||
- name: Run AX Chrome smoke (Linux, via xvfb)
|
||||
if: runner.os == 'Linux'
|
||||
env:
|
||||
CHROME_PATH: ${{ steps.setup-chrome.outputs.chrome-path }}
|
||||
OPENCLI_AX_E2E: '1'
|
||||
run: |
|
||||
xvfb-run --auto-servernum --server-args="-screen 0 1280x720x24" \
|
||||
npx vitest run --project e2e tests/e2e/browser-ax-chrome.test.ts --reporter=verbose
|
||||
|
||||
- name: Run AX Chrome smoke (macOS / Windows)
|
||||
if: runner.os != 'Linux'
|
||||
env:
|
||||
CHROME_PATH: ${{ steps.setup-chrome.outputs.chrome-path }}
|
||||
OPENCLI_AX_E2E: '1'
|
||||
run: npx vitest run --project e2e tests/e2e/browser-ax-chrome.test.ts --reporter=verbose
|
||||
|
||||
- name: Run E2E tests (Linux, via xvfb)
|
||||
if: runner.os == 'Linux'
|
||||
env:
|
||||
OPENCLI_AX_E2E: '0'
|
||||
run: |
|
||||
xvfb-run --auto-servernum --server-args="-screen 0 1280x720x24" \
|
||||
npx vitest run tests/e2e/ --reporter=verbose
|
||||
|
||||
- name: Run E2E tests (macOS / Windows)
|
||||
if: runner.os != 'Linux'
|
||||
env:
|
||||
OPENCLI_AX_E2E: '0'
|
||||
run: npx vitest run tests/e2e/ --reporter=verbose
|
||||
|
||||
-256
@@ -4,268 +4,12 @@
|
||||
|
||||
### Features
|
||||
|
||||
* **weread-official** — integrate WeRead's official Agent Gateway as the `weread-official` CLI namespace. Pure HTTP, Bearer auth via `WEREAD_API_KEY` (no browser, no cookies). 8 commands cover the official skill bundle: `search`, `shelf`, `book` (info + chapters + progress 3-in-1), `notes` (notebook overview or per-book highlights/thoughts), `review`, `readdata` (weekly/monthly/annually/overall), `discover` (recommend or similar-book), `list-apis`. Adapter surfaces typed errors for all documented failure modes — `AuthRequiredError` on missing/rejected key (errcodes -2010/-2012), `CommandExecutionError` on HTTP/`upgrade_info`/non-zero errcode, `EmptyResultError` on empty payloads. Coexists with the existing cookie-based `weread` adapter.
|
||||
|
||||
### Bug Fixes
|
||||
|
||||
* **youtube/transcript** — scope timedtext URL matching to the current `videoId` across the in-page resource-buffer scan (`findTimedtextUrl`), the in-page fetch/XHR hook (`isJson3TimedtextUrl`), and the Node-side CDP capture (`extractSegmentsFromNetworkCapture`). Previously, SPA-style watch→watch navigation kept stale timedtext URLs from prior videos in `performance.getEntriesByType('resource')`, letting the polling loop fetch a same-language predecessor's captions and return them as the current video's transcript. Most likely to hit callers that reuse a single daemon tab to fetch many transcripts back-to-back (e.g. ml-scout).
|
||||
* **adapters** — surface the remaining `silent-empty-fallback` adapter failures as typed errors. Douyin user video comment fetch failures, Jike SSR JSON parse failures, and WeRead search-page fetch failures now throw `CommandExecutionError`; true empty Douyin/Jike/WeRead result sets now throw `EmptyResultError`.
|
||||
* **browser** — recover `Page.goto()` from stale page identities by clearing the cached targetId and retrying navigation once through the session lease; classify CDP `-32000 Cannot find default execution context` as retryable target navigation.
|
||||
* **deps** — restore Node 20 runtime compatibility by pinning runtime `undici` back to the 6.x line, and clear the docs build audit chain by overriding VitePress' Vite/PostCSS transitive dependencies to patched versions.
|
||||
* **download** — keep custom media filenames inside the requested output directory by stripping path components and sanitizing generated fallback names.
|
||||
|
||||
### Internal
|
||||
|
||||
* **audit** — stop flagging sentinel fallback strings inside thrown error messages as `silent-sentinel` violations. These are typed failure diagnostics rather than fake row data, reducing the typed-error baseline to actual adapter output fallbacks.
|
||||
|
||||
## [1.7.22](https://github.com/jackwener/opencli/compare/v1.7.21...v1.7.22) (2026-05-15)
|
||||
|
||||
External CLI ergonomics + two adapter envelope/auth fixes. New `longbridge` external CLI entry; `opencli list` / root help now render human-readable brand labels for executables whose bare name is ambiguous.
|
||||
|
||||
### Features
|
||||
|
||||
* **external** — add the Longbridge CLI as a built-in external CLI passthrough (`opencli longbridge ...`) for Longbridge OpenAPI market data, account, and trading commands. ([#1584](https://github.com/jackwener/opencli/issues/1584))
|
||||
* **external-cli** — render brand alias `name(package)` in `opencli list` and root help when the bare executable name is ambiguous. Built-in entries `ntn` → `ntn(notion)`, `dws` → `dws(DingTalk Workspace)`, `wecom-cli` → `wecom-cli(企业微信)` now self-explain in help output. `package` field is repurposed to cover both upstream distribution names (e.g. `tg-cli`) and human-readable brand labels (e.g. `notion`, `企业微信`). ([#1585](https://github.com/jackwener/opencli/issues/1585))
|
||||
|
||||
### Bug Fixes
|
||||
|
||||
* **boss** — map `code=24` (identity mismatch) to `AuthRequiredError` so re-login is signaled instead of surfacing as a generic API error. ([#1573](https://github.com/jackwener/opencli/issues/1573))
|
||||
* **weibo** — unwrap Browser Bridge `page.evaluate` envelopes in read adapters. ([#1568](https://github.com/jackwener/opencli/issues/1568))
|
||||
|
||||
## [1.7.21](https://github.com/jackwener/opencli/compare/v1.7.20...v1.7.21) (2026-05-14)
|
||||
|
||||
Adapter polish release: new web search adapters, better Browser Bridge tab group reuse, and social adapters returning to one-shot tab leases. Extension package version is bumped to 1.0.15 for the Browser Bridge fix.
|
||||
|
||||
### Features
|
||||
|
||||
* **search** — add DuckDuckGo, Brave, and Yahoo web search adapters. ([#1546](https://github.com/jackwener/opencli/issues/1546))
|
||||
* **boss** — support job-seeker `chatlist` and `chatmsg` adapters. ([#1539](https://github.com/jackwener/opencli/issues/1539))
|
||||
|
||||
### Bug Fixes
|
||||
|
||||
* **extension** — reuse existing `OpenCLI Adapter` tab groups before creating new ones, including cross-window discovery, legacy `OpenCLI` title fallback, and deterministic candidate selection. ([#1541](https://github.com/jackwener/opencli/issues/1541))
|
||||
* **twitter, reddit** — default browser-backed social adapters back to ephemeral tab leases. Twitter/X and Reddit commands now release their site tab after each run while keeping the shared Adapter window available for reuse; persistent sessions remain reserved for AI/chat-style adapters that need long-lived conversation state. ([#1569](https://github.com/jackwener/opencli/issues/1569))
|
||||
* **xiaohongshu, rednote** — unwrap Browser Bridge `page.evaluate` envelopes in search adapters. ([#1561](https://github.com/jackwener/opencli/issues/1561))
|
||||
* **facebook/feed** — add fallback extraction for empty article nodes. ([#1538](https://github.com/jackwener/opencli/issues/1538))
|
||||
|
||||
### Internal
|
||||
|
||||
* **ci** — add Windows native binding lockfile entries for Rolldown/Rollup optional packages. ([#1563](https://github.com/jackwener/opencli/issues/1563))
|
||||
* **extension** — add regression coverage for the adapter tab group `groupId` tiebreaker. ([#1566](https://github.com/jackwener/opencli/issues/1566))
|
||||
|
||||
## [1.7.20](https://github.com/jackwener/opencli/compare/v1.7.19...v1.7.20) (2026-05-14)
|
||||
|
||||
External CLI surface cleanup + Browser Bridge WebSocket lifecycle hardening. Two BREAKING changes around external CLIs: built-in `tg`/`discord`/`wx` (was `tg-cli`/`discord-cli`/`wx-cli`) now match their real binary names, and Notion's in-tree CDP adapter is replaced by the official `ntn` external CLI.
|
||||
|
||||
### ⚠ BREAKING CHANGES
|
||||
|
||||
* **notion** — remove the in-tree `clis/notion/` CDP-on-Desktop adapter (8 commands: `status` / `search` / `read` / `new` / `write` / `sidebar` / `favorites` / `export`). Notion has shipped an official CLI at <https://ntn.dev>, registered as a first-class external CLI in `external-clis.yaml`. Migration: install `ntn` from <https://ntn.dev> (`curl -fsSL https://ntn.dev | bash`), then use `opencli ntn <command>`. Auto-install is intentionally not configured because the official installer is a shell script while OpenCLI external installs run shell-free command strings. The official CLI uses the public Notion API rather than reverse-engineering the Desktop UI, so it survives Notion app updates and exposes a wider command surface (blocks / databases / properties / comments) than the reverse-engineered adapter could. ([#1559](https://github.com/jackwener/opencli/issues/1559))
|
||||
* **external** — drop the `-cli` suffix from built-in external CLI subcommand names. `opencli tg-cli`, `opencli discord-cli`, `opencli wx-cli` are now `opencli tg`, `opencli discord`, `opencli wx`, matching the real binary names that those tools install as. Root help still shows the package lineage as `tg(tg-cli)` / `discord(discord-cli)` / `wx(wx-cli)`. ([#1544](https://github.com/jackwener/opencli/issues/1544))
|
||||
|
||||
### Features
|
||||
|
||||
* **twitter** — `bookmarks` and `bookmark-folder` now include media via `extractMedia`, reaching parity with `timeline` / `search`. ([#1555](https://github.com/jackwener/opencli/issues/1555))
|
||||
* **twitter/list-tweets** — include media via `extractMedia` (parity with `timeline` / `search`). ([#1464](https://github.com/jackwener/opencli/issues/1464))
|
||||
|
||||
### Bug Fixes
|
||||
|
||||
* **daemon** — report ambiguous browser command outcomes with a distinct `command_result_unknown` errorCode and `503` when the extension WebSocket drops between command dispatch and result delivery. `sendCommandRaw()` treats this code as hard non-retryable, so write-side commands (`navigate` / `click` / `type` / `eval`) won't be silently re-issued and double-executed. Daemon exposes a `commandResultUnknown` counter on `/status` for future observability. ([#1558](https://github.com/jackwener/opencli/issues/1558))
|
||||
* **extension** — keep active daemon WebSocket; stale sockets no longer clobber active connection (`onopen` / `onclose` / `onmessage` are all gated by `ws !== thisWs` short-circuit), and `safeSend` only fires when `readyState === OPEN`. ([#1540](https://github.com/jackwener/opencli/issues/1540))
|
||||
* **extension** — coalesce concurrent daemon WebSocket connects via an in-flight promise. Startup / keepalive / reconnect triggering `connect()` during the daemon-probe or context-lookup async gap no longer creates duplicate real WebSocket connections. ([#1554](https://github.com/jackwener/opencli/issues/1554))
|
||||
* **external** — distinguish external CLI executable names from distribution/project names in root help. Built-in aliases such as `tg`, `discord`, `wx` remain the callable `opencli <name> ...` entrypoints while help renders `tg(tg-cli)`, `discord(discord-cli)`, `wx(wx-cli)` to show their package lineage. ([#1560](https://github.com/jackwener/opencli/issues/1560))
|
||||
|
||||
### Docs
|
||||
|
||||
* **browser** — clarify named session lifecycle in the Browser Bridge guide. ([#1542](https://github.com/jackwener/opencli/issues/1542))
|
||||
|
||||
## [1.7.19](https://github.com/jackwener/opencli/compare/v1.7.18...v1.7.19) (2026-05-14)
|
||||
|
||||
Major hotfix + simplification batch. Extension bumped to 1.0.14. Node floor lowered to v20 so the long tail of Node v20–v21.6 users no longer crashes at module load. `opencli browser` user surface replaces required-flag `--session <name>` with a `<session>` positional. `page.evaluate(fn, ...args)` adds a type-safe alternative to the implicit auto-IIFE string form. Twitter cursor pagination no longer silently caps at ~500 items.
|
||||
|
||||
### ⚠ BREAKING CHANGES
|
||||
|
||||
* **browser** — replace the `--session <name>` flag with a `<session>` positional argument that immediately follows `browser`. `opencli browser work click 12` instead of `opencli browser --session work click 12`; `opencli browser work bind` instead of `opencli browser bind --session work`. Required-flag semantics are now encoded structurally as a positional, matching the Docker/git convention for required operation-target identifiers. The internal `--session` flag is preserved for the daemon protocol and for direct `program.parseAsync` callers but is no longer part of the user-facing surface. ([#1505](https://github.com/jackwener/opencli/issues/1505))
|
||||
* **env** — remove `OPENCLI_KEEP_TAB`. The flag was a debugging shortcut, not a config dimension: `--keep-tab true|false` on the command line is the single source of truth, and adapter `siteSession: 'persistent'` already pins persistent site tabs as a hard constraint. Removing the env eliminates a globally-leaking process state that overrode every browser command in the shell. ([#1509](https://github.com/jackwener/opencli/issues/1509))
|
||||
* **extension** — remove the internal `surface\\0session` command-session backdoor. Browser Bridge commands now route only through structured `session` + `surface` fields; lease-key strings remain an extension-internal registry detail. ([#1510](https://github.com/jackwener/opencli/issues/1510))
|
||||
|
||||
### Features
|
||||
|
||||
* **browser** — add `page.evaluate(fn, ...args)` for type-safe browser-context evaluation with JSON-serialized arguments. String evaluation remains supported, but new adapter code should use function form to avoid implicit `wrapForEval` auto-IIFE magic. ([#1508](https://github.com/jackwener/opencli/issues/1508))
|
||||
* **twitter** — default `tweets` command to the logged-in user when `user` is omitted, and fix the sibling envelope-unwrap silent bug. ([#1531](https://github.com/jackwener/opencli/issues/1531))
|
||||
* **zhihu** — add `answer-detail` to fetch a single answer's full content. ([#1528](https://github.com/jackwener/opencli/issues/1528))
|
||||
* **zhihu** — paginate question answers and recommendations. ([#1517](https://github.com/jackwener/opencli/issues/1517))
|
||||
* **reddit/read** — `--expand-more` via `/api/morechildren` + 7-kind typed errors. ([#1492](https://github.com/jackwener/opencli/issues/1492))
|
||||
* **reddit** — add `whoami`, `home`, `subreddit-info` read commands. ([#1491](https://github.com/jackwener/opencli/issues/1491))
|
||||
* **ctrip** — add `hotel-search` + flight browser-mode commands. ([#1489](https://github.com/jackwener/opencli/issues/1489))
|
||||
|
||||
### Bug Fixes
|
||||
|
||||
* **browser** — `page.evaluate()` / `evaluateInFrame()` now return the user JavaScript value directly. Browser Bridge `exec` previously routed through a shared `pageScopedResult` helper that spread / wrapped the lease's `session` into the result `data`, contaminating arbitrary user returns: array / primitive returns came back as `{ session, data }` envelopes, and plain-object returns had an extra `session` key injected (overwriting any user `session` field). `google search` and `xiaohongshu search` were the visible repro — Chrome rendered results correctly but adapters extracted an empty array. Fixed in extension 1.0.14 by reverting `pageScopedResult` to its pre-1461 form (`{ id, ok, data, page }`); no client-side unwrap is needed. ([#1518](https://github.com/jackwener/opencli/issues/1518))
|
||||
* **twitter** — raise fixed cursor-pagination caps in `bookmarks` / `likes` / `tweets` / `timeline` / `bookmark-folder` / `list-tweets` / `search` / `following`. The old `i < 5` / `i < 10` literals and following's `Math.ceil(limit / 50) + 2` formula imposed hidden result ceilings below `--limit`; the loop now treats the page count as a high runaway guard while `--limit` and cursor exhaustion control normal pagination. ([#1532](https://github.com/jackwener/opencli/issues/1532))
|
||||
* **twitter** — repair `list-add` / `list-tweets` / `lists` / `following` after 2026-05 site changes. ([#1503](https://github.com/jackwener/opencli/issues/1503))
|
||||
* **twitter** — repair `search` and `tweets` readback. ([#1512](https://github.com/jackwener/opencli/issues/1512))
|
||||
* **twitter** — make reply submission robust. ([#1511](https://github.com/jackwener/opencli/issues/1511))
|
||||
* **google/search** — wait for `#rso a h3` before extracting, falling back to the existing fixed wait. On Chrome 148 + Linux Wayland the DOM can settle before SERP anchors are populated, making extraction return empty even with the envelope bug fixed. ([#1518](https://github.com/jackwener/opencli/issues/1518))
|
||||
* **google/search** — wrap evaluate return value in object to fix serialization. ([#1523](https://github.com/jackwener/opencli/issues/1523))
|
||||
* **google-scholar/search** — wrap evaluate return to fix serialization. ([#1525](https://github.com/jackwener/opencli/issues/1525))
|
||||
* **xiaohongshu/search** — extract initially visible cards before scrolling, then merge post-scroll rows by URL. Xiaohongshu's virtualized masonry layout can evict the initial cards from the DOM after scroll, so the previous always-scroll-then-extract flow could lose the top results. ([#1518](https://github.com/jackwener/opencli/issues/1518))
|
||||
* **xiaohongshu** — `parseLikes` handles `2.1w` / `1.5万` / `1.2k` shortforms. ([#1504](https://github.com/jackwener/opencli/issues/1504))
|
||||
* **xiaohongshu+rednote/search** — fall back to href-based note cards when `section.note-item` class is dropped. ([#1507](https://github.com/jackwener/opencli/issues/1507))
|
||||
* **xueqiu** — `kline` / `earnings-date` format dates in Asia/Shanghai instead of UTC. ([#1498](https://github.com/jackwener/opencli/issues/1498))
|
||||
* **download** — clamp progress percentages. ([#1520](https://github.com/jackwener/opencli/issues/1520))
|
||||
|
||||
### Internal
|
||||
|
||||
* **runtime** — lower the Node floor to `>=20.0.0`. Three coupled changes: drop all `util.styleText()` usage (added in Node v21.7.0 / v20.12.0; previously crashed v21.0–v21.6 at module load), downgrade `undici` from `^8.0.2` (engines `>=22.19.0`) to `^6.25.0` (engines `>=18.17`, retains `Agent` / `EnvHttpProxyAgent` / `fetch`), and lower `MIN_SUPPORTED_NODE_MAJOR` from 21 to 20 so the startup guard matches the declared `engines.node`. Smoke-tested on v20.0.0 / v21.2.0 / v22.22.2. The semantic markers (`[OK]` / `[WARN]` / `[FAIL]` / `ℹ` / `⚠` / `✖`) keep their meaning; ANSI colors were redundant for the primarily agent-facing CLI. ([#1524](https://github.com/jackwener/opencli/issues/1524))
|
||||
* **extension 1.0.14** — `pageScopedResult` no longer injects `session` into `data`. The field had no consumers and contaminated `exec` results with arbitrary user-JS shapes; routing-relevant identity is already exposed via `Result.page`. ([#1518](https://github.com/jackwener/opencli/issues/1518))
|
||||
* **extension 1.0.13** — remove the internal command-session lease-key backdoor. ([#1510](https://github.com/jackwener/opencli/issues/1510))
|
||||
* **ci** — drop `e2e-headed` and `adapter-test` from `pull_request` triggers (kept on `push` to main / nightly / `workflow_dispatch`). PR-time CI now targets ~2 min wall-time. ([#1521](https://github.com/jackwener/opencli/issues/1521), [#1522](https://github.com/jackwener/opencli/issues/1522))
|
||||
* **scripts** — auto-refresh `dist/` before `build-manifest`. ([#1490](https://github.com/jackwener/opencli/issues/1490))
|
||||
|
||||
## [1.7.18](https://github.com/jackwener/opencli/compare/v1.7.17...v1.7.18) (2026-05-12)
|
||||
|
||||
Hotfix release for the 1.7.17 doctor regression: `opencli doctor` failed connectivity probe with `Browser session is required` because the doctor probe didn't pass a session to the new strict-session browser bridge. Also adds new adapters and adapter fixes that were ready immediately after 1.7.17.
|
||||
|
||||
### Bug Fixes
|
||||
|
||||
* **doctor** — pass an internal `__doctor__` browser session to the live connectivity probe so `opencli doctor` works again under the explicit-session browser model introduced in 1.7.17. ([#1485](https://github.com/jackwener/opencli/issues/1485))
|
||||
* **browser** — `--session <name>` is now declared as a `requiredOption` so Commander itself rejects calls missing the flag before runtime, and the help line is marked `(required)` instead of being hidden under `Options:`. ([#1485](https://github.com/jackwener/opencli/issues/1485))
|
||||
* **doubao/ask** — restore Assistant detection after the 2026-05 DOM refactor. ([#1484](https://github.com/jackwener/opencli/issues/1484))
|
||||
* **youtube** — request `srv3` format for caption URLs. ([#1422](https://github.com/jackwener/opencli/issues/1422))
|
||||
|
||||
### Features
|
||||
|
||||
* **rednote** — add `rednote.com` adapter mirroring xiaohongshu read commands. ([#1475](https://github.com/jackwener/opencli/issues/1475))
|
||||
* **reddit** — add `reply` command for replying to comments. ([#1428](https://github.com/jackwener/opencli/issues/1428))
|
||||
|
||||
## [1.7.17](https://github.com/jackwener/opencli/compare/v1.7.16...v1.7.17) (2026-05-12)
|
||||
|
||||
Extension bumped to 1.0.12 (workspace → session lease routing, drop `handleSessions` handler). Major simplification pass: browser/adapter session model rewrite, `--workspace` removed, doctor surface trimmed to its core job.
|
||||
|
||||
### ⚠ BREAKING CHANGES
|
||||
|
||||
* **browser session model** — replace the browser-facing `--workspace` model with explicit `--session <name>` on `opencli browser *`. Browser commands now require a session name, `browser bind`/`unbind` use `--session`, and bind no longer accepts `--domain`, `--path-prefix`, or `--allow-navigate-bound`. Browser primitives keep their session tab by design; the browser namespace no longer exposes `--keep-tab`. ([#1461](https://github.com/jackwener/opencli/issues/1461))
|
||||
* **adapter site sessions** — replace adapter metadata `browserSession: { reuse: 'site' }` with `siteSession: 'persistent'`, and replace the user override `--reuse <none|site>` / `OPENCLI_BROWSER_REUSE` with `--site-session <ephemeral|persistent>`. Persistent site sessions keep a stable site tab open without idle expiry. ([#1462](https://github.com/jackwener/opencli/issues/1462))
|
||||
* **doctor** — remove `--no-live` and `--sessions` flags from `opencli doctor`. Doctor always runs the live browser connectivity probe (that's its core job); session enumeration was never part of health diagnosis. The underlying `'sessions'` daemon protocol action and the `BrowserSessionInfo` public type are removed as dead code. ([#1470](https://github.com/jackwener/opencli/issues/1470))
|
||||
|
||||
### Features
|
||||
|
||||
* **chatgpt** — `ask` and `send` now accept local image paths and upload them through the composer before submitting the prompt. ([#1476](https://github.com/jackwener/opencli/issues/1476))
|
||||
|
||||
### Internal
|
||||
|
||||
* **extension 1.0.12** — drop `handleSessions` action handler (no remaining consumers after doctor cleanup).
|
||||
* **extension 1.0.11** — switch Browser Bridge lease routing from user-facing workspaces to explicit browser sessions.
|
||||
|
||||
## [1.7.16](https://github.com/jackwener/opencli/compare/v1.7.15...v1.7.16) (2026-05-11)
|
||||
|
||||
Extension bumped to 1.0.10 (rename adapter-owned tab group `OpenCLI Automation` → `OpenCLI Adapter`). Performance and stability sweep across browser-backed adapters; new external CLI integrations (tg-cli, discord-cli, wx-cli).
|
||||
|
||||
### Features
|
||||
|
||||
* **openreview** — add `author` command for ID-explicit publication lookup. ([#1365](https://github.com/jackwener/opencli/issues/1365))
|
||||
* **external** — register `tg-cli`, `discord-cli`, and `wx-cli` as external CLI integrations. ([#1458](https://github.com/jackwener/opencli/issues/1458))
|
||||
|
||||
### Bug Fixes
|
||||
|
||||
* **xiaohongshu** — fall back to base64 upload when CDP `DOM.setFileInputFiles` returns `Not allowed` on creator center. ([#1374](https://github.com/jackwener/opencli/issues/1374))
|
||||
* **chatgpt** — switch to locale-stable send button selector so non-English UIs don't break send. ([#1354](https://github.com/jackwener/opencli/issues/1354))
|
||||
|
||||
### Performance
|
||||
|
||||
* **adapters** — hoist cookie reads to `page.getCookies` across Tier 1 (25 files), eliminating per-call CDP round trips. ([#1450](https://github.com/jackwener/opencli/issues/1450))
|
||||
* **twitter** — drop redundant `goto + wait` in adapter steps; framework auto pre-navigates. ([#1451](https://github.com/jackwener/opencli/issues/1451))
|
||||
* **twitter** — enable `browserSession.reuse: 'site'` on 17 read-only adapters so repeated reads share one tab. ([#1454](https://github.com/jackwener/opencli/issues/1454))
|
||||
* **reddit** — opt 13 browser-backed adapters into shared site-tab lease. ([#1455](https://github.com/jackwener/opencli/issues/1455))
|
||||
* **claude** — replace fixed-sleep waits with selector-based readiness on streaming flows. ([#1452](https://github.com/jackwener/opencli/issues/1452))
|
||||
* **deepseek** — replace fixed-sleep waits with selector-based readiness on streaming flows. ([#1449](https://github.com/jackwener/opencli/issues/1449))
|
||||
* **chatgpt** — replace fixed-sleep waits with selector-based readiness (D3). ([#1456](https://github.com/jackwener/opencli/issues/1456))
|
||||
|
||||
### Refactor
|
||||
|
||||
* **browser** — split interactive and automation windows so `opencli browser *` and adapter-driven background commands no longer share one Chrome window; tab groups are isolated by role.
|
||||
|
||||
### Internal
|
||||
|
||||
* **extension 1.0.10** — rename the adapter-owned Chrome tab group from `OpenCLI Automation` to `OpenCLI Adapter`. ([#1457](https://github.com/jackwener/opencli/issues/1457))
|
||||
* **docs** — list `tg-cli`, `discord-cli`, `wx-cli` in External CLI README sections. ([#1459](https://github.com/jackwener/opencli/issues/1459))
|
||||
|
||||
## [1.7.15](https://github.com/jackwener/opencli/compare/v1.7.14...v1.7.15) (2026-05-10)
|
||||
|
||||
Extension bumped to 1.0.9 (Accessibility.enable allowlist + downloads permission + cross-origin frame target attach for AX). Major Browser Agent Runtime release: full Phase 0/1/2 alignment with `vercel-labs/agent-browser` model — CDP-primary input, AX snapshot/refs with stale recovery, semantic locators across all primitives, full form toolbelt (hover/focus/dblclick/check/uncheck/upload/drag/wait-download), annotated screenshots, and same-origin iframe AX routing. Cross-origin OOPIF AX is best-effort (Chrome extension API limitation).
|
||||
|
||||
### ⚠ BREAKING CHANGES
|
||||
|
||||
* **browser lifecycle** — replace `--focus` / `OPENCLI_WINDOW_FOCUSED` with `--window foreground|background` / `OPENCLI_WINDOW`, and replace `--live` / `OPENCLI_LIVE` with `--keep-tab true|false` / `OPENCLI_KEEP_TAB`. `opencli browser *` defaults to a foreground window and keeps its tab; browser-backed adapter commands default to a background automation window and release their tab unless the adapter uses site-level reuse.
|
||||
|
||||
### Features
|
||||
|
||||
* **help / browser** — `opencli browser --help -f yaml|json` now emits a structured, agent-ready index of all browser leaf commands (including nested `tab`, `get`, and `dialog` commands), their positionals, command options, namespace options, and root global options. Individual browser commands also support structured help, backed by a shared Commander option/argument spec extractor.
|
||||
* **help / built-in namespaces** — `opencli daemon|plugin|adapter|profile --help -f yaml|json` now emit the same structured payload as `browser`. One agent call returns every leaf's positionals, options, descriptions, and global options — no per-leaf `--help` follow-ups needed. Original namespace descriptions are preserved through `applyRootSubcommandSummaries()` via a snapshot at namespace declaration time.
|
||||
* **browser state** — add opt-in AX snapshot refs via `browser state --source ax`, including backend-node click resolution and role/name stale-ref recovery for the Phase 0 browser-agent runtime prototype.
|
||||
* **browser state** — AX snapshots now include same-origin iframe refs, and `browser state --compare-sources` prints DOM-vs-AX observation metrics for the Phase 1 default-source decision without dumping page contents.
|
||||
* **browser locators** — `browser find`, `browser click`, and `browser get text|value|attributes` now accept semantic locator flags (`--role`, `--name`, `--label`, `--text`, `--testid`) so agents can act on common controls without a separate state-ref lookup.
|
||||
* **browser locators** — semantic locator flags now work across input/action primitives (`type`, `fill`, `select`, `hover`, `focus`, `dblclick`, `check`, `uncheck`, `upload`) plus prefixed `--from-*` / `--to-*` locators for `drag`.
|
||||
* **browser actions** — add `browser hover`, `browser focus`, and `browser dblclick` primitives backed by the same target resolver and CDP input path as `browser click`.
|
||||
* **browser actions** — add `browser check` and `browser uncheck` primitives that ensure checkbox / radio / aria-checked controls reach the requested state instead of blindly toggling.
|
||||
* **browser upload** — add `browser upload <target> <file...>` to attach local files to `input[type=file]` targets through CDP `DOM.setFileInputFiles`, with local path validation and file-input verification.
|
||||
* **browser actions** — add `browser drag <source> <target>` for CDP mouse drag sequences between two resolved element centers.
|
||||
* **browser wait / extension 1.0.8** — add `browser wait download [pattern]` backed by Chrome's downloads lifecycle API, so agents can wait for file downloads by filename/URL pattern and receive completed/failed download metadata.
|
||||
* **browser state / extension 1.0.9** — AX snapshots can now route same-origin iframe refs through `frameId`. Cross-origin OOPIF AX routing is best-effort because real Chrome extension smoke tests show `chrome.debugger` may not expose attachable iframe targets to extensions.
|
||||
* **browser screenshot** — add `browser screenshot --annotate`, which refreshes DOM refs and overlays visible `[N]` labels on the screenshot so visual inspection maps back to `browser click <ref>` targets.
|
||||
|
||||
### Bug Fixes
|
||||
|
||||
* **browser click** — `browser click` now prefers CDP `Input.dispatchMouseEvent` over DOM `el.click()`, so custom dropdowns that depend on pointer/mouse events (Radix, shadcn, Material UI, Mercury-style category pickers) open and select reliably while retaining JS click as a fallback for older backends or zero-rect targets.
|
||||
* **browser state / extension 1.0.7** — `browser state --source ax` now enables the CDP Accessibility domain before reading the AX tree, fixing real-Chrome snapshots that previously returned only `RootWebArea` with zero refs.
|
||||
* **help / build** — every positional arg must now declare a non-empty `help` string. The build-manifest step fails closed when a positional has empty / whitespace-only / missing `help`, so `opencli <site> <cmd> --help` always shows callers what each parameter is for. Pre-existing offenders (`twitter followers/following/list-add/list-remove/list-tweets/search/thread`, `reddit search/subreddit/user/user-comments/user-posts`, `douyin stats/update`, `bilibili subtitle`, `jike search`) now have explicit help text — most notably `twitter followers [user]` and `following [user]` now document that omitting the user fetches the currently logged-in account.
|
||||
|
||||
## [1.7.14](https://github.com/jackwener/opencli/compare/v1.7.13...v1.7.14) (2026-05-08)
|
||||
|
||||
### Features
|
||||
|
||||
* **help** — adapter help is now agent-friendly: per-command listings drop the `[options]` noise from globally-shared options (`--format`, `--trace`, `-v`, `-h`, etc.) and only mention them at the site level, so `opencli twitter` etc. read like a flat command index. ([#1401](https://github.com/jackwener/opencli/issues/1401))
|
||||
* **twitter** — write-action symmetry P0: add `unlike`, `retweet`, `unretweet`, and `quote` to round out the read/write coverage. ([#1400](https://github.com/jackwener/opencli/issues/1400))
|
||||
|
||||
### Bug Fixes
|
||||
|
||||
* **browser daemon** — `npm install -g @jackwener/opencli@latest` now correctly auto-restarts a stale ready-state daemon so users pick up the new version without a manual `opencli daemon restart`. ([#1399](https://github.com/jackwener/opencli/issues/1399))
|
||||
|
||||
## [1.7.13](https://github.com/jackwener/opencli/compare/v1.7.12...v1.7.13) (2026-05-07)
|
||||
|
||||
Extension bumped to 1.0.6 (screenshot `--width` / `--height` / `--full-page` flags, automation tab group color marker, automation container reuse fix).
|
||||
|
||||
### ⚠ BREAKING CHANGES
|
||||
|
||||
* **linux-do** — remove deprecated compatibility shims `linux-do hot`, `linux-do category`, `linux-do latest`. Use `linux-do feed --view top --period <period>`, `linux-do feed --category <id-or-name>`, and `linux-do feed --view latest` instead.
|
||||
* **grok ask** — drop the `--web` flag and the legacy `<textarea>` composer path. The default flow is now the only path and uses the current ProseMirror+TipTap composer (the path that used to require `--web true`). Existing scripts passing `--web` will get an "unknown option" error from commander; remove the flag.
|
||||
* **env** — rename `OPENCLI_BROWSER_TIMEOUT` to `OPENCLI_BROWSER_IDLE_TIMEOUT`. The variable controls workspace lease idle release time, not per-command runtime; the new name reflects that. Old name was undocumented and removed without a fallback.
|
||||
* **registry** — remove the unused `Strategy.HEADER`; adapter authors should use `Strategy.COOKIE` and set headers explicitly inside browser-side fetches.
|
||||
|
||||
### Features
|
||||
|
||||
* **observation** — add trace artifact primitives, `browser console`, `browser network --since/--follow/--failed`, and adapter `--trace=retain-on-failure` for failure-retained browser evidence.
|
||||
* **autofix** — retire `OPENCLI_DIAGNOSTIC`; adapter repair now uses `--trace retain-on-failure`, trace `summary.md`, and error-envelope trace metadata.
|
||||
* **browser** — `bind` attaches `bound:*` workspaces to user-owned Chrome tabs without taking over window lifecycle; `sessions` reports `idleMsRemaining: null` for bound workspaces because they do not schedule idle close timers. ([#1169](https://github.com/jackwener/opencli/issues/1169), [#929](https://github.com/jackwener/opencli/issues/929))
|
||||
* **browser lifecycle** — owned browser workspaces now lease tabs inside a shared dedicated automation container instead of owning one Chrome window per workspace; lease state is persisted for MV3 service-worker reconciliation and idle cleanup is backed by alarms.
|
||||
* **browser session** — adapter commands can opt into site-level tab reuse with `browserSession.reuse = 'site'`; Grok and other browser-backed LLM adapters now keep a shared site tab by default, and users can override with `--reuse <none|site>`.
|
||||
* **chatgpt** — add browser-web baseline commands: `ask`, `send`, `read`, `history`, `detail`, `new`, and `status`.
|
||||
* **grok** — add browser-web baseline commands: `read`, `history`, `detail`, `new`, `send`, and `status` (existing `ask` and `image` unchanged).
|
||||
* **yuanbao** — add browser-web baseline commands: `send`, `status`, `read`, `history`, and `detail` (joining the existing `ask` and `new`).
|
||||
* **qwen** — add `detail` command for opening a specific historical conversation by id.
|
||||
* **web read** — make page extraction render-aware: same-origin iframe content is merged into the Markdown source, `--wait-for` can wait inside main/iframe documents, `--wait-until networkidle` waits for captured requests to settle, and `--diagnose` reports frames, empty containers, and API-like XHRs for shell/AJAX pages.
|
||||
|
||||
### Bug Fixes
|
||||
|
||||
* **pipeline / capabilityRouting** — the `fill` pipeline step (introduced in [#1222](https://github.com/jackwener/opencli/issues/1222)) now correctly triggers a browser session and gets transient retry coverage; previously a pipeline using only `fill` could crash on a missing page object. ([#1393](https://github.com/jackwener/opencli/issues/1393))
|
||||
* **xiaohongshu publish** — improve image publishing reliability via creator-center URL routing, tab priority handling, and DataTransfer fallback.
|
||||
* **youtube** — use watch-page HTML for transcript captions to recover when the public transcript API is unavailable.
|
||||
* **desktop adapters** — restore 11 desktop adapter commands that were lost from the manifest due to a factory-pattern regression.
|
||||
|
||||
### Internal
|
||||
|
||||
* **cleanup** — remove dead `src/analysis.ts` (179 lines, 0 importers), retire `OPENCLI_DIAGNOSTIC` test residue, derive validator step allowlist from the live pipeline registry to prevent future drift.
|
||||
|
||||
## [1.7.8](https://github.com/jackwener/opencli/compare/v1.7.7...v1.7.8) (2026-04-25)
|
||||
|
||||
### Features
|
||||
|
||||
@@ -1,8 +1,7 @@
|
||||
# OpenCLI
|
||||
|
||||
> **Convert any website into a CLI & run Browser Use on your logged-in Chrome.**
|
||||
> Turn websites, browser sessions, Electron apps, and local tools into deterministic interfaces for humans and AI agents.
|
||||
> Or run Browser Use against any page — navigate, fill forms, click, extract, automate.
|
||||
> **Turn websites, browser sessions, Electron apps, and local tools into deterministic interfaces for humans and AI agents.**
|
||||
> Reuse your logged-in browser, automate live workflows, and crystallize repeated actions into reusable CLI commands.
|
||||
|
||||
[](./README.zh-CN.md)
|
||||
[](https://www.npmjs.com/package/@jackwener/opencli)
|
||||
@@ -12,10 +11,24 @@
|
||||
OpenCLI gives you one surface for three different kinds of automation:
|
||||
|
||||
- **Use built-in adapters** for sites like Bilibili, Zhihu, Xiaohongshu, Reddit, HackerNews, Twitter/X, and [many more](#built-in-commands).
|
||||
- **Let AI Agents operate any website** — install the `opencli-browser` skill in your AI agent (Claude Code, Cursor, etc.), and it can navigate, click, type/fill, extract, and inspect any page through your logged-in browser via `opencli browser` primitives.
|
||||
- **Let AI Agents operate any website** — install the `opencli-adapter-author` skill in your AI agent (Claude Code, Cursor, etc.), and it can navigate, click, type, extract, and inspect any page through your logged-in browser via `opencli browser` primitives.
|
||||
- **Write new adapters** end-to-end with `opencli browser` + the `opencli-adapter-author` skill, which guides from first recon through field decoding, code, and `opencli browser verify`.
|
||||
|
||||
It also works as a **CLI hub** for local tools such as `gh`, `docker`, `longbridge`, `tg`, `discord`, `wx`, `ntn` (Notion), and other binaries you register yourself, plus **desktop app adapters** for Electron apps like Cursor, Codex, Antigravity, and ChatGPT.
|
||||
It also works as a **CLI hub** for local tools such as `gh`, `docker`, and other binaries you register yourself, plus **desktop app adapters** for Electron apps like Cursor, Codex, Antigravity, ChatGPT, and Notion.
|
||||
|
||||
## Highlights
|
||||
|
||||
- **Desktop App Control** — Drive Electron apps (Cursor, Codex, ChatGPT, Notion, etc.) directly from the terminal via CDP.
|
||||
- **Browser Automation for AI Agents** — Install the `opencli-adapter-author` skill, and your AI agent can operate any website: navigate, click, type, extract, screenshot — all through your logged-in Chrome session.
|
||||
- **Multi-profile Browser Bridge** — Install the extension in each Chrome profile you want to use, then route commands with `--profile`, `OPENCLI_PROFILE`, or `opencli profile use`.
|
||||
- **Website → CLI** — Turn any website into a deterministic CLI: 100+ site surfaces are already registered, or write your own with the `opencli-adapter-author` skill + `opencli browser verify`.
|
||||
- **Account-safe** — Reuses Chrome/Chromium logged-in state; your credentials never leave the browser.
|
||||
- **AI Agent ready** — One skill takes you from site recon through API discovery, field decoding, adapter writing, and verification.
|
||||
- **CLI Hub** — Discover, auto-install, and passthrough commands to any external CLI (gh, docker, obsidian, etc).
|
||||
- **Zero LLM cost** — No tokens consumed at runtime. Run 10,000 times and pay nothing.
|
||||
- **Deterministic** — Same command, same output schema, every time. Pipeable, scriptable, CI-friendly.
|
||||
|
||||
---
|
||||
|
||||
## Quick Start
|
||||
|
||||
@@ -112,15 +125,15 @@ npx skills add jackwener/opencli --skill smart-search
|
||||
|
||||
| Skill | When to use | Example prompt to your AI agent |
|
||||
|-------|------------|-------------------------------|
|
||||
| **opencli-adapter-author** | Write a reusable adapter for a new site or add a command to an existing site | "Write an adapter for douyin trending" / "Make a command that grabs the top posts from this page" |
|
||||
| **opencli-adapter-author** | Operate a site in real time, or write a reusable adapter for a new site | "Help me check my Xiaohongshu notifications" / "Write an adapter for douyin trending" / "Make a command that grabs the top posts from this page" |
|
||||
| **opencli-autofix** | Repair a broken adapter when a built-in command fails | "`opencli zhihu hot` is returning empty — fix it" |
|
||||
| **opencli-browser** | Drive a real Chrome page ad-hoc — navigate, fill forms, click, extract | "Help me check my Xiaohongshu notifications" / "Help me fill out this form" / "Use browser commands to scrape this page" |
|
||||
| **opencli-browser** | Browser automation reference for AI agents | "Use browser commands to scrape this page" |
|
||||
| **opencli-usage** | Quick reference for all OpenCLI commands and sites | "What commands does OpenCLI have for Twitter?" |
|
||||
| **smart-search** | Search across existing OpenCLI capabilities | "Find me a Bilibili trending adapter" |
|
||||
|
||||
### How it works
|
||||
|
||||
Once `opencli-browser` is installed, your AI agent can:
|
||||
Once `opencli-adapter-author` is installed, your AI agent can:
|
||||
|
||||
1. **Navigate** to any URL using your logged-in browser
|
||||
2. **Read** page content via structured DOM snapshots (not screenshots)
|
||||
@@ -131,23 +144,23 @@ Once `opencli-browser` is installed, your AI agent can:
|
||||
The agent handles all the `opencli browser` commands internally — you just describe what you want done in natural language.
|
||||
|
||||
**Skill references:**
|
||||
- [`skills/opencli-browser/SKILL.md`](./skills/opencli-browser/SKILL.md) — drive Chrome ad-hoc (navigate, fill forms, click, extract)
|
||||
- [`skills/opencli-adapter-author/SKILL.md`](./skills/opencli-adapter-author/SKILL.md) — write a new adapter end-to-end
|
||||
- [`skills/opencli-adapter-author/SKILL.md`](./skills/opencli-adapter-author/SKILL.md) — browser operation + adapter authoring, end-to-end
|
||||
- [`skills/opencli-autofix/SKILL.md`](./skills/opencli-autofix/SKILL.md) — repair broken adapters
|
||||
- [`skills/opencli-browser/SKILL.md`](./skills/opencli-browser/SKILL.md) — browser automation reference
|
||||
- [`skills/opencli-usage/SKILL.md`](./skills/opencli-usage/SKILL.md) — command and site reference
|
||||
- [`skills/smart-search/SKILL.md`](./skills/smart-search/SKILL.md) — capability search
|
||||
|
||||
Available browser commands include `open`, `state`, `click`, `type`, `fill`, `select`, `keys`, `wait`, `get`, `find`, `extract`, `frames`, `screenshot`, `scroll`, `back`, `eval`, `network`, `tab list`, `tab new`, `tab select`, `tab close`, `init`, `verify`, and `close`.
|
||||
Available browser commands include `open`, `state`, `click`, `type`, `select`, `keys`, `wait`, `get`, `find`, `extract`, `frames`, `screenshot`, `scroll`, `back`, `eval`, `network`, `tab list`, `tab new`, `tab select`, `tab close`, `init`, `verify`, and `close`.
|
||||
|
||||
`opencli browser` commands require a `<session>` positional immediately after `browser`. `opencli browser work open <url>` and `opencli browser work tab new [url]` both return a target ID. Use `opencli browser work tab list` to inspect target IDs, then pass `--tab <targetId>` to route a command to a specific tab. `tab new` creates a new tab without changing the default browser target; only `tab select <targetId>` promotes that tab to the default target for later untargeted commands in the same session.
|
||||
`opencli browser open <url>` and `opencli browser tab new [url]` both return a target ID. Use `opencli browser tab list` to inspect the target IDs of tabs that already exist, then pass `--tab <targetId>` to route a command to a specific tab. `tab new` creates a new tab without changing the default browser target; only `tab select <targetId>` promotes that tab to the default target for later untargeted `opencli browser ...` commands.
|
||||
|
||||
## Core Concepts
|
||||
|
||||
### `browser`: AI Agent browser control
|
||||
|
||||
`opencli browser` commands are the low-level primitives that AI Agents use to operate websites. You don't run these manually — instead, install the `opencli-browser` skill into your AI agent, describe what you want in natural language, and the agent handles the browser operations.
|
||||
`opencli browser` commands are the low-level primitives that AI Agents use to operate websites. You don't run these manually — instead, install the `opencli-adapter-author` skill into your AI agent, describe what you want in natural language, and the agent handles the browser operations.
|
||||
|
||||
For example, tell your agent: *"Help me check my Xiaohongshu notifications"* — the agent will use `opencli browser <session> open`, `state`, `click`, etc. under the hood.
|
||||
For example, tell your agent: *"Help me check my Xiaohongshu notifications"* — the agent will use `opencli browser open`, `state`, `click`, etc. under the hood.
|
||||
|
||||
### Built-in adapters: stable commands
|
||||
|
||||
@@ -159,16 +172,16 @@ When the site you need is not yet covered, use the `opencli-adapter-author` skil
|
||||
|
||||
1. Recon the site and classify its pattern (SPA / SSR / JSONP / Token / Streaming).
|
||||
2. Discover the right endpoint — network inspection, initial state, bundle search, token trace, or interceptor fallback.
|
||||
3. Decide the auth strategy — `PUBLIC` / `COOKIE` / `INTERCEPT` / `UI` / `LOCAL`.
|
||||
3. Decide the auth strategy — `PUBLIC` / `COOKIE` / `HEADER` / `INTERCEPT`.
|
||||
4. Decode response fields and design output columns.
|
||||
5. `opencli browser recon analyze <url>` for one-shot recon, then `opencli browser recon init <site>/<name>` → write adapter → `opencli browser recon verify <site>/<name>`.
|
||||
5. `opencli browser analyze <url>` for one-shot recon, then `opencli browser init <site>/<name>` → write adapter → `opencli browser verify <site>/<name>`.
|
||||
6. Persist site knowledge to `~/.opencli/sites/<site>/` so the next adapter for the same site is faster.
|
||||
|
||||
### CLI Hub and desktop adapters
|
||||
|
||||
OpenCLI is not only for websites. It can also:
|
||||
|
||||
- expose local binaries like `gh`, `docker`, `obsidian`, `tg`, `discord`, `wx`, or custom tools through `opencli <tool> ...`
|
||||
- expose local binaries like `gh`, `docker`, `obsidian`, or custom tools through `opencli <tool> ...`
|
||||
- control Electron desktop apps through dedicated adapters and CDP-backed integrations
|
||||
|
||||
## Prerequisites
|
||||
@@ -185,7 +198,8 @@ OpenCLI is not only for websites. It can also:
|
||||
|----------|---------|-------------|
|
||||
| `OPENCLI_DAEMON_PORT` | `19825` | HTTP port for the daemon-extension bridge |
|
||||
| `OPENCLI_PROFILE` | — | Browser Bridge profile alias/contextId to use when multiple Chrome profiles are connected |
|
||||
| `OPENCLI_WINDOW` | command default | Set to `foreground` or `background` to override Browser Bridge window placement. Browser-backed commands also accept `--window <foreground\|background>`. |
|
||||
| `OPENCLI_WINDOW_FOCUSED` | `false` | Set to `1` to open the automation container in the foreground (useful for debugging). The `--focus` flag sets this. |
|
||||
| `OPENCLI_LIVE` | `false` | Set to `1` to keep the automation lease open after an adapter command finishes (useful for inspection). The `--live` flag sets this. |
|
||||
| `OPENCLI_BROWSER_CONNECT_TIMEOUT` | `30` | Seconds to wait for browser connection |
|
||||
| `OPENCLI_BROWSER_COMMAND_TIMEOUT` | `60` | Seconds to wait for a single browser command |
|
||||
| `OPENCLI_CDP_ENDPOINT` | — | Chrome DevTools Protocol endpoint for remote browser or Electron apps |
|
||||
@@ -193,7 +207,7 @@ OpenCLI is not only for websites. It can also:
|
||||
| `OPENCLI_VERBOSE` | `false` | Enable verbose logging (`-v` flag also works) |
|
||||
| `DEBUG_SNAPSHOT` | — | Set to `1` for DOM snapshot debug output |
|
||||
|
||||
`opencli browser *` requires an explicit `<session>` positional, uses a foreground browser window by default, and keeps that session's tab lease until `opencli browser <session> close` or idle cleanup. Browser-backed adapters use a background adapter window and release one-shot tab leases by default. Interactive adapters can declare `siteSession: 'persistent'` to keep a stable site tab for continuity; pass `--site-session ephemeral` for a one-shot tab.
|
||||
`--focus` works for both `opencli browser *` and browser-backed adapter commands. `--live` is mainly for adapter commands: browser subcommands already keep the automation lease open until you run `opencli browser close` or the idle timeout expires.
|
||||
|
||||
## Update
|
||||
|
||||
@@ -236,8 +250,7 @@ To load the source Browser Bridge extension:
|
||||
| Site | Commands |
|
||||
|------|----------|
|
||||
| **xiaohongshu** | `search` `note` `comments` `feed` `user` `download` `publish` `notifications` `creator-notes` `creator-notes-summary` `creator-note-detail` `creator-profile` `creator-stats` |
|
||||
| **rednote** | `search` `note` `comments` `user` `download` `feed` `notifications` |
|
||||
| **bilibili** | `hot` `search` `history` `feed` `ranking` `download` `comments` `dynamic` `favorite` `following` `me` `subtitle` `summary` `video` `user-videos` |
|
||||
| **bilibili** | `hot` `search` `history` `feed` `ranking` `download` `comments` `dynamic` `favorite` `following` `me` `subtitle` `video` `user-videos` |
|
||||
| **tieba** | `hot` `posts` `search` `read` |
|
||||
| **hupu** | `hot` `search` `detail` `mentions` `reply` `like` `unlike` |
|
||||
| **twitter** | `trending` `search` `timeline` `tweets` `lists` `list-tweets` `list-add` `list-remove` `bookmarks` `post` `download` `profile` `article` `like` `likes` `notifications` `reply` `reply-dm` `thread` `follow` `unfollow` `followers` `following` `block` `unblock` `bookmark` `unbookmark` `delete` `hide-reply` `accept` |
|
||||
@@ -251,7 +264,7 @@ To load the source Browser Bridge extension:
|
||||
| **yuanbao** | `new` `ask` |
|
||||
| **notebooklm** | `status` `list` `open` `current` `get` `history` `summary` `note-list` `notes-get` `source-list` `source-get` `source-fulltext` `source-guide` |
|
||||
| **spotify** | `auth` `status` `play` `pause` `next` `prev` `volume` `search` `queue` `shuffle` `repeat` |
|
||||
| **xianyu** | `search` `item` `inbox` `messages` `chat` `reply` `publish` |
|
||||
| **xianyu** | `search` `item` `chat` |
|
||||
| **xiaoe** | `courses` `detail` `catalog` `play-url` `content` |
|
||||
| **quark** | `ls` `mkdir` `mv` `rename` `rm` `save` `share-tree` |
|
||||
| **uiverse** | `code` `preview` |
|
||||
@@ -262,9 +275,7 @@ To load the source Browser Bridge extension:
|
||||
| **nowcoder** | `hot` `trending` `topics` `recommend` `creators` `companies` `jobs` `search` `suggest` `experience` `referral` `salary` `papers` `practice` `notifications` `detail` |
|
||||
| **wanfang** | `search` |
|
||||
| **hackernews** | `top` `new` `best` `ask` `show` `jobs` `search` `user` |
|
||||
| **linkedin-learning** | `search` `trending` `course` |
|
||||
| **xiaoyuzhou** | `auth*` `podcast*` `podcast-episodes*` `episode*` `download*` `transcript*` |
|
||||
| **youdao** | `note` |
|
||||
|
||||
100+ site surfaces in total — **[→ see all supported sites & commands](./docs/adapters/index.md)**
|
||||
|
||||
@@ -272,21 +283,16 @@ To load the source Browser Bridge extension:
|
||||
|
||||
## CLI Hub
|
||||
|
||||
OpenCLI acts as a universal hub for your existing command-line tools — unified discovery, pure passthrough execution, and auto-install when a safe package-manager command is configured.
|
||||
OpenCLI acts as a universal hub for your existing command-line tools — unified discovery, pure passthrough execution, and auto-install (if a tool isn't installed, OpenCLI runs `brew install <tool>` automatically before re-running the command).
|
||||
|
||||
| External CLI | Description | Example |
|
||||
|--------------|-------------|---------|
|
||||
| **gh** | GitHub CLI | `opencli gh pr list --limit 5` |
|
||||
| **obsidian** | Obsidian vault management | `opencli obsidian search query="AI"` |
|
||||
| **docker** | Docker | `opencli docker ps` |
|
||||
| **longbridge** | Longbridge CLI — market data, account management, and trading via Longbridge OpenAPI | `opencli longbridge quote TSLA.US --format json` |
|
||||
| **ntn** | Notion CLI — official Notion API CLI for pages, databases, blocks, search, comments | `opencli ntn pages list` |
|
||||
| **lark-cli** | Lark/Feishu — messages, docs, calendar, tasks, 200+ commands | `opencli lark-cli calendar +agenda` |
|
||||
| **dws** | DingTalk — cross-platform CLI for DingTalk's full suite, designed for humans and AI agents | `opencli dws msg send --to user "hello"` |
|
||||
| **wecom-cli** | WeCom/企业微信 — CLI for WeCom open platform, for humans and AI agents | `opencli wecom-cli msg send --to user "hello"` |
|
||||
| **tg(tg-cli)** | Telegram — local-first sync, search, and export via MTProto for AI agents | `opencli tg search "AI news" -f json` |
|
||||
| **discord(discord-cli)** | Discord — local-first sync, search, and export via SQLite for AI agents | `opencli discord recent --channel general` |
|
||||
| **wx(wx-cli)** | WeChat — query local WeChat data: sessions, messages, search, contacts, export | `opencli wx search "OpenCLI"` |
|
||||
| **vercel** | Vercel — deploy projects, manage domains, env vars, logs | `opencli vercel deploy --prod` |
|
||||
|
||||
**Register your own** — add any local CLI so AI agents can discover it via `opencli list`:
|
||||
@@ -295,8 +301,6 @@ OpenCLI acts as a universal hub for your existing command-line tools — unified
|
||||
opencli external register mycli
|
||||
```
|
||||
|
||||
**Manual install** — some external CLIs use official shell-script installers rather than shell-free package-manager commands. For `ntn`, install from <https://ntn.dev> first, then run `opencli ntn ...`.
|
||||
|
||||
### Desktop App Adapters
|
||||
|
||||
Control Electron desktop apps directly from the terminal. Each adapter has its own detailed documentation:
|
||||
@@ -308,6 +312,7 @@ Control Electron desktop apps directly from the terminal. Each adapter has its o
|
||||
| **Antigravity** | Control Antigravity Ultra from terminal | [Doc](./docs/adapters/desktop/antigravity.md) |
|
||||
| **ChatGPT App** | Automate ChatGPT macOS desktop app | [Doc](./docs/adapters/desktop/chatgpt-app.md) |
|
||||
| **ChatWise** | Multi-LLM client (GPT-4, Claude, Gemini) | [Doc](./docs/adapters/desktop/chatwise.md) |
|
||||
| **Notion** | Search, read, write Notion pages | [Doc](./docs/adapters/desktop/notion.md) |
|
||||
| **Discord** | Discord Desktop — messages, channels, servers | [Doc](./docs/adapters/desktop/discord.md) |
|
||||
| **Doubao** | Control Doubao AI desktop app via CDP | [Doc](./docs/adapters/desktop/doubao-app.md) |
|
||||
|
||||
@@ -320,7 +325,6 @@ OpenCLI supports downloading images, videos, and articles from supported platfor
|
||||
| Platform | Content Types | Notes |
|
||||
|----------|---------------|-------|
|
||||
| **xiaohongshu** | Images, Videos | Downloads all media from a note |
|
||||
| **rednote** | Images, Videos | Downloads all media from a signed rednote note URL |
|
||||
| **bilibili** | Videos | Requires `yt-dlp` installed |
|
||||
| **twitter** | Images, Videos | From user media tab or single tweet |
|
||||
| **douban** | Images | Poster / still image lists |
|
||||
@@ -335,7 +339,6 @@ For video downloads, install `yt-dlp` first: `brew install yt-dlp`
|
||||
```bash
|
||||
opencli xiaohongshu download "https://www.xiaohongshu.com/search_result/<id>?xsec_token=..." --output ./xhs
|
||||
opencli xiaohongshu download "https://xhslink.com/..." --output ./xhs
|
||||
opencli rednote download "https://www.rednote.com/search_result/<id>?xsec_token=..." --output ./rednote
|
||||
opencli bilibili download BV1xxx --output ./bilibili
|
||||
opencli twitter download elonmusk --limit 20 --output ./twitter
|
||||
opencli 1688 download 841141931191 --output ./1688-downloads
|
||||
@@ -402,10 +405,10 @@ See [Plugins Guide](./docs/guide/plugins.md) for creating your own plugin.
|
||||
Before writing any adapter code, read the [`opencli-adapter-author` skill](./skills/opencli-adapter-author/SKILL.md). It takes you end-to-end:
|
||||
|
||||
- Recon the site and pick a pattern (SPA / SSR / JSONP / Token / Streaming).
|
||||
- Discover the right endpoint via `opencli browser <session> network`, `eval`, or the interceptor fallback.
|
||||
- Decide auth strategy (`PUBLIC` / `COOKIE` / `INTERCEPT` / `UI` / `LOCAL`).
|
||||
- Run `opencli browser recon analyze <url>` for one-shot recon, decode response fields, design columns, scaffold with `opencli browser recon init`.
|
||||
- Verify with `opencli browser recon verify <site>/<name>` before shipping.
|
||||
- Discover the right endpoint via `opencli browser network`, `eval`, or the interceptor fallback.
|
||||
- Decide auth strategy (`PUBLIC` / `COOKIE` / `HEADER` / `INTERCEPT`).
|
||||
- Run `opencli browser analyze <url>` for one-shot recon, decode response fields, design columns, scaffold with `opencli browser init`.
|
||||
- Verify with `opencli browser verify <site>/<name>` before shipping.
|
||||
|
||||
For long-lived personal commands that should live in your own Git repo, use a local plugin instead; see [Extending OpenCLI](./docs/guide/extending-opencli.md). Quick private adapters can still live at `~/.opencli/clis/<site>/<name>.js`. Site knowledge (endpoints, field maps, fixtures) accumulates in `~/.opencli/sites/<site>/` so the next adapter for the same site starts from context instead of zero.
|
||||
|
||||
|
||||
+42
-40
@@ -1,8 +1,7 @@
|
||||
# OpenCLI
|
||||
|
||||
> **把任意网站变成 CLI & 在你的登录态浏览器上跑 Browser Use。**
|
||||
> 把网站、浏览器会话、Electron 应用和本地工具,统一变成适合人类与 AI Agent 使用的确定性接口。
|
||||
> 或者在任意页面上跑 Browser Use —— 导航、填表单、点击、抓取、自动化。
|
||||
> **把网站、浏览器会话、Electron 应用和本地工具,统一变成适合人类与 AI Agent 使用的确定性接口。**
|
||||
> 复用浏览器登录态,先自动化真实操作,再把高频流程沉淀成可复用的 CLI 命令。
|
||||
|
||||
[](./README.md)
|
||||
[](https://www.npmjs.com/package/@jackwener/opencli)
|
||||
@@ -12,10 +11,21 @@
|
||||
OpenCLI 可以用同一套 CLI 做三类事情:
|
||||
|
||||
- **直接使用现成适配器**:B站、知乎、小红书、Twitter/X、Reddit、HackerNews 等 [100+ 站点](#内置命令) 开箱即用。
|
||||
- **让 AI Agent 操作任意网站**:在你的 AI Agent(Claude Code、Cursor 等)中安装 `opencli-browser` skill,Agent 就能用你的已登录浏览器导航、点击、输入/填充、提取任意网页内容。
|
||||
- **让 AI Agent 操作任意网站**:在你的 AI Agent(Claude Code、Cursor 等)中安装 `opencli-adapter-author` skill,Agent 就能用你的已登录浏览器导航、点击、输入、提取任意网页内容。
|
||||
- **把新网站写成 CLI**:用 `opencli browser` 原语 + `opencli-adapter-author` skill,从站点侦察、API 发现、字段解码到 `opencli browser verify` 一条龙。
|
||||
|
||||
除了网站能力,OpenCLI 还是一个 **CLI 枢纽**:你可以把 `gh`、`docker`、`longbridge`、`tg`、`discord`、`wx`、`ntn`(Notion)等本地工具统一注册到 `opencli` 下,也可以通过桌面端适配器控制 Cursor、Codex、Antigravity、ChatGPT 等 Electron 应用。
|
||||
除了网站能力,OpenCLI 还是一个 **CLI 枢纽**:你可以把 `gh`、`docker` 等本地工具统一注册到 `opencli` 下,也可以通过桌面端适配器控制 Cursor、Codex、Antigravity、ChatGPT、Notion 等 Electron 应用。
|
||||
|
||||
## 亮点
|
||||
|
||||
- **桌面应用控制** — 通过 CDP 直接在终端驱动 Electron 应用(Cursor、Codex、ChatGPT、Notion 等)。
|
||||
- **AI Agent 浏览器自动化** — 安装 `opencli-adapter-author` skill,你的 AI Agent 就能操作任意网站:导航、点击、输入、提取、截图——全部通过你的已登录 Chrome 会话完成。
|
||||
- **网站 → CLI** — 把任何网站变成确定性 CLI:100+ 站点能力已注册,或用 `opencli-adapter-author` skill + `opencli browser verify` 自己写。
|
||||
- **账号安全** — 复用 Chrome/Chromium 登录态,凭证永远不会离开浏览器。
|
||||
- **面向 AI Agent** — 一个 skill 带你走完站点侦察、API 发现、字段解码、适配器编写、验证的全流程。
|
||||
- **CLI 枢纽** — 统一发现、自动安装、纯透传任何外部 CLI(gh、docker、obsidian 等)。
|
||||
- **零 LLM 成本** — 运行时不消耗模型 token,跑 10,000 次也不花一分钱。
|
||||
- **确定性输出** — 相同命令,相同输出结构,每次一致。可管道、可脚本、CI 友好。
|
||||
|
||||
## 快速开始
|
||||
|
||||
@@ -99,15 +109,15 @@ npx skills add jackwener/opencli --skill smart-search
|
||||
|
||||
| Skill | 适用场景 | 你对 AI Agent 说的话 |
|
||||
|-------|---------|-------------------|
|
||||
| **opencli-adapter-author** | 为新站点写可复用适配器,或给已有站点添加命令 | "帮我做一个抖音热门的适配器" / "帮我做一个抓取这个页面热帖的命令" |
|
||||
| **opencli-adapter-author** | 实时操作任意网站,或为新站点写可复用适配器 | "帮我看看小红书的通知" / "帮我做一个抖音热门的适配器" / "帮我做一个抓取这个页面热帖的命令" |
|
||||
| **opencli-autofix** | 内置命令失败时修复已有适配器 | "`opencli zhihu hot` 返回空了,修一下" |
|
||||
| **opencli-browser** | 实时驱动 Chrome 页面——导航、填表单、点击、抓取 | "帮我看看小红书的通知" / "帮我填一下这个表单" / "用浏览器命令抓取这个页面" |
|
||||
| **opencli-browser** | 浏览器自动化参考文档 | "用浏览器命令抓取这个页面" |
|
||||
| **opencli-usage** | 所有命令和站点的快速参考 | "OpenCLI 有哪些 Twitter 相关的命令?" |
|
||||
| **smart-search** | 在现有 OpenCLI 能力里搜索 | "帮我找个 B 站热门相关的适配器" |
|
||||
|
||||
### 工作原理
|
||||
|
||||
安装 `opencli-browser` skill 后,你的 AI Agent 可以:
|
||||
安装 `opencli-adapter-author` skill 后,你的 AI Agent 可以:
|
||||
|
||||
1. **导航**到任意 URL,使用你的已登录浏览器
|
||||
2. **读取**页面内容——通过结构化 DOM 快照(不是截图)
|
||||
@@ -118,23 +128,23 @@ npx skills add jackwener/opencli --skill smart-search
|
||||
Agent 在内部自动处理所有 `opencli browser` 命令——你只需用自然语言描述想做的事。
|
||||
|
||||
**Skill 参考文档:**
|
||||
- [`skills/opencli-browser/SKILL.md`](./skills/opencli-browser/SKILL.md) — 实时驱动 Chrome(导航、填表单、点击、抓取)
|
||||
- [`skills/opencli-adapter-author/SKILL.md`](./skills/opencli-adapter-author/SKILL.md) — 给新站点写适配器,全流程
|
||||
- [`skills/opencli-adapter-author/SKILL.md`](./skills/opencli-adapter-author/SKILL.md) — 浏览器操作 + 适配器编写,全流程
|
||||
- [`skills/opencli-autofix/SKILL.md`](./skills/opencli-autofix/SKILL.md) — 修复已有适配器
|
||||
- [`skills/opencli-browser/SKILL.md`](./skills/opencli-browser/SKILL.md) — 浏览器自动化参考
|
||||
- [`skills/opencli-usage/SKILL.md`](./skills/opencli-usage/SKILL.md) — 命令和站点参考
|
||||
- [`skills/smart-search/SKILL.md`](./skills/smart-search/SKILL.md) — 能力搜索
|
||||
|
||||
`browser` 可用命令包括:`open`、`state`、`click`、`type`、`fill`、`select`、`keys`、`wait`、`get`、`find`、`extract`、`frames`、`screenshot`、`scroll`、`back`、`eval`、`network`、`tab list`、`tab new`、`tab select`、`tab close`、`init`、`verify`、`close`。
|
||||
`browser` 可用命令包括:`open`、`state`、`click`、`type`、`select`、`keys`、`wait`、`get`、`find`、`extract`、`frames`、`screenshot`、`scroll`、`back`、`eval`、`network`、`tab list`、`tab new`、`tab select`、`tab close`、`init`、`verify`、`close`。
|
||||
|
||||
`opencli browser` 命令必须紧跟一个 `<session>` 位置参数。`opencli browser work open <url>` 和 `opencli browser work tab new [url]` 都会返回 target ID。`opencli browser work tab list` 用来查看当前已存在 tab 的 target ID,再通过 `--tab <targetId>` 把命令明确路由到某个 tab。`tab new` 只会新建 tab,不会改变默认浏览器目标;只有显式执行 `tab select <targetId>`,才会把该 tab 设为同一 session 后续未指定 target 的默认目标。
|
||||
`opencli browser open <url>` 和 `opencli browser tab new [url]` 都会返回 target ID。`opencli browser tab list` 用来查看当前已存在 tab 的 target ID,再通过 `--tab <targetId>` 把命令明确路由到某个 tab。`tab new` 只会新建 tab,不会改变默认浏览器目标;只有显式执行 `tab select <targetId>`,才会把该 tab 设为后续未指定 target 的 `opencli browser ...` 命令的默认目标。
|
||||
|
||||
## 核心概念
|
||||
|
||||
### `browser`:AI Agent 的浏览器控制层
|
||||
|
||||
`opencli browser` 命令是 AI Agent 操作网站的底层原语。你不需要手动运行这些命令——把 `opencli-browser` skill 安装到你的 AI Agent 中,用自然语言描述你想做的事,Agent 会自动处理浏览器操作。
|
||||
`opencli browser` 命令是 AI Agent 操作网站的底层原语。你不需要手动运行这些命令——把 `opencli-adapter-author` skill 安装到你的 AI Agent 中,用自然语言描述你想做的事,Agent 会自动处理浏览器操作。
|
||||
|
||||
比如你告诉 Agent:*"帮我看看小红书的通知"*——Agent 会在底层调用 `opencli browser <session> open`、`state`、`click` 等命令。
|
||||
比如你告诉 Agent:*"帮我看看小红书的通知"*——Agent 会在底层调用 `opencli browser open`、`state`、`click` 等命令。
|
||||
|
||||
### 内置适配器:稳定命令
|
||||
|
||||
@@ -146,16 +156,16 @@ Agent 在内部自动处理所有 `opencli browser` 命令——你只需用自
|
||||
|
||||
1. 侦察站点,分类 pattern(SPA / SSR / JSONP / Token / Streaming)
|
||||
2. 发现目标 endpoint——network 精读、initial state、bundle 搜索、token 溯源,或 interceptor 兜底
|
||||
3. 定认证策略——`PUBLIC` / `COOKIE` / `INTERCEPT` / `UI` / `LOCAL`
|
||||
3. 定认证策略——`PUBLIC` / `COOKIE` / `HEADER` / `INTERCEPT`
|
||||
4. 字段解码 + 设计输出列
|
||||
5. `opencli browser recon analyze <url>` 一步侦察,再 `opencli browser recon init <site>/<name>` → 写适配器 → `opencli browser recon verify <site>/<name>`
|
||||
5. `opencli browser analyze <url>` 一步侦察,再 `opencli browser init <site>/<name>` → 写适配器 → `opencli browser verify <site>/<name>`
|
||||
6. 把站点知识沉到 `~/.opencli/sites/<site>/`,下次写同站点的其他命令直接吃缓存
|
||||
|
||||
### CLI 枢纽与桌面端适配器
|
||||
|
||||
OpenCLI 不只是网站 CLI,还可以:
|
||||
|
||||
- 统一代理本地二进制工具,例如 `gh`、`docker`、`obsidian`、`tg`、`discord`、`wx`
|
||||
- 统一代理本地二进制工具,例如 `gh`、`docker`、`obsidian`
|
||||
- 通过专门适配器和 CDP 集成控制 Electron 桌面应用
|
||||
|
||||
## 前置要求
|
||||
@@ -171,7 +181,8 @@ OpenCLI 不只是网站 CLI,还可以:
|
||||
| 变量 | 默认值 | 说明 |
|
||||
|------|--------|------|
|
||||
| `OPENCLI_DAEMON_PORT` | `19825` | daemon-extension 通信端口 |
|
||||
| `OPENCLI_WINDOW` | 命令默认值 | 设为 `foreground` 或 `background` 来覆盖 Browser Bridge 窗口位置。浏览器型命令也支持 `--window <foreground\|background>` |
|
||||
| `OPENCLI_WINDOW_FOCUSED` | `false` | 设为 `1` 时 automation 窗口在前台打开(适合调试)。`--focus` 标志会设置此变量 |
|
||||
| `OPENCLI_LIVE` | `false` | 设为 `1` 时 adapter 命令执行完后保留 automation 窗口不关闭(适合检查页面)。`--live` 标志会设置此变量 |
|
||||
| `OPENCLI_BROWSER_CONNECT_TIMEOUT` | `30` | 浏览器连接超时(秒) |
|
||||
| `OPENCLI_BROWSER_COMMAND_TIMEOUT` | `60` | 单个浏览器命令超时(秒) |
|
||||
| `OPENCLI_CDP_ENDPOINT` | — | Chrome DevTools Protocol 端点,用于远程浏览器或 Electron 应用 |
|
||||
@@ -179,7 +190,7 @@ OpenCLI 不只是网站 CLI,还可以:
|
||||
| `OPENCLI_VERBOSE` | `false` | 启用详细日志(`-v` 也可以) |
|
||||
| `DEBUG_SNAPSHOT` | — | 设为 `1` 输出 DOM 快照调试信息 |
|
||||
|
||||
`opencli browser *` 必须紧跟一个 `<session>` 位置参数,默认使用前台窗口,并保留该 session 的 tab lease,直到你手动执行 `opencli browser <session> close` 或等空闲超时。浏览器型 adapter 默认使用后台 adapter 窗口并在命令结束后释放一次性 tab lease;如果需要调试最终页面,可以传 `--window foreground --keep-tab true`。
|
||||
`--focus` 同时适用于 `opencli browser *` 和浏览器型 adapter 命令。`--live` 主要是给 adapter 命令用的:`browser` 子命令本来就会一直保留 automation window,直到你手动执行 `opencli browser close` 或等空闲超时。
|
||||
|
||||
## 更新
|
||||
|
||||
@@ -226,18 +237,18 @@ npm link
|
||||
| **tieba** | `hot` `posts` `search` `read` | 浏览器 |
|
||||
| **hupu** | `hot` `search` `detail` `mentions` `reply` `like` `unlike` | 浏览器 |
|
||||
| **cursor** | `status` `send` `read` `new` `dump` `composer` `model` `extract-code` `ask` `screenshot` `history` `export` | 桌面端 |
|
||||
| **bilibili** | `hot` `search` `me` `favorite` `history` `feed` `subtitle` `summary` `video` `comments` `dynamic` `ranking` `following` `user-videos` `download` | 浏览器 |
|
||||
| **codex** | `status` `send` `read` `new` `dump` `extract-diff` `model` `ask` `screenshot` `projects` `history` `export` | 桌面端 |
|
||||
| **bilibili** | `hot` `search` `me` `favorite` `history` `feed` `subtitle` `video` `comments` `dynamic` `ranking` `following` `user-videos` `download` | 浏览器 |
|
||||
| **codex** | `status` `send` `read` `new` `dump` `extract-diff` `model` `ask` `screenshot` `history` `export` | 桌面端 |
|
||||
| **chatwise** | `status` `new` `send` `read` `ask` `model` `history` `export` `screenshot` | 桌面端 |
|
||||
| **doubao** | `status` `new` `send` `read` `ask` `history` `detail` `meeting-summary` `meeting-transcript` | 浏览器 |
|
||||
| **doubao-app** | `status` `new` `send` `read` `ask` `screenshot` `dump` | 桌面端 |
|
||||
| **notion** | `status` `search` `read` `new` `write` `sidebar` `favorites` `export` | 桌面端 |
|
||||
| **discord-app** | `status` `send` `read` `channels` `servers` `search` `members` | 桌面端 |
|
||||
| **v2ex** | `hot` `latest` `topic` `node` `user` `member` `replies` `nodes` `daily` `me` `notifications` | 公开 / 浏览器 |
|
||||
| **xueqiu** | `feed` `hot-stock` `hot` `search` `stock` `comments` `watchlist` `earnings-date` `fund-holdings` `fund-snapshot` | 浏览器 |
|
||||
| **antigravity** | `status` `send` `read` `new` `dump` `extract-code` `model` `watch` `serve` | 桌面端 |
|
||||
| **chatgpt-app** | `status` `new` `send` `read` `ask` `model` | 桌面端 |
|
||||
| **xiaohongshu** | `search` `note` `comments` `notifications` `feed` `user` `download` `publish` `creator-notes` `creator-note-detail` `creator-notes-summary` `creator-profile` `creator-stats` | 浏览器 |
|
||||
| **rednote** | `search` `note` `comments` `user` `download` `feed` `notifications` | 浏览器 |
|
||||
| **xiaoe** | `courses` `detail` `catalog` `play-url` `content` | 浏览器 |
|
||||
| **quark** | `ls` `mkdir` `mv` `rename` `rm` `save` `share-tree` | 浏览器 |
|
||||
| **uiverse** | `code` `preview` | 浏览器 |
|
||||
@@ -252,7 +263,6 @@ npm link
|
||||
| **zhihu** | `hot` `search` `question` `download` `follow` `like` `favorite` `comment` `answer` | 浏览器 |
|
||||
| **weixin** | `download` | 浏览器 |
|
||||
| **youtube** | `search` `video` `transcript` `comments` `channel` `playlist` `feed` `history` `watch-later` `subscriptions` `like` `unlike` `subscribe` `unsubscribe` | 浏览器 |
|
||||
| **youdao** | `note` | 公开 |
|
||||
| **boss** | `search` `detail` `recommend` `joblist` `greet` `batchgreet` `send` `chatlist` `chatmsg` `invite` `mark` `exchange` `resume` `stats` | 浏览器 |
|
||||
| **coupang** | `search` `add-to-cart` | 浏览器 |
|
||||
| **bbc** | `news` | 公共 API |
|
||||
@@ -261,18 +271,15 @@ npm link
|
||||
| **devto** | `top` `tag` `user` | 公开 |
|
||||
| **dictionary** | `search` `synonyms` `examples` | 公开 |
|
||||
| **arxiv** | `search` `paper` | 公开 |
|
||||
| **pubmed** | `search` `article` `author` `citations` `related` | 公开 |
|
||||
| **openreview** | `search` `venue` `paper` `reviews` | 公开 |
|
||||
| **paperreview** | `submit` `review` `feedback` | 公开 |
|
||||
| **wikipedia** | `search` `summary` `random` `trending` | 公开 |
|
||||
| **hackernews** | `top` `new` `best` `ask` `show` `jobs` `search` `user` | 公共 API |
|
||||
| **jd** | `item` | 浏览器 |
|
||||
| **linkedin** | `search` `timeline` | 浏览器 |
|
||||
| **linkedin-learning** | `search` `trending` `course` | 浏览器 |
|
||||
| **reuters** | `search` | 浏览器 |
|
||||
| **smzdm** | `search` | 浏览器 |
|
||||
| **web** | `read` | 浏览器 |
|
||||
| **weibo** | `hot` `search` `feed` `user` `user-posts` `me` `post` `favorites` `publish` `delete` `comments` | 浏览器 |
|
||||
| **weibo** | `hot` `search` `feed` `user` `me` `post` `comments` | 浏览器 |
|
||||
| **yahoo-finance** | `quote` | 浏览器 |
|
||||
| **sinafinance** | `news` | 🌐 公开 |
|
||||
| **barchart** | `quote` `options` `greeks` `flow` | 浏览器 |
|
||||
@@ -282,7 +289,7 @@ npm link
|
||||
| **jike** | `feed` `search` `create` `like` `comment` `repost` `notifications` `post` `topic` `user` | 浏览器 |
|
||||
| **jimeng** | `generate` `history` | 浏览器 |
|
||||
| **yollomi** | `generate` `video` `edit` `upload` `models` `remove-bg` `upscale` `face-swap` `restore` `try-on` `background` `object-remover` | 浏览器 |
|
||||
| **linux-do** | `feed` `search` `categories` `tags` `topic` `topic-content` `user-posts` `user-topics` | 浏览器 |
|
||||
| **linux-do** | `hot` `latest` `feed` `search` `categories` `category` `tags` `topic` `topic-content` `user-posts` `user-topics` | 浏览器 |
|
||||
| **stackoverflow** | `hot` `search` `bounties` `unanswered` | 公开 |
|
||||
| **steam** | `top-sellers` | 公开 |
|
||||
| **weread** | `shelf` `search` `book` `highlights` `notes` `notebooks` `ranking` | 浏览器 |
|
||||
@@ -307,7 +314,7 @@ npm link
|
||||
| **pixiv** | `ranking` `search` `user` `illusts` `detail` `download` | 浏览器 |
|
||||
| **tiktok** | `explore` `search` `profile` `user` `following` `follow` `unfollow` `like` `unlike` `comment` `save` `unsave` `live` `notifications` `friends` | 浏览器 |
|
||||
| **bluesky** | `search` `trending` `user` `profile` `thread` `feeds` `followers` `following` `starter-packs` | 公开 |
|
||||
| **xianyu** | `search` `item` `inbox` `messages` `chat` `reply` `publish` | 浏览器 |
|
||||
| **xianyu** | `search` `item` `chat` | 浏览器 |
|
||||
| **douyin** | `videos` `publish` `drafts` `draft` `delete` `stats` `profile` `update` `hashtag` `location` `activities` `collections` | 浏览器 |
|
||||
| **yuanbao** | `new` `ask` | 浏览器 |
|
||||
|
||||
@@ -324,19 +331,14 @@ OpenCLI 也可以作为你现有命令行工具的统一入口,负责发现、
|
||||
| **gh** | GitHub CLI | `opencli gh pr list --limit 5` |
|
||||
| **obsidian** | Obsidian 仓库管理 | `opencli obsidian search query="AI"` |
|
||||
| **docker** | Docker 命令行工具 | `opencli docker ps` |
|
||||
| **longbridge** | Longbridge CLI — 通过 Longbridge OpenAPI 获取行情、账户和交易能力 | `opencli longbridge quote TSLA.US --format json` |
|
||||
| **ntn** | Notion CLI — 基于官方 Notion API 的页面、数据库、块、搜索、评论命令 | `opencli ntn pages list` |
|
||||
| **lark-cli** | 飞书 CLI — 消息、文档、日历、任务,200+ 命令 | `opencli lark-cli calendar +agenda` |
|
||||
| **dws** | 钉钉 CLI — 钉钉全套产品能力的跨平台命令行工具,支持人类和 AI Agent 使用 | `opencli dws msg send --to user "hello"` |
|
||||
| **wecom-cli** | 企业微信 CLI — 企业微信开放平台命令行工具,支持人类和 AI Agent 使用 | `opencli wecom-cli msg send --to user "hello"` |
|
||||
| **tg(tg-cli)** | Telegram CLI — 基于 MTProto 的本地优先同步、搜索、导出,面向 AI Agent | `opencli tg search "AI news" -f json` |
|
||||
| **discord(discord-cli)** | Discord CLI — 基于 SQLite 的本地优先同步、搜索、导出,面向 AI Agent | `opencli discord recent --channel general` |
|
||||
| **wx(wx-cli)** | 微信本地数据 CLI — 会话、聊天记录、搜索、联系人、导出 | `opencli wx search "OpenCLI"` |
|
||||
| **vercel** | Vercel — 部署项目、管理域名、环境变量、日志 | `opencli vercel deploy --prod` |
|
||||
|
||||
**零配置透传**:OpenCLI 会把你的输入原样转发给底层二进制,保留原生 stdout / stderr 行为。
|
||||
|
||||
**自动安装**:如果某个外部 CLI 配置了安全的包管理器安装命令,OpenCLI 会优先尝试安装后再执行;`ntn` 的官方安装方式是 shell 脚本,请先按 <https://ntn.dev> 手动安装。
|
||||
**自动安装**:如果你运行 `opencli gh ...` 时系统中还没有 `gh`,OpenCLI 会优先尝试通过系统包管理器安装,然后自动重试命令。
|
||||
|
||||
**注册自定义本地 CLI**:
|
||||
|
||||
@@ -355,6 +357,7 @@ opencli register mycli
|
||||
| **Antigravity** | 在终端直接控制 Antigravity Ultra | [Doc](./docs/adapters/desktop/antigravity.md) |
|
||||
| **ChatGPT App** | 自动化操作 ChatGPT macOS 桌面客户端 | [Doc](./docs/adapters/desktop/chatgpt-app.md) |
|
||||
| **ChatWise** | 多 LLM 客户端(GPT-4、Claude、Gemini) | [Doc](./docs/adapters/desktop/chatwise.md) |
|
||||
| **Notion** | 搜索、读取、写入 Notion 页面 | [Doc](./docs/adapters/desktop/notion.md) |
|
||||
| **Discord** | Discord 桌面版 — 消息、频道、服务器 | [Doc](./docs/adapters/desktop/discord.md) |
|
||||
| **Doubao** | 通过 CDP 控制豆包桌面应用 | [Doc](./docs/adapters/desktop/doubao-app.md) |
|
||||
|
||||
@@ -393,7 +396,6 @@ brew install yt-dlp
|
||||
# 下载小红书笔记中的图片/视频
|
||||
opencli xiaohongshu download "https://www.xiaohongshu.com/search_result/<id>?xsec_token=..." --output ./xhs
|
||||
opencli xiaohongshu download "https://xhslink.com/..." --output ./xhs
|
||||
opencli rednote download "https://www.rednote.com/search_result/<id>?xsec_token=..." --output ./rednote
|
||||
|
||||
# 下载B站视频(需要 yt-dlp)
|
||||
opencli bilibili download BV1xxx --output ./bilibili
|
||||
@@ -501,10 +503,10 @@ opencli plugin uninstall my-tool # 卸载
|
||||
在动代码前,先读 [`opencli-adapter-author` skill](./skills/opencli-adapter-author/SKILL.md)。它把整个流程串起来:
|
||||
|
||||
- 侦察站点,选定 pattern(SPA / SSR / JSONP / Token / Streaming)
|
||||
- 用 `opencli browser <name> network`、`eval`、interceptor 等找到目标 endpoint
|
||||
- 定认证策略(`PUBLIC` / `COOKIE` / `INTERCEPT` / `UI` / `LOCAL`)
|
||||
- 先用 `opencli browser recon analyze <url>` 一步侦察,再字段解码、设计 columns、`opencli browser recon init` 生成骨架
|
||||
- 交付前用 `opencli browser recon verify <site>/<name>` 验证
|
||||
- 用 `opencli browser network`、`eval`、interceptor 等找到目标 endpoint
|
||||
- 定认证策略(`PUBLIC` / `COOKIE` / `HEADER` / `INTERCEPT`)
|
||||
- 先用 `opencli browser analyze <url>` 一步侦察,再字段解码、设计 columns、`opencli browser init` 生成骨架
|
||||
- 交付前用 `opencli browser verify <site>/<name>` 验证
|
||||
|
||||
在仓库外写的私有适配器放到 `~/.opencli/clis/<site>/<name>.js`;每个站点的 endpoint、字段映射、抓包样本会累积在 `~/.opencli/sites/<site>/`,下次写同站点的其他命令可以直接复用。
|
||||
|
||||
|
||||
@@ -1,9 +0,0 @@
|
||||
# Use Cases
|
||||
|
||||
Real-world examples of how people use OpenCLI.
|
||||
|
||||
## Contributing
|
||||
|
||||
Want to share your use case? Submit a PR that adds a new `.md` file to this directory.
|
||||
|
||||
Each file is one use case — describe what you wanted to do, which commands you used, and the result.
|
||||
@@ -1,56 +0,0 @@
|
||||
# Daily RL research monitor
|
||||
|
||||
A 30-second morning routine that surfaces what changed overnight in reinforcement-learning and large-model research, without opening a browser.
|
||||
|
||||
## What I wanted
|
||||
|
||||
Before reading anything, decide where to spend my 20 minutes of paper time:
|
||||
|
||||
- which `cs.LG` and `cs.AI` papers landed in the last 24 hours
|
||||
- which OpenReview submissions at recent venues (NeurIPS 2025 right now, ICLR 2024 / NeurIPS 2024 as historical reference) carry titles and primary areas relevant to my work
|
||||
- which papers the Hugging Face Daily Papers community is talking about today
|
||||
|
||||
Skim signals, then drill in. The point is to filter, not to read everything.
|
||||
|
||||
## Commands
|
||||
|
||||
```bash
|
||||
# 1. arxiv recent in the two relevant categories (newest 30 each)
|
||||
opencli arxiv recent cs.LG --limit 30 -f json > /tmp/lg.json
|
||||
opencli arxiv recent cs.AI --limit 30 -f json > /tmp/ai.json
|
||||
|
||||
# 2. NeurIPS 2025 oral track from OpenReview (use natural-language
|
||||
# venue text; the EMPTY_RESULT error helpfully echoes valid syntax
|
||||
# if a venue is not yet open)
|
||||
opencli openreview venue "NeurIPS 2025 oral" --limit 50 -f json > /tmp/neurips.json
|
||||
|
||||
# 3. Hugging Face Daily Papers (community-upvoted research)
|
||||
opencli hf top --period daily --limit 20 -f json > /tmp/hf.json
|
||||
```
|
||||
|
||||
That is the entire collection step. The four files together are the whole signal surface for one morning.
|
||||
|
||||
## What I do with the output
|
||||
|
||||
Pipe the four JSON files into a one-shot LLM digest with a fixed prompt:
|
||||
|
||||
```
|
||||
Here are four JSON arrays of papers from the last 24 hours.
|
||||
Group them into:
|
||||
1. Direct hits on RLHF / preference optimization / reasoning RL.
|
||||
2. Adjacent (offline RL, world models, agent benchmarks).
|
||||
3. Notable infra (training, evaluation, data).
|
||||
For each, give me title + arxiv id + one-sentence why-it-matters.
|
||||
Skip everything that is review / survey / position paper.
|
||||
```
|
||||
|
||||
The LLM compresses ~120 entries into a 10-line shortlist in seconds. I then open whichever 2 to 3 papers actually clear the bar.
|
||||
|
||||
## Why CLI beats the browser version
|
||||
|
||||
- Four pages of clicking and scrolling collapses into four `opencli` calls.
|
||||
- The output is structured JSON, so the digest prompt can reason about it deterministically. No copy-paste, no "I missed paper 14".
|
||||
- Works inside any agent loop. A scheduled task can run the four commands, push them to an LLM, and message the digest somewhere. No browser kept open.
|
||||
- Zero token cost on the OpenCLI side. The only paid step is the digest call at the end.
|
||||
|
||||
The arxiv adapter's `recent <category>` (added in #1289) is the lever here. Without it I would have to fall back to the arxiv listings page, which means scraping HTML in agent code instead of consuming a structured listing.
|
||||
@@ -1,57 +0,0 @@
|
||||
# Find a paper's implementation and follow-up work
|
||||
|
||||
Given a single paper title or arxiv id, walk three sources in one chain to find the canonical reference, follow-up citations, and any community-fine-tuned models or Spaces that already build on it.
|
||||
|
||||
## What I wanted
|
||||
|
||||
I read a paper abstract, decide it is interesting, and want to answer three questions before deciding to actually re-read the paper or reproduce it:
|
||||
|
||||
1. Has anyone already implemented or fine-tuned on top of it (Hugging Face)?
|
||||
2. Who has cited or extended it (dblp / OpenReview)?
|
||||
3. What is the canonical bibliographic record (dblp key for citation, full arxiv metadata for reading)?
|
||||
|
||||
Doing this in a browser means three tabs and two minutes of context-switching. The point is to compress that into one shell pipeline.
|
||||
|
||||
## Commands
|
||||
|
||||
Worked example: "Direct Preference Optimization" (DPO).
|
||||
|
||||
```bash
|
||||
# 1. Canonical arxiv record (full abstract, authors, pdf url, categories).
|
||||
# Note: arxiv free-text search ranks by recency, so the original DPO
|
||||
# paper does not always come back first. When the canonical id is
|
||||
# already known, hit `arxiv paper <id>` directly.
|
||||
opencli arxiv search "Direct Preference Optimization" --limit 5 -f json
|
||||
opencli arxiv paper 2305.18290 -f json
|
||||
|
||||
# 2. dblp bibliography record + co-authors + venue history
|
||||
opencli dblp search "Direct Preference Optimization" --limit 5 -f json
|
||||
|
||||
# 3. Community uptake on Hugging Face: trending Daily Papers that mention DPO
|
||||
opencli hf top --period monthly --limit 50 -f json | jq '.[] | select(.title | test("DPO|preference"; "i"))'
|
||||
|
||||
# 4. Conference review record (if posted to OpenReview)
|
||||
opencli openreview search "Direct Preference Optimization" --limit 5 -f json
|
||||
```
|
||||
|
||||
Three of the four are public-strategy adapters, no browser session needed. The OpenReview call also lands without auth for public venues.
|
||||
|
||||
## What I do with the output
|
||||
|
||||
For DPO the chain produces:
|
||||
|
||||
- arxiv record: paper id `2305.18290`, full abstract, pdf link.
|
||||
- dblp record: canonical key `conf/nips/RafailovSMMEF23`, NeurIPS 2023, co-author list (useful to find related work by same lab).
|
||||
- HF Daily Papers (last 30 days): every paper whose title mentions DPO or preference. Each one is a candidate "follow-up work I should know about".
|
||||
- OpenReview: the original submission's review thread, if posted (lets me see what reviewers actually pushed back on, which is more useful than the published abstract).
|
||||
|
||||
I dump all four JSON outputs into a single LLM call with the prompt: *"Build a one-paragraph 'state of the field' summary for this paper as of today. Cite each follow-up by arxiv id."* That gives me a research-debt brief in 30 seconds.
|
||||
|
||||
## Why this is worth a CLI chain
|
||||
|
||||
- Each adapter alone is just "search a website". The value is the chain. Four `opencli` calls feed into one LLM call. No browser, no copy-paste.
|
||||
- Output is identifier-rich (arxiv id, dblp key, venue id, HF paper id). I can re-feed any of those into the next call, e.g. once I find a follow-up arxiv id from HF Daily Papers I run `opencli arxiv paper <new-id>` immediately.
|
||||
- Survives use inside an agent loop. Same chain runs unattended for a batch of 20 papers from a reading list.
|
||||
- Zero token cost for the discovery half. Only the final summary step pays for inference.
|
||||
|
||||
Without `opencli dblp search` (added in #1299) and `opencli openreview search` (added in #1294), this whole pipeline used to require either web scraping in agent code or paying for a research-paper API. Both adapters being public-strategy means they slot in cleanly.
|
||||
@@ -1,75 +0,0 @@
|
||||
# Track a conference's accepted papers and reviews from the terminal
|
||||
|
||||
Once an OpenReview venue opens its decisions (or releases reviews publicly during the discussion phase), I want a one-shot way to pull the full venue listing and dive into individual review threads, without clicking through 200+ submission pages.
|
||||
|
||||
## What I wanted
|
||||
|
||||
For each major venue I follow (ICLR, NeurIPS, ICML), the same three things every time decisions are visible:
|
||||
|
||||
1. The full list of accepted papers at the venue, with titles and forum ids.
|
||||
2. For any paper I flagged interesting from the list: the full review thread, including reviewer scores, rebuttals, and the AC's decision rationale.
|
||||
3. A way to pipe both into LLM-driven shortlisting ("which of these 100 oral papers actually intersect with my research direction").
|
||||
|
||||
The OpenReview UI is fine for one paper at a time, but unusable for batch reasoning across the whole acceptance list.
|
||||
|
||||
## Commands
|
||||
|
||||
Worked example: ICLR 2024 oral track, then drill into one paper's reviews using a real forum id.
|
||||
|
||||
```bash
|
||||
# 1. Full list of papers at a venue (natural-language venue text;
|
||||
# if the venue is not yet open OpenReview returns EMPTY_RESULT
|
||||
# with a help line listing valid forms)
|
||||
opencli openreview venue "ICLR 2024 oral" --limit 200 -f json > /tmp/iclr-2024.json
|
||||
|
||||
# 2. Pick a forum id from the listing, fetch the full review thread.
|
||||
# Example: "Proving Test Set Contamination in Black-Box Language Models"
|
||||
opencli openreview reviews KS8mIvetg2 -f json > /tmp/reviews.json
|
||||
|
||||
# 3. Single paper metadata if needed
|
||||
opencli openreview paper KS8mIvetg2 -f json
|
||||
```
|
||||
|
||||
`venue` returns each entry with a forum id you can hand straight back into `reviews` and `paper`. No id lookup gymnastics. `reviews` returns the full thread as a JSON array: a `PAPER` row with the abstract, then one `REVIEW` row per reviewer (with `rating`, `confidence`, summary, weaknesses, questions), followed by author rebuttals and the AC's decision rationale.
|
||||
|
||||
## What I do with the output
|
||||
|
||||
Two distinct workflows depending on the phase of the venue:
|
||||
|
||||
### Phase A: filtering the acceptance list
|
||||
|
||||
After `venue` returns 200 entries, dump the JSON into an LLM with the prompt:
|
||||
|
||||
```
|
||||
Here is the full acceptance list at <venue>. Filter to papers that intersect
|
||||
with my research interests:
|
||||
- reinforcement learning from preference / reward feedback
|
||||
- reasoning training (process reward, RLVR, RLHF variants)
|
||||
- long-horizon agent benchmarks
|
||||
For each match: title + forum_id + one-sentence why-it-matters.
|
||||
```
|
||||
|
||||
This collapses 200 papers to a 10-paper shortlist in seconds. The forum ids are the keys I will use in Phase B.
|
||||
|
||||
### Phase B: depth-reading the shortlist
|
||||
|
||||
For each shortlisted forum id, run `opencli openreview reviews <forum-id>` and feed the JSON to an LLM with the prompt:
|
||||
|
||||
```
|
||||
Summarize the review thread:
|
||||
- reviewer scores
|
||||
- the strongest critique
|
||||
- whether the rebuttal addressed it
|
||||
- final decision and AC rationale
|
||||
```
|
||||
|
||||
This is faster than reading three reviews + rebuttal + meta-review per paper. For 10 papers this turns 60 minutes of OpenReview clicking into 10 minutes of summary reading, then I open the actual reviews only for papers where the summary flagged something worth knowing.
|
||||
|
||||
## Why this beats opening OpenReview
|
||||
|
||||
- One `venue` call replaces scrolling a paginated UI for 200+ papers.
|
||||
- `reviews` returns the entire thread as JSON, so an LLM can reason over the whole review-rebuttal-decision arc at once. The web view forces you to scroll three reviews + N rebuttals + meta separately.
|
||||
- Forum ids returned from `venue` are stable and reusable across calls. Easy to keep a personal reading list as `forum-ids.txt` and run `for id in $(cat forum-ids.txt); do opencli openreview reviews $id; done`.
|
||||
- The whole loop is public-strategy. No login required for venues with public reviewing.
|
||||
|
||||
`opencli openreview` (added in #1294) is the lever. Before this adapter existed, the same workflow needed either OpenReview's Python client or HTML scraping inside agent code. Both have higher friction than `opencli openreview reviews <forum-id>` returning structured JSON in one shot.
|
||||
+527
-9693
File diff suppressed because it is too large
Load Diff
@@ -1,73 +0,0 @@
|
||||
/**
|
||||
* 12306 account summary for the logged-in user.
|
||||
*
|
||||
* Returns non-sensitive identity fields plus masked email / mobile.
|
||||
* Use `--include-sensitive` to surface unmasked values from 12306's
|
||||
* own response (12306 already masks the ID number server-side; this
|
||||
* adapter never decodes that mask).
|
||||
*/
|
||||
import { cli, Strategy } from '@jackwener/opencli/registry';
|
||||
import { AuthRequiredError, CommandExecutionError } from '@jackwener/opencli/errors';
|
||||
import { isAuthLikePayload, maskEmail, maskMobile, maskChineseName, require12306Login, requireEvaluateObject } from './utils.js';
|
||||
|
||||
const ACCOUNT_INFO_URL = 'https://kyfw.12306.cn/otn/modifyUser/initQueryUserInfoApi';
|
||||
|
||||
cli({
|
||||
site: '12306',
|
||||
name: 'me',
|
||||
access: 'read',
|
||||
description: 'Show the logged-in 12306 account summary. Sensitive fields (real name, email, mobile, birth date) are masked by default; pass --include-sensitive to opt in.',
|
||||
domain: 'kyfw.12306.cn',
|
||||
strategy: Strategy.COOKIE,
|
||||
browser: true,
|
||||
args: [
|
||||
{ name: 'include-sensitive', type: 'boolean', default: false, help: 'Reveal unmasked real name / email / mobile / birth date. The 12306 ID-number mask is server-side and never decoded.' },
|
||||
],
|
||||
columns: ['username', 'real_name', 'email', 'mobile', 'birth_date', 'sex', 'country', 'user_type', 'member', 'active'],
|
||||
func: async (page, kwargs) => {
|
||||
if (!page) throw new CommandExecutionError('Browser session required for 12306 me');
|
||||
await page.goto('https://kyfw.12306.cn/otn/view/index.html');
|
||||
await require12306Login(page, AuthRequiredError);
|
||||
const json = requireEvaluateObject(await page.evaluate(`async () => {
|
||||
const r = await fetch(${JSON.stringify(ACCOUNT_INFO_URL)}, { credentials: 'include' });
|
||||
if (!r.ok) return { __http: r.status };
|
||||
try {
|
||||
return await r.json();
|
||||
} catch (err) {
|
||||
return { __parse: String(err && err.message || err) };
|
||||
}
|
||||
}`), 'account info');
|
||||
if (json?.__http) {
|
||||
if ([401, 403].includes(Number(json.__http))) {
|
||||
throw new AuthRequiredError('kyfw.12306.cn', '12306 account info requires a valid login session');
|
||||
}
|
||||
throw new CommandExecutionError(`12306 returned HTTP ${json.__http} for account info`);
|
||||
}
|
||||
if (json?.__parse) {
|
||||
throw new CommandExecutionError(`12306 account info returned non-JSON body: ${json.__parse}`);
|
||||
}
|
||||
if (isAuthLikePayload(json)) {
|
||||
throw new AuthRequiredError('kyfw.12306.cn', '12306 account info requires a valid login session');
|
||||
}
|
||||
if (json?.status !== true || !json?.data?.userDTO) {
|
||||
throw new CommandExecutionError('12306 account info payload missing userDTO');
|
||||
}
|
||||
const dto = json.data.userDTO;
|
||||
const loginDto = dto.loginUserDTO || {};
|
||||
const username = loginDto.user_name || loginDto.name || '';
|
||||
const realName = loginDto.real_name || loginDto.realname || '';
|
||||
const include = kwargs['include-sensitive'] === true;
|
||||
return [{
|
||||
username,
|
||||
real_name: include ? realName : maskChineseName(realName),
|
||||
email: include ? (dto.email || '') : maskEmail(dto.email || ''),
|
||||
mobile: include ? (dto.mobile_no || '') : maskMobile(dto.mobile_no || ''),
|
||||
birth_date: include ? (dto.born_date || '') : (dto.born_date || '').slice(0, 4),
|
||||
sex: dto.sex_code === 'M' ? '男' : (dto.sex_code === 'F' ? '女' : ''),
|
||||
country: dto.country_code || '',
|
||||
user_type: json.data.userTypeName || '',
|
||||
member: dto.flag_member === '1',
|
||||
active: dto.is_active === '1',
|
||||
}];
|
||||
},
|
||||
});
|
||||
@@ -1,96 +0,0 @@
|
||||
/**
|
||||
* 12306 in-progress orders for the logged-in user.
|
||||
*
|
||||
* Returns orders that have not yet been ridden / refunded / completed
|
||||
* (the `noComplete` slice). Order history covering completed and
|
||||
* refunded tickets uses a separate endpoint that requires extra
|
||||
* referer / page-state handshakes and is left for a follow-up so this
|
||||
* command can ship reliably.
|
||||
*/
|
||||
import { cli, Strategy } from '@jackwener/opencli/registry';
|
||||
import { AuthRequiredError, CommandExecutionError, EmptyResultError } from '@jackwener/opencli/errors';
|
||||
import { isAuthLikePayload, maskChineseName, require12306Login, requireEvaluateObject } from './utils.js';
|
||||
|
||||
const NO_COMPLETE_URL = 'https://kyfw.12306.cn/otn/queryOrder/queryMyOrderNoComplete';
|
||||
|
||||
cli({
|
||||
site: '12306',
|
||||
name: 'orders',
|
||||
access: 'read',
|
||||
description: 'List in-progress 12306 orders (not yet ridden, refunded, or completed) for the logged-in user',
|
||||
domain: 'kyfw.12306.cn',
|
||||
strategy: Strategy.COOKIE,
|
||||
browser: true,
|
||||
args: [
|
||||
{ name: 'include-sensitive', type: 'boolean', default: false, help: 'Reveal unmasked passenger names in order rows. Masked by default.' },
|
||||
],
|
||||
columns: ['order_id', 'order_date', 'train_code', 'from_station', 'to_station', 'departure', 'passengers', 'status', 'amount'],
|
||||
func: async (page, kwargs) => {
|
||||
if (!page) throw new CommandExecutionError('Browser session required for 12306 orders');
|
||||
await page.goto('https://kyfw.12306.cn/otn/view/index.html');
|
||||
await require12306Login(page, AuthRequiredError);
|
||||
const include = kwargs['include-sensitive'] === true;
|
||||
const json = requireEvaluateObject(await page.evaluate(`async () => {
|
||||
const r = await fetch(${JSON.stringify(NO_COMPLETE_URL)}, {
|
||||
method: 'POST',
|
||||
headers: { 'Content-Type': 'application/x-www-form-urlencoded' },
|
||||
body: '_json_att=', credentials: 'include',
|
||||
});
|
||||
if (!r.ok) return { __http: r.status };
|
||||
try {
|
||||
return await r.json();
|
||||
} catch (err) {
|
||||
return { __parse: String(err && err.message || err) };
|
||||
}
|
||||
}`), 'orders');
|
||||
if (json?.__http) {
|
||||
if ([401, 403].includes(Number(json.__http))) {
|
||||
throw new AuthRequiredError('kyfw.12306.cn', '12306 orders requires a valid login session');
|
||||
}
|
||||
throw new CommandExecutionError(`12306 returned HTTP ${json.__http} for queryMyOrderNoComplete`);
|
||||
}
|
||||
if (json?.__parse) {
|
||||
throw new CommandExecutionError(`12306 orders returned non-JSON body: ${json.__parse}`);
|
||||
}
|
||||
if (isAuthLikePayload(json)) {
|
||||
throw new AuthRequiredError('kyfw.12306.cn', '12306 orders requires a valid login session');
|
||||
}
|
||||
if (json?.status !== true) {
|
||||
throw new CommandExecutionError('12306 queryMyOrderNoComplete returned a failure status');
|
||||
}
|
||||
let orders;
|
||||
if (Array.isArray(json?.data?.orderDBList)) {
|
||||
orders = json.data.orderDBList;
|
||||
} else if (Array.isArray(json?.data?.orderDTODataList)) {
|
||||
orders = json.data.orderDTODataList;
|
||||
} else if (Array.isArray(json?.data?.orders)) {
|
||||
orders = json.data.orders;
|
||||
} else if (Array.isArray(json?.data)) {
|
||||
orders = json.data;
|
||||
} else {
|
||||
throw new CommandExecutionError('12306 queryMyOrderNoComplete payload missing order list array');
|
||||
}
|
||||
if (orders.length === 0) {
|
||||
throw new EmptyResultError('No in-progress 12306 orders on this account');
|
||||
}
|
||||
return orders.map((o) => {
|
||||
const tickets = Array.isArray(o.tickets) ? o.tickets : [];
|
||||
const passengerNames = tickets
|
||||
.map((t) => t.passenger_name || '')
|
||||
.filter(Boolean)
|
||||
.map((name) => include ? name : maskChineseName(name))
|
||||
.join(', ');
|
||||
return {
|
||||
order_id: o.sequence_no || o.order_id || o.sequenceNo || '',
|
||||
order_date: o.order_date || '',
|
||||
train_code: o.train_code_page || o.station_train_code || o.train_code || '',
|
||||
from_station: o.from_station_name_page || o.from_station_name || '',
|
||||
to_station: o.to_station_name_page || o.to_station_name || '',
|
||||
departure: o.start_train_date_page || o.start_train_date || '',
|
||||
passengers: passengerNames,
|
||||
status: o.ticket_status_name || o.order_status_name || o.statusName || '',
|
||||
amount: o.ticket_total_price_page || o.ticket_total_price || '',
|
||||
};
|
||||
});
|
||||
},
|
||||
});
|
||||
@@ -1,90 +0,0 @@
|
||||
/**
|
||||
* 12306 saved passenger list for the logged-in user.
|
||||
*
|
||||
* 12306 already masks ID numbers (`xxxx***********xxx`) and mobile
|
||||
* numbers (`138****xxxx`) server-side. This adapter further masks the
|
||||
* passenger's Chinese real name and birth date by default; pass
|
||||
* `--include-sensitive` to surface the unmasked-by-12306 fields.
|
||||
*/
|
||||
import { cli, Strategy } from '@jackwener/opencli/registry';
|
||||
import { ArgumentError, AuthRequiredError, CommandExecutionError, EmptyResultError } from '@jackwener/opencli/errors';
|
||||
import { isAuthLikePayload, maskChineseName, require12306Login, requireEvaluateObject } from './utils.js';
|
||||
|
||||
const PASSENGER_QUERY_URL = 'https://kyfw.12306.cn/otn/passengers/query';
|
||||
const MAX_PAGE_SIZE = 50;
|
||||
|
||||
function normalizeLimit(value, defaultValue, max) {
|
||||
if (value === undefined || value === null || value === '') return defaultValue;
|
||||
const n = Number(value);
|
||||
if (!Number.isInteger(n) || n < 1) throw new ArgumentError(`limit must be a positive integer (1-${max})`);
|
||||
if (n > max) throw new ArgumentError(`limit must be <= ${max}`);
|
||||
return n;
|
||||
}
|
||||
|
||||
cli({
|
||||
site: '12306',
|
||||
name: 'passengers',
|
||||
access: 'read',
|
||||
description: 'List the logged-in user\'s saved 12306 passengers. Sensitive fields are masked by default; pass --include-sensitive to opt in.',
|
||||
domain: 'kyfw.12306.cn',
|
||||
strategy: Strategy.COOKIE,
|
||||
browser: true,
|
||||
args: [
|
||||
{ name: 'limit', type: 'int', default: 20, help: `Max passengers to return (1-${MAX_PAGE_SIZE})` },
|
||||
{ name: 'include-sensitive', type: 'boolean', default: false, help: 'Reveal unmasked real names and birth dates. The 12306 ID-number / mobile masks are server-side and never decoded.' },
|
||||
],
|
||||
columns: ['name', 'sex', 'born_year', 'id_type', 'id_no', 'mobile', 'passenger_type', 'country'],
|
||||
func: async (page, kwargs) => {
|
||||
if (!page) throw new CommandExecutionError('Browser session required for 12306 passengers');
|
||||
const limit = normalizeLimit(kwargs.limit, 20, MAX_PAGE_SIZE);
|
||||
const include = kwargs['include-sensitive'] === true;
|
||||
|
||||
await page.goto('https://kyfw.12306.cn/otn/view/index.html');
|
||||
await require12306Login(page, AuthRequiredError);
|
||||
const json = requireEvaluateObject(await page.evaluate(`async () => {
|
||||
const body = "pageIndex=1&pageSize=${MAX_PAGE_SIZE}";
|
||||
const r = await fetch(${JSON.stringify(PASSENGER_QUERY_URL)}, {
|
||||
method: 'POST',
|
||||
headers: { 'Content-Type': 'application/x-www-form-urlencoded' },
|
||||
body, credentials: 'include',
|
||||
});
|
||||
if (!r.ok) return { __http: r.status };
|
||||
try {
|
||||
return await r.json();
|
||||
} catch (err) {
|
||||
return { __parse: String(err && err.message || err) };
|
||||
}
|
||||
}`), 'passengers');
|
||||
if (json?.__http) {
|
||||
if ([401, 403].includes(Number(json.__http))) {
|
||||
throw new AuthRequiredError('kyfw.12306.cn', '12306 passengers requires a valid login session');
|
||||
}
|
||||
throw new CommandExecutionError(`12306 returned HTTP ${json.__http} for passengers/query`);
|
||||
}
|
||||
if (json?.__parse) {
|
||||
throw new CommandExecutionError(`12306 passengers returned non-JSON body: ${json.__parse}`);
|
||||
}
|
||||
if (isAuthLikePayload(json)) {
|
||||
throw new AuthRequiredError('kyfw.12306.cn', '12306 passengers requires a valid login session');
|
||||
}
|
||||
if (json?.status !== true || !Array.isArray(json?.data?.datas)) {
|
||||
throw new CommandExecutionError('12306 passengers payload missing data.datas array');
|
||||
}
|
||||
const datas = json.data.datas;
|
||||
if (datas.length === 0) {
|
||||
throw new EmptyResultError('No saved passengers on this 12306 account');
|
||||
}
|
||||
return datas.slice(0, limit).map((p) => ({
|
||||
name: include ? (p.passenger_name || '') : maskChineseName(p.passenger_name || ''),
|
||||
sex: p.sex_name || '',
|
||||
born_year: (p.born_date || '').slice(0, 4),
|
||||
id_type: p.passenger_id_type_name || '',
|
||||
id_no: p.passenger_id_no || '',
|
||||
mobile: p.mobile_no || '',
|
||||
passenger_type: p.passenger_type_name || '',
|
||||
country: p.country_code || '',
|
||||
}));
|
||||
},
|
||||
});
|
||||
|
||||
export const __test__ = { normalizeLimit };
|
||||
@@ -1,166 +0,0 @@
|
||||
/**
|
||||
* 12306 ticket price lookup for a single train + segment.
|
||||
*
|
||||
* Cascades three anonymous API calls:
|
||||
* 1. /otn/leftTicket/init: mint session cookies
|
||||
* 2. /otn/czxx/queryByTrainNo: resolve from/to station_no within the
|
||||
* train route (price endpoint addresses stops by station_no, not
|
||||
* telecode)
|
||||
* 3. /otn/leftTicket/queryTicketPrice: ticket prices keyed by seat
|
||||
* letter (M=一等座, O=二等座, A9=商务座, A1=硬座, A3=硬卧,
|
||||
* A4=软卧, F=动卧, P=特等座, WZ=无座, etc.)
|
||||
*/
|
||||
import { cli, Strategy } from '@jackwener/opencli/registry';
|
||||
import { ArgumentError, CommandExecutionError, EmptyResultError } from '@jackwener/opencli/errors';
|
||||
import { fetchStationBundle, mintSession, resolveStation, validateDate } from './utils.js';
|
||||
|
||||
const UA = 'Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/127.0 Safari/537.36';
|
||||
const TRAIN_NO_RE = /^[0-9A-Z]{8,18}$/;
|
||||
const SEAT_TYPES_RE = /^[A-Z0-9]{1,32}$/;
|
||||
|
||||
const SEAT_LETTERS = {
|
||||
'A9': '商务座',
|
||||
'P': '特等座',
|
||||
'M': '一等座',
|
||||
'O': '二等座',
|
||||
'A1': '硬座',
|
||||
'A3': '硬卧',
|
||||
'A4': '软卧',
|
||||
'F': '动卧',
|
||||
'WZ': '无座',
|
||||
};
|
||||
|
||||
async function queryStopsForPrice(cookieHeader, trainNo, fromCode, toCode, date, fetchImpl = fetch) {
|
||||
const url = `https://kyfw.12306.cn/otn/czxx/queryByTrainNo?train_no=${trainNo}&from_station_telecode=${fromCode}&to_station_telecode=${toCode}&depart_date=${date}`;
|
||||
const resp = await fetchImpl(url, {
|
||||
headers: {
|
||||
'User-Agent': UA,
|
||||
'Referer': 'https://kyfw.12306.cn/otn/leftTicket/init',
|
||||
'Cookie': cookieHeader,
|
||||
},
|
||||
});
|
||||
if (!resp.ok) throw new CommandExecutionError(`12306 queryByTrainNo returned HTTP ${resp.status}`);
|
||||
let json;
|
||||
try {
|
||||
json = await resp.json();
|
||||
} catch {
|
||||
throw new CommandExecutionError('12306 queryByTrainNo returned non-JSON body');
|
||||
}
|
||||
if (json?.status !== true || !Array.isArray(json?.data?.data)) {
|
||||
throw new CommandExecutionError('12306 queryByTrainNo returned an unexpected payload shape');
|
||||
}
|
||||
return json.data.data;
|
||||
}
|
||||
|
||||
function pickStationNos(stops, fromCode, toCode, fromName, toName) {
|
||||
const matches = (s, code, name) => (s.station_name && name && s.station_name === name);
|
||||
const fromStop = stops.find((s) => matches(s, fromCode, fromName));
|
||||
const toStop = stops.find((s) => matches(s, toCode, toName));
|
||||
if (!fromStop) throw new CommandExecutionError(`Train does not stop at ${fromName}`);
|
||||
if (!toStop) throw new CommandExecutionError(`Train does not stop at ${toName}`);
|
||||
return { fromNo: fromStop.station_no, toNo: toStop.station_no };
|
||||
}
|
||||
|
||||
async function queryPrice(cookieHeader, trainNo, fromNo, toNo, seatTypes, date, fetchImpl = fetch) {
|
||||
const url = `https://kyfw.12306.cn/otn/leftTicket/queryTicketPrice?train_no=${trainNo}&from_station_no=${fromNo}&to_station_no=${toNo}&seat_types=${seatTypes}&train_date=${date}`;
|
||||
const resp = await fetchImpl(url, {
|
||||
headers: {
|
||||
'User-Agent': UA,
|
||||
'Referer': 'https://kyfw.12306.cn/otn/leftTicket/init',
|
||||
'Cookie': cookieHeader,
|
||||
},
|
||||
});
|
||||
if (!resp.ok) throw new CommandExecutionError(`12306 queryTicketPrice returned HTTP ${resp.status}`);
|
||||
let json;
|
||||
try {
|
||||
json = await resp.json();
|
||||
} catch {
|
||||
throw new CommandExecutionError('12306 queryTicketPrice returned non-JSON body');
|
||||
}
|
||||
if (json?.status !== true || !json?.data) {
|
||||
throw new CommandExecutionError('12306 queryTicketPrice returned an unexpected payload shape');
|
||||
}
|
||||
return json.data;
|
||||
}
|
||||
|
||||
function parsePriceData(priceData) {
|
||||
const rows = [];
|
||||
for (const [letter, value] of Object.entries(priceData)) {
|
||||
if (letter === 'train_no' || letter === 'OT') continue;
|
||||
if (typeof value !== 'string' || !value) continue;
|
||||
// 12306 doubles up some prices as bare numerics ("9": "21580"), which
|
||||
// mirror their letter sibling ("A9": "¥2158.0") in cents/no-decimal
|
||||
// form. Skip the bare numeric letter codes to avoid duplicates.
|
||||
if (/^\d+$/.test(letter)) continue;
|
||||
if (!/^[A-Z]/.test(letter)) continue;
|
||||
const numeric = value.replace(/^¥/, '');
|
||||
if (!/^[\d.]+$/.test(numeric)) continue;
|
||||
rows.push({
|
||||
seat_code: letter,
|
||||
seat_name: SEAT_LETTERS[letter] || letter,
|
||||
price: numeric,
|
||||
currency: 'CNY',
|
||||
});
|
||||
}
|
||||
rows.sort((a, b) => Number(b.price) - Number(a.price));
|
||||
return rows;
|
||||
}
|
||||
|
||||
cli({
|
||||
site: '12306',
|
||||
name: 'price',
|
||||
access: 'read',
|
||||
description: 'Look up 12306 ticket prices by seat class for one train on a given date and segment (anonymous, no login required)',
|
||||
domain: 'kyfw.12306.cn',
|
||||
strategy: Strategy.PUBLIC,
|
||||
browser: false,
|
||||
args: [
|
||||
{ name: 'train-no', positional: true, required: true, help: 'Internal train_no from `12306 trains` (e.g. 24000000G10L)' },
|
||||
{ name: 'from', required: true, help: 'Origin station (Chinese name, telecode, or pinyin) - must be a stop of this train' },
|
||||
{ name: 'to', required: true, help: 'Destination station - must be a stop of this train' },
|
||||
{ name: 'date', required: true, help: 'Departure date in YYYY-MM-DD' },
|
||||
{ name: 'seat-types', default: 'OM9PA1A3A4FWZ', help: 'Seat-type letters to query (default covers the common classes). Examples: OM9 (二等/一等/商务), A1A3A4 (硬座/硬卧/软卧).' },
|
||||
],
|
||||
columns: ['seat_code', 'seat_name', 'price', 'currency'],
|
||||
func: async (kwargs) => {
|
||||
const trainNo = String(kwargs['train-no'] ?? '').trim();
|
||||
if (!trainNo) throw new ArgumentError('<train-no> must not be empty');
|
||||
if (!TRAIN_NO_RE.test(trainNo)) {
|
||||
throw new ArgumentError(
|
||||
`<train-no> "${trainNo}" does not look like a 12306 internal train_no`,
|
||||
'Use the train_no field from `12306 trains` output (e.g. 24000000G10L), not the public code (G1).',
|
||||
);
|
||||
}
|
||||
const fromArg = String(kwargs.from ?? '').trim();
|
||||
const toArg = String(kwargs.to ?? '').trim();
|
||||
if (!fromArg) throw new ArgumentError('--from station must not be empty');
|
||||
if (!toArg) throw new ArgumentError('--to station must not be empty');
|
||||
const date = validateDate(kwargs.date);
|
||||
const seatTypes = String(kwargs['seat-types'] ?? '').trim() || 'OM9PA1A3A4FWZ';
|
||||
if (!SEAT_TYPES_RE.test(seatTypes)) {
|
||||
throw new ArgumentError('--seat-types must contain only 12306 seat letters/digits (A-Z, 0-9)');
|
||||
}
|
||||
|
||||
const stations = await fetchStationBundle();
|
||||
const fromStation = resolveStation(stations, fromArg);
|
||||
const toStation = resolveStation(stations, toArg);
|
||||
if (fromStation.code === toStation.code) {
|
||||
throw new ArgumentError(`--from and --to must differ; both resolved to ${fromStation.name} (${fromStation.code})`);
|
||||
}
|
||||
|
||||
const cookieHeader = await mintSession();
|
||||
const stops = await queryStopsForPrice(cookieHeader, trainNo, fromStation.code, toStation.code, date);
|
||||
const { fromNo, toNo } = pickStationNos(stops, fromStation.code, toStation.code, fromStation.name, toStation.name);
|
||||
const priceData = await queryPrice(cookieHeader, trainNo, fromNo, toNo, seatTypes, date);
|
||||
const rows = parsePriceData(priceData);
|
||||
if (rows.length === 0) {
|
||||
throw new EmptyResultError(
|
||||
`No prices returned for train_no=${trainNo} ${fromStation.name} -> ${toStation.name} on ${date}`,
|
||||
'Try a different seat-types letter set, or check that this train operates on the date.',
|
||||
);
|
||||
}
|
||||
return rows;
|
||||
},
|
||||
});
|
||||
|
||||
export const __test__ = { parsePriceData, pickStationNos, queryStopsForPrice, queryPrice, SEAT_LETTERS };
|
||||
@@ -1,66 +0,0 @@
|
||||
/**
|
||||
* 12306 station search.
|
||||
*
|
||||
* Queries the public `station_name.js` bundle and filters by the user's
|
||||
* keyword. Anonymous, no session needed.
|
||||
*/
|
||||
import { cli, Strategy } from '@jackwener/opencli/registry';
|
||||
import { ArgumentError, EmptyResultError } from '@jackwener/opencli/errors';
|
||||
import { fetchStationBundle } from './utils.js';
|
||||
|
||||
const MAX_LIMIT = 50;
|
||||
|
||||
function normalizeLimit(value, defaultValue, max) {
|
||||
if (value === undefined || value === null || value === '') return defaultValue;
|
||||
const n = Number(value);
|
||||
if (!Number.isInteger(n) || n < 1) {
|
||||
throw new ArgumentError(`limit must be a positive integer (1-${max})`);
|
||||
}
|
||||
if (n > max) {
|
||||
throw new ArgumentError(`limit must be <= ${max}`);
|
||||
}
|
||||
return n;
|
||||
}
|
||||
|
||||
cli({
|
||||
site: '12306',
|
||||
name: 'stations',
|
||||
access: 'read',
|
||||
description: 'Search 12306 (China Railway) stations by Chinese name, telecode, or pinyin keyword',
|
||||
domain: 'kyfw.12306.cn',
|
||||
strategy: Strategy.PUBLIC,
|
||||
browser: false,
|
||||
args: [
|
||||
{ name: 'keyword', positional: true, required: true, help: 'Chinese substring (上海), telecode (AOH), or pinyin (shanghai)' },
|
||||
{ name: 'limit', type: 'int', default: 20, help: `Maximum results (1-${MAX_LIMIT})` },
|
||||
],
|
||||
columns: ['name', 'code', 'pinyin', 'abbr', 'city'],
|
||||
func: async (kwargs) => {
|
||||
const keyword = String(kwargs.keyword ?? '').trim();
|
||||
if (!keyword) throw new ArgumentError('keyword must not be empty');
|
||||
const limit = normalizeLimit(kwargs.limit, 20, MAX_LIMIT);
|
||||
|
||||
const stations = await fetchStationBundle();
|
||||
const lower = keyword.toLowerCase();
|
||||
const matches = stations.filter((s) =>
|
||||
s.name.includes(keyword)
|
||||
|| s.code === keyword.toUpperCase()
|
||||
|| s.pinyin.includes(lower)
|
||||
|| s.abbr.includes(lower)
|
||||
|| s.short.includes(lower)
|
||||
|| s.city.includes(keyword),
|
||||
);
|
||||
if (matches.length === 0) {
|
||||
throw new EmptyResultError(`No 12306 stations match "${keyword}"`);
|
||||
}
|
||||
return matches.slice(0, limit).map((s) => ({
|
||||
name: s.name,
|
||||
code: s.code,
|
||||
pinyin: s.pinyin,
|
||||
abbr: s.abbr,
|
||||
city: s.city,
|
||||
}));
|
||||
},
|
||||
});
|
||||
|
||||
export const __test__ = { normalizeLimit };
|
||||
@@ -1,91 +0,0 @@
|
||||
/**
|
||||
* 12306 train stop details - list every station a train calls at,
|
||||
* with arrival / departure / stopover time.
|
||||
*
|
||||
* Requires the internal `train_no` returned by `12306 trains`
|
||||
* (`24000000G10L`), not the public train code (`G1`).
|
||||
*/
|
||||
import { cli, Strategy } from '@jackwener/opencli/registry';
|
||||
import { ArgumentError, CommandExecutionError, EmptyResultError } from '@jackwener/opencli/errors';
|
||||
import { fetchStationBundle, mintSession, resolveStation, validateDate } from './utils.js';
|
||||
|
||||
const UA = 'Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/127.0 Safari/537.36';
|
||||
const TRAIN_NO_RE = /^[0-9A-Z]{8,18}$/;
|
||||
|
||||
async function queryStops(cookieHeader, trainNo, fromCode, toCode, date, fetchImpl = fetch) {
|
||||
const url = `https://kyfw.12306.cn/otn/czxx/queryByTrainNo?train_no=${trainNo}&from_station_telecode=${fromCode}&to_station_telecode=${toCode}&depart_date=${date}`;
|
||||
const resp = await fetchImpl(url, {
|
||||
headers: {
|
||||
'User-Agent': UA,
|
||||
'Referer': 'https://kyfw.12306.cn/otn/leftTicket/init',
|
||||
'Cookie': cookieHeader,
|
||||
},
|
||||
});
|
||||
if (!resp.ok) {
|
||||
throw new CommandExecutionError(`12306 queryByTrainNo returned HTTP ${resp.status}`);
|
||||
}
|
||||
let json;
|
||||
try {
|
||||
json = await resp.json();
|
||||
} catch {
|
||||
throw new CommandExecutionError('12306 queryByTrainNo returned non-JSON body');
|
||||
}
|
||||
if (json?.status !== true || !Array.isArray(json?.data?.data)) {
|
||||
throw new CommandExecutionError(`12306 queryByTrainNo returned an unexpected payload shape`);
|
||||
}
|
||||
return json.data.data;
|
||||
}
|
||||
|
||||
cli({
|
||||
site: '12306',
|
||||
name: 'train',
|
||||
access: 'read',
|
||||
description: 'List every station a 12306 train calls at, with arrival / departure / stopover time (anonymous, no login required)',
|
||||
domain: 'kyfw.12306.cn',
|
||||
strategy: Strategy.PUBLIC,
|
||||
browser: false,
|
||||
args: [
|
||||
{ name: 'train-no', positional: true, required: true, help: 'Internal train_no from `12306 trains` (e.g. 24000000G10L), not the public code (G1)' },
|
||||
{ name: 'from', required: true, help: 'Origin station for the segment: Chinese name, telecode, or pinyin' },
|
||||
{ name: 'to', required: true, help: 'Destination station for the segment' },
|
||||
{ name: 'date', required: true, help: 'Departure date in YYYY-MM-DD' },
|
||||
],
|
||||
columns: ['station_no', 'station_name', 'arrive_time', 'start_time', 'stopover_time'],
|
||||
func: async (kwargs) => {
|
||||
const trainNo = String(kwargs['train-no'] ?? '').trim();
|
||||
if (!trainNo) throw new ArgumentError('<train-no> must not be empty');
|
||||
if (!TRAIN_NO_RE.test(trainNo)) {
|
||||
throw new ArgumentError(
|
||||
`<train-no> "${trainNo}" does not look like a 12306 internal train_no`,
|
||||
'Use the train_no field from `12306 trains` output (e.g. 24000000G10L), not the public code (G1).',
|
||||
);
|
||||
}
|
||||
const fromArg = String(kwargs.from ?? '').trim();
|
||||
const toArg = String(kwargs.to ?? '').trim();
|
||||
if (!fromArg) throw new ArgumentError('--from station must not be empty');
|
||||
if (!toArg) throw new ArgumentError('--to station must not be empty');
|
||||
const date = validateDate(kwargs.date);
|
||||
|
||||
const stations = await fetchStationBundle();
|
||||
const fromStation = resolveStation(stations, fromArg);
|
||||
const toStation = resolveStation(stations, toArg);
|
||||
if (fromStation.code === toStation.code) {
|
||||
throw new ArgumentError(`--from and --to must differ; both resolved to ${fromStation.name} (${fromStation.code})`);
|
||||
}
|
||||
|
||||
const cookieHeader = await mintSession();
|
||||
const stops = await queryStops(cookieHeader, trainNo, fromStation.code, toStation.code, date);
|
||||
if (stops.length === 0) {
|
||||
throw new EmptyResultError(`No stops returned for train_no=${trainNo} on ${date}`);
|
||||
}
|
||||
return stops.map((s) => ({
|
||||
station_no: s.station_no || '',
|
||||
station_name: s.station_name || '',
|
||||
arrive_time: s.arrive_time === '----' ? '' : (s.arrive_time || ''),
|
||||
start_time: s.start_time === '----' ? '' : (s.start_time || ''),
|
||||
stopover_time: s.stopover_time === '----' ? '' : (s.stopover_time || ''),
|
||||
}));
|
||||
},
|
||||
});
|
||||
|
||||
export const __test__ = { queryStops, TRAIN_NO_RE };
|
||||
@@ -1,119 +0,0 @@
|
||||
/**
|
||||
* 12306 train availability between two stations on a given date.
|
||||
*
|
||||
* Flow:
|
||||
* 1. Fetch the station bundle (cached implicitly via per-process module state).
|
||||
* 2. Mint anonymous session cookies via /otn/leftTicket/init.
|
||||
* 3. Query /otn/leftTicket/queryG; if 12306 returns
|
||||
* `{c_url: "leftTicket/queryX"}` (endpoint rotation), retry once
|
||||
* against the suggested name.
|
||||
* 4. Parse the `|`-separated train records.
|
||||
*/
|
||||
import { cli, Strategy } from '@jackwener/opencli/registry';
|
||||
import { ArgumentError, CommandExecutionError, EmptyResultError } from '@jackwener/opencli/errors';
|
||||
import { fetchStationBundle, mintSession, resolveStation, validateDate, parseTrainRecord } from './utils.js';
|
||||
|
||||
const UA = 'Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/127.0 Safari/537.36';
|
||||
const QUERY_ENDPOINTS = ['queryG', 'queryO', 'queryZ', 'queryA'];
|
||||
const MAX_LIMIT = 100;
|
||||
|
||||
function normalizeLimit(value, defaultValue, max) {
|
||||
if (value === undefined || value === null || value === '') return defaultValue;
|
||||
const n = Number(value);
|
||||
if (!Number.isInteger(n) || n < 1) {
|
||||
throw new ArgumentError(`limit must be a positive integer (1-${max})`);
|
||||
}
|
||||
if (n > max) {
|
||||
throw new ArgumentError(`limit must be <= ${max}`);
|
||||
}
|
||||
return n;
|
||||
}
|
||||
|
||||
async function queryLeftTickets(cookieHeader, fromCode, toCode, date) {
|
||||
const headers = {
|
||||
'User-Agent': UA,
|
||||
'Referer': 'https://kyfw.12306.cn/otn/leftTicket/init',
|
||||
'Cookie': cookieHeader,
|
||||
};
|
||||
const queryParams = `leftTicketDTO.train_date=${date}&leftTicketDTO.from_station=${fromCode}&leftTicketDTO.to_station=${toCode}&purpose_codes=ADULT`;
|
||||
let lastResponseText = '';
|
||||
for (const endpoint of QUERY_ENDPOINTS) {
|
||||
const url = `https://kyfw.12306.cn/otn/leftTicket/${endpoint}?${queryParams}`;
|
||||
const resp = await fetch(url, { headers });
|
||||
if (!resp.ok) {
|
||||
if (resp.status === 302) continue;
|
||||
throw new CommandExecutionError(`12306 ${endpoint} returned HTTP ${resp.status}`);
|
||||
}
|
||||
const text = await resp.text();
|
||||
lastResponseText = text;
|
||||
let json;
|
||||
try { json = JSON.parse(text); } catch {
|
||||
throw new CommandExecutionError(`12306 ${endpoint} returned non-JSON body`);
|
||||
}
|
||||
if (json?.c_url && typeof json.c_url === 'string') {
|
||||
const rotated = json.c_url.replace('leftTicket/', '').trim();
|
||||
if (rotated && !QUERY_ENDPOINTS.includes(rotated)) {
|
||||
QUERY_ENDPOINTS.unshift(rotated);
|
||||
}
|
||||
continue;
|
||||
}
|
||||
if (Array.isArray(json?.data?.result)) {
|
||||
return json.data.result;
|
||||
}
|
||||
throw new CommandExecutionError(`12306 ${endpoint} returned an unexpected payload shape`);
|
||||
}
|
||||
throw new CommandExecutionError(`12306 rejected every known query endpoint name (${QUERY_ENDPOINTS.join(', ')}); the wire protocol may have changed. Last body: ${lastResponseText.slice(0, 200)}`);
|
||||
}
|
||||
|
||||
cli({
|
||||
site: '12306',
|
||||
name: 'trains',
|
||||
access: 'read',
|
||||
description: 'List trains between two 12306 stations on a given date (anonymous, no login required)',
|
||||
domain: 'kyfw.12306.cn',
|
||||
strategy: Strategy.PUBLIC,
|
||||
browser: false,
|
||||
args: [
|
||||
{ name: 'from', positional: true, required: true, help: 'Origin station: Chinese name (北京), telecode (BJP), or pinyin (beijing)' },
|
||||
{ name: 'to', positional: true, required: true, help: 'Destination station: same forms as <from>' },
|
||||
{ name: 'date', required: true, help: 'Departure date in YYYY-MM-DD' },
|
||||
{ name: 'limit', type: 'int', default: 50, help: `Maximum rows (1-${MAX_LIMIT})` },
|
||||
],
|
||||
columns: [
|
||||
'code', 'from_station', 'to_station', 'start_time', 'arrive_time',
|
||||
'duration', 'available', 'business_seat', 'first_seat', 'second_seat',
|
||||
'soft_sleeper', 'hard_sleeper', 'hard_seat', 'no_seat', 'train_no',
|
||||
],
|
||||
func: async (kwargs) => {
|
||||
const fromArg = String(kwargs.from ?? '').trim();
|
||||
const toArg = String(kwargs.to ?? '').trim();
|
||||
if (!fromArg) throw new ArgumentError('<from> station must not be empty');
|
||||
if (!toArg) throw new ArgumentError('<to> station must not be empty');
|
||||
const date = validateDate(kwargs.date);
|
||||
const limit = normalizeLimit(kwargs.limit, 50, MAX_LIMIT);
|
||||
|
||||
const stations = await fetchStationBundle();
|
||||
const fromStation = resolveStation(stations, fromArg);
|
||||
const toStation = resolveStation(stations, toArg);
|
||||
if (fromStation.code === toStation.code) {
|
||||
throw new ArgumentError(`<from> and <to> must differ; both resolved to ${fromStation.name} (${fromStation.code})`);
|
||||
}
|
||||
const stationByCode = new Map(stations.map((s) => [s.code, s]));
|
||||
|
||||
const cookieHeader = await mintSession();
|
||||
const rawRows = await queryLeftTickets(cookieHeader, fromStation.code, toStation.code, date);
|
||||
const decoded = rawRows
|
||||
.map((line) => parseTrainRecord(decodeURIComponent(line.replace(/%0A/g, '')), stationByCode))
|
||||
.filter(Boolean);
|
||||
|
||||
if (decoded.length === 0) {
|
||||
throw new EmptyResultError(
|
||||
`No trains found from ${fromStation.name} to ${toStation.name} on ${date}`,
|
||||
'Try a different date or check whether the route is operated by 12306.',
|
||||
);
|
||||
}
|
||||
return decoded.slice(0, limit);
|
||||
},
|
||||
});
|
||||
|
||||
export const __test__ = { normalizeLimit, queryLeftTickets };
|
||||
@@ -1,272 +0,0 @@
|
||||
/**
|
||||
* 12306 (中国铁路) shared helpers.
|
||||
*
|
||||
* - Station lookup: parses the public `station_name.js` bundle into
|
||||
* structured records.
|
||||
* - Cookie session: 12306's query endpoints reject anonymous requests
|
||||
* with `HTTP 302 -> error.html`, so callers must hit `/otn/leftTicket/init`
|
||||
* first to mint the JSESSIONID / route / BIGipServerotn cookies.
|
||||
* - Query endpoint rotation: 12306 rotates the train-query endpoint
|
||||
* name (queryO / queryZ / queryA / queryG / ...) every few weeks.
|
||||
* When the wrong name is hit, the server returns
|
||||
* `{"c_url":"leftTicket/queryG","c_name":"CLeftTicketUrl","status":false}`
|
||||
* pointing to the current correct name; retry once with that name.
|
||||
*/
|
||||
import { ArgumentError, CommandExecutionError } from '@jackwener/opencli/errors';
|
||||
|
||||
const STATION_BUNDLE_URL = 'https://kyfw.12306.cn/otn/resources/js/framework/station_name.js';
|
||||
const INIT_URL = 'https://kyfw.12306.cn/otn/leftTicket/init';
|
||||
const UA = 'Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/127.0 Safari/537.36';
|
||||
|
||||
const DATE_RE = /^\d{4}-\d{2}-\d{2}$/;
|
||||
const STATION_CODE_RE = /^[A-Z]{2,4}$/;
|
||||
|
||||
/**
|
||||
* Parse the `station_name.js` bundle into a station record array.
|
||||
*
|
||||
* Bundle format (single line, `@`-delimited records, each `|`-delimited):
|
||||
* `var station_names ='@bjb|北京北|VAP|beijingbei|bjb|0|0357|北京|||...';`
|
||||
*
|
||||
* Per-record fields (positional):
|
||||
* [0] short pinyin alias (e.g. `bjb`)
|
||||
* [1] Chinese station name (e.g. `北京北`)
|
||||
* [2] telecode (3-4 uppercase letters, e.g. `VAP`) - this is the
|
||||
* wire format 12306 uses for `from_station` / `to_station`.
|
||||
* [3] full pinyin (e.g. `beijingbei`)
|
||||
* [4] short alias (duplicate of [0] usually)
|
||||
* [5] index/rank
|
||||
* [6] city code
|
||||
* [7] city name (e.g. `北京`)
|
||||
*/
|
||||
export function parseStationBundle(text) {
|
||||
const match = text.match(/'([^']+)'/);
|
||||
if (!match) {
|
||||
throw new CommandExecutionError('Failed to parse 12306 station_name.js: source string not found');
|
||||
}
|
||||
const raw = match[1];
|
||||
const records = raw.split('@').filter(Boolean);
|
||||
const stations = [];
|
||||
for (const r of records) {
|
||||
const parts = r.split('|');
|
||||
if (parts.length < 8 || !parts[2]) continue;
|
||||
stations.push({
|
||||
short: parts[0] || '',
|
||||
name: parts[1] || '',
|
||||
code: parts[2] || '',
|
||||
pinyin: parts[3] || '',
|
||||
abbr: parts[4] || '',
|
||||
city: parts[7] || '',
|
||||
});
|
||||
}
|
||||
if (stations.length === 0) {
|
||||
throw new CommandExecutionError('Failed to parse 12306 station_name.js: no station records found');
|
||||
}
|
||||
return stations;
|
||||
}
|
||||
|
||||
/**
|
||||
* Resolve a user-supplied station identifier to a telecode.
|
||||
*
|
||||
* Accepts Chinese name (`上海虹桥`), telecode (`AOH`), pinyin
|
||||
* (`shanghaihongqiao`), short alias (`shh`), or city name with a
|
||||
* preference for the city's main station.
|
||||
*/
|
||||
export function resolveStation(stations, input) {
|
||||
const trimmed = String(input ?? '').trim();
|
||||
if (!trimmed) throw new ArgumentError('station must not be empty');
|
||||
if (STATION_CODE_RE.test(trimmed)) {
|
||||
const exact = stations.find((s) => s.code === trimmed);
|
||||
if (exact) return exact;
|
||||
throw new ArgumentError(`Unknown 12306 station telecode "${trimmed}"`);
|
||||
}
|
||||
const lower = trimmed.toLowerCase();
|
||||
const exactName = stations.find((s) => s.name === trimmed);
|
||||
if (exactName) return exactName;
|
||||
const exactPinyin = stations.find((s) => s.pinyin === lower);
|
||||
if (exactPinyin) return exactPinyin;
|
||||
const exactAbbr = stations.find((s) => s.abbr === lower || s.short === lower);
|
||||
if (exactAbbr) return exactAbbr;
|
||||
throw new ArgumentError(`Unknown 12306 station "${trimmed}"`, 'Try the Chinese name (上海虹桥), the 3-4 letter telecode (AOH), or full pinyin (shanghaihongqiao).');
|
||||
}
|
||||
|
||||
export function validateDate(value) {
|
||||
if (!DATE_RE.test(String(value ?? ''))) {
|
||||
throw new ArgumentError(`date must be YYYY-MM-DD, got "${value}"`);
|
||||
}
|
||||
const [y, m, d] = value.split('-').map(Number);
|
||||
const date = new Date(Date.UTC(y, m - 1, d));
|
||||
if (date.getUTCFullYear() !== y || date.getUTCMonth() !== m - 1 || date.getUTCDate() !== d) {
|
||||
throw new ArgumentError(`date "${value}" is not a real calendar date`);
|
||||
}
|
||||
return value;
|
||||
}
|
||||
|
||||
/** Extract Set-Cookie header values into a single `Cookie:` header string. */
|
||||
export function buildCookieHeader(setCookieHeaders) {
|
||||
if (!Array.isArray(setCookieHeaders) || setCookieHeaders.length === 0) return '';
|
||||
return setCookieHeaders
|
||||
.map((line) => line.split(';')[0])
|
||||
.filter(Boolean)
|
||||
.join('; ');
|
||||
}
|
||||
|
||||
export async function fetchStationBundle(fetchImpl = fetch) {
|
||||
const resp = await fetchImpl(STATION_BUNDLE_URL, {
|
||||
headers: { 'User-Agent': UA },
|
||||
});
|
||||
if (!resp.ok) {
|
||||
throw new CommandExecutionError(`Failed to fetch 12306 station bundle: HTTP ${resp.status}`);
|
||||
}
|
||||
return parseStationBundle(await resp.text());
|
||||
}
|
||||
|
||||
/** Mint a 12306 anonymous session by hitting /otn/leftTicket/init. */
|
||||
export async function mintSession(fetchImpl = fetch) {
|
||||
const resp = await fetchImpl(INIT_URL, {
|
||||
headers: { 'User-Agent': UA },
|
||||
redirect: 'follow',
|
||||
});
|
||||
if (!resp.ok) {
|
||||
throw new CommandExecutionError(`Failed to mint 12306 session: HTTP ${resp.status}`);
|
||||
}
|
||||
const setCookies = typeof resp.headers.getSetCookie === 'function'
|
||||
? resp.headers.getSetCookie()
|
||||
: resp.headers.raw?.()['set-cookie'] || [];
|
||||
const cookieHeader = buildCookieHeader(setCookies);
|
||||
if (!cookieHeader) {
|
||||
throw new CommandExecutionError('12306 init returned no session cookies');
|
||||
}
|
||||
return cookieHeader;
|
||||
}
|
||||
|
||||
/**
|
||||
* Twelve-row train query record (LEFT_TICKET_DTO).
|
||||
*
|
||||
* 12306 returns each train as a `|`-separated string with ~36 fields.
|
||||
* Positions used here come from the public web client; unused
|
||||
* positions are documented inline so future maintainers can extend
|
||||
* the row shape without re-reverse-engineering.
|
||||
*/
|
||||
export function parseTrainRecord(line, stationByCode) {
|
||||
const f = line.split('|');
|
||||
if (f.length < 33) return null;
|
||||
return {
|
||||
train_no: f[2] || '',
|
||||
code: f[3] || '',
|
||||
from_station: stationByCode.get(f[6])?.name || f[6] || '',
|
||||
to_station: stationByCode.get(f[7])?.name || f[7] || '',
|
||||
from_code: f[6] || '',
|
||||
to_code: f[7] || '',
|
||||
start_time: f[8] || '',
|
||||
arrive_time: f[9] || '',
|
||||
duration: f[10] || '',
|
||||
available: (f[1] || '').trim() === '预订' || (f[11] || '').trim() === 'Y',
|
||||
business_seat: f[32] || '',
|
||||
first_seat: f[31] || '',
|
||||
second_seat: f[30] || '',
|
||||
soft_sleeper: f[23] || '',
|
||||
hard_sleeper: f[28] || '',
|
||||
hard_seat: f[29] || '',
|
||||
no_seat: f[26] || '',
|
||||
};
|
||||
}
|
||||
|
||||
/**
|
||||
* Mask helpers for sensitive identity fields rendered by 12306.
|
||||
*
|
||||
* 12306 already masks ID numbers and mobile numbers server-side
|
||||
* (`xxxx***********xxx` / `138****xxxx`); these helpers handle the
|
||||
* remaining fields (email, real Chinese name) so the adapter never
|
||||
* leaks unmasked PII without an explicit `--include-sensitive` opt-in.
|
||||
*/
|
||||
export function maskEmail(value) {
|
||||
const v = String(value || '').trim();
|
||||
if (!v) return '';
|
||||
const at = v.indexOf('@');
|
||||
if (at <= 0) return v;
|
||||
const local = v.slice(0, at);
|
||||
const domain = v.slice(at);
|
||||
if (local.length <= 2) return local[0] + '*' + domain;
|
||||
return local[0] + '*'.repeat(Math.max(1, local.length - 2)) + local.slice(-1) + domain;
|
||||
}
|
||||
|
||||
export function maskMobile(value) {
|
||||
const v = String(value || '').trim();
|
||||
if (!v) return '';
|
||||
if (/\*/.test(v)) return v;
|
||||
if (v.length < 7) return v.replace(/.(?=.)/g, '*');
|
||||
return v.slice(0, 3) + '*'.repeat(v.length - 7) + v.slice(-4);
|
||||
}
|
||||
|
||||
export function maskChineseName(value) {
|
||||
const v = String(value || '').trim();
|
||||
if (!v) return '';
|
||||
if (v.length === 1) return v;
|
||||
if (v.length === 2) return v[0] + '*';
|
||||
return v[0] + '*'.repeat(v.length - 2) + v.slice(-1);
|
||||
}
|
||||
|
||||
export function unwrapEvaluateResult(value) {
|
||||
if (
|
||||
value
|
||||
&& typeof value === 'object'
|
||||
&& !Array.isArray(value)
|
||||
&& Object.prototype.hasOwnProperty.call(value, 'session')
|
||||
&& Object.prototype.hasOwnProperty.call(value, 'data')
|
||||
) {
|
||||
return value.data;
|
||||
}
|
||||
return value;
|
||||
}
|
||||
|
||||
export function requireEvaluateObject(value, label) {
|
||||
const payload = unwrapEvaluateResult(value);
|
||||
if (!payload || typeof payload !== 'object' || Array.isArray(payload)) {
|
||||
throw new CommandExecutionError(`12306 ${label} returned a malformed browser payload`);
|
||||
}
|
||||
return payload;
|
||||
}
|
||||
|
||||
export function isAuthLikePayload(payload) {
|
||||
if (!payload || typeof payload !== 'object') return false;
|
||||
const parts = [];
|
||||
if (Array.isArray(payload.messages)) parts.push(...payload.messages);
|
||||
if (payload.message) parts.push(payload.message);
|
||||
if (payload.msg) parts.push(payload.msg);
|
||||
if (payload.validateMessages && typeof payload.validateMessages === 'object') {
|
||||
parts.push(...Object.values(payload.validateMessages).flat());
|
||||
}
|
||||
const text = parts.map((item) => String(item ?? '')).join(' ');
|
||||
return /未登录|登录|请登录|身份|认证|session|Session|login/i.test(text);
|
||||
}
|
||||
|
||||
/**
|
||||
* Detect the 12306 login marker by reading `document.cookie` from the
|
||||
* current adapter page. Cannot use `page.getCookies({url})` here:
|
||||
* 12306 sets the auth cookie `tk` and `JSESSIONID` with `Path=/otn`,
|
||||
* and CDP `Network.getCookies` with a bare URL filter excludes
|
||||
* cookies whose path does not match the URL path. `document.cookie`
|
||||
* returns all non-httponly cookies visible to the current page
|
||||
* regardless of path, which is what we need to confirm login.
|
||||
*/
|
||||
export async function require12306Login(page, AuthRequiredErrorClass) {
|
||||
const docCookie = unwrapEvaluateResult(await page.evaluate(`document.cookie || ''`));
|
||||
const cookieStr = typeof docCookie === 'string' ? docCookie : '';
|
||||
if (!/\btk=/.test(cookieStr) || !/JSESSIONID=/.test(cookieStr)) {
|
||||
throw new AuthRequiredErrorClass('kyfw.12306.cn', 'Not logged into 12306. Sign in at https://kyfw.12306.cn first.');
|
||||
}
|
||||
}
|
||||
|
||||
export const __test__ = {
|
||||
parseStationBundle,
|
||||
resolveStation,
|
||||
validateDate,
|
||||
buildCookieHeader,
|
||||
parseTrainRecord,
|
||||
maskEmail,
|
||||
maskMobile,
|
||||
maskChineseName,
|
||||
unwrapEvaluateResult,
|
||||
requireEvaluateObject,
|
||||
isAuthLikePayload,
|
||||
};
|
||||
@@ -1,331 +0,0 @@
|
||||
import { describe, expect, it } from 'vitest';
|
||||
import { ArgumentError, AuthRequiredError, CommandExecutionError, EmptyResultError } from '@jackwener/opencli/errors';
|
||||
import { getRegistry } from '@jackwener/opencli/registry';
|
||||
import { __test__ } from './utils.js';
|
||||
import { __test__ as priceTest } from './price.js';
|
||||
import { __test__ as trainTest } from './train.js';
|
||||
import './orders.js';
|
||||
|
||||
const { parseStationBundle, resolveStation, validateDate, buildCookieHeader, parseTrainRecord, maskEmail, maskMobile, maskChineseName, unwrapEvaluateResult, requireEvaluateObject, isAuthLikePayload } = __test__;
|
||||
const { parsePriceData, queryStopsForPrice, queryPrice } = priceTest;
|
||||
const { queryStops } = trainTest;
|
||||
|
||||
describe('12306 utils - parseStationBundle', () => {
|
||||
it('parses the `@`-delimited station bundle into structured records', () => {
|
||||
const bundle = "var station_names ='@bjb|北京北|VAP|beijingbei|bjb|0|0357|北京|||@bji|北京|BJP|beijing|bj|2|0357|北京|||@aoh|上海虹桥|AOH|shanghaihongqiao|shhq|10|7600|上海|||';";
|
||||
const stations = parseStationBundle(bundle);
|
||||
expect(stations).toHaveLength(3);
|
||||
expect(stations[1]).toEqual({
|
||||
short: 'bji', name: '北京', code: 'BJP', pinyin: 'beijing', abbr: 'bj', city: '北京',
|
||||
});
|
||||
});
|
||||
|
||||
it('skips records that lack a telecode', () => {
|
||||
const bundle = "var station_names ='@xxx|||||||||@bji|北京|BJP|beijing|bj|2|0357|北京|||';";
|
||||
const stations = parseStationBundle(bundle);
|
||||
expect(stations).toHaveLength(1);
|
||||
expect(stations[0].code).toBe('BJP');
|
||||
});
|
||||
|
||||
it('throws CommandExecutionError when the bundle has no parseable station rows', () => {
|
||||
expect(() => parseStationBundle("var station_names ='@xxx|||||||||';")).toThrow(CommandExecutionError);
|
||||
});
|
||||
});
|
||||
|
||||
describe('12306 utils - resolveStation', () => {
|
||||
const stations = [
|
||||
{ short: 'bjb', name: '北京北', code: 'VAP', pinyin: 'beijingbei', abbr: 'bjb', city: '北京' },
|
||||
{ short: 'bji', name: '北京', code: 'BJP', pinyin: 'beijing', abbr: 'bj', city: '北京' },
|
||||
{ short: 'aoh', name: '上海虹桥', code: 'AOH', pinyin: 'shanghaihongqiao', abbr: 'shhq', city: '上海' },
|
||||
];
|
||||
|
||||
it('matches by exact Chinese name', () => {
|
||||
expect(resolveStation(stations, '上海虹桥').code).toBe('AOH');
|
||||
});
|
||||
|
||||
it('matches by uppercase telecode', () => {
|
||||
expect(resolveStation(stations, 'BJP').code).toBe('BJP');
|
||||
});
|
||||
|
||||
it('matches by full pinyin (case-insensitive)', () => {
|
||||
expect(resolveStation(stations, 'Beijing').code).toBe('BJP');
|
||||
});
|
||||
|
||||
it('matches by short alias / abbr', () => {
|
||||
expect(resolveStation(stations, 'shhq').code).toBe('AOH');
|
||||
});
|
||||
|
||||
it('throws ArgumentError for empty input', () => {
|
||||
expect(() => resolveStation(stations, ' ')).toThrow(ArgumentError);
|
||||
});
|
||||
|
||||
it('throws ArgumentError for unknown station', () => {
|
||||
expect(() => resolveStation(stations, '某不存在站')).toThrow(ArgumentError);
|
||||
});
|
||||
|
||||
it('throws ArgumentError for telecode-shaped but unknown input', () => {
|
||||
expect(() => resolveStation(stations, 'XYZ')).toThrow(ArgumentError);
|
||||
});
|
||||
});
|
||||
|
||||
describe('12306 utils - validateDate', () => {
|
||||
it('accepts valid YYYY-MM-DD', () => {
|
||||
expect(validateDate('2026-05-22')).toBe('2026-05-22');
|
||||
});
|
||||
|
||||
it('throws ArgumentError on wrong format', () => {
|
||||
expect(() => validateDate('2026/05/22')).toThrow(ArgumentError);
|
||||
expect(() => validateDate('26-05-22')).toThrow(ArgumentError);
|
||||
expect(() => validateDate('today')).toThrow(ArgumentError);
|
||||
expect(() => validateDate('')).toThrow(ArgumentError);
|
||||
});
|
||||
|
||||
it('throws ArgumentError on impossible calendar dates', () => {
|
||||
expect(() => validateDate('2026-02-30')).toThrow(ArgumentError);
|
||||
expect(() => validateDate('2026-13-01')).toThrow(ArgumentError);
|
||||
});
|
||||
});
|
||||
|
||||
describe('12306 utils - buildCookieHeader', () => {
|
||||
it('joins set-cookie lines into a single Cookie header', () => {
|
||||
const headers = [
|
||||
'JSESSIONID=ABC123; Path=/otn',
|
||||
'BIGipServerotn=xxx.yyy; Path=/',
|
||||
'route=zzz; Expires=Sat, 01 Jan 2027 00:00:00 GMT',
|
||||
];
|
||||
expect(buildCookieHeader(headers)).toBe('JSESSIONID=ABC123; BIGipServerotn=xxx.yyy; route=zzz');
|
||||
});
|
||||
|
||||
it('returns empty string for empty input', () => {
|
||||
expect(buildCookieHeader([])).toBe('');
|
||||
expect(buildCookieHeader(undefined)).toBe('');
|
||||
});
|
||||
});
|
||||
|
||||
describe('12306 utils - parseTrainRecord', () => {
|
||||
const stationByCode = new Map([
|
||||
['VNP', { name: '北京南', code: 'VNP' }],
|
||||
['AOH', { name: '上海虹桥', code: 'AOH' }],
|
||||
]);
|
||||
|
||||
it('extracts the canonical train fields from a wire record', () => {
|
||||
// 33 `|`-separated fields, with positions used by parseTrainRecord populated.
|
||||
const fields = new Array(36).fill('');
|
||||
fields[0] = 'SECRET_TOKEN';
|
||||
fields[1] = '预订';
|
||||
fields[2] = '240000G54700';
|
||||
fields[3] = 'G547';
|
||||
fields[6] = 'VNP';
|
||||
fields[7] = 'AOH';
|
||||
fields[8] = '06:18';
|
||||
fields[9] = '12:11';
|
||||
fields[10] = '05:53';
|
||||
fields[11] = 'Y';
|
||||
fields[23] = ''; // soft sleeper
|
||||
fields[26] = '无'; // no seat
|
||||
fields[28] = ''; // hard sleeper
|
||||
fields[29] = ''; // hard seat
|
||||
fields[30] = '有'; // second seat
|
||||
fields[31] = '有'; // first seat
|
||||
fields[32] = '无'; // business seat
|
||||
const row = parseTrainRecord(fields.join('|'), stationByCode);
|
||||
expect(row).toEqual({
|
||||
train_no: '240000G54700',
|
||||
code: 'G547',
|
||||
from_station: '北京南',
|
||||
to_station: '上海虹桥',
|
||||
from_code: 'VNP',
|
||||
to_code: 'AOH',
|
||||
start_time: '06:18',
|
||||
arrive_time: '12:11',
|
||||
duration: '05:53',
|
||||
available: true,
|
||||
business_seat: '无',
|
||||
first_seat: '有',
|
||||
second_seat: '有',
|
||||
soft_sleeper: '',
|
||||
hard_sleeper: '',
|
||||
hard_seat: '',
|
||||
no_seat: '无',
|
||||
});
|
||||
});
|
||||
|
||||
it('does not expose the booking-handshake secret token', () => {
|
||||
const fields = new Array(36).fill('');
|
||||
fields[0] = 'SECRET_TOKEN_DO_NOT_LEAK';
|
||||
fields[2] = 't_no'; fields[3] = 'X1'; fields[6] = 'VNP'; fields[7] = 'AOH';
|
||||
const row = parseTrainRecord(fields.join('|'), stationByCode);
|
||||
expect(Object.values(row)).not.toContain('SECRET_TOKEN_DO_NOT_LEAK');
|
||||
expect('secret' in row).toBe(false);
|
||||
});
|
||||
|
||||
it('falls back to the telecode when the station bundle has no name', () => {
|
||||
const fields = new Array(36).fill('');
|
||||
fields[2] = 'X'; fields[3] = 'X'; fields[6] = 'ZZZ'; fields[7] = 'YYY';
|
||||
const row = parseTrainRecord(fields.join('|'), stationByCode);
|
||||
expect(row.from_station).toBe('ZZZ');
|
||||
expect(row.to_station).toBe('YYY');
|
||||
});
|
||||
|
||||
it('returns null for short records', () => {
|
||||
expect(parseTrainRecord('a|b|c', stationByCode)).toBeNull();
|
||||
});
|
||||
});
|
||||
|
||||
describe('12306 utils - mask helpers', () => {
|
||||
it('masks the local-part of an email', () => {
|
||||
expect(maskEmail('hello@example.com')).toBe('h***o@example.com');
|
||||
expect(maskEmail('ab@x.cn')).toBe('a*@x.cn');
|
||||
expect(maskEmail('a@x.cn')).toBe('a*@x.cn');
|
||||
expect(maskEmail('')).toBe('');
|
||||
expect(maskEmail('not-an-email')).toBe('not-an-email');
|
||||
});
|
||||
|
||||
it('masks Chinese mobile numbers while preserving 12306-side masks', () => {
|
||||
expect(maskMobile('13800001234')).toBe('138****1234');
|
||||
expect(maskMobile('138****1234')).toBe('138****1234');
|
||||
expect(maskMobile('')).toBe('');
|
||||
expect(maskMobile('123')).toBe('**3');
|
||||
});
|
||||
|
||||
it('masks Chinese real names', () => {
|
||||
expect(maskChineseName('张三')).toBe('张*');
|
||||
expect(maskChineseName('李四明')).toBe('李*明');
|
||||
expect(maskChineseName('欧阳锋')).toBe('欧*锋');
|
||||
expect(maskChineseName('张')).toBe('张');
|
||||
expect(maskChineseName('')).toBe('');
|
||||
});
|
||||
});
|
||||
|
||||
describe('12306 price - parsePriceData', () => {
|
||||
it('returns seat rows sorted by descending price and drops dup numeric codes', () => {
|
||||
const data = {
|
||||
train_no: '24000000G10L',
|
||||
'OT': [],
|
||||
'A9': '¥2158.0',
|
||||
'9': '21580',
|
||||
'P': '¥1163.0',
|
||||
'M': '¥1035.0',
|
||||
'O': '¥626.0',
|
||||
'WZ': '¥626.0',
|
||||
'INVALID': 'not-a-price',
|
||||
};
|
||||
const rows = parsePriceData(data);
|
||||
const codes = rows.map((r) => r.seat_code);
|
||||
expect(codes).not.toContain('9');
|
||||
expect(codes).not.toContain('OT');
|
||||
expect(codes).not.toContain('train_no');
|
||||
expect(codes).not.toContain('INVALID');
|
||||
expect(codes).toEqual(['A9', 'P', 'M', 'O', 'WZ']);
|
||||
expect(rows[0]).toEqual({ seat_code: 'A9', seat_name: '商务座', price: '2158.0', currency: 'CNY' });
|
||||
expect(rows[4]).toEqual({ seat_code: 'WZ', seat_name: '无座', price: '626.0', currency: 'CNY' });
|
||||
});
|
||||
|
||||
it('keeps unknown letter codes with the letter as the name', () => {
|
||||
const data = { 'A9': '¥100.0', 'ZZ': '¥50.0' };
|
||||
const rows = parsePriceData(data);
|
||||
const zz = rows.find((r) => r.seat_code === 'ZZ');
|
||||
expect(zz?.seat_name).toBe('ZZ');
|
||||
});
|
||||
});
|
||||
|
||||
describe('12306 public API typed boundaries', () => {
|
||||
const nonJsonFetch = async () => ({
|
||||
ok: true,
|
||||
json: async () => {
|
||||
throw new SyntaxError('Unexpected token <');
|
||||
},
|
||||
});
|
||||
|
||||
it('wraps non-JSON train stop bodies as CommandExecutionError', async () => {
|
||||
await expect(queryStops('cookie=1', '24000000G10L', 'BJP', 'AOH', '2026-05-22', nonJsonFetch))
|
||||
.rejects.toBeInstanceOf(CommandExecutionError);
|
||||
});
|
||||
|
||||
it('wraps non-JSON price helper bodies as CommandExecutionError', async () => {
|
||||
await expect(queryStopsForPrice('cookie=1', '24000000G10L', 'BJP', 'AOH', '2026-05-22', nonJsonFetch))
|
||||
.rejects.toBeInstanceOf(CommandExecutionError);
|
||||
await expect(queryPrice('cookie=1', '24000000G10L', '01', '02', 'OM9', '2026-05-22', nonJsonFetch))
|
||||
.rejects.toBeInstanceOf(CommandExecutionError);
|
||||
});
|
||||
});
|
||||
|
||||
describe('12306 browser evaluate boundaries', () => {
|
||||
it('unwraps Browser Bridge {session,data} evaluate envelopes only at the boundary', () => {
|
||||
expect(unwrapEvaluateResult({ session: 's1', data: 'JSESSIONID=1; tk=2' })).toBe('JSESSIONID=1; tk=2');
|
||||
expect(unwrapEvaluateResult({ status: true, data: { value: 1 } })).toEqual({ status: true, data: { value: 1 } });
|
||||
expect(requireEvaluateObject({ session: 's1', data: { status: true } }, 'test')).toEqual({ status: true });
|
||||
expect(() => requireEvaluateObject({ session: 's1', data: null }, 'test')).toThrow(CommandExecutionError);
|
||||
});
|
||||
|
||||
it('classifies 12306 login-like API envelopes as auth failures', () => {
|
||||
expect(isAuthLikePayload({ status: false, messages: ['用户未登录'] })).toBe(true);
|
||||
expect(isAuthLikePayload({ status: false, validateMessages: { global: ['请登录后再试'] } })).toBe(true);
|
||||
expect(isAuthLikePayload({ status: false, messages: ['系统繁忙'] })).toBe(false);
|
||||
});
|
||||
|
||||
it('masks passenger names in orders by default and supports explicit sensitive opt-in', async () => {
|
||||
const command = getRegistry().get('12306/orders');
|
||||
const makePage = () => ({
|
||||
goto: async () => {},
|
||||
evaluate: async (script) => {
|
||||
if (script === "document.cookie || ''") return { session: 'browser', data: 'JSESSIONID=abc; tk=def' };
|
||||
return {
|
||||
session: 'browser',
|
||||
data: {
|
||||
status: true,
|
||||
data: {
|
||||
orderDBList: [{
|
||||
sequence_no: 'E123',
|
||||
order_date: '2026-05-18 10:00',
|
||||
train_code_page: 'G1',
|
||||
from_station_name_page: '北京南',
|
||||
to_station_name_page: '上海虹桥',
|
||||
start_train_date_page: '2026-05-22 07:00',
|
||||
ticket_status_name: '未出行',
|
||||
ticket_total_price_page: '626.0',
|
||||
tickets: [{ passenger_name: '张三' }, { passenger_name: '李四明' }],
|
||||
}],
|
||||
},
|
||||
},
|
||||
};
|
||||
},
|
||||
});
|
||||
|
||||
await expect(command.func(makePage(), {})).resolves.toMatchObject([
|
||||
{ order_id: 'E123', passengers: '张*, 李*明' },
|
||||
]);
|
||||
await expect(command.func(makePage(), { 'include-sensitive': true })).resolves.toMatchObject([
|
||||
{ order_id: 'E123', passengers: '张三, 李四明' },
|
||||
]);
|
||||
});
|
||||
|
||||
it('maps login-like order payloads to AuthRequiredError instead of parser drift', async () => {
|
||||
const command = getRegistry().get('12306/orders');
|
||||
const page = {
|
||||
goto: async () => {},
|
||||
evaluate: async (script) => {
|
||||
if (script === "document.cookie || ''") return 'JSESSIONID=abc; tk=def';
|
||||
return { status: false, messages: ['用户未登录'] };
|
||||
},
|
||||
};
|
||||
|
||||
await expect(command.func(page, {})).rejects.toBeInstanceOf(AuthRequiredError);
|
||||
});
|
||||
|
||||
it('treats missing order list shape as parser drift but known empty arrays as empty result', async () => {
|
||||
const command = getRegistry().get('12306/orders');
|
||||
const makePage = (payload) => ({
|
||||
goto: async () => {},
|
||||
evaluate: async (script) => {
|
||||
if (script === "document.cookie || ''") return 'JSESSIONID=abc; tk=def';
|
||||
return payload;
|
||||
},
|
||||
});
|
||||
|
||||
await expect(command.func(makePage({ status: true, data: {} }), {}))
|
||||
.rejects.toBeInstanceOf(CommandExecutionError);
|
||||
await expect(command.func(makePage({ status: true, data: { orderDBList: [] } }), {}))
|
||||
.rejects.toBeInstanceOf(EmptyResultError);
|
||||
});
|
||||
});
|
||||
@@ -183,7 +183,6 @@ export async function extractAssetsForInput(page, input) {
|
||||
cli({
|
||||
site: '1688',
|
||||
name: 'assets',
|
||||
access: 'read',
|
||||
description: '列出 1688 商品页可提取的图片/视频素材',
|
||||
domain: 'www.1688.com',
|
||||
strategy: Strategy.COOKIE,
|
||||
|
||||
@@ -42,7 +42,6 @@ function toDownloadItems(offerId, assets) {
|
||||
cli({
|
||||
site: '1688',
|
||||
name: 'download',
|
||||
access: 'read',
|
||||
description: '批量下载 1688 商品页可提取的图片和视频素材',
|
||||
domain: 'www.1688.com',
|
||||
strategy: Strategy.COOKIE,
|
||||
|
||||
@@ -156,7 +156,6 @@ async function readItemPayload(page, itemUrl) {
|
||||
cli({
|
||||
site: '1688',
|
||||
name: 'item',
|
||||
access: 'read',
|
||||
description: '1688 商品详情(公开商品字段、价格阶梯、卖家基础信息)',
|
||||
domain: 'www.1688.com',
|
||||
strategy: Strategy.COOKIE,
|
||||
|
||||
+1
-2
@@ -275,7 +275,6 @@ async function collectSearchRows(page, query, limit) {
|
||||
cli({
|
||||
site: '1688',
|
||||
name: 'search',
|
||||
access: 'read',
|
||||
description: '1688 商品搜索(结果候选、卖家链接、价格/MOQ/销量文本)',
|
||||
domain: 'www.1688.com',
|
||||
strategy: Strategy.COOKIE,
|
||||
@@ -294,7 +293,7 @@ cli({
|
||||
help: `结果数量上限(默认 ${SEARCH_LIMIT_DEFAULT},最大 ${SEARCH_LIMIT_MAX})`,
|
||||
},
|
||||
],
|
||||
columns: ['rank', 'offer_id', 'title', 'item_url', 'price_text', 'moq_text', 'seller_name', 'member_id', 'location'],
|
||||
columns: ['rank', 'title', 'price_text', 'moq_text', 'seller_name', 'location'],
|
||||
func: async (page, kwargs) => {
|
||||
const query = String(kwargs.query ?? '');
|
||||
const limit = parseSearchLimit(kwargs.limit);
|
||||
|
||||
@@ -167,7 +167,6 @@ function hasAnyEvidence(storePayload, contactPayload, seed) {
|
||||
cli({
|
||||
site: '1688',
|
||||
name: 'store',
|
||||
access: 'read',
|
||||
description: '1688 店铺/供应商公开信息(联系方式、主营、入驻年限、公开服务信号)',
|
||||
domain: 'www.1688.com',
|
||||
strategy: Strategy.COOKIE,
|
||||
|
||||
@@ -1,35 +0,0 @@
|
||||
/**
|
||||
* 一亩三分地 精华帖 — Discuz guide=digest view.
|
||||
*/
|
||||
import { cli, Strategy } from '@jackwener/opencli/registry';
|
||||
import { fetchHtml, parseThreadList, normalizeLimit, BASE } from './utils.js';
|
||||
|
||||
cli({
|
||||
site: '1point3acres',
|
||||
name: 'digest',
|
||||
access: 'read',
|
||||
description: '一亩三分地 精华帖(编辑推荐 / 加精)',
|
||||
domain: 'www.1point3acres.com',
|
||||
strategy: Strategy.PUBLIC,
|
||||
browser: false,
|
||||
args: [
|
||||
{ name: 'limit', type: 'int', default: 20, help: '返回条数(默认 20,最多 50)' },
|
||||
],
|
||||
columns: ['rank', 'tid', 'title', 'forum', 'author', 'replies', 'views', 'lastReplyTime', 'url'],
|
||||
func: async (args) => {
|
||||
const limit = normalizeLimit(args.limit, 20, 50);
|
||||
const html = await fetchHtml(`${BASE}/forum.php?mod=guide&view=digest`);
|
||||
const items = parseThreadList(html);
|
||||
return items.slice(0, limit).map((t, i) => ({
|
||||
rank: i + 1,
|
||||
tid: t.tid,
|
||||
title: t.title,
|
||||
forum: t.forum,
|
||||
author: t.author,
|
||||
replies: t.replies,
|
||||
views: t.views,
|
||||
lastReplyTime: t.lastReplyTime,
|
||||
url: t.url,
|
||||
}));
|
||||
},
|
||||
});
|
||||
@@ -1,51 +0,0 @@
|
||||
/**
|
||||
* 一亩三分地 版块帖子列表 — /bbs/forum-<fid>-<page>.html
|
||||
*/
|
||||
import { cli, Strategy } from '@jackwener/opencli/registry';
|
||||
import { ArgumentError } from '@jackwener/opencli/errors';
|
||||
import { fetchHtml, parseThreadList, parseThreadRows, normalizeLimit, BASE } from './utils.js';
|
||||
|
||||
cli({
|
||||
site: '1point3acres',
|
||||
name: 'forum',
|
||||
access: 'read',
|
||||
description: '浏览一亩三分地某个版块的帖子列表(按 fid)',
|
||||
domain: 'www.1point3acres.com',
|
||||
strategy: Strategy.PUBLIC,
|
||||
browser: false,
|
||||
args: [
|
||||
{ name: 'fid', required: true, positional: true, help: '版块 ID,例如 145(海外面经)、198(海外职位内推)、27(研究生申请)' },
|
||||
{ name: 'page', type: 'int', default: 1, help: '页码(默认 1)' },
|
||||
{ name: 'limit', type: 'int', default: 20, help: '返回条数(默认 20,最多 50)' },
|
||||
],
|
||||
columns: ['rank', 'tid', 'kind', 'title', 'author', 'replies', 'views', 'lastReplyTime', 'url'],
|
||||
func: async (args) => {
|
||||
const fid = String(args.fid || '').trim();
|
||||
if (!/^\d+$/.test(fid)) {
|
||||
throw new ArgumentError('fid must be a numeric forum id', 'e.g. 145 for 海外面经');
|
||||
}
|
||||
const pageNum = Number(args.page ?? 1);
|
||||
if (!Number.isInteger(pageNum) || pageNum <= 0) {
|
||||
throw new ArgumentError('page must be a positive integer');
|
||||
}
|
||||
const limit = normalizeLimit(args.limit, 20, 50);
|
||||
const html = await fetchHtml(`${BASE}/forum-${fid}-${pageNum}.html`);
|
||||
const rows = parseThreadRows(html);
|
||||
if (rows.length === 0) {
|
||||
// Forum may be sub-category-only — surface gracefully as empty with hint.
|
||||
return [];
|
||||
}
|
||||
const items = parseThreadList(html);
|
||||
return items.slice(0, limit).map((t, i) => ({
|
||||
rank: i + 1,
|
||||
tid: t.tid,
|
||||
kind: t.kind === 'stickthread' ? '置顶' : '普通',
|
||||
title: t.title,
|
||||
author: t.author,
|
||||
replies: t.replies,
|
||||
views: t.views,
|
||||
lastReplyTime: t.lastReplyTime,
|
||||
url: t.url,
|
||||
}));
|
||||
},
|
||||
});
|
||||
@@ -1,44 +0,0 @@
|
||||
/**
|
||||
* 一亩三分地 所有版块清单 — parsed from /bbs/forum.php
|
||||
*
|
||||
* Each forum card has:
|
||||
* <a href="forum-<fid>-1.html" ... class="... overflow-hidden whitespace-nowrap hidden desktop:block">版块名</a>
|
||||
* and an adjacent description element. We dedupe by fid and return name + url.
|
||||
*/
|
||||
import { cli, Strategy } from '@jackwener/opencli/registry';
|
||||
import { fetchHtml, decodeEntities, BASE } from './utils.js';
|
||||
|
||||
cli({
|
||||
site: '1point3acres',
|
||||
name: 'forums',
|
||||
access: 'read',
|
||||
description: '一亩三分地 所有版块(fid + 版块名)',
|
||||
domain: 'www.1point3acres.com',
|
||||
strategy: Strategy.PUBLIC,
|
||||
browser: false,
|
||||
args: [
|
||||
{ name: 'filter', type: 'string', default: '', help: '按版块名关键字过滤(子串匹配,中英文)' },
|
||||
],
|
||||
columns: ['fid', 'name', 'url'],
|
||||
func: async (args) => {
|
||||
const html = await fetchHtml(`${BASE}/forum.php`);
|
||||
const seen = new Map();
|
||||
const re = /<a href="forum-(\d+)-1\.html"[^>]*class="[^"]*overflow-hidden[^"]*"[^>]*>\s*([^<]+?)\s*<\/a>/g;
|
||||
let m;
|
||||
while ((m = re.exec(html))) {
|
||||
const fid = m[1];
|
||||
let name = decodeEntities(m[2].trim());
|
||||
// Some subforum labels are wrapped in brackets — unwrap for display parity.
|
||||
name = name.replace(/^\[(.+)\]$/, '$1').trim();
|
||||
if (!name || seen.has(fid)) continue;
|
||||
seen.set(fid, name);
|
||||
}
|
||||
const filter = String(args.filter || '').toLowerCase().trim();
|
||||
const out = [];
|
||||
for (const [fid, name] of seen) {
|
||||
if (filter && !name.toLowerCase().includes(filter)) continue;
|
||||
out.push({ fid, name, url: `${BASE}/forum-${fid}-1.html` });
|
||||
}
|
||||
return out;
|
||||
},
|
||||
});
|
||||
@@ -1,35 +0,0 @@
|
||||
/**
|
||||
* 一亩三分地 热门帖子 — Discuz guide=hot view.
|
||||
*/
|
||||
import { cli, Strategy } from '@jackwener/opencli/registry';
|
||||
import { fetchHtml, parseThreadList, normalizeLimit, BASE } from './utils.js';
|
||||
|
||||
cli({
|
||||
site: '1point3acres',
|
||||
name: 'hot',
|
||||
access: 'read',
|
||||
description: '一亩三分地 今日热门帖子(按热度排序,约 50 条)',
|
||||
domain: 'www.1point3acres.com',
|
||||
strategy: Strategy.PUBLIC,
|
||||
browser: false,
|
||||
args: [
|
||||
{ name: 'limit', type: 'int', default: 20, help: '返回条数(默认 20,最多 50)' },
|
||||
],
|
||||
columns: ['rank', 'tid', 'title', 'forum', 'author', 'replies', 'views', 'lastReplyTime', 'url'],
|
||||
func: async (args) => {
|
||||
const limit = normalizeLimit(args.limit, 20, 50);
|
||||
const html = await fetchHtml(`${BASE}/forum.php?mod=guide&view=hot`);
|
||||
const items = parseThreadList(html);
|
||||
return items.slice(0, limit).map((t, i) => ({
|
||||
rank: i + 1,
|
||||
tid: t.tid,
|
||||
title: t.title,
|
||||
forum: t.forum,
|
||||
author: t.author,
|
||||
replies: t.replies,
|
||||
views: t.views,
|
||||
lastReplyTime: t.lastReplyTime,
|
||||
url: t.url,
|
||||
}));
|
||||
},
|
||||
});
|
||||
@@ -1,35 +0,0 @@
|
||||
/**
|
||||
* 一亩三分地 最新帖子 — Discuz guide=new view.
|
||||
*/
|
||||
import { cli, Strategy } from '@jackwener/opencli/registry';
|
||||
import { fetchHtml, parseThreadList, normalizeLimit, BASE } from './utils.js';
|
||||
|
||||
cli({
|
||||
site: '1point3acres',
|
||||
name: 'latest',
|
||||
access: 'read',
|
||||
description: '一亩三分地 最新发帖(按发帖时间倒序)',
|
||||
domain: 'www.1point3acres.com',
|
||||
strategy: Strategy.PUBLIC,
|
||||
browser: false,
|
||||
args: [
|
||||
{ name: 'limit', type: 'int', default: 20, help: '返回条数(默认 20,最多 50)' },
|
||||
],
|
||||
columns: ['rank', 'tid', 'title', 'forum', 'author', 'replies', 'views', 'postTime', 'url'],
|
||||
func: async (args) => {
|
||||
const limit = normalizeLimit(args.limit, 20, 50);
|
||||
const html = await fetchHtml(`${BASE}/forum.php?mod=guide&view=new`);
|
||||
const items = parseThreadList(html);
|
||||
return items.slice(0, limit).map((t, i) => ({
|
||||
rank: i + 1,
|
||||
tid: t.tid,
|
||||
title: t.title,
|
||||
forum: t.forum,
|
||||
author: t.author,
|
||||
replies: t.replies,
|
||||
views: t.views,
|
||||
postTime: t.postTime,
|
||||
url: t.url,
|
||||
}));
|
||||
},
|
||||
});
|
||||
@@ -1,64 +0,0 @@
|
||||
/**
|
||||
* 一亩三分地 我的通知 — 坛友互动 / 点评 / @我 等
|
||||
*
|
||||
* /bbs/home.php?mod=space&do=notice&view=interactive needs login cookie.
|
||||
*/
|
||||
import { cli, Strategy } from '@jackwener/opencli/registry';
|
||||
import { AuthRequiredError, EmptyResultError } from '@jackwener/opencli/errors';
|
||||
import { fetchHtml, decodeEntities, getCookie, stripHtml, truncate, normalizePositiveInteger, BASE } from './utils.js';
|
||||
|
||||
cli({
|
||||
site: '1point3acres',
|
||||
name: 'notifications',
|
||||
access: 'read',
|
||||
description: '一亩三分地 站内通知(互动 / 点评 / @ 我;需要登录)',
|
||||
domain: 'www.1point3acres.com',
|
||||
strategy: Strategy.COOKIE,
|
||||
browser: true,
|
||||
navigateBefore: false,
|
||||
args: [
|
||||
{ name: 'kind', type: 'string', default: 'mypost',
|
||||
help: '通知类型:mypost(我的帖子) / interactive(互动) / system(系统) / app(应用)' },
|
||||
{ name: 'limit', type: 'int', default: 20, help: '返回条数' },
|
||||
],
|
||||
columns: ['index', 'from', 'summary', 'time', 'threadUrl'],
|
||||
func: async (page, args) => {
|
||||
const kind = String(args.kind || 'mypost').trim();
|
||||
const cookie = await getCookie(page);
|
||||
const url = `${BASE}/home.php?mod=space&do=notice&view=${encodeURIComponent(kind)}`;
|
||||
const html = await fetchHtml(url, { cookie, headers: { Referer: `${BASE}/` } });
|
||||
|
||||
if (/<title>提示信息/.test(html) && /请登录/.test(html)) {
|
||||
throw new AuthRequiredError('www.1point3acres.com', '请先登录一亩三分地');
|
||||
}
|
||||
|
||||
// "No notifications" is a real empty result, not a synthetic data row.
|
||||
if (/暂时没有提醒内容/.test(html)) {
|
||||
throw new EmptyResultError('1point3acres notifications', '暂时没有提醒内容');
|
||||
}
|
||||
|
||||
const rows = [];
|
||||
const limit = normalizePositiveInteger(args.limit, 20, 'limit');
|
||||
|
||||
// Pattern 1: standard Discuz <dl class="cl">…</dl> block per notice.
|
||||
const dlRe = /<dl class="[^"]*cl[^"]*"[^>]*>([\s\S]*?)<\/dl>/g;
|
||||
let m;
|
||||
let i = 0;
|
||||
while ((m = dlRe.exec(html)) && rows.length < limit) {
|
||||
const block = m[1];
|
||||
const from = decodeEntities((block.match(/<dt>([\s\S]*?)<\/dt>/) || [, ''])[1])
|
||||
.replace(/<[^>]+>/g, '').trim();
|
||||
const summaryRaw = (block.match(/<dd class="ntc_body">([\s\S]*?)<\/dd>/) ||
|
||||
block.match(/<dd>([\s\S]*?)<\/dd>/) || [, ''])[1];
|
||||
const summary = truncate(stripHtml(summaryRaw), 200);
|
||||
const time = ((block.match(/<dd class="[^"]*xg1[^"]*"[^>]*>([\s\S]*?)<\/dd>/) || [, ''])[1] || '')
|
||||
.replace(/<[^>]+>/g, '').trim();
|
||||
const linkMatch = summaryRaw.match(/href="([^"]*thread-\d+[^"]*)"/);
|
||||
const threadUrl = linkMatch ? (linkMatch[1].startsWith('http') ? linkMatch[1] : `${BASE}/${linkMatch[1]}`) : '';
|
||||
i += 1;
|
||||
if (!from && !summary) continue;
|
||||
rows.push({ index: i, from, summary, time, threadUrl });
|
||||
}
|
||||
return rows;
|
||||
},
|
||||
});
|
||||
@@ -1,71 +0,0 @@
|
||||
/**
|
||||
* 一亩三分地 站内搜索 — /bbs/search.php?mod=forum
|
||||
*
|
||||
* Guests get a "请登录" alert page, so this command needs the live browser
|
||||
* session's cookie. Discuz routes search through a 302 redirect to
|
||||
* search.php?searchid=<ID>. Node fetch follows redirects automatically as
|
||||
* long as we pass the session cookie along.
|
||||
*/
|
||||
import { cli, Strategy } from '@jackwener/opencli/registry';
|
||||
import { ArgumentError, EmptyResultError } from '@jackwener/opencli/errors';
|
||||
import { fetchHtml, parseSearchList, assertNotGuestAlert, getCookie, decodeEntities, normalizeLimit, BASE } from './utils.js';
|
||||
|
||||
cli({
|
||||
site: '1point3acres',
|
||||
name: 'search',
|
||||
access: 'read',
|
||||
description: '一亩三分地 站内关键字搜索(需要登录)',
|
||||
domain: 'www.1point3acres.com',
|
||||
strategy: Strategy.COOKIE,
|
||||
browser: true,
|
||||
navigateBefore: false,
|
||||
args: [
|
||||
{ name: 'query', required: true, positional: true, help: '搜索关键字' },
|
||||
{ name: 'limit', type: 'int', default: 20, help: '返回条数(默认 20,最多 50)' },
|
||||
{ name: 'fid', type: 'string', default: '', help: '限定版块 ID(可选)' },
|
||||
],
|
||||
columns: ['rank', 'tid', 'title', 'forum', 'author', 'replies', 'views', 'postTime', 'url'],
|
||||
func: async (page, args) => {
|
||||
const query = String(args.query || '').trim();
|
||||
if (!query) throw new ArgumentError('query 不能为空');
|
||||
const limit = normalizeLimit(args.limit, 20, 50);
|
||||
const fid = String(args.fid || '').trim();
|
||||
|
||||
const cookie = await getCookie(page);
|
||||
const qs = new URLSearchParams({
|
||||
mod: 'forum',
|
||||
srchtxt: query,
|
||||
searchsubmit: 'yes',
|
||||
...(fid ? { srchfid: fid } : {}),
|
||||
});
|
||||
const url = `${BASE}/search.php?${qs.toString()}`;
|
||||
|
||||
// Node fetch with the session cookie — Discuz's 302 to search.php?searchid=…
|
||||
// is followed by default.
|
||||
const html = await fetchHtml(url, {
|
||||
cookie,
|
||||
headers: { Referer: `${BASE}/` },
|
||||
});
|
||||
assertNotGuestAlert(html);
|
||||
|
||||
const items = parseSearchList(html);
|
||||
if (items.length === 0) {
|
||||
const hint = html.match(/<p>([^<]*?抱歉[^<]*?)<\/p>/);
|
||||
if (hint) {
|
||||
throw new EmptyResultError('1point3acres search', decodeEntities(hint[1].trim()));
|
||||
}
|
||||
throw new EmptyResultError('1point3acres search', `No results for "${query}"`);
|
||||
}
|
||||
return items.slice(0, limit).map((t, i) => ({
|
||||
rank: i + 1,
|
||||
tid: t.tid,
|
||||
title: t.title,
|
||||
forum: t.forum,
|
||||
author: t.author,
|
||||
replies: t.replies,
|
||||
views: t.views,
|
||||
postTime: t.postTime,
|
||||
url: t.url,
|
||||
}));
|
||||
},
|
||||
});
|
||||
@@ -1,117 +0,0 @@
|
||||
/**
|
||||
* 一亩三分地 帖子详情 — /bbs/thread-<tid>-<page>-1.html
|
||||
*
|
||||
* Returns one row per post on the requested page. First row (floor=1) is the
|
||||
* main post; the rest are replies. Columns are shaped so `--limit 1` gives
|
||||
* just the main post, and larger limits walk down the thread.
|
||||
*/
|
||||
import { cli, Strategy } from '@jackwener/opencli/registry';
|
||||
import { ArgumentError, EmptyResultError } from '@jackwener/opencli/errors';
|
||||
import { fetchHtml, decodeEntities, stripHtml, truncate, normalizePositiveInteger, BASE } from './utils.js';
|
||||
|
||||
function extract(html, regex, group = 1) {
|
||||
const m = html.match(regex);
|
||||
return m ? m[group] : '';
|
||||
}
|
||||
|
||||
cli({
|
||||
site: '1point3acres',
|
||||
name: 'thread',
|
||||
access: 'read',
|
||||
description: '一亩三分地 帖子详情 + 楼层(主楼 + 回复)',
|
||||
domain: 'www.1point3acres.com',
|
||||
strategy: Strategy.PUBLIC,
|
||||
browser: false,
|
||||
args: [
|
||||
{ name: 'tid', required: true, positional: true, help: '帖子 ID(数字,见 `hot`/`latest` 返回的 tid)' },
|
||||
{ name: 'page', type: 'int', default: 1, help: '楼层分页页码(默认 1)' },
|
||||
{ name: 'limit', type: 'int', default: 10, help: '返回楼层条数(默认 10,含主楼)' },
|
||||
{ name: 'contentLimit', type: 'int', default: 400, help: '每楼正文截断长度(默认 400 字符,最少 50)' },
|
||||
],
|
||||
columns: ['floor', 'pid', 'author', 'postTime', 'content', 'url'],
|
||||
func: async (args) => {
|
||||
const tid = String(args.tid || '').trim();
|
||||
if (!/^\d+$/.test(tid)) {
|
||||
throw new ArgumentError('tid must be a numeric thread id');
|
||||
}
|
||||
const page = normalizePositiveInteger(args.page, 1, 'page');
|
||||
const limit = normalizePositiveInteger(args.limit, 10, 'limit');
|
||||
const contentLimit = normalizePositiveInteger(args.contentLimit, 400, 'contentLimit', { min: 50 });
|
||||
|
||||
const url = `${BASE}/thread-${tid}-${page}-1.html`;
|
||||
const html = await fetchHtml(url);
|
||||
|
||||
// Sanity: real thread page will contain postlist + at least one post div.
|
||||
if (!/id="postlist"/.test(html) && !/id="post_\d+"/.test(html)) {
|
||||
throw new EmptyResultError('1point3acres thread', `帖子 ${tid} 不存在或被删除`);
|
||||
}
|
||||
|
||||
// Split posts: each post block is bounded by <div id="post_<PID>">…</div> next post or postlist end.
|
||||
// NOTE: intermediate objects intentionally use postId/body/offset (not pid/html/start) to
|
||||
// avoid being mistaken for row-shaped objects by the silent-column-drop audit.
|
||||
const postBlocks = [];
|
||||
const re = /<div id="post_(\d+)"[^>]*>/g;
|
||||
const offsets = [];
|
||||
let m;
|
||||
while ((m = re.exec(html))) offsets.push({ postId: m[1], offset: m.index });
|
||||
for (let i = 0; i < offsets.length; i++) {
|
||||
const segStart = offsets[i].offset;
|
||||
const segEnd = i + 1 < offsets.length ? offsets[i + 1].offset : html.length;
|
||||
postBlocks.push({ postId: offsets[i].postId, body: html.slice(segStart, segEnd) });
|
||||
}
|
||||
|
||||
const rows = [];
|
||||
for (let i = 0; i < postBlocks.length && rows.length < limit; i++) {
|
||||
const { postId: pid, body: block } = postBlocks[i];
|
||||
// Discuz authi block holds the author link + post time metadata.
|
||||
const authiMatch = block.match(/<div class="authi"[\s\S]*?<\/div>/);
|
||||
const authiBlock = authiMatch ? authiMatch[0] : '';
|
||||
const authorCandidates = [
|
||||
/<a [^>]*class="[^"]*\bxi2\b[^"]*"[^>]*>\s*([^<]+?)\s*<\/a>/,
|
||||
/<a [^>]*href="space-uid-\d+\.html"[^>]*>\s*([^<]+?)\s*<\/a>/,
|
||||
/<a [^>]*class="[^"]*\bxw1\b[^"]*"[^>]*>\s*([^<]+?)\s*<\/a>/,
|
||||
];
|
||||
let author = '';
|
||||
for (const re of authorCandidates) {
|
||||
const v = decodeEntities(extract(authiBlock || block, re));
|
||||
if (v && !/匿名卡|变色卡|关贴卡/.test(v)) { author = v; break; }
|
||||
}
|
||||
// Time: prefer <span title="YYYY-MM-DD HH:MM:SS"> (per-post, precise).
|
||||
// <meta itemprop="datePublished"> is the *thread* publish time on this site — avoid.
|
||||
const postTime = extract(authiBlock, /<span title="([^"]+)">/) ||
|
||||
extract(block, /id="authorposton\d+"[^>]*>\s*<span title="([^"]+)">/) ||
|
||||
extract(block, /id="authorposton\d+"[^>]*>\s*([^<]+?)\s*</) ||
|
||||
extract(block, /<meta itemprop="datePublished" content="([^"]+)"/);
|
||||
// Floor: first post on page 1 is the 楼主, subsequent posts carry <em>N#</em>.
|
||||
const floorEm = extract(block, /<em>(\d+)<\/em>\s*#?\s*<\/a>/) ||
|
||||
extract(block, /id="postnum\d+"[^>]*>\s*<em>(\d+)<\/em>/);
|
||||
const isMainPost = page === 1 && i === 0;
|
||||
const floor = floorEm ? Number(floorEm) : (isMainPost ? 1 : (page - 1) * 10 + i + 1);
|
||||
const contentMatch = block.match(/id="postmessage_\d+"[^>]*>([\s\S]*?)<\/td>/);
|
||||
const content = truncate(stripHtml(contentMatch ? contentMatch[1] : ''), contentLimit);
|
||||
rows.push({
|
||||
floor,
|
||||
pid,
|
||||
author,
|
||||
postTime: postTime.trim(),
|
||||
content,
|
||||
url: `${BASE}/forum.php?mod=redirect&goto=findpost&ptid=${tid}&pid=${pid}`,
|
||||
});
|
||||
}
|
||||
|
||||
// Attach the thread title + forum name as a leading synthetic row only when rows exist
|
||||
// and only for page 1, so agents get the title without needing a separate call.
|
||||
if (page === 1 && rows.length > 0) {
|
||||
const title = decodeEntities(
|
||||
extract(html, /<span id="thread_subject">([^<]+)<\/span>/).trim() ||
|
||||
extract(html, /<title>([^<]+?)\s*[-|]/).trim()
|
||||
);
|
||||
rows[0].content = title ? `【${title}】\n${rows[0].content}` : rows[0].content;
|
||||
}
|
||||
|
||||
if (!rows.length) {
|
||||
throw new EmptyResultError('1point3acres thread', `帖子 ${tid} 第 ${page} 页没有可读取楼层`);
|
||||
}
|
||||
return rows;
|
||||
},
|
||||
});
|
||||
@@ -1,77 +0,0 @@
|
||||
/**
|
||||
* 一亩三分地 用户资料 — /bbs/space-uid-<uid>.html or /bbs/space-username-<name>.html
|
||||
*
|
||||
* Guest-visible fields: username, uid, user group, register/last-access times,
|
||||
* post/thread/digest counts, credits, rice (大米 — site currency), profile URL.
|
||||
* Users can be queried by numeric uid or by username (both routes are public).
|
||||
*/
|
||||
import { cli, Strategy } from '@jackwener/opencli/registry';
|
||||
import { ArgumentError, EmptyResultError } from '@jackwener/opencli/errors';
|
||||
import { fetchHtml, decodeEntities, BASE } from './utils.js';
|
||||
|
||||
cli({
|
||||
site: '1point3acres',
|
||||
name: 'user',
|
||||
access: 'read',
|
||||
description: '一亩三分地 用户空间(用户组 / 积分 / 大米 / 帖子数 等)',
|
||||
domain: 'www.1point3acres.com',
|
||||
strategy: Strategy.PUBLIC,
|
||||
browser: false,
|
||||
args: [
|
||||
{ name: 'who', required: true, positional: true, help: '用户名或 uid(纯数字按 uid 查,否则按用户名)' },
|
||||
],
|
||||
columns: [
|
||||
'uid', 'username', 'group', 'credits', 'rice',
|
||||
'posts', 'threads', 'digests', 'registerTime', 'lastAccess', 'profileUrl',
|
||||
],
|
||||
func: async (args) => {
|
||||
const who = String(args.who || '').trim();
|
||||
if (!who) throw new ArgumentError('who 不能为空', '传用户名或数字 uid');
|
||||
const url = /^\d+$/.test(who)
|
||||
? `${BASE}/space-uid-${who}.html`
|
||||
: `${BASE}/space-username-${encodeURIComponent(who)}.html`;
|
||||
|
||||
const html = await fetchHtml(url);
|
||||
if (/<title>提示信息/.test(html) && /(没有找到|不存在)/.test(html)) {
|
||||
throw new EmptyResultError('1point3acres user', `用户 "${who}" 不存在`);
|
||||
}
|
||||
|
||||
const pick = (re) => {
|
||||
const m = html.match(re);
|
||||
return m ? decodeEntities(m[1].trim()) : '';
|
||||
};
|
||||
// <li>KEY: VAL</li> — tolerant of optional <span>, colons fullwidth/半角, 颗/根/粒 suffixes.
|
||||
const pickLi = (label) => {
|
||||
const re = new RegExp(`<li>\\s*${label}[::\\s]*(?:<[^>]+>)?\\s*([^<]+?)\\s*(?:<|$)`);
|
||||
const m = html.match(re);
|
||||
return m ? decodeEntities(m[1].trim()) : '';
|
||||
};
|
||||
|
||||
const username =
|
||||
pick(/<p class="mtm[^"]*"[^>]*>\s*<a [^>]*>([^<]+?)<\/a>/) ||
|
||||
pick(/<title>([^<]+?)的个人资料/);
|
||||
const uid = pick(/uid=(\d+)/) || pick(/space-uid-(\d+)\.html/);
|
||||
const group = pickLi('用户组');
|
||||
const credits = pickLi('积分');
|
||||
const rice = pickLi('大米');
|
||||
const posts = pickLi('帖子数');
|
||||
const threads = pickLi('主题数');
|
||||
const digests = pickLi('精华数');
|
||||
const registerTime = pickLi('注册时间');
|
||||
const lastAccess = pickLi('最后访问');
|
||||
|
||||
return [{
|
||||
uid,
|
||||
username,
|
||||
group,
|
||||
credits,
|
||||
rice,
|
||||
posts,
|
||||
threads,
|
||||
digests,
|
||||
registerTime,
|
||||
lastAccess,
|
||||
profileUrl: uid ? `${BASE}/space-uid-${uid}.html` : url,
|
||||
}];
|
||||
},
|
||||
});
|
||||
@@ -1,247 +0,0 @@
|
||||
/**
|
||||
* Shared helpers for 一亩三分地 (1point3acres.com) adapters.
|
||||
*
|
||||
* Site is a Discuz!X PHP BBS that serves GBK-encoded HTML.
|
||||
* - Thread listings: /bbs/forum.php?mod=guide&view={hot|new|digest|newthread}
|
||||
* - Forum: /bbs/forum-<fid>-<page>.html
|
||||
* - Thread detail: /bbs/thread-<tid>-<page>-1.html
|
||||
* - User profile: /bbs/space-uid-<uid>.html or /bbs/space-username-<name>.html
|
||||
* - Search: /bbs/search.php?mod=forum (COOKIE — guests get an alert page)
|
||||
*/
|
||||
import { AuthRequiredError, ArgumentError, CommandExecutionError } from '@jackwener/opencli/errors';
|
||||
|
||||
export const BASE = 'https://www.1point3acres.com/bbs';
|
||||
|
||||
/**
|
||||
* Validate `limit` per typed-fail-fast convention (no silent clamp).
|
||||
* Throws ArgumentError on non-positive / non-integer / out-of-range input.
|
||||
*/
|
||||
export function normalizeLimit(value, defaultValue, maxValue, label = 'limit') {
|
||||
const limit = normalizePositiveInteger(value, defaultValue, label);
|
||||
if (limit > maxValue) {
|
||||
throw new ArgumentError(`${label} must be <= ${maxValue}`);
|
||||
}
|
||||
return limit;
|
||||
}
|
||||
|
||||
/** Validate a positive integer argument without silently flooring/clamping. */
|
||||
export function normalizePositiveInteger(value, defaultValue, label = 'value', { min = 1 } = {}) {
|
||||
const raw = value ?? defaultValue;
|
||||
const limit = Number(raw);
|
||||
if (!Number.isInteger(limit) || limit <= 0) {
|
||||
throw new ArgumentError(`${label} must be a positive integer`);
|
||||
}
|
||||
if (limit < min) {
|
||||
throw new ArgumentError(`${label} must be >= ${min}`);
|
||||
}
|
||||
return limit;
|
||||
}
|
||||
|
||||
const UA = 'Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/146.0 Safari/537.36';
|
||||
|
||||
/** Fetch a GBK-encoded Discuz page and return decoded UTF-8 HTML. */
|
||||
export async function fetchHtml(url, { headers = {}, cookie = '' } = {}) {
|
||||
let res;
|
||||
try {
|
||||
res = await fetch(url, {
|
||||
headers: {
|
||||
'User-Agent': UA,
|
||||
'Accept': 'text/html,application/xhtml+xml',
|
||||
'Accept-Language': 'zh-CN,zh;q=0.9,en;q=0.8',
|
||||
...(cookie ? { Cookie: cookie } : {}),
|
||||
...headers,
|
||||
},
|
||||
redirect: 'follow',
|
||||
});
|
||||
} catch (error) {
|
||||
throw new CommandExecutionError(`1point3acres request failed: ${error?.message || error}`);
|
||||
}
|
||||
if (!res.ok) {
|
||||
throw new CommandExecutionError(`1point3acres request failed: HTTP ${res.status} ${res.statusText} from ${url}`);
|
||||
}
|
||||
const buf = await res.arrayBuffer();
|
||||
return new TextDecoder('gbk').decode(buf);
|
||||
}
|
||||
|
||||
/** Pull cookie string from the live browser session for this domain.
|
||||
* Discuz auth cookies (4Oaf_61d6_*, session) are HttpOnly and set on the
|
||||
* root domain `.1point3acres.com`, so we need `getCookies` (not document.cookie)
|
||||
* AND we need to query both host + root domain and merge.
|
||||
*/
|
||||
export async function getCookie(page) {
|
||||
if (!page) return '';
|
||||
const seen = new Map();
|
||||
if (typeof page.getCookies === 'function') {
|
||||
for (const opts of [{ domain: 'www.1point3acres.com' }, { domain: '.1point3acres.com' }]) {
|
||||
try {
|
||||
const cookies = await page.getCookies(opts);
|
||||
for (const c of cookies || []) {
|
||||
if (!seen.has(c.name)) seen.set(c.name, c.value);
|
||||
}
|
||||
} catch { /* try next */ }
|
||||
}
|
||||
}
|
||||
if (seen.size > 0) {
|
||||
return [...seen].map(([k, v]) => `${k}=${v}`).join('; ');
|
||||
}
|
||||
try {
|
||||
const result = await page.evaluate('document.cookie');
|
||||
return typeof result === 'string' ? result : '';
|
||||
} catch {
|
||||
return '';
|
||||
}
|
||||
}
|
||||
|
||||
/** Detect the "you are a guest" alert page that Discuz returns for protected actions. */
|
||||
export function assertNotGuestAlert(html, domain = 'www.1point3acres.com') {
|
||||
if (/<title>提示信息 \| 一亩三分地<\/title>/.test(html) && /无法进行此操作/.test(html)) {
|
||||
throw new AuthRequiredError(domain, '需要登录一亩三分地后再使用该命令');
|
||||
}
|
||||
}
|
||||
|
||||
const ENTITY_MAP = {
|
||||
' ': ' ', '&': '&', '<': '<', '>': '>',
|
||||
'"': '"', ''': "'", ''': "'",
|
||||
};
|
||||
|
||||
/** Decode HTML entities (numeric + common named). */
|
||||
export function decodeEntities(s) {
|
||||
if (!s) return '';
|
||||
return s
|
||||
.replace(/&#(\d+);/g, (_, n) => String.fromCodePoint(Number(n)))
|
||||
.replace(/&#[xX]([0-9a-fA-F]+);/g, (_, n) => String.fromCodePoint(parseInt(n, 16)))
|
||||
.replace(/&(nbsp|amp|lt|gt|quot|#39|apos);/g, m => ENTITY_MAP[m] || m);
|
||||
}
|
||||
|
||||
/** Strip HTML tags and collapse whitespace, returning plain text. */
|
||||
export function stripHtml(html) {
|
||||
if (!html) return '';
|
||||
return decodeEntities(
|
||||
String(html)
|
||||
.replace(/<br\s*\/?>/gi, '\n')
|
||||
.replace(/<\/(p|div|li|tr)>/gi, '\n')
|
||||
.replace(/<[^>]+>/g, '')
|
||||
).replace(/[ \t]+\n/g, '\n').replace(/\n{3,}/g, '\n\n').trim();
|
||||
}
|
||||
|
||||
/** Truncate text to n characters with ellipsis. */
|
||||
export function truncate(s, n = 300) {
|
||||
if (!s) return '';
|
||||
return s.length > n ? s.slice(0, n) + '…' : s;
|
||||
}
|
||||
|
||||
/** Extract all <tbody id="normalthread_*"> blocks from a forum/guide page. */
|
||||
export function parseThreadRows(html) {
|
||||
const rows = [];
|
||||
const re = /<tbody id="(normalthread|stickthread)_(\d+)"[^>]*>([\s\S]*?)<\/tbody>/g;
|
||||
let m;
|
||||
while ((m = re.exec(html))) {
|
||||
const [, kind, tid, inner] = m;
|
||||
rows.push({ kind, tid, inner });
|
||||
}
|
||||
return rows;
|
||||
}
|
||||
|
||||
/** Parse a single Discuz thread row (inner HTML of the tbody). */
|
||||
export function parseThreadRow({ kind, tid, inner }) {
|
||||
const titleMatches = [...inner.matchAll(/<a [^>]*class="[^"]*\bxst\b[^"]*"[^>]*>([^<]+)<\/a>/g)];
|
||||
const title = titleMatches.length
|
||||
? decodeEntities(titleMatches[titleMatches.length - 1][1].trim())
|
||||
: '';
|
||||
|
||||
const forumMatch = inner.match(/<a href="forum-(\d+)-1\.html"[^>]*target="_blank"[^>]*>([^<]+)<\/a>/);
|
||||
const fid = forumMatch ? forumMatch[1] : '';
|
||||
const forumName = forumMatch ? decodeEntities(forumMatch[2].trim()) : '';
|
||||
|
||||
// <td class="by"> blocks; first with <cite> = author, last with <cite> = last reply
|
||||
const byBlocks = [...inner.matchAll(/<td class="by"[^>]*>([\s\S]*?)<\/td>/g)].map(m => m[1]);
|
||||
const readCite = (block) => {
|
||||
const m = block.match(/<cite[^>]*>([\s\S]*?)<\/cite>/);
|
||||
if (!m) return '';
|
||||
return decodeEntities(m[1].replace(/<[^>]+>/g, '').trim());
|
||||
};
|
||||
const readTime = (block) => {
|
||||
const titleM = block.match(/<span [^>]*title="([^"]+)"[^>]*>/);
|
||||
if (titleM) return titleM[1].trim();
|
||||
const plainA = block.match(/<em>[\s\S]*?<a [^>]*>\s*([^<]+?)\s*<\/a>/);
|
||||
if (plainA) return decodeEntities(plainA[1].trim());
|
||||
const plainSpan = block.match(/<em>[\s\S]*?<span[^>]*>\s*([^<]+?)\s*<\/span>/);
|
||||
if (plainSpan) return decodeEntities(plainSpan[1].trim());
|
||||
const bare = block.match(/<em>\s*([^<]+?)\s*<\/em>/);
|
||||
return bare ? decodeEntities(bare[1].trim()) : '';
|
||||
};
|
||||
let authorBlock = '';
|
||||
let lastBlock = '';
|
||||
for (const b of byBlocks) {
|
||||
if (/<cite/.test(b)) {
|
||||
if (!authorBlock) authorBlock = b;
|
||||
lastBlock = b;
|
||||
}
|
||||
}
|
||||
const author = authorBlock ? readCite(authorBlock) : '';
|
||||
const postTime = authorBlock ? readTime(authorBlock) : '';
|
||||
const lastReplyUser = lastBlock && lastBlock !== authorBlock ? readCite(lastBlock) : '';
|
||||
const lastReplyTime = lastBlock && lastBlock !== authorBlock ? readTime(lastBlock) : '';
|
||||
|
||||
const numMatch = inner.match(/<td class="num"[^>]*>\s*<a[^>]*class="xi2"[^>]*>(\d+)<\/a>(?:\s*<em>(\d+)<\/em>)?/);
|
||||
const replies = numMatch ? Number(numMatch[1]) : 0;
|
||||
const views = numMatch && numMatch[2] ? Number(numMatch[2]) : 0;
|
||||
return {
|
||||
tid,
|
||||
kind,
|
||||
title,
|
||||
author,
|
||||
forum: forumName,
|
||||
fid,
|
||||
replies,
|
||||
views,
|
||||
postTime,
|
||||
lastReplyUser,
|
||||
lastReplyTime,
|
||||
url: `${BASE}/thread-${tid}-1-1.html`,
|
||||
};
|
||||
}
|
||||
|
||||
/** Quick one-shot listing parser used by hot/latest/digest/forum. */
|
||||
export function parseThreadList(html) {
|
||||
return parseThreadRows(html).map(parseThreadRow).filter(t => t.title);
|
||||
}
|
||||
|
||||
/**
|
||||
* Parse Discuz search results page (different HTML shape than forum listings).
|
||||
* Each hit is <li class="pbw" id="TID"> containing h3 > a[href*="tid=TID"],
|
||||
* <p class="xg1">N 个回复 - M 次查看</p>, and a time/author/forum <p>.
|
||||
*/
|
||||
export function parseSearchList(html) {
|
||||
const items = [];
|
||||
const re = /<li class="pbw" id="(\d+)">([\s\S]*?)<\/li>/g;
|
||||
let m;
|
||||
while ((m = re.exec(html))) {
|
||||
const [, tid, inner] = m;
|
||||
const titleMatch = inner.match(/<h3[^>]*>\s*<a [^>]*>([\s\S]*?)<\/a>/);
|
||||
const titleRaw = titleMatch ? titleMatch[1] : '';
|
||||
const title = decodeEntities(titleRaw.replace(/<[^>]+>/g, '')).trim();
|
||||
if (!title) continue;
|
||||
|
||||
const statsMatch = inner.match(/<p class="xg1">\s*([\d,]+)\s*个回复\s*-\s*([\d,]+)\s*次查看\s*<\/p>/);
|
||||
const replies = statsMatch ? Number(statsMatch[1].replace(/,/g, '')) : 0;
|
||||
const views = statsMatch ? Number(statsMatch[2].replace(/,/g, '')) : 0;
|
||||
|
||||
const metaMatch = inner.match(/<p>\s*<span>([^<]+)<\/span>[\s\S]*?<a [^>]*space-uid-\d+[^>]*>([^<]+?)<\/a>[\s\S]*?<a [^>]*href="forum-(\d+)-[^"]*"[^>]*>([^<]+?)<\/a>/);
|
||||
const postTime = metaMatch ? decodeEntities(metaMatch[1].trim()) : '';
|
||||
const author = metaMatch ? decodeEntities(metaMatch[2].trim()) : '';
|
||||
const fid = metaMatch ? metaMatch[3] : '';
|
||||
const forumName = metaMatch ? decodeEntities(metaMatch[4].trim()) : '';
|
||||
|
||||
items.push({
|
||||
tid, title, author, forum: forumName, fid,
|
||||
replies, views, postTime,
|
||||
// Search pages don't show lastReplyTime separately — surface postTime instead.
|
||||
lastReplyUser: '', lastReplyTime: postTime,
|
||||
url: `${BASE}/thread-${tid}-1-1.html`,
|
||||
});
|
||||
}
|
||||
return items;
|
||||
}
|
||||
|
||||
export { UA };
|
||||
@@ -13,7 +13,6 @@ function parseArticleId(input) {
|
||||
cli({
|
||||
site: '36kr',
|
||||
name: 'article',
|
||||
access: 'read',
|
||||
description: '获取36氪文章正文内容',
|
||||
domain: 'www.36kr.com',
|
||||
strategy: Strategy.INTERCEPT,
|
||||
@@ -52,15 +51,12 @@ cli({
|
||||
if (!data?.title) {
|
||||
throw new CliError('NOT_FOUND', 'Article not found or failed to load', 'Check the article ID');
|
||||
}
|
||||
if (!data.body) {
|
||||
throw new CliError('PARSE_ERROR', 'Article body not found', '36kr page loaded but no article body paragraphs were extracted');
|
||||
}
|
||||
return [
|
||||
{ field: 'title', value: data.title },
|
||||
{ field: 'author', value: data.author || '' },
|
||||
{ field: 'date', value: data.date || '' },
|
||||
{ field: 'author', value: data.author || '-' },
|
||||
{ field: 'date', value: data.date || '-' },
|
||||
{ field: 'url', value: `https://36kr.com/p/${articleId}` },
|
||||
{ field: 'body', value: data.body || '' },
|
||||
{ field: 'body', value: data.body || '-' },
|
||||
];
|
||||
},
|
||||
});
|
||||
|
||||
@@ -1,46 +0,0 @@
|
||||
import { describe, expect, it, vi } from 'vitest';
|
||||
import { getRegistry } from '@jackwener/opencli/registry';
|
||||
import { CliError } from '@jackwener/opencli/errors';
|
||||
import './article.js';
|
||||
|
||||
function makePage(evaluateResult) {
|
||||
return {
|
||||
installInterceptor: vi.fn().mockResolvedValue(undefined),
|
||||
goto: vi.fn().mockResolvedValue(undefined),
|
||||
wait: vi.fn().mockResolvedValue(undefined),
|
||||
evaluate: vi.fn().mockResolvedValue(evaluateResult),
|
||||
};
|
||||
}
|
||||
|
||||
describe('36kr article', () => {
|
||||
it('emits empty-string for missing optional author / date instead of a sentinel', async () => {
|
||||
const command = getRegistry().get('36kr/article');
|
||||
expect(command?.func).toBeDefined();
|
||||
const page = makePage({ title: 'Real Title', author: '', date: '', body: 'Real article body' });
|
||||
const rows = await command.func(page, { id: '1234567' });
|
||||
const byField = Object.fromEntries(rows.map((r) => [r.field, r.value]));
|
||||
expect(byField.title).toBe('Real Title');
|
||||
expect(byField.author).toBe('');
|
||||
expect(byField.date).toBe('');
|
||||
expect(byField.body).toBe('Real article body');
|
||||
expect(byField.url).toBe('https://36kr.com/p/1234567');
|
||||
});
|
||||
|
||||
it('throws CliError NOT_FOUND when the page exposes no title', async () => {
|
||||
const command = getRegistry().get('36kr/article');
|
||||
const page = makePage({ title: '', author: 'x', date: 'y', body: 'z' });
|
||||
await expect(command.func(page, { id: '1234567' })).rejects.toBeInstanceOf(CliError);
|
||||
});
|
||||
|
||||
it('throws CliError PARSE_ERROR when the page exposes title but no body', async () => {
|
||||
const command = getRegistry().get('36kr/article');
|
||||
const page = makePage({ title: 'Real Title', author: 'x', date: 'y', body: '' });
|
||||
await expect(command.func(page, { id: '1234567' })).rejects.toMatchObject({ code: 'PARSE_ERROR' });
|
||||
});
|
||||
|
||||
it('throws CliError INVALID_ARGUMENT when no numeric id can be parsed', async () => {
|
||||
const command = getRegistry().get('36kr/article');
|
||||
const page = makePage({});
|
||||
await expect(command.func(page, { id: 'not-a-url' })).rejects.toBeInstanceOf(CliError);
|
||||
});
|
||||
});
|
||||
@@ -26,7 +26,6 @@ function buildHotListUrl(listType, date = new Date()) {
|
||||
cli({
|
||||
site: '36kr',
|
||||
name: 'hot',
|
||||
access: 'read',
|
||||
description: '36氪热榜 — trending articles (renqi/zonghe/shoucang/catalog)',
|
||||
domain: 'www.36kr.com',
|
||||
strategy: Strategy.PUBLIC,
|
||||
|
||||
@@ -5,7 +5,6 @@ import { cli, Strategy } from '@jackwener/opencli/registry';
|
||||
cli({
|
||||
site: '36kr',
|
||||
name: 'news',
|
||||
access: 'read',
|
||||
description: 'Latest tech/startup news from 36kr (36氪)',
|
||||
domain: 'www.36kr.com',
|
||||
strategy: Strategy.PUBLIC,
|
||||
|
||||
@@ -8,7 +8,6 @@ import { CliError } from '@jackwener/opencli/errors';
|
||||
cli({
|
||||
site: '36kr',
|
||||
name: 'search',
|
||||
access: 'read',
|
||||
description: '搜索36氪文章',
|
||||
domain: 'www.36kr.com',
|
||||
strategy: Strategy.PUBLIC,
|
||||
|
||||
@@ -13,7 +13,6 @@ import { JOBS_ORIGIN, requirePage, navigateTo, parseCompanyJobCard } from './uti
|
||||
cli({
|
||||
site: '51job',
|
||||
name: 'company',
|
||||
access: 'read',
|
||||
description: '51job 公司简介 + 在招职位(按 encCoId)',
|
||||
domain: 'jobs.51job.com',
|
||||
strategy: Strategy.COOKIE,
|
||||
|
||||
@@ -13,7 +13,6 @@ import { JOBS_ORIGIN, requirePage, navigateTo } from './utils.js';
|
||||
cli({
|
||||
site: '51job',
|
||||
name: 'detail',
|
||||
access: 'read',
|
||||
description: '51job 职位详情(按 jobId)',
|
||||
domain: 'jobs.51job.com',
|
||||
strategy: Strategy.COOKIE,
|
||||
|
||||
@@ -16,7 +16,6 @@ import {
|
||||
cli({
|
||||
site: '51job',
|
||||
name: 'hot',
|
||||
access: 'read',
|
||||
description: '51job 推荐职位(按城市/行业/排序浏览)',
|
||||
domain: 'we.51job.com',
|
||||
strategy: Strategy.COOKIE,
|
||||
|
||||
@@ -20,7 +20,6 @@ import {
|
||||
cli({
|
||||
site: '51job',
|
||||
name: 'search',
|
||||
access: 'read',
|
||||
description: '51job 前程无忧关键词职位搜索',
|
||||
domain: 'we.51job.com',
|
||||
strategy: Strategy.COOKIE,
|
||||
|
||||
@@ -14,7 +14,6 @@ export function makeScreenshotCommand(site, displayName, extra = {}) {
|
||||
...extra,
|
||||
site,
|
||||
name: 'screenshot',
|
||||
access: 'read',
|
||||
description: `Capture a snapshot of the current ${label} window (DOM + Accessibility tree)`,
|
||||
domain: 'localhost',
|
||||
strategy: Strategy.UI,
|
||||
@@ -47,7 +46,6 @@ export function makeStatusCommand(site, displayName, extra = {}) {
|
||||
...extra,
|
||||
site,
|
||||
name: 'status',
|
||||
access: 'read',
|
||||
description: `Check active CDP connection to ${label}`,
|
||||
domain: 'localhost',
|
||||
strategy: Strategy.UI,
|
||||
@@ -69,7 +67,6 @@ export function makeNewCommand(site, displayName, extra = {}) {
|
||||
...extra,
|
||||
site,
|
||||
name: 'new',
|
||||
access: 'write',
|
||||
description: `Start a new ${label} session`,
|
||||
domain: 'localhost',
|
||||
strategy: Strategy.UI,
|
||||
@@ -90,7 +87,6 @@ export function makeDumpCommand(site) {
|
||||
return cli({
|
||||
site,
|
||||
name: 'dump',
|
||||
access: 'read',
|
||||
description: `Dump the DOM and Accessibility tree of ${site} for reverse-engineering`,
|
||||
domain: 'localhost',
|
||||
strategy: Strategy.UI,
|
||||
|
||||
@@ -1,70 +0,0 @@
|
||||
import { ArgumentError, CommandExecutionError, EmptyResultError } from '@jackwener/opencli/errors';
|
||||
|
||||
export function requireSearchQuery(value, label = 'keyword') {
|
||||
const query = String(value ?? '').trim();
|
||||
if (!query) {
|
||||
throw new ArgumentError(`${label} cannot be empty`);
|
||||
}
|
||||
return query;
|
||||
}
|
||||
|
||||
export function requireBoundedInteger(value, defaultValue, min, max, label) {
|
||||
const raw = value ?? defaultValue;
|
||||
const parsed = typeof raw === 'number' ? raw : Number(raw);
|
||||
if (!Number.isInteger(parsed)) {
|
||||
throw new ArgumentError(`${label} must be an integer between ${min} and ${max}, got ${JSON.stringify(value)}`);
|
||||
}
|
||||
if (parsed < min || parsed > max) {
|
||||
throw new ArgumentError(`${label} must be between ${min} and ${max}, got ${parsed}`);
|
||||
}
|
||||
return parsed;
|
||||
}
|
||||
|
||||
export function requireNonNegativeInteger(value, defaultValue, label) {
|
||||
const raw = value ?? defaultValue;
|
||||
const parsed = typeof raw === 'number' ? raw : Number(raw);
|
||||
if (!Number.isInteger(parsed) || parsed < 0) {
|
||||
throw new ArgumentError(`${label} must be a non-negative integer, got ${JSON.stringify(value)}`);
|
||||
}
|
||||
return parsed;
|
||||
}
|
||||
|
||||
export function unwrapBrowserResult(value) {
|
||||
if (value && typeof value === 'object' && !Array.isArray(value) && 'session' in value && 'data' in value) {
|
||||
return value.data;
|
||||
}
|
||||
return value;
|
||||
}
|
||||
|
||||
export function requireRows(value, label) {
|
||||
const rows = unwrapBrowserResult(value);
|
||||
if (!Array.isArray(rows)) {
|
||||
throw new CommandExecutionError(`${label} returned an unexpected payload shape; expected an array of result rows.`);
|
||||
}
|
||||
return rows;
|
||||
}
|
||||
|
||||
export function toHttpsUrl(value, baseUrl) {
|
||||
const raw = String(value ?? '').trim();
|
||||
if (!raw) return '';
|
||||
try {
|
||||
const url = new URL(raw, baseUrl);
|
||||
if (url.protocol !== 'http:' && url.protocol !== 'https:') return '';
|
||||
return url.href;
|
||||
} catch {
|
||||
return '';
|
||||
}
|
||||
}
|
||||
|
||||
export function emptySearchResults(site, query) {
|
||||
return new EmptyResultError(`${site} search`, `No ${site} results matched "${query}".`);
|
||||
}
|
||||
|
||||
export async function runBrowserStep(label, fn) {
|
||||
try {
|
||||
return await fn();
|
||||
} catch (error) {
|
||||
if (error?.code || error?.name === 'ArgumentError') throw error;
|
||||
throw new CommandExecutionError(`${label} failed: ${error?.message ?? error}`);
|
||||
}
|
||||
}
|
||||
@@ -1,110 +0,0 @@
|
||||
import { cli, Strategy } from '@jackwener/opencli/registry';
|
||||
import { ArgumentError, CommandExecutionError, EmptyResultError, getErrorMessage } from '@jackwener/opencli/errors';
|
||||
|
||||
const AIBASE_DAILY_URL = 'https://www.aibase.com/zh/daily';
|
||||
const DEFAULT_LIMIT = 20;
|
||||
const MAX_LIMIT = 50;
|
||||
|
||||
function normalizeLimit(value) {
|
||||
const raw = value ?? DEFAULT_LIMIT;
|
||||
const limit = Number(raw);
|
||||
if (!Number.isInteger(limit) || limit <= 0) {
|
||||
throw new ArgumentError('limit must be a positive integer', `Example: opencli aibase news --limit ${DEFAULT_LIMIT}`);
|
||||
}
|
||||
if (limit > MAX_LIMIT) {
|
||||
throw new ArgumentError(`limit must be <= ${MAX_LIMIT}`, `Example: opencli aibase news --limit ${MAX_LIMIT}`);
|
||||
}
|
||||
return limit;
|
||||
}
|
||||
|
||||
function normalizeText(value) {
|
||||
return String(value ?? '').replace(/\s+/g, ' ').trim();
|
||||
}
|
||||
|
||||
function buildExtractAibaseNewsJs() {
|
||||
return `
|
||||
(() => {
|
||||
const anchors = Array.from(document.querySelectorAll('.bg-white .grid a[href], a[href*="/zh/daily/"]'))
|
||||
.filter((anchor) => {
|
||||
const href = anchor.getAttribute('href') || '';
|
||||
const text = (anchor.innerText || anchor.textContent || '').trim();
|
||||
return text && href && !href.endsWith('/zh/daily') && !href.endsWith('/zh/daily/');
|
||||
});
|
||||
if (anchors.length === 0) {
|
||||
return {
|
||||
ok: false,
|
||||
reason: 'selector-missing',
|
||||
title: document.title || '',
|
||||
bodyText: (document.body?.innerText || document.body?.textContent || '').slice(0, 500),
|
||||
};
|
||||
}
|
||||
const seen = new Set();
|
||||
const rows = [];
|
||||
for (const anchor of anchors) {
|
||||
const url = new URL(anchor.getAttribute('href'), location.href).href;
|
||||
if (seen.has(url)) continue;
|
||||
seen.add(url);
|
||||
rows.push({
|
||||
rank: rows.length + 1,
|
||||
title: anchor.innerText || anchor.textContent || '',
|
||||
url,
|
||||
});
|
||||
}
|
||||
return { ok: true, rows };
|
||||
})()
|
||||
`;
|
||||
}
|
||||
|
||||
function toRows(payload, limit) {
|
||||
if (!payload || typeof payload !== 'object') {
|
||||
throw new CommandExecutionError('AIbase daily page returned an unreadable payload');
|
||||
}
|
||||
if (!payload.ok) {
|
||||
const reason = typeof payload.reason === 'string' && payload.reason.trim() ? payload.reason.trim() : 'selector-drift';
|
||||
throw new CommandExecutionError(
|
||||
`AIbase daily selector drift: ${reason}`,
|
||||
payload.title ? `Page title: ${payload.title}` : undefined,
|
||||
);
|
||||
}
|
||||
const rows = (Array.isArray(payload.rows) ? payload.rows : [])
|
||||
.map((row, index) => ({
|
||||
rank: index + 1,
|
||||
title: normalizeText(row.title),
|
||||
url: normalizeText(row.url),
|
||||
}))
|
||||
.filter((row) => row.title && row.url);
|
||||
if (rows.length === 0) {
|
||||
throw new EmptyResultError('aibase news', 'AIbase daily page loaded, but no article rows with title and URL were extracted.');
|
||||
}
|
||||
return rows.slice(0, limit).map((row, index) => ({ ...row, rank: index + 1 }));
|
||||
}
|
||||
|
||||
async function loadAibaseNews(page, args) {
|
||||
const limit = normalizeLimit(args.limit);
|
||||
await page.goto(AIBASE_DAILY_URL, { waitUntil: 'load', settleMs: 3000 });
|
||||
const payload = await page.evaluate(buildExtractAibaseNewsJs()).catch((error) => {
|
||||
throw new CommandExecutionError(`Failed to extract AIbase daily news: ${getErrorMessage(error)}`);
|
||||
});
|
||||
return toRows(payload, limit);
|
||||
}
|
||||
|
||||
export const aibaseNewsCommand = cli({
|
||||
site: 'aibase',
|
||||
name: 'news',
|
||||
access: 'read',
|
||||
description: 'AIbase 日报 - 每天三分钟关注AI行业趋势',
|
||||
domain: 'www.aibase.com',
|
||||
strategy: Strategy.PUBLIC,
|
||||
browser: true,
|
||||
args: [
|
||||
{ name: 'limit', type: 'int', default: DEFAULT_LIMIT, help: `Number of news items to return (max ${MAX_LIMIT})` },
|
||||
],
|
||||
columns: ['rank', 'title', 'url'],
|
||||
func: loadAibaseNews,
|
||||
});
|
||||
|
||||
export const __test__ = {
|
||||
buildExtractAibaseNewsJs,
|
||||
normalizeLimit,
|
||||
toRows,
|
||||
};
|
||||
@@ -1,59 +0,0 @@
|
||||
import { JSDOM } from 'jsdom';
|
||||
import { describe, expect, it, vi } from 'vitest';
|
||||
import { ArgumentError, CommandExecutionError, EmptyResultError } from '@jackwener/opencli/errors';
|
||||
import { aibaseNewsCommand, __test__ } from './news.js';
|
||||
|
||||
function runBrowserScript(html, script, url = 'https://www.aibase.com/zh/daily') {
|
||||
const dom = new JSDOM(html, { url, runScripts: 'outside-only' });
|
||||
return dom.window.eval(script);
|
||||
}
|
||||
|
||||
function makePage(evaluateResult) {
|
||||
return {
|
||||
goto: vi.fn().mockResolvedValue(undefined),
|
||||
evaluate: vi.fn().mockResolvedValue(evaluateResult),
|
||||
};
|
||||
}
|
||||
|
||||
describe('aibase/news', () => {
|
||||
it('registers stable URL in columns', () => {
|
||||
expect(aibaseNewsCommand.access).toBe('read');
|
||||
expect(aibaseNewsCommand.columns).toEqual(['rank', 'title', 'url']);
|
||||
});
|
||||
|
||||
it('validates limit before browser navigation', async () => {
|
||||
const page = makePage({ ok: true, rows: [] });
|
||||
await expect(aibaseNewsCommand.func(page, { limit: 0 })).rejects.toBeInstanceOf(ArgumentError);
|
||||
await expect(aibaseNewsCommand.func(page, { limit: 51 })).rejects.toBeInstanceOf(ArgumentError);
|
||||
expect(page.goto).not.toHaveBeenCalled();
|
||||
});
|
||||
|
||||
it('extracts and deduplicates AIbase daily rows', async () => {
|
||||
const html = `
|
||||
<div class="bg-white">
|
||||
<div class="grid">
|
||||
<a href="/zh/daily/123"> First AI daily item </a>
|
||||
<a href="/zh/daily/123"> First AI daily item duplicate </a>
|
||||
<a href="/zh/daily/456"> Second AI daily item </a>
|
||||
</div>
|
||||
</div>
|
||||
`;
|
||||
const payload = runBrowserScript(html, __test__.buildExtractAibaseNewsJs());
|
||||
const page = makePage(payload);
|
||||
|
||||
const rows = await aibaseNewsCommand.func(page, { limit: 2 });
|
||||
|
||||
expect(page.goto).toHaveBeenCalledWith('https://www.aibase.com/zh/daily', { waitUntil: 'load', settleMs: 3000 });
|
||||
expect(rows).toEqual([
|
||||
{ rank: 1, title: 'First AI daily item', url: 'https://www.aibase.com/zh/daily/123' },
|
||||
{ rank: 2, title: 'Second AI daily item', url: 'https://www.aibase.com/zh/daily/456' },
|
||||
]);
|
||||
});
|
||||
|
||||
it('maps selector drift and empty rows to typed errors', async () => {
|
||||
await expect(aibaseNewsCommand.func(makePage({ ok: false, reason: 'selector-missing' }), { limit: 1 }))
|
||||
.rejects.toBeInstanceOf(CommandExecutionError);
|
||||
await expect(aibaseNewsCommand.func(makePage({ ok: true, rows: [{ title: '', url: '' }] }), { limit: 1 }))
|
||||
.rejects.toBeInstanceOf(EmptyResultError);
|
||||
});
|
||||
});
|
||||
@@ -2,7 +2,6 @@ import { cli } from '@jackwener/opencli/registry';
|
||||
import { createRankingCliOptions } from './rankings.js';
|
||||
cli(createRankingCliOptions({
|
||||
commandName: 'bestsellers',
|
||||
access: 'read',
|
||||
listType: 'bestsellers',
|
||||
description: 'Amazon Best Sellers pages for category candidate discovery',
|
||||
}));
|
||||
|
||||
@@ -85,7 +85,6 @@ async function readDiscussionPayload(page, input, limit) {
|
||||
cli({
|
||||
site: 'amazon',
|
||||
name: 'discussion',
|
||||
access: 'read',
|
||||
description: 'Amazon review summary and sample customer discussion from product review pages',
|
||||
domain: 'amazon.com',
|
||||
strategy: Strategy.COOKIE,
|
||||
|
||||
@@ -3,8 +3,35 @@ import { AuthRequiredError } from '@jackwener/opencli/errors';
|
||||
import { getRegistry } from '@jackwener/opencli/registry';
|
||||
import { __test__ } from './discussion.js';
|
||||
import './discussion.js';
|
||||
import { createPageMock } from '../test-utils.js';
|
||||
|
||||
function createPageMock(evaluateResults) {
|
||||
const evaluate = vi.fn();
|
||||
for (const result of evaluateResults) {
|
||||
evaluate.mockResolvedValueOnce(result);
|
||||
}
|
||||
return {
|
||||
goto: vi.fn().mockResolvedValue(undefined),
|
||||
wait: vi.fn().mockResolvedValue(undefined),
|
||||
evaluate,
|
||||
snapshot: vi.fn().mockResolvedValue(undefined),
|
||||
click: vi.fn().mockResolvedValue(undefined),
|
||||
typeText: vi.fn().mockResolvedValue(undefined),
|
||||
pressKey: vi.fn().mockResolvedValue(undefined),
|
||||
scrollTo: vi.fn().mockResolvedValue(undefined),
|
||||
getFormState: vi.fn().mockResolvedValue({ forms: [], orphanFields: [] }),
|
||||
tabs: vi.fn().mockResolvedValue([]),
|
||||
selectTab: vi.fn().mockResolvedValue(undefined),
|
||||
networkRequests: vi.fn().mockResolvedValue([]),
|
||||
consoleMessages: vi.fn().mockResolvedValue([]),
|
||||
scroll: vi.fn().mockResolvedValue(undefined),
|
||||
autoScroll: vi.fn().mockResolvedValue(undefined),
|
||||
installInterceptor: vi.fn().mockResolvedValue(undefined),
|
||||
getInterceptedRequests: vi.fn().mockResolvedValue([]),
|
||||
getCookies: vi.fn().mockResolvedValue([]),
|
||||
screenshot: vi.fn().mockResolvedValue(''),
|
||||
waitForCapture: vi.fn().mockResolvedValue(undefined),
|
||||
};
|
||||
}
|
||||
|
||||
describe('amazon discussion normalization', () => {
|
||||
it('normalizes review summary and sample reviews', () => {
|
||||
|
||||
@@ -2,7 +2,6 @@ import { cli } from '@jackwener/opencli/registry';
|
||||
import { createRankingCliOptions } from './rankings.js';
|
||||
cli(createRankingCliOptions({
|
||||
commandName: 'movers-shakers',
|
||||
access: 'read',
|
||||
listType: 'movers_shakers',
|
||||
description: 'Amazon Movers & Shakers pages for short-term growth signals',
|
||||
}));
|
||||
|
||||
@@ -2,7 +2,6 @@ import { cli } from '@jackwener/opencli/registry';
|
||||
import { createRankingCliOptions } from './rankings.js';
|
||||
cli(createRankingCliOptions({
|
||||
commandName: 'new-releases',
|
||||
access: 'read',
|
||||
listType: 'new_releases',
|
||||
description: 'Amazon New Releases pages for early momentum discovery',
|
||||
}));
|
||||
|
||||
@@ -106,7 +106,6 @@ async function readOfferPayload(page, input) {
|
||||
cli({
|
||||
site: 'amazon',
|
||||
name: 'offer',
|
||||
access: 'read',
|
||||
description: 'Amazon seller, buy box, and fulfillment facts from the product page',
|
||||
domain: 'amazon.com',
|
||||
strategy: Strategy.COOKIE,
|
||||
|
||||
@@ -65,7 +65,6 @@ async function readProductPayload(page, input) {
|
||||
cli({
|
||||
site: 'amazon',
|
||||
name: 'product',
|
||||
access: 'read',
|
||||
description: 'Amazon product page facts for candidate validation',
|
||||
domain: 'amazon.com',
|
||||
strategy: Strategy.COOKIE,
|
||||
|
||||
@@ -142,7 +142,6 @@ export function createRankingCliOptions(definition) {
|
||||
return {
|
||||
site: 'amazon',
|
||||
name: definition.commandName,
|
||||
access: definition.access ?? 'read',
|
||||
description: definition.description,
|
||||
domain: 'amazon.com',
|
||||
strategy: Strategy.COOKIE,
|
||||
|
||||
@@ -49,7 +49,6 @@ async function readSearchPayload(page, query) {
|
||||
cli({
|
||||
site: 'amazon',
|
||||
name: 'search',
|
||||
access: 'read',
|
||||
description: 'Amazon search results for product discovery and coarse filtering',
|
||||
domain: 'amazon.com',
|
||||
strategy: Strategy.COOKIE,
|
||||
|
||||
@@ -3,7 +3,6 @@ import * as fs from 'node:fs';
|
||||
export const dumpCommand = cli({
|
||||
site: 'antigravity',
|
||||
name: 'dump',
|
||||
access: 'read',
|
||||
description: 'Dump the DOM to help AI understand the UI',
|
||||
domain: 'localhost',
|
||||
strategy: Strategy.UI,
|
||||
|
||||
@@ -2,7 +2,6 @@ import { cli, Strategy } from '@jackwener/opencli/registry';
|
||||
export const extractCodeCommand = cli({
|
||||
site: 'antigravity',
|
||||
name: 'extract-code',
|
||||
access: 'read',
|
||||
description: 'Extract multi-line code blocks from the current Antigravity conversation',
|
||||
domain: 'localhost',
|
||||
strategy: Strategy.UI,
|
||||
|
||||
@@ -2,7 +2,6 @@ import { cli, Strategy } from '@jackwener/opencli/registry';
|
||||
export const modelCommand = cli({
|
||||
site: 'antigravity',
|
||||
name: 'model',
|
||||
access: 'read',
|
||||
description: 'Switch the active LLM model in Antigravity',
|
||||
domain: 'localhost',
|
||||
strategy: Strategy.UI,
|
||||
|
||||
@@ -2,7 +2,6 @@ import { cli, Strategy } from '@jackwener/opencli/registry';
|
||||
export const newCommand = cli({
|
||||
site: 'antigravity',
|
||||
name: 'new',
|
||||
access: 'read',
|
||||
description: 'Start a new conversation / clear context in Antigravity',
|
||||
domain: 'localhost',
|
||||
strategy: Strategy.UI,
|
||||
|
||||
@@ -2,7 +2,6 @@ import { cli, Strategy } from '@jackwener/opencli/registry';
|
||||
export const readCommand = cli({
|
||||
site: 'antigravity',
|
||||
name: 'read',
|
||||
access: 'read',
|
||||
description: 'Read the latest chat messages from Antigravity AI',
|
||||
domain: 'localhost',
|
||||
strategy: Strategy.UI,
|
||||
|
||||
@@ -2,7 +2,6 @@ import { cli, Strategy } from '@jackwener/opencli/registry';
|
||||
export const sendCommand = cli({
|
||||
site: 'antigravity',
|
||||
name: 'send',
|
||||
access: 'write',
|
||||
description: 'Send a message to Antigravity AI via the internal Lexical editor',
|
||||
domain: 'localhost',
|
||||
strategy: Strategy.UI,
|
||||
|
||||
@@ -2,7 +2,6 @@ import { cli, Strategy } from '@jackwener/opencli/registry';
|
||||
export const statusCommand = cli({
|
||||
site: 'antigravity',
|
||||
name: 'status',
|
||||
access: 'read',
|
||||
description: 'Check Antigravity CDP connection and get current page state',
|
||||
domain: 'localhost',
|
||||
strategy: Strategy.UI,
|
||||
|
||||
@@ -2,14 +2,12 @@ import { cli, Strategy } from '@jackwener/opencli/registry';
|
||||
export const watchCommand = cli({
|
||||
site: 'antigravity',
|
||||
name: 'watch',
|
||||
access: 'read',
|
||||
description: 'Stream new chat messages from Antigravity in real-time',
|
||||
domain: 'localhost',
|
||||
strategy: Strategy.UI,
|
||||
browser: true,
|
||||
args: [
|
||||
{ name: 'timeout', type: 'int', required: false, default: 86400, help: 'Max seconds to keep watching (default: 86400 — 24h)' },
|
||||
],
|
||||
args: [],
|
||||
timeoutSeconds: 86400, // Run for up to 24 hours
|
||||
columns: [], // We use direct stdout streaming
|
||||
func: async (page) => {
|
||||
console.log('Watching Antigravity chat... (Press Ctrl+C to stop)');
|
||||
|
||||
@@ -41,26 +41,6 @@ describe('apple-podcasts search command', () => {
|
||||
}),
|
||||
]);
|
||||
});
|
||||
it('emits empty-string for missing trackCount and primaryGenreName instead of a sentinel', async () => {
|
||||
const cmd = getRegistry().get('apple-podcasts/search');
|
||||
const fetchMock = vi.fn().mockResolvedValue({
|
||||
ok: true,
|
||||
json: () => Promise.resolve({
|
||||
results: [
|
||||
{
|
||||
collectionId: 99,
|
||||
collectionName: 'No-Meta Show',
|
||||
artistName: 'Anon Host',
|
||||
collectionViewUrl: 'https://example.com/p/99',
|
||||
},
|
||||
],
|
||||
}),
|
||||
});
|
||||
vi.stubGlobal('fetch', fetchMock);
|
||||
const result = await cmd.func({ query: 'no-meta', limit: 1 });
|
||||
expect(result[0].episodes).toBe('');
|
||||
expect(result[0].genre).toBe('');
|
||||
});
|
||||
});
|
||||
describe('apple-podcasts top command', () => {
|
||||
beforeEach(() => {
|
||||
|
||||
@@ -4,7 +4,6 @@ import { itunesFetch, formatDuration, formatDate } from './utils.js';
|
||||
cli({
|
||||
site: 'apple-podcasts',
|
||||
name: 'episodes',
|
||||
access: 'read',
|
||||
description: 'List recent episodes of an Apple Podcast (use ID from search)',
|
||||
strategy: Strategy.PUBLIC,
|
||||
browser: false,
|
||||
|
||||
@@ -4,7 +4,6 @@ import { itunesFetch } from './utils.js';
|
||||
cli({
|
||||
site: 'apple-podcasts',
|
||||
name: 'search',
|
||||
access: 'read',
|
||||
description: 'Search Apple Podcasts',
|
||||
strategy: Strategy.PUBLIC,
|
||||
browser: false,
|
||||
@@ -23,8 +22,8 @@ cli({
|
||||
id: p.collectionId,
|
||||
title: p.collectionName,
|
||||
author: p.artistName,
|
||||
episodes: p.trackCount ?? '',
|
||||
genre: p.primaryGenreName ?? '',
|
||||
episodes: p.trackCount ?? '-',
|
||||
genre: p.primaryGenreName ?? '-',
|
||||
url: p.collectionViewUrl || '',
|
||||
}));
|
||||
},
|
||||
|
||||
@@ -6,7 +6,6 @@ const CHARTS_TIMEOUT_MS = 15_000;
|
||||
cli({
|
||||
site: 'apple-podcasts',
|
||||
name: 'top',
|
||||
access: 'read',
|
||||
description: 'Top podcasts chart on Apple Podcasts',
|
||||
strategy: Strategy.PUBLIC,
|
||||
browser: false,
|
||||
|
||||
@@ -1,44 +0,0 @@
|
||||
// arxiv author — list papers authored by a person, newest first.
|
||||
//
|
||||
// arXiv's public API supports `au:` prefix queries. Author names on arXiv are
|
||||
// not stable IDs, so this is a best-effort fuzzy match — the same person can
|
||||
// appear under multiple spellings ("Y. Bengio" vs "Yoshua Bengio").
|
||||
import { cli, Strategy } from '@jackwener/opencli/registry';
|
||||
import { ArgumentError, EmptyResultError } from '@jackwener/opencli/errors';
|
||||
import { arxivFetch, normalizeArxivLimit, parseEntries } from './utils.js';
|
||||
|
||||
cli({
|
||||
site: 'arxiv',
|
||||
name: 'author',
|
||||
access: 'read',
|
||||
description: 'List arXiv papers by a given author (newest first)',
|
||||
strategy: Strategy.PUBLIC,
|
||||
browser: false,
|
||||
args: [
|
||||
{ name: 'author', positional: true, required: true, help: 'Author name (e.g. "Yoshua Bengio" or "Y Bengio")' },
|
||||
{ name: 'limit', type: 'int', default: 20, help: 'Max papers to return (max 50)' },
|
||||
],
|
||||
columns: ['id', 'title', 'authors', 'published', 'primary_category', 'url'],
|
||||
func: async (args) => {
|
||||
const authorText = String(args.author || '').trim();
|
||||
if (!authorText) {
|
||||
throw new ArgumentError('arxiv author cannot be empty', 'Example: opencli arxiv author "Yoshua Bengio"');
|
||||
}
|
||||
const limit = normalizeArxivLimit(args.limit, 20, 50);
|
||||
// Quote the value so multi-word author names match as a phrase.
|
||||
const query = encodeURIComponent(`au:"${authorText}"`);
|
||||
const xml = await arxivFetch(`search_query=${query}&max_results=${limit}&sortBy=submittedDate&sortOrder=descending`);
|
||||
const entries = parseEntries(xml);
|
||||
if (!entries.length) {
|
||||
throw new EmptyResultError('arxiv author', `No papers found for author "${authorText}". Try alternate spellings (e.g. initials).`);
|
||||
}
|
||||
return entries.map(e => ({
|
||||
id: e.id,
|
||||
title: e.title,
|
||||
authors: e.authors,
|
||||
published: e.published,
|
||||
primary_category: e.primary_category,
|
||||
url: e.url,
|
||||
}));
|
||||
},
|
||||
});
|
||||
@@ -4,7 +4,6 @@ import { arxivFetch, parseEntries } from './utils.js';
|
||||
cli({
|
||||
site: 'arxiv',
|
||||
name: 'paper',
|
||||
access: 'read',
|
||||
description: 'Get arXiv paper details by ID',
|
||||
strategy: Strategy.PUBLIC,
|
||||
browser: false,
|
||||
|
||||
@@ -4,7 +4,6 @@ import { arxivFetch, normalizeArxivCategory, normalizeArxivLimit, parseEntries }
|
||||
cli({
|
||||
site: 'arxiv',
|
||||
name: 'recent',
|
||||
access: 'read',
|
||||
description: 'List recent arXiv submissions in a category',
|
||||
strategy: Strategy.PUBLIC,
|
||||
browser: false,
|
||||
|
||||
@@ -4,7 +4,6 @@ import { arxivFetch, normalizeArxivLimit, parseEntries } from './utils.js';
|
||||
cli({
|
||||
site: 'arxiv',
|
||||
name: 'search',
|
||||
access: 'read',
|
||||
description: 'Search arXiv papers',
|
||||
strategy: Strategy.PUBLIC,
|
||||
browser: false,
|
||||
|
||||
@@ -4,7 +4,6 @@ import { clampInt, requireNonEmptyQuery } from '../_shared/common.js';
|
||||
cli({
|
||||
site: 'baidu-scholar',
|
||||
name: 'search',
|
||||
access: 'read',
|
||||
description: '百度学术搜索',
|
||||
domain: 'xueshu.baidu.com',
|
||||
strategy: Strategy.PUBLIC,
|
||||
@@ -14,6 +13,7 @@ cli({
|
||||
{ name: 'limit', type: 'int', default: 10, help: '返回结果数量 (max 20)' },
|
||||
],
|
||||
columns: ['rank', 'title', 'authors', 'journal', 'year', 'cited', 'url'],
|
||||
navigateBefore: false,
|
||||
func: async (page, kwargs) => {
|
||||
const limit = clampInt(kwargs.limit, 10, 1, 20);
|
||||
const query = requireNonEmptyQuery(kwargs.query);
|
||||
|
||||
@@ -13,7 +13,6 @@ import { cli, Strategy } from '@jackwener/opencli/registry';
|
||||
cli({
|
||||
site: 'band',
|
||||
name: 'bands',
|
||||
access: 'read',
|
||||
description: 'List all Bands you belong to',
|
||||
domain: 'www.band.us',
|
||||
strategy: Strategy.COOKIE,
|
||||
|
||||
@@ -12,7 +12,6 @@ import { cli, Strategy } from '@jackwener/opencli/registry';
|
||||
cli({
|
||||
site: 'band',
|
||||
name: 'mentions',
|
||||
access: 'read',
|
||||
description: 'Show Band notifications where you are @mentioned',
|
||||
domain: 'www.band.us',
|
||||
strategy: Strategy.INTERCEPT,
|
||||
|
||||
@@ -18,7 +18,6 @@ import { cli, Strategy } from '@jackwener/opencli/registry';
|
||||
cli({
|
||||
site: 'band',
|
||||
name: 'post',
|
||||
access: 'read',
|
||||
description: 'Export full content of a post including comments',
|
||||
domain: 'www.band.us',
|
||||
strategy: Strategy.COOKIE,
|
||||
|
||||
@@ -10,7 +10,6 @@ import { cli, Strategy } from '@jackwener/opencli/registry';
|
||||
cli({
|
||||
site: 'band',
|
||||
name: 'posts',
|
||||
access: 'read',
|
||||
description: 'List posts from a Band',
|
||||
domain: 'www.band.us',
|
||||
strategy: Strategy.COOKIE,
|
||||
|
||||
@@ -7,7 +7,6 @@ import { cli, Strategy } from '@jackwener/opencli/registry';
|
||||
cli({
|
||||
site: 'barchart',
|
||||
name: 'flow',
|
||||
access: 'read',
|
||||
description: 'Barchart unusual options activity / options flow',
|
||||
domain: 'www.barchart.com',
|
||||
strategy: Strategy.COOKIE,
|
||||
|
||||
+56
-145
@@ -4,70 +4,28 @@
|
||||
* Auth: CSRF token from <meta name="csrf-token"> + session cookies.
|
||||
*/
|
||||
import { cli, Strategy } from '@jackwener/opencli/registry';
|
||||
import { ArgumentError, CommandExecutionError, EmptyResultError } from '@jackwener/opencli/errors';
|
||||
|
||||
const DEFAULT_LIMIT = 10;
|
||||
const MIN_LIMIT = 1;
|
||||
const MAX_LIMIT = 100;
|
||||
|
||||
function normalizeSymbol(value) {
|
||||
const symbol = String(value ?? '').trim().toUpperCase();
|
||||
if (!symbol) throw new ArgumentError('symbol is required');
|
||||
return symbol;
|
||||
}
|
||||
|
||||
function normalizeExpiration(value) {
|
||||
const expiration = String(value ?? '').trim();
|
||||
if (!expiration) return '';
|
||||
if (!/^\d{4}-\d{2}-\d{2}$/.test(expiration)) {
|
||||
throw new ArgumentError('--expiration must use YYYY-MM-DD format');
|
||||
}
|
||||
const parsed = new Date(`${expiration}T00:00:00Z`);
|
||||
if (Number.isNaN(parsed.getTime()) || parsed.toISOString().slice(0, 10) !== expiration) {
|
||||
throw new ArgumentError('--expiration must be a valid calendar date');
|
||||
}
|
||||
return expiration;
|
||||
}
|
||||
|
||||
function parseLimit(value) {
|
||||
if (value === undefined || value === null || value === '') return DEFAULT_LIMIT;
|
||||
const limit = Number(value);
|
||||
if (!Number.isInteger(limit) || limit < MIN_LIMIT || limit > MAX_LIMIT) {
|
||||
throw new ArgumentError(`--limit must be an integer between ${MIN_LIMIT} and ${MAX_LIMIT}`);
|
||||
}
|
||||
return limit;
|
||||
}
|
||||
|
||||
function unwrapBrowserResult(value) {
|
||||
if (value && typeof value === 'object' && 'session' in value && 'data' in value) {
|
||||
return value.data;
|
||||
}
|
||||
return value;
|
||||
}
|
||||
|
||||
cli({
|
||||
site: 'barchart',
|
||||
name: 'greeks',
|
||||
access: 'read',
|
||||
description: 'Barchart options greeks overview (IV, delta, gamma, theta, vega)',
|
||||
domain: 'www.barchart.com',
|
||||
strategy: Strategy.COOKIE,
|
||||
args: [
|
||||
{ name: 'symbol', required: true, positional: true, help: 'Stock ticker (e.g. AAPL)' },
|
||||
{ name: 'expiration', type: 'str', help: 'Expiration date (YYYY-MM-DD). Defaults to the nearest available expiration.' },
|
||||
{ name: 'limit', type: 'int', default: DEFAULT_LIMIT, help: 'Number of near-the-money strikes per type (1-100)' },
|
||||
{ name: 'limit', type: 'int', default: 10, help: 'Number of near-the-money strikes per type' },
|
||||
],
|
||||
columns: [
|
||||
'type', 'strike', 'last', 'iv', 'delta', 'gamma', 'theta', 'vega', 'rho',
|
||||
'volume', 'openInterest', 'expiration',
|
||||
],
|
||||
func: async (page, kwargs) => {
|
||||
const symbol = normalizeSymbol(kwargs.symbol);
|
||||
const expiration = normalizeExpiration(kwargs.expiration);
|
||||
const limit = parseLimit(kwargs.limit);
|
||||
const symbol = kwargs.symbol.toUpperCase().trim();
|
||||
const expiration = kwargs.expiration ?? '';
|
||||
const limit = kwargs.limit ?? 10;
|
||||
await page.goto(`https://www.barchart.com/stocks/quotes/${encodeURIComponent(symbol)}/options`);
|
||||
await page.wait(4);
|
||||
const data = unwrapBrowserResult(await page.evaluate(`
|
||||
const data = await page.evaluate(`
|
||||
(async () => {
|
||||
const sym = ${JSON.stringify(symbol)};
|
||||
const expDate = ${JSON.stringify(expiration)};
|
||||
@@ -86,53 +44,39 @@ cli({
|
||||
+ '&fields=' + fields + '&raw=1';
|
||||
if (expDate) url += '&expirationDate=' + encodeURIComponent(expDate);
|
||||
const resp = await fetch(url, { credentials: 'include', headers });
|
||||
if (!resp.ok) {
|
||||
return { ok: false, reason: 'http', status: resp.status, statusText: resp.statusText || '' };
|
||||
}
|
||||
if (resp.ok) {
|
||||
const d = await resp.json();
|
||||
let items = d?.data || [];
|
||||
|
||||
const d = await resp.json();
|
||||
const allItems = d?.data;
|
||||
if (!Array.isArray(allItems)) {
|
||||
return { ok: false, reason: 'malformed' };
|
||||
}
|
||||
let items = allItems;
|
||||
|
||||
if (!expDate) {
|
||||
const expirations = items
|
||||
.map(i => (i.raw || i).expirationDate || null)
|
||||
.filter(Boolean)
|
||||
.sort((a, b) => {
|
||||
const aTime = Date.parse(a);
|
||||
const bTime = Date.parse(b);
|
||||
if (Number.isNaN(aTime) && Number.isNaN(bTime)) return 0;
|
||||
if (Number.isNaN(aTime)) return 1;
|
||||
if (Number.isNaN(bTime)) return -1;
|
||||
return aTime - bTime;
|
||||
});
|
||||
const nearestExpiration = expirations[0];
|
||||
if (nearestExpiration) {
|
||||
items = items.filter(i => ((i.raw || i).expirationDate || null) === nearestExpiration);
|
||||
if (!expDate) {
|
||||
const expirations = items
|
||||
.map(i => (i.raw || i).expirationDate || null)
|
||||
.filter(Boolean)
|
||||
.sort((a, b) => {
|
||||
const aTime = Date.parse(a);
|
||||
const bTime = Date.parse(b);
|
||||
if (Number.isNaN(aTime) && Number.isNaN(bTime)) return 0;
|
||||
if (Number.isNaN(aTime)) return 1;
|
||||
if (Number.isNaN(bTime)) return -1;
|
||||
return aTime - bTime;
|
||||
});
|
||||
const nearestExpiration = expirations[0];
|
||||
if (nearestExpiration) {
|
||||
items = items.filter(i => ((i.raw || i).expirationDate || null) === nearestExpiration);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// Separate calls and puts, sort by distance from current price.
|
||||
const calls = items
|
||||
.filter(i => ((i.raw || i).optionType || '').toLowerCase() === 'call')
|
||||
.sort((a, b) => Math.abs((a.raw || a).percentFromLast || 999) - Math.abs((b.raw || b).percentFromLast || 999))
|
||||
.slice(0, limit);
|
||||
const puts = items
|
||||
.filter(i => ((i.raw || i).optionType || '').toLowerCase() === 'put')
|
||||
.sort((a, b) => Math.abs((a.raw || a).percentFromLast || 999) - Math.abs((b.raw || b).percentFromLast || 999))
|
||||
.slice(0, limit);
|
||||
const selected = [...calls, ...puts];
|
||||
// Separate calls and puts, sort by distance from current price
|
||||
const calls = items
|
||||
.filter(i => ((i.raw || i).optionType || '').toLowerCase() === 'call')
|
||||
.sort((a, b) => Math.abs((a.raw || a).percentFromLast || 999) - Math.abs((b.raw || b).percentFromLast || 999))
|
||||
.slice(0, limit);
|
||||
const puts = items
|
||||
.filter(i => ((i.raw || i).optionType || '').toLowerCase() === 'put')
|
||||
.sort((a, b) => Math.abs((a.raw || a).percentFromLast || 999) - Math.abs((b.raw || b).percentFromLast || 999))
|
||||
.slice(0, limit);
|
||||
|
||||
if (items.length > 0 && selected.length === 0) {
|
||||
return { ok: false, reason: 'malformed', message: 'options rows did not include call or put identities' };
|
||||
}
|
||||
|
||||
return {
|
||||
ok: true,
|
||||
rows: selected.map(i => {
|
||||
return [...calls, ...puts].map(i => {
|
||||
const r = i.raw || i;
|
||||
return {
|
||||
type: r.optionType,
|
||||
@@ -148,61 +92,28 @@ cli({
|
||||
openInterest: r.openInterest,
|
||||
expiration: r.expirationDate,
|
||||
};
|
||||
})
|
||||
};
|
||||
} catch(e) {
|
||||
return { ok: false, reason: 'exception', message: e?.message || String(e) };
|
||||
}
|
||||
});
|
||||
}
|
||||
} catch(e) {}
|
||||
|
||||
return [];
|
||||
})()
|
||||
`));
|
||||
if (!data || data.ok !== true) {
|
||||
if (data?.reason === 'http') {
|
||||
throw new CommandExecutionError(`Barchart greeks request failed: HTTP ${data.status}${data.statusText ? ` ${data.statusText}` : ''}`);
|
||||
}
|
||||
if (data?.reason === 'malformed') {
|
||||
throw new CommandExecutionError(`Barchart greeks returned an unreadable options payload${data.message ? `: ${data.message}` : ''}`);
|
||||
}
|
||||
if (data?.reason === 'exception') {
|
||||
throw new CommandExecutionError(`Barchart greeks request failed: ${data.message || 'unknown error'}`);
|
||||
}
|
||||
throw new CommandExecutionError(`Failed to fetch Barchart greeks for ${symbol}`);
|
||||
}
|
||||
if (!Array.isArray(data.rows)) {
|
||||
throw new CommandExecutionError('Barchart greeks returned an unreadable options payload');
|
||||
}
|
||||
if (data.rows.length === 0) {
|
||||
throw new EmptyResultError('barchart greeks', `No option greeks were returned for ${symbol}. Confirm the symbol, expiration, and Barchart login state.`);
|
||||
}
|
||||
return data.rows.map(r => {
|
||||
if (!r || typeof r !== 'object' || Array.isArray(r)) {
|
||||
throw new CommandExecutionError('Barchart greeks returned a malformed option row');
|
||||
}
|
||||
const type = String(r.type || '').trim();
|
||||
const expirationValue = String(r.expiration || '').trim();
|
||||
if (!/^(call|put)$/i.test(type) || r.strike === undefined || r.strike === null || r.strike === '' || !expirationValue) {
|
||||
throw new CommandExecutionError('Barchart greeks returned a malformed option row identity');
|
||||
}
|
||||
return {
|
||||
type,
|
||||
strike: r.strike,
|
||||
last: r.last != null ? Number(Number(r.last).toFixed(2)) : null,
|
||||
iv: r.iv != null ? Number(Number(r.iv).toFixed(2)) + '%' : null,
|
||||
delta: r.delta != null ? Number(Number(r.delta).toFixed(4)) : null,
|
||||
gamma: r.gamma != null ? Number(Number(r.gamma).toFixed(4)) : null,
|
||||
theta: r.theta != null ? Number(Number(r.theta).toFixed(4)) : null,
|
||||
vega: r.vega != null ? Number(Number(r.vega).toFixed(4)) : null,
|
||||
rho: r.rho != null ? Number(Number(r.rho).toFixed(4)) : null,
|
||||
volume: r.volume,
|
||||
openInterest: r.openInterest,
|
||||
expiration: expirationValue,
|
||||
};
|
||||
});
|
||||
`);
|
||||
if (!data || !Array.isArray(data))
|
||||
return [];
|
||||
return data.map(r => ({
|
||||
type: r.type || '',
|
||||
strike: r.strike,
|
||||
last: r.last != null ? Number(Number(r.last).toFixed(2)) : null,
|
||||
iv: r.iv != null ? Number(Number(r.iv).toFixed(2)) + '%' : null,
|
||||
delta: r.delta != null ? Number(Number(r.delta).toFixed(4)) : null,
|
||||
gamma: r.gamma != null ? Number(Number(r.gamma).toFixed(4)) : null,
|
||||
theta: r.theta != null ? Number(Number(r.theta).toFixed(4)) : null,
|
||||
vega: r.vega != null ? Number(Number(r.vega).toFixed(4)) : null,
|
||||
rho: r.rho != null ? Number(Number(r.rho).toFixed(4)) : null,
|
||||
volume: r.volume,
|
||||
openInterest: r.openInterest,
|
||||
expiration: r.expiration ?? null,
|
||||
}));
|
||||
},
|
||||
});
|
||||
|
||||
export const __test__ = {
|
||||
normalizeSymbol,
|
||||
normalizeExpiration,
|
||||
parseLimit,
|
||||
unwrapBrowserResult,
|
||||
};
|
||||
|
||||
@@ -1,138 +0,0 @@
|
||||
import { describe, expect, it, vi } from 'vitest';
|
||||
import { getRegistry } from '@jackwener/opencli/registry';
|
||||
import { ArgumentError, CommandExecutionError, EmptyResultError } from '@jackwener/opencli/errors';
|
||||
import './greeks.js';
|
||||
|
||||
const { normalizeExpiration, normalizeSymbol, parseLimit, unwrapBrowserResult } = await import('./greeks.js').then((m) => m.__test__);
|
||||
|
||||
function makePage(evaluateResult) {
|
||||
return {
|
||||
goto: vi.fn().mockResolvedValue(undefined),
|
||||
wait: vi.fn().mockResolvedValue(undefined),
|
||||
evaluate: vi.fn().mockResolvedValue(evaluateResult),
|
||||
};
|
||||
}
|
||||
|
||||
describe('barchart greeks command', () => {
|
||||
const command = getRegistry().get('barchart/greeks');
|
||||
|
||||
it('registers with the expected shape', () => {
|
||||
expect(command).toBeDefined();
|
||||
expect(command.access).toBe('read');
|
||||
expect(command.browser).toBe(true);
|
||||
expect(command.columns).toEqual([
|
||||
'type', 'strike', 'last', 'iv', 'delta', 'gamma', 'theta', 'vega', 'rho',
|
||||
'volume', 'openInterest', 'expiration',
|
||||
]);
|
||||
});
|
||||
|
||||
it('maps returned option rows without changing the declared output shape', async () => {
|
||||
const page = makePage({
|
||||
session: 'site:barchart',
|
||||
data: {
|
||||
ok: true,
|
||||
rows: [
|
||||
{
|
||||
type: 'Call',
|
||||
strike: 190,
|
||||
last: 3.456,
|
||||
iv: 21.234,
|
||||
delta: 0.56789,
|
||||
gamma: 0.01234,
|
||||
theta: -0.12345,
|
||||
vega: 0.23456,
|
||||
rho: 0.03456,
|
||||
volume: 123,
|
||||
openInterest: 456,
|
||||
expiration: '2026-06-19',
|
||||
},
|
||||
],
|
||||
},
|
||||
});
|
||||
|
||||
const rows = await command.func(page, { symbol: 'aapl', limit: 1 });
|
||||
|
||||
expect(page.goto).toHaveBeenCalledWith('https://www.barchart.com/stocks/quotes/AAPL/options');
|
||||
expect(page.wait).toHaveBeenCalledWith(4);
|
||||
expect(rows).toEqual([
|
||||
{
|
||||
type: 'Call',
|
||||
strike: 190,
|
||||
last: 3.46,
|
||||
iv: '21.23%',
|
||||
delta: 0.5679,
|
||||
gamma: 0.0123,
|
||||
theta: -0.1235,
|
||||
vega: 0.2346,
|
||||
rho: 0.0346,
|
||||
volume: 123,
|
||||
openInterest: 456,
|
||||
expiration: '2026-06-19',
|
||||
},
|
||||
]);
|
||||
});
|
||||
|
||||
it('validates args before browser navigation and unwraps bridge envelopes', async () => {
|
||||
expect(normalizeSymbol(' aapl ')).toBe('AAPL');
|
||||
expect(normalizeExpiration('2026-06-19')).toBe('2026-06-19');
|
||||
expect(parseLimit(undefined)).toBe(10);
|
||||
expect(parseLimit(100)).toBe(100);
|
||||
expect(unwrapBrowserResult({ session: 'site:barchart', data: { ok: true } })).toEqual({ ok: true });
|
||||
|
||||
await expect(command.func(makePage({ ok: true, rows: [] }), { symbol: '', limit: 1 }))
|
||||
.rejects.toBeInstanceOf(ArgumentError);
|
||||
await expect(command.func(makePage({ ok: true, rows: [] }), { symbol: 'AAPL', expiration: '2026-02-30', limit: 1 }))
|
||||
.rejects.toBeInstanceOf(ArgumentError);
|
||||
await expect(command.func(makePage({ ok: true, rows: [] }), { symbol: 'AAPL', limit: 101 }))
|
||||
.rejects.toBeInstanceOf(ArgumentError);
|
||||
});
|
||||
|
||||
it('embeds expiration and limit in the browser-side request script', async () => {
|
||||
const page = makePage({
|
||||
ok: true,
|
||||
rows: [{
|
||||
type: 'Put',
|
||||
strike: 185,
|
||||
last: null,
|
||||
iv: null,
|
||||
delta: null,
|
||||
gamma: null,
|
||||
theta: null,
|
||||
vega: null,
|
||||
rho: null,
|
||||
volume: 0,
|
||||
openInterest: 0,
|
||||
expiration: '2026-07-17',
|
||||
}],
|
||||
});
|
||||
|
||||
await command.func(page, { symbol: 'MSFT', expiration: '2026-07-17', limit: 7 });
|
||||
const script = page.evaluate.mock.calls[0][0];
|
||||
|
||||
expect(script).toContain('const expDate = "2026-07-17"');
|
||||
expect(script).toContain('const limit = 7');
|
||||
expect(script).toContain("url += '&expirationDate=' + encodeURIComponent(expDate)");
|
||||
});
|
||||
|
||||
it('throws CommandExecutionError for HTTP, malformed, exception, and missing payload states', async () => {
|
||||
await expect(command.func(makePage({ ok: false, reason: 'http', status: 403, statusText: 'Forbidden' }), { symbol: 'AAPL' }))
|
||||
.rejects.toBeInstanceOf(CommandExecutionError);
|
||||
await expect(command.func(makePage({ ok: false, reason: 'malformed' }), { symbol: 'AAPL' }))
|
||||
.rejects.toBeInstanceOf(CommandExecutionError);
|
||||
await expect(command.func(makePage({ ok: false, reason: 'exception', message: 'network down' }), { symbol: 'AAPL' }))
|
||||
.rejects.toBeInstanceOf(CommandExecutionError);
|
||||
await expect(command.func(makePage({ ok: false, reason: 'malformed', message: 'options rows did not include call or put identities' }), { symbol: 'AAPL' }))
|
||||
.rejects.toThrow('call or put identities');
|
||||
await expect(command.func(makePage(null), { symbol: 'AAPL' }))
|
||||
.rejects.toBeInstanceOf(CommandExecutionError);
|
||||
await expect(command.func(makePage({ ok: true, rows: 'bad' }), { symbol: 'AAPL' }))
|
||||
.rejects.toBeInstanceOf(CommandExecutionError);
|
||||
await expect(command.func(makePage({ ok: true, rows: [{ type: 'Call', strike: null, expiration: '' }] }), { symbol: 'AAPL' }))
|
||||
.rejects.toThrow('malformed option row identity');
|
||||
});
|
||||
|
||||
it('throws EmptyResultError when Barchart returns no greeks rows', async () => {
|
||||
await expect(command.func(makePage({ ok: true, rows: [] }), { symbol: 'AAPL' }))
|
||||
.rejects.toBeInstanceOf(EmptyResultError);
|
||||
});
|
||||
});
|
||||
@@ -6,7 +6,6 @@ import { cli, Strategy } from '@jackwener/opencli/registry';
|
||||
cli({
|
||||
site: 'barchart',
|
||||
name: 'options',
|
||||
access: 'read',
|
||||
description: 'Barchart options chain with greeks, IV, volume, and open interest',
|
||||
domain: 'www.barchart.com',
|
||||
strategy: Strategy.COOKIE,
|
||||
|
||||
@@ -7,7 +7,6 @@ import { CommandExecutionError } from '@jackwener/opencli/errors';
|
||||
cli({
|
||||
site: 'barchart',
|
||||
name: 'quote',
|
||||
access: 'read',
|
||||
description: 'Barchart stock quote with price, volume, and key metrics',
|
||||
domain: 'www.barchart.com',
|
||||
strategy: Strategy.COOKIE,
|
||||
|
||||
@@ -5,7 +5,6 @@ import { cli, Strategy } from '@jackwener/opencli/registry';
|
||||
cli({
|
||||
site: 'bbc',
|
||||
name: 'news',
|
||||
access: 'read',
|
||||
description: 'BBC News headlines (RSS)',
|
||||
domain: 'www.bbc.com',
|
||||
strategy: Strategy.PUBLIC,
|
||||
|
||||
@@ -1,57 +0,0 @@
|
||||
// bbc topic — BBC News headlines for a specific category, via public RSS.
|
||||
//
|
||||
// BBC publishes per-section RSS feeds at
|
||||
// `https://feeds.bbci.co.uk/news/<topic>/rss.xml`. We expose the eight
|
||||
// canonical sections and reject anything else with a typed argument error
|
||||
// so the user knows the supported set.
|
||||
import { cli, Strategy } from '@jackwener/opencli/registry';
|
||||
import { ArgumentError, EmptyResultError } from '@jackwener/opencli/errors';
|
||||
import { bbcFetchRss, parseRssItems, pubDateToIso, requireBoundedInt } from './utils.js';
|
||||
|
||||
const TOPICS = [
|
||||
'world',
|
||||
'business',
|
||||
'politics',
|
||||
'health',
|
||||
'education',
|
||||
'science_and_environment',
|
||||
'technology',
|
||||
'entertainment_and_arts',
|
||||
];
|
||||
|
||||
cli({
|
||||
site: 'bbc',
|
||||
name: 'topic',
|
||||
access: 'read',
|
||||
description: 'BBC News headlines for a specific section (RSS feed)',
|
||||
domain: 'www.bbc.com',
|
||||
strategy: Strategy.PUBLIC,
|
||||
browser: false,
|
||||
args: [
|
||||
{ name: 'topic', positional: true, required: true, help: `Section name (${TOPICS.join(' / ')})` },
|
||||
{ name: 'limit', type: 'int', default: 20, help: 'Max headlines (1-50)' },
|
||||
],
|
||||
columns: ['rank', 'title', 'description', 'pubDate', 'url'],
|
||||
func: async (args) => {
|
||||
const raw = String(args.topic ?? '').trim().toLowerCase().replace(/[\s-]+/g, '_');
|
||||
if (!TOPICS.includes(raw)) {
|
||||
throw new ArgumentError(
|
||||
`bbc topic "${args.topic}" is not supported`,
|
||||
`Supported topics: ${TOPICS.join(', ')}`,
|
||||
);
|
||||
}
|
||||
const limit = requireBoundedInt(args.limit, 20, 50);
|
||||
const xml = await bbcFetchRss(`${raw}/rss.xml`, `bbc topic ${raw}`);
|
||||
const items = parseRssItems(xml);
|
||||
if (!items.length) {
|
||||
throw new EmptyResultError('bbc topic', `BBC ${raw} feed returned no items.`);
|
||||
}
|
||||
return items.slice(0, limit).map((it, i) => ({
|
||||
rank: i + 1,
|
||||
title: it.title,
|
||||
description: it.description,
|
||||
pubDate: pubDateToIso(it.pubDate),
|
||||
url: it.link,
|
||||
}));
|
||||
},
|
||||
});
|
||||
@@ -1,79 +0,0 @@
|
||||
// Shared helpers for the bbc adapters that hit BBC's public RSS feeds.
|
||||
import { ArgumentError, CommandExecutionError } from '@jackwener/opencli/errors';
|
||||
|
||||
export const BBC_FEED_BASE = 'https://feeds.bbci.co.uk/news';
|
||||
const UA = 'opencli-bbc-adapter (+https://github.com/jackwener/opencli)';
|
||||
|
||||
const HTML_ENTITIES = {
|
||||
'&': '&', '<': '<', '>': '>', '"': '"', ''': "'", ''': "'", ' ': ' ',
|
||||
};
|
||||
|
||||
export function decodeHtmlEntities(value) {
|
||||
return String(value ?? '')
|
||||
.replace(/&#x([0-9a-fA-F]+);/g, (_, h) => String.fromCodePoint(parseInt(h, 16)))
|
||||
.replace(/&#(\d+);/g, (_, d) => String.fromCodePoint(parseInt(d, 10)))
|
||||
.replace(/&(amp|lt|gt|quot|apos|#39|nbsp);/g, (m) => HTML_ENTITIES[m] || m);
|
||||
}
|
||||
|
||||
/** Extract `<tag>…</tag>` (CDATA-aware) from a block. */
|
||||
export function extractRssTag(block, tag) {
|
||||
const cdata = block.match(new RegExp(`<${tag}[^>]*>\\s*<!\\[CDATA\\[([\\s\\S]*?)\\]\\]>\\s*<\\/${tag}>`));
|
||||
if (cdata) return cdata[1];
|
||||
const plain = block.match(new RegExp(`<${tag}[^>]*>([\\s\\S]*?)<\\/${tag}>`));
|
||||
return plain ? plain[1] : '';
|
||||
}
|
||||
|
||||
export function parseRssItems(xml) {
|
||||
const out = [];
|
||||
const re = /<item[^>]*>([\s\S]*?)<\/item>/g;
|
||||
let m;
|
||||
while ((m = re.exec(String(xml || ''))) !== null) {
|
||||
const block = m[1];
|
||||
out.push({
|
||||
title: decodeHtmlEntities(extractRssTag(block, 'title')).trim(),
|
||||
description: decodeHtmlEntities(extractRssTag(block, 'description')).trim(),
|
||||
link: decodeHtmlEntities(extractRssTag(block, 'link')).trim(),
|
||||
pubDate: decodeHtmlEntities(extractRssTag(block, 'pubDate')).trim(),
|
||||
guid: decodeHtmlEntities(extractRssTag(block, 'guid')).trim(),
|
||||
});
|
||||
}
|
||||
return out;
|
||||
}
|
||||
|
||||
export function requireBoundedInt(value, defaultValue, maxValue, label = 'limit') {
|
||||
const raw = value ?? defaultValue;
|
||||
const n = typeof raw === 'number' ? raw : Number(raw);
|
||||
if (!Number.isInteger(n) || n <= 0) {
|
||||
throw new ArgumentError(`bbc ${label} must be a positive integer`);
|
||||
}
|
||||
if (n > maxValue) {
|
||||
throw new ArgumentError(`bbc ${label} must be <= ${maxValue}`);
|
||||
}
|
||||
return n;
|
||||
}
|
||||
|
||||
export async function bbcFetchRss(path, label) {
|
||||
const url = `${BBC_FEED_BASE}/${path}`;
|
||||
let resp;
|
||||
try {
|
||||
resp = await fetch(url, { headers: { 'user-agent': UA, accept: 'application/rss+xml, application/xml' } });
|
||||
}
|
||||
catch (err) {
|
||||
throw new CommandExecutionError(
|
||||
`${label} request failed: ${err?.message ?? err}`,
|
||||
'Check that feeds.bbci.co.uk is reachable from this network.',
|
||||
);
|
||||
}
|
||||
if (!resp.ok) {
|
||||
throw new CommandExecutionError(`${label} returned HTTP ${resp.status} (${url})`);
|
||||
}
|
||||
return resp.text();
|
||||
}
|
||||
|
||||
/** Convert RFC-822 pubDate to ISO `YYYY-MM-DD`; empty string on parse failure. */
|
||||
export function pubDateToIso(value) {
|
||||
if (!value) return '';
|
||||
const d = new Date(value);
|
||||
if (Number.isNaN(d.getTime())) return '';
|
||||
return d.toISOString().slice(0, 10);
|
||||
}
|
||||
@@ -7,7 +7,6 @@ import { apiGet, resolveBvid } from './utils.js';
|
||||
cli({
|
||||
site: 'bilibili',
|
||||
name: 'comments',
|
||||
access: 'read',
|
||||
description: '获取 B站视频评论(使用官方 API + WBI 签名)',
|
||||
domain: 'www.bilibili.com',
|
||||
strategy: Strategy.COOKIE,
|
||||
|
||||
@@ -14,7 +14,6 @@ import { resolveBvid } from './utils.js';
|
||||
cli({
|
||||
site: 'bilibili',
|
||||
name: 'download',
|
||||
access: 'read',
|
||||
description: '下载B站视频(需要 yt-dlp)',
|
||||
domain: 'www.bilibili.com',
|
||||
strategy: Strategy.COOKIE,
|
||||
|
||||
@@ -3,7 +3,6 @@ import { apiGet } from './utils.js';
|
||||
cli({
|
||||
site: 'bilibili',
|
||||
name: 'dynamic',
|
||||
access: 'read',
|
||||
description: 'Get Bilibili user dynamic feed',
|
||||
domain: 'www.bilibili.com',
|
||||
strategy: Strategy.COOKIE,
|
||||
|
||||
@@ -3,7 +3,6 @@ import { apiGet, payloadData, getSelfUid } from './utils.js';
|
||||
cli({
|
||||
site: 'bilibili',
|
||||
name: 'favorite',
|
||||
access: 'write',
|
||||
description: '我的收藏夹',
|
||||
domain: 'www.bilibili.com',
|
||||
strategy: Strategy.COOKIE,
|
||||
|
||||
@@ -65,7 +65,6 @@ function parseItem(item) {
|
||||
cli({
|
||||
site: 'bilibili',
|
||||
name: 'feed',
|
||||
access: 'read',
|
||||
description: '动态时间线(不传 uid 查关注时间线,传 uid 查指定用户动态)',
|
||||
domain: 'www.bilibili.com',
|
||||
strategy: Strategy.COOKIE,
|
||||
@@ -135,7 +134,6 @@ cli({
|
||||
cli({
|
||||
site: 'bilibili',
|
||||
name: 'feed-detail',
|
||||
access: 'read',
|
||||
description: '查看 Bilibili 动态详情(支持充电专属内容)',
|
||||
domain: 'www.bilibili.com',
|
||||
strategy: Strategy.COOKIE,
|
||||
|
||||
@@ -4,7 +4,6 @@ import { fetchJson, getSelfUid, resolveUid } from './utils.js';
|
||||
cli({
|
||||
site: 'bilibili',
|
||||
name: 'following',
|
||||
access: 'read',
|
||||
description: '获取 Bilibili 用户的关注列表',
|
||||
domain: 'www.bilibili.com',
|
||||
strategy: Strategy.COOKIE,
|
||||
|
||||
@@ -3,7 +3,6 @@ import { apiGet, payloadData } from './utils.js';
|
||||
cli({
|
||||
site: 'bilibili',
|
||||
name: 'history',
|
||||
access: 'read',
|
||||
description: '我的观看历史',
|
||||
domain: 'www.bilibili.com',
|
||||
strategy: Strategy.COOKIE,
|
||||
|
||||
@@ -2,7 +2,6 @@ import { cli } from '@jackwener/opencli/registry';
|
||||
cli({
|
||||
site: 'bilibili',
|
||||
name: 'hot',
|
||||
access: 'read',
|
||||
description: 'B站热门视频',
|
||||
domain: 'www.bilibili.com',
|
||||
args: [
|
||||
|
||||
+1
-1
@@ -1,7 +1,7 @@
|
||||
import { cli, Strategy } from '@jackwener/opencli/registry';
|
||||
import { apiGet, getSelfUid } from './utils.js';
|
||||
cli({
|
||||
site: 'bilibili', name: 'me', access: 'read', description: 'My Bilibili profile info', domain: 'www.bilibili.com', strategy: Strategy.COOKIE,
|
||||
site: 'bilibili', name: 'me', description: 'My Bilibili profile info', domain: 'www.bilibili.com', strategy: Strategy.COOKIE,
|
||||
args: [],
|
||||
columns: ['name', 'uid', 'level', 'coins', 'followers', 'following'],
|
||||
func: async (page) => {
|
||||
|
||||
@@ -3,7 +3,6 @@ import { apiGet } from './utils.js';
|
||||
cli({
|
||||
site: 'bilibili',
|
||||
name: 'ranking',
|
||||
access: 'read',
|
||||
description: 'Get Bilibili video ranking board',
|
||||
domain: 'www.bilibili.com',
|
||||
strategy: Strategy.COOKIE,
|
||||
|
||||
@@ -1,7 +1,7 @@
|
||||
import { cli, Strategy } from '@jackwener/opencli/registry';
|
||||
import { apiGet, stripHtml } from './utils.js';
|
||||
cli({
|
||||
site: 'bilibili', name: 'search', access: 'read', description: 'Search Bilibili videos or users', domain: 'www.bilibili.com', strategy: Strategy.COOKIE,
|
||||
site: 'bilibili', name: 'search', description: 'Search Bilibili videos or users', domain: 'www.bilibili.com', strategy: Strategy.COOKIE,
|
||||
args: [
|
||||
{ name: 'query', required: true, positional: true, help: 'Search keyword' },
|
||||
{ name: 'type', default: 'video', help: 'video or user' },
|
||||
|
||||
@@ -4,11 +4,10 @@ import { apiGet, resolveBvid } from './utils.js';
|
||||
cli({
|
||||
site: 'bilibili',
|
||||
name: 'subtitle',
|
||||
access: 'read',
|
||||
description: '获取 Bilibili 视频的字幕',
|
||||
strategy: Strategy.COOKIE,
|
||||
args: [
|
||||
{ name: 'bvid', required: true, positional: true, help: 'Bilibili 视频 BV ID(如 BV1xx411c7mD),或视频 URL / b23.tv 短链' },
|
||||
{ name: 'bvid', required: true, positional: true },
|
||||
{ name: 'lang', required: false, help: '字幕语言代码 (如 zh-CN, en-US, ai-zh),默认取第一个' },
|
||||
],
|
||||
columns: ['index', 'from', 'to', 'content'],
|
||||
|
||||
@@ -1,167 +0,0 @@
|
||||
/**
|
||||
* Bilibili summary — fetches the official AI-generated video summary (the "AI总结"
|
||||
* shown on the video page) via /x/web-interface/view/conclusion/get.
|
||||
*/
|
||||
import { cli, Strategy } from '@jackwener/opencli/registry';
|
||||
import { ArgumentError, AuthRequiredError, CommandExecutionError, EmptyResultError } from '@jackwener/opencli/errors';
|
||||
import { apiGet, resolveBvid } from './utils.js';
|
||||
|
||||
const BILIBILI_HOST_RE = /(^|\.)bilibili\.com$/i;
|
||||
const B23_HOST_RE = /(^|\.)b23\.tv$/i;
|
||||
const BVID_RE = /^BV[A-Za-z0-9]+$/;
|
||||
|
||||
function formatTime(seconds) {
|
||||
const s = Math.max(0, Math.floor(Number(seconds) || 0));
|
||||
const h = Math.floor(s / 3600);
|
||||
const m = Math.floor((s % 3600) / 60);
|
||||
const sec = s % 60;
|
||||
const pad = (n) => String(n).padStart(2, '0');
|
||||
return h > 0 ? `${h}:${pad(m)}:${pad(sec)}` : `${pad(m)}:${pad(sec)}`;
|
||||
}
|
||||
|
||||
async function readBvid(raw) {
|
||||
const input = String(raw ?? '').trim();
|
||||
if (!input) {
|
||||
throw new ArgumentError('bilibili summary bvid cannot be empty', 'Pass a BV ID, Bilibili video URL, or b23.tv short link.');
|
||||
}
|
||||
if (BVID_RE.test(input)) {
|
||||
return input;
|
||||
}
|
||||
let parsed = null;
|
||||
try {
|
||||
parsed = new URL(input);
|
||||
} catch {
|
||||
// Bare b23.tv short codes are accepted by the shared resolver.
|
||||
}
|
||||
if (parsed) {
|
||||
if (parsed.protocol !== 'https:' && parsed.protocol !== 'http:') {
|
||||
throw new ArgumentError('Bilibili summary URL must use http or https');
|
||||
}
|
||||
if (BILIBILI_HOST_RE.test(parsed.hostname)) {
|
||||
const match = parsed.pathname.match(/\/(?:video|bangumi\/play)\/(BV[A-Za-z0-9]+)/i);
|
||||
if (!match) {
|
||||
throw new ArgumentError('Bilibili summary URL must contain a BV video id');
|
||||
}
|
||||
return match[1];
|
||||
}
|
||||
if (!B23_HOST_RE.test(parsed.hostname)) {
|
||||
throw new ArgumentError('Bilibili summary URL must be a bilibili.com or b23.tv URL');
|
||||
}
|
||||
}
|
||||
try {
|
||||
return await resolveBvid(input);
|
||||
} catch (error) {
|
||||
throw new ArgumentError(`Cannot resolve Bilibili BV ID from input: ${input}`, error instanceof Error ? error.message : String(error));
|
||||
}
|
||||
}
|
||||
|
||||
function requireOkPayload(payload, label) {
|
||||
if (!payload || typeof payload !== 'object' || Array.isArray(payload)) {
|
||||
throw new CommandExecutionError(`Bilibili ${label} API returned a malformed payload`);
|
||||
}
|
||||
if (payload.code !== 0) {
|
||||
const message = payload.message ?? 'unknown error';
|
||||
if (payload.code === -101 || payload.code === -403 || /登录|权限|forbidden|permission|login/i.test(String(message))) {
|
||||
throw new AuthRequiredError('bilibili.com', `Bilibili ${label} API requires login or permission: ${message} (${payload.code})`);
|
||||
}
|
||||
throw new CommandExecutionError(`Bilibili ${label} API failed: ${message} (${payload.code})`);
|
||||
}
|
||||
return payload.data;
|
||||
}
|
||||
|
||||
function readModelResult(data, bvid) {
|
||||
if (!data || typeof data !== 'object' || Array.isArray(data)) {
|
||||
throw new CommandExecutionError('Bilibili conclusion API returned malformed data');
|
||||
}
|
||||
if (data.code !== 0) {
|
||||
throw new EmptyResultError('bilibili summary', `Bilibili has not generated an AI summary for ${bvid}.`);
|
||||
}
|
||||
let modelResult = data.model_result;
|
||||
if (typeof modelResult === 'string') {
|
||||
try {
|
||||
modelResult = JSON.parse(modelResult);
|
||||
} catch {
|
||||
throw new CommandExecutionError('Bilibili conclusion API returned malformed model_result JSON');
|
||||
}
|
||||
}
|
||||
if (!modelResult || typeof modelResult !== 'object' || Array.isArray(modelResult)) {
|
||||
throw new CommandExecutionError('Bilibili conclusion API returned malformed model_result');
|
||||
}
|
||||
const summary = String(modelResult.summary ?? '').trim();
|
||||
if (!summary) {
|
||||
throw new EmptyResultError('bilibili summary', `Bilibili has not generated an AI summary for ${bvid}.`);
|
||||
}
|
||||
const outline = modelResult.outline ?? [];
|
||||
if (!Array.isArray(outline)) {
|
||||
throw new CommandExecutionError('Bilibili conclusion API returned malformed outline');
|
||||
}
|
||||
return { summary, outline };
|
||||
}
|
||||
|
||||
function rowsFromModel(model) {
|
||||
const rows = [{ time: '', content: model.summary }];
|
||||
for (const section of model.outline) {
|
||||
if (!section || typeof section !== 'object' || Array.isArray(section)) {
|
||||
throw new CommandExecutionError('Bilibili conclusion API returned malformed outline section');
|
||||
}
|
||||
const sectionTitle = String(section.title ?? '').trim();
|
||||
const sectionTime = formatTime(section.timestamp);
|
||||
if (sectionTitle) {
|
||||
rows.push({ time: sectionTime, content: `# ${sectionTitle}` });
|
||||
}
|
||||
const points = section.part_outline ?? [];
|
||||
if (!Array.isArray(points)) {
|
||||
throw new CommandExecutionError('Bilibili conclusion API returned malformed part outline');
|
||||
}
|
||||
for (const point of points) {
|
||||
if (!point || typeof point !== 'object' || Array.isArray(point)) {
|
||||
throw new CommandExecutionError('Bilibili conclusion API returned malformed outline point');
|
||||
}
|
||||
const content = String(point.content ?? '').trim();
|
||||
if (content) {
|
||||
rows.push({ time: formatTime(point.timestamp), content });
|
||||
}
|
||||
}
|
||||
}
|
||||
return rows;
|
||||
}
|
||||
|
||||
var command = cli({
|
||||
site: 'bilibili',
|
||||
name: 'summary',
|
||||
access: 'read',
|
||||
description: '获取 B站视频的官方 AI 总结(视频页「AI总结」同款,含分段大纲与时间戳)',
|
||||
domain: 'www.bilibili.com',
|
||||
strategy: Strategy.COOKIE,
|
||||
args: [
|
||||
{ name: 'bvid', required: true, positional: true, help: 'Video BV ID / URL / b23.tv short link' },
|
||||
],
|
||||
columns: ['time', 'content'],
|
||||
func: async (page, kwargs) => {
|
||||
if (!page) {
|
||||
throw new CommandExecutionError('Browser session required for bilibili summary');
|
||||
}
|
||||
const bvid = await readBvid(kwargs.bvid);
|
||||
const view = await apiGet(page, '/x/web-interface/view', { params: { bvid } });
|
||||
const viewData = requireOkPayload(view, 'view');
|
||||
const cid = viewData?.cid;
|
||||
const upMid = viewData?.owner?.mid;
|
||||
if (!cid || !upMid) {
|
||||
throw new CommandExecutionError(`Bilibili view API did not return cid/up_mid for ${bvid}`);
|
||||
}
|
||||
const conclusion = await apiGet(page, '/x/web-interface/view/conclusion/get', {
|
||||
params: { bvid, cid, up_mid: upMid },
|
||||
signed: true,
|
||||
});
|
||||
const conclusionData = requireOkPayload(conclusion, 'conclusion');
|
||||
return rowsFromModel(readModelResult(conclusionData, bvid));
|
||||
},
|
||||
});
|
||||
|
||||
export const __test__ = {
|
||||
command,
|
||||
formatTime,
|
||||
readBvid,
|
||||
readModelResult,
|
||||
rowsFromModel,
|
||||
};
|
||||
@@ -1,210 +0,0 @@
|
||||
import { beforeEach, describe, expect, it, vi } from 'vitest';
|
||||
import { ArgumentError, AuthRequiredError, CommandExecutionError, EmptyResultError } from '@jackwener/opencli/errors';
|
||||
|
||||
const { mockApiGet, mockResolveBvid } = vi.hoisted(() => ({
|
||||
mockApiGet: vi.fn(),
|
||||
mockResolveBvid: vi.fn(),
|
||||
}));
|
||||
|
||||
vi.mock('./utils.js', async (importOriginal) => ({
|
||||
...(await importOriginal()),
|
||||
apiGet: mockApiGet,
|
||||
resolveBvid: mockResolveBvid,
|
||||
}));
|
||||
|
||||
import { getRegistry } from '@jackwener/opencli/registry';
|
||||
import './summary.js';
|
||||
|
||||
describe('bilibili summary', () => {
|
||||
const command = getRegistry().get('bilibili/summary');
|
||||
const page = {};
|
||||
|
||||
beforeEach(() => {
|
||||
mockApiGet.mockReset();
|
||||
mockResolveBvid.mockReset();
|
||||
mockResolveBvid.mockRejectedValue(new Error('short link not found'));
|
||||
});
|
||||
|
||||
function mockView(data = { aid: 114, cid: 222, owner: { mid: 333 } }) {
|
||||
mockApiGet.mockResolvedValueOnce({ code: 0, data });
|
||||
}
|
||||
|
||||
function mockConclusion(modelResult) {
|
||||
mockApiGet.mockResolvedValueOnce({
|
||||
code: 0,
|
||||
data: {
|
||||
code: 0,
|
||||
model_result: modelResult,
|
||||
},
|
||||
});
|
||||
}
|
||||
|
||||
it('returns the summary plus timestamped outline rows', async () => {
|
||||
mockView();
|
||||
mockConclusion({
|
||||
summary: '整体总结',
|
||||
outline: [
|
||||
{
|
||||
title: '第一节',
|
||||
timestamp: 0,
|
||||
part_outline: [
|
||||
{ timestamp: 12, content: '要点A' },
|
||||
{ timestamp: 3725, content: '要点B' },
|
||||
],
|
||||
},
|
||||
],
|
||||
});
|
||||
|
||||
const result = await command.func(page, { bvid: 'BV1xxx' });
|
||||
|
||||
expect(mockApiGet).toHaveBeenNthCalledWith(1, page, '/x/web-interface/view', { params: { bvid: 'BV1xxx' } });
|
||||
expect(mockApiGet).toHaveBeenNthCalledWith(2, page, '/x/web-interface/view/conclusion/get', {
|
||||
params: { bvid: 'BV1xxx', cid: 222, up_mid: 333 },
|
||||
signed: true,
|
||||
});
|
||||
expect(result).toEqual([
|
||||
{ time: '', content: '整体总结' },
|
||||
{ time: '00:00', content: '# 第一节' },
|
||||
{ time: '00:12', content: '要点A' },
|
||||
{ time: '1:02:05', content: '要点B' },
|
||||
]);
|
||||
});
|
||||
|
||||
it('returns just the summary when the video has no outline', async () => {
|
||||
mockView({ aid: 1, cid: 2, owner: { mid: 3 } });
|
||||
mockConclusion({ summary: '只有总结', outline: [] });
|
||||
|
||||
await expect(command.func(page, { bvid: 'BV1xxx' })).resolves.toEqual([
|
||||
{ time: '', content: '只有总结' },
|
||||
]);
|
||||
});
|
||||
|
||||
it('parses model_result when Bilibili returns it as a JSON string', async () => {
|
||||
mockView({ aid: 1, cid: 2, owner: { mid: 3 } });
|
||||
mockConclusion(JSON.stringify({ summary: '字符串总结', outline: [] }));
|
||||
|
||||
await expect(command.func(page, { bvid: 'BV1xxx' })).resolves.toEqual([
|
||||
{ time: '', content: '字符串总结' },
|
||||
]);
|
||||
});
|
||||
|
||||
it('normalizes Bilibili video URLs before calling the APIs', async () => {
|
||||
mockView({ aid: 1, cid: 2, owner: { mid: 3 } });
|
||||
mockConclusion({ summary: 'URL 总结', outline: [] });
|
||||
|
||||
await command.func(page, {
|
||||
bvid: 'https://www.bilibili.com/video/BV1abc12345/?spm_id_from=333.1007',
|
||||
});
|
||||
|
||||
expect(mockApiGet).toHaveBeenNthCalledWith(1, page, '/x/web-interface/view', { params: { bvid: 'BV1abc12345' } });
|
||||
});
|
||||
|
||||
it('resolves b23.tv short links through the shared resolver', async () => {
|
||||
mockResolveBvid.mockResolvedValueOnce('BVshort12345');
|
||||
mockView({ aid: 1, cid: 2, owner: { mid: 3 } });
|
||||
mockConclusion({ summary: '短链总结', outline: [] });
|
||||
|
||||
await command.func(page, { bvid: 'https://b23.tv/abc' });
|
||||
|
||||
expect(mockResolveBvid).toHaveBeenCalledWith('https://b23.tv/abc');
|
||||
expect(mockApiGet).toHaveBeenNthCalledWith(1, page, '/x/web-interface/view', { params: { bvid: 'BVshort12345' } });
|
||||
});
|
||||
|
||||
it('rejects invalid inputs before calling Bilibili APIs', async () => {
|
||||
const cases = [
|
||||
'',
|
||||
'javascript:alert(1)',
|
||||
'https://example.com/video/BV1abc12345',
|
||||
'https://share.note.youdao.com/video/BV1abc12345',
|
||||
'https://www.bilibili.com/read/cv12345',
|
||||
];
|
||||
|
||||
for (const bvid of cases) {
|
||||
await expect(command.func(page, { bvid })).rejects.toBeInstanceOf(ArgumentError);
|
||||
}
|
||||
expect(mockApiGet).not.toHaveBeenCalled();
|
||||
});
|
||||
|
||||
it('maps unresolved short-code inputs to ArgumentError without calling APIs', async () => {
|
||||
await expect(command.func(page, { bvid: 'not-a-bv' })).rejects.toBeInstanceOf(ArgumentError);
|
||||
|
||||
expect(mockResolveBvid).toHaveBeenCalledWith('not-a-bv');
|
||||
expect(mockApiGet).not.toHaveBeenCalled();
|
||||
});
|
||||
|
||||
it('throws EmptyResultError when Bilibili has not generated an AI summary for the video', async () => {
|
||||
mockView({ aid: 1, cid: 2, owner: { mid: 3 } });
|
||||
mockApiGet.mockResolvedValueOnce({ code: 0, data: { code: 1, model_result: {} } });
|
||||
|
||||
await expect(command.func(page, { bvid: 'BV1xxx' })).rejects.toBeInstanceOf(EmptyResultError);
|
||||
});
|
||||
|
||||
it('throws CommandExecutionError when the view payload is malformed', async () => {
|
||||
mockApiGet.mockResolvedValueOnce({ code: 0, data: {} });
|
||||
|
||||
await expect(command.func(page, { bvid: 'BVbroken' })).rejects.toSatisfy(
|
||||
(err) => err instanceof CommandExecutionError && /cid\/up_mid/.test(err.message),
|
||||
);
|
||||
});
|
||||
|
||||
it('throws CommandExecutionError when the view API returns a non-auth error', async () => {
|
||||
mockApiGet.mockResolvedValueOnce({ code: -404, message: '啥都木有' });
|
||||
|
||||
await expect(command.func(page, { bvid: 'BVbroken' })).rejects.toSatisfy(
|
||||
(err) => err instanceof CommandExecutionError && /啥都木有.*-404/.test(err.message),
|
||||
);
|
||||
});
|
||||
|
||||
it('maps conclusion auth or permission errors to AuthRequiredError', async () => {
|
||||
mockView({ aid: 1, cid: 2, owner: { mid: 3 } });
|
||||
mockApiGet.mockResolvedValueOnce({ code: -403, message: '访问权限不足' });
|
||||
|
||||
await expect(command.func(page, { bvid: 'BV1xxx' })).rejects.toBeInstanceOf(AuthRequiredError);
|
||||
});
|
||||
|
||||
it('maps conclusion non-auth API errors to CommandExecutionError', async () => {
|
||||
mockView({ aid: 1, cid: 2, owner: { mid: 3 } });
|
||||
mockApiGet.mockResolvedValueOnce({ code: -500, message: 'server error' });
|
||||
|
||||
await expect(command.func(page, { bvid: 'BV1xxx' })).rejects.toSatisfy(
|
||||
(err) => err instanceof CommandExecutionError && /server error.*-500/.test(err.message),
|
||||
);
|
||||
});
|
||||
|
||||
it('throws CommandExecutionError for malformed conclusion API payloads', async () => {
|
||||
mockView({ aid: 1, cid: 2, owner: { mid: 3 } });
|
||||
mockApiGet.mockResolvedValueOnce(null);
|
||||
|
||||
await expect(command.func(page, { bvid: 'BV1xxx' })).rejects.toBeInstanceOf(CommandExecutionError);
|
||||
});
|
||||
|
||||
it('throws CommandExecutionError for malformed model_result JSON', async () => {
|
||||
mockView({ aid: 1, cid: 2, owner: { mid: 3 } });
|
||||
mockConclusion('{bad json');
|
||||
|
||||
await expect(command.func(page, { bvid: 'BV1xxx' })).rejects.toSatisfy(
|
||||
(err) => err instanceof CommandExecutionError && /model_result JSON/.test(err.message),
|
||||
);
|
||||
});
|
||||
|
||||
it('throws CommandExecutionError for malformed outline shapes', async () => {
|
||||
mockView({ aid: 1, cid: 2, owner: { mid: 3 } });
|
||||
mockConclusion({ summary: '坏 outline', outline: {} });
|
||||
|
||||
await expect(command.func(page, { bvid: 'BV1xxx' })).rejects.toSatisfy(
|
||||
(err) => err instanceof CommandExecutionError && /outline/.test(err.message),
|
||||
);
|
||||
});
|
||||
|
||||
it('throws CommandExecutionError for malformed part outline shapes', async () => {
|
||||
mockView({ aid: 1, cid: 2, owner: { mid: 3 } });
|
||||
mockConclusion({
|
||||
summary: '坏 part_outline',
|
||||
outline: [{ title: '段落', timestamp: 0, part_outline: {} }],
|
||||
});
|
||||
|
||||
await expect(command.func(page, { bvid: 'BV1xxx' })).rejects.toSatisfy(
|
||||
(err) => err instanceof CommandExecutionError && /part outline/.test(err.message),
|
||||
);
|
||||
});
|
||||
});
|
||||
Some files were not shown because too many files have changed in this diff Show More
Reference in New Issue
Block a user