mirror of
https://github.com/jackwener/OpenCLI.git
synced 2026-09-14 18:25:42 +08:00
80eef46b4e
* refactor: move adapters from src/clis/ to root clis/ for monorepo separation Separates CLI adapters from the core runtime to prepare for independent adapter distribution via postinstall fetch. Key changes: - Move src/clis/ → clis/ (adapters at repo root) - Change tsconfig rootDir from "src" to "." so tsc compiles both - Create root-level shim files (registry.ts, errors.ts, etc.) so adapter relative imports (../../registry.js) resolve correctly - Update build-manifest.ts, main.ts paths for new dist/src/ structure - Expand ensureUserCliCompatShims() to cover all adapter import targets (types, utils, logger, launcher, browser/*, download/*, pipeline/*) - Add scripts/fetch-adapters.js postinstall for ~/.opencli/clis/ sync - Update vitest.config.ts adapter test paths - Add package.json files field to exclude adapters from npm package Official adapter files are unconditionally overwritten on update; user-created files not in the manifest are preserved. * fix: add dist/clis/ and cli-manifest.json to npm files, harden fetch-adapters - Add dist/clis/ and dist/cli-manifest.json to package.json files field so built-in adapters and manifest ship with the npm package - Replace execSync with execFileSync to prevent command injection - Add version check to skip redundant adapter fetches - Track tmpRoot explicitly for reliable cleanup * fix: address review blockers — manifest-based updates, global-only fetch, first-run fallback 1. Manifest-based update strategy: - Read old manifest to identify previously-official files - Clean up files removed upstream (in old manifest but not new) - User-created files (never in any manifest) remain untouched 2. Only run fetch-adapters on global install (npm_config_global=true) or explicit OPENCLI_FETCH=1, preventing heavy side effects for local/dev installs 3. First-run fallback in discovery.ts: - ensureUserAdapters() checks for adapter-manifest.json - If missing and ~/.opencli/clis/ is empty, spawns fetch-adapters.js - Guarantees adapters are available even with --ignore-scripts * fix: remove OPENCLI_FETCH env var, use internal _OPENCLI_FIRST_RUN instead * feat: also support OPENCLI_FETCH=1 for explicit adapter fetch trigger * simplify: replace git clone with local copy from dist/clis/ Adapters already ship in the npm package (dist/clis/), so there's no need to clone from GitHub. Copy directly from the installed package: - Eliminates git, curl, tar dependencies - No network calls in postinstall - No timeout/offline issues - Version always matches the installed CLI - ~65 lines of clone/download code replaced by one cpSync loop
70 lines
2.6 KiB
TypeScript
70 lines
2.6 KiB
TypeScript
/**
|
|
* 36kr article detail — INTERCEPT strategy.
|
|
*
|
|
* Fetches the full content of a 36kr article given its ID or URL.
|
|
*/
|
|
import { cli, Strategy } from '../../registry.js';
|
|
import { CliError } from '../../errors.js';
|
|
import type { IPage } from '../../types.js';
|
|
|
|
/** Extract article ID from a full URL or a bare numeric ID string */
|
|
function parseArticleId(input: string): string {
|
|
const m = input.match(/\/p\/(\d+)/);
|
|
return m ? m[1] : input.replace(/\D/g, '');
|
|
}
|
|
|
|
cli({
|
|
site: '36kr',
|
|
name: 'article',
|
|
description: '获取36氪文章正文内容',
|
|
domain: 'www.36kr.com',
|
|
strategy: Strategy.INTERCEPT,
|
|
args: [
|
|
{ name: 'id', positional: true, required: true, help: 'Article ID or full 36kr article URL' },
|
|
],
|
|
columns: ['field', 'value'],
|
|
func: async (page: IPage, args) => {
|
|
const articleId = parseArticleId(String(args.id ?? ''));
|
|
if (!articleId) {
|
|
throw new CliError('INVALID_ARGUMENT', 'Invalid article ID or URL');
|
|
}
|
|
|
|
await page.installInterceptor('36kr.com/api');
|
|
await page.goto(`https://www.36kr.com/p/${articleId}`);
|
|
await page.wait(5);
|
|
|
|
const data: any = await page.evaluate(`
|
|
(() => {
|
|
// Title: 36kr uses class "article-title" on h1
|
|
const title = document.querySelector('.article-title, h1')?.textContent?.trim() || '';
|
|
// Author: second .author-name (first is empty nav link, second has real name)
|
|
const authorEls = document.querySelectorAll('.author-name');
|
|
const author = Array.from(authorEls).map(el => el.textContent?.trim()).filter(Boolean)[0] || '';
|
|
// Date: 36kr uses class "title-icon-item item-time" for the publish date
|
|
const dateRaw = document.querySelector('.item-time')?.textContent?.trim() || '';
|
|
const date = dateRaw.replace(/^[·\s]+/, '').trim();
|
|
// Article body paragraphs
|
|
const bodyEls = document.querySelectorAll('[class*="article-content"] p, [class*="rich-text"] p, .article p');
|
|
const body = Array.from(bodyEls)
|
|
.map(el => el.textContent?.trim())
|
|
.filter(t => t && t.length > 10)
|
|
.join(' ')
|
|
.slice(0, 800);
|
|
return { title, author, date, body };
|
|
})()
|
|
`);
|
|
|
|
if (!data?.title) {
|
|
throw new CliError('NOT_FOUND', 'Article not found or failed to load', 'Check the article ID');
|
|
}
|
|
|
|
return [
|
|
{ field: 'title', value: data.title },
|
|
{ field: 'author', value: data.author || '-' },
|
|
{ field: 'date', value: data.date || '-' },
|
|
{ field: 'url', value: `https://36kr.com/p/${articleId}` },
|
|
{ field: 'body', value: data.body || '-' },
|
|
];
|
|
},
|
|
});
|