mirror of
https://github.com/jackwener/OpenCLI.git
synced 2026-09-14 18:25:42 +08:00
80eef46b4e
* refactor: move adapters from src/clis/ to root clis/ for monorepo separation Separates CLI adapters from the core runtime to prepare for independent adapter distribution via postinstall fetch. Key changes: - Move src/clis/ → clis/ (adapters at repo root) - Change tsconfig rootDir from "src" to "." so tsc compiles both - Create root-level shim files (registry.ts, errors.ts, etc.) so adapter relative imports (../../registry.js) resolve correctly - Update build-manifest.ts, main.ts paths for new dist/src/ structure - Expand ensureUserCliCompatShims() to cover all adapter import targets (types, utils, logger, launcher, browser/*, download/*, pipeline/*) - Add scripts/fetch-adapters.js postinstall for ~/.opencli/clis/ sync - Update vitest.config.ts adapter test paths - Add package.json files field to exclude adapters from npm package Official adapter files are unconditionally overwritten on update; user-created files not in the manifest are preserved. * fix: add dist/clis/ and cli-manifest.json to npm files, harden fetch-adapters - Add dist/clis/ and dist/cli-manifest.json to package.json files field so built-in adapters and manifest ship with the npm package - Replace execSync with execFileSync to prevent command injection - Add version check to skip redundant adapter fetches - Track tmpRoot explicitly for reliable cleanup * fix: address review blockers — manifest-based updates, global-only fetch, first-run fallback 1. Manifest-based update strategy: - Read old manifest to identify previously-official files - Clean up files removed upstream (in old manifest but not new) - User-created files (never in any manifest) remain untouched 2. Only run fetch-adapters on global install (npm_config_global=true) or explicit OPENCLI_FETCH=1, preventing heavy side effects for local/dev installs 3. First-run fallback in discovery.ts: - ensureUserAdapters() checks for adapter-manifest.json - If missing and ~/.opencli/clis/ is empty, spawns fetch-adapters.js - Guarantees adapters are available even with --ignore-scripts * fix: remove OPENCLI_FETCH env var, use internal _OPENCLI_FIRST_RUN instead * feat: also support OPENCLI_FETCH=1 for explicit adapter fetch trigger * simplify: replace git clone with local copy from dist/clis/ Adapters already ship in the npm package (dist/clis/), so there's no need to clone from GitHub. Copy directly from the installed package: - Eliminates git, curl, tar dependencies - No network calls in postinstall - No timeout/offline issues - Version always matches the installed CLI - ~65 lines of clone/download code replaced by one cpSync loop
198 lines
8.2 KiB
TypeScript
198 lines
8.2 KiB
TypeScript
import type { IPage } from '../../types.js';
|
||
import { clamp } from '../_shared/common.js';
|
||
|
||
const clampLimit = (limit: number) => clamp(limit || 20, 1, 50);
|
||
|
||
export function buildSinaBlogSearchUrl(keyword: string): string {
|
||
return `https://search.sina.com.cn/search?q=${encodeURIComponent(keyword)}&tp=mix`;
|
||
}
|
||
|
||
export function buildSinaBlogUserUrl(uid: string): string {
|
||
return `https://blog.sina.com.cn/s/articlelist_${encodeURIComponent(uid)}_0_1.html`;
|
||
}
|
||
|
||
export async function loadSinaBlogArticle(page: IPage, url: string): Promise<any> {
|
||
await page.goto(url);
|
||
await page.wait({ selector: 'h1', timeout: 3 });
|
||
return page.evaluate(`
|
||
(async () => {
|
||
await new Promise((resolve) => setTimeout(resolve, 1500));
|
||
const normalize = (value) => (value || '').replace(/\\s+/g, ' ').trim();
|
||
const title = normalize(document.querySelector('.articalTitle h2, .title h2, h1, h2.titName')?.textContent);
|
||
const titleParts = normalize(document.title).split('_').map((part) => normalize(part)).filter(Boolean);
|
||
const author = titleParts[1] || title.split(/[::]/)[0] || '';
|
||
const timeText = normalize(document.querySelector('.time, .articalInfo .time')?.textContent).replace(/[()]/g, '');
|
||
const date = timeText || normalize(document.body.innerText.match(/\\b\\d{4}-\\d{2}-\\d{2}(?:\\s+\\d{2}:\\d{2}:\\d{2})?\\b/)?.[0]);
|
||
const category = normalize(document.querySelector('.articalTag .blog_class a, .blog_class a')?.textContent);
|
||
const tags = Array.from(document.querySelectorAll('.blog_tag h3, .blog_tag a, .tag a, .artical_tag a'))
|
||
.map((node) => normalize(node.textContent))
|
||
.filter(Boolean);
|
||
const content = normalize(document.querySelector('.articalContent, .blog_content, .content, #sina_keyword_ad_area2')?.textContent).slice(0, 500);
|
||
const images = Array.from(document.querySelectorAll('.articalContent img, .blog_content img, .content img'))
|
||
.map((img) => img.getAttribute('src') || img.getAttribute('real_src') || '')
|
||
.filter((src) => src && !src.includes('icon'))
|
||
.slice(0, 5);
|
||
return {
|
||
title,
|
||
author,
|
||
date,
|
||
category,
|
||
tags: tags.join(', '),
|
||
readCount: '',
|
||
commentCount: '',
|
||
content: content + (content.length >= 500 ? '...' : ''),
|
||
images: images.join(', '),
|
||
url: ${JSON.stringify(url)},
|
||
};
|
||
})()
|
||
`);
|
||
}
|
||
|
||
export async function loadSinaBlogHot(page: IPage, limit: number): Promise<any[]> {
|
||
const safeLimit = clampLimit(limit);
|
||
await page.goto('https://blog.sina.com.cn/');
|
||
await page.wait({ selector: 'h1', timeout: 3 });
|
||
const data = await page.evaluate(`
|
||
(async () => {
|
||
await new Promise((resolve) => setTimeout(resolve, 1500));
|
||
const normalize = (value) => (value || '').replace(/\\s+/g, ' ').trim();
|
||
const limit = ${safeLimit};
|
||
const abs = (href) => {
|
||
if (!href) return '';
|
||
if (href.startsWith('//')) return 'https:' + href;
|
||
if (href.startsWith('http')) return href;
|
||
return 'https://blog.sina.com.cn' + (href.startsWith('/') ? '' : '/') + href;
|
||
};
|
||
const parseArticle = (doc, fallback) => {
|
||
const title = normalize(doc.querySelector('.articalTitle h2, .title h2, h1, h2.titName')?.textContent) || fallback.title;
|
||
const titleParts = normalize(doc.title).split('_').map((part) => normalize(part)).filter(Boolean);
|
||
const timeText = normalize(doc.querySelector('.time, .articalInfo .time')?.textContent).replace(/[()]/g, '');
|
||
const articleId = fallback.url.match(/blog_([a-zA-Z0-9]+)\\.html/)?.[1] || '';
|
||
return {
|
||
articleId,
|
||
title,
|
||
author: titleParts[1] || title.split(/[::]/)[0] || '',
|
||
date: timeText || '',
|
||
readCount: '',
|
||
description: normalize(doc.querySelector('.articalContent, .blog_content, .content, #sina_keyword_ad_area2')?.textContent).slice(0, 150),
|
||
};
|
||
};
|
||
|
||
const seeds = [];
|
||
const seen = new Set();
|
||
for (const link of Array.from(document.querySelectorAll('.day-hot-rank .art-list a[href*="/s/blog_"], .hot-rank .art-list a[href*="/s/blog_"]'))) {
|
||
const title = normalize(link.textContent);
|
||
const url = abs(link.getAttribute('href') || '');
|
||
if (!title || !url || seen.has(url)) continue;
|
||
seen.add(url);
|
||
seeds.push({ rank: seeds.length + 1, title, url });
|
||
if (seeds.length >= limit) break;
|
||
}
|
||
|
||
const results = [];
|
||
for (const item of seeds) {
|
||
let merged = {
|
||
rank: item.rank,
|
||
articleId: item.url.match(/blog_([a-zA-Z0-9]+)\\.html/)?.[1] || '',
|
||
title: item.title,
|
||
author: '',
|
||
date: '',
|
||
readCount: '',
|
||
description: '',
|
||
url: item.url,
|
||
};
|
||
try {
|
||
const resp = await fetch(item.url, { credentials: 'include' });
|
||
if (resp.ok) {
|
||
const html = await resp.text();
|
||
const doc = new DOMParser().parseFromString(html, 'text/html');
|
||
merged = Object.assign(merged, parseArticle(doc, item));
|
||
}
|
||
} catch {}
|
||
results.push(merged);
|
||
}
|
||
return results;
|
||
})()
|
||
`);
|
||
|
||
return Array.isArray(data) ? data : [];
|
||
}
|
||
|
||
export async function loadSinaBlogSearch(page: IPage, keyword: string, limit: number): Promise<any[]> {
|
||
const safeLimit = clampLimit(limit);
|
||
await page.goto(buildSinaBlogSearchUrl(keyword));
|
||
await page.wait({ selector: '.result-item', timeout: 5 });
|
||
const data = await page.evaluate(`
|
||
(async () => {
|
||
const sleep = (ms) => new Promise((resolve) => setTimeout(resolve, ms));
|
||
for (let i = 0; i < 20; i += 1) {
|
||
if (document.querySelector('.result-item')) break;
|
||
await sleep(500);
|
||
}
|
||
const normalize = (value) => (value || '').replace(/\\s+/g, ' ').trim();
|
||
const limit = ${safeLimit};
|
||
const items = Array.from(document.querySelectorAll('.result-item'));
|
||
const results = [];
|
||
for (const item of items) {
|
||
const link = item.querySelector('.result-title a[href*="blog.sina.com.cn/s/blog_"]');
|
||
const title = normalize(link?.textContent);
|
||
const url = link?.getAttribute('href') || '';
|
||
if (!title || !url) continue;
|
||
results.push({
|
||
rank: results.length + 1,
|
||
title,
|
||
author: normalize(item.querySelector('.result-meta .source')?.textContent),
|
||
date: normalize(item.querySelector('.result-meta .time')?.textContent),
|
||
description: normalize(item.querySelector('.result-intro')?.textContent).slice(0, 150),
|
||
url,
|
||
});
|
||
if (results.length >= limit) break;
|
||
}
|
||
return results;
|
||
})()
|
||
`);
|
||
|
||
return Array.isArray(data) ? data : [];
|
||
}
|
||
|
||
export async function loadSinaBlogUser(page: IPage, uid: string, limit: number): Promise<any[]> {
|
||
const safeLimit = clampLimit(limit);
|
||
await page.goto(buildSinaBlogUserUrl(uid));
|
||
await page.wait({ selector: 'h1', timeout: 3 });
|
||
const data = await page.evaluate(`
|
||
(async () => {
|
||
await new Promise((resolve) => setTimeout(resolve, 1000));
|
||
const normalize = (value) => (value || '').replace(/\\s+/g, ' ').trim();
|
||
const limit = ${safeLimit};
|
||
const author = normalize(document.title).split('_').map((part) => normalize(part)).filter(Boolean)[1] || '';
|
||
const abs = (href) => {
|
||
if (!href) return '';
|
||
if (href.startsWith('//')) return 'https:' + href;
|
||
if (href.startsWith('http')) return href;
|
||
return 'https://blog.sina.com.cn' + (href.startsWith('/') ? '' : '/') + href;
|
||
};
|
||
const results = [];
|
||
for (const item of Array.from(document.querySelectorAll('.articleList .articleCell'))) {
|
||
const link = item.querySelector('.atc_title a[href*="/s/blog_"]');
|
||
const title = normalize(link?.textContent);
|
||
const url = abs(link?.getAttribute('href') || '');
|
||
if (!title || !url) continue;
|
||
results.push({
|
||
rank: results.length + 1,
|
||
articleId: url.match(/blog_([a-zA-Z0-9]+)\\.html/)?.[1] || '',
|
||
title,
|
||
author,
|
||
date: normalize(item.querySelector('.atc_tm')?.textContent),
|
||
readCount: '',
|
||
description: '',
|
||
url,
|
||
});
|
||
if (results.length >= limit) break;
|
||
}
|
||
return results;
|
||
})()
|
||
`);
|
||
|
||
return Array.isArray(data) ? data : [];
|
||
}
|