Files
jackwener__opencli/clis/substack/utils.ts
jakevin 80eef46b4e refactor: monorepo adapter separation (clis/ at root) (#782)
* refactor: move adapters from src/clis/ to root clis/ for monorepo separation

Separates CLI adapters from the core runtime to prepare for independent
adapter distribution via postinstall fetch.

Key changes:
- Move src/clis/ → clis/ (adapters at repo root)
- Change tsconfig rootDir from "src" to "." so tsc compiles both
- Create root-level shim files (registry.ts, errors.ts, etc.) so adapter
  relative imports (../../registry.js) resolve correctly
- Update build-manifest.ts, main.ts paths for new dist/src/ structure
- Expand ensureUserCliCompatShims() to cover all adapter import targets
  (types, utils, logger, launcher, browser/*, download/*, pipeline/*)
- Add scripts/fetch-adapters.js postinstall for ~/.opencli/clis/ sync
- Update vitest.config.ts adapter test paths
- Add package.json files field to exclude adapters from npm package

Official adapter files are unconditionally overwritten on update;
user-created files not in the manifest are preserved.

* fix: add dist/clis/ and cli-manifest.json to npm files, harden fetch-adapters

- Add dist/clis/ and dist/cli-manifest.json to package.json files field
  so built-in adapters and manifest ship with the npm package
- Replace execSync with execFileSync to prevent command injection
- Add version check to skip redundant adapter fetches
- Track tmpRoot explicitly for reliable cleanup

* fix: address review blockers — manifest-based updates, global-only fetch, first-run fallback

1. Manifest-based update strategy:
   - Read old manifest to identify previously-official files
   - Clean up files removed upstream (in old manifest but not new)
   - User-created files (never in any manifest) remain untouched

2. Only run fetch-adapters on global install (npm_config_global=true)
   or explicit OPENCLI_FETCH=1, preventing heavy side effects for
   local/dev installs

3. First-run fallback in discovery.ts:
   - ensureUserAdapters() checks for adapter-manifest.json
   - If missing and ~/.opencli/clis/ is empty, spawns fetch-adapters.js
   - Guarantees adapters are available even with --ignore-scripts

* fix: remove OPENCLI_FETCH env var, use internal _OPENCLI_FIRST_RUN instead

* feat: also support OPENCLI_FETCH=1 for explicit adapter fetch trigger

* simplify: replace git clone with local copy from dist/clis/

Adapters already ship in the npm package (dist/clis/), so there's no
need to clone from GitHub. Copy directly from the installed package:

- Eliminates git, curl, tar dependencies
- No network calls in postinstall
- No timeout/offline issues
- Version always matches the installed CLI
- ~65 lines of clone/download code replaced by one cpSync loop
2026-04-05 01:46:36 +08:00

142 lines
5.2 KiB
TypeScript

import { CommandExecutionError } from '../../errors.js';
import type { IPage } from '../../types.js';
const FEED_POST_LINK_SELECTOR = 'a[href*="/home/post/"], a[href*="/p/"]';
const ARCHIVE_POST_LINK_SELECTOR = 'a[href*="/p/"]';
export function buildSubstackBrowseUrl(category?: string): string {
if (!category || category === 'all') return 'https://substack.com/';
const slug = category === 'tech' ? 'technology' : category;
return `https://substack.com/browse/${slug}`;
}
export async function loadSubstackFeed(page: IPage, url: string, limit: number): Promise<any[]> {
if (!page) throw new CommandExecutionError('Browser session required for substack feed');
await page.goto(url);
await page.wait({ selector: FEED_POST_LINK_SELECTOR, timeout: 5 });
const data = await page.evaluate(`
(async () => {
await new Promise((resolve) => setTimeout(resolve, 3000));
const limit = ${Math.max(1, Math.min(limit, 50))};
const normalize = (value) => (value || '').replace(/\\s+/g, ' ').trim();
const posts = [];
const seen = new Set();
const allLinks = Array.from(document.querySelectorAll('a')).filter((link) => {
const href = link.getAttribute('href') || '';
return href.includes('/home/post/') || href.includes('/p/');
});
for (const linkEl of allLinks) {
let postUrl = linkEl.getAttribute('href') || '';
if (!postUrl) continue;
if (!postUrl.startsWith('http')) postUrl = 'https://substack.com' + postUrl;
if (seen.has(postUrl)) continue;
const lines = (linkEl.innerText || '')
.split('\\n')
.map((line) => normalize(line))
.filter(Boolean);
const readMeta = lines.find((line) => /\\b(read|watch|listen)\\b/i.test(line)) || '';
if (!readMeta) continue;
const date = lines.find((line) => /^[A-Z]{3}\\s+\\d{1,2}$/i.test(line)) || '';
const contentLines = lines.filter((line) =>
line &&
line !== date &&
line !== readMeta &&
line.toLowerCase() !== 'save' &&
line.toLowerCase() !== 'more' &&
!/^(sign in|create account|get app)$/i.test(line),
);
const metaParts = readMeta.split('∙').map((part) => normalize(part));
const author = metaParts[0] || '';
const readTime = metaParts.slice(1).join(' ∙ ') || readMeta;
const title = contentLines.length >= 2 ? contentLines[1] : (contentLines[0] || '');
const description = contentLines.length >= 3 ? contentLines.slice(2).join(' ') : '';
if (!title) continue;
seen.add(postUrl);
posts.push({
rank: posts.length + 1,
title,
author,
date,
readTime,
description: description.slice(0, 150),
url: postUrl,
});
if (posts.length >= limit) break;
}
return posts;
})()
`);
return Array.isArray(data) ? data : [];
}
export async function loadSubstackArchive(page: IPage, baseUrl: string, limit: number): Promise<any[]> {
if (!page) throw new CommandExecutionError('Browser session required for substack archive');
await page.goto(`${baseUrl}/archive`);
await page.wait({ selector: ARCHIVE_POST_LINK_SELECTOR, timeout: 5 });
const data = await page.evaluate(`
(async () => {
await new Promise((resolve) => setTimeout(resolve, 3000));
const normalize = (value) => (value || '').replace(/\\s+/g, ' ').trim();
const limit = ${Math.max(1, Math.min(limit, 50))};
const grouped = new Map();
for (const link of Array.from(document.querySelectorAll('a[href*="/p/"]'))) {
const rawHref = link.getAttribute('href') || '';
if (!rawHref || rawHref === '/p/upgrade') continue;
const url = rawHref.startsWith('http') ? rawHref : ${JSON.stringify(baseUrl)} + rawHref;
const text = normalize(link.textContent);
if (!text) continue;
if (/^(subscribe|paid|home|about|latest|top|discussions)$/i.test(text)) continue;
if (/^[\\d,]+$/.test(text)) continue;
const entry = grouped.get(url) || { texts: new Set(), date: '' };
entry.texts.add(text);
const container = link.closest('article, section, div') || link.parentElement || link;
const containerText = normalize(container.textContent);
if (!entry.date) {
entry.date = containerText.match(/\\b(?:[A-Z]{3}\\s+\\d{1,2}|[A-Z][a-z]{2}\\s+\\d{1,2})\\b/)?.[0] || '';
}
grouped.set(url, entry);
}
const posts = [];
for (const [url, entry] of Array.from(grouped.entries())) {
const texts = Array.from(entry.texts).map((text) => normalize(text)).filter((text) => text.length > 3).sort((a, b) => a.length - b.length);
const title = texts[0] || '';
const description = texts.find((text) => text !== title) || '';
if (!title) continue;
posts.push({
rank: posts.length + 1,
title,
date: entry.date,
description: description.slice(0, 150),
url,
});
if (posts.length >= limit) break;
}
return posts;
})()
`);
return Array.isArray(data) ? data : [];
}
export const __test__ = {
FEED_POST_LINK_SELECTOR,
ARCHIVE_POST_LINK_SELECTOR,
};