Improve extraction scripts, validation, and documentation

Enhance article extraction and validation systems with better error handling, retry logic, and content cleaning.

Script improvements:
- Add retry logic with exponential backoff for API calls
- Implement 60-second timeout for Anthropic API requests
- Improve HTML cleaning with Substack-specific filters
- Remove navigation, footer, and UI noise from extracted content
- Better HTML entity decoding

Validation enhancements:
- Add duplicate URL detection across articles
- Validate topic assignments against known topics
- Enhanced error reporting with specific issue types

GitHub Actions:
- Add build output validation step
- Verify /out directory exists before PR creation
- Count generated HTML files for sanity check

Documentation updates:
- Update project stats (26 articles, ~30 pages)
- Document v1.2 improvements
- Add changelog entries

Type safety:
- Improve topics.ts type definitions

🤖 Generated with [Claude Code](https://claude.com/claude-code)

Co-Authored-By: Claude <noreply@anthropic.com>
This commit is contained in:
kbanc85 2025-11-09 14:06:19 -05:00
parent 050f29370b
commit 9956c9dee4
7 changed files with 195 additions and 49 deletions

View File

@ -49,10 +49,23 @@ jobs:
- name: Build site - name: Build site
if: steps.check_changes.outputs.has_changes == 'true' if: steps.check_changes.outputs.has_changes == 'true'
run: npm run build id: build
run: |
npm run build
echo "build_status=success" >> $GITHUB_OUTPUT
- name: Validate build output
if: steps.check_changes.outputs.has_changes == 'true'
run: |
if [ ! -d "out" ]; then
echo "❌ Build failed: /out directory not found"
exit 1
fi
FILE_COUNT=$(find out -name "*.html" | wc -l)
echo "✅ Build validation passed: Found $FILE_COUNT HTML files"
- name: Create Pull Request - name: Create Pull Request
if: steps.check_changes.outputs.has_changes == 'true' if: steps.check_changes.outputs.has_changes == 'true' && steps.build.outputs.build_status == 'success'
uses: peter-evans/create-pull-request@v6 uses: peter-evans/create-pull-request@v6
with: with:
token: ${{ secrets.GITHUB_TOKEN }} token: ${{ secrets.GITHUB_TOKEN }}

View File

@ -60,10 +60,10 @@ Complete reference guide for all project documentation.
## 📊 Current Project Stats ## 📊 Current Project Stats
**Last verified:** 2025-11-04 **Last verified:** 2025-11-07
- **Total pages:** 15 static HTML files - **Total pages:** ~30 static HTML files
- **Claims pages:** 10 individual articles - **Claims pages:** 26 individual articles
- **Build time:** ~3 seconds - **Build time:** ~3 seconds
- **Deployment:** Auto-deploy via GitHub → Netlify - **Deployment:** Auto-deploy via GitHub → Netlify
- **Image policy:** White background + black text + red accents ONLY - **Image policy:** White background + black text + red accents ONLY
@ -233,6 +233,15 @@ npm run build # Must succeed before committing
- Strict image criteria implemented - Strict image criteria implemented
- Comprehensive documentation created - Comprehensive documentation created
**v1.2** (Nov 7, 2025)
- Expanded to 26 claim articles
- Enhanced validation system (duplicate URLs, topic validation)
- Improved HTML cleaning with Substack-specific filters
- Added retry logic and timeouts to API calls
- Better type safety in topics.ts
- Enhanced GitHub Actions build validation
- ~30 total pages
--- ---
## 🎓 Learning Resources ## 🎓 Learning Resources

View File

@ -10,8 +10,8 @@ Next.js site for kbanc.com with static site generation for optimal GEO (Generati
This repository is connected to Netlify. Every push to main branch triggers automatic deployment. This repository is connected to Netlify. Every push to main branch triggers automatic deployment.
## Project Stats ## Project Stats
- **Total Pages:** 15 static HTML files - **Total Pages:** ~30 static HTML files
- **Claims Pages:** 10 individual claim articles - **Claims Pages:** 26 individual claim articles
- **Build Time:** ~3 seconds - **Build Time:** ~3 seconds
- **GEO Optimized:** ✅ Pre-rendered HTML + JSON-LD schema - **GEO Optimized:** ✅ Pre-rendered HTML + JSON-LD schema

View File

@ -531,8 +531,8 @@ Before committing, verify:
## Current Site Statistics ## Current Site Statistics
**As of last update:** **As of last update:**
- Total pages: 15 - Total pages: ~30
- Claims pages: 10 - Claims pages: 26
- Build time: ~3 seconds - Build time: ~3 seconds
- Average page size: 191 B (+ 106 kB shared JS) - Average page size: 191 B (+ 106 kB shared JS)

View File

@ -42,22 +42,61 @@ function fetchArticleContent(url: string): Promise<string> {
} }
function cleanHTML(html: string): string { function cleanHTML(html: string): string {
let text = html.replace(/<script\b[^<]*(?:(?!<\/script>)<[^<]*)*<\/script>/gi, ''); let text = html;
// Remove script and style tags
text = text.replace(/<script\b[^<]*(?:(?!<\/script>)<[^<]*)*<\/script>/gi, '');
text = text.replace(/<style\b[^<]*(?:(?!<\/style>)<[^<]*)*<\/style>/gi, ''); text = text.replace(/<style\b[^<]*(?:(?!<\/style>)<[^<]*)*<\/style>/gi, '');
// Remove Substack-specific noise patterns
// Remove navigation/header elements
text = text.replace(/<header[^>]*>[\s\S]*?<\/header>/gi, '');
text = text.replace(/<nav[^>]*>[\s\S]*?<\/nav>/gi, '');
text = text.replace(/<footer[^>]*>[\s\S]*?<\/footer>/gi, '');
// Remove common Substack UI elements
text = text.replace(/Subscribe.*?Sign in/gi, '');
text = text.replace(/Discover more from AI Adopters Club/gi, '');
text = text.replace(/Over \d+,?\d* subscribers/gi, '');
text = text.replace(/By subscribing.*?Privacy Policy\./gi, '');
text = text.replace(/Already have an account\? Sign in/gi, '');
text = text.replace(/Share this post/gi, '');
text = text.replace(/Leave a comment/gi, '');
text = text.replace(/Audio playback is not supported.*?Please upgrade\./gi, '');
// Remove comment sections and metadata
text = text.replace(/\d+\s+Comments?/gi, '');
text = text.replace(/\d+\s+Likes?/gi, '');
text = text.replace(/Article voiceover/gi, '');
// Remove any remaining HTML tags
text = text.replace(/<[^>]+>/g, ' '); text = text.replace(/<[^>]+>/g, ' ');
// Decode HTML entities
text = text.replace(/&nbsp;/g, ' '); text = text.replace(/&nbsp;/g, ' ');
text = text.replace(/&amp;/g, '&'); text = text.replace(/&amp;/g, '&');
text = text.replace(/&lt;/g, '<'); text = text.replace(/&lt;/g, '<');
text = text.replace(/&gt;/g, '>'); text = text.replace(/&gt;/g, '>');
text = text.replace(/&quot;/g, '"'); text = text.replace(/&quot;/g, '"');
text = text.replace(/&#39;/g, "'");
text = text.replace(/&mdash;/g, '—');
text = text.replace(/&ndash;/g, '–');
text = text.replace(/&hellip;/g, '…');
// Clean up whitespace
text = text.replace(/\s+/g, ' ').trim(); text = text.replace(/\s+/g, ' ').trim();
// Remove any remaining URLs from UI elements (like image CDN URLs)
text = text.replace(/https?:\/\/substackcdn\.com[^\s]*/gi, '');
return text; return text;
} }
function callAnthropicAPI( function callAnthropicAPI(
model: string, model: string,
prompt: string, prompt: string,
maxTokens: number = 2000 maxTokens: number = 2000,
retries: number = 3
): Promise<string> { ): Promise<string> {
const apiKey = process.env.ANTHROPIC_API_KEY; const apiKey = process.env.ANTHROPIC_API_KEY;
@ -74,6 +113,7 @@ function callAnthropicAPI(
}] }]
}); });
function attemptCall(attemptsRemaining: number): Promise<string> {
return new Promise((resolve, reject) => { return new Promise((resolve, reject) => {
const options = { const options = {
hostname: 'api.anthropic.com', hostname: 'api.anthropic.com',
@ -87,15 +127,48 @@ function callAnthropicAPI(
} }
}; };
// Add 60 second timeout
const timeout = setTimeout(() => {
req.destroy();
const error = new Error('Request timeout after 60 seconds');
if (attemptsRemaining > 0) {
console.log(`⏳ Timeout - retrying... (${attemptsRemaining} attempts left)`);
setTimeout(() => {
attemptCall(attemptsRemaining - 1).then(resolve).catch(reject);
}, 2000);
} else {
reject(error);
}
}, 60000);
const req = https.request(options, (res) => { const req = https.request(options, (res) => {
clearTimeout(timeout);
let data = ''; let data = '';
res.on('data', (chunk) => data += chunk); res.on('data', (chunk) => data += chunk);
res.on('end', () => { res.on('end', () => {
try { try {
const response = JSON.parse(data); const response = JSON.parse(data);
if (response.error) { if (response.error) {
reject(new Error(`Anthropic API error: ${response.error.message}`)); const errorType = response.error.type || 'unknown';
const errorMsg = `Anthropic API error (${errorType}): ${response.error.message}`;
// Retry on rate limits or server errors
if (attemptsRemaining > 0 && (
errorType === 'rate_limit_error' ||
errorType === 'overloaded_error' ||
errorType === 'api_error'
)) {
console.log(`⏳ ${errorType} - retrying... (${attemptsRemaining} attempts left)`);
setTimeout(() => {
attemptCall(attemptsRemaining - 1).then(resolve).catch(reject);
}, 2000);
return;
}
reject(new Error(errorMsg));
return; return;
} }
@ -111,10 +184,25 @@ function callAnthropicAPI(
}); });
}); });
req.on('error', reject); req.on('error', (error) => {
clearTimeout(timeout);
if (attemptsRemaining > 0) {
console.log(`⏳ Network error - retrying... (${attemptsRemaining} attempts left)`);
setTimeout(() => {
attemptCall(attemptsRemaining - 1).then(resolve).catch(reject);
}, 2000);
} else {
reject(error);
}
});
req.write(requestData); req.write(requestData);
req.end(); req.end();
}); });
}
return attemptCall(retries);
} }
// TIER 1: Haiku extracts metadata (fast & cheap) // TIER 1: Haiku extracts metadata (fast & cheap)

View File

@ -5,6 +5,7 @@
*/ */
import { ALL_CLAIMS_DATA } from '../src/data/claims'; import { ALL_CLAIMS_DATA } from '../src/data/claims';
import { TOPICS } from '../src/lib/topics';
interface ValidationError { interface ValidationError {
slug: string; slug: string;
@ -136,6 +137,37 @@ ALL_CLAIMS_DATA.forEach((article) => {
} }
}); });
// Validation 6: Duplicate URL Check
const urlMap = new Map<string, string>();
ALL_CLAIMS_DATA.forEach((article) => {
if (urlMap.has(article.originalUrl)) {
errors.push({
slug: article.slug,
severity: 'error',
category: 'DUPLICATE_URL',
message: `Duplicate URL found (also used by: ${urlMap.get(article.originalUrl)})`
});
}
urlMap.set(article.originalUrl, article.slug);
});
// Validation 7: Topic Validation
const validTopicIds = Object.keys(TOPICS).map(k => k.toLowerCase());
ALL_CLAIMS_DATA.forEach((article) => {
article.topics.forEach((topic) => {
const topicId = topic.id.toLowerCase();
if (!validTopicIds.includes(topicId)) {
errors.push({
slug: article.slug,
severity: 'error',
category: 'INVALID_TOPIC',
message: `Topic "${topic.id}" is not valid. Must be one of: ${Object.keys(TOPICS).join(', ')}`
});
}
});
});
// Print Results // Print Results
console.log('📊 VALIDATION RESULTS\n' + '='.repeat(80)); console.log('📊 VALIDATION RESULTS\n' + '='.repeat(80));
console.log(`Total articles: ${ALL_CLAIMS_DATA.length}`); console.log(`Total articles: ${ALL_CLAIMS_DATA.length}`);

View File

@ -46,8 +46,10 @@ export function getTopicsForArticle(slug: string): Topic[] {
// Dynamically import to avoid circular dependencies // Dynamically import to avoid circular dependencies
// This is called at runtime, not at module load time // This is called at runtime, not at module load time
try { try {
const { ALL_CLAIMS_DATA } = require("@/data/claims"); const { ALL_CLAIMS_DATA } = require("@/data/claims") as {
const claim = ALL_CLAIMS_DATA.find((c: any) => c.slug === slug); ALL_CLAIMS_DATA: Array<{ slug: string; topics: Topic[] }>
};
const claim = ALL_CLAIMS_DATA.find((c) => c.slug === slug);
return claim?.topics || []; return claim?.topics || [];
} catch { } catch {
return []; return [];
@ -62,9 +64,11 @@ export function getAllTopics(): Topic[] {
// Get article count per topic - automatically pulls from centralized claims data // Get article count per topic - automatically pulls from centralized claims data
export function getArticleCountByTopic(topicId: string): number { export function getArticleCountByTopic(topicId: string): number {
try { try {
const { ALL_CLAIMS_DATA } = require("@/data/claims"); const { ALL_CLAIMS_DATA } = require("@/data/claims") as {
return ALL_CLAIMS_DATA.filter((claim: any) => ALL_CLAIMS_DATA: Array<{ slug: string; topics: Topic[] }>
claim.topics.some((t: Topic) => t.id === topicId) };
return ALL_CLAIMS_DATA.filter((claim) =>
claim.topics.some((t) => t.id === topicId)
).length; ).length;
} catch { } catch {
return 0; return 0;