feat(curation): data quality pipeline — Phases 1-3

Add comprehensive data curation system to clean up the 197K skill
dataset and show only quality browse-ready skills to users.

Phase 1 — Database exploration:
- Explore scripts (explore.ts, explore.mjs, explore.sql) for analysis
- Discovered: 69% duplicates, 77% aggregator/fork noise

Phase 2 — Data cleanup and classification:
- Schema: 6 new curation columns + 4 indexes
- curate.mjs: 8-step pipeline (classify, dedup, fork detection, etc.)
- Result: 197K → 60K unique → 16K browse-ready skills
- Bug fix: securityStatus was computed but never stored during crawl

Phase 3 — UI browse-ready filters:
- browseReadyFilter applied to 17+ query functions
- Homepage stats show accurate browse-ready counts
- Stats API filtered (previously had no WHERE clause)
- Category counts recalculated (e.g. 45K → 3.1K)
- Featured skills exclude duplicates and aggregators

Co-Authored-By: Claude Opus 4.6 <noreply@anthropic.com>
This commit is contained in:
airano
2026-02-19 14:31:25 +03:30
parent 31f5df0900
commit caca09fbe7
12 changed files with 2615 additions and 28 deletions

505
scripts/curation/curate.mjs Normal file
View File

@@ -0,0 +1,505 @@
#!/usr/bin/env node
/**
* Phase 2: Data Cleanup & Classification
*
* Runs all curation steps in order:
* Step 1: Compute repo_skill_count for every skill
* Step 2: Mark aggregators (repos with 50+ skills)
* Step 3: Classify remaining repos (collection / standalone / project-bound)
* Step 4: Fill missing content_hash values
* Step 5: Mark duplicates by content_hash (canonical = highest stars or oldest)
* Step 6: Detect fork marketplace repos
* Step 7: Recalculate category skill counts (browse-ready only)
* Step 8: Summary report
*
* Usage:
* DATABASE_URL=postgres://... node scripts/curation/curate.mjs
* # Or with local Docker:
* DATABASE_URL=postgresql://postgres:postgres@localhost:5432/skillhub node scripts/curation/curate.mjs
*
* Options:
* --dry-run Show what would change without writing
* --step=N Run only step N (1-8)
*/
import { createRequire } from 'module';
import { resolve, dirname } from 'path';
import { fileURLToPath } from 'url';
const __dirname = dirname(fileURLToPath(import.meta.url));
// ─── Find pg module ───
let pg;
const tryPaths = [
resolve(__dirname, '../..', 'package.json'),
'/tmp/package.json',
process.env.APPDATA ? resolve(process.env.APPDATA, '..', 'Local', 'Temp', 'package.json') : null,
].filter(Boolean);
for (const p of tryPaths) {
try { pg = createRequire(p)('pg'); break; } catch {}
}
if (!pg) { console.error('pg not found. Run: npm install pg'); process.exit(1); }
// ─── Config ───
const DATABASE_URL = process.env.DATABASE_URL || 'postgresql://postgres:postgres@localhost:5432/skillhub';
const DRY_RUN = process.argv.includes('--dry-run');
const stepArg = process.argv.find(a => a.startsWith('--step='));
const ONLY_STEP = stepArg ? parseInt(stepArg.split('=')[1]) : null;
// ─── Helpers ───
function header(step, title) {
console.log(`\n${'='.repeat(70)}\n STEP ${step}: ${title}\n${'='.repeat(70)}`);
}
const n = v => Number(v ?? 0).toLocaleString('en-US');
// ─── Database ───
let client;
async function connect() {
client = new pg.Client({
connectionString: DATABASE_URL,
ssl: false,
connectionTimeoutMillis: 15000,
query_timeout: 600000, // 10 min for big updates
keepAlive: true,
});
await client.connect();
console.log('Connected to database');
}
async function query(sql, params = []) {
const result = await client.query(sql, params);
return result;
}
// ─── Step 1: Compute repo_skill_count ───
async function step1_repoSkillCount() {
header(1, 'COMPUTE repo_skill_count');
const countResult = await query(`
SELECT COUNT(*)::int AS total
FROM skills
WHERE is_blocked = false AND repo_skill_count IS NULL
`);
console.log(` Skills without repo_skill_count: ${n(countResult.rows[0].total)}`);
if (DRY_RUN) { console.log(' [DRY RUN] Would update all skills'); return; }
const result = await query(`
UPDATE skills s SET repo_skill_count = sub.cnt
FROM (
SELECT github_owner, github_repo, COUNT(*)::int AS cnt
FROM skills WHERE is_blocked = false
GROUP BY github_owner, github_repo
) sub
WHERE s.github_owner = sub.github_owner
AND s.github_repo = sub.github_repo
AND s.is_blocked = false
`);
console.log(` Updated: ${n(result.rowCount)} skills`);
}
// ─── Step 2: Mark aggregators ───
async function step2_markAggregators() {
header(2, 'MARK AGGREGATORS (repos with 50+ skills)');
// Preview
const preview = await query(`
SELECT github_owner || '/' || github_repo AS repo, COUNT(*)::int AS skills,
MAX(github_stars)::int AS stars
FROM skills WHERE is_blocked = false
GROUP BY github_owner, github_repo
HAVING COUNT(*) >= 50
ORDER BY skills DESC LIMIT 20
`);
console.log(` Top aggregator repos (${preview.rowCount} total):`);
for (const r of preview.rows.slice(0, 10)) {
console.log(` ${r.repo}: ${n(r.skills)} skills (${n(r.stars)} stars)`);
}
if (DRY_RUN) { console.log(' [DRY RUN] Would mark skills in these repos'); return; }
const result = await query(`
UPDATE skills s SET skill_type = 'aggregator'
FROM (
SELECT github_owner, github_repo
FROM skills WHERE is_blocked = false
GROUP BY github_owner, github_repo
HAVING COUNT(*) >= 50
) agg
WHERE s.github_owner = agg.github_owner
AND s.github_repo = agg.github_repo
AND s.is_blocked = false
AND s.skill_type IS NULL
`);
console.log(` Marked as aggregator: ${n(result.rowCount)} skills`);
}
// ─── Step 3: Classify remaining repos ───
async function step3_classifyRemaining() {
header(3, 'CLASSIFY REMAINING REPOS');
// 3a: repos with 10-49 skills, name contains marketplace/awesome/collection → aggregator
if (!DRY_RUN) {
const aggByName = await query(`
UPDATE skills s SET skill_type = 'aggregator'
FROM (
SELECT github_owner, github_repo
FROM skills WHERE is_blocked = false AND skill_type IS NULL
GROUP BY github_owner, github_repo
HAVING COUNT(*) >= 10
AND (github_repo ILIKE '%marketplace%'
OR github_repo ILIKE '%awesome%'
OR github_repo ILIKE '%collection%'
OR github_repo ILIKE '%registry%')
) agg
WHERE s.github_owner = agg.github_owner
AND s.github_repo = agg.github_repo
AND s.is_blocked = false
AND s.skill_type IS NULL
`);
console.log(` Aggregator by name (10+ & marketplace/awesome/registry): ${n(aggByName.rowCount)}`);
}
// 3b: repos with 3-49 skills → collection
if (!DRY_RUN) {
const collections = await query(`
UPDATE skills s SET skill_type = 'collection'
FROM (
SELECT github_owner, github_repo
FROM skills WHERE is_blocked = false AND skill_type IS NULL
GROUP BY github_owner, github_repo
HAVING COUNT(*) >= 3
) col
WHERE s.github_owner = col.github_owner
AND s.github_repo = col.github_repo
AND s.is_blocked = false
AND s.skill_type IS NULL
`);
console.log(` Collection (3+ skills in repo): ${n(collections.rowCount)}`);
}
// 3c: single/two-skill repos with project-bound name patterns → project-bound
if (!DRY_RUN) {
const projectBound = await query(`
UPDATE skills SET skill_type = 'project-bound'
WHERE is_blocked = false AND skill_type IS NULL
AND repo_skill_count <= 2
AND (name ILIKE '%my-%' OR name ILIKE '%my\\_%' ESCAPE '\\'
OR name ILIKE '%project%' OR name ILIKE '%team%' OR name ILIKE '%internal%'
OR name ILIKE '%.mdc' OR name ILIKE '%cursorrule%'
OR name ILIKE '%config%' OR name ILIKE '%setup%')
`);
console.log(` Project-bound (name pattern): ${n(projectBound.rowCount)}`);
}
// 3d: remaining → standalone
if (!DRY_RUN) {
const standalone = await query(`
UPDATE skills SET skill_type = 'standalone'
WHERE is_blocked = false AND skill_type IS NULL
`);
console.log(` Standalone (remaining): ${n(standalone.rowCount)}`);
}
// Summary
const summary = await query(`
SELECT skill_type, COUNT(*)::int AS count
FROM skills WHERE is_blocked = false
GROUP BY skill_type ORDER BY count DESC
`);
console.log('\n Classification summary:');
for (const r of summary.rows) {
console.log(` ${(r.skill_type || 'null').padEnd(15)} ${n(r.count)}`);
}
}
// ─── Step 4: Fill missing content_hash ───
async function step4_fillContentHash() {
header(4, 'FILL MISSING content_hash');
const countResult = await query(`
SELECT COUNT(*)::int AS total
FROM skills WHERE is_blocked = false AND content_hash IS NULL AND raw_content IS NOT NULL
`);
const missing = countResult.rows[0].total;
console.log(` Skills missing content_hash: ${n(missing)}`);
if (missing === 0) { console.log(' Nothing to do'); return; }
if (DRY_RUN) { console.log(' [DRY RUN] Would compute hash for these skills'); return; }
// Use md5 as a reliable hash function available in PostgreSQL
const result = await query(`
UPDATE skills SET content_hash = md5(raw_content)
WHERE is_blocked = false AND content_hash IS NULL AND raw_content IS NOT NULL
`);
console.log(` Computed content_hash (md5) for: ${n(result.rowCount)} skills`);
// Note: existing hashes use a JS numeric hash (analyzer.ts hashContent).
// We now use md5 for missing ones. For dedup purposes, same content → same hash
// regardless of algorithm, since we compare within hash groups.
// But mixed algorithms mean same content could have different hashes.
// Solution: re-hash ALL with md5 for consistency.
console.log(' Re-hashing ALL skills with md5 for consistency...');
const rehash = await query(`
UPDATE skills SET content_hash = md5(raw_content)
WHERE is_blocked = false AND raw_content IS NOT NULL
`);
console.log(` Re-hashed: ${n(rehash.rowCount)} skills`);
}
// ─── Step 5: Mark duplicates ───
async function step5_markDuplicates() {
header(5, 'MARK DUPLICATES BY content_hash');
// Find duplicate groups
const dupGroups = await query(`
SELECT content_hash, COUNT(*)::int AS copies
FROM skills
WHERE is_blocked = false AND content_hash IS NOT NULL AND is_duplicate = false
GROUP BY content_hash
HAVING COUNT(*) > 1
ORDER BY copies DESC
`);
const totalDupGroups = dupGroups.rowCount;
const totalDupSkills = dupGroups.rows.reduce((sum, r) => sum + r.copies, 0);
const removable = dupGroups.rows.reduce((sum, r) => sum + r.copies - 1, 0);
console.log(` Duplicate groups: ${n(totalDupGroups)}`);
console.log(` Total skills in dup groups: ${n(totalDupSkills)}`);
console.log(` Removable (keeping 1 per group): ${n(removable)}`);
console.log(` Top 5 by copies:`);
for (const r of dupGroups.rows.slice(0, 5)) {
console.log(` hash=${r.content_hash}: ${r.copies} copies`);
}
if (DRY_RUN) { console.log(' [DRY RUN] Would mark duplicates'); return; }
// For each duplicate group:
// - canonical = highest github_stars, then oldest created_at
// - rest = is_duplicate = true, canonical_skill_id = canonical.id
const markResult = await query(`
WITH ranked AS (
SELECT id, content_hash,
ROW_NUMBER() OVER (
PARTITION BY content_hash
ORDER BY github_stars DESC NULLS LAST,
created_at ASC
) AS rn
FROM skills
WHERE is_blocked = false AND content_hash IS NOT NULL AND is_duplicate = false
),
canonicals AS (
SELECT content_hash, id AS canonical_id
FROM ranked WHERE rn = 1
)
UPDATE skills s
SET is_duplicate = true,
canonical_skill_id = c.canonical_id
FROM ranked r
JOIN canonicals c ON r.content_hash = c.content_hash
WHERE s.id = r.id
AND r.rn > 1
`);
console.log(` Marked as duplicate: ${n(markResult.rowCount)} skills`);
// Verify
const verify = await query(`
SELECT
COUNT(*) FILTER (WHERE is_duplicate = false)::int AS unique_skills,
COUNT(*) FILTER (WHERE is_duplicate = true)::int AS duplicates
FROM skills WHERE is_blocked = false
`);
console.log(` After dedup: ${n(verify.rows[0].unique_skills)} unique, ${n(verify.rows[0].duplicates)} duplicates`);
}
// ─── Step 6: Detect fork marketplace repos ───
async function step6_detectForks() {
header(6, 'DETECT FORK MARKETPLACE REPOS');
// Known fork patterns: repos with same name across many owners
const forkPatterns = await query(`
SELECT github_repo, COUNT(DISTINCT github_owner)::int AS owners,
COUNT(*)::int AS total_skills
FROM skills
WHERE is_blocked = false
AND repo_skill_count >= 20
GROUP BY github_repo
HAVING COUNT(DISTINCT github_owner) >= 3
ORDER BY owners DESC
`);
console.log(` Repo names appearing in 3+ owners (with 20+ skills each):`);
for (const r of forkPatterns.rows) {
console.log(` ${r.github_repo}: ${r.owners} owners, ${n(r.total_skills)} skills`);
}
if (DRY_RUN) { console.log(' [DRY RUN] Would mark fork repo skills'); return; }
// For fork repos, all copies should be checked as duplicates
// The canonical is already handled by step 5 (content_hash dedup)
// Here we just ensure they're typed correctly as aggregator
if (forkPatterns.rows.length > 0) {
const repoNames = forkPatterns.rows.map(r => r.github_repo);
const placeholders = repoNames.map((_, i) => `$${i + 1}`).join(',');
const markForks = await query(`
UPDATE skills SET skill_type = 'aggregator'
WHERE is_blocked = false
AND github_repo IN (${placeholders})
AND repo_skill_count >= 20
AND (skill_type IS NULL OR skill_type != 'aggregator')
`, repoNames);
console.log(` Re-classified as aggregator (fork repos): ${n(markForks.rowCount)}`);
}
}
// ─── Step 7: Recalculate category counts ───
async function step7_categoryCounts() {
header(7, 'RECALCULATE CATEGORY COUNTS');
// Update skill_count on each category to only count browse-ready skills
const result = await query(`
UPDATE categories c
SET skill_count = sub.cnt
FROM (
SELECT sc.category_id, COUNT(*)::int AS cnt
FROM skill_categories sc
JOIN skills s ON sc.skill_id = s.id
WHERE s.is_blocked = false
AND s.is_duplicate = false
AND (s.skill_type IS NULL OR s.skill_type != 'aggregator')
GROUP BY sc.category_id
) sub
WHERE c.id = sub.category_id
AND c.skill_count IS DISTINCT FROM sub.cnt
`);
console.log(` Updated ${result.rowCount} category counts (browse-ready only)`);
// Also zero out categories that have no browse-ready skills
const zeroResult = await query(`
UPDATE categories c
SET skill_count = 0
WHERE c.skill_count > 0
AND NOT EXISTS (
SELECT 1 FROM skill_categories sc
JOIN skills s ON sc.skill_id = s.id
WHERE sc.category_id = c.id
AND s.is_blocked = false
AND s.is_duplicate = false
AND (s.skill_type IS NULL OR s.skill_type != 'aggregator')
)
`);
if (zeroResult.rowCount > 0) {
console.log(` Zeroed ${zeroResult.rowCount} categories with no browse-ready skills`);
}
// Show category counts
const cats = await query(`
SELECT name, skill_count FROM categories
WHERE id NOT LIKE 'parent-%'
ORDER BY skill_count DESC
LIMIT 10
`);
console.log(' Top 10 categories by browse-ready count:');
for (const r of cats.rows) {
console.log(` ${n(r.skill_count).padStart(6)} ${r.name}`);
}
}
// ─── Step 8: Summary ───
async function step8_summary() {
header(8, 'FINAL SUMMARY');
const summary = await query(`
SELECT
COUNT(*)::int AS total,
COUNT(*) FILTER (WHERE is_duplicate = false)::int AS unique_skills,
COUNT(*) FILTER (WHERE is_duplicate = true)::int AS duplicates,
COUNT(*) FILTER (WHERE skill_type = 'standalone' AND is_duplicate = false)::int AS standalone,
COUNT(*) FILTER (WHERE skill_type = 'collection' AND is_duplicate = false)::int AS collection,
COUNT(*) FILTER (WHERE skill_type = 'aggregator' AND is_duplicate = false)::int AS aggregator,
COUNT(*) FILTER (WHERE skill_type = 'project-bound' AND is_duplicate = false)::int AS project_bound,
COUNT(*) FILTER (WHERE skill_type IS NULL AND is_duplicate = false)::int AS unclassified
FROM skills WHERE is_blocked = false
`);
const s = summary.rows[0];
console.log(`
┌─────────────────────────────────────────────────────┐
│ CURATION SUMMARY │
├─────────────────────────────────────────────────────┤
│ Total skills: ${n(s.total).padStart(8)}
│ Unique skills: ${n(s.unique_skills).padStart(8)}
│ Duplicates: ${n(s.duplicates).padStart(8)}
├─────────────────────────────────────────────────────┤
│ Unique by type: │
│ Standalone: ${n(s.standalone).padStart(8)}
│ Collection: ${n(s.collection).padStart(8)}
│ Aggregator: ${n(s.aggregator).padStart(8)}
│ Project-bound: ${n(s.project_bound).padStart(8)}
│ Unclassified: ${n(s.unclassified).padStart(8)}
└─────────────────────────────────────────────────────┘
`);
// Interesting standalone skills
const topStandalone = await query(`
SELECT id, name, github_stars AS stars, COALESCE(download_count,0) AS dl,
LEFT(description, 70) AS description
FROM skills
WHERE is_blocked = false AND is_duplicate = false
AND skill_type = 'standalone'
ORDER BY github_stars DESC NULLS LAST
LIMIT 20
`);
console.log(' Top 20 standalone skills by stars:');
for (const r of topStandalone.rows) {
console.log(` [${n(r.stars)}${n(r.dl)}↓] ${r.id}`);
console.log(` ${r.description}`);
}
// Browse-ready count (what users would see)
const browseReady = await query(`
SELECT COUNT(*)::int AS count
FROM skills
WHERE is_blocked = false
AND is_duplicate = false
AND skill_type IN ('standalone', 'collection')
`);
console.log(`\n Browse-ready skills (standalone + collection, unique): ${n(browseReady.rows[0].count)}`);
}
// ─── Main ───
async function main() {
const t0 = Date.now();
await connect();
if (DRY_RUN) console.log('\n *** DRY RUN MODE — no changes will be made ***\n');
const steps = [
step1_repoSkillCount,
step2_markAggregators,
step3_classifyRemaining,
step4_fillContentHash,
step5_markDuplicates,
step6_detectForks,
step7_categoryCounts,
step8_summary,
];
for (let i = 0; i < steps.length; i++) {
if (ONLY_STEP && ONLY_STEP !== i + 1) continue;
await steps[i]();
}
const dur = ((Date.now() - t0) / 1000).toFixed(1);
console.log(`\nCompleted in ${dur}s`);
await client.end().catch(() => {});
}
main().catch(async e => {
console.error('FAILED:', e.message || e);
await client?.end().catch(() => {});
process.exit(1);
});