#!/usr/bin/env bash set -euo pipefail SCRIPT_DIR="$(cd "$(dirname "$0")" && pwd)" PROJECT_ROOT="$(cd "$SCRIPT_DIR/.." && pwd)" KNOWLEDGE_ROOT="$PROJECT_ROOT/knowledge/entries" OUTPUT="$PROJECT_ROOT/knowledge-index.html" if [ ! -d "$KNOWLEDGE_ROOT" ]; then KNOWLEDGE_ROOT="$PROJECT_ROOT" fi # ---- Phase 1: Scan directories ---- echo "[1/4] Scanning knowledge directories..." dirs=() while IFS= read -r d; do dirs+=("${d%/}") done < <(ls -d "$KNOWLEDGE_ROOT"/knowledge_*/ 2>/dev/null || true) if [ ${#dirs[@]} -eq 0 ]; then echo " No knowledge_*/ directories found in $KNOWLEDGE_ROOT" echo " Generated empty index (will show 'no entries' message)." fi n=${#dirs[@]}; echo " Found $n knowledge director$([ "$n" -ne 1 ] && echo 'ies' || echo 'y')." # ---- Phase 2: Extract metadata ---- echo "[2/4] Extracting metadata..." entries_js="" count=0 for dir in "${dirs[@]}"; do dirname=$(basename "$dir") # Parse directory name: knowledge_YYYYMMDD_Slug date_part=$(echo "$dirname" | sed -E 's/^knowledge_([0-9]{8})_.*/\1/') slug=$(echo "$dirname" | sed -E 's/^knowledge_[0-9]{8}_//') if [ -z "$date_part" ] || [ "$date_part" = "$dirname" ]; then echo " WARNING: '$dirname' doesn't match knowledge_YYYYMMDD_Slug pattern, skipping..." continue fi # Find .md file md_file="$dir/${dirname}.md" if [ ! -f "$md_file" ]; then # Try to find any .md file in the directory md_file=$(ls "$dir"/*.md 2>/dev/null | head -1) if [ -z "$md_file" ]; then echo " WARNING: No .md file found in '$dirname', skipping..." continue fi fi # Extract YAML frontmatter (between first and second ---) yaml=$(sed -n '/^---$/,/^---$/p' "$md_file" | sed '1d;$d' 2>/dev/null || echo "") # Extract title title=$(echo "$yaml" | grep -i "^title:" | head -1 | sed 's/^[Tt]itle: *//' | sed 's/^"//;s/"$//;s/^'"'"'//;s/'"'"'$//' || echo "") # Extract author author=$(echo "$yaml" | grep -i "^author:" | head -1 | sed 's/^[Aa]uthor: *//' | sed 's/^"//;s/"$//' || echo "") # Extract tags: handle both ["a","b"] and [a,b] formats tags_raw=$(echo "$yaml" | grep -i "^tags:" | head -1 | sed 's/^[Tt]ags: *//' || echo "") # Normalize: remove brackets, split by comma, strip quotes/spaces tags_csv=$(echo "$tags_raw" | sed 's/^\[//;s/\]$//' | sed 's/"//g' | sed "s/'//g" | tr ',' '\n' | sed 's/^[[:space:]]*//;s/[[:space:]]*$//' | grep -v '^$' | paste -sd ',' - 2>/dev/null || echo "") # Format tags as JS array if [ -n "$tags_csv" ]; then tags_js="" IFS=',' read -ra TAG_ARR <<< "$tags_csv" for t in "${TAG_ARR[@]}"; do t_trimmed=$(echo "$t" | sed 's/^[[:space:]]*//;s/[[:space:]]*$//') if [ -n "$t_trimmed" ]; then [ -n "$tags_js" ] && tags_js+=", " # Escape single quotes in tag t_escaped=$(echo "$t_trimmed" | sed "s/'/\\\\'/g") tags_js+="'$t_escaped'" fi done else tags_js="" fi # Fallback title if [ -z "$title" ]; then title=$(echo "$slug" | sed 's/_/ /g') fi [ -z "$author" ] && author="unknown" # Extract body text (everything after the second ---) body=$(awk 'BEGIN{found=0} /^---$/{found++; next} found>=2{print}' "$md_file" 2>/dev/null || echo "") # Escape for JS string: backslash, single quote, newline title_esc=$(echo "$title" | sed "s/\\\\/\\\\\\\\/g; s/'/\\\\'/g") author_esc=$(echo "$author" | sed "s/\\\\/\\\\\\\\/g; s/'/\\\\'/g") body_esc=$(echo "$body" | sed "s/\\\\/\\\\\\\\/g; s/'/\\\\'/g" | tr '\n' ' ' | sed 's/ */ /g') slug_esc=$(echo "$slug" | sed "s/\\\\/\\\\\\\\/g; s/'/\\\\'/g") rel_dir=$(realpath --relative-to="$PROJECT_ROOT" "$dir" 2>/dev/null || echo "$dirname") rel_dir=${rel_dir//\\//} rel_dir_esc=$(echo "$rel_dir" | sed "s/\\\\/\\\\\\\\/g; s/'/\\\\'/g") # Build JS entry entries_js+=" {" entries_js+=" slug: '$slug_esc'," entries_js+=" path: '$rel_dir_esc'," entries_js+=" date: '$date_part'," entries_js+=" title: '$title_esc'," entries_js+=" author: '$author_esc'," entries_js+=" tags: [$tags_js]," entries_js+=" body: '$body_esc'" entries_js+=" }," entries_js+=$'\n' count=$((count + 1)) echo " ✓ $dirname" done # ---- Phase 3: Generate HTML ---- echo "[3/4] Generating knowledge-index.html..." entries_js=$(echo "$entries_js" | sed '$ s/,$//') # Write HTML header (before __DATA_PLACEHOLDER__) cat > "$OUTPUT" << 'HTMLEOF'