Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
19 changes: 19 additions & 0 deletions tests/test_basic.py
Original file line number Diff line number Diff line change
Expand Up @@ -823,3 +823,22 @@ def test_sitemap_includes_compare():
print(f"\n{passed} passed, {failed} failed out of {passed + failed} tests")
if failed > 0:
sys.exit(1)

def test_ml_similarity_score_returns_float():
from utils.recommender import ml_similarity_score, parse_skills
projects = load_all_projects()
score = ml_similarity_score(
projects[0],
parse_skills("Python"),
"Beginner",
"Data",
"Low",
projects,
)
assert isinstance(score, float)
assert score >= 0

def test_ml_recommendation_prefers_relevant_python_data_project():
results = get_recommendations("Python, pandas", "Intermediate", "Data", "High")
titles = [project["title"] for project in results]
assert any("Data" in title or "Pipeline" in title for title in titles)
240 changes: 95 additions & 145 deletions utils/recommender.py
Original file line number Diff line number Diff line change
@@ -1,105 +1,116 @@
# utils/recommender.py
# Contains all recommendation logic: scoring, filtering, and related projects.
# Kept separate from routing so it can be tested and extended independently.
# Contains all recommendation logic: scoring and filtering projects.

import math
import re
from collections import Counter

import json
import os

from utils.data_loader import load_all_projects

# Maximum number of recommendations returned to the user
MAX_RESULTS = 3

# Maximum number of "you might also like" projects returned alongside results
MAX_RELATED = 3

# Scoring weights used by the recommendation engine.
# Higher weights mean that criterion has more influence
# on the final recommendation score.
SCORING_WEIGHTS = {
"skill": 3,
"level": 2,
"skill": 3,
"level": 2,
"interest": 2,
"time": 1,
"time": 1,
}

# Common aliases and abbreviations for skills.
# This improves recommendation accuracy by normalizing user input.
SKILL_ALIASES = {
"js": "javascript",
"py": "python",
"html5": "html",
"css3": "css",
"c++": "cpp",
"js": "javascript",
"py": "python",
"html5": "html",
"css3": "css",
"c++": "cpp",
"web dev": "javascript",
}

# Path to the precomputed cluster assignments.
# Generated by: python scripts/cluster_projects.py
_CLUSTERS_PATH = os.path.join(
os.path.dirname(os.path.dirname(os.path.abspath(__file__))),
"data",
"clusters.json",
)


# ---------------------------------------------------------------------------
# Skill parsing
# ---------------------------------------------------------------------------

def parse_skills(skills_string):
"""
Convert a raw comma-separated skills string into
a normalized lowercase list.

Example:
"JS, HTML5, CSS3" -> ["javascript", "html", "css"]
"""
raw_skills = [
s.strip().lower()
for s in skills_string.split(",")
if s.strip()
]
return [SKILL_ALIASES.get(skill, skill) for skill in raw_skills]

def _tokenize(text):
return re.findall(r"[a-z0-9]+", str(text).lower())

def _project_text(project):
parts = [
project.get("title", ""),
project.get("level", ""),
project.get("interest", ""),
project.get("time", ""),
project.get("description", ""),
" ".join(project.get("skills", [])),
" ".join(project.get("tech_stack", [])),
" ".join(project.get("features", [])),
]
return " ".join(parts)

def _user_text(user_skills, level, interest, time_availability):
return " ".join(user_skills + [level, interest, time_availability])

# ---------------------------------------------------------------------------
# Scoring
# ---------------------------------------------------------------------------
def _tf(tokens):
counts = Counter(tokens)
total = len(tokens) or 1
return {token: count / total for token, count in counts.items()}

def score_single_project(project, user_skills, level, interest, time_availability):
"""
Calculate a numeric relevance score for one project.
def _idf(documents):
total_docs = len(documents)
idf_scores = {}

Scoring rules:
- Each matching skill: +3
- Level match: +2
- Interest match: +2
- Time match: +1
all_tokens = set(token for doc in documents for token in set(doc))

Time filtering: projects that require MORE time than the user has
available are excluded entirely (score returned as 0).
for token in all_tokens:
docs_with_token = sum(1 for doc in documents if token in doc)
idf_scores[token] = math.log((1 + total_docs) / (1 + docs_with_token)) + 1

Returns an integer score (0 means no match or time mismatch).
"""
TIME_RANKS = ["low", "medium", "high"]
return idf_scores

user_time = time_availability.strip().lower()
project_time = project.get("time", "").strip().lower()
def _tfidf_vector(tokens, idf_scores):
tf_scores = _tf(tokens)
return {
token: tf_scores[token] * idf_scores.get(token, 0)
for token in tf_scores
}

# If the project needs more time than the user has, exclude it.
if project_time not in TIME_RANKS or user_time not in TIME_RANKS:
return 0
if TIME_RANKS.index(project_time) > TIME_RANKS.index(user_time):
def _cosine_similarity(vec_a, vec_b):
shared_tokens = set(vec_a) & set(vec_b)

dot_product = sum(vec_a[token] * vec_b[token] for token in shared_tokens)
magnitude_a = math.sqrt(sum(value ** 2 for value in vec_a.values()))
magnitude_b = math.sqrt(sum(value ** 2 for value in vec_b.values()))

if magnitude_a == 0 or magnitude_b == 0:
return 0

return dot_product / (magnitude_a * magnitude_b)

def ml_similarity_score(project, user_skills, level, interest, time_availability, all_projects):
project_documents = [_tokenize(_project_text(p)) for p in all_projects]
user_tokens = _tokenize(_user_text(user_skills, level, interest, time_availability))

idf_scores = _idf(project_documents + [user_tokens])

user_vector = _tfidf_vector(user_tokens, idf_scores)
project_vector = _tfidf_vector(_tokenize(_project_text(project)), idf_scores)

return _cosine_similarity(user_vector, project_vector)

def score_single_project(project, user_skills, level, interest, time_availability):
score = 0

# Compare user's skills against the project's required skills
project_skills = [SKILL_ALIASES.get(s.lower(), s.lower()) for s in project.get("skills", [])]
# Count how many user skills overlap with the
# skills required by the current project.
matched_skills = sum(1 for skill in user_skills if skill in project_skills)

score += matched_skills * SCORING_WEIGHTS["skill"]

if project.get("level", "").lower() == level.lower():
Expand All @@ -113,105 +124,44 @@ def score_single_project(project, user_skills, level, interest, time_availabilit

return score


# ---------------------------------------------------------------------------
# Clustering helpers
# ---------------------------------------------------------------------------

def _load_clusters():
"""
Load clusters.json if it exists.

Returns the parsed dict, or None if the file is missing or unreadable.
A missing file is a soft failure — the recommender still works,
it just won't return related projects.
"""
if not os.path.exists(_CLUSTERS_PATH):
return None
try:
with open(_CLUSTERS_PATH, "r", encoding="utf-8") as f:
return json.load(f)
except (json.JSONDecodeError, OSError):
return None


def _get_related(recommended_ids, all_projects, cluster_data):
"""
Find projects in the same cluster(s) as the recommended projects,
excluding the ones already recommended.

Returns up to MAX_RELATED project dicts.
"""
clusters = cluster_data.get("clusters", {}) # {str(pid): cid}
members = cluster_data.get("members", {}) # {str(cid): [pid, ...]}

# Collect which clusters the recommended projects belong to.
relevant_cluster_ids = set()
for pid in recommended_ids:
cid = clusters.get(str(pid))
if cid is not None:
relevant_cluster_ids.add(str(cid))

if not relevant_cluster_ids:
return []

# Gather candidate IDs from those clusters, excluding already recommended.
candidate_ids = []
for cid in relevant_cluster_ids:
for pid in members.get(cid, []):
if pid not in recommended_ids and pid not in candidate_ids:
candidate_ids.append(pid)

id_to_project = {p["id"]: p for p in all_projects}
related = [id_to_project[pid] for pid in candidate_ids if pid in id_to_project]
return related[:MAX_RELATED]


# ---------------------------------------------------------------------------
# Public API
# ---------------------------------------------------------------------------

def get_recommendations(skills_string, level, interest, time_availability):
"""
Return the top N recommended projects for the given user inputs,
along with related projects from the same cluster.

Return shape:
{
"recommendations": [ <project>, ... ], # up to MAX_RESULTS
"related": [ <project>, ... ], # up to MAX_RELATED
}

The "related" list is empty when clusters.json does not exist yet.
Run scripts/cluster_projects.py to generate it.
"""
user_skills = parse_skills(skills_string)
user_skills = parse_skills(skills_string)
all_projects = load_all_projects()

scored = []
for project in all_projects:
score = score_single_project(
project, user_skills, level, interest, time_availability
rule_score = score_single_project(
project,
user_skills,
level,
interest,
time_availability,
)
if score >= SCORING_WEIGHTS["skill"]:
scored.append({"project": project, "score": score})

scored.sort(key=lambda item: item["score"], reverse=True)
top_projects = [item["project"] for item in scored[:MAX_RESULTS]]
top_ids = [p["id"] for p in top_projects]

cluster_data = _load_clusters()
related = _get_related(top_ids, all_projects, cluster_data) if cluster_data else []
similarity_score = ml_similarity_score(
project,
user_skills,
level,
interest,
time_availability,
all_projects,
)

return {
"recommendations": top_projects,
"related": related,
}
final_score = rule_score + similarity_score

if final_score > 0:
scored_projects.append({
"project": project,
"score": final_score,
})

VALID_LEVELS = ["beginner", "intermediate", "advanced"]
VALID_TIME_AVAILABILITY = ["low", "medium", "high"]
scored_projects.sort(key=lambda item: item["score"], reverse=True)

return [item["project"] for item in scored_projects[:MAX_RESULTS]]

def validate_recommendation_inputs(skills, level, interest, time_availability):
errors = []
Expand Down
Loading