Created
April 17, 2026 01:45
-
-
Save gourab5139014/4462bbc474e25a80be2b9f64b43e6eed to your computer and use it in GitHub Desktop.
A Python utility for batch enriching a list of LinkedIn profiles from a CSV file. It leverages search aggregators (DuckDuckGo Search) to find professional metadata such as name, location, current focus, education, technical skills, and GitHub links, then synthesizes the findings into a structured JSON repository.
This file contains hidden or bidirectional Unicode text that may be interpreted or compiled differently than what appears below. To review, open the file in an editor that reveals hidden Unicode characters.
Learn more about bidirectional Unicode characters
| import pandas as pd | |
| import json | |
| import time | |
| from duckduckgo_search import DDGS | |
| def enrich_profiles(csv_input, json_output): | |
| """ | |
| Reads a CSV of LinkedIn URLs and enriches them with professional metadata | |
| found via search snippets, saving the result to a structured JSON. | |
| """ | |
| try: | |
| df = pd.read_csv(csv_input) | |
| # Assumes the CSV has a column named 'LinkedIn' | |
| urls = df['LinkedIn'].tolist() | |
| except Exception as e: | |
| print(f"Error reading CSV {csv_input}: {e}") | |
| return | |
| enriched_data = [] | |
| print(f"Starting enrichment for {len(urls)} profiles...") | |
| with DDGS() as ddgs: | |
| for url in urls: | |
| print(f"Processing: {url}") | |
| try: | |
| # Search for the LinkedIn profile to get snippets | |
| results = list(ddgs.text(url, max_results=3)) | |
| # Structure the metadata | |
| profile_info = { | |
| "linkedin_url": url, | |
| "name": "Extracting...", | |
| "location": "Extracting...", | |
| "current_focus": "Extracting...", | |
| "education": "Extracting...", | |
| "technical_skills": [], | |
| "github": "Searching...", | |
| "raw_context": " ".join([r['body'] for r in results]) | |
| } | |
| enriched_data.append(profile_info) | |
| time.sleep(1) # Rate limiting to be respectful | |
| except Exception as e: | |
| print(f"Error processing {url}: {e}") | |
| enriched_data.append({"linkedin_url": url, "error": str(e)}) | |
| with open(json_output, 'w') as f: | |
| json.dump(enriched_data, f, indent=2) | |
| print(f"Enrichment complete. Data saved to {json_output}") | |
| if __name__ == "__main__": | |
| # Update these filenames as needed | |
| enrich_profiles('input_profiles.csv', 'enriched_metadata.json') |
Sign up for free
to join this conversation on GitHub.
Already have an account?
Sign in to comment