Skip to content

Instantly share code, notes, and snippets.

@gourab5139014
Created April 17, 2026 01:45
Show Gist options
  • Select an option

  • Save gourab5139014/4462bbc474e25a80be2b9f64b43e6eed to your computer and use it in GitHub Desktop.

Select an option

Save gourab5139014/4462bbc474e25a80be2b9f64b43e6eed to your computer and use it in GitHub Desktop.
A Python utility for batch enriching a list of LinkedIn profiles from a CSV file. It leverages search aggregators (DuckDuckGo Search) to find professional metadata such as name, location, current focus, education, technical skills, and GitHub links, then synthesizes the findings into a structured JSON repository.
import pandas as pd
import json
import time
from duckduckgo_search import DDGS
def enrich_profiles(csv_input, json_output):
"""
Reads a CSV of LinkedIn URLs and enriches them with professional metadata
found via search snippets, saving the result to a structured JSON.
"""
try:
df = pd.read_csv(csv_input)
# Assumes the CSV has a column named 'LinkedIn'
urls = df['LinkedIn'].tolist()
except Exception as e:
print(f"Error reading CSV {csv_input}: {e}")
return
enriched_data = []
print(f"Starting enrichment for {len(urls)} profiles...")
with DDGS() as ddgs:
for url in urls:
print(f"Processing: {url}")
try:
# Search for the LinkedIn profile to get snippets
results = list(ddgs.text(url, max_results=3))
# Structure the metadata
profile_info = {
"linkedin_url": url,
"name": "Extracting...",
"location": "Extracting...",
"current_focus": "Extracting...",
"education": "Extracting...",
"technical_skills": [],
"github": "Searching...",
"raw_context": " ".join([r['body'] for r in results])
}
enriched_data.append(profile_info)
time.sleep(1) # Rate limiting to be respectful
except Exception as e:
print(f"Error processing {url}: {e}")
enriched_data.append({"linkedin_url": url, "error": str(e)})
with open(json_output, 'w') as f:
json.dump(enriched_data, f, indent=2)
print(f"Enrichment complete. Data saved to {json_output}")
if __name__ == "__main__":
# Update these filenames as needed
enrich_profiles('input_profiles.csv', 'enriched_metadata.json')
Sign up for free to join this conversation on GitHub. Already have an account? Sign in to comment