Spaces:
Running
Running
Download ingestion/entity_extraction.py from VedantDhavan/graphrag-benchmark: direct link, hf CLI and curl.
- Browser
- Download file 823 Bytes
-
https://huggingface.co/spaces/VedantDhavan/graphrag-benchmark/resolve/main/ingestion/entity_extraction.py
- Command line
-
hf download hf://spaces/VedantDhavan/graphrag-benchmark/ingestion/entity_extraction.py
-
curl -L -o entity_extraction.py https://huggingface.co/spaces/VedantDhavan/graphrag-benchmark/resolve/main/ingestion/entity_extraction.py
823 Bytes
| import re | |
| STOPWORDS = { | |
| "about", | |
| "after", | |
| "also", | |
| "because", | |
| "being", | |
| "between", | |
| "could", | |
| "does", | |
| "from", | |
| "for", | |
| "have", | |
| "into", | |
| "more", | |
| "other", | |
| "paper", | |
| "than", | |
| "that", | |
| "their", | |
| "there", | |
| "these", | |
| "this", | |
| "through", | |
| "using", | |
| "what", | |
| "when", | |
| "where", | |
| "which", | |
| "with", | |
| } | |
| def extract_entities(chunks): | |
| entities = set() | |
| for text in chunks: | |
| words = re.findall(r"\b[A-Z][A-Za-z0-9]*(?:-[A-Z0-9][A-Za-z0-9]*)*\b", text) | |
| words.extend(re.findall(r"\b[a-z][a-z0-9-]{4,}\b", text)) | |
| for w in words: | |
| normalized = w.strip("-").lower() | |
| if normalized and normalized not in STOPWORDS: | |
| entities.add((normalized, "Concept")) | |
| return list(entities) | |