Spaces:
Build error
Build error
Download lex_rank_text2vec_v1.py from timpan/summary-simi-check4qee: direct link, hf CLI and curl.
- Browser
- Download file 1.52 kB
-
https://huggingface.co/spaces/timpan/summary-simi-check4qee/resolve/main/lex_rank_text2vec_v1.py
- Command line
-
hf download hf://spaces/timpan/summary-simi-check4qee/lex_rank_text2vec_v1.py
-
curl -L -o lex_rank_text2vec_v1.py https://huggingface.co/spaces/timpan/summary-simi-check4qee/resolve/main/lex_rank_text2vec_v1.py
1.52 kB
| import numpy, nltk | |
| nltk.download('punkt') | |
| from harvesttext import HarvestText | |
| from lex_rank_util import degree_centrality_scores, find_siblings_by_index | |
| from sentence_transformers import SentenceTransformer, util | |
| class LexRankText2VecV1(object): | |
| def __init__(self): | |
| self.model = SentenceTransformer('shibing624/text2vec-base-chinese-paraphrase') | |
| self.ht = HarvestText() | |
| def find_central(self, content: str, num=10, siblings=0): | |
| if self.contains_chinese(content): | |
| sentences = self.ht.cut_sentences(content) | |
| else: | |
| sentences = nltk.sent_tokenize(content) | |
| embeddings = self.model.encode(sentences, convert_to_tensor=True).cpu() | |
| # Compute the pair-wise cosine similarities | |
| cos_scores = util.cos_sim(embeddings, embeddings).numpy() | |
| # Compute the centrality for each sentence | |
| centrality_scores = degree_centrality_scores(cos_scores, threshold=None) | |
| # We argsort so that the first element is the sentence with the highest score | |
| most_central_sentence_indices = numpy.argsort(-centrality_scores) | |
| central_and_siblings = find_siblings_by_index(sentences, most_central_sentence_indices, siblings, num) | |
| res = [] | |
| for index in central_and_siblings: | |
| res.append(sentences[index]) | |
| return res | |
| def contains_chinese(self, content: str): | |
| for _char in content: | |
| if '\u4e00' <= _char <= '\u9fa5': | |
| return True | |
| return False | |