INNER CODE UNIT · Python
get_chunks_from_hits
eugeneyan/obsidian-copilot · src/app.py:115
def get_chunks_from_hits(
hits: List[dict], model_name: str = "gpt-3.5-turbo", max_tokens: int = 3200
) -> List[dict]:
"""
Deduplicates and scores a list of chunks. (There may be duplicate chunks as we query multiple indices.)
Args:
hits: List of hits from opensearch, semantic index, etc.
model_name: Downstream model for retrieval-augmented generation. Used to tokenize chunks and limit the size of
input based on LLM context window size. Defaults to 'gpt-3.5-turbo'.
max_tokens: Maximum tokens to allow in chunks. Defaults to 3,200.
Returns:
List of chunks for retrieval-augmented generation.
"""
# Combine os and semantic hits and rank them
df = pd.DataFrame(hits)
df["score"] = df["rank"].apply(lambda x: 10 - x)