INNER CODE UNIT · Python

get_chunks_from_hits

eugeneyan/obsidian-copilot · src/app.py:115

def get_chunks_from_hits(
    hits: List[dict], model_name: str = "gpt-3.5-turbo", max_tokens: int = 3200
) -> List[dict]:
    """
    Deduplicates and scores a list of chunks. (There may be duplicate chunks as we query multiple indices.)

    Args:
        hits: List of hits from opensearch, semantic index, etc.
        model_name: Downstream model for retrieval-augmented generation. Used to tokenize chunks and limit the size of
            input based on LLM context window size. Defaults to 'gpt-3.5-turbo'.
        max_tokens: Maximum tokens to allow in chunks. Defaults to 3,200.

    Returns:
        List of chunks for retrieval-augmented generation.
    """
    # Combine os and semantic hits and rank them
    df = pd.DataFrame(hits)
    df["score"] = df["rank"].apply(lambda x: 10 - x)

View source record →

📰 Research Paper
Loading…
⏳ Fetching content…