Repository navigation
Expand file tree
/
Copy pathdocumentation-embedder.js
More file actions
91 lines (76 loc) · 3.92 KB
/
Copy pathdocumentation-embedder.js
File metadata and controls
91 lines (76 loc) · 3.92 KB
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
import { RecursiveCharacterTextSplitter } from "langchain/text_splitter";
import fs from 'fs';
import * as path from "node:path";
import { download } from '@guoyunhe/downloader';
import dotenv from 'dotenv'
import Anthropic from '@anthropic-ai/sdk'
import { WriteMode } from "vectordb";
dotenv.config()
const lancedb = await import("vectordb");
const { pipeline } = await import('@xenova/transformers')
const pipe = await pipeline('feature-extraction', 'Xenova/all-MiniLM-L6-v2');
const anthropic = new Anthropic({
apiKey: process.env.CLAUDE_API,
});
async function read_data(){
var docs = [];
const docsPath = "data/python-3.11.4-docs-text"
var subfolders = fs.readdirSync(docsPath);
for (let i = 0; i < subfolders.length; i++) {
const subfolder = docsPath + "/" + subfolders[i];
console.log(subfolder)
if (!fs.lstatSync(subfolder).isDirectory()) { continue; }
if (fs.existsSync(subfolder)) {
for (const p of fs.readdirSync(subfolder).filter((f) => f.endsWith('.txt'))) {
const docPath = path.join(subfolder, p);
console.log(docPath);
try {
var output = fs.readFileSync(docPath).toString();
const prompt = `\n\n${Anthropic.HUMAN_PROMPT}: You are to act as a summarizer bot whose task is to read the following code documentation about a python function and summarize it in a few paragraphs without bullet points. You should include what the function does and potentially a few examples on how to use it, in multiple paragraphs, but leave out any unrelated information that is not about the functionality of it in python, including a preface to the response, such as 'Here is a summary...'. \n\nREMEMBER: \nDo NOT begin your response with an introduction. \nMake sure your entire response can fit in a research paper. \nDo not use bullet points. \nKeep responses in paragraph form. \nDo not respond with extra context or your introduction to the reponse.\n\nNow act like the summarizer bot, and follow all instructions. Do not add any additional context or introduction in your response. Here is the documentation file:\n${output}\n\n${Anthropic.AI_PROMPT}:`
let completion = await anthropic.completions.create({
model: 'claude-2',
max_tokens_to_sample: 100000,
prompt: prompt,
});
completion = completion.completion.split(":\n\n").slice(1);
completion = completion.join(":\n\n");
docs = docs.concat(completion);
} catch (err) {
console.log(err);
}
}
}
}
fs.writeFileSync("data/all_python_docs.txt", docs.join("\n\n---\n\n"))
return docs;
};
const embed_fun = {}
embed_fun.sourceColumn = 'text'
embed_fun.embed = async function (batch) {
let result = []
for (let text of batch) {
const res = await pipe(text, { pooling: 'mean', normalize: true })
result.push(Array.from(res['data']))
}
return (result)
};
(async () => {
const db = await lancedb.connect("data/sample-lancedb")
// await download("https://docs.python.org/3/archives/python-3.11.4-docs-text.zip", "data/", { extract: true })
// var docs = await read_data();
var docs = fs.readFileSync("data/all_python_docs.txt").toString().split("\n\n---\n\n");
// make table here
const splitter = new RecursiveCharacterTextSplitter({
chunkSize: 250,
chunkOverlap: 50,
});
docs = await splitter.createDocuments(docs);
console.log(docs[0])
let data = [];
for (let doc of docs) {
data.push({text: doc['pageContent'], metadata: JSON.stringify(doc['metadata'])});
}
console.log("creating table");
const _ = await db.createTable("python_docs", data, embed_fun, { writeMode: WriteMode.Overwrite });
console.log("table created");
})();