assistants/tools/file-search.md +0 −1367 deleted
File Deleted View Diff
1# Assistants File Search
2
3> For the complete documentation index, see [llms.txt](/llms.txt). Markdown versions of documentation pages are available by appending `.md` to the page URL.
4
5After achieving feature parity in the Responses API, we've deprecated the Assistants API. It will shut down on August 26, 2026. Follow the [migration guide](https://developers.openai.com/platform/assistants/migration) to update your integration. [Learn more](https://platform.openai.com/docs/guides/migrate-to-responses).
6
7## Overview
8
9File Search augments the Assistant with knowledge from outside its model, such as proprietary product information or documents provided by your users. OpenAI automatically parses and chunks your documents, creates and stores the embeddings, and use both vector and keyword search to retrieve relevant content to answer user queries.
10
11## Quickstart
12
13In this example, we’ll create an assistant that can help answer questions about companies’ financial statements.
14
15### Step 1: Create a new Assistant with File Search Enabled
16
17Create a new assistant with `file_search` enabled in the `tools` parameter of the Assistant.
18
19```javascript
20import OpenAI from "openai";
21const openai = new OpenAI();
22
23async function main() {
24 const assistant = await openai.beta.assistants.create({
25 name: "Financial Analyst Assistant",
26 instructions:
27 "You are an expert financial analyst. Use you knowledge base to answer questions about audited financial statements.",
28 model: "gpt-4o",
29 tools: [{ type: "file_search" }],
30 });
31}
32
33main();
34```
35
36```python
37from openai import OpenAI
38
39client = OpenAI()
40
41assistant = client.beta.assistants.create(
42 name="Financial Analyst Assistant",
43 instructions="You are an expert financial analyst. Use you knowledge base to answer questions about audited financial statements.",
44 model="gpt-4o",
45 tools=[{"type": "file_search"}],
46)
47```
48
49```go
50assistant, err := client.Beta.Assistants.New(context.Background(), openai.BetaAssistantNewParams{
51 Name: openai.String("Financial Analyst Assistant"),
52 Instructions: openai.String("You are an expert financial analyst. Use your knowledge base to answer questions about audited financial statements."),
53 Model: shared.ChatModelGPT4o,
54 Tools: []openai.AssistantToolUnionParam{{OfFileSearch: &openai.FileSearchToolParam{}}},
55})
56if err != nil {
57 panic(err)
58}
59```
60
61```java
62import com.openai.client.OpenAIClient;
63import com.openai.client.okhttp.OpenAIOkHttpClient;
64import com.openai.models.beta.assistants.AssistantCreateParams;
65import com.openai.models.beta.assistants.FileSearchTool;
66
67var assistant =
68 client
69 .beta()
70 .assistants()
71 .create(
72 AssistantCreateParams.builder()
73 .model("gpt-4o")
74 .name("Financial Analyst Assistant")
75 .instructions(
76 "You are an expert financial analyst. Use you knowledge base to answer"
77 + " questions about audited financial statements.")
78 .addTool(FileSearchTool.builder().build())
79 .build());
80
81System.out.println(assistant.id());
82```
83
84```ruby
85require "openai"
86
87client = OpenAI::Client.new
88assistant = client.beta.assistants.create(
89 model: "gpt-4o",
90 name: "Financial Analyst Assistant",
91 instructions: "Use the knowledge base to answer questions about audited financial statements.",
92 tools: [{type: :file_search}]
93)
94puts(assistant.id)
95```
96
97```bash
98curl https://api.openai.com/v1/assistants \
99-H "Content-Type: application/json" \
100-H "Authorization: Bearer $OPENAI_API_KEY" \
101-H "OpenAI-Beta: assistants=v2" \
102-d '{
103"name": "Financial Analyst Assistant",
104"instructions": "You are an expert financial analyst. Use you knowledge base to answer questions about audited financial statements.",
105"tools": [{"type": "file_search"}],
106"model": "gpt-4o"
107}'
108```
109
110
111Once the `file_search` tool is enabled, the model decides when to retrieve content based on user messages.
112
113### Step 2: Upload files and add them to a Vector Store
114
115To access your files, the `file_search` tool uses the Vector Store object.
116Upload your files and create a Vector Store to contain them.
117Once the Vector Store is created, you should poll its status until all files are out of the `in_progress` state to
118ensure that all content has finished processing. The SDK provides helpers to uploading and polling in one shot.
119
120```javascript
121const fileStreams = [
122 fs.createReadStream("edgar/goog-10k.pdf"),
123 fs.createReadStream("edgar/brka-10k.txt"),
124];
125
126// Create a vector store including our two files.
127let vectorStore = await openai.vectorStores.create({
128 name: "Financial Statement",
129});
130
131await openai.vectorStores.fileBatches.uploadAndPoll(vectorStore.id, {
132 files: fileStreams,
133});
134```
135
136```python
137# Create a vector store called "Financial Statements"
138vector_store = client.vector_stores.create(name="Financial Statements")
139
140# Ready the files for upload to OpenAI
141
142file_paths = ["edgar/goog-10k.pdf", "edgar/brka-10k.txt"]
143file_streams = [open(path, "rb") for path in file_paths]
144
145# Use the upload and poll SDK helper to upload the files, add them to the vector store,
146
147# and poll the status of the file batch for completion.
148
149file_batch = client.vector_stores.file_batches.upload_and_poll(
150 vector_store_id=vector_store.id, files=file_streams
151)
152
153# You can print the status and the file counts of the batch to see the result of this operation.
154
155print(file_batch.status)
156print(file_batch.file_counts)
157```
158
159
160### Step 3: Update the assistant to use the new Vector Store
161
162To make the files accessible to your assistant, update the assistant’s `tool_resources` with the new `vector_store` id.
163
164```javascript
165await openai.beta.assistants.update(assistant.id, {
166 tool_resources: { file_search: { vector_store_ids: [vectorStore.id] } },
167});
168```
169
170```python
171assistant = client.beta.assistants.update(
172 assistant_id=assistant.id,
173 tool_resources={"file_search": {"vector_store_ids": [vector_store.id]}},
174)
175```
176
177```go
178_, err := client.Beta.Assistants.Update(context.Background(), "asst_abc123", openai.BetaAssistantUpdateParams{
179 ToolResources: openai.BetaAssistantUpdateParamsToolResources{
180 FileSearch: openai.BetaAssistantUpdateParamsToolResourcesFileSearch{
181 VectorStoreIDs: []string{"vs_abc123"},
182 },
183 },
184})
185if err != nil {
186 panic(err)
187}
188```
189
190```java
191import com.openai.client.OpenAIClient;
192import com.openai.client.okhttp.OpenAIOkHttpClient;
193import com.openai.models.beta.assistants.AssistantUpdateParams;
194import com.openai.models.beta.assistants.FileSearchTool;
195
196String assistantId = "asst_abc123";
197
198String vectorStoreId = "vs_abc123";
199
200var assistant =
201 client
202 .beta()
203 .assistants()
204 .update(
205 assistantId,
206 AssistantUpdateParams.builder()
207 .addTool(FileSearchTool.builder().build())
208 .toolResources(
209 AssistantUpdateParams.ToolResources.builder()
210 .fileSearch(
211 AssistantUpdateParams.ToolResources.FileSearch.builder()
212 .addVectorStoreId(vectorStoreId)
213 .build())
214 .build())
215 .build());
216
217System.out.println(assistant.id());
218```
219
220```ruby
221require "openai"
222
223client = OpenAI::Client.new
224assistant = client.beta.assistants.update(
225 "asst_abc123",
226 tool_resources: {
227 file_search: {vector_store_ids: ["vs_abc123"]}
228 }
229)
230puts(assistant.id)
231```
232
233
234### Step 4: Create a thread
235
236You can also attach files as Message attachments on your thread. Doing so will create another `vector_store` associated with the thread, or, if there is already a vector store attached to this thread, attach the new files to the existing thread vector store. When you create a Run on this thread, the file search tool will query both the `vector_store` from your assistant and the `vector_store` on the thread.
237
238In this example, the user attached a copy of Apple’s latest 10-K filing.
239
240```javascript
241// A user wants to attach a file to a specific message, let's upload it.
242const aapl10k = await openai.files.create({
243 file: fs.createReadStream("edgar/aapl-10k.pdf"),
244 purpose: "assistants",
245});
246
247const thread = await openai.beta.threads.create({
248 messages: [
249 {
250 role: "user",
251 content:
252 "How many shares of AAPL were outstanding at the end of October 2023?",
253 // Attach the new file to the message.
254 attachments: [{ file_id: aapl10k.id, tools: [{ type: "file_search" }] }],
255 },
256 ],
257});
258
259// The thread now has a vector store in its tool resources.
260console.log(thread.tool_resources?.file_search);
261```
262
263```python
264# Upload the user provided file to OpenAI
265message_file = client.files.create(
266 file=open("edgar/aapl-10k.pdf", "rb"), purpose="assistants"
267)
268
269# Create a thread and attach the file to the message
270
271thread = client.beta.threads.create(
272 messages=[
273 {
274 "role": "user",
275 "content": "How many shares of AAPL were outstanding at the end of of October 2023?", # Attach the new file to the message.
276 "attachments": [
277 {"file_id": message_file.id, "tools": [{"type": "file_search"}]}
278 ],
279 }
280 ]
281)
282
283# The thread now has a vector store with that file in its tool resources.
284
285print(thread.tool_resources.file_search)
286```
287
288
289Vector stores created using message attachments have a default expiration policy of 7 days after they were last active (defined as the last time the vector store was part of a run). This default exists to help you manage your vector storage costs. You can override these expiration policies at any time. Learn more [here](#managing-costs-with-expiration-policies).
290
291### Step 5: Create a run and check the output
292
293Now, create a Run and observe that the model uses the File Search tool to provide a response to the user’s question.
294
295
296
297With streaming
298
299```javascript
300const stream = openai.beta.threads.runs
301 .stream(thread.id, {
302 assistant_id: assistant.id,
303 })
304 .on("textCreated", () => console.log("assistant >"))
305 .on("toolCallCreated", (event) => console.log("assistant " + event.type))
306 .on("messageDone", async (event) => {
307 if (event.content[0].type === "text") {
308 const { text } = event.content[0];
309 const { annotations } = text;
310 const citations = [];
311
312 let index = 0;
313 for (const annotation of annotations) {
314 text.value = text.value.replace(annotation.text, `[${index}]`);
315 if (annotation.type === "file_citation") {
316 const citedFile = await openai.files.retrieve(
317 annotation.file_citation.file_id
318 );
319 citations.push(`[${index}]${citedFile.filename}`);
320 }
321 index++;
322 }
323
324 console.log(text.value);
325 console.log(citations.join("\n"));
326 }
327 });
328```
329
330```python
331from typing_extensions import override
332from openai import AssistantEventHandler, OpenAI
333
334client = OpenAI()
335
336class EventHandler(AssistantEventHandler):
337 @override
338 def on_text_created(self, text) -> None:
339 print("\nassistant > ", end="", flush=True)
340
341 @override
342 def on_tool_call_created(self, tool_call):
343 print(f"\nassistant > {tool_call.type}\n", flush=True)
344
345 @override
346 def on_message_done(self, message) -> None:
347 # print a citation to the file searched
348 message_content = message.content[0].text
349 annotations = message_content.annotations
350 citations = []
351 for index, annotation in enumerate(annotations):
352 message_content.value = message_content.value.replace(
353 annotation.text, f"[{index}]"
354 )
355 if file_citation := getattr(annotation, "file_citation", None):
356 cited_file = client.files.retrieve(file_citation.file_id)
357 citations.append(f"[{index}] {cited_file.filename}")
358
359 print(message_content.value)
360 print("\n".join(citations))
361
362# Then, we use the stream SDK helper
363
364# with the EventHandler class to create the Run
365
366# and stream the response.
367
368with client.beta.threads.runs.stream(
369 thread_id=thread.id,
370 assistant_id=assistant.id,
371 instructions="Please address the user as Jane Doe. The user has a premium account.",
372 event_handler=EventHandler(),
373) as stream:
374 stream.until_done()
375```
376
377
378
379
380
381
382Without streaming
383
384```javascript
385const run = await openai.beta.threads.runs.createAndPoll(thread.id, {
386 assistant_id: assistant.id,
387});
388
389const messages = await openai.beta.threads.messages.list(thread.id, {
390 run_id: run.id,
391});
392
393const message = messages.data.pop();
394if (message.content[0].type === "text") {
395 const { text } = message.content[0];
396 const { annotations } = text;
397 const citations = [];
398
399 let index = 0;
400 for (const annotation of annotations) {
401 text.value = text.value.replace(annotation.text, `[${index}]`);
402 if (annotation.type === "file_citation") {
403 const citedFile = await openai.files.retrieve(
404 annotation.file_citation.file_id
405 );
406 citations.push(`[${index}]${citedFile.filename}`);
407 }
408 index++;
409 }
410
411 console.log(text.value);
412 console.log(citations.join("\n"));
413}
414```
415
416```python
417# Use the create and poll SDK helper to create a run and poll the status of
418# the run until it's in a terminal state.
419
420run = client.beta.threads.runs.create_and_poll(
421 thread_id=thread.id,
422 assistant_id=assistant.id,
423)
424
425messages = list(
426 client.beta.threads.messages.list(thread_id=thread.id, run_id=run.id)
427)
428
429message_content = messages[0].content[0].text
430annotations = message_content.annotations
431citations = []
432for index, annotation in enumerate(annotations):
433 message_content.value = message_content.value.replace(
434 annotation.text, f"[{index}]"
435 )
436 if file_citation := getattr(annotation, "file_citation", None):
437 cited_file = client.files.retrieve(file_citation.file_id)
438 citations.append(f"[{index}] {cited_file.filename}")
439
440print(message_content.value)
441print("\n".join(citations))
442```
443
444
445
446Your new assistant will query both attached vector stores (one containing `goog-10k.pdf` and `brka-10k.txt`, and the other containing `aapl-10k.pdf`) and return this result from `aapl-10k.pdf`.
447
448To retrieve the contents of the file search results that were used by the model, use the `include` query parameter and provide a value of `step_details.tool_calls[*].file_search.results[*].content` in the format `?include[]=step_details.tool_calls[*].file_search.results[*].content`.
449
450
451## How it works
452
453The `file_search` tool implements several retrieval best practices out of the box to help you extract the right data from your files and augment the model’s responses. The `file_search` tool:
454
455- Rewrites user queries to optimize them for search.
456- Breaks down complex user queries into multiple searches it can run in parallel.
457- Runs both keyword and semantic searches across both assistant and thread vector stores.
458- Reranks search results to pick the most relevant ones before generating the final response.
459
460By default, the `file_search` tool uses the following settings but these can be [configured](#customizing-file-search-settings) to suit your needs:
461
462- Chunk size: 800 tokens
463- Chunk overlap: 400 tokens
464- Embedding model: `text-embedding-3-large` at 256 dimensions
465- Maximum number of chunks added to context: 20 (could be fewer)
466- Ranker: `auto` (OpenAI will choose which ranker to use)
467- Score threshold: 0 minimum ranking score
468
469**Known Limitations**
470
471We have a few known limitations we're working on adding support for in the coming months:
472
4731. Support for deterministic pre-search filtering using custom metadata.
4742. Support for parsing images within documents (including images of charts, graphs, tables etc.)
4753. Support for retrievals over structured file formats (like `csv` or `jsonl`).
4764. Better support for summarization — the tool today is optimized for search queries.
477
478## Vector stores
479
480Vector Store objects give the File Search tool the ability to search your files. Adding a file to a `vector_store` automatically parses, chunks, embeds and stores the file in a vector database that's capable of both keyword and semantic search. Each `vector_store` can hold up to 10,000 files. For vector stores created starting in November 2025, this limit is 100,000,000 files. Vector stores can be attached to both Assistants and Threads. Today, you can attach at most one vector store to an assistant and at most one vector store to a thread.
481
482#### Creating vector stores and adding files
483
484You can create a vector store and add files to it in a single API call:
485
486```javascript
487const vectorStore = await openai.vectorStores.create({
488 name: "Product Documentation",
489 file_ids: [
490 "file_1",
491 "file_2",
492 "file_3",
493 "file_4",
494 "file_5",
495 ],
496});
497```
498
499```python
500vector_store = client.vector_stores.create(
501 name="Product Documentation",
502 file_ids=[
503 "file_1",
504 "file_2",
505 "file_3",
506 "file_4",
507 "file_5",
508 ],
509)
510```
511
512```go
513vectorStore, err := client.VectorStores.New(context.Background(), openai.VectorStoreNewParams{
514 Name: openai.String("Product Documentation"),
515 FileIDs: []string{"file_1", "file_2", "file_3", "file_4", "file_5"},
516})
517if err != nil {
518 panic(err)
519}
520```
521
522```java
523import com.openai.client.OpenAIClient;
524import com.openai.client.okhttp.OpenAIOkHttpClient;
525import com.openai.models.vectorstores.VectorStoreCreateParams;
526
527String fileId1 = "file_1";
528
529String fileId2 = "file_2";
530
531String fileId3 = "file_3";
532
533String fileId4 = "file_4";
534
535String fileId5 = "file_5";
536
537var store =
538 client
539 .vectorStores()
540 .create(
541 VectorStoreCreateParams.builder()
542 .name("Product Documentation")
543 .addFileId(fileId1)
544 .addFileId(fileId2)
545 .addFileId(fileId3)
546 .addFileId(fileId4)
547 .addFileId(fileId5)
548 .build());
549
550System.out.println(store.id());
551```
552
553```ruby
554require "openai"
555
556client = OpenAI::Client.new
557store = client.vector_stores.create(
558 name: "Product Documentation",
559 file_ids: [
560 "file_1",
561 "file_2",
562 "file_3",
563 "file_4",
564 "file_5"
565 ]
566)
567puts(store.id)
568```
569
570
571Adding files to vector stores is an async operation. To ensure the operation is complete, we recommend that you use the 'create and poll' helpers in our official SDKs. If you're not using the SDKs, you can retrieve the `vector_store` object and monitor its [`file_counts`](https://developers.openai.com/api/reference/resources/vector_stores#vector-stores/object-file_counts) property to see the result of the file ingestion operation.
572
573Files can also be added to a vector store after it's created by [creating vector store files](https://developers.openai.com/api/reference/resources/vector_stores/subresources/files/methods/create).
574
575Adding files is rate limited per vector store ID. Requests to `/vector_stores/{vector_store_id}/files` and `/vector_stores/{vector_store_id}/file_batches` share a per-vector-store limit of 300 requests per minute.
576
577```javascript
578const file = await openai.vectorStores.files.createAndPoll(
579 "vs_abc123",
580 {
581 file_id: "file-abc123",
582 }
583);
584```
585
586```python
587file = client.vector_stores.files.create_and_poll(
588 vector_store_id="vs_abc123", file_id="file-abc123"
589)
590```
591
592```go
593_, err := client.VectorStores.Files.NewAndPoll(context.Background(), "vs_abc123", openai.VectorStoreFileNewParams{
594 FileID: "file-abc123",
595}, 1000)
596if err != nil {
597 panic(err)
598}
599```
600
601```java
602import com.openai.client.OpenAIClient;
603import com.openai.client.okhttp.OpenAIOkHttpClient;
604import com.openai.models.vectorstores.files.FileCreateParams;
605import com.openai.models.vectorstores.files.FileRetrieveParams;
606import com.openai.models.vectorstores.files.VectorStoreFile;
607
608String vectorStoreId = "vs_abc123";
609String fileId = "file-abc123";
610var file =
611 client
612 .vectorStores()
613 .files()
614 .create(vectorStoreId, FileCreateParams.builder().fileId(fileId).build());
615while (file.status().equals(VectorStoreFile.Status.IN_PROGRESS)) {
616 Thread.sleep(1000);
617 file =
618 client
619 .vectorStores()
620 .files()
621 .retrieve(
622 file.id(), FileRetrieveParams.builder().vectorStoreId(vectorStoreId).build());
623}
624System.out.println(file.status());
625```
626
627```ruby
628require "openai"
629
630client = OpenAI::Client.new
631file = client.vector_stores.files.create(
632 "vs_abc123",
633 file_id: "file-abc123"
634)
635until [:completed, :failed, :cancelled].include?(file.status)
636 sleep(1)
637 file = client.vector_stores.files.retrieve(
638 file.id,
639 vector_store_id: "vs_abc123"
640 )
641end
642puts(file.status)
643```
644
645
646Alternatively, you can add several files to a vector store by [creating batches](https://developers.openai.com/api/reference/resources/vector_stores/subresources/file_batches/methods/create) of up to 500 files.
647
648Batch creation accepts either a simple list of `file_ids` or a `files` array made up of objects with a `file_id` plus optional `attributes` and `chunking_strategy`. Use `files` when you need per-file metadata or chunking settings, and note that `file_ids` and `files` are mutually exclusive in a single request.
649
650For high-throughput ingestion into one vector store, prefer file batches whenever possible to reduce request volume and improve latency.
651
652```javascript
653const batch = await openai.vectorStores.fileBatches.createAndPoll(
654 "vs_abc123",
655 {
656 files: [
657 {
658 file_id: "file_1",
659 attributes: { category: "finance" },
660 },
661 {
662 file_id: "file_2",
663 chunking_strategy: {
664 type: "static",
665 static: {
666 max_chunk_size_tokens: 1000,
667 chunk_overlap_tokens: 200,
668 },
669 },
670 },
671 ],
672 }
673);
674```
675
676```python
677batch = client.vector_stores.file_batches.create_and_poll(
678 vector_store_id="vs_abc123",
679 files=[
680 {"file_id": "file_1", "attributes": {"category": "finance"}},
681 {
682 "file_id": "file_2",
683 "chunking_strategy": {
684 "type": "static",
685 "max_chunk_size_tokens": 1000,
686 "chunk_overlap_tokens": 200,
687 },
688 },
689 ],
690)
691```
692
693```go
694_, err := client.VectorStores.FileBatches.NewAndPoll(context.Background(), "vs_abc123", openai.VectorStoreFileBatchNewParams{
695 Files: []openai.VectorStoreFileBatchNewParamsFile{
696 {
697 FileID: "file_1",
698 Attributes: map[string]openai.VectorStoreFileBatchNewParamsFileAttributeUnion{
699 "category": {OfString: openai.String("finance")},
700 },
701 },
702 {
703 FileID: "file_2",
704 ChunkingStrategy: openai.FileChunkingStrategyParamUnion{OfStatic: &openai.StaticFileChunkingStrategyObjectParam{
705 Static: openai.StaticFileChunkingStrategyParam{MaxChunkSizeTokens: 1000, ChunkOverlapTokens: 200},
706 }},
707 },
708 },
709}, 1000)
710if err != nil {
711 panic(err)
712}
713```
714
715```java
716import com.openai.client.OpenAIClient;
717import com.openai.client.okhttp.OpenAIOkHttpClient;
718import com.openai.core.JsonValue;
719import com.openai.models.vectorstores.StaticFileChunkingStrategy;
720import com.openai.models.vectorstores.filebatches.FileBatchCreateParams;
721import com.openai.models.vectorstores.filebatches.FileBatchRetrieveParams;
722import com.openai.models.vectorstores.filebatches.VectorStoreFileBatch;
723
724String vectorStoreId = "vs_abc123";
725String fileId = "file_1";
726String fileId2 = "file_2";
727var first =
728 FileBatchCreateParams.File.builder()
729 .fileId(fileId)
730 .attributes(
731 FileBatchCreateParams.File.Attributes.builder()
732 .putAdditionalProperty("category", JsonValue.from("finance"))
733 .build())
734 .build();
735var second =
736 FileBatchCreateParams.File.builder()
737 .fileId(fileId2)
738 .staticChunkingStrategy(
739 StaticFileChunkingStrategy.builder()
740 .maxChunkSizeTokens(1000)
741 .chunkOverlapTokens(200)
742 .build())
743 .build();
744
745var batch =
746 client
747 .vectorStores()
748 .fileBatches()
749 .create(
750 vectorStoreId,
751 FileBatchCreateParams.builder().addFile(first).addFile(second).build());
752while (batch.status().equals(VectorStoreFileBatch.Status.IN_PROGRESS)) {
753 Thread.sleep(1000);
754 batch =
755 client
756 .vectorStores()
757 .fileBatches()
758 .retrieve(
759 batch.id(),
760 FileBatchRetrieveParams.builder().vectorStoreId(vectorStoreId).build());
761}
762System.out.println(batch.status());
763```
764
765```ruby
766require "openai"
767
768client = OpenAI::Client.new
769batch = client.vector_stores.file_batches.create(
770 "vs_abc123",
771 files: [
772 {file_id: "file_1", attributes: {category: "finance"}},
773 {
774 file_id: "file_2",
775 chunking_strategy: {
776 type: :static,
777 max_chunk_size_tokens: 1_000,
778 chunk_overlap_tokens: 200
779 }
780 }
781 ]
782)
783until [:completed, :failed, :cancelled].include?(batch.status)
784 sleep(1)
785 batch = client.vector_stores.file_batches.retrieve(
786 batch.id,
787 vector_store_id: "vs_abc123"
788 )
789end
790puts(batch.status)
791```
792
793
794Similarly, these files can be removed from a vector store by either:
795
796- Deleting the [vector store file object](https://developers.openai.com/api/reference/resources/vector_stores/subresources/files/methods/delete) or,
797- By deleting the underlying [file object](https://developers.openai.com/api/reference/resources/files/methods/delete) (which removes the file it from all `vector_store` and `code_interpreter` configurations across all assistants and threads in your organization)
798
799The maximum file size is 512 MB. Each file should contain no more than 5,000,000 tokens per file (computed automatically when you attach a file).
800
801File Search supports a variety of file formats including `.pdf`, `.md`, and `.docx`. More details on the file extensions (and their corresponding MIME-types) supported can be found in the [Supported files](#supported-files) section below.
802
803#### Attaching vector stores
804
805You can attach vector stores to your Assistant or Thread using the `tool_resources` parameter.
806
807```javascript
808const assistant = await openai.beta.assistants.create({
809 instructions:
810 "You are a helpful product support assistant and you answer questions based on the files provided to you.",
811 model: "gpt-4o",
812 tools: [{ type: "file_search" }],
813 tool_resources: {
814 file_search: {
815 vector_store_ids: ["vs_1"],
816 },
817 },
818});
819
820const thread = await openai.beta.threads.create({
821 messages: [{ role: "user", content: "How do I cancel my subscription?" }],
822 tool_resources: {
823 file_search: {
824 vector_store_ids: ["vs_2"],
825 },
826 },
827});
828```
829
830```python
831assistant = client.beta.assistants.create(
832 instructions="You are a helpful product support assistant and you answer questions based on the files provided to you.",
833 model="gpt-4o",
834 tools=[{"type": "file_search"}],
835 tool_resources={"file_search": {"vector_store_ids": ["vs_1"]}},
836)
837
838thread = client.beta.threads.create(
839 messages=[{"role": "user", "content": "How do I cancel my subscription?"}],
840 tool_resources={"file_search": {"vector_store_ids": ["vs_2"]}},
841)
842```
843
844```go
845assistant, err := client.Beta.Assistants.New(context.Background(), openai.BetaAssistantNewParams{
846 Instructions: openai.String("You are a helpful product support assistant and you answer questions based on the files provided to you."),
847 Model: shared.ChatModelGPT4o,
848 Tools: []openai.AssistantToolUnionParam{{OfFileSearch: &openai.FileSearchToolParam{}}},
849 ToolResources: openai.BetaAssistantNewParamsToolResources{
850 FileSearch: openai.BetaAssistantNewParamsToolResourcesFileSearch{VectorStoreIDs: []string{"vs_1"}},
851 },
852})
853if err != nil {
854 panic(err)
855}
856thread, err := client.Beta.Threads.New(context.Background(), openai.BetaThreadNewParams{
857 Messages: []openai.BetaThreadNewParamsMessage{{
858 Role: "user",
859 Content: openai.BetaThreadNewParamsMessageContentUnion{OfString: openai.String("How do I cancel my subscription?")},
860 }},
861 ToolResources: openai.BetaThreadNewParamsToolResources{
862 FileSearch: openai.BetaThreadNewParamsToolResourcesFileSearch{VectorStoreIDs: []string{"vs_2"}},
863 },
864})
865if err != nil {
866 panic(err)
867}
868```
869
870```java
871import com.openai.client.OpenAIClient;
872import com.openai.client.okhttp.OpenAIOkHttpClient;
873import com.openai.models.beta.assistants.AssistantCreateParams;
874import com.openai.models.beta.assistants.FileSearchTool;
875import com.openai.models.beta.threads.ThreadCreateParams;
876
877String vectorStoreId = "vs_1";
878
879String vectorStoreId2 = "vs_2";
880
881var assistant =
882 client
883 .beta()
884 .assistants()
885 .create(
886 AssistantCreateParams.builder()
887 .model("gpt-4o")
888 .instructions(
889 "You are a helpful product support assistant and you answer questions based"
890 + " on the files provided to you.")
891 .addTool(FileSearchTool.builder().build())
892 .toolResources(
893 AssistantCreateParams.ToolResources.builder()
894 .fileSearch(
895 AssistantCreateParams.ToolResources.FileSearch.builder()
896 .addVectorStoreId(vectorStoreId)
897 .build())
898 .build())
899 .build());
900
901var thread =
902 client
903 .beta()
904 .threads()
905 .create(
906 ThreadCreateParams.builder()
907 .addMessage(
908 ThreadCreateParams.Message.builder()
909 .role(ThreadCreateParams.Message.Role.USER)
910 .content("How do I cancel my subscription?")
911 .build())
912 .toolResources(
913 ThreadCreateParams.ToolResources.builder()
914 .fileSearch(
915 ThreadCreateParams.ToolResources.FileSearch.builder()
916 .addVectorStoreId(vectorStoreId2)
917 .build())
918 .build())
919 .build());
920
921System.out.println(assistant.id() + " " + thread.id());
922```
923
924```ruby
925require "openai"
926
927client = OpenAI::Client.new
928assistant = client.beta.assistants.create(
929 instructions: "Answer product support questions using the provided files.",
930 model: "gpt-4o",
931 tools: [{type: :file_search}],
932 tool_resources: {
933 file_search: {vector_store_ids: ["vs_1"]}
934 }
935)
936thread = client.beta.threads.create(
937 messages: [{role: :user, content: "How do I cancel my subscription?"}],
938 tool_resources: {
939 file_search: {vector_store_ids: ["vs_2"]}
940 }
941)
942puts([assistant.id, thread.id])
943```
944
945
946You can also attach a vector store to Threads or Assistants after they're created by updating them with the right `tool_resources`.
947
948#### Ensuring vector store readiness before creating runs
949
950We highly recommend that you ensure all files in a `vector_store` are fully processed before you create a run. This will ensure that all the data in your `vector_store` is searchable. You can check for `vector_store` readiness by using the polling helpers in our SDKs, or by manually polling the `vector_store` object to ensure the [`status`](https://developers.openai.com/api/reference/resources/vector_stores#vector-stores/object-status) is `completed`.
951
952As a fallback, we've built a **60 second maximum wait** in the Run object when the **thread’s** vector store contains files that are still being processed. This is to ensure that any files your users upload in a thread a fully searchable before the run proceeds. This fallback wait _does not_ apply to the assistant's vector store.
953
954#### Customizing File Search settings
955
956You can customize how the `file_search` tool chunks your data and how many chunks it returns to the model context.
957
958**Chunking configuration**
959
960By default, `max_chunk_size_tokens` is set to `800` and `chunk_overlap_tokens` is set to `400`, meaning every file is indexed by being split up into 800-token chunks, with 400-token overlap between consecutive chunks.
961
962You can adjust this by setting [`chunking_strategy`](https://developers.openai.com/api/reference/resources/vector_stores/subresources/files/methods/create#vector-stores-files-createfile-chunking_strategy) when adding files to the vector store. There are certain limitations to `chunking_strategy`:
963
964- `max_chunk_size_tokens` must be between 100 and 4096 inclusive.
965- `chunk_overlap_tokens` must be non-negative and should not exceed `max_chunk_size_tokens / 2`.
966
967**Number of chunks**
968
969By default, the `file_search` tool outputs up to 20 chunks for `gpt-4*` and o-series models and up to 5 chunks for `gpt-3.5-turbo`. You can adjust this by setting [`file_search.max_num_results`](https://developers.openai.com/api/reference/resources/beta/subresources/assistants/methods/create#assistants-createassistant-tools) in the tool when creating the assistant or the run.
970
971Note that the `file_search` tool may output fewer than this number for a myriad of reasons:
972
973- The total number of chunks is fewer than `max_num_results`.
974- The total token size of all the retrieved chunks exceeds the token "budget" assigned to the `file_search` tool. The `file_search` tool currently has a token budget of:
975 - 4,000 tokens for `gpt-3.5-turbo`
976 - 16,000 tokens for `gpt-4*` models
977 - 16,000 tokens for o-series models
978
979#### Improve file search result relevance with chunk ranking
980
981By default, the file search tool will return all search results to the model that it thinks have any level of relevance when generating a response. However, if responses are generated using content that has low relevance, it can lead to lower quality responses. You can adjust this behavior by both inspecting the file search results that are returned when generating responses, and then tuning the behavior of the file search tool's ranker to change how relevant results must be before they are used to generate a response.
982
983**Inspecting file search chunks**
984
985The first step in improving the quality of your file search results is inspecting the current behavior of your assistant. Most often, this will involve investigating responses from your assistant that are not not performing well. You can get [granular information about a past run step](https://developers.openai.com/api/reference/resources/beta/subresources/threads/subresources/runs/subresources/steps/methods/retrieve) using the REST API, specifically using the `include` query parameter to get the file chunks that are being used to generate results.
986
987Include file search results in response when creating a run
988
989```javascript
990import OpenAI from "openai";
991
992const openai = new OpenAI();
993
994const runStep = await openai.beta.threads.runs.steps.retrieve("step_abc123", {
995 thread_id: "thread_abc123",
996 run_id: "run_abc123",
997 include: ["step_details.tool_calls[*].file_search.results[*].content"],
998});
999
1000console.log(runStep);
1001```
1002
1003```python
1004from openai import OpenAI
1005
1006client = OpenAI()
1007
1008run_step = client.beta.threads.runs.steps.retrieve(
1009 thread_id="thread_abc123",
1010 run_id="run_abc123",
1011 step_id="step_abc123",
1012 include=["step_details.tool_calls[*].file_search.results[*].content"],
1013)
1014
1015print(run_step)
1016```
1017
1018```go
1019runStep, err := client.Beta.Threads.Runs.Steps.Get(
1020 context.Background(),
1021 "thread_abc123",
1022 "run_abc123",
1023 "step_abc123",
1024 openai.BetaThreadRunStepGetParams{Include: []openai.RunStepInclude{
1025 openai.RunStepIncludeStepDetailsToolCallsFileSearchResultsContent,
1026 }},
1027)
1028if err != nil {
1029 panic(err)
1030}
1031fmt.Println(runStep)
1032```
1033
1034```ruby
1035require "openai"
1036
1037client = OpenAI::Client.new
1038step = client.beta.threads.runs.steps.retrieve(
1039 "step_abc123",
1040 thread_id: "thread_abc123",
1041 run_id: "run_abc123",
1042 include: ["step_details.tool_calls[*].file_search.results[*].content"]
1043)
1044puts(step)
1045```
1046
1047```bash
1048curl -g https://api.openai.com/v1/threads/thread_abc123/runs/run_abc123/steps/step_abc123?include[]=step_details.tool_calls[*].file_search.results[*].content \
1049-H "Authorization: Bearer $OPENAI_API_KEY" \
1050-H "Content-Type: application/json" \
1051-H "OpenAI-Beta: assistants=v2"
1052```
1053
1054
1055You can then log and inspect the search results used during the run step, and determine whether or not they are consistently relevant to the responses your assistant should generate.
1056
1057**Configure ranking options**
1058
1059If you have determined that your file search results are not sufficiently relevant to generate high quality responses, you can adjust the settings of the result ranker used to choose which search results should be used to generate responses. You can adjust this setting [`file_search.ranking_options`](https://developers.openai.com/api/reference/resources/beta/subresources/assistants/methods/create#assistants-createassistant-tools) in the tool when **creating the assistant** or **creating the run**.
1060
1061The settings you can configure are:
1062
1063- `ranker` - Which ranker to use in determining which chunks to use. The available values are `auto`, which uses the latest available ranker, and `default_2024_08_21`.
1064- `score_threshold` - a ranking between 0.0 and 1.0, with 1.0 being the highest ranking. A higher number will constrain the file chunks used to generate a result to only chunks with a higher possible relevance, at the cost of potentially leaving out relevant chunks.
1065- `hybrid_search.embedding_weight` (also referred to as `rrf_embedding_weight`) - determines how much weight to give to semantic similarity when combining dense (embedding) and sparse (text) rankings with [reciprocal rank fusion](https://en.wikipedia.org/wiki/Reciprocal_rank_fusion). Increase this weight to favor chunks that are close in embedding space.
1066- `hybrid_search.text_weight` (also referred to as `rrf_text_weight`) - determines how much weight to give to keyword/text matching when hybrid search is enabled. Increase this weight to favor chunks that share exact terms with the query.
1067
1068At least one of `hybrid_search.embedding_weight` or `hybrid_search.text_weight` must be greater than zero when hybrid search is configured.
1069
1070#### Managing costs with expiration policies
1071
1072The `file_search` tool uses the `vector_stores` object as its resource and you will be billed based on the [size](https://developers.openai.com/api/reference/resources/vector_stores#vector-stores/object-bytes) of the `vector_store` objects created. The size of the vector store object is the sum of all the parsed chunks from your files and their corresponding embeddings.
1073
1074You first GB is free and beyond that, usage is billed at $0.10/GB/day of vector storage. There are no other costs associated with vector store operations.
1075
1076In order to help you manage the costs associated with these `vector_store` objects, we have added support for expiration policies in the `vector_store` object. You can set these policies when creating or updating the `vector_store` object.
1077
1078```javascript
1079let vectorStore = await openai.vectorStores.create({
1080 name: "rag-store",
1081 file_ids: [
1082 "file_1",
1083 "file_2",
1084 "file_3",
1085 "file_4",
1086 "file_5",
1087 ],
1088 expires_after: {
1089 anchor: "last_active_at",
1090 days: 7,
1091 },
1092});
1093```
1094
1095```python
1096vector_store = client.vector_stores.create(
1097 name="Product Documentation",
1098 file_ids=[
1099 "file_1",
1100 "file_2",
1101 "file_3",
1102 "file_4",
1103 "file_5",
1104 ],
1105 expires_after={"anchor": "last_active_at", "days": 7},
1106)
1107```
1108
1109```go
1110vectorStore, err := client.VectorStores.New(context.Background(), openai.VectorStoreNewParams{
1111 Name: openai.String("Product Documentation"),
1112 FileIDs: []string{"file_1", "file_2", "file_3", "file_4", "file_5"},
1113 ExpiresAfter: openai.VectorStoreNewParamsExpiresAfter{Days: 7},
1114})
1115if err != nil {
1116 panic(err)
1117}
1118```
1119
1120```java
1121import com.openai.client.OpenAIClient;
1122import com.openai.client.okhttp.OpenAIOkHttpClient;
1123import com.openai.core.JsonValue;
1124import com.openai.models.vectorstores.VectorStoreCreateParams;
1125
1126String fileId1 = "file_1";
1127
1128String fileId2 = "file_2";
1129
1130String fileId3 = "file_3";
1131
1132String fileId4 = "file_4";
1133
1134String fileId5 = "file_5";
1135
1136var store =
1137 client
1138 .vectorStores()
1139 .create(
1140 VectorStoreCreateParams.builder()
1141 .name("Product Documentation")
1142 .addFileId(fileId1)
1143 .addFileId(fileId2)
1144 .addFileId(fileId3)
1145 .addFileId(fileId4)
1146 .addFileId(fileId5)
1147 .expiresAfter(
1148 VectorStoreCreateParams.ExpiresAfter.builder()
1149 .anchor(JsonValue.from("last_active_at"))
1150 .days(7)
1151 .build())
1152 .build());
1153
1154System.out.println(store.id());
1155```
1156
1157```ruby
1158require "openai"
1159
1160client = OpenAI::Client.new
1161store = client.vector_stores.create(
1162 name: "Product Documentation",
1163 file_ids: [
1164 "file_1",
1165 "file_2",
1166 "file_3",
1167 "file_4",
1168 "file_5"
1169 ],
1170 expires_after: {anchor: :last_active_at, days: 7}
1171)
1172puts(store.id)
1173```
1174
1175
1176**Thread vector stores have default expiration policies**
1177
1178Vector stores created using thread helpers (like [`tool_resources.file_search.vector_stores`](https://developers.openai.com/api/reference/resources/beta/subresources/threads/methods/create#threads-createthread-tool_resources) in Threads or [message.attachments](https://developers.openai.com/api/reference/resources/beta/subresources/threads/subresources/messages/methods/create#messages-createmessage-attachments) in Messages) have a default expiration policy of 7 days after they were last active (defined as the last time the vector store was part of a run).
1179
1180When a vector store expires, runs on that thread will fail. To fix this, you can simply recreate a new `vector_store` with the same files and reattach it to the thread.
1181
1182```javascript
1183const fileIds = [];
1184for await (const file of openai.vectorStores.files.list(
1185 "vs_expired"
1186)) {
1187 fileIds.push(file.id);
1188}
1189
1190const vectorStore = await openai.vectorStores.create({
1191 name: "rag-store",
1192});
1193await openai.beta.threads.update("thread_abc123", {
1194 tool_resources: { file_search: { vector_store_ids: [vectorStore.id] } },
1195});
1196
1197for (const fileBatch of _.chunk(fileIds, 100)) {
1198 await openai.vectorStores.fileBatches.create(vectorStore.id, {
1199 file_ids: fileBatch,
1200 });
1201}
1202```
1203
1204```python
1205all_files = list(client.vector_stores.files.list("vs_expired"))
1206
1207vector_store = client.vector_stores.create(name="rag-store")
1208client.beta.threads.update(
1209 "thread_abc123",
1210 tool_resources={"file_search": {"vector_store_ids": [vector_store.id]}},
1211)
1212
1213for file_batch in chunked(all_files, 100):
1214 client.vector_stores.file_batches.create_and_poll(
1215 vector_store_id=vector_store.id,
1216 file_ids=[file.id for file in file_batch],
1217 )
1218```
1219
1220```go
1221pager := client.VectorStores.Files.ListAutoPaging(context.Background(), "vs_expired", openai.VectorStoreFileListParams{})
1222fileIDs := make([]string, 0)
1223for pager.Next() {
1224 fileIDs = append(fileIDs, pager.Current().ID)
1225}
1226if err := pager.Err(); err != nil {
1227 panic(err)
1228}
1229vectorStore, err := client.VectorStores.New(context.Background(), openai.VectorStoreNewParams{
1230 Name: openai.String("rag-store"),
1231})
1232if err != nil {
1233 panic(err)
1234}
1235_, err = client.Beta.Threads.Update(context.Background(), "thread_abc123", openai.BetaThreadUpdateParams{
1236 ToolResources: openai.BetaThreadUpdateParamsToolResources{
1237 FileSearch: openai.BetaThreadUpdateParamsToolResourcesFileSearch{VectorStoreIDs: []string{vectorStore.ID}},
1238 },
1239})
1240if err != nil {
1241 panic(err)
1242}
1243for start := 0; start < len(fileIDs); start += 100 {
1244 end := min(start+100, len(fileIDs))
1245 if _, err := client.VectorStores.FileBatches.NewAndPoll(context.Background(), vectorStore.ID, openai.VectorStoreFileBatchNewParams{
1246 FileIDs: fileIDs[start:end],
1247 }, 1000); err != nil {
1248 panic(err)
1249 }
1250}
1251```
1252
1253```java
1254import com.openai.client.OpenAIClient;
1255import com.openai.client.okhttp.OpenAIOkHttpClient;
1256import com.openai.models.beta.threads.ThreadUpdateParams;
1257import com.openai.models.vectorstores.VectorStoreCreateParams;
1258import com.openai.models.vectorstores.filebatches.FileBatchCreateParams;
1259import com.openai.models.vectorstores.filebatches.FileBatchRetrieveParams;
1260import com.openai.models.vectorstores.filebatches.VectorStoreFileBatch;
1261import java.util.ArrayList;
1262
1263String vectorStoreId = "vs_expired";
1264String threadId = "thread_abc123";
1265var files = client.vectorStores().files().list(vectorStoreId).autoPager();
1266var replacement =
1267 client.vectorStores().create(VectorStoreCreateParams.builder().name("rag-store").build());
1268client
1269 .beta()
1270 .threads()
1271 .update(
1272 threadId,
1273 ThreadUpdateParams.builder()
1274 .toolResources(
1275 ThreadUpdateParams.ToolResources.builder()
1276 .fileSearch(
1277 ThreadUpdateParams.ToolResources.FileSearch.builder()
1278 .addVectorStoreId(replacement.id())
1279 .build())
1280 .build())
1281 .build());
1282
1283var fileIds = new ArrayList<String>();
1284for (var file : files) fileIds.add(file.id());
1285for (int offset = 0; offset < fileIds.size(); offset += 100) {
1286 var ids = fileIds.subList(offset, Math.min(offset + 100, fileIds.size()));
1287 var batch =
1288 client
1289 .vectorStores()
1290 .fileBatches()
1291 .create(replacement.id(), FileBatchCreateParams.builder().fileIds(ids).build());
1292 while (batch.status().equals(VectorStoreFileBatch.Status.IN_PROGRESS)) {
1293 Thread.sleep(1000);
1294 batch =
1295 client
1296 .vectorStores()
1297 .fileBatches()
1298 .retrieve(
1299 batch.id(),
1300 FileBatchRetrieveParams.builder().vectorStoreId(replacement.id()).build());
1301 }
1302 if (!batch.status().equals(VectorStoreFileBatch.Status.COMPLETED)) {
1303 throw new IllegalStateException("File batch ended with status: " + batch.status());
1304 }
1305}
1306System.out.println(replacement.id());
1307```
1308
1309```ruby
1310require "openai"
1311
1312client = OpenAI::Client.new
1313files = client.vector_stores.files.list("vs_expired")
1314store = client.vector_stores.create(name: "rag-store")
1315client.beta.threads.update(
1316 "thread_abc123",
1317 tool_resources: {file_search: {vector_store_ids: [store.id]}}
1318)
1319file_ids = []
1320files.auto_paging_each { |file| file_ids << file.id }
1321file_ids.each_slice(100) do |batch_ids|
1322 batch = client.vector_stores.file_batches.create(store.id, file_ids: batch_ids)
1323 while batch.status == OpenAI::VectorStores::VectorStoreFileBatch::Status::IN_PROGRESS
1324 sleep(2)
1325 batch = client.vector_stores.file_batches.retrieve(
1326 batch.id,
1327 vector_store_id: store.id
1328 )
1329 end
1330 unless batch.status == OpenAI::VectorStores::VectorStoreFileBatch::Status::COMPLETED
1331 raise "File batch ended with status: #{batch.status}"
1332 end
1333end
1334puts(store.id)
1335```
1336
1337
1338## Supported files
1339
1340_For `text/` MIME types, the encoding must be one of `utf-8`, `utf-16`, or `ascii`._
1341
1342{/* Keep this table in sync with RETRIEVAL_SUPPORTED_EXTENSIONS in the agentapi service */}
1343
1344| File format | MIME type |
1345| ----------- | --------------------------------------------------------------------------- |
1346| `.c` | `text/x-c` |
1347| `.cpp` | `text/x-c++` |
1348| `.cs` | `text/x-csharp` |
1349| `.css` | `text/css` |
1350| `.doc` | `application/msword` |
1351| `.docx` | `application/vnd.openxmlformats-officedocument.wordprocessingml.document` |
1352| `.go` | `text/x-golang` |
1353| `.html` | `text/html` |
1354| `.java` | `text/x-java` |
1355| `.js` | `text/javascript` |
1356| `.json` | `application/json` |
1357| `.md` | `text/markdown` |
1358| `.pdf` | `application/pdf` |
1359| `.php` | `text/x-php` |
1360| `.pptx` | `application/vnd.openxmlformats-officedocument.presentationml.presentation` |
1361| `.py` | `text/x-python` |
1362| `.py` | `text/x-script.python` |
1363| `.rb` | `text/x-ruby` |
1364| `.sh` | `application/x-sh` |
1365| `.tex` | `text/x-tex` |
1366| `.ts` | `application/typescript` |
1367| `.txt` | `text/plain` |