From 4e4c3433fcf3f36af98f635a96a7af50309afbb4 Mon Sep 17 00:00:00 2001 From: Paul Date: Wed, 26 Jun 2024 21:08:02 +0000 Subject: [PATCH] Hotfix llama3 --- schemas/meta_meta-llama-3-70b-instruct.json | 346 ++++---------------- schemas/meta_meta-llama-3-8b-instruct.json | 315 ++++-------------- 2 files changed, 134 insertions(+), 527 deletions(-) diff --git a/schemas/meta_meta-llama-3-70b-instruct.json b/schemas/meta_meta-llama-3-70b-instruct.json index a1584bd..4bd449f 100644 --- a/schemas/meta_meta-llama-3-70b-instruct.json +++ b/schemas/meta_meta-llama-3-70b-instruct.json @@ -4,236 +4,16 @@ "name": "meta-llama-3-70b-instruct", "description": "A 70 billion parameter language model from Meta, fine tuned for chat completions", "visibility": "public", - "github_url": "https://github.com/meta-llama/llama3", + "github_url": null, "paper_url": null, - "license_url": "https://github.com/meta-llama/llama3/blob/main/LICENSE", - "run_count": 47428809, - "cover_image_url": "https://tjzk.replicate.delivery/models_models_cover_image/12ed0c29-9236-4a21-ac5a-7faff3045ab1/meta-logo.png", - "default_example": { - "id": "7zr9g2asx9rgj0cey8698vm4d4", - "model": "meta/meta-llama-3-70b-instruct", - "version": "fbfb20b472b2f3bdd101412a9f70a0ed4fc0ced78a77ff00970ee7a2383c575d", - "status": "succeeded", - "input": { - "top_p": 0.9, - "prompt": "Work through this problem step by step:\n\nQ: Sarah has 7 llamas. Her friend gives her 3 more trucks of llamas. Each truck has 5 llamas. How many llamas does Sarah have in total?", - "max_tokens": 512, - "min_tokens": 0, - "temperature": 0.6, - "prompt_template": "<|begin_of_text|><|start_header_id|>system<|end_header_id|>\n\nYou are a helpful assistant<|eot_id|><|start_header_id|>user<|end_header_id|>\n\n{prompt}<|eot_id|><|start_header_id|>assistant<|end_header_id|>\n\n", - "presence_penalty": 1.15, - "frequency_penalty": 0.2 - }, - "output": [ - "Let", - "'s", - " break", - " this", - " problem", - " down", - " step", - " by", - " step", - ".\n\n", - "Step", - " ", - "1", - ":", - " Sarah", - " already", - " has", - " ", - "7", - " ll", - "amas", - ".\n\n", - "Step", - " ", - "2", - ":", - " Her", - " friend", - " gives", - " her", - " ", - "3", - " trucks", - " of", - " ll", - "amas", - ".", - " We", - " need", - " to", - " find", - " out", - " how", - " many", - " ll", - "amas", - " are", - " in", - " these", - " ", - "3", - " trucks", - ".\n\n", - "Step", - " ", - "3", - ":", - " Each", - " truck", - " has", - " ", - "5", - " ll", - "amas", - ",", - " so", - " we", - " multiply", - " the", - " number", - " of", - " trucks", - " (", - "3", - ")", - " by", - " the", - " number", - " of", - " ll", - "amas", - " per", - " truck", - " (", - "5", - "):\n\n", - "3", - " trucks", - " x", - " ", - "5", - " ll", - "amas", - "/tr", - "uck", - " =", - " ", - "3", - " x", - " ", - "5", - " =", - " ", - "15", - " ll", - "amas", - "\n\n", - "Step", - " ", - "4", - ":", - " Sarah", - " already", - " had", - " ", - "7", - " ll", - "amas", - ",", - " and", - " now", - " she", - " gets", - " ", - "15", - " more", - " ll", - "amas", - " from", - " her", - " friend", - ".", - " To", - " find", - " the", - " total", - " number", - " of", - " ll", - "amas", - " Sarah", - " has", - ",", - " we", - " add", - " the", - " two", - " numbers", - " together", - ":\n\n", - "7", - " ll", - "amas", - " (", - "already", - " had", - ")", - " +", - " ", - "15", - " ll", - "amas", - " (", - "from", - " her", - " friend", - ")", - " =", - " ", - "22", - " ll", - "amas", - "\n\n", - "Therefore", - ",", - " Sarah", - " has", - " a", - " total", - " of", - " ", - "22", - " ll", - "amas", - ".", - "" - ], - "logs": "", - "error": "", - "metrics": { - "total_time": 3.47, - "input_token_count": 54, - "tokens_per_second": 41.72998441835564, - "output_token_count": 166, - "predict_time": 4.045511 - }, - "created_at": "2024-04-18T16:31:19.530000Z", - "started_at": "2024-04-18T16:31:19Z", - "completed_at": "2024-04-18T16:31:23Z", - "urls": { - "stream": "https://streaming-api.svc.us.c.replicate.net/v1/streams/cvg64spwkdzlpdltjkmv6jotmwn2cq5btgxirmsbejvsxx7xp2la", - "get": "https://api.replicate.com/v1/predictions/7zr9g2asx9rgj0cey8698vm4d4", - "cancel": "https://api.replicate.com/v1/predictions/7zr9g2asx9rgj0cey8698vm4d4/cancel" - } - }, + "license_url": null, + "run_count": 0, + "cover_image_url": null, + "default_example": null, "latest_version": { - "id": "fbfb20b472b2f3bdd101412a9f70a0ed4fc0ced78a77ff00970ee7a2383c575d", - "created_at": "2024-04-17T21:58:54.806446+00:00", - "cog_version": "0.9.4", + "id": "2ce90869e475b267cdbb7f2b74b00851c0f3307edda4bde88eb88085832d891e", + "created_at": "2024-06-19T19:19:18.233909+00:00", + "cog_version": "0.10.0-alpha13", "openapi_schema": { "info": { "title": "Cog", @@ -258,24 +38,6 @@ "operationId": "root__get" } }, - "/ready": { - "get": { - "summary": "Ready", - "responses": { - "200": { - "content": { - "application/json": { - "schema": { - "title": "Response Ready Ready Get" - } - } - }, - "description": "Successful Response" - } - }, - "operationId": "ready_ready_get" - } - }, "/shutdown": { "post": { "summary": "Start Shutdown", @@ -472,69 +234,105 @@ "Input": { "type": "object", "title": "Input", + "required": [ + "prompt" + ], "properties": { + "seed": { + "type": "integer", + "title": "Seed", + "x-order": 10, + "description": "Random seed. Leave blank to randomize the seed." + }, "top_k": { "type": "integer", "title": "Top K", - "default": 50, - "x-order": 5, - "description": "The number of highest probability tokens to consider for generating the output. If > 0, only keep the top k tokens with highest probability (top-k filtering)." + "default": 0, + "minimum": -1, + "x-order": 6, + "description": "When decoding text, samples from the top k most likely tokens; lower to ignore less likely tokens." }, "top_p": { "type": "number", "title": "Top P", - "default": 0.9, - "x-order": 4, - "description": "A probability threshold for generating the output. If < 1.0, only keep the top tokens with cumulative probability >= top_p (nucleus filtering). Nucleus filtering is described in Holtzman et al. (http://arxiv.org/abs/1904.09751)." + "default": 0.95, + "maximum": 1, + "minimum": 0, + "x-order": 5, + "description": "When decoding text, samples from the top p percentage of most likely tokens; lower to ignore less likely tokens." }, "prompt": { "type": "string", "title": "Prompt", - "default": "", "x-order": 0, - "description": "Prompt" + "description": "Prompt to send to the model." }, "max_tokens": { "type": "integer", "title": "Max Tokens", "default": 512, + "minimum": 1, "x-order": 2, - "description": "The maximum number of tokens the model should generate as output." + "description": "Maximum number of tokens to generate. A word is generally 2-3 tokens." }, "min_tokens": { "type": "integer", "title": "Min Tokens", - "default": 0, - "x-order": 1, - "description": "The minimum number of tokens the model should generate as output." + "minimum": -1, + "x-order": 3, + "description": "Minimum number of tokens to generate. To disable, set to -1. A word is generally 2-3 tokens." }, "temperature": { "type": "number", "title": "Temperature", - "default": 0.6, - "x-order": 3, - "description": "The value used to modulate the next token probabilities." + "default": 0.7, + "maximum": 5, + "minimum": 0, + "x-order": 4, + "description": "Adjusts randomness of outputs, greater than 1 is random and 0 is deterministic, 0.75 is a good starting value." + }, + "system_prompt": { + "type": "string", + "title": "System Prompt", + "default": "You are a helpful assistant", + "x-order": 1, + "description": "System prompt to send to the model. This is prepended to the prompt and helps guide system behavior." + }, + "length_penalty": { + "type": "number", + "title": "Length Penalty", + "default": 1, + "maximum": 5, + "minimum": 0, + "x-order": 8, + "description": "A parameter that controls how long the outputs are. If < 1, the model will tend to generate shorter outputs, and > 1 will tend to generate longer outputs." + }, + "stop_sequences": { + "type": "string", + "title": "Stop Sequences", + "default": "<|end_of_text|>,<|eot_id|>", + "x-order": 7, + "description": "A comma-separated list of sequences to stop generation at. For example, ',' will stop generation at the first instance of 'end' or ''." }, "prompt_template": { "type": "string", "title": "Prompt Template", - "default": "{prompt}", - "x-order": 8, - "description": "Prompt template. The string `{prompt}` will be substituted for the input prompt. If you want to generate dialog output, use this template as a starting point and construct the prompt string manually, leaving `prompt_template={prompt}`." + "default": "<|begin_of_text|><|start_header_id|>system<|end_header_id|>\n\n{system_prompt}<|eot_id|><|start_header_id|>user<|end_header_id|>\n\n{prompt}<|eot_id|><|start_header_id|>assistant<|end_header_id|>", + "x-order": 11, + "description": "Template for formatting the prompt. Can be an arbitrary string, but must contain the substring `{prompt}`." }, "presence_penalty": { "type": "number", "title": "Presence Penalty", - "default": 1.15, - "x-order": 6, - "description": "Presence penalty" + "default": 0, + "x-order": 9, + "description": "A parameter that penalizes repeated tokens regardless of the number of appearances. As the value increases, the model will be less likely to repeat tokens in the output." }, - "frequency_penalty": { - "type": "number", - "title": "Frequency Penalty", - "default": 0.2, - "x-order": 7, - "description": "Frequency penalty" + "log_performance_metrics": { + "type": "boolean", + "title": "Log Performance Metrics", + "default": false, + "x-order": 12 } } }, @@ -712,4 +510,4 @@ } } } -} \ No newline at end of file +} diff --git a/schemas/meta_meta-llama-3-8b-instruct.json b/schemas/meta_meta-llama-3-8b-instruct.json index 585c1b7..532c8e1 100644 --- a/schemas/meta_meta-llama-3-8b-instruct.json +++ b/schemas/meta_meta-llama-3-8b-instruct.json @@ -4,204 +4,16 @@ "name": "meta-llama-3-8b-instruct", "description": "An 8 billion parameter language model from Meta, fine tuned for chat completions", "visibility": "public", - "github_url": "https://github.com/meta-llama/llama3", + "github_url": null, "paper_url": null, - "license_url": "https://github.com/meta-llama/llama3/blob/main/LICENSE", - "run_count": 21431865, - "cover_image_url": "https://tjzk.replicate.delivery/models_models_cover_image/927d3fce-75e3-4af9-92da-f537bc34072a/meta-logo.png", - "default_example": { - "id": "855g9wxd7hrgp0cf7tsv1ewzgc", - "model": "replicate-internal/llama-3-8b-instruct-int8-triton", - "version": "b63acc3f54e3c08cb3b081f049ebc881420035dfc6db48f554530e9c4bc02ba3", - "status": "succeeded", - "input": { - "top_p": 0.95, - "prompt": "Johnny has 8 billion parameters. His friend Tommy has 70 billion parameters. What does this mean when it comes to speed?", - "temperature": 0.7, - "system_prompt": "You are a helpful assistant", - "length_penalty": 1, - "max_new_tokens": 512, - "stop_sequences": "<|end_of_text|>,<|eot_id|>", - "prompt_template": "<|begin_of_text|><|start_header_id|>system<|end_header_id|>\n\n{system_prompt}<|eot_id|><|start_header_id|>user<|end_header_id|>\n\n{prompt}<|eot_id|><|start_header_id|>assistant<|end_header_id|>\n\n", - "presence_penalty": 0 - }, - "output": [ - "The", - " number", - " of", - " parameters", - " in", - " a", - " neural", - " network", - " can", - " impact", - " its", - " speed", - ",", - " but", - " it", - "'s", - " not", - " the", - " only", - " factor", - ".\n\n", - "In", - " general", - ",", - " a", - " larger", - " number", - " of", - " parameters", - " can", - " lead", - " to", - ":\n\n", - "1", - ".", - " Increased", - " computational", - " complexity", - ":", - " More", - " parameters", - " mean", - " more", - " calculations", - " are", - " required", - " to", - " process", - " the", - " data", - ".\n", - "2", - ".", - " Increased", - " memory", - " requirements", - ":", - " Larger", - " models", - " require", - " more", - " memory", - " to", - " store", - " their", - " parameters", - ",", - " which", - " can", - " impact", - " system", - " performance", - ".\n\n", - "However", - ",", - " it", - "'s", - " worth", - " noting", - " that", - " the", - " relationship", - " between", - " the", - " number", - " of", - " parameters", - " and", - " speed", - " is", - " not", - " always", - " linear", - ".", - " Other", - " factors", - ",", - " such", - " as", - ":\n\n", - "*", - " Model", - " architecture", - "\n", - "*", - " Optim", - "izer", - " choice", - "\n", - "*", - " Hyper", - "parameter", - " tuning", - "\n\n", - "can", - " also", - " impact", - " the", - " speed", - " of", - " a", - " neural", - " network", - ".\n\n", - "In", - " the", - " case", - " of", - " Johnny", - " and", - " Tommy", - ",", - " it", - "'s", - " difficult", - " to", - " say", - " which", - " one", - "'s", - " model", - " will", - " be", - " faster", - " without", - " more", - " information", - " about", - " the", - " models", - " themselves", - "." - ], - "logs": "Random seed used: `57440`\nNote: Random seed will not impact output if greedy decoding is used.\nFormatted prompt: `<|begin_of_text|><|start_header_id|>system<|end_header_id|>\n\nYou are a helpful assistant<|eot_id|><|start_header_id|>user<|end_header_id|>\n\nJohnny has 8 billion parameters. His friend Tommy has 70 billion parameters. What does this mean when it comes to speed?<|eot_id|><|start_header_id|>assistant<|end_header_id|>\n\n`Random seed used: `57440`\nNote: Random seed will not impact output if greedy decoding is used.\nFormatted prompt: `<|begin_of_text|><|start_header_id|>system<|end_header_id|>\n\nYou are a helpful assistant<|eot_id|><|start_header_id|>user<|end_header_id|>\n\nJohnny has 8 billion parameters. His friend Tommy has 70 billion parameters. What does this mean when it comes to speed?<|eot_id|><|start_header_id|>assistant<|end_header_id|>\n\n`", - "error": null, - "metrics": { - "total_time": 1.657073, - "input_token_count": 39, - "tokens_per_second": 92.80206135476371, - "output_token_count": 149, - "predict_time": 1.652461, - "time_to_first_token": 0.060728942999999994 - }, - "created_at": "2024-05-03T13:45:13.788000Z", - "started_at": "2024-05-03T13:45:13.792612Z", - "completed_at": "2024-05-03T13:45:15.445073Z", - "urls": { - "stream": "https://streaming-api.svc.us.c.replicate.net/v1/streams/hscsfwedhigbbnorfpq7c4i3lbac5srhaqgvb4b5m2iuof3rotwq", - "get": "https://api.replicate.com/v1/predictions/855g9wxd7hrgp0cf7tsv1ewzgc", - "cancel": "https://api.replicate.com/v1/predictions/855g9wxd7hrgp0cf7tsv1ewzgc/cancel" - } - }, + "license_url": null, + "run_count": 0, + "cover_image_url": null, + "default_example": null, "latest_version": { - "id": "5a6809ca6288247d06daf6365557e5e429063f32a21146b2a807c682652136b8", - "created_at": "2024-04-17T21:58:45.726276+00:00", - "cog_version": "0.9.4", + "id": "b63acc3f54e3c08cb3b081f049ebc881420035dfc6db48f554530e9c4bc02ba3", + "created_at": "2024-04-19T21:06:45.886873+00:00", + "cog_version": "0.10.0-alpha5", "openapi_schema": { "info": { "title": "Cog", @@ -226,24 +38,6 @@ "operationId": "root__get" } }, - "/ready": { - "get": { - "summary": "Ready", - "responses": { - "200": { - "content": { - "application/json": { - "schema": { - "title": "Response Ready Ready Get" - } - } - }, - "description": "Successful Response" - } - }, - "operationId": "ready_ready_get" - } - }, "/shutdown": { "post": { "summary": "Start Shutdown", @@ -440,69 +234,84 @@ "Input": { "type": "object", "title": "Input", + "required": [ + "prompt" + ], "properties": { + "seed": { + "type": "integer", + "title": "Seed", + "x-order": 10, + "description": "Random seed. Leave blank to randomize the seed" + }, "top_k": { "type": "integer", "title": "Top K", - "default": 50, - "x-order": 5, - "description": "The number of highest probability tokens to consider for generating the output. If > 0, only keep the top k tokens with highest probability (top-k filtering)." + "default": 0, + "minimum": -1, + "x-order": 6, + "description": "When decoding text, samples from the top k most likely tokens; lower to ignore less likely tokens" }, "top_p": { "type": "number", "title": "Top P", - "default": 0.9, - "x-order": 4, - "description": "A probability threshold for generating the output. If < 1.0, only keep the top tokens with cumulative probability >= top_p (nucleus filtering). Nucleus filtering is described in Holtzman et al. (http://arxiv.org/abs/1904.09751)." + "default": 0.95, + "maximum": 1, + "minimum": 0, + "x-order": 5, + "description": "When decoding text, samples from the top p percentage of most likely tokens; lower to ignore less likely tokens" }, "prompt": { "type": "string", "title": "Prompt", - "default": "", "x-order": 0, - "description": "Prompt" - }, - "max_tokens": { - "type": "integer", - "title": "Max Tokens", - "default": 512, - "x-order": 2, - "description": "The maximum number of tokens the model should generate as output." - }, - "min_tokens": { - "type": "integer", - "title": "Min Tokens", - "default": 0, - "x-order": 1, - "description": "The minimum number of tokens the model should generate as output." + "description": "Prompt to send to the model." }, "temperature": { "type": "number", "title": "Temperature", - "default": 0.6, - "x-order": 3, - "description": "The value used to modulate the next token probabilities." + "default": 0.7, + "maximum": 5, + "minimum": 0, + "x-order": 4, + "description": "Adjusts randomness of outputs, greater than 1 is random and 0 is deterministic, 0.75 is a good starting value." + }, + "system_prompt": { + "type": "string", + "title": "System Prompt", + "default": "You are a helpful assistant", + "x-order": 1, + "description": "System prompt to send to the model. This is prepended to the prompt and helps guide system behavior." + }, + "length_penalty": { + "type": "number", + "title": "Length Penalty", + "default": 1, + "maximum": 5, + "minimum": 0, + "x-order": 8, + "description": "A parameter that controls how long the outputs are. If < 1, the model will tend to generate shorter outputs, and > 1 will tend to generate longer outputs." + }, + "stop_sequences": { + "type": "string", + "title": "Stop Sequences", + "default": "<|end_of_text|>,<|eot_id|>", + "x-order": 7, + "description": "A comma-separated list of sequences to stop generation at. For example, ',' will stop generation at the first instance of 'end' or ''." }, "prompt_template": { "type": "string", "title": "Prompt Template", - "default": "{prompt}", - "x-order": 8, - "description": "Prompt template. The string `{prompt}` will be substituted for the input prompt. If you want to generate dialog output, use this template as a starting point and construct the prompt string manually, leaving `prompt_template={prompt}`." + "default": "<|begin_of_text|><|start_header_id|>system<|end_header_id|>\n\n{system_prompt}<|eot_id|><|start_header_id|>user<|end_header_id|>\n\n{prompt}<|eot_id|><|start_header_id|>assistant<|end_header_id|>\n\n", + "x-order": 11, + "description": "Template for formatting the prompt. Can be an arbitrary string, but must contain the substring `{prompt}`." }, "presence_penalty": { "type": "number", "title": "Presence Penalty", - "default": 1.15, - "x-order": 6, - "description": "Presence penalty" - }, - "frequency_penalty": { - "type": "number", - "title": "Frequency Penalty", - "default": 0.2, - "x-order": 7, - "description": "Frequency penalty" + "default": 0, + "x-order": 9, + "description": "A parameter that penalizes repeated tokens regardless of the number of appearances. As the value increases, the model will be less likely to repeat tokens in the output." } } }, @@ -680,4 +489,4 @@ } } } -} \ No newline at end of file +}