You signed in with another tab or window. Reload to refresh your session.You signed out in another tab or window. Reload to refresh your session.You switched accounts on another tab or window. Reload to refresh your session.Dismiss alert
Copy file name to clipboardExpand all lines: .speakeasy/out.openapi.yaml
+178Lines changed: 178 additions & 0 deletions
Original file line number
Diff line number
Diff line change
@@ -21800,6 +21800,19 @@ components:
21800
21800
model_id: 'openai/gpt-4'
21801
21801
model_name: 'GPT-4'
21802
21802
name: 'OpenAI: GPT-4'
21803
+
perf_last_30m_by_workload:
21804
+
text_generation:
21805
+
latency:
21806
+
p50: 250
21807
+
p75: 350
21808
+
p90: 480
21809
+
p99: 850
21810
+
request_count: 1000
21811
+
throughput:
21812
+
p50: 45.2
21813
+
p75: 38.5
21814
+
p90: 28.3
21815
+
p99: 15.1
21803
21816
pricing:
21804
21817
completion: '0.00006'
21805
21818
image: '0'
@@ -21850,6 +21863,171 @@ components:
21850
21863
type: 'string'
21851
21864
name:
21852
21865
type: 'string'
21866
+
perf_last_30m_by_workload:
21867
+
additionalProperties: false
21868
+
description: 'Endpoint performance over the last 30 minutes, keyed by the kind of request served (e.g. `text_generation`, `image_generation`). Additive to the legacy singular latency and throughput fields; image and video generation report end-to-end latency. Only visible when authenticated with an API key or cookie.'
21869
+
properties:
21870
+
embeddings:
21871
+
properties:
21872
+
latency:
21873
+
allOf:
21874
+
- $ref: '#/components/schemas/PercentileStats'
21875
+
- description: 'Latency percentiles in milliseconds for this workload. Image and video generation report full end-to-end generation time, because their raw latency only measures request acknowledgement; every other workload reports time to first token or result.'
21876
+
request_count:
21877
+
description: 'Total requests admitted for this workload in the window.'
21878
+
type:
21879
+
- 'integer'
21880
+
- 'null'
21881
+
throughput:
21882
+
allOf:
21883
+
- $ref: '#/components/schemas/PercentileStats'
21884
+
- description: 'Throughput percentiles in tokens per second. Only meaningful for text generation; null for workloads without token throughput.'
21885
+
required:
21886
+
- 'latency'
21887
+
- 'throughput'
21888
+
- 'request_count'
21889
+
type: 'object'
21890
+
image_generation:
21891
+
properties:
21892
+
latency:
21893
+
allOf:
21894
+
- $ref: '#/components/schemas/PercentileStats'
21895
+
- description: 'Latency percentiles in milliseconds for this workload. Image and video generation report full end-to-end generation time, because their raw latency only measures request acknowledgement; every other workload reports time to first token or result.'
21896
+
request_count:
21897
+
description: 'Total requests admitted for this workload in the window.'
21898
+
type:
21899
+
- 'integer'
21900
+
- 'null'
21901
+
throughput:
21902
+
allOf:
21903
+
- $ref: '#/components/schemas/PercentileStats'
21904
+
- description: 'Throughput percentiles in tokens per second. Only meaningful for text generation; null for workloads without token throughput.'
21905
+
required:
21906
+
- 'latency'
21907
+
- 'throughput'
21908
+
- 'request_count'
21909
+
type: 'object'
21910
+
rerank:
21911
+
properties:
21912
+
latency:
21913
+
allOf:
21914
+
- $ref: '#/components/schemas/PercentileStats'
21915
+
- description: 'Latency percentiles in milliseconds for this workload. Image and video generation report full end-to-end generation time, because their raw latency only measures request acknowledgement; every other workload reports time to first token or result.'
21916
+
request_count:
21917
+
description: 'Total requests admitted for this workload in the window.'
21918
+
type:
21919
+
- 'integer'
21920
+
- 'null'
21921
+
throughput:
21922
+
allOf:
21923
+
- $ref: '#/components/schemas/PercentileStats'
21924
+
- description: 'Throughput percentiles in tokens per second. Only meaningful for text generation; null for workloads without token throughput.'
21925
+
required:
21926
+
- 'latency'
21927
+
- 'throughput'
21928
+
- 'request_count'
21929
+
type: 'object'
21930
+
stt:
21931
+
properties:
21932
+
latency:
21933
+
allOf:
21934
+
- $ref: '#/components/schemas/PercentileStats'
21935
+
- description: 'Latency percentiles in milliseconds for this workload. Image and video generation report full end-to-end generation time, because their raw latency only measures request acknowledgement; every other workload reports time to first token or result.'
21936
+
request_count:
21937
+
description: 'Total requests admitted for this workload in the window.'
21938
+
type:
21939
+
- 'integer'
21940
+
- 'null'
21941
+
throughput:
21942
+
allOf:
21943
+
- $ref: '#/components/schemas/PercentileStats'
21944
+
- description: 'Throughput percentiles in tokens per second. Only meaningful for text generation; null for workloads without token throughput.'
21945
+
required:
21946
+
- 'latency'
21947
+
- 'throughput'
21948
+
- 'request_count'
21949
+
type: 'object'
21950
+
text_generation:
21951
+
properties:
21952
+
latency:
21953
+
allOf:
21954
+
- $ref: '#/components/schemas/PercentileStats'
21955
+
- description: 'Latency percentiles in milliseconds for this workload. Image and video generation report full end-to-end generation time, because their raw latency only measures request acknowledgement; every other workload reports time to first token or result.'
21956
+
request_count:
21957
+
description: 'Total requests admitted for this workload in the window.'
21958
+
type:
21959
+
- 'integer'
21960
+
- 'null'
21961
+
throughput:
21962
+
allOf:
21963
+
- $ref: '#/components/schemas/PercentileStats'
21964
+
- description: 'Throughput percentiles in tokens per second. Only meaningful for text generation; null for workloads without token throughput.'
21965
+
required:
21966
+
- 'latency'
21967
+
- 'throughput'
21968
+
- 'request_count'
21969
+
type: 'object'
21970
+
tts:
21971
+
properties:
21972
+
latency:
21973
+
allOf:
21974
+
- $ref: '#/components/schemas/PercentileStats'
21975
+
- description: 'Latency percentiles in milliseconds for this workload. Image and video generation report full end-to-end generation time, because their raw latency only measures request acknowledgement; every other workload reports time to first token or result.'
21976
+
request_count:
21977
+
description: 'Total requests admitted for this workload in the window.'
21978
+
type:
21979
+
- 'integer'
21980
+
- 'null'
21981
+
throughput:
21982
+
allOf:
21983
+
- $ref: '#/components/schemas/PercentileStats'
21984
+
- description: 'Throughput percentiles in tokens per second. Only meaningful for text generation; null for workloads without token throughput.'
21985
+
required:
21986
+
- 'latency'
21987
+
- 'throughput'
21988
+
- 'request_count'
21989
+
type: 'object'
21990
+
unknown:
21991
+
properties:
21992
+
latency:
21993
+
allOf:
21994
+
- $ref: '#/components/schemas/PercentileStats'
21995
+
- description: 'Latency percentiles in milliseconds for this workload. Image and video generation report full end-to-end generation time, because their raw latency only measures request acknowledgement; every other workload reports time to first token or result.'
21996
+
request_count:
21997
+
description: 'Total requests admitted for this workload in the window.'
21998
+
type:
21999
+
- 'integer'
22000
+
- 'null'
22001
+
throughput:
22002
+
allOf:
22003
+
- $ref: '#/components/schemas/PercentileStats'
22004
+
- description: 'Throughput percentiles in tokens per second. Only meaningful for text generation; null for workloads without token throughput.'
22005
+
required:
22006
+
- 'latency'
22007
+
- 'throughput'
22008
+
- 'request_count'
22009
+
type: 'object'
22010
+
video_generation:
22011
+
properties:
22012
+
latency:
22013
+
allOf:
22014
+
- $ref: '#/components/schemas/PercentileStats'
22015
+
- description: 'Latency percentiles in milliseconds for this workload. Image and video generation report full end-to-end generation time, because their raw latency only measures request acknowledgement; every other workload reports time to first token or result.'
22016
+
request_count:
22017
+
description: 'Total requests admitted for this workload in the window.'
22018
+
type:
22019
+
- 'integer'
22020
+
- 'null'
22021
+
throughput:
22022
+
allOf:
22023
+
- $ref: '#/components/schemas/PercentileStats'
22024
+
- description: 'Throughput percentiles in tokens per second. Only meaningful for text generation; null for workloads without token throughput.'
0 commit comments