Spaces:
Sleeping
Sleeping
Upload folder using huggingface_hub (part 7)
Browse filesThis view is limited to 50 files because it contains too many changes. See raw diff
- node_modules/@huggingface/tasks/dist/esm/tasks/image-classification/data.d.ts.map +1 -0
- node_modules/@huggingface/tasks/dist/esm/tasks/image-classification/data.js +83 -0
- node_modules/@huggingface/tasks/dist/esm/tasks/image-classification/inference.d.ts +54 -0
- node_modules/@huggingface/tasks/dist/esm/tasks/image-classification/inference.d.ts.map +1 -0
- node_modules/@huggingface/tasks/dist/esm/tasks/image-classification/inference.js +1 -0
- node_modules/@huggingface/tasks/dist/esm/tasks/image-feature-extraction/data.d.ts +4 -0
- node_modules/@huggingface/tasks/dist/esm/tasks/image-feature-extraction/data.d.ts.map +1 -0
- node_modules/@huggingface/tasks/dist/esm/tasks/image-feature-extraction/data.js +60 -0
- node_modules/@huggingface/tasks/dist/esm/tasks/image-segmentation/data.d.ts +4 -0
- node_modules/@huggingface/tasks/dist/esm/tasks/image-segmentation/data.d.ts.map +1 -0
- node_modules/@huggingface/tasks/dist/esm/tasks/image-segmentation/data.js +93 -0
- node_modules/@huggingface/tasks/dist/esm/tasks/image-segmentation/inference.d.ts +68 -0
- node_modules/@huggingface/tasks/dist/esm/tasks/image-segmentation/inference.d.ts.map +1 -0
- node_modules/@huggingface/tasks/dist/esm/tasks/image-segmentation/inference.js +1 -0
- node_modules/@huggingface/tasks/dist/esm/tasks/image-text-to-image/data.d.ts +4 -0
- node_modules/@huggingface/tasks/dist/esm/tasks/image-text-to-image/data.d.ts.map +1 -0
- node_modules/@huggingface/tasks/dist/esm/tasks/image-text-to-image/data.js +48 -0
- node_modules/@huggingface/tasks/dist/esm/tasks/image-text-to-image/inference.d.ts +76 -0
- node_modules/@huggingface/tasks/dist/esm/tasks/image-text-to-image/inference.d.ts.map +1 -0
- node_modules/@huggingface/tasks/dist/esm/tasks/image-text-to-image/inference.js +1 -0
- node_modules/@huggingface/tasks/dist/esm/tasks/image-text-to-text/data.d.ts +4 -0
- node_modules/@huggingface/tasks/dist/esm/tasks/image-text-to-text/data.d.ts.map +1 -0
- node_modules/@huggingface/tasks/dist/esm/tasks/image-text-to-text/data.js +81 -0
- node_modules/@huggingface/tasks/dist/esm/tasks/image-text-to-video/data.d.ts +4 -0
- node_modules/@huggingface/tasks/dist/esm/tasks/image-text-to-video/data.d.ts.map +1 -0
- node_modules/@huggingface/tasks/dist/esm/tasks/image-text-to-video/data.js +48 -0
- node_modules/@huggingface/tasks/dist/esm/tasks/image-text-to-video/inference.d.ts +78 -0
- node_modules/@huggingface/tasks/dist/esm/tasks/image-text-to-video/inference.d.ts.map +1 -0
- node_modules/@huggingface/tasks/dist/esm/tasks/image-text-to-video/inference.js +1 -0
- node_modules/@huggingface/tasks/dist/esm/tasks/image-to-3d/data.d.ts +4 -0
- node_modules/@huggingface/tasks/dist/esm/tasks/image-to-3d/data.d.ts.map +1 -0
- node_modules/@huggingface/tasks/dist/esm/tasks/image-to-3d/data.js +72 -0
- node_modules/@huggingface/tasks/dist/esm/tasks/image-to-image/data.d.ts +4 -0
- node_modules/@huggingface/tasks/dist/esm/tasks/image-to-image/data.d.ts.map +1 -0
- node_modules/@huggingface/tasks/dist/esm/tasks/image-to-image/data.js +89 -0
- node_modules/@huggingface/tasks/dist/esm/tasks/image-to-image/inference.d.ts +69 -0
- node_modules/@huggingface/tasks/dist/esm/tasks/image-to-image/inference.d.ts.map +1 -0
- node_modules/@huggingface/tasks/dist/esm/tasks/image-to-image/inference.js +1 -0
- node_modules/@huggingface/tasks/dist/esm/tasks/image-to-text/data.d.ts +4 -0
- node_modules/@huggingface/tasks/dist/esm/tasks/image-to-text/data.d.ts.map +1 -0
- node_modules/@huggingface/tasks/dist/esm/tasks/image-to-text/data.js +58 -0
- node_modules/@huggingface/tasks/dist/esm/tasks/image-to-text/inference.d.ts +135 -0
- node_modules/@huggingface/tasks/dist/esm/tasks/image-to-text/inference.d.ts.map +1 -0
- node_modules/@huggingface/tasks/dist/esm/tasks/image-to-text/inference.js +1 -0
- node_modules/@huggingface/tasks/dist/esm/tasks/image-to-video/data.d.ts +4 -0
- node_modules/@huggingface/tasks/dist/esm/tasks/image-to-video/data.d.ts.map +1 -0
- node_modules/@huggingface/tasks/dist/esm/tasks/image-to-video/data.js +117 -0
- node_modules/@huggingface/tasks/dist/esm/tasks/image-to-video/inference.d.ts +75 -0
- node_modules/@huggingface/tasks/dist/esm/tasks/image-to-video/inference.d.ts.map +1 -0
- node_modules/@huggingface/tasks/dist/esm/tasks/image-to-video/inference.js +1 -0
node_modules/@huggingface/tasks/dist/esm/tasks/image-classification/data.d.ts.map
ADDED
|
@@ -0,0 +1 @@
|
|
|
|
|
|
|
| 1 |
+
{"version":3,"file":"data.d.ts","sourceRoot":"","sources":["../../../../src/tasks/image-classification/data.ts"],"names":[],"mappings":"AAAA,OAAO,KAAK,EAAE,cAAc,EAAE,MAAM,aAAa,CAAC;AAElD,QAAA,MAAM,QAAQ,EAAE,cAkFf,CAAC;AAEF,eAAe,QAAQ,CAAC"}
|
node_modules/@huggingface/tasks/dist/esm/tasks/image-classification/data.js
ADDED
|
@@ -0,0 +1,83 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
const taskData = {
|
| 2 |
+
datasets: [
|
| 3 |
+
{
|
| 4 |
+
// TODO write proper description
|
| 5 |
+
description: "Benchmark dataset used for image classification with images that belong to 100 classes.",
|
| 6 |
+
id: "cifar100",
|
| 7 |
+
},
|
| 8 |
+
{
|
| 9 |
+
// TODO write proper description
|
| 10 |
+
description: "Dataset consisting of images of garments.",
|
| 11 |
+
id: "fashion_mnist",
|
| 12 |
+
},
|
| 13 |
+
],
|
| 14 |
+
demo: {
|
| 15 |
+
inputs: [
|
| 16 |
+
{
|
| 17 |
+
filename: "image-classification-input.jpeg",
|
| 18 |
+
type: "img",
|
| 19 |
+
},
|
| 20 |
+
],
|
| 21 |
+
outputs: [
|
| 22 |
+
{
|
| 23 |
+
type: "chart",
|
| 24 |
+
data: [
|
| 25 |
+
{
|
| 26 |
+
label: "Egyptian cat",
|
| 27 |
+
score: 0.514,
|
| 28 |
+
},
|
| 29 |
+
{
|
| 30 |
+
label: "Tabby cat",
|
| 31 |
+
score: 0.193,
|
| 32 |
+
},
|
| 33 |
+
{
|
| 34 |
+
label: "Tiger cat",
|
| 35 |
+
score: 0.068,
|
| 36 |
+
},
|
| 37 |
+
],
|
| 38 |
+
},
|
| 39 |
+
],
|
| 40 |
+
},
|
| 41 |
+
metrics: [
|
| 42 |
+
{
|
| 43 |
+
description: "",
|
| 44 |
+
id: "accuracy",
|
| 45 |
+
},
|
| 46 |
+
{
|
| 47 |
+
description: "",
|
| 48 |
+
id: "recall",
|
| 49 |
+
},
|
| 50 |
+
{
|
| 51 |
+
description: "",
|
| 52 |
+
id: "precision",
|
| 53 |
+
},
|
| 54 |
+
{
|
| 55 |
+
description: "",
|
| 56 |
+
id: "f1",
|
| 57 |
+
},
|
| 58 |
+
],
|
| 59 |
+
models: [
|
| 60 |
+
{
|
| 61 |
+
description: "A strong image classification model.",
|
| 62 |
+
id: "google/vit-base-patch16-224",
|
| 63 |
+
},
|
| 64 |
+
{
|
| 65 |
+
description: "A robust image classification model.",
|
| 66 |
+
id: "facebook/deit-base-distilled-patch16-224",
|
| 67 |
+
},
|
| 68 |
+
{
|
| 69 |
+
description: "A strong image classification model.",
|
| 70 |
+
id: "facebook/convnext-large-224",
|
| 71 |
+
},
|
| 72 |
+
],
|
| 73 |
+
spaces: [
|
| 74 |
+
{
|
| 75 |
+
description: "A leaderboard to evaluate different image classification models.",
|
| 76 |
+
id: "timm/leaderboard",
|
| 77 |
+
},
|
| 78 |
+
],
|
| 79 |
+
summary: "Image classification is the task of assigning a label or class to an entire image. Images are expected to have only one class for each image. Image classification models take an image as input and return a prediction about which class the image belongs to.",
|
| 80 |
+
widgetModels: ["google/vit-base-patch16-224"],
|
| 81 |
+
youtubeId: "tjAIM7BOYhw",
|
| 82 |
+
};
|
| 83 |
+
export default taskData;
|
node_modules/@huggingface/tasks/dist/esm/tasks/image-classification/inference.d.ts
ADDED
|
@@ -0,0 +1,54 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
/**
|
| 2 |
+
* Inference code generated from the JSON schema spec in ./spec
|
| 3 |
+
*
|
| 4 |
+
* Using src/scripts/inference-codegen
|
| 5 |
+
*/
|
| 6 |
+
/**
|
| 7 |
+
* Inputs for Image Classification inference
|
| 8 |
+
*/
|
| 9 |
+
export interface ImageClassificationInput {
|
| 10 |
+
/**
|
| 11 |
+
* The input image data as a base64-encoded string. If no `parameters` are provided, you can
|
| 12 |
+
* also provide the image data as a raw bytes payload.
|
| 13 |
+
*/
|
| 14 |
+
inputs: Blob;
|
| 15 |
+
/**
|
| 16 |
+
* Additional inference parameters for Image Classification
|
| 17 |
+
*/
|
| 18 |
+
parameters?: ImageClassificationParameters;
|
| 19 |
+
[property: string]: unknown;
|
| 20 |
+
}
|
| 21 |
+
/**
|
| 22 |
+
* Additional inference parameters for Image Classification
|
| 23 |
+
*/
|
| 24 |
+
export interface ImageClassificationParameters {
|
| 25 |
+
/**
|
| 26 |
+
* The function to apply to the model outputs in order to retrieve the scores.
|
| 27 |
+
*/
|
| 28 |
+
function_to_apply?: ClassificationOutputTransform;
|
| 29 |
+
/**
|
| 30 |
+
* When specified, limits the output to the top K most probable classes.
|
| 31 |
+
*/
|
| 32 |
+
top_k?: number;
|
| 33 |
+
[property: string]: unknown;
|
| 34 |
+
}
|
| 35 |
+
/**
|
| 36 |
+
* The function to apply to the model outputs in order to retrieve the scores.
|
| 37 |
+
*/
|
| 38 |
+
export type ClassificationOutputTransform = "sigmoid" | "softmax" | "none";
|
| 39 |
+
export type ImageClassificationOutput = ImageClassificationOutputElement[];
|
| 40 |
+
/**
|
| 41 |
+
* Outputs of inference for the Image Classification task
|
| 42 |
+
*/
|
| 43 |
+
export interface ImageClassificationOutputElement {
|
| 44 |
+
/**
|
| 45 |
+
* The predicted class label.
|
| 46 |
+
*/
|
| 47 |
+
label: string;
|
| 48 |
+
/**
|
| 49 |
+
* The corresponding probability.
|
| 50 |
+
*/
|
| 51 |
+
score: number;
|
| 52 |
+
[property: string]: unknown;
|
| 53 |
+
}
|
| 54 |
+
//# sourceMappingURL=inference.d.ts.map
|
node_modules/@huggingface/tasks/dist/esm/tasks/image-classification/inference.d.ts.map
ADDED
|
@@ -0,0 +1 @@
|
|
|
|
|
|
|
| 1 |
+
{"version":3,"file":"inference.d.ts","sourceRoot":"","sources":["../../../../src/tasks/image-classification/inference.ts"],"names":[],"mappings":"AAAA;;;;GAIG;AACH;;GAEG;AACH,MAAM,WAAW,wBAAwB;IACxC;;;OAGG;IACH,MAAM,EAAE,IAAI,CAAC;IACb;;OAEG;IACH,UAAU,CAAC,EAAE,6BAA6B,CAAC;IAC3C,CAAC,QAAQ,EAAE,MAAM,GAAG,OAAO,CAAC;CAC5B;AACD;;GAEG;AACH,MAAM,WAAW,6BAA6B;IAC7C;;OAEG;IACH,iBAAiB,CAAC,EAAE,6BAA6B,CAAC;IAClD;;OAEG;IACH,KAAK,CAAC,EAAE,MAAM,CAAC;IACf,CAAC,QAAQ,EAAE,MAAM,GAAG,OAAO,CAAC;CAC5B;AACD;;GAEG;AACH,MAAM,MAAM,6BAA6B,GAAG,SAAS,GAAG,SAAS,GAAG,MAAM,CAAC;AAC3E,MAAM,MAAM,yBAAyB,GAAG,gCAAgC,EAAE,CAAC;AAC3E;;GAEG;AACH,MAAM,WAAW,gCAAgC;IAChD;;OAEG;IACH,KAAK,EAAE,MAAM,CAAC;IACd;;OAEG;IACH,KAAK,EAAE,MAAM,CAAC;IACd,CAAC,QAAQ,EAAE,MAAM,GAAG,OAAO,CAAC;CAC5B"}
|
node_modules/@huggingface/tasks/dist/esm/tasks/image-classification/inference.js
ADDED
|
@@ -0,0 +1 @@
|
|
|
|
|
|
|
| 1 |
+
export {};
|
node_modules/@huggingface/tasks/dist/esm/tasks/image-feature-extraction/data.d.ts
ADDED
|
@@ -0,0 +1,4 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
import type { TaskDataCustom } from "../index.js";
|
| 2 |
+
declare const taskData: TaskDataCustom;
|
| 3 |
+
export default taskData;
|
| 4 |
+
//# sourceMappingURL=data.d.ts.map
|
node_modules/@huggingface/tasks/dist/esm/tasks/image-feature-extraction/data.d.ts.map
ADDED
|
@@ -0,0 +1 @@
|
|
|
|
|
|
|
| 1 |
+
{"version":3,"file":"data.d.ts","sourceRoot":"","sources":["../../../../src/tasks/image-feature-extraction/data.ts"],"names":[],"mappings":"AAAA,OAAO,KAAK,EAAE,cAAc,EAAE,MAAM,aAAa,CAAC;AAElD,QAAA,MAAM,QAAQ,EAAE,cA2Df,CAAC;AAEF,eAAe,QAAQ,CAAC"}
|
node_modules/@huggingface/tasks/dist/esm/tasks/image-feature-extraction/data.js
ADDED
|
@@ -0,0 +1,60 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
const taskData = {
|
| 2 |
+
datasets: [
|
| 3 |
+
{
|
| 4 |
+
description: "ImageNet-1K is a image classification dataset in which images are used to train image-feature-extraction models.",
|
| 5 |
+
id: "imagenet-1k",
|
| 6 |
+
},
|
| 7 |
+
],
|
| 8 |
+
demo: {
|
| 9 |
+
inputs: [
|
| 10 |
+
{
|
| 11 |
+
filename: "mask-generation-input.png",
|
| 12 |
+
type: "img",
|
| 13 |
+
},
|
| 14 |
+
],
|
| 15 |
+
outputs: [
|
| 16 |
+
{
|
| 17 |
+
table: [
|
| 18 |
+
["Dimension 1", "Dimension 2", "Dimension 3"],
|
| 19 |
+
["0.21236686408519745", "1.0919708013534546", "0.8512550592422485"],
|
| 20 |
+
["0.809657871723175", "-0.18544459342956543", "-0.7851548194885254"],
|
| 21 |
+
["1.3103108406066895", "-0.2479034662246704", "-0.9107287526130676"],
|
| 22 |
+
["1.8536205291748047", "-0.36419737339019775", "0.09717650711536407"],
|
| 23 |
+
],
|
| 24 |
+
type: "tabular",
|
| 25 |
+
},
|
| 26 |
+
],
|
| 27 |
+
},
|
| 28 |
+
metrics: [],
|
| 29 |
+
models: [
|
| 30 |
+
{
|
| 31 |
+
description: "A powerful image feature extraction model.",
|
| 32 |
+
id: "timm/vit_large_patch14_dinov2.lvd142m",
|
| 33 |
+
},
|
| 34 |
+
{
|
| 35 |
+
description: "A strong image feature extraction model.",
|
| 36 |
+
id: "nvidia/MambaVision-T-1K",
|
| 37 |
+
},
|
| 38 |
+
{
|
| 39 |
+
description: "A robust image feature extraction model.",
|
| 40 |
+
id: "facebook/dino-vitb16",
|
| 41 |
+
},
|
| 42 |
+
{
|
| 43 |
+
description: "Cutting-edge image feature extraction model.",
|
| 44 |
+
id: "apple/aimv2-large-patch14-336-distilled",
|
| 45 |
+
},
|
| 46 |
+
{
|
| 47 |
+
description: "Strong image feature extraction model that can be used on images and documents.",
|
| 48 |
+
id: "OpenGVLab/InternViT-6B-448px-V1-2",
|
| 49 |
+
},
|
| 50 |
+
],
|
| 51 |
+
spaces: [
|
| 52 |
+
{
|
| 53 |
+
description: "A leaderboard to evaluate different image-feature-extraction models on classification performances",
|
| 54 |
+
id: "timm/leaderboard",
|
| 55 |
+
},
|
| 56 |
+
],
|
| 57 |
+
summary: "Image feature extraction is the task of extracting features learnt in a computer vision model.",
|
| 58 |
+
widgetModels: [],
|
| 59 |
+
};
|
| 60 |
+
export default taskData;
|
node_modules/@huggingface/tasks/dist/esm/tasks/image-segmentation/data.d.ts
ADDED
|
@@ -0,0 +1,4 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
import type { TaskDataCustom } from "../index.js";
|
| 2 |
+
declare const taskData: TaskDataCustom;
|
| 3 |
+
export default taskData;
|
| 4 |
+
//# sourceMappingURL=data.d.ts.map
|
node_modules/@huggingface/tasks/dist/esm/tasks/image-segmentation/data.d.ts.map
ADDED
|
@@ -0,0 +1 @@
|
|
|
|
|
|
|
| 1 |
+
{"version":3,"file":"data.d.ts","sourceRoot":"","sources":["../../../../src/tasks/image-segmentation/data.ts"],"names":[],"mappings":"AAAA,OAAO,KAAK,EAAE,cAAc,EAAE,MAAM,aAAa,CAAC;AAElD,QAAA,MAAM,QAAQ,EAAE,cA8Ff,CAAC;AAEF,eAAe,QAAQ,CAAC"}
|
node_modules/@huggingface/tasks/dist/esm/tasks/image-segmentation/data.js
ADDED
|
@@ -0,0 +1,93 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
const taskData = {
|
| 2 |
+
datasets: [
|
| 3 |
+
{
|
| 4 |
+
description: "Scene segmentation dataset.",
|
| 5 |
+
id: "scene_parse_150",
|
| 6 |
+
},
|
| 7 |
+
],
|
| 8 |
+
demo: {
|
| 9 |
+
inputs: [
|
| 10 |
+
{
|
| 11 |
+
filename: "image-segmentation-input.jpeg",
|
| 12 |
+
type: "img",
|
| 13 |
+
},
|
| 14 |
+
],
|
| 15 |
+
outputs: [
|
| 16 |
+
{
|
| 17 |
+
filename: "image-segmentation-output.png",
|
| 18 |
+
type: "img",
|
| 19 |
+
},
|
| 20 |
+
],
|
| 21 |
+
},
|
| 22 |
+
metrics: [
|
| 23 |
+
{
|
| 24 |
+
description: "Average Precision (AP) is the Area Under the PR Curve (AUC-PR). It is calculated for each semantic class separately",
|
| 25 |
+
id: "Average Precision",
|
| 26 |
+
},
|
| 27 |
+
{
|
| 28 |
+
description: "Mean Average Precision (mAP) is the overall average of the AP values",
|
| 29 |
+
id: "Mean Average Precision",
|
| 30 |
+
},
|
| 31 |
+
{
|
| 32 |
+
description: "Intersection over Union (IoU) is the overlap of segmentation masks. Mean IoU is the average of the IoU of all semantic classes",
|
| 33 |
+
id: "Mean Intersection over Union",
|
| 34 |
+
},
|
| 35 |
+
{
|
| 36 |
+
description: "APα is the Average Precision at the IoU threshold of a α value, for example, AP50 and AP75",
|
| 37 |
+
id: "APα",
|
| 38 |
+
},
|
| 39 |
+
],
|
| 40 |
+
models: [
|
| 41 |
+
{
|
| 42 |
+
// TO DO: write description
|
| 43 |
+
description: "Solid panoptic segmentation model trained on COCO.",
|
| 44 |
+
id: "tue-mps/coco_panoptic_eomt_large_640",
|
| 45 |
+
},
|
| 46 |
+
{
|
| 47 |
+
description: "Background removal model.",
|
| 48 |
+
id: "briaai/RMBG-1.4",
|
| 49 |
+
},
|
| 50 |
+
{
|
| 51 |
+
description: "A multipurpose image segmentation model for high resolution images.",
|
| 52 |
+
id: "ZhengPeng7/BiRefNet",
|
| 53 |
+
},
|
| 54 |
+
{
|
| 55 |
+
description: "Powerful human-centric image segmentation model.",
|
| 56 |
+
id: "facebook/sapiens-seg-1b",
|
| 57 |
+
},
|
| 58 |
+
{
|
| 59 |
+
description: "Panoptic segmentation model trained on the COCO (common objects) dataset.",
|
| 60 |
+
id: "facebook/mask2former-swin-large-coco-panoptic",
|
| 61 |
+
},
|
| 62 |
+
],
|
| 63 |
+
spaces: [
|
| 64 |
+
{
|
| 65 |
+
description: "A semantic segmentation application that can predict unseen instances out of the box.",
|
| 66 |
+
id: "facebook/ov-seg",
|
| 67 |
+
},
|
| 68 |
+
{
|
| 69 |
+
description: "One of the strongest segmentation applications.",
|
| 70 |
+
id: "jbrinkma/segment-anything",
|
| 71 |
+
},
|
| 72 |
+
{
|
| 73 |
+
description: "A human-centric segmentation model.",
|
| 74 |
+
id: "facebook/sapiens-pose",
|
| 75 |
+
},
|
| 76 |
+
{
|
| 77 |
+
description: "An instance segmentation application to predict neuronal cell types from microscopy images.",
|
| 78 |
+
id: "rashmi/sartorius-cell-instance-segmentation",
|
| 79 |
+
},
|
| 80 |
+
{
|
| 81 |
+
description: "An application that segments videos.",
|
| 82 |
+
id: "ArtGAN/Segment-Anything-Video",
|
| 83 |
+
},
|
| 84 |
+
{
|
| 85 |
+
description: "An panoptic segmentation application built for outdoor environments.",
|
| 86 |
+
id: "segments/panoptic-segment-anything",
|
| 87 |
+
},
|
| 88 |
+
],
|
| 89 |
+
summary: "Image Segmentation divides an image into segments where each pixel in the image is mapped to an object. This task has multiple variants such as instance segmentation, panoptic segmentation and semantic segmentation.",
|
| 90 |
+
widgetModels: ["nvidia/segformer-b0-finetuned-ade-512-512"],
|
| 91 |
+
youtubeId: "dKE8SIt9C-w",
|
| 92 |
+
};
|
| 93 |
+
export default taskData;
|
node_modules/@huggingface/tasks/dist/esm/tasks/image-segmentation/inference.d.ts
ADDED
|
@@ -0,0 +1,68 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
/**
|
| 2 |
+
* Inference code generated from the JSON schema spec in ./spec
|
| 3 |
+
*
|
| 4 |
+
* Using src/scripts/inference-codegen
|
| 5 |
+
*/
|
| 6 |
+
/**
|
| 7 |
+
* Inputs for Image Segmentation inference
|
| 8 |
+
*/
|
| 9 |
+
export interface ImageSegmentationInput {
|
| 10 |
+
/**
|
| 11 |
+
* The input image data as a base64-encoded string. If no `parameters` are provided, you can
|
| 12 |
+
* also provide the image data as a raw bytes payload.
|
| 13 |
+
*/
|
| 14 |
+
inputs: Blob;
|
| 15 |
+
/**
|
| 16 |
+
* Additional inference parameters for Image Segmentation
|
| 17 |
+
*/
|
| 18 |
+
parameters?: ImageSegmentationParameters;
|
| 19 |
+
[property: string]: unknown;
|
| 20 |
+
}
|
| 21 |
+
/**
|
| 22 |
+
* Additional inference parameters for Image Segmentation
|
| 23 |
+
*/
|
| 24 |
+
export interface ImageSegmentationParameters {
|
| 25 |
+
/**
|
| 26 |
+
* Threshold to use when turning the predicted masks into binary values.
|
| 27 |
+
*/
|
| 28 |
+
mask_threshold?: number;
|
| 29 |
+
/**
|
| 30 |
+
* Mask overlap threshold to eliminate small, disconnected segments.
|
| 31 |
+
*/
|
| 32 |
+
overlap_mask_area_threshold?: number;
|
| 33 |
+
/**
|
| 34 |
+
* Segmentation task to be performed, depending on model capabilities.
|
| 35 |
+
*/
|
| 36 |
+
subtask?: ImageSegmentationSubtask;
|
| 37 |
+
/**
|
| 38 |
+
* Probability threshold to filter out predicted masks.
|
| 39 |
+
*/
|
| 40 |
+
threshold?: number;
|
| 41 |
+
[property: string]: unknown;
|
| 42 |
+
}
|
| 43 |
+
/**
|
| 44 |
+
* Segmentation task to be performed, depending on model capabilities.
|
| 45 |
+
*/
|
| 46 |
+
export type ImageSegmentationSubtask = "instance" | "panoptic" | "semantic";
|
| 47 |
+
export type ImageSegmentationOutput = ImageSegmentationOutputElement[];
|
| 48 |
+
/**
|
| 49 |
+
* Outputs of inference for the Image Segmentation task
|
| 50 |
+
*
|
| 51 |
+
* A predicted mask / segment
|
| 52 |
+
*/
|
| 53 |
+
export interface ImageSegmentationOutputElement {
|
| 54 |
+
/**
|
| 55 |
+
* The label of the predicted segment.
|
| 56 |
+
*/
|
| 57 |
+
label: string;
|
| 58 |
+
/**
|
| 59 |
+
* The corresponding mask as a black-and-white image (base64-encoded).
|
| 60 |
+
*/
|
| 61 |
+
mask: string;
|
| 62 |
+
/**
|
| 63 |
+
* The score or confidence degree the model has.
|
| 64 |
+
*/
|
| 65 |
+
score?: number;
|
| 66 |
+
[property: string]: unknown;
|
| 67 |
+
}
|
| 68 |
+
//# sourceMappingURL=inference.d.ts.map
|
node_modules/@huggingface/tasks/dist/esm/tasks/image-segmentation/inference.d.ts.map
ADDED
|
@@ -0,0 +1 @@
|
|
|
|
|
|
|
| 1 |
+
{"version":3,"file":"inference.d.ts","sourceRoot":"","sources":["../../../../src/tasks/image-segmentation/inference.ts"],"names":[],"mappings":"AAAA;;;;GAIG;AACH;;GAEG;AACH,MAAM,WAAW,sBAAsB;IACtC;;;OAGG;IACH,MAAM,EAAE,IAAI,CAAC;IACb;;OAEG;IACH,UAAU,CAAC,EAAE,2BAA2B,CAAC;IACzC,CAAC,QAAQ,EAAE,MAAM,GAAG,OAAO,CAAC;CAC5B;AACD;;GAEG;AACH,MAAM,WAAW,2BAA2B;IAC3C;;OAEG;IACH,cAAc,CAAC,EAAE,MAAM,CAAC;IACxB;;OAEG;IACH,2BAA2B,CAAC,EAAE,MAAM,CAAC;IACrC;;OAEG;IACH,OAAO,CAAC,EAAE,wBAAwB,CAAC;IACnC;;OAEG;IACH,SAAS,CAAC,EAAE,MAAM,CAAC;IACnB,CAAC,QAAQ,EAAE,MAAM,GAAG,OAAO,CAAC;CAC5B;AACD;;GAEG;AACH,MAAM,MAAM,wBAAwB,GAAG,UAAU,GAAG,UAAU,GAAG,UAAU,CAAC;AAC5E,MAAM,MAAM,uBAAuB,GAAG,8BAA8B,EAAE,CAAC;AACvE;;;;GAIG;AACH,MAAM,WAAW,8BAA8B;IAC9C;;OAEG;IACH,KAAK,EAAE,MAAM,CAAC;IACd;;OAEG;IACH,IAAI,EAAE,MAAM,CAAC;IACb;;OAEG;IACH,KAAK,CAAC,EAAE,MAAM,CAAC;IACf,CAAC,QAAQ,EAAE,MAAM,GAAG,OAAO,CAAC;CAC5B"}
|
node_modules/@huggingface/tasks/dist/esm/tasks/image-segmentation/inference.js
ADDED
|
@@ -0,0 +1 @@
|
|
|
|
|
|
|
| 1 |
+
export {};
|
node_modules/@huggingface/tasks/dist/esm/tasks/image-text-to-image/data.d.ts
ADDED
|
@@ -0,0 +1,4 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
import type { TaskDataCustom } from "../index.js";
|
| 2 |
+
declare const taskData: TaskDataCustom;
|
| 3 |
+
export default taskData;
|
| 4 |
+
//# sourceMappingURL=data.d.ts.map
|
node_modules/@huggingface/tasks/dist/esm/tasks/image-text-to-image/data.d.ts.map
ADDED
|
@@ -0,0 +1 @@
|
|
|
|
|
|
|
| 1 |
+
{"version":3,"file":"data.d.ts","sourceRoot":"","sources":["../../../../src/tasks/image-text-to-image/data.ts"],"names":[],"mappings":"AAAA,OAAO,KAAK,EAAE,cAAc,EAAE,MAAM,aAAa,CAAC;AAElD,QAAA,MAAM,QAAQ,EAAE,cAiDf,CAAC;AAEF,eAAe,QAAQ,CAAC"}
|
node_modules/@huggingface/tasks/dist/esm/tasks/image-text-to-image/data.js
ADDED
|
@@ -0,0 +1,48 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
const taskData = {
|
| 2 |
+
datasets: [],
|
| 3 |
+
demo: {
|
| 4 |
+
inputs: [
|
| 5 |
+
{
|
| 6 |
+
filename: "image-text-to-image-input.jpeg",
|
| 7 |
+
type: "img",
|
| 8 |
+
},
|
| 9 |
+
{
|
| 10 |
+
label: "Input",
|
| 11 |
+
content: "A city above clouds, pastel colors, Victorian style",
|
| 12 |
+
type: "text",
|
| 13 |
+
},
|
| 14 |
+
],
|
| 15 |
+
outputs: [
|
| 16 |
+
{
|
| 17 |
+
filename: "image-text-to-image-output.png",
|
| 18 |
+
type: "img",
|
| 19 |
+
},
|
| 20 |
+
],
|
| 21 |
+
},
|
| 22 |
+
metrics: [
|
| 23 |
+
{
|
| 24 |
+
description: "The Fréchet Inception Distance (FID) calculates the distance between distributions between synthetic and real samples. A lower FID score indicates better similarity between the distributions of real and generated images.",
|
| 25 |
+
id: "FID",
|
| 26 |
+
},
|
| 27 |
+
{
|
| 28 |
+
description: "CLIP Score measures the similarity between the generated image and the text prompt using CLIP embeddings. A higher score indicates better alignment with the text prompt.",
|
| 29 |
+
id: "CLIP",
|
| 30 |
+
},
|
| 31 |
+
],
|
| 32 |
+
models: [
|
| 33 |
+
{
|
| 34 |
+
description: "A powerful model for image-text-to-image generation.",
|
| 35 |
+
id: "black-forest-labs/FLUX.2-dev",
|
| 36 |
+
},
|
| 37 |
+
],
|
| 38 |
+
spaces: [
|
| 39 |
+
{
|
| 40 |
+
description: "An application for image-text-to-image generation.",
|
| 41 |
+
id: "black-forest-labs/FLUX.2-dev",
|
| 42 |
+
},
|
| 43 |
+
],
|
| 44 |
+
summary: "Image-text-to-image models take an image and a text prompt as input and generate a new image based on the reference image and text instructions. These models are useful for image editing, style transfer, image variations, and guided image generation tasks.",
|
| 45 |
+
widgetModels: ["black-forest-labs/FLUX.2-dev"],
|
| 46 |
+
youtubeId: undefined,
|
| 47 |
+
};
|
| 48 |
+
export default taskData;
|
node_modules/@huggingface/tasks/dist/esm/tasks/image-text-to-image/inference.d.ts
ADDED
|
@@ -0,0 +1,76 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
/**
|
| 2 |
+
* Inference code generated from the JSON schema spec in ./spec
|
| 3 |
+
*
|
| 4 |
+
* Using src/scripts/inference-codegen
|
| 5 |
+
*/
|
| 6 |
+
/**
|
| 7 |
+
* Inputs for Image Text To Image inference. Either inputs (image) or prompt (in parameters)
|
| 8 |
+
* must be provided, or both.
|
| 9 |
+
*/
|
| 10 |
+
export interface ImageTextToImageInput {
|
| 11 |
+
/**
|
| 12 |
+
* The input image data as a base64-encoded string. If no `parameters` are provided, you can
|
| 13 |
+
* also provide the image data as a raw bytes payload. Either this or prompt must be
|
| 14 |
+
* provided.
|
| 15 |
+
*/
|
| 16 |
+
inputs?: Blob;
|
| 17 |
+
/**
|
| 18 |
+
* Additional inference parameters for Image Text To Image
|
| 19 |
+
*/
|
| 20 |
+
parameters?: ImageTextToImageParameters;
|
| 21 |
+
[property: string]: unknown;
|
| 22 |
+
}
|
| 23 |
+
/**
|
| 24 |
+
* Additional inference parameters for Image Text To Image
|
| 25 |
+
*/
|
| 26 |
+
export interface ImageTextToImageParameters {
|
| 27 |
+
/**
|
| 28 |
+
* For diffusion models. A higher guidance scale value encourages the model to generate
|
| 29 |
+
* images closely linked to the text prompt at the expense of lower image quality.
|
| 30 |
+
*/
|
| 31 |
+
guidance_scale?: number;
|
| 32 |
+
/**
|
| 33 |
+
* One prompt to guide what NOT to include in image generation.
|
| 34 |
+
*/
|
| 35 |
+
negative_prompt?: string;
|
| 36 |
+
/**
|
| 37 |
+
* For diffusion models. The number of denoising steps. More denoising steps usually lead to
|
| 38 |
+
* a higher quality image at the expense of slower inference.
|
| 39 |
+
*/
|
| 40 |
+
num_inference_steps?: number;
|
| 41 |
+
/**
|
| 42 |
+
* The text prompt to guide the image generation. Either this or inputs (image) must be
|
| 43 |
+
* provided.
|
| 44 |
+
*/
|
| 45 |
+
prompt?: string;
|
| 46 |
+
/**
|
| 47 |
+
* Seed for the random number generator.
|
| 48 |
+
*/
|
| 49 |
+
seed?: number;
|
| 50 |
+
/**
|
| 51 |
+
* The size in pixels of the output image. This parameter is only supported by some
|
| 52 |
+
* providers and for specific models. It will be ignored when unsupported.
|
| 53 |
+
*/
|
| 54 |
+
target_size?: TargetSize;
|
| 55 |
+
[property: string]: unknown;
|
| 56 |
+
}
|
| 57 |
+
/**
|
| 58 |
+
* The size in pixels of the output image. This parameter is only supported by some
|
| 59 |
+
* providers and for specific models. It will be ignored when unsupported.
|
| 60 |
+
*/
|
| 61 |
+
export interface TargetSize {
|
| 62 |
+
height: number;
|
| 63 |
+
width: number;
|
| 64 |
+
[property: string]: unknown;
|
| 65 |
+
}
|
| 66 |
+
/**
|
| 67 |
+
* Outputs of inference for the Image Text To Image task
|
| 68 |
+
*/
|
| 69 |
+
export interface ImageTextToImageOutput {
|
| 70 |
+
/**
|
| 71 |
+
* The generated image returned as raw bytes in the payload.
|
| 72 |
+
*/
|
| 73 |
+
image: unknown;
|
| 74 |
+
[property: string]: unknown;
|
| 75 |
+
}
|
| 76 |
+
//# sourceMappingURL=inference.d.ts.map
|
node_modules/@huggingface/tasks/dist/esm/tasks/image-text-to-image/inference.d.ts.map
ADDED
|
@@ -0,0 +1 @@
|
|
|
|
|
|
|
| 1 |
+
{"version":3,"file":"inference.d.ts","sourceRoot":"","sources":["../../../../src/tasks/image-text-to-image/inference.ts"],"names":[],"mappings":"AAAA;;;;GAIG;AACH;;;GAGG;AACH,MAAM,WAAW,qBAAqB;IACrC;;;;OAIG;IACH,MAAM,CAAC,EAAE,IAAI,CAAC;IACd;;OAEG;IACH,UAAU,CAAC,EAAE,0BAA0B,CAAC;IACxC,CAAC,QAAQ,EAAE,MAAM,GAAG,OAAO,CAAC;CAC5B;AACD;;GAEG;AACH,MAAM,WAAW,0BAA0B;IAC1C;;;OAGG;IACH,cAAc,CAAC,EAAE,MAAM,CAAC;IACxB;;OAEG;IACH,eAAe,CAAC,EAAE,MAAM,CAAC;IACzB;;;OAGG;IACH,mBAAmB,CAAC,EAAE,MAAM,CAAC;IAC7B;;;OAGG;IACH,MAAM,CAAC,EAAE,MAAM,CAAC;IAChB;;OAEG;IACH,IAAI,CAAC,EAAE,MAAM,CAAC;IACd;;;OAGG;IACH,WAAW,CAAC,EAAE,UAAU,CAAC;IACzB,CAAC,QAAQ,EAAE,MAAM,GAAG,OAAO,CAAC;CAC5B;AACD;;;GAGG;AACH,MAAM,WAAW,UAAU;IAC1B,MAAM,EAAE,MAAM,CAAC;IACf,KAAK,EAAE,MAAM,CAAC;IACd,CAAC,QAAQ,EAAE,MAAM,GAAG,OAAO,CAAC;CAC5B;AACD;;GAEG;AACH,MAAM,WAAW,sBAAsB;IACtC;;OAEG;IACH,KAAK,EAAE,OAAO,CAAC;IACf,CAAC,QAAQ,EAAE,MAAM,GAAG,OAAO,CAAC;CAC5B"}
|
node_modules/@huggingface/tasks/dist/esm/tasks/image-text-to-image/inference.js
ADDED
|
@@ -0,0 +1 @@
|
|
|
|
|
|
|
| 1 |
+
export {};
|
node_modules/@huggingface/tasks/dist/esm/tasks/image-text-to-text/data.d.ts
ADDED
|
@@ -0,0 +1,4 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
import type { TaskDataCustom } from "../index.js";
|
| 2 |
+
declare const taskData: TaskDataCustom;
|
| 3 |
+
export default taskData;
|
| 4 |
+
//# sourceMappingURL=data.d.ts.map
|
node_modules/@huggingface/tasks/dist/esm/tasks/image-text-to-text/data.d.ts.map
ADDED
|
@@ -0,0 +1 @@
|
|
|
|
|
|
|
| 1 |
+
{"version":3,"file":"data.d.ts","sourceRoot":"","sources":["../../../../src/tasks/image-text-to-text/data.ts"],"names":[],"mappings":"AAAA,OAAO,KAAK,EAAE,cAAc,EAAE,MAAM,aAAa,CAAC;AAElD,QAAA,MAAM,QAAQ,EAAE,cAiFf,CAAC;AAEF,eAAe,QAAQ,CAAC"}
|
node_modules/@huggingface/tasks/dist/esm/tasks/image-text-to-text/data.js
ADDED
|
@@ -0,0 +1,81 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
const taskData = {
|
| 2 |
+
datasets: [
|
| 3 |
+
{
|
| 4 |
+
description: "Instructions composed of image and text.",
|
| 5 |
+
id: "liuhaotian/LLaVA-Instruct-150K",
|
| 6 |
+
},
|
| 7 |
+
{
|
| 8 |
+
description: "Collection of image-text pairs on scientific topics.",
|
| 9 |
+
id: "DAMO-NLP-SG/multimodal_textbook",
|
| 10 |
+
},
|
| 11 |
+
{
|
| 12 |
+
description: "A collection of datasets made for model fine-tuning.",
|
| 13 |
+
id: "HuggingFaceM4/the_cauldron",
|
| 14 |
+
},
|
| 15 |
+
{
|
| 16 |
+
description: "Screenshots of websites with their HTML/CSS codes.",
|
| 17 |
+
id: "HuggingFaceM4/WebSight",
|
| 18 |
+
},
|
| 19 |
+
],
|
| 20 |
+
demo: {
|
| 21 |
+
inputs: [
|
| 22 |
+
{
|
| 23 |
+
filename: "image-text-to-text-input.png",
|
| 24 |
+
type: "img",
|
| 25 |
+
},
|
| 26 |
+
{
|
| 27 |
+
label: "Text Prompt",
|
| 28 |
+
content: "Describe the position of the bee in detail.",
|
| 29 |
+
type: "text",
|
| 30 |
+
},
|
| 31 |
+
],
|
| 32 |
+
outputs: [
|
| 33 |
+
{
|
| 34 |
+
label: "Answer",
|
| 35 |
+
content: "The bee is sitting on a pink flower, surrounded by other flowers. The bee is positioned in the center of the flower, with its head and front legs sticking out.",
|
| 36 |
+
type: "text",
|
| 37 |
+
},
|
| 38 |
+
],
|
| 39 |
+
},
|
| 40 |
+
metrics: [],
|
| 41 |
+
models: [
|
| 42 |
+
{
|
| 43 |
+
description: "Small and efficient yet powerful vision language model.",
|
| 44 |
+
id: "HuggingFaceTB/SmolVLM-Instruct",
|
| 45 |
+
},
|
| 46 |
+
{
|
| 47 |
+
description: "Cutting-edge reasoning vision language model.",
|
| 48 |
+
id: "zai-org/GLM-4.5V",
|
| 49 |
+
},
|
| 50 |
+
{
|
| 51 |
+
description: "Cutting-edge small vision language model to convert documents to text.",
|
| 52 |
+
id: "rednote-hilab/dots.ocr",
|
| 53 |
+
},
|
| 54 |
+
{
|
| 55 |
+
description: "Small yet powerful model.",
|
| 56 |
+
id: "Qwen/Qwen2.5-VL-3B-Instruct",
|
| 57 |
+
},
|
| 58 |
+
{
|
| 59 |
+
description: "Image-text-to-text model with agentic capabilities.",
|
| 60 |
+
id: "microsoft/Magma-8B",
|
| 61 |
+
},
|
| 62 |
+
],
|
| 63 |
+
spaces: [
|
| 64 |
+
{
|
| 65 |
+
description: "Leaderboard to evaluate vision language models.",
|
| 66 |
+
id: "opencompass/open_vlm_leaderboard",
|
| 67 |
+
},
|
| 68 |
+
{
|
| 69 |
+
description: "An application that compares object detection capabilities of different vision language models.",
|
| 70 |
+
id: "sergiopaniego/vlm_object_understanding",
|
| 71 |
+
},
|
| 72 |
+
{
|
| 73 |
+
description: "An application to compare different OCR models.",
|
| 74 |
+
id: "prithivMLmods/Multimodal-OCR",
|
| 75 |
+
},
|
| 76 |
+
],
|
| 77 |
+
summary: "Image-text-to-text models take in an image and text prompt and output text. These models are also called vision-language models, or VLMs. The difference from image-to-text models is that these models take an additional text input, not restricting the model to certain use cases like image captioning, and may also be trained to accept a conversation as input.",
|
| 78 |
+
widgetModels: ["zai-org/GLM-4.5V"],
|
| 79 |
+
youtubeId: "IoGaGfU1CIg",
|
| 80 |
+
};
|
| 81 |
+
export default taskData;
|
node_modules/@huggingface/tasks/dist/esm/tasks/image-text-to-video/data.d.ts
ADDED
|
@@ -0,0 +1,4 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
import type { TaskDataCustom } from "../index.js";
|
| 2 |
+
declare const taskData: TaskDataCustom;
|
| 3 |
+
export default taskData;
|
| 4 |
+
//# sourceMappingURL=data.d.ts.map
|
node_modules/@huggingface/tasks/dist/esm/tasks/image-text-to-video/data.d.ts.map
ADDED
|
@@ -0,0 +1 @@
|
|
|
|
|
|
|
| 1 |
+
{"version":3,"file":"data.d.ts","sourceRoot":"","sources":["../../../../src/tasks/image-text-to-video/data.ts"],"names":[],"mappings":"AAAA,OAAO,KAAK,EAAE,cAAc,EAAE,MAAM,aAAa,CAAC;AAElD,QAAA,MAAM,QAAQ,EAAE,cAiDf,CAAC;AAEF,eAAe,QAAQ,CAAC"}
|
node_modules/@huggingface/tasks/dist/esm/tasks/image-text-to-video/data.js
ADDED
|
@@ -0,0 +1,48 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
const taskData = {
|
| 2 |
+
datasets: [],
|
| 3 |
+
demo: {
|
| 4 |
+
inputs: [
|
| 5 |
+
{
|
| 6 |
+
filename: "image-text-to-video-input.jpg",
|
| 7 |
+
type: "img",
|
| 8 |
+
},
|
| 9 |
+
{
|
| 10 |
+
label: "Input",
|
| 11 |
+
content: "Darth Vader is surfing on the waves.",
|
| 12 |
+
type: "text",
|
| 13 |
+
},
|
| 14 |
+
],
|
| 15 |
+
outputs: [
|
| 16 |
+
{
|
| 17 |
+
filename: "image-text-to-video-output.gif",
|
| 18 |
+
type: "img",
|
| 19 |
+
},
|
| 20 |
+
],
|
| 21 |
+
},
|
| 22 |
+
metrics: [
|
| 23 |
+
{
|
| 24 |
+
description: "Frechet Video Distance uses a model that captures coherence for changes in frames and the quality of each frame. A smaller score indicates better video generation.",
|
| 25 |
+
id: "fvd",
|
| 26 |
+
},
|
| 27 |
+
{
|
| 28 |
+
description: "CLIPSIM measures similarity between video frames and text using an image-text similarity model. A higher score indicates better video generation.",
|
| 29 |
+
id: "clipsim",
|
| 30 |
+
},
|
| 31 |
+
],
|
| 32 |
+
models: [
|
| 33 |
+
{
|
| 34 |
+
description: "A powerful model for image-text-to-video generation.",
|
| 35 |
+
id: "Lightricks/LTX-Video",
|
| 36 |
+
},
|
| 37 |
+
],
|
| 38 |
+
spaces: [
|
| 39 |
+
{
|
| 40 |
+
description: "An application for image-text-to-video generation.",
|
| 41 |
+
id: "Lightricks/ltx-video-distilled",
|
| 42 |
+
},
|
| 43 |
+
],
|
| 44 |
+
summary: "Image-text-to-video models take an reference image and a text instructions as and generate a video based on them. These models are useful for animating still images, creating dynamic content from static references, and generating videos with specific motion or transformation guidance.",
|
| 45 |
+
widgetModels: ["Lightricks/LTX-Video"],
|
| 46 |
+
youtubeId: undefined,
|
| 47 |
+
};
|
| 48 |
+
export default taskData;
|
node_modules/@huggingface/tasks/dist/esm/tasks/image-text-to-video/inference.d.ts
ADDED
|
@@ -0,0 +1,78 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
/**
|
| 2 |
+
* Inference code generated from the JSON schema spec in ./spec
|
| 3 |
+
*
|
| 4 |
+
* Using src/scripts/inference-codegen
|
| 5 |
+
*/
|
| 6 |
+
/**
|
| 7 |
+
* Inputs for Image Text To Video inference. Either inputs (image) or prompt (in parameters)
|
| 8 |
+
* must be provided, or both.
|
| 9 |
+
*/
|
| 10 |
+
export interface ImageTextToVideoInput {
|
| 11 |
+
/**
|
| 12 |
+
* The input image data as a base64-encoded string. If no `parameters` are provided, you can
|
| 13 |
+
* also provide the image data as a raw bytes payload. Either this or prompt must be
|
| 14 |
+
* provided.
|
| 15 |
+
*/
|
| 16 |
+
inputs?: Blob;
|
| 17 |
+
/**
|
| 18 |
+
* Additional inference parameters for Image Text To Video
|
| 19 |
+
*/
|
| 20 |
+
parameters?: ImageTextToVideoParameters;
|
| 21 |
+
[property: string]: unknown;
|
| 22 |
+
}
|
| 23 |
+
/**
|
| 24 |
+
* Additional inference parameters for Image Text To Video
|
| 25 |
+
*/
|
| 26 |
+
export interface ImageTextToVideoParameters {
|
| 27 |
+
/**
|
| 28 |
+
* For diffusion models. A higher guidance scale value encourages the model to generate
|
| 29 |
+
* videos closely linked to the text prompt at the expense of lower image quality.
|
| 30 |
+
*/
|
| 31 |
+
guidance_scale?: number;
|
| 32 |
+
/**
|
| 33 |
+
* One prompt to guide what NOT to include in video generation.
|
| 34 |
+
*/
|
| 35 |
+
negative_prompt?: string;
|
| 36 |
+
/**
|
| 37 |
+
* The num_frames parameter determines how many video frames are generated.
|
| 38 |
+
*/
|
| 39 |
+
num_frames?: number;
|
| 40 |
+
/**
|
| 41 |
+
* The number of denoising steps. More denoising steps usually lead to a higher quality
|
| 42 |
+
* video at the expense of slower inference.
|
| 43 |
+
*/
|
| 44 |
+
num_inference_steps?: number;
|
| 45 |
+
/**
|
| 46 |
+
* The text prompt to guide the video generation. Either this or inputs (image) must be
|
| 47 |
+
* provided.
|
| 48 |
+
*/
|
| 49 |
+
prompt?: string;
|
| 50 |
+
/**
|
| 51 |
+
* Seed for the random number generator.
|
| 52 |
+
*/
|
| 53 |
+
seed?: number;
|
| 54 |
+
/**
|
| 55 |
+
* The size in pixel of the output video frames.
|
| 56 |
+
*/
|
| 57 |
+
target_size?: TargetSize;
|
| 58 |
+
[property: string]: unknown;
|
| 59 |
+
}
|
| 60 |
+
/**
|
| 61 |
+
* The size in pixel of the output video frames.
|
| 62 |
+
*/
|
| 63 |
+
export interface TargetSize {
|
| 64 |
+
height: number;
|
| 65 |
+
width: number;
|
| 66 |
+
[property: string]: unknown;
|
| 67 |
+
}
|
| 68 |
+
/**
|
| 69 |
+
* Outputs of inference for the Image Text To Video task
|
| 70 |
+
*/
|
| 71 |
+
export interface ImageTextToVideoOutput {
|
| 72 |
+
/**
|
| 73 |
+
* The generated video returned as raw bytes in the payload.
|
| 74 |
+
*/
|
| 75 |
+
video: unknown;
|
| 76 |
+
[property: string]: unknown;
|
| 77 |
+
}
|
| 78 |
+
//# sourceMappingURL=inference.d.ts.map
|
node_modules/@huggingface/tasks/dist/esm/tasks/image-text-to-video/inference.d.ts.map
ADDED
|
@@ -0,0 +1 @@
|
|
|
|
|
|
|
| 1 |
+
{"version":3,"file":"inference.d.ts","sourceRoot":"","sources":["../../../../src/tasks/image-text-to-video/inference.ts"],"names":[],"mappings":"AAAA;;;;GAIG;AACH;;;GAGG;AACH,MAAM,WAAW,qBAAqB;IACrC;;;;OAIG;IACH,MAAM,CAAC,EAAE,IAAI,CAAC;IACd;;OAEG;IACH,UAAU,CAAC,EAAE,0BAA0B,CAAC;IACxC,CAAC,QAAQ,EAAE,MAAM,GAAG,OAAO,CAAC;CAC5B;AACD;;GAEG;AACH,MAAM,WAAW,0BAA0B;IAC1C;;;OAGG;IACH,cAAc,CAAC,EAAE,MAAM,CAAC;IACxB;;OAEG;IACH,eAAe,CAAC,EAAE,MAAM,CAAC;IACzB;;OAEG;IACH,UAAU,CAAC,EAAE,MAAM,CAAC;IACpB;;;OAGG;IACH,mBAAmB,CAAC,EAAE,MAAM,CAAC;IAC7B;;;OAGG;IACH,MAAM,CAAC,EAAE,MAAM,CAAC;IAChB;;OAEG;IACH,IAAI,CAAC,EAAE,MAAM,CAAC;IACd;;OAEG;IACH,WAAW,CAAC,EAAE,UAAU,CAAC;IACzB,CAAC,QAAQ,EAAE,MAAM,GAAG,OAAO,CAAC;CAC5B;AACD;;GAEG;AACH,MAAM,WAAW,UAAU;IAC1B,MAAM,EAAE,MAAM,CAAC;IACf,KAAK,EAAE,MAAM,CAAC;IACd,CAAC,QAAQ,EAAE,MAAM,GAAG,OAAO,CAAC;CAC5B;AACD;;GAEG;AACH,MAAM,WAAW,sBAAsB;IACtC;;OAEG;IACH,KAAK,EAAE,OAAO,CAAC;IACf,CAAC,QAAQ,EAAE,MAAM,GAAG,OAAO,CAAC;CAC5B"}
|
node_modules/@huggingface/tasks/dist/esm/tasks/image-text-to-video/inference.js
ADDED
|
@@ -0,0 +1 @@
|
|
|
|
|
|
|
| 1 |
+
export {};
|
node_modules/@huggingface/tasks/dist/esm/tasks/image-to-3d/data.d.ts
ADDED
|
@@ -0,0 +1,4 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
import type { TaskDataCustom } from "../index.js";
|
| 2 |
+
declare const taskData: TaskDataCustom;
|
| 3 |
+
export default taskData;
|
| 4 |
+
//# sourceMappingURL=data.d.ts.map
|
node_modules/@huggingface/tasks/dist/esm/tasks/image-to-3d/data.d.ts.map
ADDED
|
@@ -0,0 +1 @@
|
|
|
|
|
|
|
| 1 |
+
{"version":3,"file":"data.d.ts","sourceRoot":"","sources":["../../../../src/tasks/image-to-3d/data.ts"],"names":[],"mappings":"AAAA,OAAO,KAAK,EAAE,cAAc,EAAE,MAAM,aAAa,CAAC;AAElD,QAAA,MAAM,QAAQ,EAAE,cAsEf,CAAC;AAEF,eAAe,QAAQ,CAAC"}
|
node_modules/@huggingface/tasks/dist/esm/tasks/image-to-3d/data.js
ADDED
|
@@ -0,0 +1,72 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
const taskData = {
|
| 2 |
+
datasets: [
|
| 3 |
+
{
|
| 4 |
+
description: "A large dataset of over 10 million 3D objects.",
|
| 5 |
+
id: "allenai/objaverse-xl",
|
| 6 |
+
},
|
| 7 |
+
{
|
| 8 |
+
description: "A dataset of isolated object images for evaluating image-to-3D models.",
|
| 9 |
+
id: "dylanebert/iso3d",
|
| 10 |
+
},
|
| 11 |
+
],
|
| 12 |
+
demo: {
|
| 13 |
+
inputs: [
|
| 14 |
+
{
|
| 15 |
+
filename: "image-to-3d-image-input.png",
|
| 16 |
+
type: "img",
|
| 17 |
+
},
|
| 18 |
+
],
|
| 19 |
+
outputs: [
|
| 20 |
+
{
|
| 21 |
+
label: "Result",
|
| 22 |
+
content: "image-to-3d-3d-output-filename.glb",
|
| 23 |
+
type: "text",
|
| 24 |
+
},
|
| 25 |
+
],
|
| 26 |
+
},
|
| 27 |
+
metrics: [],
|
| 28 |
+
models: [
|
| 29 |
+
{
|
| 30 |
+
description: "Fast image-to-3D mesh model by Tencent.",
|
| 31 |
+
id: "TencentARC/InstantMesh",
|
| 32 |
+
},
|
| 33 |
+
{
|
| 34 |
+
description: "3D world generation model.",
|
| 35 |
+
id: "tencent/HunyuanWorld-1",
|
| 36 |
+
},
|
| 37 |
+
{
|
| 38 |
+
description: "A scaled up image-to-3D mesh model derived from TripoSR.",
|
| 39 |
+
id: "hwjiang/Real3D",
|
| 40 |
+
},
|
| 41 |
+
{
|
| 42 |
+
description: "Consistent image-to-3d generation model.",
|
| 43 |
+
id: "stabilityai/stable-point-aware-3d",
|
| 44 |
+
},
|
| 45 |
+
],
|
| 46 |
+
spaces: [
|
| 47 |
+
{
|
| 48 |
+
description: "Leaderboard to evaluate image-to-3D models.",
|
| 49 |
+
id: "dylanebert/3d-arena",
|
| 50 |
+
},
|
| 51 |
+
{
|
| 52 |
+
description: "Image-to-3D demo with mesh outputs.",
|
| 53 |
+
id: "TencentARC/InstantMesh",
|
| 54 |
+
},
|
| 55 |
+
{
|
| 56 |
+
description: "Image-to-3D demo.",
|
| 57 |
+
id: "stabilityai/stable-point-aware-3d",
|
| 58 |
+
},
|
| 59 |
+
{
|
| 60 |
+
description: "Image-to-3D demo with mesh outputs.",
|
| 61 |
+
id: "hwjiang/Real3D",
|
| 62 |
+
},
|
| 63 |
+
{
|
| 64 |
+
description: "Image-to-3D demo with splat outputs.",
|
| 65 |
+
id: "dylanebert/LGM-mini",
|
| 66 |
+
},
|
| 67 |
+
],
|
| 68 |
+
summary: "Image-to-3D models take in image input and produce 3D output.",
|
| 69 |
+
widgetModels: [],
|
| 70 |
+
youtubeId: "",
|
| 71 |
+
};
|
| 72 |
+
export default taskData;
|
node_modules/@huggingface/tasks/dist/esm/tasks/image-to-image/data.d.ts
ADDED
|
@@ -0,0 +1,4 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
import type { TaskDataCustom } from "../index.js";
|
| 2 |
+
declare const taskData: TaskDataCustom;
|
| 3 |
+
export default taskData;
|
| 4 |
+
//# sourceMappingURL=data.d.ts.map
|
node_modules/@huggingface/tasks/dist/esm/tasks/image-to-image/data.d.ts.map
ADDED
|
@@ -0,0 +1 @@
|
|
|
|
|
|
|
| 1 |
+
{"version":3,"file":"data.d.ts","sourceRoot":"","sources":["../../../../src/tasks/image-to-image/data.ts"],"names":[],"mappings":"AAAA,OAAO,KAAK,EAAE,cAAc,EAAE,MAAM,aAAa,CAAC;AAElD,QAAA,MAAM,QAAQ,EAAE,cA2Ff,CAAC;AAEF,eAAe,QAAQ,CAAC"}
|
node_modules/@huggingface/tasks/dist/esm/tasks/image-to-image/data.js
ADDED
|
@@ -0,0 +1,89 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
const taskData = {
|
| 2 |
+
datasets: [
|
| 3 |
+
{
|
| 4 |
+
description: "Synthetic dataset, for image relighting",
|
| 5 |
+
id: "VIDIT",
|
| 6 |
+
},
|
| 7 |
+
{
|
| 8 |
+
description: "Multiple images of celebrities, used for facial expression translation",
|
| 9 |
+
id: "huggan/CelebA-faces",
|
| 10 |
+
},
|
| 11 |
+
{
|
| 12 |
+
description: "12M image-caption pairs.",
|
| 13 |
+
id: "Spawning/PD12M",
|
| 14 |
+
},
|
| 15 |
+
],
|
| 16 |
+
demo: {
|
| 17 |
+
inputs: [
|
| 18 |
+
{
|
| 19 |
+
filename: "image-to-image-input.jpeg",
|
| 20 |
+
type: "img",
|
| 21 |
+
},
|
| 22 |
+
],
|
| 23 |
+
outputs: [
|
| 24 |
+
{
|
| 25 |
+
filename: "image-to-image-output.png",
|
| 26 |
+
type: "img",
|
| 27 |
+
},
|
| 28 |
+
],
|
| 29 |
+
},
|
| 30 |
+
isPlaceholder: false,
|
| 31 |
+
metrics: [
|
| 32 |
+
{
|
| 33 |
+
description: "Peak Signal to Noise Ratio (PSNR) is an approximation of the human perception, considering the ratio of the absolute intensity with respect to the variations. Measured in dB, a high value indicates a high fidelity.",
|
| 34 |
+
id: "PSNR",
|
| 35 |
+
},
|
| 36 |
+
{
|
| 37 |
+
description: "Structural Similarity Index (SSIM) is a perceptual metric which compares the luminance, contrast and structure of two images. The values of SSIM range between -1 and 1, and higher values indicate closer resemblance to the original image.",
|
| 38 |
+
id: "SSIM",
|
| 39 |
+
},
|
| 40 |
+
{
|
| 41 |
+
description: "Inception Score (IS) is an analysis of the labels predicted by an image classification model when presented with a sample of the generated images.",
|
| 42 |
+
id: "IS",
|
| 43 |
+
},
|
| 44 |
+
],
|
| 45 |
+
models: [
|
| 46 |
+
{
|
| 47 |
+
description: "An image-to-image model to improve image resolution.",
|
| 48 |
+
id: "fal/AuraSR-v2",
|
| 49 |
+
},
|
| 50 |
+
{
|
| 51 |
+
description: "Powerful image editing model.",
|
| 52 |
+
id: "black-forest-labs/FLUX.1-Kontext-dev",
|
| 53 |
+
},
|
| 54 |
+
{
|
| 55 |
+
description: "Virtual try-on model.",
|
| 56 |
+
id: "yisol/IDM-VTON",
|
| 57 |
+
},
|
| 58 |
+
{
|
| 59 |
+
description: "Image re-lighting model.",
|
| 60 |
+
id: "kontext-community/relighting-kontext-dev-lora-v3",
|
| 61 |
+
},
|
| 62 |
+
{
|
| 63 |
+
description: "Strong model for inpainting and outpainting.",
|
| 64 |
+
id: "black-forest-labs/FLUX.1-Fill-dev",
|
| 65 |
+
},
|
| 66 |
+
{
|
| 67 |
+
description: "Strong model for image editing using depth maps.",
|
| 68 |
+
id: "black-forest-labs/FLUX.1-Depth-dev-lora",
|
| 69 |
+
},
|
| 70 |
+
],
|
| 71 |
+
spaces: [
|
| 72 |
+
{
|
| 73 |
+
description: "Image editing application.",
|
| 74 |
+
id: "black-forest-labs/FLUX.1-Kontext-Dev",
|
| 75 |
+
},
|
| 76 |
+
{
|
| 77 |
+
description: "Image relighting application.",
|
| 78 |
+
id: "lllyasviel/iclight-v2-vary",
|
| 79 |
+
},
|
| 80 |
+
{
|
| 81 |
+
description: "An application for image upscaling.",
|
| 82 |
+
id: "jasperai/Flux.1-dev-Controlnet-Upscaler",
|
| 83 |
+
},
|
| 84 |
+
],
|
| 85 |
+
summary: "Image-to-image is the task of transforming an input image through a variety of possible manipulations and enhancements, such as super-resolution, image inpainting, colorization, and more.",
|
| 86 |
+
widgetModels: ["Qwen/Qwen-Image"],
|
| 87 |
+
youtubeId: "",
|
| 88 |
+
};
|
| 89 |
+
export default taskData;
|
node_modules/@huggingface/tasks/dist/esm/tasks/image-to-image/inference.d.ts
ADDED
|
@@ -0,0 +1,69 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
/**
|
| 2 |
+
* Inference code generated from the JSON schema spec in ./spec
|
| 3 |
+
*
|
| 4 |
+
* Using src/scripts/inference-codegen
|
| 5 |
+
*/
|
| 6 |
+
/**
|
| 7 |
+
* Inputs for Image To Image inference
|
| 8 |
+
*/
|
| 9 |
+
export interface ImageToImageInput {
|
| 10 |
+
/**
|
| 11 |
+
* The input image data as a base64-encoded string. If no `parameters` are provided, you can
|
| 12 |
+
* also provide the image data as a raw bytes payload.
|
| 13 |
+
*/
|
| 14 |
+
inputs: Blob;
|
| 15 |
+
/**
|
| 16 |
+
* Additional inference parameters for Image To Image
|
| 17 |
+
*/
|
| 18 |
+
parameters?: ImageToImageParameters;
|
| 19 |
+
[property: string]: unknown;
|
| 20 |
+
}
|
| 21 |
+
/**
|
| 22 |
+
* Additional inference parameters for Image To Image
|
| 23 |
+
*/
|
| 24 |
+
export interface ImageToImageParameters {
|
| 25 |
+
/**
|
| 26 |
+
* For diffusion models. A higher guidance scale value encourages the model to generate
|
| 27 |
+
* images closely linked to the text prompt at the expense of lower image quality.
|
| 28 |
+
*/
|
| 29 |
+
guidance_scale?: number;
|
| 30 |
+
/**
|
| 31 |
+
* One prompt to guide what NOT to include in image generation.
|
| 32 |
+
*/
|
| 33 |
+
negative_prompt?: string;
|
| 34 |
+
/**
|
| 35 |
+
* For diffusion models. The number of denoising steps. More denoising steps usually lead to
|
| 36 |
+
* a higher quality image at the expense of slower inference.
|
| 37 |
+
*/
|
| 38 |
+
num_inference_steps?: number;
|
| 39 |
+
/**
|
| 40 |
+
* The text prompt to guide the image generation.
|
| 41 |
+
*/
|
| 42 |
+
prompt?: string;
|
| 43 |
+
/**
|
| 44 |
+
* The size in pixels of the output image. This parameter is only supported by some
|
| 45 |
+
* providers and for specific models. It will be ignored when unsupported.
|
| 46 |
+
*/
|
| 47 |
+
target_size?: TargetSize;
|
| 48 |
+
[property: string]: unknown;
|
| 49 |
+
}
|
| 50 |
+
/**
|
| 51 |
+
* The size in pixels of the output image. This parameter is only supported by some
|
| 52 |
+
* providers and for specific models. It will be ignored when unsupported.
|
| 53 |
+
*/
|
| 54 |
+
export interface TargetSize {
|
| 55 |
+
height: number;
|
| 56 |
+
width: number;
|
| 57 |
+
[property: string]: unknown;
|
| 58 |
+
}
|
| 59 |
+
/**
|
| 60 |
+
* Outputs of inference for the Image To Image task
|
| 61 |
+
*/
|
| 62 |
+
export interface ImageToImageOutput {
|
| 63 |
+
/**
|
| 64 |
+
* The output image returned as raw bytes in the payload.
|
| 65 |
+
*/
|
| 66 |
+
image?: unknown;
|
| 67 |
+
[property: string]: unknown;
|
| 68 |
+
}
|
| 69 |
+
//# sourceMappingURL=inference.d.ts.map
|
node_modules/@huggingface/tasks/dist/esm/tasks/image-to-image/inference.d.ts.map
ADDED
|
@@ -0,0 +1 @@
|
|
|
|
|
|
|
| 1 |
+
{"version":3,"file":"inference.d.ts","sourceRoot":"","sources":["../../../../src/tasks/image-to-image/inference.ts"],"names":[],"mappings":"AAAA;;;;GAIG;AACH;;GAEG;AACH,MAAM,WAAW,iBAAiB;IACjC;;;OAGG;IACH,MAAM,EAAE,IAAI,CAAC;IACb;;OAEG;IACH,UAAU,CAAC,EAAE,sBAAsB,CAAC;IACpC,CAAC,QAAQ,EAAE,MAAM,GAAG,OAAO,CAAC;CAC5B;AACD;;GAEG;AACH,MAAM,WAAW,sBAAsB;IACtC;;;OAGG;IACH,cAAc,CAAC,EAAE,MAAM,CAAC;IACxB;;OAEG;IACH,eAAe,CAAC,EAAE,MAAM,CAAC;IACzB;;;OAGG;IACH,mBAAmB,CAAC,EAAE,MAAM,CAAC;IAC7B;;OAEG;IACH,MAAM,CAAC,EAAE,MAAM,CAAC;IAChB;;;OAGG;IACH,WAAW,CAAC,EAAE,UAAU,CAAC;IACzB,CAAC,QAAQ,EAAE,MAAM,GAAG,OAAO,CAAC;CAC5B;AACD;;;GAGG;AACH,MAAM,WAAW,UAAU;IAC1B,MAAM,EAAE,MAAM,CAAC;IACf,KAAK,EAAE,MAAM,CAAC;IACd,CAAC,QAAQ,EAAE,MAAM,GAAG,OAAO,CAAC;CAC5B;AACD;;GAEG;AACH,MAAM,WAAW,kBAAkB;IAClC;;OAEG;IACH,KAAK,CAAC,EAAE,OAAO,CAAC;IAChB,CAAC,QAAQ,EAAE,MAAM,GAAG,OAAO,CAAC;CAC5B"}
|
node_modules/@huggingface/tasks/dist/esm/tasks/image-to-image/inference.js
ADDED
|
@@ -0,0 +1 @@
|
|
|
|
|
|
|
| 1 |
+
export {};
|
node_modules/@huggingface/tasks/dist/esm/tasks/image-to-text/data.d.ts
ADDED
|
@@ -0,0 +1,4 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
import type { TaskDataCustom } from "../index.js";
|
| 2 |
+
declare const taskData: TaskDataCustom;
|
| 3 |
+
export default taskData;
|
| 4 |
+
//# sourceMappingURL=data.d.ts.map
|
node_modules/@huggingface/tasks/dist/esm/tasks/image-to-text/data.d.ts.map
ADDED
|
@@ -0,0 +1 @@
|
|
|
|
|
|
|
| 1 |
+
{"version":3,"file":"data.d.ts","sourceRoot":"","sources":["../../../../src/tasks/image-to-text/data.ts"],"names":[],"mappings":"AAAA,OAAO,KAAK,EAAE,cAAc,EAAE,MAAM,aAAa,CAAC;AAElD,QAAA,MAAM,QAAQ,EAAE,cAyDf,CAAC;AAEF,eAAe,QAAQ,CAAC"}
|
node_modules/@huggingface/tasks/dist/esm/tasks/image-to-text/data.js
ADDED
|
@@ -0,0 +1,58 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
const taskData = {
|
| 2 |
+
datasets: [
|
| 3 |
+
{
|
| 4 |
+
// TODO write proper description
|
| 5 |
+
description: "Dataset from 12M image-text of Reddit",
|
| 6 |
+
id: "red_caps",
|
| 7 |
+
},
|
| 8 |
+
{
|
| 9 |
+
// TODO write proper description
|
| 10 |
+
description: "Dataset from 3.3M images of Google",
|
| 11 |
+
id: "datasets/conceptual_captions",
|
| 12 |
+
},
|
| 13 |
+
],
|
| 14 |
+
demo: {
|
| 15 |
+
inputs: [
|
| 16 |
+
{
|
| 17 |
+
filename: "savanna.jpg",
|
| 18 |
+
type: "img",
|
| 19 |
+
},
|
| 20 |
+
],
|
| 21 |
+
outputs: [
|
| 22 |
+
{
|
| 23 |
+
label: "Detailed description",
|
| 24 |
+
content: "a herd of giraffes and zebras grazing in a field",
|
| 25 |
+
type: "text",
|
| 26 |
+
},
|
| 27 |
+
],
|
| 28 |
+
},
|
| 29 |
+
metrics: [],
|
| 30 |
+
models: [
|
| 31 |
+
{
|
| 32 |
+
description: "Strong OCR model.",
|
| 33 |
+
id: "allenai/olmOCR-7B-0725",
|
| 34 |
+
},
|
| 35 |
+
{
|
| 36 |
+
description: "Powerful image captioning model.",
|
| 37 |
+
id: "fancyfeast/llama-joycaption-beta-one-hf-llava",
|
| 38 |
+
},
|
| 39 |
+
],
|
| 40 |
+
spaces: [
|
| 41 |
+
{
|
| 42 |
+
description: "SVG generator app from images.",
|
| 43 |
+
id: "multimodalart/OmniSVG-3B",
|
| 44 |
+
},
|
| 45 |
+
{
|
| 46 |
+
description: "An application that converts documents to markdown.",
|
| 47 |
+
id: "numind/NuMarkdown-8B-Thinking",
|
| 48 |
+
},
|
| 49 |
+
{
|
| 50 |
+
description: "An application that can caption images.",
|
| 51 |
+
id: "fancyfeast/joy-caption-beta-one",
|
| 52 |
+
},
|
| 53 |
+
],
|
| 54 |
+
summary: "Image to text models output a text from a given image. Image captioning or optical character recognition can be considered as the most common applications of image to text.",
|
| 55 |
+
widgetModels: ["Salesforce/blip-image-captioning-large"],
|
| 56 |
+
youtubeId: "",
|
| 57 |
+
};
|
| 58 |
+
export default taskData;
|
node_modules/@huggingface/tasks/dist/esm/tasks/image-to-text/inference.d.ts
ADDED
|
@@ -0,0 +1,135 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
/**
|
| 2 |
+
* Inference code generated from the JSON schema spec in ./spec
|
| 3 |
+
*
|
| 4 |
+
* Using src/scripts/inference-codegen
|
| 5 |
+
*/
|
| 6 |
+
/**
|
| 7 |
+
* Inputs for Image To Text inference
|
| 8 |
+
*/
|
| 9 |
+
export interface ImageToTextInput {
|
| 10 |
+
/**
|
| 11 |
+
* The input image data
|
| 12 |
+
*/
|
| 13 |
+
inputs: Blob;
|
| 14 |
+
/**
|
| 15 |
+
* Additional inference parameters for Image To Text
|
| 16 |
+
*/
|
| 17 |
+
parameters?: ImageToTextParameters;
|
| 18 |
+
[property: string]: unknown;
|
| 19 |
+
}
|
| 20 |
+
/**
|
| 21 |
+
* Additional inference parameters for Image To Text
|
| 22 |
+
*/
|
| 23 |
+
export interface ImageToTextParameters {
|
| 24 |
+
/**
|
| 25 |
+
* Parametrization of the text generation process
|
| 26 |
+
*/
|
| 27 |
+
generation_parameters?: GenerationParameters;
|
| 28 |
+
/**
|
| 29 |
+
* The amount of maximum tokens to generate.
|
| 30 |
+
*/
|
| 31 |
+
max_new_tokens?: number;
|
| 32 |
+
[property: string]: unknown;
|
| 33 |
+
}
|
| 34 |
+
/**
|
| 35 |
+
* Parametrization of the text generation process
|
| 36 |
+
*/
|
| 37 |
+
export interface GenerationParameters {
|
| 38 |
+
/**
|
| 39 |
+
* Whether to use sampling instead of greedy decoding when generating new tokens.
|
| 40 |
+
*/
|
| 41 |
+
do_sample?: boolean;
|
| 42 |
+
/**
|
| 43 |
+
* Controls the stopping condition for beam-based methods.
|
| 44 |
+
*/
|
| 45 |
+
early_stopping?: EarlyStoppingUnion;
|
| 46 |
+
/**
|
| 47 |
+
* If set to float strictly between 0 and 1, only tokens with a conditional probability
|
| 48 |
+
* greater than epsilon_cutoff will be sampled. In the paper, suggested values range from
|
| 49 |
+
* 3e-4 to 9e-4, depending on the size of the model. See [Truncation Sampling as Language
|
| 50 |
+
* Model Desmoothing](https://hf.co/papers/2210.15191) for more details.
|
| 51 |
+
*/
|
| 52 |
+
epsilon_cutoff?: number;
|
| 53 |
+
/**
|
| 54 |
+
* Eta sampling is a hybrid of locally typical sampling and epsilon sampling. If set to
|
| 55 |
+
* float strictly between 0 and 1, a token is only considered if it is greater than either
|
| 56 |
+
* eta_cutoff or sqrt(eta_cutoff) * exp(-entropy(softmax(next_token_logits))). The latter
|
| 57 |
+
* term is intuitively the expected next token probability, scaled by sqrt(eta_cutoff). In
|
| 58 |
+
* the paper, suggested values range from 3e-4 to 2e-3, depending on the size of the model.
|
| 59 |
+
* See [Truncation Sampling as Language Model Desmoothing](https://hf.co/papers/2210.15191)
|
| 60 |
+
* for more details.
|
| 61 |
+
*/
|
| 62 |
+
eta_cutoff?: number;
|
| 63 |
+
/**
|
| 64 |
+
* The maximum length (in tokens) of the generated text, including the input.
|
| 65 |
+
*/
|
| 66 |
+
max_length?: number;
|
| 67 |
+
/**
|
| 68 |
+
* The maximum number of tokens to generate. Takes precedence over max_length.
|
| 69 |
+
*/
|
| 70 |
+
max_new_tokens?: number;
|
| 71 |
+
/**
|
| 72 |
+
* The minimum length (in tokens) of the generated text, including the input.
|
| 73 |
+
*/
|
| 74 |
+
min_length?: number;
|
| 75 |
+
/**
|
| 76 |
+
* The minimum number of tokens to generate. Takes precedence over min_length.
|
| 77 |
+
*/
|
| 78 |
+
min_new_tokens?: number;
|
| 79 |
+
/**
|
| 80 |
+
* Number of groups to divide num_beams into in order to ensure diversity among different
|
| 81 |
+
* groups of beams. See [this paper](https://hf.co/papers/1610.02424) for more details.
|
| 82 |
+
*/
|
| 83 |
+
num_beam_groups?: number;
|
| 84 |
+
/**
|
| 85 |
+
* Number of beams to use for beam search.
|
| 86 |
+
*/
|
| 87 |
+
num_beams?: number;
|
| 88 |
+
/**
|
| 89 |
+
* The value balances the model confidence and the degeneration penalty in contrastive
|
| 90 |
+
* search decoding.
|
| 91 |
+
*/
|
| 92 |
+
penalty_alpha?: number;
|
| 93 |
+
/**
|
| 94 |
+
* The value used to modulate the next token probabilities.
|
| 95 |
+
*/
|
| 96 |
+
temperature?: number;
|
| 97 |
+
/**
|
| 98 |
+
* The number of highest probability vocabulary tokens to keep for top-k-filtering.
|
| 99 |
+
*/
|
| 100 |
+
top_k?: number;
|
| 101 |
+
/**
|
| 102 |
+
* If set to float < 1, only the smallest set of most probable tokens with probabilities
|
| 103 |
+
* that add up to top_p or higher are kept for generation.
|
| 104 |
+
*/
|
| 105 |
+
top_p?: number;
|
| 106 |
+
/**
|
| 107 |
+
* Local typicality measures how similar the conditional probability of predicting a target
|
| 108 |
+
* token next is to the expected conditional probability of predicting a random token next,
|
| 109 |
+
* given the partial text already generated. If set to float < 1, the smallest set of the
|
| 110 |
+
* most locally typical tokens with probabilities that add up to typical_p or higher are
|
| 111 |
+
* kept for generation. See [this paper](https://hf.co/papers/2202.00666) for more details.
|
| 112 |
+
*/
|
| 113 |
+
typical_p?: number;
|
| 114 |
+
/**
|
| 115 |
+
* Whether the model should use the past last key/values attentions to speed up decoding
|
| 116 |
+
*/
|
| 117 |
+
use_cache?: boolean;
|
| 118 |
+
[property: string]: unknown;
|
| 119 |
+
}
|
| 120 |
+
/**
|
| 121 |
+
* Controls the stopping condition for beam-based methods.
|
| 122 |
+
*/
|
| 123 |
+
export type EarlyStoppingUnion = boolean | "never";
|
| 124 |
+
/**
|
| 125 |
+
* Outputs of inference for the Image To Text task
|
| 126 |
+
*/
|
| 127 |
+
export interface ImageToTextOutput {
|
| 128 |
+
generatedText: unknown;
|
| 129 |
+
/**
|
| 130 |
+
* The generated text.
|
| 131 |
+
*/
|
| 132 |
+
generated_text?: string;
|
| 133 |
+
[property: string]: unknown;
|
| 134 |
+
}
|
| 135 |
+
//# sourceMappingURL=inference.d.ts.map
|
node_modules/@huggingface/tasks/dist/esm/tasks/image-to-text/inference.d.ts.map
ADDED
|
@@ -0,0 +1 @@
|
|
|
|
|
|
|
| 1 |
+
{"version":3,"file":"inference.d.ts","sourceRoot":"","sources":["../../../../src/tasks/image-to-text/inference.ts"],"names":[],"mappings":"AAAA;;;;GAIG;AACH;;GAEG;AACH,MAAM,WAAW,gBAAgB;IAChC;;OAEG;IACH,MAAM,EAAE,IAAI,CAAC;IACb;;OAEG;IACH,UAAU,CAAC,EAAE,qBAAqB,CAAC;IACnC,CAAC,QAAQ,EAAE,MAAM,GAAG,OAAO,CAAC;CAC5B;AACD;;GAEG;AACH,MAAM,WAAW,qBAAqB;IACrC;;OAEG;IACH,qBAAqB,CAAC,EAAE,oBAAoB,CAAC;IAC7C;;OAEG;IACH,cAAc,CAAC,EAAE,MAAM,CAAC;IACxB,CAAC,QAAQ,EAAE,MAAM,GAAG,OAAO,CAAC;CAC5B;AACD;;GAEG;AACH,MAAM,WAAW,oBAAoB;IACpC;;OAEG;IACH,SAAS,CAAC,EAAE,OAAO,CAAC;IACpB;;OAEG;IACH,cAAc,CAAC,EAAE,kBAAkB,CAAC;IACpC;;;;;OAKG;IACH,cAAc,CAAC,EAAE,MAAM,CAAC;IACxB;;;;;;;;OAQG;IACH,UAAU,CAAC,EAAE,MAAM,CAAC;IACpB;;OAEG;IACH,UAAU,CAAC,EAAE,MAAM,CAAC;IACpB;;OAEG;IACH,cAAc,CAAC,EAAE,MAAM,CAAC;IACxB;;OAEG;IACH,UAAU,CAAC,EAAE,MAAM,CAAC;IACpB;;OAEG;IACH,cAAc,CAAC,EAAE,MAAM,CAAC;IACxB;;;OAGG;IACH,eAAe,CAAC,EAAE,MAAM,CAAC;IACzB;;OAEG;IACH,SAAS,CAAC,EAAE,MAAM,CAAC;IACnB;;;OAGG;IACH,aAAa,CAAC,EAAE,MAAM,CAAC;IACvB;;OAEG;IACH,WAAW,CAAC,EAAE,MAAM,CAAC;IACrB;;OAEG;IACH,KAAK,CAAC,EAAE,MAAM,CAAC;IACf;;;OAGG;IACH,KAAK,CAAC,EAAE,MAAM,CAAC;IACf;;;;;;OAMG;IACH,SAAS,CAAC,EAAE,MAAM,CAAC;IACnB;;OAEG;IACH,SAAS,CAAC,EAAE,OAAO,CAAC;IACpB,CAAC,QAAQ,EAAE,MAAM,GAAG,OAAO,CAAC;CAC5B;AACD;;GAEG;AACH,MAAM,MAAM,kBAAkB,GAAG,OAAO,GAAG,OAAO,CAAC;AACnD;;GAEG;AACH,MAAM,WAAW,iBAAiB;IACjC,aAAa,EAAE,OAAO,CAAC;IACvB;;OAEG;IACH,cAAc,CAAC,EAAE,MAAM,CAAC;IACxB,CAAC,QAAQ,EAAE,MAAM,GAAG,OAAO,CAAC;CAC5B"}
|
node_modules/@huggingface/tasks/dist/esm/tasks/image-to-text/inference.js
ADDED
|
@@ -0,0 +1 @@
|
|
|
|
|
|
|
| 1 |
+
export {};
|
node_modules/@huggingface/tasks/dist/esm/tasks/image-to-video/data.d.ts
ADDED
|
@@ -0,0 +1,4 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
import type { TaskDataCustom } from "../index.js";
|
| 2 |
+
declare const taskData: TaskDataCustom;
|
| 3 |
+
export default taskData;
|
| 4 |
+
//# sourceMappingURL=data.d.ts.map
|
node_modules/@huggingface/tasks/dist/esm/tasks/image-to-video/data.d.ts.map
ADDED
|
@@ -0,0 +1 @@
|
|
|
|
|
|
|
| 1 |
+
{"version":3,"file":"data.d.ts","sourceRoot":"","sources":["../../../../src/tasks/image-to-video/data.ts"],"names":[],"mappings":"AAAA,OAAO,KAAK,EAAE,cAAc,EAAE,MAAM,aAAa,CAAC;AAElD,QAAA,MAAM,QAAQ,EAAE,cAyHf,CAAC;AAEF,eAAe,QAAQ,CAAC"}
|
node_modules/@huggingface/tasks/dist/esm/tasks/image-to-video/data.js
ADDED
|
@@ -0,0 +1,117 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
const taskData = {
|
| 2 |
+
datasets: [
|
| 3 |
+
{
|
| 4 |
+
description: "A benchmark dataset for reference image controlled video generation.",
|
| 5 |
+
id: "ali-vilab/VACE-Benchmark",
|
| 6 |
+
},
|
| 7 |
+
{
|
| 8 |
+
description: "A dataset of video generation style preferences.",
|
| 9 |
+
id: "Rapidata/sora-video-generation-style-likert-scoring",
|
| 10 |
+
},
|
| 11 |
+
{
|
| 12 |
+
description: "A dataset with videos and captions throughout the videos.",
|
| 13 |
+
id: "BestWishYsh/ChronoMagic",
|
| 14 |
+
},
|
| 15 |
+
],
|
| 16 |
+
demo: {
|
| 17 |
+
inputs: [
|
| 18 |
+
{
|
| 19 |
+
filename: "image-to-video-input.jpg",
|
| 20 |
+
type: "img",
|
| 21 |
+
},
|
| 22 |
+
{
|
| 23 |
+
label: "Optional Text Prompt",
|
| 24 |
+
content: "This penguin is dancing",
|
| 25 |
+
type: "text",
|
| 26 |
+
},
|
| 27 |
+
],
|
| 28 |
+
outputs: [
|
| 29 |
+
{
|
| 30 |
+
filename: "image-to-video-output.gif",
|
| 31 |
+
type: "img",
|
| 32 |
+
},
|
| 33 |
+
],
|
| 34 |
+
},
|
| 35 |
+
metrics: [
|
| 36 |
+
{
|
| 37 |
+
description: "Fréchet Video Distance (FVD) measures the perceptual similarity between the distributions of generated videos and a set of real videos, assessing overall visual quality and temporal coherence of the video generated from an input image.",
|
| 38 |
+
id: "fvd",
|
| 39 |
+
},
|
| 40 |
+
{
|
| 41 |
+
description: "CLIP Score measures the semantic similarity between a textual prompt (if provided alongside the input image) and the generated video frames. It evaluates how well the video's generated content and motion align with the textual description, conditioned on the initial image.",
|
| 42 |
+
id: "clip_score",
|
| 43 |
+
},
|
| 44 |
+
{
|
| 45 |
+
description: "First Frame Fidelity, often measured using LPIPS (Learned Perceptual Image Patch Similarity), PSNR, or SSIM, quantifies how closely the first frame of the generated video matches the input conditioning image.",
|
| 46 |
+
id: "lpips",
|
| 47 |
+
},
|
| 48 |
+
{
|
| 49 |
+
description: "Identity Preservation Score measures the consistency of identity (e.g., a person's face or a specific object's characteristics) between the input image and throughout the generated video frames, often calculated using features from specialized models like face recognition (e.g., ArcFace) or re-identification models.",
|
| 50 |
+
id: "identity_preservation",
|
| 51 |
+
},
|
| 52 |
+
{
|
| 53 |
+
description: "Motion Score evaluates the quality, realism, and temporal consistency of motion in the video generated from a static image. This can be based on optical flow analysis (e.g., smoothness, magnitude), consistency of object trajectories, or specific motion plausibility assessments.",
|
| 54 |
+
id: "motion_score",
|
| 55 |
+
},
|
| 56 |
+
],
|
| 57 |
+
models: [
|
| 58 |
+
{
|
| 59 |
+
description: "LTX-Video, a 13B parameter model for high quality video generation",
|
| 60 |
+
id: "Lightricks/LTX-Video-0.9.7-dev",
|
| 61 |
+
},
|
| 62 |
+
{
|
| 63 |
+
description: "A 14B parameter model for reference image controlled video generation",
|
| 64 |
+
id: "Wan-AI/Wan2.1-VACE-14B",
|
| 65 |
+
},
|
| 66 |
+
{
|
| 67 |
+
description: "An image-to-video generation model using FramePack F1 methodology with Hunyuan-DiT architecture",
|
| 68 |
+
id: "lllyasviel/FramePack_F1_I2V_HY_20250503",
|
| 69 |
+
},
|
| 70 |
+
{
|
| 71 |
+
description: "A distilled version of the LTX-Video-0.9.7-dev model for faster inference",
|
| 72 |
+
id: "Lightricks/LTX-Video-0.9.7-distilled",
|
| 73 |
+
},
|
| 74 |
+
{
|
| 75 |
+
description: "An image-to-video generation model by Skywork AI, 14B parameters, producing 720p videos.",
|
| 76 |
+
id: "Skywork/SkyReels-V2-I2V-14B-720P",
|
| 77 |
+
},
|
| 78 |
+
{
|
| 79 |
+
description: "Image-to-video variant of Tencent's HunyuanVideo.",
|
| 80 |
+
id: "tencent/HunyuanVideo-I2V",
|
| 81 |
+
},
|
| 82 |
+
{
|
| 83 |
+
description: "A 14B parameter model for 720p image-to-video generation by Wan-AI.",
|
| 84 |
+
id: "Wan-AI/Wan2.1-I2V-14B-720P",
|
| 85 |
+
},
|
| 86 |
+
{
|
| 87 |
+
description: "A Diffusers version of the Wan2.1-I2V-14B-720P model for 720p image-to-video generation.",
|
| 88 |
+
id: "Wan-AI/Wan2.1-I2V-14B-720P-Diffusers",
|
| 89 |
+
},
|
| 90 |
+
],
|
| 91 |
+
spaces: [
|
| 92 |
+
{
|
| 93 |
+
description: "An application to generate videos fast.",
|
| 94 |
+
id: "Lightricks/ltx-video-distilled",
|
| 95 |
+
},
|
| 96 |
+
{
|
| 97 |
+
description: "Generate videos with the FramePack-F1",
|
| 98 |
+
id: "linoyts/FramePack-F1",
|
| 99 |
+
},
|
| 100 |
+
{
|
| 101 |
+
description: "Generate videos with the FramePack",
|
| 102 |
+
id: "lisonallen/framepack-i2v",
|
| 103 |
+
},
|
| 104 |
+
{
|
| 105 |
+
description: "Wan2.1 with CausVid LoRA",
|
| 106 |
+
id: "multimodalart/wan2-1-fast",
|
| 107 |
+
},
|
| 108 |
+
{
|
| 109 |
+
description: "A demo for Stable Video Diffusion",
|
| 110 |
+
id: "multimodalart/stable-video-diffusion",
|
| 111 |
+
},
|
| 112 |
+
],
|
| 113 |
+
summary: "Image-to-video models take a still image as input and generate a video. These models can be guided by text prompts to influence the content and style of the output video.",
|
| 114 |
+
widgetModels: [],
|
| 115 |
+
youtubeId: undefined,
|
| 116 |
+
};
|
| 117 |
+
export default taskData;
|
node_modules/@huggingface/tasks/dist/esm/tasks/image-to-video/inference.d.ts
ADDED
|
@@ -0,0 +1,75 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
/**
|
| 2 |
+
* Inference code generated from the JSON schema spec in ./spec
|
| 3 |
+
*
|
| 4 |
+
* Using src/scripts/inference-codegen
|
| 5 |
+
*/
|
| 6 |
+
/**
|
| 7 |
+
* Inputs for Image To Video inference
|
| 8 |
+
*/
|
| 9 |
+
export interface ImageToVideoInput {
|
| 10 |
+
/**
|
| 11 |
+
* The input image data as a base64-encoded string. If no `parameters` are provided, you can
|
| 12 |
+
* also provide the image data as a raw bytes payload.
|
| 13 |
+
*/
|
| 14 |
+
inputs: Blob;
|
| 15 |
+
/**
|
| 16 |
+
* Additional inference parameters for Image To Video
|
| 17 |
+
*/
|
| 18 |
+
parameters?: ImageToVideoParameters;
|
| 19 |
+
[property: string]: unknown;
|
| 20 |
+
}
|
| 21 |
+
/**
|
| 22 |
+
* Additional inference parameters for Image To Video
|
| 23 |
+
*/
|
| 24 |
+
export interface ImageToVideoParameters {
|
| 25 |
+
/**
|
| 26 |
+
* For diffusion models. A higher guidance scale value encourages the model to generate
|
| 27 |
+
* videos closely linked to the text prompt at the expense of lower image quality.
|
| 28 |
+
*/
|
| 29 |
+
guidance_scale?: number;
|
| 30 |
+
/**
|
| 31 |
+
* One prompt to guide what NOT to include in video generation.
|
| 32 |
+
*/
|
| 33 |
+
negative_prompt?: string;
|
| 34 |
+
/**
|
| 35 |
+
* The num_frames parameter determines how many video frames are generated.
|
| 36 |
+
*/
|
| 37 |
+
num_frames?: number;
|
| 38 |
+
/**
|
| 39 |
+
* The number of denoising steps. More denoising steps usually lead to a higher quality
|
| 40 |
+
* video at the expense of slower inference.
|
| 41 |
+
*/
|
| 42 |
+
num_inference_steps?: number;
|
| 43 |
+
/**
|
| 44 |
+
* The text prompt to guide the video generation.
|
| 45 |
+
*/
|
| 46 |
+
prompt?: string;
|
| 47 |
+
/**
|
| 48 |
+
* Seed for the random number generator.
|
| 49 |
+
*/
|
| 50 |
+
seed?: number;
|
| 51 |
+
/**
|
| 52 |
+
* The size in pixel of the output video frames.
|
| 53 |
+
*/
|
| 54 |
+
target_size?: TargetSize;
|
| 55 |
+
[property: string]: unknown;
|
| 56 |
+
}
|
| 57 |
+
/**
|
| 58 |
+
* The size in pixel of the output video frames.
|
| 59 |
+
*/
|
| 60 |
+
export interface TargetSize {
|
| 61 |
+
height: number;
|
| 62 |
+
width: number;
|
| 63 |
+
[property: string]: unknown;
|
| 64 |
+
}
|
| 65 |
+
/**
|
| 66 |
+
* Outputs of inference for the Image To Video task
|
| 67 |
+
*/
|
| 68 |
+
export interface ImageToVideoOutput {
|
| 69 |
+
/**
|
| 70 |
+
* The generated video returned as raw bytes in the payload.
|
| 71 |
+
*/
|
| 72 |
+
video: unknown;
|
| 73 |
+
[property: string]: unknown;
|
| 74 |
+
}
|
| 75 |
+
//# sourceMappingURL=inference.d.ts.map
|
node_modules/@huggingface/tasks/dist/esm/tasks/image-to-video/inference.d.ts.map
ADDED
|
@@ -0,0 +1 @@
|
|
|
|
|
|
|
| 1 |
+
{"version":3,"file":"inference.d.ts","sourceRoot":"","sources":["../../../../src/tasks/image-to-video/inference.ts"],"names":[],"mappings":"AAAA;;;;GAIG;AACH;;GAEG;AACH,MAAM,WAAW,iBAAiB;IACjC;;;OAGG;IACH,MAAM,EAAE,IAAI,CAAC;IACb;;OAEG;IACH,UAAU,CAAC,EAAE,sBAAsB,CAAC;IACpC,CAAC,QAAQ,EAAE,MAAM,GAAG,OAAO,CAAC;CAC5B;AACD;;GAEG;AACH,MAAM,WAAW,sBAAsB;IACtC;;;OAGG;IACH,cAAc,CAAC,EAAE,MAAM,CAAC;IACxB;;OAEG;IACH,eAAe,CAAC,EAAE,MAAM,CAAC;IACzB;;OAEG;IACH,UAAU,CAAC,EAAE,MAAM,CAAC;IACpB;;;OAGG;IACH,mBAAmB,CAAC,EAAE,MAAM,CAAC;IAC7B;;OAEG;IACH,MAAM,CAAC,EAAE,MAAM,CAAC;IAChB;;OAEG;IACH,IAAI,CAAC,EAAE,MAAM,CAAC;IACd;;OAEG;IACH,WAAW,CAAC,EAAE,UAAU,CAAC;IACzB,CAAC,QAAQ,EAAE,MAAM,GAAG,OAAO,CAAC;CAC5B;AACD;;GAEG;AACH,MAAM,WAAW,UAAU;IAC1B,MAAM,EAAE,MAAM,CAAC;IACf,KAAK,EAAE,MAAM,CAAC;IACd,CAAC,QAAQ,EAAE,MAAM,GAAG,OAAO,CAAC;CAC5B;AACD;;GAEG;AACH,MAAM,WAAW,kBAAkB;IAClC;;OAEG;IACH,KAAK,EAAE,OAAO,CAAC;IACf,CAAC,QAAQ,EAAE,MAAM,GAAG,OAAO,CAAC;CAC5B"}
|
node_modules/@huggingface/tasks/dist/esm/tasks/image-to-video/inference.js
ADDED
|
@@ -0,0 +1 @@
|
|
|
|
|
|
|
| 1 |
+
export {};
|