Instructions to use facebook/sapiens2-pose-5b with libraries, inference providers, notebooks, and local apps. Follow these links to get started.
- Libraries
- sapiens2
How to use facebook/sapiens2-pose-5b with sapiens2:
# No code snippets available yet for this library. # To use this model, check the repository files and the library's documentation. # Want to help? PRs adding snippets are welcome at: # https://github.com/huggingface/huggingface.js
- sapiens
How to use facebook/sapiens2-pose-5b with sapiens:
# No code snippets available yet for this library. # To use this model, check the repository files and the library's documentation. # Want to help? PRs adding snippets are welcome at: # https://github.com/huggingface/huggingface.js
- Transformers
How to use facebook/sapiens2-pose-5b with Transformers:
# pip install -U transformers accelerate # Load model directly from transformers import AutoImageProcessor, Sapiens2ForPoseEstimation processor = AutoImageProcessor.from_pretrained("facebook/sapiens2-pose-5b") model = Sapiens2ForPoseEstimation.from_pretrained("facebook/sapiens2-pose-5b", device_map="auto") - Notebooks
- Google Colab
- Kaggle
Update config.json and preprocessor_config.json
Browse files- config.json +49 -41
config.json
CHANGED
|
@@ -595,6 +595,7 @@
|
|
| 595 |
"_name_or_path": "",
|
| 596 |
"architectures": null,
|
| 597 |
"chunk_size_feed_forward": 0,
|
|
|
|
| 598 |
"conv_kernel_sizes": [
|
| 599 |
1,
|
| 600 |
1,
|
|
@@ -620,10 +621,12 @@
|
|
| 620 |
"output_hidden_states": false,
|
| 621 |
"problem_type": null,
|
| 622 |
"return_dict": true,
|
|
|
|
| 623 |
"scale_conv_kernel_sizes": null,
|
| 624 |
"scale_conv_out_channels": null,
|
| 625 |
"scale_final_hidden_sizes": null,
|
| 626 |
"scale_final_input_size": null,
|
|
|
|
| 627 |
"upsample_kernel_sizes": [
|
| 628 |
4,
|
| 629 |
4
|
|
@@ -1263,13 +1266,16 @@
|
|
| 1263 |
"LABEL_98": 98,
|
| 1264 |
"LABEL_99": 99
|
| 1265 |
},
|
| 1266 |
-
"layer_norm_eps": 1e-
|
| 1267 |
"layerscale_value": 1.0,
|
| 1268 |
"mlp_bias": true,
|
| 1269 |
"model_type": "sapiens2",
|
|
|
|
| 1270 |
"num_attention_heads": 32,
|
| 1271 |
"num_channels": 3,
|
|
|
|
| 1272 |
"num_hidden_layers": 56,
|
|
|
|
| 1273 |
"num_key_value_heads_per_layer": [
|
| 1274 |
32,
|
| 1275 |
32,
|
|
@@ -1279,46 +1285,46 @@
|
|
| 1279 |
32,
|
| 1280 |
32,
|
| 1281 |
32,
|
| 1282 |
-
|
| 1283 |
-
|
| 1284 |
-
|
| 1285 |
-
|
| 1286 |
-
|
| 1287 |
-
|
| 1288 |
-
|
| 1289 |
-
|
| 1290 |
-
|
| 1291 |
-
|
| 1292 |
-
|
| 1293 |
-
|
| 1294 |
-
|
| 1295 |
-
|
| 1296 |
-
|
| 1297 |
-
|
| 1298 |
-
|
| 1299 |
-
|
| 1300 |
-
|
| 1301 |
-
|
| 1302 |
-
|
| 1303 |
-
|
| 1304 |
-
|
| 1305 |
-
|
| 1306 |
-
|
| 1307 |
-
|
| 1308 |
-
|
| 1309 |
-
|
| 1310 |
-
|
| 1311 |
-
|
| 1312 |
-
|
| 1313 |
-
|
| 1314 |
-
|
| 1315 |
-
|
| 1316 |
-
|
| 1317 |
-
|
| 1318 |
-
|
| 1319 |
-
|
| 1320 |
-
|
| 1321 |
-
|
| 1322 |
32,
|
| 1323 |
32,
|
| 1324 |
32,
|
|
@@ -1328,6 +1334,7 @@
|
|
| 1328 |
32,
|
| 1329 |
32
|
| 1330 |
],
|
|
|
|
| 1331 |
"num_register_tokens": 8,
|
| 1332 |
"out_features": [
|
| 1333 |
"stage56"
|
|
@@ -1342,6 +1349,7 @@
|
|
| 1342 |
"proj_bias": true,
|
| 1343 |
"query_bias": true,
|
| 1344 |
"reshape_hidden_states": true,
|
|
|
|
| 1345 |
"rope_theta": 100.0,
|
| 1346 |
"semantic_loss_ignore_index": 255,
|
| 1347 |
"stage_names": [
|
|
|
|
| 595 |
"_name_or_path": "",
|
| 596 |
"architectures": null,
|
| 597 |
"chunk_size_feed_forward": 0,
|
| 598 |
+
"conv_kernel_size": 1,
|
| 599 |
"conv_kernel_sizes": [
|
| 600 |
1,
|
| 601 |
1,
|
|
|
|
| 621 |
"output_hidden_states": false,
|
| 622 |
"problem_type": null,
|
| 623 |
"return_dict": true,
|
| 624 |
+
"scale_conv_kernel_size": 1,
|
| 625 |
"scale_conv_kernel_sizes": null,
|
| 626 |
"scale_conv_out_channels": null,
|
| 627 |
"scale_final_hidden_sizes": null,
|
| 628 |
"scale_final_input_size": null,
|
| 629 |
+
"upsample_kernel_size": 4,
|
| 630 |
"upsample_kernel_sizes": [
|
| 631 |
4,
|
| 632 |
4
|
|
|
|
| 1266 |
"LABEL_98": 98,
|
| 1267 |
"LABEL_99": 99
|
| 1268 |
},
|
| 1269 |
+
"layer_norm_eps": 1e-05,
|
| 1270 |
"layerscale_value": 1.0,
|
| 1271 |
"mlp_bias": true,
|
| 1272 |
"model_type": "sapiens2",
|
| 1273 |
+
"normalize_backbone_outputs": true,
|
| 1274 |
"num_attention_heads": 32,
|
| 1275 |
"num_channels": 3,
|
| 1276 |
+
"num_first_full_attention_layers": 8,
|
| 1277 |
"num_hidden_layers": 56,
|
| 1278 |
+
"num_key_value_attention_heads": 8,
|
| 1279 |
"num_key_value_heads_per_layer": [
|
| 1280 |
32,
|
| 1281 |
32,
|
|
|
|
| 1285 |
32,
|
| 1286 |
32,
|
| 1287 |
32,
|
| 1288 |
+
8,
|
| 1289 |
+
8,
|
| 1290 |
+
8,
|
| 1291 |
+
8,
|
| 1292 |
+
8,
|
| 1293 |
+
8,
|
| 1294 |
+
8,
|
| 1295 |
+
8,
|
| 1296 |
+
8,
|
| 1297 |
+
8,
|
| 1298 |
+
8,
|
| 1299 |
+
8,
|
| 1300 |
+
8,
|
| 1301 |
+
8,
|
| 1302 |
+
8,
|
| 1303 |
+
8,
|
| 1304 |
+
8,
|
| 1305 |
+
8,
|
| 1306 |
+
8,
|
| 1307 |
+
8,
|
| 1308 |
+
8,
|
| 1309 |
+
8,
|
| 1310 |
+
8,
|
| 1311 |
+
8,
|
| 1312 |
+
8,
|
| 1313 |
+
8,
|
| 1314 |
+
8,
|
| 1315 |
+
8,
|
| 1316 |
+
8,
|
| 1317 |
+
8,
|
| 1318 |
+
8,
|
| 1319 |
+
8,
|
| 1320 |
+
8,
|
| 1321 |
+
8,
|
| 1322 |
+
8,
|
| 1323 |
+
8,
|
| 1324 |
+
8,
|
| 1325 |
+
8,
|
| 1326 |
+
8,
|
| 1327 |
+
8,
|
| 1328 |
32,
|
| 1329 |
32,
|
| 1330 |
32,
|
|
|
|
| 1334 |
32,
|
| 1335 |
32
|
| 1336 |
],
|
| 1337 |
+
"num_last_full_attention_layers": 8,
|
| 1338 |
"num_register_tokens": 8,
|
| 1339 |
"out_features": [
|
| 1340 |
"stage56"
|
|
|
|
| 1349 |
"proj_bias": true,
|
| 1350 |
"query_bias": true,
|
| 1351 |
"reshape_hidden_states": true,
|
| 1352 |
+
"rms_norm_eps": 1e-06,
|
| 1353 |
"rope_theta": 100.0,
|
| 1354 |
"semantic_loss_ignore_index": 255,
|
| 1355 |
"stage_names": [
|