masking: hide sky and cloud from features in 360-camera preset

This commit is contained in:
Harry Chen
2026-09-30 12:54:48 -04:00
parent 09de422c80
commit 9f7ab0d697
39 changed files with 923 additions and 201 deletions
+6
View File
@@ -71,6 +71,12 @@ the first, middle and last of them: an RGBA export that is opaque everywhere is
no mask. Training decodes each such image twice, once for its colour and once
for its alpha.
A dataset the GUI built may also carry `feature_masks/`, mirroring `masks/`:
what its "Hide from the reconstruction only" prompt matched (the 360-camera
preset's `sky; cloud`). Only the reconstruction reads it -- `spirula sfm
--feature-masks`, intersected with `--masks` -- so training never sees it and
still learns the sky.
## Seed points
The splats start from the dataset's point cloud. `random_init` decides when
+41
View File
@@ -1211,4 +1211,45 @@ int64_t apply_frame_stencil(const FrameStencilRun& run,
return written;
}
int64_t intersect_mask_trees(const std::string& image_dir, const std::string& a,
bool flip_a, const std::string& b,
const std::string& out, const std::atomic<bool>* cancel,
std::string& error) {
struct Item { std::string dst, a, b; int w = 0, h = 0; };
std::vector<Item> items;
std::error_code ec;
for (const auto& [rel, files] : group_frames_by_camera(image_dir)) {
for (const std::string& f : files) {
const fs::path name = fs::path(rel) / (fs::path(f).stem().string() + ".png");
Item it;
if (!a.empty() && fs::exists(fs::path(a) / name, ec))
it.a = (fs::path(a) / name).string();
if (!b.empty() && fs::exists(fs::path(b) / name, ec))
it.b = (fs::path(b) / name).string();
if ((it.a.empty() && it.b.empty()) || !image_size(f, it.w, it.h)) continue;
it.dst = (fs::path(out) / name).string();
fs::create_directories(fs::path(it.dst).parent_path(), ec);
items.push_back(std::move(it));
}
}
std::atomic<bool> failed{false};
std::mutex mu;
nn::parallel_for((int64_t)items.size(), [&](int64_t lo, int64_t hi) {
for (int64_t k = lo; k < hi; k++) {
if (failed.load() || (cancel && cancel->load())) return;
const Item& it = items[(size_t)k];
std::vector<uint8_t> px((size_t)it.w * it.h, 255);
if (!it.a.empty()) intersect_with_file(px, it.w, it.h, it.a, flip_a);
if (!it.b.empty()) intersect_with_file(px, it.w, it.h, it.b, false);
if (!stbi_write_png(it.dst.c_str(), it.w, it.h, 1, px.data(), it.w)) {
std::lock_guard<std::mutex> lk(mu);
error = it.dst;
failed = true;
return;
}
}
}, 1);
return failed.load() ? -1 : (int64_t)items.size();
}
} // namespace app
+8
View File
@@ -208,4 +208,12 @@ struct FrameStencilSinks {
int64_t apply_frame_stencil(const FrameStencilRun& run,
const FrameStencilSinks& sinks, std::string& error);
// Masks `a` and `b` intersected into `out`, one per image of `image_dir` that
// has either, named <rel>/<stem>.png as every writer here names them. For a
// reader that takes one mask tree: COLMAP. Returns how many, or -1.
int64_t intersect_mask_trees(const std::string& image_dir, const std::string& a,
bool flip_a, const std::string& b,
const std::string& out, const std::atomic<bool>* cancel,
std::string& error);
} // namespace app
+1 -1
View File
@@ -405,7 +405,7 @@ void check_dataset_stage(const BatchRow& row, const BatchCapabilities& caps,
// on and a preset cannot carry them -- so the text prompt is the only
// prompt there is.
const bool prompted = !caps.mask_model_prompted || caps.mask_model_prompted(s.mask_model_id);
if (prompted && s.mask.prompt.empty())
if (prompted && s.mask.prompt.empty() && s.mask.feature_prompt.empty())
out.push_back(issue_of(msg::chk_mask_no_prompt, kSt, true));
if (!caps.masking) {
out.push_back(issue_of(msg::chk_masking_unavailable, kSt, true));
+23 -2
View File
@@ -271,6 +271,7 @@ void ColmapRunner::take_masking(PrepJob& prep) {
prep.mask_enable = _live.mask_enable;
prep.mask_prompt = _live.mask_prompt;
prep.mask_negative_prompt = _live.mask_negative_prompt;
prep.mask_feature_prompt = _live.mask_feature_prompt;
prep.mask_keep_subject = _live.mask_keep_subject;
prep.mask_max_image_size = _live.mask_max_image_size;
prep.mask_dilate_ratio = _live.mask_dilate_ratio;
@@ -459,6 +460,7 @@ static std::vector<std::string> colmap_recon_args(const ColmapJob& job) {
"--vocab-tree", job.vocab_tree_path,
"--masks", flag(job.mask_enable && job.mask_features),
"--mask-prompt", job.mask_enable ? job.mask_prompt : std::string(),
"--feature-prompt", job.mask_enable ? job.mask_feature_prompt : std::string(),
};
}
@@ -529,6 +531,7 @@ void ColmapRunner::run(ColmapJob job) {
pj.mask_enable = job.mask_enable;
pj.mask_prompt = job.mask_prompt;
pj.mask_negative_prompt = job.mask_negative_prompt;
pj.mask_feature_prompt = job.mask_feature_prompt;
pj.mask_keep_subject = job.mask_keep_subject;
pj.mask_max_image_size = job.mask_max_image_size;
pj.mask_dilate_ratio = job.mask_dilate_ratio;
@@ -658,9 +661,26 @@ void ColmapRunner::run(ColmapJob job) {
shared.push_back("--SiftExtraction.estimate_affine_shape");
shared.push_back("1");
}
if (have_masks && job.mask_features) {
// COLMAP reads one mask tree, so the feature-only masks are
// intersected into a copy when training's are wanted there too.
std::string colmap_masks = have_masks && job.mask_features ? prep.mask_dir : "";
const fs::path both = ws / ".colmap_masks";
remove_tree(both);
if (!prep.feature_mask_dir.empty()) {
if (colmap_masks.empty()) {
colmap_masks = prep.feature_mask_dir;
} else {
std::string err;
if (app::intersect_mask_trees(images, colmap_masks, prep.mask_dir_flipped,
prep.feature_mask_dir, both.string(),
&_cancel, err) < 0)
return fail("could not write " + err);
colmap_masks = both.string();
}
}
if (!colmap_masks.empty()) {
shared.push_back("--ImageReader.mask_path");
shared.push_back(prep.mask_dir);
shared.push_back(colmap_masks);
}
int rc = 0;
std::vector<std::vector<std::string>> passes;
@@ -702,6 +722,7 @@ void ColmapRunner::run(ColmapJob job) {
if (rc == kCancelled) return fail("cancelled");
if (rc != 0) return fail("colmap feature_extractor failed (see log)");
}
remove_tree(both);
// ---- 4. matching -----------------------------------------------------
// An explicit choice: the GUI presets sequential for video and
+1
View File
@@ -152,6 +152,7 @@ struct ColmapJob {
bool mask_enable = false;
std::string mask_prompt; // "people; cars; ..."
std::string mask_negative_prompt;
std::string mask_feature_prompt; // what features skip, training keeps
bool mask_keep_subject = false; // prompt names what to KEEP
std::string mask_model_path;
std::string mask_detector_path; // Grounding DINO, when one is paired
+107 -44
View File
@@ -768,7 +768,8 @@ void collect_image_folders(const fs::path& dir, const std::string& rel, int dept
it.increment(ec)) {
if (it->is_directory(ec)) {
if (!is_mask_folder(it->path().string()) &&
!is_mask_edits_folder(it->path().string()))
!is_mask_edits_folder(it->path().string()) &&
!is_feature_mask_folder(it->path().string()))
sub.push_back(it->path());
} else if (!here && it->is_regular_file(ec) && is_image_file(it->path())) {
here = true;
@@ -886,7 +887,8 @@ WorkspaceState probe_workspace(const std::string& workspace,
};
st.frames = has_content(ws / "images") && !is_input(ws / "images", false);
st.masks = has_content(ws / "masks") && !is_input(ws / "masks", true);
st.masks = (has_content(ws / "masks") && !is_input(ws / "masks", true)) ||
has_content(ws / kFeatureMaskDirName);
for (const PrepInput& in : inputs)
st.input_masks = st.input_masks || (!in.mask_dir.empty() && has_content(in.mask_dir));
st.features = has_content(ws / "features") || fs::exists(ws / "matches.bin", ec) ||
@@ -916,7 +918,8 @@ std::vector<std::string> workspace_artifacts(const std::string& workspace,
};
if (!is_input_folder(ws / "images", inputs, false)) add("images");
if (!is_input_folder(ws / "masks", inputs, true)) add("masks");
for (const char* name : {"features", "sparse", "colmap", "normals", "depths",
for (const char* name : {kFeatureMaskDirName, "features", "sparse", "colmap",
"normals", "depths",
".progress", sfm::resume::kDir, "matches.bin",
"database.db", kReconStampFile})
add(name);
@@ -931,6 +934,10 @@ bool is_mask_edits_folder(const std::string& path) {
return named(fs::path(path), gui::mask::kLayerDirName);
}
bool is_feature_mask_folder(const std::string& path) {
return named(fs::path(path), kFeatureMaskDirName);
}
void resolve_photo_folder(const std::string& picked, std::string& images,
std::string& masks) {
std::error_code ec;
@@ -1152,6 +1159,7 @@ bool DatasetPrep::run(const PrepJob& job_in, PrepResult& out, std::string& error
// can run per input (see generate_masks) instead of over one flat tree.
struct Prepared {
std::string images, masks; // absolute
std::string feature_masks; // absolute; "" without a feature prompt
// This input's masks already exist: brought along by the input, taken
// from its alpha channel, or kept by a resumed run.
bool have_masks = false;
@@ -1298,20 +1306,29 @@ bool DatasetPrep::run(const PrepJob& job_in, PrepResult& out, std::string& error
// answer the user gave while watching them go by.
if (refresh_masks) refresh_masks(job);
std::vector<int64_t> mask_planned(job.inputs.size(), 0);
if (job.mask_enable) {
// A subject model (BiRefNet) finds what to mask by itself.
// A subject model (BiRefNet) finds what to mask by itself, and reads no
// words to find the sky with.
bool subject = false;
#ifdef SS_BUILD_SAM
subject = sam::is_subject_model(job.mask_model_path);
subject = job.mask_enable && sam::is_subject_model(job.mask_model_path);
#endif
if (!subject && job.mask_prompt.empty() && job.mask_clicks.empty()) {
const bool want_features =
job.mask_enable && !subject && !job.mask_feature_prompt.empty();
if (want_features)
for (size_t i = 0; i < job.inputs.size(); i++)
per[i].feature_masks =
under((ws / kFeatureMaskDirName).string(), job.inputs[i].subdir).string();
std::vector<int64_t> mask_planned(job.inputs.size(), 0);
if (job.mask_enable) {
if (!subject && job.mask_prompt.empty() && job.mask_clicks.empty() &&
!want_features) {
error = lmsg::err_mask_no_target.get();
return false;
}
// Half-masking reads as a masking run that worked, so refuse instead:
// without a text prompt, every input needs clicks of its own.
if (!subject && job.mask_prompt.empty()) {
if (!subject && job.mask_prompt.empty() && !job.mask_clicks.empty()) {
std::string unprompted;
for (size_t i = 0; i < job.inputs.size(); i++) {
if (per[i].have_masks || !clicks_for(job, job.inputs[i]).empty())
@@ -1325,7 +1342,7 @@ bool DatasetPrep::run(const PrepJob& job_in, PrepResult& out, std::string& error
}
}
for (size_t i = 0; i < job.inputs.size(); i++)
if (!per[i].have_masks) {
if (!per[i].have_masks || want_features) {
mask_planned[i] = count_images(per[i].images, per[i].masks);
_masks_tally.plan(mask_planned[i]);
}
@@ -1352,11 +1369,13 @@ bool DatasetPrep::run(const PrepJob& job_in, PrepResult& out, std::string& error
// masks an input brought with it are safe.
if (job.redo_masks && job.inputs[i].mask_dir.empty())
clear_generated(per[i].masks, ws);
// Already masked by whoever drew the masks -- or the alpha -- this
// input arrived with. Segmenting over those would replace an
// answer the user already has.
if (per[i].have_masks) continue;
if (job.redo_masks && want_features) clear_generated(per[i].feature_masks, ws);
// Masks the input arrived with, or its alpha, are an answer the user
// already has and are not segmented over; the feature prompt asks
// another question, so it still runs.
if (per[i].have_masks && !want_features) continue;
if (!generate_masks(job, job.inputs[i], per[i].images, per[i].masks,
per[i].feature_masks, !per[i].have_masks,
per[i].stencil_folded, error))
return false;
_masks_tally.settle(mask_planned[i], mask_planned[i]);
@@ -1386,6 +1405,10 @@ bool DatasetPrep::run(const PrepJob& job_in, PrepResult& out, std::string& error
out.mask_dir = (ws / "masks").string();
out.mask_dir_cfg = "masks";
}
const fs::path feature_root = ws / kFeatureMaskDirName;
if (want_features && fs::is_directory(feature_root, mec) &&
!fs::is_empty(feature_root, mec))
out.feature_mask_dir = feature_root.string();
// Hand corrections outlive a re-run: whatever wrote masks/ this time, the
// editor's layers are re-applied over every mask whose bytes changed.
const fs::path layer_root = ws / gui::mask::kLayerDirName;
@@ -2510,6 +2533,7 @@ bool DatasetPrep::split_packed_frames(const PrepInput& in,
bool DatasetPrep::generate_masks(const PrepJob& job, const PrepInput& in,
const std::string& images,
const std::string& masks,
const std::string& feature_masks, bool train,
bool& folded, std::string& error) {
if (!backends().builtin_masking) {
error = lmsg::err_no_builtin_segmentation.get();
@@ -2519,15 +2543,18 @@ bool DatasetPrep::generate_masks(const PrepJob& job, const PrepInput& in,
error = lmsg::err_mask_model_not_downloaded.get();
return false;
}
return generate_masks_builtin(job, in, images, masks, folded, error);
return generate_masks_builtin(job, in, images, masks, feature_masks, train, folded,
error);
}
bool DatasetPrep::generate_masks_builtin(const PrepJob& job, const PrepInput& in,
const std::string& images,
const std::string& masks,
const std::string& feature_masks, bool train,
bool& folded, std::string& error) {
#ifndef SS_BUILD_SAM
(void)job; (void)in; (void)images; (void)masks; (void)folded;
(void)job; (void)in; (void)images; (void)masks; (void)feature_masks; (void)train;
(void)folded;
error = backends().masking_reason;
return false;
#else
@@ -2537,7 +2564,14 @@ bool DatasetPrep::generate_masks_builtin(const PrepJob& job, const PrepInput& in
// Never the masks themselves: a nested masks/ is only possible for photos
// read where they are, and those already have masks -- but the guard costs
// nothing and the alternative is masking a folder of masks.
const std::vector<fs::path> files = walk_images(images, masks);
std::vector<fs::path> files = walk_images(images, masks);
const bool features = !feature_masks.empty();
if (features)
files.erase(std::remove_if(files.begin(), files.end(),
[&](const fs::path& f) {
return inside(f, fs::path(feature_masks));
}),
files.end());
if (files.empty()) {
error = lmsg::err_no_images_to_mask.get();
return false;
@@ -2565,8 +2599,9 @@ bool DatasetPrep::generate_masks_builtin(const PrepJob& job, const PrepInput& in
mo.detector = job.mask_detector_path;
mo.detector_threshold = job.mask_detector_threshold;
mo.device = job.device;
mo.text = job.mask_prompt;
mo.neg_text = job.mask_negative_prompt;
mo.text = train ? job.mask_prompt : "";
mo.neg_text = train ? job.mask_negative_prompt : "";
mo.feature_text = features ? job.mask_feature_prompt : "";
mo.keep_prompted = job.mask_keep_subject;
mo.max_size = job.mask_max_image_size;
mo.dilate_ratio = job.mask_dilate_ratio;
@@ -2578,8 +2613,9 @@ bool DatasetPrep::generate_masks_builtin(const PrepJob& job, const PrepInput& in
// carry one that does not apply -- and a video only when it was asked for.
// A click needs one either way, since it says nothing about any frame but
// its own.
const std::vector<MaskClick> clicks = clicks_for(job, in);
mo.video = (in.is_video && job.mask_memory) || !clicks.empty();
const std::vector<MaskClick> clicks =
train ? clicks_for(job, in) : std::vector<MaskClick>{};
mo.video = (in.is_video && job.mask_memory && train) || !clicks.empty();
// A click's own frame number survives whenever the numbering it was
// recorded against did. ffmpeg resampled the video, and a packed photo's
// preview counted the files before the cut: there only the fraction holds.
@@ -2587,11 +2623,19 @@ bool DatasetPrep::generate_masks_builtin(const PrepJob& job, const PrepInput& in
clicks, cameras, ids,
/*exact=*/(!in.is_video && in.packed_lenses < 2) || by_stem);
sam::Masker masker;
if (!masker.init(mo, error)) return false;
// Nothing named for training still writes masks/ when there is a stencil
// to put in it, and an all-white one when there is not would be noise.
const bool write_train = train && (masker.hasTarget() || !in.stencil.empty());
const bool write_features = features && masker.hasFeatureMask();
if (!write_train && !write_features) return true;
// The stencil goes in here rather than in a pass of its own: it is one AND
// over a mask that is already in memory, against a decode and a re-encode
// of every PNG on disk (~131 ms per 2880-square fisheye frame).
StencilRaster stencil;
if (!in.stencil.empty()) {
if (write_train && !in.stencil.empty()) {
stencil.build(in.stencil, image_root, files,
[&](const std::string& camera, const app::BorderDetect& d) {
const std::string name =
@@ -2609,31 +2653,39 @@ bool DatasetPrep::generate_masks_builtin(const PrepJob& job, const PrepInput& in
folded = true;
}
sam::Masker masker;
if (!masker.init(mo, error)) return false;
const fs::path mask_root(masks), feature_root(feature_masks);
const fs::path mask_root(masks);
fs::create_directories(mask_root, ec);
// What a resumed run still has to do, decided before anything is read: the
// prefetcher below reads ahead, and reading a frame whose mask is already
// on disk is the one decode that buys nothing.
// What a resumed run still has to do, decided before the prefetcher reads
// ahead: a frame whose masks are all on disk is a decode that buys nothing.
struct Dst {
fs::path path; // "" = a mask this run does not make
bool write = false; // false = already on disk
};
std::vector<size_t> todo;
std::vector<fs::path> todo_files, todo_dst;
std::vector<fs::path> todo_files;
std::vector<Dst> todo_dst, todo_fdst;
int done = 0;
for (size_t i = 0; i < files.size(); i++) {
auto wanted = [&](bool on, const fs::path& root, const fs::path& rel) {
Dst d;
if (!on) return d;
// Masks mirror the image tree, so cam0/ and cam1/ keep their names.
d.path = root / rel.parent_path() / (rel.stem().string() + ".png");
d.write = !(job.resume && !job.redo_masks && fs::exists(d.path, ec));
if (d.write) fs::create_directories(d.path.parent_path(), ec);
return d;
};
for (size_t i = 0; i < files.size(); i++) {
const fs::path rel = under_root(files[i], image_root);
const fs::path dst = mask_root / rel.parent_path() /
(rel.stem().string() + ".png");
if (job.resume && !job.redo_masks && fs::exists(dst, ec)) {
Dst dst = wanted(write_train, mask_root, rel);
Dst fdst = wanted(write_features, feature_root, rel);
if (!dst.write && !fdst.write) {
++done;
continue;
}
fs::create_directories(dst.parent_path(), ec);
todo.push_back(i);
todo_files.push_back(files[i]);
todo_dst.push_back(dst);
todo_dst.push_back(std::move(dst));
todo_fdst.push_back(std::move(fdst));
}
// Encoding a 1080p mask costs about a third of what the model costs to
@@ -2650,33 +2702,44 @@ bool DatasetPrep::generate_masks_builtin(const PrepJob& job, const PrepInput& in
log(fmt(lmsg::warn_unreadable_skipped, {todo_files[k].string()}), false);
continue;
}
sam::Mask mask;
if (!masker.run(img, mask, nullptr, ids[todo[k]])) {
sam::Mask mask, fmask;
if (!masker.run(img, mask, nullptr, ids[todo[k]],
write_features ? &fmask : nullptr)) {
error = fmt(lmsg::err_masking_failed_on,
{todo_files[k].filename().string(), masker.lastError()});
return false;
}
// Still the way up the model saw it, which is the frame the stencil's
// shapes were drawn in.
if (!stencil.apply(todo_files[k], image_root, mask, error)) return false;
if (write_train && !stencil.apply(todo_files[k], image_root, mask, error))
return false;
// And back into the frame the file stores, which is the one the
// trainer reads it against (docs/datasets.md, "EXIF orientation") and
// the one the reel re-reads a frame nobody watched go by in.
const sfm::ExifTransform back = app::inverse_turn(turn);
app::turn_pixels(back, 1, mask.data, mask.width, mask.height);
app::turn_pixels(back, 1, fmask.data, fmask.width, fmask.height);
app::turn_pixels(back, img.channels, img.data, img.width, img.height);
if (_films.masks) {
FilmFrame f;
f.name = under_root(todo_files[k], image_root).generic_string();
f.image_path = todo_files[k].string();
f.mask_path = todo_dst[k].string();
f.mask_path = todo_dst[k].path.string();
f.feature_mask_path = todo_fdst[k].path.string();
_films.masks->add(f, _films.masks->wants() ? img.data.data() : nullptr,
img.width, img.height, mask.data.data());
img.width, img.height,
write_train ? mask.data.data() : nullptr, {},
fmask.data.empty() ? nullptr : fmask.data.data());
}
auto write = [&](sam::Mask& m, const Dst& dst) {
if (!dst.write || m.data.empty()) return;
app::WriteJob wj;
wj.mask = std::move(mask);
wj.path = todo_dst[k].string();
wj.mask = std::move(m);
wj.path = dst.path.string();
writers.submit(std::move(wj));
};
write(mask, todo_dst[k]);
write(fmask, todo_fdst[k]);
// Sequenced deliberately: `update(++done, done == n)` leaves the
// argument's read of `done` unsequenced against the increment, so
// whether the last image is reported as the last one came down to the
+17 -8
View File
@@ -273,6 +273,10 @@ struct PrepJob {
bool mask_enable = false;
std::string mask_prompt; // "people; cars; ..."
std::string mask_negative_prompt;
// What reconstruction skips and training keeps ("sky; cloud"): its own
// tree, feature_masks/, which SfM intersects with masks/. Needs a text
// model, and is run even over an input that brought its own masks.
std::string mask_feature_prompt;
bool mask_keep_subject = false; // prompt names what to KEEP, not remove
// Share of its own size every matched object's boundary moves by before
// the mask is written, so the PNGs on disk carry it. SIGNED as
@@ -394,6 +398,9 @@ struct PrepResult {
// only where the run handed them on untouched; what it wrote itself is in
// the usual convention and the readers need no flag.
bool mask_dir_flipped = false;
// Absolute, "" when this run asked for none: PrepJob::mask_feature_prompt's
// masks, for feature extraction only. Never flipped.
std::string feature_mask_dir;
int n_images = 0;
// images/ came out holding one sub-folder per camera -- several inputs, or
// a multi-track video -- so intrinsics must not be shared across them.
@@ -582,7 +589,7 @@ bool folder_looks_like_dataset(const std::string& dir);
struct WorkspaceState {
bool frames = false; // images/ this run would extract into
bool features = false; // features/, matches.bin, database.db -- reusable
bool masks = false; // masks/ this run would generate into
bool masks = false; // masks/ or feature_masks/ this run would generate into
bool input_masks = false; // masks an input came with (PrepInput::mask_dir)
// A reconstruction any dataset reader can open: this run's own sparse/, or
// the transforms.json, root-level COLMAP files or Metashape export of a
@@ -614,6 +621,10 @@ bool is_mask_folder(const std::string& path);
// finished dataset carries beside images/ and masks/ and which holds PNGs.
bool is_mask_edits_folder(const std::string& path);
// Where PrepJob::mask_feature_prompt's masks go, beside masks/ and mirroring it.
inline constexpr const char* kFeatureMaskDirName = "feature_masks";
bool is_feature_mask_folder(const std::string& path);
// One counter for a whole step, rather than one per input: a job with three
// videos in it should fill the bar once and never wind it back, which is the
// only thing a user watching a long extraction is reading it for.
@@ -704,18 +715,16 @@ private:
bool gather_photos(const PrepJob& job, const PrepInput& in,
const std::string& images, const std::string& masks,
bool& have_masks, std::string& error);
// Masks for ONE input's images. Run per input rather than over the whole
// tree so the tracker's memory bank never crosses from one capture into the
// next, and so clicks reach only the input they were drawn on.
//
// `folded` comes back true when the input's stencil was intersected into
// the masks as they were produced, which is what lets apply_stencil be
// skipped -- it would otherwise decode and re-encode every mask again.
// ONE input's masks, so a memory bank or a click never crosses captures.
// `folded`: its stencil went in as they were made, sparing apply_stencil a
// re-encode. `train` false writes only `feature_masks` ("" for none).
bool generate_masks(const PrepJob& job, const PrepInput& in,
const std::string& images, const std::string& masks,
const std::string& feature_masks, bool train,
bool& folded, std::string& error);
bool generate_masks_builtin(const PrepJob& job, const PrepInput& in,
const std::string& images, const std::string& masks,
const std::string& feature_masks, bool train,
bool& folded, std::string& error);
// The static stencil on its own, for the masks segmentation did not make.
// `merge_from` names the masks it folds in when they are not the ones it
+4
View File
@@ -48,6 +48,7 @@ namespace {
X("mask_text_detector", mask_detector_id) \
X("mask_prompt", mask.prompt) \
X("mask_negative_prompt", mask.negative_prompt) \
X("mask_feature_prompt", mask.feature_prompt) \
X("mask_keep_subject", mask.keep_subject) \
X("mask_dilate_ratio", mask.dilate_ratio) \
X("mask_shrink_ratio", mask.shrink_ratio) \
@@ -170,6 +171,9 @@ bool dataset_apply_preset(DatasetSettings& s, const std::string& name) {
// whatever they carry and their shadow.
s.sfm.prep.mask_enable = true;
s.mask.prompt = "person; hand; backpack; shadow of person";
// Outdoors half of every frame is sky, and a clear one yields no
// feature points while a cloudy one yields points that drift.
s.mask.feature_prompt = "sky; cloud";
s.sfm.mask_features = true;
s.border_enable = true;
return true;
+1 -1
View File
@@ -27,7 +27,7 @@ struct DatasetSettings {
ColmapJob colmap;
MaskSettings mask; // clicks excluded -- see the header note
std::string mask_model_id = "sam2.1-base-plus";
std::string mask_detector_id = "gdino-tiny"; // a TextDetector, ModelCache.h
std::string mask_detector_id = "gdino-base"; // a TextDetector, ModelCache.h
bool use_found_masks = true;
bool border_enable = false;
// A saved stencil's name (StencilPreset.h), drawn on every input.
+6 -4
View File
@@ -44,7 +44,8 @@ std::vector<KeyPoint2D> FilmReel::thin(const KeyPoint2D* pts, size_t count,
}
void FilmReel::add(const FilmFrame& f, const uint8_t* rgb, int w, int h,
const uint8_t* mask, FramePoints points) {
const uint8_t* mask, FramePoints points,
const uint8_t* feature_mask) {
Picture pic;
std::vector<KeyPoint2D> pts;
if (rgb) {
@@ -53,7 +54,7 @@ void FilmReel::add(const FilmFrame& f, const uint8_t* rgb, int w, int h,
std::lock_guard<std::mutex> lk(_mu);
target = _target;
}
make_picture(rgb, w, h, mask, target, pic);
make_picture(rgb, w, h, mask, target, pic, feature_mask);
pts = thin(points.pts, points.count, w, h);
}
append(f, std::move(pic), std::move(pts));
@@ -67,7 +68,7 @@ void FilmReel::add_loaded(const FilmFrame& f) {
}
Picture pic;
if (!f.panels.empty()) load_picture_row(f.panels, target, pic);
else load_picture(f.image_path, f.mask_path, target, pic);
else load_picture(f.image_path, f.mask_path, target, pic, false, f.feature_mask_path);
append(f, std::move(pic), {});
}
@@ -176,7 +177,8 @@ void FilmReel::loader_loop() {
try {
if (!f.panels.empty()) {
load_picture_row(f.panels, target, pic);
} else if (load_picture(f.image_path, f.mask_path, target, pic) &&
} else if (load_picture(f.image_path, f.mask_path, target, pic, false,
f.feature_mask_path) &&
!f.points_path.empty()) {
std::vector<KeyPoint2D> kp;
if (read_keypoints_file(f.points_path, pic.src_w, pic.src_h, kp))
+6 -4
View File
@@ -47,6 +47,7 @@ struct FilmFrame {
std::string name; // the caption; the image's name in the dataset
std::string image_path;
std::string mask_path; // "" when there is none
std::string feature_mask_path; // for feature extraction alone; "" for none
std::string points_path; // feature file to overlay; "" for none
// Set instead of the three above when the frame is a ROW -- the geometry
// step's photograph, normal map and depth map of one frame, which are
@@ -73,11 +74,12 @@ public:
// keeps the copy off a decode running at hundreds of frames a second. The
// frame is registered either way, so the slider still reaches it.
bool wants(double min_interval_s = 0.12) const;
// Appends `f`. `rgb` (w*h*3) and `mask` (w*h, 255 = keep) are the pixels
// the producer already holds; without them the picture is read from
// `f.image_path` when it is asked for.
// Appends `f`. `rgb` (w*h*3), `mask` and `feature_mask` (w*h, 255 = keep)
// are the pixels the producer already holds; without them the picture is
// read from `f.image_path` when it is asked for.
void add(const FilmFrame& f, const uint8_t* rgb = nullptr, int w = 0,
int h = 0, const uint8_t* mask = nullptr, FramePoints points = {});
int h = 0, const uint8_t* mask = nullptr, FramePoints points = {},
const uint8_t* feature_mask = nullptr);
// Appends `f` AND reads its files now, on the calling thread. The reel
// follows the newest picture it HOLDS (see draw), so a producer that only
// registers paths never advances the slider.
+8
View File
@@ -3405,6 +3405,7 @@ void GuiApp::sync_dataset_jobs() {
const MaskModelFiles mask_model = selected_mask_model();
prep.mask_prompt = mask_model.text ? _mask.prompt : "";
prep.mask_negative_prompt = mask_model.text ? _mask.negative_prompt : "";
prep.mask_feature_prompt = mask_model.text ? _mask.feature_prompt : "";
prep.mask_keep_subject = _mask.keep_subject;
prep.mask_max_image_size = _mask.max_image_size;
prep.mask_dilate_ratio = _mask.boundary_ratio();
@@ -3442,6 +3443,7 @@ void GuiApp::sync_dataset_jobs() {
_colmap_job.mask_enable = prep.mask_enable;
_colmap_job.mask_prompt = prep.mask_prompt;
_colmap_job.mask_negative_prompt = prep.mask_negative_prompt;
_colmap_job.mask_feature_prompt = prep.mask_feature_prompt;
_colmap_job.mask_keep_subject = prep.mask_keep_subject;
_colmap_job.mask_max_image_size = prep.mask_max_image_size;
_colmap_job.mask_dilate_ratio = prep.mask_dilate_ratio;
@@ -4942,6 +4944,12 @@ void GuiApp::draw_masking_options() {
}
if (ui::CollapsingHeader(dmsg::mask_advanced)) {
if (text) {
ImGui::SetNextItemWidth(px(320.0f));
ui::InputTextEnglish(dmsg::mask_features_only, "sky; cloud",
&_mask.feature_prompt);
ui::help_on_hover(dmsg::mask_features_only_help);
}
// The preview reads these off the same fields. Grounding DINO scores
// on its own scale, so the slider is its threshold there, and it keeps
// every box, so there is no NMS to set.
+1
View File
@@ -15,6 +15,7 @@ namespace gui {
struct MaskSettings {
std::string prompt; // "people; cars"
std::string negative_prompt;
std::string feature_prompt; // "sky; cloud": PrepJob::mask_feature_prompt
bool keep_subject = false; // prompt names what to KEEP
// How far the boundary moves from where the model drew it, as a share of
// the object's own size. Two, because the polarities want opposite
+32
View File
@@ -0,0 +1,32 @@
#pragma once
// How a masked pixel is drawn over its photograph, by the mask preview and by
// the run's reel alike, so the two read as one answer.
#include <algorithm>
#include <cstdint>
namespace gui {
// Dropped from everything: dimmed and pulled to red.
inline void tint_removed(uint8_t* px) {
px[0] = (uint8_t)(px[0] / 3 + 150);
px[1] = (uint8_t)(px[1] / 3);
px[2] = (uint8_t)(px[2] / 3);
}
// Trained on, but kept out of feature extraction: amber hatching over a light
// wash. Lightness and pattern tell it from the red, not hue alone, which a
// red-green colour-blind eye does not separate from orange.
inline void tint_features_only(uint8_t* px, int x, int y, int period) {
static constexpr int kAmber[3] = {255, 176, 32};
const int a = (x + y) % period < (period + 2) / 4 ? 170 : 80; // of 256
for (int c = 0; c < 3; c++)
px[c] = (uint8_t)((px[c] * (256 - a) + kAmber[c] * a) >> 8);
}
// The hatch spacing, in the pixels of a w x h picture, that looks the same on
// screen whatever size the picture was made at.
inline int hatch_period(int w, int h) { return std::max(6, std::max(w, h) / 90); }
} // namespace gui
+48 -32
View File
@@ -3,6 +3,7 @@
#include "app/gui/Picture.h"
#include "app/DepthColor.h"
#include "app/gui/MaskTint.h"
#include "app/FrameLook.h" // app::photo_turn
#include "app/FrameMask.h" // app::load_rgb, app::load_stencil
#include "core/ExrImage.h"
@@ -17,14 +18,6 @@ namespace gui {
namespace {
// The masked-out region, tinted the way the mask preview tints it so that the
// two read as one answer.
void tint(uint8_t* px, int r, int g, int b) {
px[0] = (uint8_t)(r / 3 + 150);
px[1] = (uint8_t)(g / 3);
px[2] = (uint8_t)(b / 3);
}
// Area average to an exact size. The row's panels arrive at three different
// resolutions and have to line up before they can be laid side by side.
void scale_to(const uint8_t* src, int sw, int sh, int dw, int dh,
@@ -117,17 +110,21 @@ void box_photo(const uint8_t* rgb, int w, int h, int step, Picture& out) {
}
}
// Tints each of `out`'s blocks in which fewer than half the source pixels are
// What a block of the picture is drawn as; the higher one wins.
enum : uint8_t { kMarkFeaturesOnly = 1, kMarkRemoved = 2 };
// Marks each of `out`'s blocks in which fewer than half the source pixels are
// kept. A mask of another size than its w x h image is sampled nearest: a
// mask that came with the capture rather than one the run made.
void tint_blocks(const uint8_t* mask, int mw, int mh, int w, int h, int step,
bool mask_flipped, Picture& out) {
void mark_blocks(const uint8_t* mask, int mw, int mh, int w, int h, int step,
bool mask_flipped, uint8_t level, const Picture& out,
std::vector<uint8_t>& marks) {
marks.resize((size_t)out.w * out.h, 0);
auto mark = [&](size_t i) { marks[i] = std::max(marks[i], level); };
const bool same = mw == w && mh == h;
if (same && step == 1) {
for (size_t i = 0; i < (size_t)w * h; i++) {
uint8_t* px = &out.rgb[i * 3];
if ((mask[i] > 127) == mask_flipped) tint(px, px[0], px[1], px[2]);
}
for (size_t i = 0; i < (size_t)w * h; i++)
if ((mask[i] > 127) == mask_flipped) mark(i);
return;
}
for (int y = 0; y < out.h; y++) {
@@ -142,27 +139,41 @@ void tint_blocks(const uint8_t* mask, int mw, int mh, int w, int h, int step,
keep += row[same ? sx : std::min(mw - 1, sx * mw / w)] > 127;
}
if (mask_flipped) keep = n - keep;
if (keep * 2 < n) {
uint8_t* px = &out.rgb[((size_t)y * out.w + x) * 3];
tint(px, px[0], px[1], px[2]);
if (keep * 2 < n) mark((size_t)y * out.w + x);
}
}
}
void tint_marked(const std::vector<uint8_t>& marks, Picture& out) {
if (marks.empty()) return;
const int period = hatch_period(out.w, out.h);
for (int y = 0; y < out.h; y++)
for (int x = 0; x < out.w; x++) {
const size_t i = (size_t)y * out.w + x;
if (marks[i] == kMarkRemoved) tint_removed(&out.rgb[i * 3]);
else if (marks[i] == kMarkFeaturesOnly)
tint_features_only(&out.rgb[i * 3], x, y, period);
}
}
} // namespace
void make_picture(const uint8_t* rgb, int w, int h, const uint8_t* mask,
int max_side, Picture& out) {
int max_side, Picture& out, const uint8_t* feature_mask) {
out = Picture{};
if (!rgb || w <= 0 || h <= 0) return;
const int step = size_picture(w, h, max_side, out);
box_photo(rgb, w, h, step, out);
if (mask) tint_blocks(mask, w, h, w, h, step, false, out);
std::vector<uint8_t> marks;
if (mask) mark_blocks(mask, w, h, w, h, step, false, kMarkRemoved, out, marks);
if (feature_mask)
mark_blocks(feature_mask, w, h, w, h, step, false, kMarkFeaturesOnly, out, marks);
tint_marked(marks, out);
}
bool load_picture(const std::string& image_path, const std::string& mask_path,
int max_side, Picture& out, bool mask_flipped) {
int max_side, Picture& out, bool mask_flipped,
const std::string& feature_mask_path) {
const auto fail = [&out] {
out.rgb.clear();
out.w = out.h = out.src_w = out.src_h = out.made_for = 0;
@@ -181,21 +192,26 @@ bool load_picture(const std::string& image_path, const std::string& mask_path,
step = size_picture(w, h, max_side, out);
box_photo(rgb.get(), w, h, step, out);
}
if (mask_path.empty()) return true;
// stb's buffer in place, unless the mask needs what load_stencil adds: an
// EXR decode, or the EXIF turn a JPEG mask may carry.
std::vector<uint8_t> marks;
auto mark_file = [&](const std::string& path, bool flipped, uint8_t level) {
if (path.empty()) return;
// stb's buffer in place, unless the mask needs what load_stencil adds:
// an EXR decode, or the EXIF turn a JPEG mask may carry.
int mw = 0, mh = 0;
StbPixels m(exr::is_exr(mask_path) ? nullptr
: stbi_load(mask_path.c_str(), &mw, &mh, &comp, 1));
if (m && app::photo_turn(mask_path).identity()) {
tint_blocks(m.get(), mw, mh, w, h, step, mask_flipped, out);
return true;
StbPixels m(exr::is_exr(path) ? nullptr
: stbi_load(path.c_str(), &mw, &mh, &comp, 1));
if (m && app::photo_turn(path).identity()) {
mark_blocks(m.get(), mw, mh, w, h, step, flipped, level, out, marks);
return;
}
m.reset();
std::vector<uint8_t> stencil;
if (app::load_stencil(mask_path, mw, mh, stencil))
tint_blocks(stencil.data(), mw, mh, w, h, step, mask_flipped, out);
if (app::load_stencil(path, mw, mh, stencil))
mark_blocks(stencil.data(), mw, mh, w, h, step, flipped, level, out, marks);
};
mark_file(mask_path, mask_flipped, kMarkRemoved);
mark_file(feature_mask_path, false, kMarkFeaturesOnly);
tint_marked(marks, out);
return true;
}
+6 -5
View File
@@ -32,17 +32,18 @@ struct Picture {
bool fits(int side) const { return made_for >= side; }
};
// Compose `rgb` (w*h*3) with `mask` (w*h, 255 = keep; null for none), box
// filtered down to `max_side` on the long edge. 0, or a size the source does
// not reach, keeps the source resolution: nothing here ever upscales.
// Compose `rgb` (w*h*3) with `mask` and `feature_mask` (w*h, 255 = keep, null
// for none; MaskTint.h), box filtered down to `max_side` on the long edge. 0,
// or a size the source does not reach, keeps the source: it never upscales.
void make_picture(const uint8_t* rgb, int w, int h, const uint8_t* mask,
int max_side, Picture& out);
int max_side, Picture& out, const uint8_t* feature_mask = nullptr);
// The same from files. `mask_path` may be empty or absent; a mask stored at
// another size than its image is sampled to it. `mask_flipped`: the file's
// 255 is drop (TrainConfig::flip_mask). Reuses `out`'s buffer.
bool load_picture(const std::string& image_path, const std::string& mask_path,
int max_side, Picture& out, bool mask_flipped = false);
int max_side, Picture& out, bool mask_flipped = false,
const std::string& feature_mask_path = "");
// One panel of a row picture.
struct PicturePanel {
+44 -14
View File
@@ -5,6 +5,7 @@
#include "app/gui/Layout.h"
#include "app/gui/MaskPrompt.h"
#include "app/gui/MaskTint.h"
#include "app/gui/Ui.h"
#include "core/PolygonFill.h"
@@ -99,6 +100,7 @@ void SegmentPanel::open(const PreviewSource& src, const MaskModelFiles& model) {
{
std::lock_guard<std::mutex> lk(_mu);
_kept_fraction = -1.0f;
_feature_fraction = -1.0f;
_frames.clear();
_all_files.clear();
_folders.clear();
@@ -438,6 +440,7 @@ void SegmentPanel::start_job(const MaskSettings& s,
_preview_h = img.h;
_preview_dirty = true;
_kept_fraction = -1.0f;
_feature_fraction = -1.0f;
};
try {
if (!_job) _job = std::make_unique<Job>();
@@ -493,7 +496,9 @@ void SegmentPanel::start_job(const MaskSettings& s,
const bool subject = model.kind == MaskModelKind::Subject;
const std::string prompt = model.text ? settings.prompt : "";
const std::string negative = model.text ? settings.negative_prompt : "";
const bool wants_model = subject || !prompt.empty() || !clicks.empty();
const std::string feature = model.text ? settings.feature_prompt : "";
const bool wants_model =
subject || !prompt.empty() || !clicks.empty() || !feature.empty();
#ifndef SS_BUILD_SAM
stencil_only();
if (wants_model)
@@ -513,7 +518,7 @@ void SegmentPanel::start_job(const MaskSettings& s,
// Every field below changes what the masker computes, so the
// signature is what decides whether it can be reused. The weights
// are the expensive part and only the model path moves them.
std::string sig = prompt + "|" + negative + "|" +
std::string sig = prompt + "|" + negative + "|" + feature + "|" +
std::to_string((int)settings.keep_subject) + "|" +
std::to_string(settings.max_image_size) + "|" +
std::to_string(settings.threshold) + "|" +
@@ -534,6 +539,7 @@ void SegmentPanel::start_job(const MaskSettings& s,
mo.device = src.device;
mo.text = prompt;
mo.neg_text = negative;
mo.feature_text = feature;
mo.keep_prompted = settings.keep_subject;
mo.max_size = settings.max_image_size;
mo.threshold = settings.threshold;
@@ -567,26 +573,30 @@ void SegmentPanel::start_job(const MaskSettings& s,
img.height = j.frame.h;
img.channels = 3;
img.data = j.frame.px;
sam::Mask mask;
sam::Mask mask, fmask;
sam::Result detections;
if (!j.masker.run(img, mask, &detections, frame.index))
if (!j.masker.run(img, mask, &detections, frame.index, &fmask))
return set_error(j.masker.lastError());
// ---- composite ----
// Kept pixels stay as they are; masked-out pixels are dimmed and
// tinted red, which reads as "this will be ignored" far better
// than a separate black-and-white mask image next to the photo.
// Tinted over the photo, which reads as "ignored" far better than a
// mask beside it: red is dropped, amber hatching features-only.
std::vector<uint8_t> rgb = j.frame.px;
size_t kept = 0;
size_t kept = 0, for_features = 0;
const size_t n = (size_t)j.frame.w * j.frame.h;
const bool features = fmask.data.size() == n;
const int period = hatch_period(j.frame.w, j.frame.h);
for (size_t i = 0; i < n && i * 3 + 2 < rgb.size(); i++) {
if (i < mask.data.size() && mask.data[i] > 127) {
if (i >= mask.data.size() || mask.data[i] <= 127) {
tint_removed(&rgb[i * 3]);
} else if (features && fmask.data[i] <= 127) {
tint_features_only(&rgb[i * 3], (int)(i % j.frame.w),
(int)(i / j.frame.w), period);
kept += stencil_keeps(i) ? 1 : 0;
continue;
} else if (stencil_keeps(i)) {
kept++;
for_features++;
}
rgb[i * 3 + 0] = (uint8_t)(rgb[i * 3 + 0] / 3 + 150);
rgb[i * 3 + 1] = (uint8_t)(rgb[i * 3 + 1] / 3);
rgb[i * 3 + 2] = (uint8_t)(rgb[i * 3 + 2] / 3);
}
{
std::lock_guard<std::mutex> lk(_mu);
@@ -595,6 +605,8 @@ void SegmentPanel::start_job(const MaskSettings& s,
_preview_h = j.frame.h;
_preview_dirty = true;
_kept_fraction = n ? (float)((double)kept / (double)n) : -1.0f;
_feature_fraction =
n && features ? (float)((double)for_features / (double)n) : -1.0f;
_status.clear();
if (detections.detections.empty() && !prompt.empty() &&
clicks.empty() && !subject)
@@ -1583,6 +1595,9 @@ void SegmentPanel::draw(MaskSettings& settings, app::FrameStencil& stencil) {
}
ui::TextDisabled(dmsg::preview_legend);
if (_model.text && _model.kind != MaskModelKind::Subject && !_model.empty() &&
!settings.feature_prompt.empty())
ui::TextDisabled(dmsg::preview_legend_features);
ImGui::Separator();
// ---- controls ----
@@ -1673,6 +1688,16 @@ void SegmentPanel::draw(MaskSettings& settings, app::FrameStencil& stencil) {
edited |= draw_margin_slider(settings.dilate_ratio, settings.shrink_ratio, keep, -1.0f,
/*inline_label=*/false);
// Below the margin, which does not apply to it.
if (_model.text) {
ImGui::Spacing();
ui::Text(dmsg::mask_features_only);
ImGui::SetNextItemWidth(-1);
edited |= ui::InputTextEnglishRaw("##featprompt", "sky; cloud",
&settings.feature_prompt);
ui::help_on_hover(dmsg::preview_features_only_help);
}
ImGui::Spacing();
ImGui::Separator();
draw_objects(settings, edited);
@@ -1732,7 +1757,7 @@ void SegmentPanel::draw(MaskSettings& settings, app::FrameStencil& stencil) {
}
ImGui::Spacing();
float kept_fraction = -1.0f;
float kept_fraction = -1.0f, feature_fraction = -1.0f;
{
std::lock_guard<std::mutex> lk(_mu);
// Segmentation errors and progress come from the inference layer.
@@ -1741,11 +1766,16 @@ void SegmentPanel::draw(MaskSettings& settings, app::FrameStencil& stencil) {
else if (!_status.empty())
ui::TextDisabledRaw(_status);
kept_fraction = _kept_fraction;
feature_fraction = _feature_fraction;
}
if (kept_fraction >= 0.0f) {
char pct[16];
std::snprintf(pct, sizeof pct, "%.0f", 100.0f * kept_fraction);
ui::Text(dmsg::preview_kept_fraction, {pct});
if (feature_fraction >= 0.0f) {
std::snprintf(pct, sizeof pct, "%.0f", 100.0f * feature_fraction);
ui::Text(dmsg::preview_features_kept_fraction, {pct});
}
if (kept_fraction < 0.05f)
ui::TextColoredWrapped(ImVec4(0.95f, 0.75f, 0.30f, 1.0f),
settings.keep_subject
+2
View File
@@ -203,6 +203,8 @@ private:
bool _preview_dirty = false;
std::string _status, _error;
float _kept_fraction = -1.0f; // guarded by _mu
// What is left for feature points; -1 without a feature prompt. Ditto.
float _feature_fraction = -1.0f;
GLuint _tex = 0;
int _tex_w = 0, _tex_h = 0;
+6
View File
@@ -289,6 +289,7 @@ void SfmRunner::take_masking(PrepJob& prep) {
prep.mask_enable = _live.prep.mask_enable;
prep.mask_prompt = _live.prep.mask_prompt;
prep.mask_negative_prompt = _live.prep.mask_negative_prompt;
prep.mask_feature_prompt = _live.prep.mask_feature_prompt;
prep.mask_keep_subject = _live.prep.mask_keep_subject;
prep.mask_max_image_size = _live.prep.mask_max_image_size;
prep.mask_dilate_ratio = _live.prep.mask_dilate_ratio;
@@ -786,6 +787,11 @@ std::vector<std::string> SfmRunner::recon_args(const SfmJob& job,
// from an earlier run with masking on.
argv.push_back("--no-masks");
}
// Whatever the checkbox says: these exist only to be kept from the features.
if (!prep.feature_mask_dir.empty()) {
argv.push_back("--feature-masks");
argv.push_back(prep.feature_mask_dir);
}
for (const std::string& a : split_args(job.extra_args))
argv.push_back(a);
return argv;
+59 -10
View File
@@ -1,9 +1,10 @@
// dataset_prep_test -- the two places DatasetPrep (app/gui/DatasetPrep.h) meets
// the mask editor's layer folder: a re-run re-applies hand corrections over
// the masks it rewrites, and the camera scan never takes mask_edits/ for a
// camera. Real DatasetPrep::run, no model: the re-mask is the frame stencil,
// which is also checked per camera folder. Built without SS_BUILD_SAM, so it
// also checks that asking for a model fails.
// dataset_prep_test -- where DatasetPrep (app/gui/DatasetPrep.h) meets the
// folders beside images/: a re-run re-applies the mask editor's corrections
// over the masks it rewrites, the camera scan never takes mask_edits/ or
// feature_masks/ for a camera, and COLMAP's one mask tree is the two ANDed.
// Real DatasetPrep::run, no model: the re-mask is the frame stencil, which is
// also checked per camera folder. Built without SS_BUILD_SAM, so it also
// checks that asking for a model fails.
#include "app/FrameMask.h"
#include "app/gui/DatasetPrep.h"
@@ -143,12 +144,12 @@ void test_rerun_reapplies_corrections() {
}
// A sibling holding the same PNGs under another name IS a camera, so only
// the name guard keeps mask_edits/ out of the list.
// the name guard keeps mask_edits/ and feature_masks/ out of the list.
void test_camera_scan_skips_mask_edits() {
const fs::path root = scratch("scan");
write_jpg(root / "cam0" / "f0.jpg", 16, 12, 0);
write_jpg(root / "cam1" / "f0.jpg", 16, 12, 1);
for (const char* dir : {mk::kLayerDirName, "lookalike"})
for (const char* dir : {mk::kLayerDirName, gui::kFeatureMaskDirName, "lookalike"})
for (const char* f : {"cam0/f0.base.png", "cam0/f0.drop.png", "cam0/f0.keep.png"})
write_png(root / dir / f, 16, 12, 255);
const std::vector<std::string> cams = gui::camera_subfolders(root.string());
@@ -157,8 +158,9 @@ void test_camera_scan_skips_mask_edits() {
check(std::find(cams.begin(), cams.end(), "lookalike/cam0") != cams.end(),
"fixture: the layer PNGs under another name are taken for a camera: " + listed);
bool edits = false;
for (const std::string& c : cams) edits |= c.rfind(mk::kLayerDirName, 0) == 0;
check(!edits, "camera scan: nothing under mask_edits/ is listed: " + listed);
for (const std::string& c : cams)
edits |= c.rfind(mk::kLayerDirName, 0) == 0 || c.rfind(gui::kFeatureMaskDirName, 0) == 0;
check(!edits, "camera scan: nothing under mask_edits/ or feature_masks/: " + listed);
check(cams.size() == 3 && cams[0] == "cam0" && cams[1] == "cam1",
"camera scan: cam0, cam1 and the lookalike, nothing else: " + listed);
check(gui::is_mask_edits_folder((root / mk::kLayerDirName).string()) &&
@@ -200,6 +202,52 @@ void test_per_camera_stencil() {
"per camera: cam1 has its own, the right half out");
}
// feature_masks/ is the run's own, so a workspace holding only it is one to
// resume and one "clear this project" empties; and COLMAP, which reads one
// mask tree, gets it intersected with masks/ -- the flipped one read flipped.
void test_feature_masks_workspace() {
const int W = 8, H = 4;
const fs::path ws = scratch("featmasks");
write_jpg(ws / "images" / "cam0" / "a.jpg", W, H, 0);
write_jpg(ws / "images" / "cam0" / "b.jpg", W, H, 1);
write_jpg(ws / "images" / "cam0" / "c.jpg", W, H, 2);
auto png = [&](const fs::path& p, const std::vector<uint8_t>& px) {
std::error_code ec;
fs::create_directories(p.parent_path(), ec);
stbi_write_png(p.string().c_str(), W, H, 1, px.data(), W);
};
png(ws / gui::kFeatureMaskDirName / "cam0" / "a.png", box(W, H, 0, 2, W, H));
png(ws / gui::kFeatureMaskDirName / "cam0" / "b.png", box(W, H, 0, 0, W, H));
const gui::WorkspaceState st = gui::probe_workspace(ws.string(), {});
const std::vector<std::string> arts = gui::workspace_artifacts(ws.string(), {});
check(st.masks && st.resumable(), "probe: feature_masks/ alone is resumable");
check(std::find(arts.begin(), arts.end(), (ws / gui::kFeatureMaskDirName).string()) !=
arts.end(),
"artifacts: feature_masks/ is the run's to clear");
// masks/ marks what to REMOVE here: the left half of a, all of c.
png(ws / "masks" / "cam0" / "a.png", box(W, H, 0, 0, W / 2, H));
png(ws / "masks" / "cam0" / "c.png", box(W, H, 0, 0, W, H));
std::string err;
const fs::path out = ws / ".colmap_masks";
const int64_t n = app::intersect_mask_trees(
(ws / "images").string(), (ws / "masks").string(), /*flip_a=*/true,
(ws / gui::kFeatureMaskDirName).string(), out.string(), nullptr, err);
check(n == 3, "intersect: one mask per image that has either: " + std::to_string(n));
int w = 0, h = 0;
std::vector<uint8_t> a, b, c;
app::load_stencil((out / "cam0" / "a.png").string(), w, h, a);
app::load_stencil((out / "cam0" / "b.png").string(), w, h, b);
app::load_stencil((out / "cam0" / "c.png").string(), w, h, c);
check(a.size() == (size_t)W * H && at(a, W, 6, 3) == 255 && at(a, W, 1, 3) == 0 &&
at(a, W, 6, 0) == 0,
"intersect: a keeps only the right half of its lower rows");
check(b.size() == (size_t)W * H && at(b, W, 0, 0) == 255 && at(b, W, 7, 3) == 255,
"intersect: b, with no masks/ file, is its feature mask");
check(c.size() == (size_t)W * H && at(c, W, 3, 2) == 0,
"intersect: c, with no feature mask, is its flipped mask");
}
void test_model_masking_needs_segmentation() {
const fs::path root = scratch("nosam");
const fs::path photos = root / "photos";
@@ -225,6 +273,7 @@ int main() {
test_rerun_reapplies_corrections();
test_camera_scan_skips_mask_edits();
test_per_camera_stencil();
test_feature_masks_workspace();
test_model_masking_needs_segmentation();
std::printf("%s: %d failure(s)\n", SS_FILE, g_failures);
return g_failures;
@@ -80,6 +80,7 @@ static void test_dataset_preset() {
s.sfm.mask_features = false;
s.mask.prompt = "people; cars";
s.mask.negative_prompt = "statue";
s.mask.feature_prompt = "sky; cloud";
s.mask.keep_subject = true;
s.mask.dilate_ratio = 0.25f;
s.mask.shrink_ratio = 0.125f;
@@ -196,6 +197,7 @@ static void test_dataset_preset() {
CHECK_EQ(b.sfm.mask_features, s.sfm.mask_features);
CHECK_EQ(b.mask.prompt, s.mask.prompt);
CHECK_EQ(b.mask.negative_prompt, s.mask.negative_prompt);
CHECK_EQ(b.mask.feature_prompt, s.mask.feature_prompt);
CHECK_EQ(b.mask.keep_subject, s.mask.keep_subject);
CHECK_EQ(b.mask.dilate_ratio, s.mask.dilate_ratio);
CHECK_EQ(b.mask.shrink_ratio, s.mask.shrink_ratio);
+140
View File
@@ -3863,6 +3863,82 @@ SS_MSG(mask_negative_help_remove,
"Необязательно."),
TR("Yukarıdaki satıra uysa bile kalacak istisnalar. İsteğe bağlı."));
SS_MSG(mask_features_only,
EN("Hide from the reconstruction only"),
JA("再構成からだけ隠す"),
ZH_HANS("只在重建时避开"),
ZH_HANT("只在重建時避開"),
KO("재구성에서만 빼기"),
DE("Nur vor der Rekonstruktion verbergen"),
FR("Cacher à la reconstruction seulement"),
ES("Ocultar solo a la reconstrucción"),
PT("Esconder só da reconstrução"),
IT("Nascondere solo alla ricostruzione"),
NL("Alleen voor de reconstructie verbergen"),
RU("Скрывать только от реконструкции"),
TR("Yalnızca yeniden kurmadan gizle"));
SS_MSG(mask_features_only_help,
EN("What the reconstruction takes no feature point from, though training "
"still uses it: the sky, whose clouds drift and whose points are too far "
"off to place a camera by. Written to feature_masks/ beside masks/, from "
"the same pass over each frame. \"Try the mask...\" shows it hatched in "
"amber."),
JA("再構成は特徴点を取らず、学習はそのまま使うものです。雲が流れ、点が遠すぎて"
"カメラの位置決めの手がかりにならない空などです。各フレームを処理する同じ"
"パスの中で求め、masks/ の隣の feature_masks/ に書き出します。「マスクを"
"試す…」では琥珀色の斜線で表示されます。"),
ZH_HANS("重建时不从中取特征点、训练却照样使用的东西,比如天空——云在飘,点又"
"太远,定不了相机的位置。在处理每一帧的同一遍里求出,写到 masks/ 旁边"
"的 feature_masks/。在“试一下蒙版…”里以琥珀色斜线显示。"),
ZH_HANT("重建時不從中取特徵點、訓練卻照樣使用的東西,比如天空——雲在飄,點又"
"太遠,定不了相機的位置。在處理每一影格的同一遍裡求出,寫到 masks/ 旁邊"
"的 feature_masks/。在「試一下遮罩…」裡以琥珀色斜線顯示。"),
KO("재구성은 특징점을 뽑지 않지만 학습은 그대로 쓰는 것입니다. 구름이 흘러가고 "
"점이 너무 멀어 카메라 위치를 잡는 근거가 되지 못하는 하늘 같은 것입니다. "
"각 프레임을 처리하는 같은 패스에서 구해 masks/ 옆의 feature_masks/ 에 "
"씁니다. \"마스크 시험해 보기…\"에서는 호박색 빗금으로 보입니다."),
DE("Woraus die Rekonstruktion keinen Merkmalspunkt nimmt, was das Training "
"aber weiter nutzt: den Himmel, dessen Wolken ziehen und dessen Punkte zu "
"fern sind, um eine Kamera daran auszurichten. Landet in feature_masks/ "
"neben masks/, im selben Durchgang über jedes Bild ermittelt. „Maske "
"ausprobieren …“ zeigt es bernsteinfarben schraffiert."),
FR("Ce dont la reconstruction ne tire aucun point d'intérêt mais que "
"l'entraînement utilise quand même : le ciel, dont les nuages dérivent et "
"dont les points sont trop lointains pour situer une caméra. Écrit dans "
"feature_masks/ à côté de masks/, lors du même passage sur chaque image. "
"« Essayer le masque… » le montre hachuré d'ambre."),
ES("Lo que la reconstrucción no usa para ningún punto característico pero el "
"entrenamiento sí: el cielo, cuyas nubes se desplazan y cuyos puntos están "
"demasiado lejos para situar una cámara. Se escribe en feature_masks/ "
"junto a masks/, en la misma pasada por cada fotograma. «Probar la "
"máscara…» lo muestra rayado en ámbar."),
PT("O que a reconstrução não usa para nenhum ponto de característica, mas o "
"treino usa: o céu, cujas nuvens se movem e cujos pontos estão longe "
"demais para situar uma câmera. Gravado em feature_masks/ ao lado de "
"masks/, na mesma passagem por cada quadro. “Testar a máscara…” mostra "
"isso hachurado em âmbar."),
IT("Ciò da cui la ricostruzione non prende alcun punto caratteristico ma che "
"l'addestramento usa comunque: il cielo, le cui nuvole si spostano e i cui "
"punti sono troppo lontani per collocare una fotocamera. Scritto in "
"feature_masks/ accanto a masks/, nello stesso passaggio su ogni "
"fotogramma. «Prova la maschera…» lo mostra tratteggiato in ambra."),
NL("Waar de reconstructie geen kenmerkpunt uit haalt, maar wat de training "
"wel gebruikt: de lucht, waarvan de wolken drijven en de punten te ver weg "
"liggen om een camera op te plaatsen. Komt in feature_masks/ naast masks/, "
"uit dezelfde doorgang over elk beeld. \"Masker uitproberen…\" toont het "
"amberkleurig gearceerd."),
RU("То, из чего реконструкция не берёт ни одной особой точки, а обучение всё "
"равно использует: небо, где плывут облака, а точки слишком далеко, чтобы "
"по ним ставить камеру. Пишется в feature_masks/ рядом с masks/, за тот же "
"проход по каждому кадру. «Проверить маску…» показывает это янтарной "
"штриховкой."),
TR("Yeniden kurmanın hiçbir öznitelik noktası almadığı ama eğitimin yine de "
"kullandığı şeyler: bulutları kayan, noktaları bir kamerayı yerleştirmeye "
"yaramayacak kadar uzak olan gökyüzü gibi. Her karenin aynı geçişinde "
"bulunur ve masks/ yanındaki feature_masks/ klasörüne yazılır. \"Maskeyi "
"dene…\" bunu kehribar renkli taramayla gösterir."));
SS_MSG(mask_advanced,
EN("Advanced masking"),
JA("マスクの詳細設定"),
@@ -7450,6 +7526,70 @@ SS_MSG(preview_kept_fraction,
RU("остаётся {0}% кадра"),
TR("karenin %{0}'i tutuluyor"));
SS_MSG(preview_features_only_help,
EN("What the reconstruction takes no feature point from, though training "
"still uses it -- the sky, say. Hatched in amber on the picture."),
JA("再構成は特徴点を取らず、学習はそのまま使うもの(空など)です。画像上では"
"琥珀色の斜線で表示されます。"),
ZH_HANS("重建时不从中取特征点、训练却照样使用的东西,比如天空。画面上以琥珀色"
"斜线显示。"),
ZH_HANT("重建時不從中取特徵點、訓練卻照樣使用的東西,比如天空。畫面上以琥珀色"
"斜線顯示。"),
KO("재구성은 특징점을 뽑지 않지만 학습은 그대로 쓰는 것(하늘 등)입니다. "
"그림에서는 호박색 빗금으로 보입니다."),
DE("Woraus die Rekonstruktion keinen Merkmalspunkt nimmt, was das Training "
"aber weiter nutzt, etwa der Himmel. Im Bild bernsteinfarben schraffiert."),
FR("Ce dont la reconstruction ne tire aucun point d'intérêt mais que "
"l'entraînement utilise quand même, comme le ciel. Hachuré d'ambre sur "
"l'image."),
ES("Lo que la reconstrucción no usa para ningún punto característico pero el "
"entrenamiento sí, como el cielo. Rayado en ámbar sobre la imagen."),
PT("O que a reconstrução não usa para nenhum ponto de característica, mas o "
"treino usa, como o céu. Hachurado em âmbar na imagem."),
IT("Ciò da cui la ricostruzione non prende alcun punto caratteristico ma che "
"l'addestramento usa comunque, come il cielo. Tratteggiato in ambra "
"sull'immagine."),
NL("Waar de reconstructie geen kenmerkpunt uit haalt, maar wat de training "
"wel gebruikt, zoals de lucht. Amberkleurig gearceerd op het beeld."),
RU("То, из чего реконструкция не берёт ни одной особой точки, а обучение всё "
"равно использует, например небо. На снимке — янтарная штриховка."),
TR("Yeniden kurmanın hiçbir öznitelik noktası almadığı ama eğitimin yine de "
"kullandığı şeyler, örneğin gökyüzü. Görüntüde kehribar renkli taramayla "
"gösterilir."));
SS_MSG(preview_legend_features,
EN("Hatched amber = trained on, but hidden from the reconstruction."),
JA("琥珀色の斜線 = 学習には使うが、再構成からは隠す部分です。"),
ZH_HANS("琥珀色斜线 = 训练照用,但重建时避开。"),
ZH_HANT("琥珀色斜線 = 訓練照用,但重建時避開。"),
KO("호박색 빗금 = 학습에는 쓰지만 재구성에서는 뺍니다."),
DE("Bernsteinfarben schraffiert = wird trainiert, aber vor der "
"Rekonstruktion verborgen."),
FR("Hachuré d'ambre = entraîné, mais caché à la reconstruction."),
ES("Rayado en ámbar = se entrena, pero se oculta a la reconstrucción."),
PT("Hachurado em âmbar = treinado, mas escondido da reconstrução."),
IT("Tratteggio ambra = addestrato, ma nascosto alla ricostruzione."),
NL("Amberkleurig gearceerd = wordt getraind, maar voor de reconstructie "
"verborgen."),
RU("Янтарная штриховка — используется в обучении, но скрыто от "
"реконструкции."),
TR("Kehribar tarama = eğitimde kullanılır ama yeniden kurmadan gizlenir."));
SS_MSG(preview_features_kept_fraction,
EN("{0}% of the frame is left for feature points"),
JA("特徴点に使えるのはフレームの {0}% です"),
ZH_HANS("这一帧有 {0}% 可以取特征点"),
ZH_HANT("這一影格有 {0}% 可以取特徵點"),
KO("특징점을 뽑을 수 있는 부분: 프레임의 {0}%"),
DE("{0} % des Bildes bleiben für Merkmalspunkte"),
FR("{0} % de l'image reste pour les points d'intérêt"),
ES("queda el {0} % del fotograma para puntos característicos"),
PT("{0}% do quadro fica para pontos de característica"),
IT("resta il {0}% del fotogramma per i punti caratteristici"),
NL("{0}% van het beeld blijft over voor kenmerkpunten"),
RU("для особых точек остаётся {0}% кадра"),
TR("karenin %{0}'i öznitelik noktalarına kalıyor"));
SS_MSG(preview_almost_nothing_kept,
EN("Almost nothing is left -- the prompt matched very little of the "
"frame."),
+43
View File
@@ -551,6 +551,49 @@ SS_MSG(flip_mask_help,
TR("Her maskede korunanla yok sayılanı yer değiştirir; korunacak alan yerine "
"KALDIRILACAK alanı boyayan dışa aktarma araçları için"));
SS_MSG(feature_masks_help,
EN("A second directory of masks, intersected with --masks: keypoints are "
"kept only where both are nonzero. For what the reconstruction should "
"not use but training should, such as the sky. Never flipped"),
JA("2 つ目のマスクのディレクトリで、--masks と重ね合わせます。キーポイントは"
"両方が 0 でない画素にあるものだけが残ります。空のように、再構成には使わない"
"が学習には使うものに向けたものです。反転はしません"),
ZH_HANS("第二个掩码目录,与 --masks 取交集:只有两者都不为 0 的像素上的关键点"
"会被保留。用于重建不该用、训练却要用的区域,例如天空。不会被反转"),
ZH_HANT("第二個遮罩目錄,與 --masks 取交集:只有兩者都不為 0 的像素上的關鍵點"
"會被保留。用於重建不該用、訓練卻要用的區域,例如天空。不會被反轉"),
KO("두 번째 마스크 디렉터리로, --masks 와 교집합을 씁니다. 키포인트는 둘 다 "
"0 이 아닌 화소에서만 남습니다. 하늘처럼 재구성에는 쓰지 않지만 학습에는 "
"쓰는 영역을 위한 것입니다. 뒤집지 않습니다"),
DE("Ein zweites Maskenverzeichnis, mit --masks geschnitten: Schlüsselpunkte "
"bleiben nur, wo beide ungleich null sind. Für das, was die Rekonstruktion "
"nicht nutzen soll, das Training aber schon, etwa den Himmel. Wird nie "
"invertiert"),
FR("Un second dossier de masques, croisé avec --masks : les points clés ne "
"restent que là où les deux sont non nuls. Pour ce que la reconstruction ne "
"doit pas utiliser mais l'entraînement si, comme le ciel. Jamais inversé"),
ES("Una segunda carpeta de máscaras, intersecada con --masks: los puntos "
"clave solo quedan donde ambas son distintas de cero. Para lo que la "
"reconstrucción no debe usar pero el entrenamiento sí, como el cielo. "
"Nunca se invierte"),
PT("Uma segunda pasta de máscaras, intersectada com --masks: os pontos-chave "
"só ficam onde ambas são diferentes de zero. Para o que a reconstrução não "
"deve usar mas o treino sim, como o céu. Nunca é invertida"),
IT("Una seconda cartella di maschere, intersecata con --masks: i punti chiave "
"restano solo dove entrambe sono diverse da zero. Per ciò che la "
"ricostruzione non deve usare ma l'addestramento sì, come il cielo. Mai "
"invertita"),
NL("Een tweede map met maskers, gesneden met --masks: sleutelpunten blijven "
"alleen waar beide niet nul zijn. Voor wat de reconstructie niet moet "
"gebruiken maar de training wel, zoals de lucht. Wordt nooit omgekeerd"),
RU("Второй каталог масок, пересекаемый с --masks: ключевые точки остаются "
"только там, где обе маски ненулевые. Для того, что реконструкции брать не "
"нужно, а обучению нужно, например неба. Никогда не инвертируется"),
TR("--masks ile kesiştirilen ikinci bir maske dizini: anahtar noktalar yalnızca "
"ikisinin de sıfır olmadığı yerde kalır. Gökyüzü gibi, yeniden yapılandırmanın "
"kullanmaması ama eğitimin kullanması gereken alanlar için. Asla ters "
"çevrilmez"));
SS_MSG(mask_dir_help,
EN("Alias of --masks"),
JA("--masks の別名"),
+67 -19
View File
@@ -106,7 +106,7 @@ struct Masker::Impl {
// more that only ever propagates what was seeded into it by hand.
std::vector<std::unique_ptr<sam::Tracker>> trackers;
std::unique_ptr<sam::Tracker> visual;
std::vector<std::string> pos, neg;
std::vector<std::string> pos, neg, feat;
MaskOptions opts;
int64_t frame = 0;
// Seeds not yet applied, in frame order, and the instance each object was
@@ -158,6 +158,10 @@ const std::string& Masker::lastError() const {
}
sam::Session& Masker::session() { return impl_->session; }
bool Masker::subjectMode() const { return impl_->subject_mode; }
bool Masker::hasTarget() const {
return impl_->subject_mode || !impl_->pos.empty() || !impl_->opts.seeds.empty();
}
bool Masker::hasFeatureMask() const { return !impl_->feat.empty(); }
void Masker::unload() {
impl_->trackers.clear();
@@ -184,6 +188,10 @@ bool Masker::init(const MaskOptions& o, std::string& error) {
impl_->opts = o;
impl_->pos = split_phrases(o.text);
impl_->neg = split_phrases(o.neg_text);
impl_->feat = split_phrases(o.feature_text);
// Exceptions to nothing: with only feature phrases there is no union for
// them to carve out of.
if (impl_->pos.empty() && o.seeds.empty()) impl_->neg.clear();
impl_->pending = o.seeds;
// Applied in frame order regardless of the order they were given in, so a
// correction drawn on frame 200 cannot land before the click on frame 3
@@ -194,6 +202,8 @@ bool Masker::init(const MaskOptions& o, std::string& error) {
});
impl_->error.clear();
impl_->subject_mode = is_subject_model(o.model);
// BiRefNet reads no words, so it has none to find the sky with.
if (impl_->subject_mode) impl_->feat.clear();
try {
if (impl_->subject_mode) {
if (impl_->loaded_subject != o.model) {
@@ -213,7 +223,7 @@ bool Masker::init(const MaskOptions& o, std::string& error) {
impl_->detector.unload();
impl_->loaded_detector.clear();
} else if (impl_->loaded_detector != o.detector &&
(!impl_->pos.empty() || !impl_->neg.empty())) {
(!impl_->pos.empty() || !impl_->neg.empty() || !impl_->feat.empty())) {
impl_->detector.load(o.detector);
impl_->loaded_detector = o.detector;
}
@@ -221,7 +231,7 @@ bool Masker::init(const MaskOptions& o, std::string& error) {
error = e.what();
return false;
}
if (impl_->pos.empty() && impl_->pending.empty()) {
if (impl_->pos.empty() && impl_->pending.empty() && impl_->feat.empty()) {
error = "nothing to segment: give --text, or --point to click on an object";
return false;
}
@@ -238,7 +248,8 @@ bool Masker::init(const MaskOptions& o, std::string& error) {
error = impl_->session.lastError();
return false;
}
if (!impl_->pos.empty() && o.detector.empty() && !impl_->session.supportsTextPrompts()) {
if ((!impl_->pos.empty() || !impl_->feat.empty()) && o.detector.empty() &&
!impl_->session.supportsTextPrompts()) {
error = "this checkpoint has no text encoder, so --text cannot be used; seed an "
"instance with --point, or pair it with --detector gdino-tiny (SAM 2 "
"checkpoints are visual-only)";
@@ -267,9 +278,13 @@ bool Masker::init(const MaskOptions& o, std::string& error) {
}
bool Masker::run(const nn::Image& image, sam::Mask& out, sam::Result* overlay_out,
int64_t frame_id) {
int64_t frame_id, sam::Mask* features_out) {
Impl& s = *impl_;
if (frame_id < 0) frame_id = s.frame;
if (features_out) *features_out = sam::Mask{};
// No decoder passes for a second mask nobody reads.
const std::vector<std::string> no_phrases;
const std::vector<std::string>& feat = features_out ? s.feat : no_phrases;
const nn::Image scaled = downscale_to_fit(image, s.opts.max_size);
const size_t n = (size_t)scaled.width * scaled.height;
const double fx = (double)scaled.width / std::max(1, image.width);
@@ -277,6 +292,7 @@ bool Masker::run(const nn::Image& image, sam::Mask& out, sam::Result* overlay_ou
std::vector<uint8_t> hit(n, 0);
sam::Result all;
sam::Result neg;
sam::Result feat_found;
s.error.clear();
if (s.subject_mode) {
@@ -374,12 +390,14 @@ bool Masker::run(const nn::Image& image, sam::Mask& out, sam::Result* overlay_ou
}
}
// Grounding DINO: one detection pass for every phrase, positive and
// negative alike, then one box-prompted SAM decode per box -- the features
// are already on the device.
if (!s.subject_mode && !s.opts.detector.empty() && (!s.pos.empty() || !s.neg.empty())) {
// Grounding DINO: one detection pass for every phrase, positive, negative
// and feature-only alike, then one box-prompted SAM decode per box -- the
// features are already on the device.
if (!s.subject_mode && !s.opts.detector.empty() &&
(!s.pos.empty() || !s.neg.empty() || !feat.empty())) {
std::vector<std::string> phrases = s.pos;
phrases.insert(phrases.end(), s.neg.begin(), s.neg.end());
phrases.insert(phrases.end(), feat.begin(), feat.end());
std::vector<gdino::Detection> boxes;
try {
gdino::DetectOptions go;
@@ -401,7 +419,11 @@ bool Masker::run(const nn::Image& image, sam::Mask& out, sam::Result* overlay_ou
r.detections.resize(1);
r.detections[0].box = vp.box;
r.detections[0].score = b.score;
Impl::append(b.phrase < (int)s.pos.size() ? all : neg, r);
const size_t k = (size_t)b.phrase;
Impl::append(k < s.pos.size() ? all
: k < s.pos.size() + s.neg.size() ? neg
: feat_found,
r);
}
}
@@ -416,30 +438,56 @@ bool Masker::run(const nn::Image& image, sam::Mask& out, sam::Result* overlay_ou
cp.nms_threshold = s.opts.nms;
Impl::append(neg, s.session.segmentConcept(cp));
}
for (const std::string& phrase : s.opts.detector.empty() ? feat : no_phrases) {
sam::ConceptPrompt cp;
cp.text = phrase;
cp.score_threshold = s.opts.threshold;
cp.nms_threshold = s.opts.nms;
Impl::append(feat_found, s.session.segmentConcept(cp));
}
// Every positive detection reached `all` as it was found, so the union and
// the margin happen once here. The boxes it measures the margin from are
// still in `scaled` pixels; the overlay below is what converts them.
compose_hit(all, neg, s.opts.dilate_ratio, hit);
// Covered -> 0 (255 under `keep`), at the caller's resolution.
auto emit = [&](const std::vector<uint8_t>& covered, bool keep, sam::Mask& m) {
std::vector<uint8_t> mask(n);
{
const uint8_t* src = hit.data();
const uint8_t* src = covered.data();
uint8_t* dst = mask.data();
const bool keep = s.opts.keep_prompted;
nn::parallel_for((int64_t)n, [src, dst, keep](int64_t lo, int64_t hi) {
for (int64_t i = lo; i < hi; ++i)
dst[i] = (uint8_t)(((src[i] != 0) == keep) ? 255 : 0);
}, /*min_chunk=*/65536);
}
out.width = image.width;
out.height = image.height;
m.width = image.width;
m.height = image.height;
if (scaled.width == image.width && scaled.height == image.height)
out.data = std::move(mask);
m.data = std::move(mask);
else
upscale_nearest(mask, scaled.width, scaled.height, out.data, image.width,
upscale_nearest(mask, scaled.width, scaled.height, m.data, image.width,
image.height);
};
// With nothing to keep by name, "keep the named" would keep nothing.
emit(hit, s.opts.keep_prompted && hasTarget(), out);
if (features_out && !feat.empty()) {
// Grounding DINO answers a phrase it cannot place with a box round the
// whole frame (0.32 for "sky" indoors, 0.59 on real sky), which SAM
// fills with everything: as a feature mask that hides the frame.
const float fw = 0.95f * scaled.width, fh = 0.95f * scaled.height;
auto& fd = feat_found.detections;
fd.erase(std::remove_if(fd.begin(), fd.end(),
[&](const sam::Detection& d) {
return d.box.x1 - d.box.x0 >= fw &&
d.box.y1 - d.box.y0 >= fh;
}),
fd.end());
// No margin: it covers the halo round a moving object, and grown into
// the skyline it would take the rooftops a distant scene registers by.
std::vector<uint8_t> covered(n, 0);
compose_hit(feat_found, sam::Result{}, 0.0f, covered);
emit(covered, false, *features_out);
}
if (overlay_out) {
// The overlay is drawn over the caller's image, so per-instance masks
+16 -13
View File
@@ -3,8 +3,9 @@
// positive phrases unioned; negative ones carved back out (a region matching
// one is KEPT); the mask says what to keep, so prompted objects come out black
// unless `keep_prompted`; a longest-side cap on what the model sees. Also the
// visual path (clicks, SeedPrompt) and the two non-SAM models, BiRefNet's
// subject and Grounding DINO's boxes (MaskOptions). src/sam/README.md.
// visual path (clicks, SeedPrompt), the two non-SAM models, BiRefNet's
// subject and Grounding DINO's boxes (MaskOptions), and the second mask that
// keeps the sky out of feature extraction only. src/sam/README.md.
#include "sam/Sam.h"
@@ -49,6 +50,10 @@ struct MaskOptions {
// SS_VK_DEVICE and then Auto in charge; a bad value fails the run.
std::string device;
std::string text, neg_text;
// Phrases kept out of SfM feature extraction but not out of training --
// the sky, whose features ride along with the clouds. Their own mask,
// run()'s `features_out`, from the same backbone pass as `text`.
std::string feature_text;
// Clicks seeding tracked instances. The only way to prompt a SAM 2
// checkpoint, and usable alongside text on a SAM 3 one.
std::vector<SeedPrompt> seeds;
@@ -106,23 +111,21 @@ public:
bool init(const MaskOptions& o, std::string& error);
// Writes a mask the size of `image`: 255 where the pixel should be KEPT.
// `overlay_out`, when given, receives the raw detections for diagnostics.
//
// `frame_id` says which frame of the source this is, and is what
// SeedPrompt::frame is matched against. Callers that hand over every frame
// in order can leave it at -1 and get a plain 0, 1, 2 counter; the video
// extractor passes the decoded index instead, because it writes only the
// sharpest frame of each window and the two do not line up. A seed lands on
// the first frame at or after the one it was drawn on, so it is never lost
// to a frame that was skipped.
// A mask the size of `image`, 255 = KEEP; `features_out` likewise for
// feature_text, left empty without one. `frame_id` is what SeedPrompt::frame
// matches, -1 counting 0, 1, 2 (src/sam/README.md, "Clicked objects").
bool run(const nn::Image& image, sam::Mask& out, sam::Result* overlay_out,
int64_t frame_id = -1);
int64_t frame_id = -1, sam::Mask* features_out = nullptr);
const std::string& lastError() const;
sam::Session& session();
// Whether init() chose BiRefNet: the mask is the subject, no prompt used.
bool subjectMode() const;
// Whether `out` answers anything: false when the only prompt is
// `feature_text`, and run() then writes a mask that keeps every pixel.
bool hasTarget() const;
// Whether run() fills `features_out`.
bool hasFeatureMask() const;
// Returns every model's weights to the device; init() loads them again.
void unload();
+4
View File
@@ -122,6 +122,10 @@ user sees in the preview is what gets written.
- a longest-side cap on what the model sees, with the mask returned at the
source resolution;
- **clicked objects** (`MaskOptions::seeds`), described below;
- **a second mask, for feature extraction only** (`MaskOptions::feature_text`,
"sky; cloud"): matched from the same backbone pass -- the same Grounding
DINO pass, on SAM 2 -- with no margin, and written to `feature_masks/`,
which the reconstruction intersects with `masks/` and training never reads;
- **two models that are not SAM**: `MaskOptions::detector` pairs a SAM
checkpoint with Grounding DINO (`src/gdino/`), which finds every phrase's
boxes in one pass for SAM to cut out -- text on SAM 2, lang-segment-anything
+18 -7
View File
@@ -70,8 +70,8 @@ void Session::unload() {
const bool was_loaded = impl_->loaded;
impl_->loaded = false;
impl_->device_selector.clear();
impl_->text_cache_valid = false;
impl_->text_cache_ids.clear();
impl_->text_cache_next = 0;
// Nothing to give back, or the device is already gone (nn::shutdown() ran
// first, which frees all of this wholesale).
if (!was_loaded || !vk::Context::initialized()) return;
@@ -151,7 +151,8 @@ bool Session::loadModel(const ModelParams& params) {
try {
impl_->model.load(params.model_path, params.img_size);
impl_->arena.reserve(arena_reserve_for(impl_->model));
impl_->text_cache_valid = false;
impl_->text_cache_ids.clear();
impl_->text_cache_next = 0;
impl_->loaded = true;
impl_->loaded_params = params;
// What the weights actually landed on -- the live context is the only
@@ -248,17 +249,27 @@ int64_t Session::Impl::buildPrompt(const ConceptPrompt& prompt, nn::Tensor& out_
{
const nn::Tensor rows = out_prompt.slice0(0, L);
const uint64_t bytes = (uint64_t)L * D * sizeof(float);
if (text_cache_valid && text_cache_ids == ids) {
const auto hit = std::find(text_cache_ids.begin(), text_cache_ids.end(), ids);
if (hit != text_cache_ids.end()) {
const uint32_t sub = (uint32_t)(hit - text_cache_ids.begin());
vk::Stream::get().copy(rows.ptr,
vk::VramPool::get().lookup(vk::PoolSlot::TextFeat, 0),
vk::VramPool::get().lookup(vk::PoolSlot::TextFeat, sub),
bytes);
} else {
model::encode_text(model, arena, ids, rows);
// 32 KiB an entry; a prompt holding more phrases than this is not
// one anybody types.
constexpr size_t kMaxCached = 32;
size_t sub = text_cache_ids.size();
if (sub < kMaxCached) {
text_cache_ids.push_back(ids);
} else {
sub = text_cache_next++ % kMaxCached;
text_cache_ids[sub] = ids;
}
nn::Tensor cached =
nn::pool_tensor(vk::PoolSlot::TextFeat, 0, nn::DType::F32, L, D);
nn::pool_tensor(vk::PoolSlot::TextFeat, (uint32_t)sub, nn::DType::F32, L, D);
vk::Stream::get().copy(cached.ptr, rows.ptr, bytes);
text_cache_ids = ids;
text_cache_valid = true;
}
}
+5 -6
View File
@@ -41,12 +41,11 @@ struct Session::Impl {
// Point/box prompt tokens for the SAM decoder. Returns the token count.
int64_t buildSparsePrompt(const VisualPrompt& prompt, nn::Tensor& out_sparse);
// The text encoder is 24 transformer blocks over a fixed 32-token context
// and depends on nothing but the prompt, yet the tracker calls buildPrompt
// once per frame with the same words every time. The encoded rows live in
// PoolSlot::TextFeat and are reused until the token ids change.
std::vector<int32_t> text_cache_ids;
bool text_cache_valid = false;
// The text encoder (24 blocks over 32 tokens) reads nothing but the prompt,
// yet every frame asks again for each phrase of a masking run: entry i's
// rows live in PoolSlot::TextFeat sub i, encoded once per model load.
std::vector<std::vector<int32_t>> text_cache_ids;
size_t text_cache_next = 0; // round-robin once full
Result runConcept(const ConceptPrompt& prompt);
+40
View File
@@ -827,6 +827,38 @@ int main() {
check(tracker.refineInstance(id, {{70.0f, 55.0f}}, {}), "refineInstance");
}
// Random weights make whether "sky" matches anything noise, so what is
// checked is the shape of the answer, not its content.
std::printf("\nFeature-only phrases (sam::Masker)\n");
{
sam::MaskOptions mo;
mo.model = path;
mo.video = false;
mo.max_size = 0;
mo.threshold = 0.0f;
mo.keep_prompted = true; // with nothing named, must still keep all
mo.feature_text = "sky; cloud";
sam::Masker masker;
std::string err;
const bool ok = masker.init(mo, err);
check(ok, "a feature prompt alone is something to segment");
if (ok) {
check(!masker.hasTarget() && masker.hasFeatureMask(),
"... with nothing named for training");
sam::Mask m, f;
check(masker.run(make_image(160, 120), m, nullptr, -1, &f),
"run with a feature mask");
const bool all_kept =
std::all_of(m.data.begin(), m.data.end(), [](uint8_t v) { return v == 255; });
check(m.width == 160 && m.height == 120 && all_kept,
"nothing named for training keeps every pixel");
check(f.width == 160 && f.height == 120 && f.data.size() == 160u * 120u,
"the feature mask comes back at the source resolution");
check(masker.run(make_image(160, 120), m, nullptr, -1, nullptr),
"and without one asked for");
}
}
std::printf("\nVRAM\n%s", session.vramReport().c_str());
// ---- the same drill against a SAM 2 checkpoint --------------------------
@@ -933,6 +965,14 @@ int main() {
}
}
}
{
sam::MaskOptions mo;
mo.model = p2;
mo.feature_text = "sky";
sam::Masker masker;
std::string err;
check(!masker.init(mo, err), "a feature prompt needs a text model");
}
std::remove(p2.c_str());
}
+47 -12
View File
@@ -1270,11 +1270,13 @@ void sweepStaleFeatures(const fs::path& outdir, const std::set<fs::path>& live)
// mtime comparison, because a re-run that regenerated the frames or the masks
// leaves everything else about the settings identical.
bool featuresAreCurrent(const fs::path& feat, const fs::path& img,
const std::string& mask, uint32_t& count) {
const std::string& mask, const std::string& feature_mask,
uint32_t& count) {
std::error_code fe, ie, me;
const auto t = fs::last_write_time(feat, fe);
if (fe || t < fs::last_write_time(img, ie) || ie) return false;
if (!mask.empty() && t < fs::last_write_time(mask, me) && !me) return false;
for (const std::string* m : {&mask, &feature_mask})
if (!m->empty() && t < fs::last_write_time(*m, me) && !me) return false;
return peekFeatures(feat.string(), count);
}
@@ -1284,16 +1286,22 @@ int extractDirectory(const std::string& imagedir, const fs::path& outdir,
const SfmConfig& cfg, ExtractStats& stats, bool reuse) {
const SiftOptions& opt = cfg.sift;
const std::string& maskdir = cfg.mask_dir;
const std::string& fmaskdir = cfg.feature_mask_dir;
// Recursive: per-folder intrinsics (ppisp) keep images in images/<camera>/
// and the folder is the grouping key (D17). A mask directory nested inside
// is skipped -- masks are PNGs too, and would double the image count with
// garbage views.
std::error_code skip_ec;
const bool skip_masks = !maskdir.empty() && fs::is_directory(maskdir, skip_ec);
std::vector<std::string> skip;
for (const std::string* d : {&maskdir, &fmaskdir})
if (!d->empty() && fs::is_directory(*d, skip_ec)) skip.push_back(*d);
std::vector<fs::path> found;
for (auto it = fs::recursive_directory_iterator(imagedir, fs::directory_options::follow_directory_symlink);
it != fs::recursive_directory_iterator(); ++it) {
if (skip_masks && it->is_directory() && fs::equivalent(it->path(), maskdir, skip_ec)) {
if (it->is_directory() &&
std::any_of(skip.begin(), skip.end(), [&](const std::string& d) {
return fs::equivalent(it->path(), d, skip_ec);
})) {
it.disable_recursion_pending();
continue;
}
@@ -1382,6 +1390,24 @@ int extractDirectory(const std::string& imagedir, const fs::path& outdir,
L::warn(Tag::Extract, M::extract_some_unmasked,
{(long long)stats.unmasked_images, stats.first_unmasked});
}
// An image without one keeps all its features, as under --masks; a tree
// matching nothing is not fatal here, since the sky is often absent.
MaskIndex fmasks(fmaskdir);
if (!fmaskdir.empty() && !fmasks.valid())
L::warn(Tag::Extract, M::extract_mask_dir_missing, {fmaskdir});
if (fmasks.valid()) {
lopt.feature_mask_paths.assign(paths.size(), std::string());
size_t matched = 0;
for (size_t k = 0; k < paths.size(); k++) {
std::string& fp = lopt.feature_mask_paths[k];
fp = fmasks.find(relativeTo(paths[k], imagedir).generic_string());
if (fp.empty()) continue;
matched++;
if (lopt.mask_paths.empty() || lopt.mask_paths[k].empty()) stats.masked_images++;
}
L::out(Tag::Extract, M::extract_masks_matched,
{(long long)matched, (long long)paths.size(), fmaskdir});
}
// Where each image's features belong, and what a previous run already put
// there. The total the bar counts is the capture, not the work left.
@@ -1402,7 +1428,10 @@ int extractDirectory(const std::string& imagedir, const fs::path& outdir,
uint32_t count = 0;
const std::string mask =
k < lopt.mask_paths.size() ? lopt.mask_paths[k] : std::string();
if (!featuresAreCurrent(outs[k], paths[k], mask, count)) {
const std::string fmask = k < lopt.feature_mask_paths.size()
? lopt.feature_mask_paths[k]
: std::string();
if (!featuresAreCurrent(outs[k], paths[k], mask, fmask, count)) {
todo.push_back(k);
continue;
}
@@ -1430,11 +1459,14 @@ int extractDirectory(const std::string& imagedir, const fs::path& outdir,
sorted_dims[i] = sorted_dims[k];
outs[i] = std::move(outs[k]);
if (!lopt.mask_paths.empty()) lopt.mask_paths[i] = std::move(lopt.mask_paths[k]);
if (!lopt.feature_mask_paths.empty())
lopt.feature_mask_paths[i] = std::move(lopt.feature_mask_paths[k]);
}
paths.resize(todo.size());
sorted_dims.resize(todo.size());
outs.resize(todo.size());
if (!lopt.mask_paths.empty()) lopt.mask_paths.resize(todo.size());
if (!lopt.feature_mask_paths.empty()) lopt.feature_mask_paths.resize(todo.size());
}
}
if (paths.empty()) {
@@ -1455,6 +1487,7 @@ int extractDirectory(const std::string& imagedir, const fs::path& outdir,
if (opt.verbose) L::err(Tag::Extract, M::extract_frontend, {ext->name()});
// Everything after extract() reads only the image and its features, so it
// runs on its own thread while the device works on the next image.
static const std::string kNoPath;
auto postProcess = [&](size_t k, GrayImage& img, FeatureSet& f) {
if (img.exif_mirror_dropped && !stats.warned_exif_mirror) {
stats.warned_exif_mirror = true;
@@ -1463,16 +1496,18 @@ int extractDirectory(const std::string& imagedir, const fs::path& outdir,
}
sampleFeatureColors(f, img);
uint32_t dropped = 0;
if (!lopt.mask_paths.empty() && !lopt.mask_paths[k].empty()) {
const std::string& mpath =
k < lopt.mask_paths.size() && !lopt.mask_paths[k].empty() ? lopt.mask_paths[k]
: k < lopt.feature_mask_paths.size() ? lopt.feature_mask_paths[k]
: kNoPath;
if (!mpath.empty()) {
if (img.mask.empty()) {
stats.mask_unreadable++;
L::warn(Tag::Extract, M::extract_mask_undecodable,
{lopt.mask_paths[k],
fs::path(paths[k]).filename().string()});
{mpath, fs::path(paths[k]).filename().string()});
} else {
// img's own size, not the probed one: `apply` turned both.
checkMaskShape(lopt.mask_paths[k], img.mask,
{img.orig_width, img.orig_height});
checkMaskShape(mpath, img.mask, {img.orig_width, img.orig_height});
const uint32_t before = f.count();
dropped = applyMask(f, img.mask);
stats.masked_out += dropped;
@@ -1482,8 +1517,7 @@ int extractDirectory(const std::string& imagedir, const fs::path& outdir,
if (before && dropped == before && !stats.warned_empty) {
stats.warned_empty = true;
L::warn(Tag::Extract, M::extract_mask_empty,
{lopt.mask_paths[k],
fs::path(paths[k]).filename().string()});
{mpath, fs::path(paths[k]).filename().string()});
}
}
}
@@ -2106,6 +2140,7 @@ AutoResult run_auto(SfmConfig& cfg, const AutoInputs& in) {
L::out(Tag::Run, M::run_data_type, {cfg.data_type});
L::out(Tag::Run, M::run_cameras, {cfg.camera_model, cfg.camera_mode});
if (!cfg.mask_dir.empty()) L::out(Tag::Run, M::run_masks, {cfg.mask_dir});
if (!cfg.feature_mask_dir.empty()) L::out(Tag::Run, M::run_masks, {cfg.feature_mask_dir});
// What the two knobs moved, so a surprising run is explainable from its own
// output rather than from reading the preset table.
for (const PresetChange& p : in.preset_changes)
+1
View File
@@ -279,6 +279,7 @@ spirula sfm auto IMAGES/ -o WORKSPACE/ # images -> sparse model
spirula sfm auto -o ws/ # ./images + ./masks, all defaults
spirula sfm auto IMAGES/ -o ws/ --data-type video --quality medium
spirula sfm auto IMAGES/ -o ws/ --masks MASKS/ # drop keypoints on masked pixels
spirula sfm auto IMAGES/ -o ws/ --masks MASKS/ --feature-masks SKY/ # ... on either
spirula sfm auto IMAGES/ -o ws/ --camera-model opencv-fisheye
spirula sfm extract IMAGES/ -o feats/
+2 -2
View File
@@ -174,8 +174,8 @@ std::string metavarFor(const std::string&, const char* name, const char* choices
// No metavar column in the table: the flag name says what it takes, and
// "DIR" reads better than "VALUE" on the handful that take a path.
std::string n = name;
if (n.find("dir") != std::string::npos || n == "masks" || n == "images" ||
n == "features" || n == "resume")
if (n.find("dir") != std::string::npos || n == "masks" || n == "feature-masks" ||
n == "images" || n == "features" || n == "resume")
return "DIR";
if (n.find("path") != std::string::npos) return "FILE";
return "VALUE";
+5
View File
@@ -119,6 +119,9 @@ struct SfmConfig {
// Swap keep and ignore in every mask, for the exporters that paint the
// region to REMOVE (sfm/core/Mask.h).
bool flip_mask = false;
// A second mask tree, intersected with mask_dir's: what extraction skips
// but training keeps, the sky. Never flipped, never guessed from a sibling.
std::string feature_mask_dir;
// The input files' colour space. Pixels convert to sRGB on decode, which
// is what the detectors and the AI models were trained on.
@@ -338,6 +341,8 @@ struct SfmConfig {
F(mask_dir, "mask-dir", CMD_AUTO | CMD_EXTRACT, Tier::Alias, "pipeline", 0, 0, "", mask_dir) \
F(flip_mask, "flip-mask", CMD_AUTO | CMD_EXTRACT, Tier::Advanced, "pipeline", 0, 0, "", \
flip_mask) \
F(feature_mask_dir, "feature-masks", CMD_AUTO | CMD_EXTRACT, Tier::Advanced, "pipeline", 0, 0, \
"", feature_masks) \
/* ---- colour ---- */ \
F(image_gamut, "image-gamut", CMD_AUTO | CMD_EXTRACT, Tier::Advanced, "colour", 0, 0, \
"Rec.709|ACES2065-1|ACEScg|Rec.2020|AdobeRGB|DCI-P3", image_gamut) \
+6 -1
View File
@@ -140,7 +140,8 @@ void applyExifOrientation(GrayImage& img) {
GrayImage loadGrayImage(const std::string& path, int max_image_size, bool want_color,
const std::string& mask_path,
const std::string& gamut, std::optional<bool> is_linear,
bool flip_mask, bool apply_exif_orientation) {
bool flip_mask, bool apply_exif_orientation,
const std::string& feature_mask_path) {
int w = 0, h = 0, chan = 0;
// Force 3 channels; we do our own luma so behavior is decoder-independent.
// An EXR decodes on this thread: the pool above already owns every core.
@@ -199,6 +200,10 @@ GrayImage loadGrayImage(const std::string& path, int max_image_size, bool want_c
img.mask = loadMask(mask_path);
if (flip_mask) img.mask.invert();
}
// Not over a first mask that failed to decode: the caller reports that by
// finding img.mask empty.
if (!feature_mask_path.empty() && (mask_path.empty() || !img.mask.empty()))
intersectMask(img.mask, loadMask(feature_mask_path));
img.exif = readExif(path); // header bytes only; see sfm/core/Exif.h
if (apply_exif_orientation) applyExifOrientation(img);
return img;
+3 -1
View File
@@ -89,7 +89,9 @@ GrayImage loadGrayImage(const std::string& path, int max_image_size = 3200,
bool flip_mask = false,
// Turns the pixels and the mask, leaving exif.orientation
// at 1; a tag's MIRROR half is dropped (docs/datasets.md).
bool apply_exif_orientation = false);
bool apply_exif_orientation = false,
// Intersected into `mask_path`'s (intersectMask), unflipped.
const std::string& feature_mask_path = "");
// Read just the pixel dimensions from an image header (no full decode).
// Returns false if the file is not a decodable image.
+9 -4
View File
@@ -65,6 +65,8 @@ struct ImageLoadOptions {
// order of magnitude -- doing them serially on the consumer thread would
// hand the GPU stage back the decode cost the pool exists to hide.
std::vector<std::string> mask_paths;
// --feature-masks, laid out as mask_paths and intersected into them.
std::vector<std::string> feature_mask_paths;
// Swap keep and ignore in every decoded mask (sfm/core/Mask.h). On the pool
// rather than the consumer thread, which the mask decode is already on.
bool flip_mask = false;
@@ -138,11 +140,12 @@ inline ImageLoadPlan planImageLoad(const std::vector<std::pair<int, int>>& dims,
// or below the image resolution in every convention we have seen, and
// 1 B/px against the image's 3 B/px leaves the budget dominated by the
// image either way.
const size_t mask_bytes = opt.mask_paths.empty() ? 0 : maxOutPix * 2;
const size_t masks = (opt.mask_paths.empty() ? 0 : 1) +
(opt.feature_mask_paths.empty() ? 0 : 1);
plan.decode_peak_bytes =
maxPix * 3 + maxOutPix * (opt.want_color ? 7 : 4) + mask_bytes;
maxPix * 3 + maxOutPix * (opt.want_color ? 7 : 4) + maxOutPix * 2 * masks;
plan.held_bytes = maxOutPix * (opt.want_color ? 7 : 4) // gray float (+ RGB u8)
+ (opt.mask_paths.empty() ? 0 : maxOutPix);
+ (masks ? maxOutPix : 0);
unsigned hc = std::thread::hardware_concurrency();
int want = opt.num_threads > 0 ? opt.num_threads : (hc > 0 ? (int)hc : 1);
@@ -177,10 +180,12 @@ inline void loadImagesInOrder(const std::vector<std::string>& paths, const Image
auto decodeOne = [&](size_t i, GrayImage& out, std::string& err) {
static const std::string kNoMask;
const std::string& mp = i < opt.mask_paths.size() ? opt.mask_paths[i] : kNoMask;
const std::string& fmp =
i < opt.feature_mask_paths.size() ? opt.feature_mask_paths[i] : kNoMask;
try {
out = loadGrayImage(paths[i], opt.max_image_size, opt.want_color, mp,
opt.gamut, opt.is_linear, opt.flip_mask,
opt.apply_exif_orientation);
opt.apply_exif_orientation, fmp);
} catch (const std::exception& e) {
err = e.what();
}
+20
View File
@@ -123,6 +123,26 @@ inline uint32_t applyMask(FeatureSet& fs, const Mask& m) {
return n - out;
}
// Keep only what both keep (--masks and --feature-masks), on the finer of the
// two grids with the other sampled in uv, so the two need not share a
// resolution. An empty mask keeps everything.
inline void intersectMask(Mask& a, const Mask& b) {
if (b.empty()) return;
if (a.empty()) {
a = b;
return;
}
const bool b_finer = b.pixels() > a.pixels();
Mask fine = b_finer ? b : a;
const Mask& coarse = b_finer ? a : b;
for (int y = 0; y < fine.height; y++)
for (int x = 0; x < fine.width; x++) {
uint8_t& v = fine.bits[(size_t)y * fine.width + x];
if (v) v = coarse.atUV((x + 0.5f) / fine.width, (y + 0.5f) / fine.height);
}
a = std::move(fine);
}
// ---- finding the mask that belongs to an image --------------------------
namespace detail {
+57
View File
@@ -12,6 +12,7 @@
#include <vector>
#include "sfm/core/Features.h"
#include "sfm/core/Image.h"
#include "sfm/core/Mask.h"
#include "sfm/tests/TestMain.h"
@@ -182,6 +183,62 @@ int cmdMaskSelftest(int, char**) {
}
}
// ---- 2b. --masks and --feature-masks intersect, at the finer grid ----
{
auto grid = [](int w, int h, bool left_half) {
Mask m;
m.width = w;
m.height = h;
m.bits.resize((size_t)w * h);
for (int y = 0; y < h; y++)
for (int x = 0; x < w; x++)
m.bits[(size_t)y * w + x] = left_half ? x < w / 2 : y < h / 2;
return m;
};
Mask a = grid(40, 30, true);
intersectMask(a, grid(120, 90, false));
const double keep = a.keepFraction();
printf("mask: 40x30 left half & 120x90 top half -> %dx%d, keep %.3f (expect 0.250)\n",
a.width, a.height, keep);
if (a.width != 120 || a.height != 90 || std::fabs(keep - 0.25) > 1e-6 ||
!a.atUV(0.1f, 0.1f) || a.atUV(0.9f, 0.1f) || a.atUV(0.1f, 0.9f)) {
printf(" FAIL: intersection of masks at two resolutions\n");
fails++;
}
Mask none;
intersectMask(none, grid(8, 8, true));
Mask kept = grid(8, 8, true);
intersectMask(kept, Mask{});
if (none.keepFraction() != 0.5 || kept.keepFraction() != 0.5) {
printf(" FAIL: an empty mask must leave the other as it is\n");
fails++;
}
// Through the loader: both files apply, and a first mask that will not
// decode is reported as such rather than hidden by the second.
const fs::path d = tmp / "both";
writePgm(d / "img.pgm", 64, 48, std::vector<uint8_t>((size_t)64 * 48, 128));
writePgm(d / "a.pgm", 64, 48, leftHalfMask(64, 48, 0.5));
std::vector<uint8_t> top((size_t)64 * 48, 0);
std::fill(top.begin(), top.begin() + (size_t)64 * 24, (uint8_t)255);
writePgm(d / "b.pgm", 64, 48, top);
std::ofstream(d / "junk.png", std::ios::binary) << "not an image";
const std::string img = (d / "img.pgm").string();
GrayImage g = loadGrayImage(img, 0, false, (d / "a.pgm").string(), "", std::nullopt,
false, false, (d / "b.pgm").string());
GrayImage only = loadGrayImage(img, 0, false, "", "", std::nullopt, false, false,
(d / "b.pgm").string());
GrayImage broken = loadGrayImage(img, 0, false, (d / "junk.png").string(), "",
std::nullopt, false, false, (d / "b.pgm").string());
if (std::fabs(g.mask.keepFraction() - 0.25) > 1e-6 ||
std::fabs(only.mask.keepFraction() - 0.5) > 1e-6 || !broken.mask.empty()) {
printf(" FAIL: loader with a feature mask kept %.3f / %.3f, broken %s\n",
g.mask.keepFraction(), only.mask.keepFraction(),
broken.mask.empty() ? "empty" : "not empty");
fails++;
}
}
// ---- 3. discovery: the naming conventions in the wild ----
{
const fs::path md = tmp / "masks";