Search in sources :

Example 6 with StringDoublePair

use of org.vitrivr.cineast.core.data.StringDoublePair in project cineast by vitrivr.

the class QueryUtil method retrieve.

public static List<StringDoublePair> retrieve(ContinuousRetrievalLogic continuousRetrievalLogic, AbstractQueryTermContainer queryTermContainer, ReadableQueryConfig config, String category) {
    float weight = MathHelper.limit(queryTermContainer.getWeight(), -1f, 1f);
    TObjectDoubleHashMap<String> scoreBySegmentId = new TObjectDoubleHashMap<>();
    retrieveAndWeight(continuousRetrievalLogic, category, scoreBySegmentId, queryTermContainer, config, weight);
    final List<StringDoublePair> list = new ArrayList<>(scoreBySegmentId.size());
    scoreBySegmentId.forEachEntry((segmentId, score) -> {
        if (score > 0) {
            list.add(new StringDoublePair(segmentId, score));
        }
        return true;
    });
    return list;
}
Also used : StringDoublePair(org.vitrivr.cineast.core.data.StringDoublePair) TObjectDoubleHashMap(gnu.trove.map.hash.TObjectDoubleHashMap) ArrayList(java.util.ArrayList)

Example 7 with StringDoublePair

use of org.vitrivr.cineast.core.data.StringDoublePair in project cineast by vitrivr.

the class AbstractQueryMessageHandler method finalizeAndSubmitResults.

/**
 * Fetches and submits all the data (e.g. {@link MediaObjectDescriptor}, {@link MediaSegmentDescriptor}) associated with the raw results produced by a similarity search in a specific category. q
 *
 * @param session  The {@link Session} object used to transmit the results.
 * @param queryId  ID of the running query.
 * @param category Name of the query category.
 * @param raw      List of raw per-category results (segmentId -> score).
 */
protected List<CompletableFuture<Void>> finalizeAndSubmitResults(Session session, String queryId, String category, int containerId, List<StringDoublePair> raw) {
    StopWatch watch = StopWatch.createStarted();
    final int stride = 50_000;
    List<CompletableFuture<Void>> futures = new ArrayList<>();
    for (int i = 0; i < Math.floorDiv(raw.size(), stride) + 1; i++) {
        final List<StringDoublePair> sub = raw.subList(i * stride, Math.min((i + 1) * stride, raw.size()));
        futures.add(this.write(session, new SimilarityQueryResult(queryId, category, containerId, sub)));
    }
    watch.stop();
    LOGGER.trace("Finalizing & submitting results took {} ms", watch.getTime(TimeUnit.MILLISECONDS));
    return futures;
}
Also used : StringDoublePair(org.vitrivr.cineast.core.data.StringDoublePair) SimilarityQueryResult(org.vitrivr.cineast.api.messages.result.SimilarityQueryResult) CompletableFuture(java.util.concurrent.CompletableFuture) ArrayList(java.util.ArrayList) StopWatch(org.apache.commons.lang3.time.StopWatch)

Example 8 with StringDoublePair

use of org.vitrivr.cineast.core.data.StringDoublePair in project cineast by vitrivr.

the class TemporalTestCases method buildTestCase1.

public void buildTestCase1() {
    Map<String, MediaSegmentDescriptor> segmentMap = new HashMap<>();
    List<List<StringDoublePair>> containerResults = new ArrayList<>();
    segmentMap.put(descriptor1_1.getSegmentId(), descriptor1_1);
    segmentMap.put(descriptor1_2.getSegmentId(), descriptor1_2);
    segmentMap.put(descriptor2_1.getSegmentId(), descriptor2_1);
    segmentMap.put(descriptor2_2.getSegmentId(), descriptor2_2);
    List<StringDoublePair> containerList0 = new ArrayList<>();
    containerList0.add(new StringDoublePair(descriptor1_1.getSegmentId(), 1d));
    containerList0.add(new StringDoublePair(descriptor2_1.getSegmentId(), 0.5d));
    List<StringDoublePair> containerList1 = new ArrayList<>();
    containerList1.add(new StringDoublePair(descriptor1_2.getSegmentId(), 1d));
    containerList1.add(new StringDoublePair(descriptor2_2.getSegmentId(), 0.5d));
    containerResults.add(0, containerList0);
    containerResults.add(1, containerList1);
    List<TemporalObject> expectedResults = new ArrayList<>();
    List<String> segments1 = new ArrayList<>();
    segments1.add(descriptor1_1.getSegmentId());
    segments1.add(descriptor1_2.getSegmentId());
    TemporalObject t1 = new TemporalObject(segments1, descriptor1_1.getObjectId(), 1f);
    expectedResults.add(t1);
    List<String> segments2 = new ArrayList<>();
    segments2.add(descriptor2_1.getSegmentId());
    segments2.add(descriptor2_2.getSegmentId());
    TemporalObject t2 = new TemporalObject(segments2, descriptor2_1.getObjectId(), 0.5f);
    expectedResults.add(t2);
    this.segmentMap = segmentMap;
    this.containerResults = containerResults;
    this.maxLength = 20f;
    this.expectedResults = expectedResults;
    List<Float> timeDistances = new ArrayList<>();
    timeDistances.add(0f);
    this.timeDistances = timeDistances;
}
Also used : HashMap(java.util.HashMap) ArrayList(java.util.ArrayList) StringDoublePair(org.vitrivr.cineast.core.data.StringDoublePair) TemporalObject(org.vitrivr.cineast.core.data.TemporalObject) MediaSegmentDescriptor(org.vitrivr.cineast.core.data.entities.MediaSegmentDescriptor) List(java.util.List) ArrayList(java.util.ArrayList)

Example 9 with StringDoublePair

use of org.vitrivr.cineast.core.data.StringDoublePair in project cineast by vitrivr.

the class TemporalQueryMessageHandler method execute.

@Override
public void execute(Session session, QueryConfig qconf, TemporalQuery message, Set<String> segmentIdsForWhichMetadataIsFetched, Set<String> objectIdsForWhichMetadataIsFetched) throws Exception {
    /* Prepare the query config and get the QueryId */
    final String uuid = qconf.getQueryId().toString();
    String qid = uuid.substring(0, 3);
    final int max = Math.min(qconf.getMaxResults().orElse(Config.sharedConfig().getRetriever().getMaxResults()), Config.sharedConfig().getRetriever().getMaxResults());
    qconf.setMaxResults(max);
    final int resultsPerModule = Math.min(qconf.getRawResultsPerModule() == -1 ? Config.sharedConfig().getRetriever().getMaxResultsPerModule() : qconf.getResultsPerModule(), Config.sharedConfig().getRetriever().getMaxResultsPerModule());
    qconf.setResultsPerModule(resultsPerModule);
    List<Thread> metadataRetrievalThreads = new ArrayList<>();
    List<CompletableFuture<Void>> futures = new ArrayList<>();
    List<Thread> cleanupThreads = new ArrayList<>();
    /* We need a set of segments and objects to be used for temporal scoring as well as a storage of all container results where are the index of the outer list is where container i was scored */
    Map<Integer, List<StringDoublePair>> containerResults = new IntObjectHashMap<>();
    Set<MediaSegmentDescriptor> segments = new HashSet<>();
    Set<String> sentSegmentIds = new HashSet<>();
    Set<String> sentObjectIds = new HashSet<>();
    /* Each container can be evaluated in parallel, provided resouces are available */
    List<Thread> ssqThreads = new ArrayList<>();
    /* Iterate over all temporal query containers independently */
    for (int containerIdx = 0; containerIdx < message.getQueries().size(); containerIdx++) {
        StagedSimilarityQuery stagedSimilarityQuery = message.getQueries().get(containerIdx);
        /* Make a new Query config for this container because the relevant segments from the previous stage will differ within this container from stage to stage.  */
        QueryConfig stageQConf = QueryConfig.clone(qconf);
        QueryConfig limitedStageQConf = QueryConfig.clone(qconf);
        /* The first stage of a container will have no relevant segments from a previous stage. The retrieval engine will handle this case. */
        HashSet<String> relevantSegments = new HashSet<>();
        HashSet<String> limitedRelevantSegments = new HashSet<>();
        /*
       * Store for each query term per category all results to be sent at a later time
       */
        List<Map<String, List<StringDoublePair>>> cache = new ArrayList<>();
        /* For the temporal scoring, we need to store the relevant results of the stage to be saved to the containerResults */
        List<StringDoublePair> stageResults = new ArrayList<>();
        int lambdaFinalContainerIdx = containerIdx;
        /*
       * The lightweight, but blocking logic of waiting for retrieval results is launched as a thread.
       * The results of this thread will be awaited after all containers have started their retrieval process
       */
        Thread ssqThread = new Thread(() -> {
            /* Iterate over all stages in their respective order as each term of one stage will be used as a filter for its successors */
            for (int stageIndex = 0; stageIndex < stagedSimilarityQuery.getStages().size(); stageIndex++) {
                /* Create hashmap for this stage as cache */
                cache.add(stageIndex, new HashMap<>());
                QueryStage stage = stagedSimilarityQuery.getStages().get(stageIndex);
                /*
           * Iterate over all QueryTerms for this stage and add their results to the list of relevant segments for the next query stage.
           * Only update the list of relevant query terms once we iterated over all terms
           */
                for (int i = 0; i < stage.terms.size(); i++) {
                    QueryTerm qt = stage.terms.get(i);
                    /* Prepare the QueryTerm and perform sanity checks */
                    if (qt == null) {
                        /* There are edge cases in which we have a null as a query stage. If this happens please report this to the developers  */
                        LOGGER.warn("QueryTerm was null for stage {}", stage);
                        return;
                    }
                    AbstractQueryTermContainer qc = qt.toContainer();
                    if (qc == null) {
                        LOGGER.warn("Likely an empty query, as it could not be converted to a query container. Ignoring it");
                        return;
                    }
                    /* We retrieve the results for each category of a QueryTerm independently. The relevant ids will not yet be changed after this call as we are still in the same stage. */
                    for (String category : qt.getCategories()) {
                        List<SegmentScoreElement> scores = continuousRetrievalLogic.retrieve(qc, category, stageQConf);
                        final List<StringDoublePair> results = scores.stream().map(elem -> new StringDoublePair(elem.getSegmentId(), elem.getScore())).filter(p -> p.value > 0d).sorted(StringDoublePair.COMPARATOR).collect(Collectors.toList());
                        if (results.isEmpty()) {
                            LOGGER.warn("No results found for category {} and qt {} in stage with id {}. Full component: {}", category, qt.getType(), lambdaFinalContainerIdx, stage);
                        }
                        if (cache.get(stageIndex).containsKey(category)) {
                            LOGGER.error("Category {} was used twice in stage {}. This erases the results of the previous category... ", category, stageIndex);
                        }
                        cache.get(stageIndex).put(category, results);
                        results.forEach(res -> relevantSegments.add(res.key));
                        /*
               * If this is the last stage, we can collect the results and send relevant results per category back the requester.
               * Otherwise we shouldn't yet send since we might send results to the requester that would be filtered at a later stage.
               */
                        if (stageIndex == stagedSimilarityQuery.getStages().size() - 1) {
                            /* We limit the results to be sent back to the requester to the max limit. This is so that the original view is not affected by the changes of temporal query version 2 */
                            List<StringDoublePair> limitedResults = results.stream().limit(max).collect(Collectors.toList());
                            results.forEach(res -> limitedRelevantSegments.add(res.key));
                            List<String> limitedSegmentIds = limitedResults.stream().map(el -> el.key).collect(Collectors.toList());
                            sentSegmentIds.addAll(limitedSegmentIds);
                            List<MediaSegmentDescriptor> limitedSegmentDescriptors = this.loadSegments(limitedSegmentIds, qid);
                            /* Store the segments and results for this staged query to be used in the temporal querying. */
                            segments.addAll(limitedSegmentDescriptors);
                            stageResults.addAll(results);
                            List<String> limitedObjectIds = this.submitPrefetchedSegmentAndObjectInformation(session, uuid, limitedSegmentDescriptors);
                            sentObjectIds.addAll(limitedObjectIds);
                            LOGGER.trace("Queueing finalization and result submission for last stage, container {}", lambdaFinalContainerIdx);
                            futures.addAll(this.finalizeAndSubmitResults(session, uuid, category, lambdaFinalContainerIdx, limitedResults));
                            List<Thread> _threads = this.submitMetadata(session, uuid, limitedSegmentIds, limitedObjectIds, segmentIdsForWhichMetadataIsFetched, objectIdsForWhichMetadataIsFetched, message.getMetadataAccessSpec());
                            metadataRetrievalThreads.addAll(_threads);
                        }
                    }
                }
                /* After having finished a stage, we add all relevant segments to the config of the next stage. */
                if (relevantSegments.size() == 0) {
                    LOGGER.warn("No relevant segments anymore, aborting staged querying");
                    /* Clear the relevant segments are there are none */
                    stageQConf.setRelevantSegmentIds(relevantSegments);
                    break;
                }
                stageQConf.setRelevantSegmentIds(relevantSegments);
                relevantSegments.clear();
            }
            limitedStageQConf.setRelevantSegmentIds(limitedRelevantSegments);
            /* At this point, we have iterated over all stages. Now, we need to go back for all stages and send the results for the relevant ids. */
            for (int stageIndex = 0; stageIndex < stagedSimilarityQuery.getStages().size() - 1; stageIndex++) {
                int finalStageIndex = stageIndex;
                /* Add the results from the last filter from all previous stages also to the list of results */
                cache.get(stageIndex).forEach((category, results) -> {
                    results.removeIf(pair -> !stageQConf.getRelevantSegmentIds().contains(pair.key));
                    stageResults.addAll(results);
                });
                /* Return the limited results from all stages that are within the filter */
                cache.get(stageIndex).forEach((category, results) -> {
                    results.removeIf(pair -> !limitedStageQConf.getRelevantSegmentIds().contains(pair.key));
                    Thread thread = new Thread(() -> {
                        LOGGER.trace("Queuing finalization & result submission for stage {} and container {}", finalStageIndex, lambdaFinalContainerIdx);
                        futures.addAll(this.finalizeAndSubmitResults(session, uuid, category, lambdaFinalContainerIdx, results));
                    });
                    thread.setName("finalization-stage" + finalStageIndex + "-" + category);
                    thread.start();
                    cleanupThreads.add(thread);
                });
            }
            /* There should be no carry-over from this block since temporal queries are executed independently */
            containerResults.put(lambdaFinalContainerIdx, stageResults);
        });
        ssqThread.setName("ssq-" + containerIdx);
        ssqThreads.add(ssqThread);
        ssqThread.start();
    }
    for (Thread ssqThread : ssqThreads) {
        ssqThread.join();
    }
    /* You can skip the computation of temporal objects in the config if you wish simply to execute all queries independently (e.g. for evaluation)*/
    if (!message.getTemporalQueryConfig().computeTemporalObjects) {
        LOGGER.debug("Not computing temporal objects due to query config");
        finish(metadataRetrievalThreads, cleanupThreads);
        return;
    }
    LOGGER.debug("Starting fusion for temporal context");
    long start = System.currentTimeMillis();
    /* Retrieve the MediaSegmentDescriptors needed for the temporal scoring retrieval */
    Map<String, MediaSegmentDescriptor> segmentMap = segments.stream().distinct().collect(Collectors.toMap(MediaSegmentDescriptor::getSegmentId, x -> x, (x1, x2) -> x1));
    /* Initialise the temporal scoring algorithms depending on timeDistances list */
    List<List<StringDoublePair>> tmpContainerResults = new ArrayList<>();
    IntStream.range(0, message.getQueries().size()).forEach(idx -> tmpContainerResults.add(containerResults.getOrDefault(idx, new ArrayList<>())));
    /* Score and retrieve the results */
    List<TemporalObject> results = TemporalScoring.score(segmentMap, tmpContainerResults, message.getTimeDistances(), message.getMaxLength());
    List<TemporalObject> finalResults = results.stream().sorted(TemporalObject.COMPARATOR.reversed()).limit(max).collect(Collectors.toList());
    LOGGER.debug("Temporal scoring done in {} ms, {} results", System.currentTimeMillis() - start, finalResults.size());
    /* Retrieve the segment Ids of the newly scored segments */
    List<String> segmentIds = finalResults.stream().map(TemporalObject::getSegments).flatMap(List::stream).collect(Collectors.toList());
    /* Send potential information not already sent  */
    /* Maybe change from list to set? */
    segmentIds = segmentIds.stream().filter(s -> !sentSegmentIds.contains(s)).collect(Collectors.toList());
    List<String> objectIds = segments.stream().map(MediaSegmentDescriptor::getObjectId).collect(Collectors.toList());
    objectIds = objectIds.stream().filter(s -> !sentObjectIds.contains(s)).collect(Collectors.toList());
    /* If necessary, send to the UI */
    if (segmentIds.size() != 0 && objectIds.size() != 0) {
        this.submitSegmentAndObjectInformationFromIds(session, uuid, segmentIds, objectIds);
        /* Retrieve and send metadata for items not already sent */
        List<Thread> _threads = this.submitMetadata(session, uuid, segmentIds, objectIds, segmentIdsForWhichMetadataIsFetched, objectIdsForWhichMetadataIsFetched, message.getMetadataAccessSpec());
        metadataRetrievalThreads.addAll(_threads);
    }
    /* Send scoring results to the frontend */
    if (finalResults.size() > 0) {
        futures.addAll(this.finalizeAndSubmitTemporalResults(session, uuid, finalResults));
        futures.forEach(CompletableFuture::join);
    }
    finish(metadataRetrievalThreads, cleanupThreads);
}
Also used : IntStream(java.util.stream.IntStream) QueryStage(org.vitrivr.cineast.api.messages.query.QueryStage) TemporalQuery(org.vitrivr.cineast.api.messages.query.TemporalQuery) StagedSimilarityQuery(org.vitrivr.cineast.api.messages.query.StagedSimilarityQuery) AbstractQueryTermContainer(org.vitrivr.cineast.core.data.query.containers.AbstractQueryTermContainer) TemporalObject(org.vitrivr.cineast.core.data.TemporalObject) HashMap(java.util.HashMap) CompletableFuture(java.util.concurrent.CompletableFuture) ArrayList(java.util.ArrayList) HashSet(java.util.HashSet) TemporalScoring(org.vitrivr.cineast.core.temporal.TemporalScoring) QueryTerm(org.vitrivr.cineast.api.messages.query.QueryTerm) Map(java.util.Map) Session(org.eclipse.jetty.websocket.api.Session) IntObjectHashMap(io.netty.util.collection.IntObjectHashMap) MediaSegmentDescriptor(org.vitrivr.cineast.core.data.entities.MediaSegmentDescriptor) ContinuousRetrievalLogic(org.vitrivr.cineast.standalone.util.ContinuousRetrievalLogic) QueryConfig(org.vitrivr.cineast.core.config.QueryConfig) Set(java.util.Set) StringDoublePair(org.vitrivr.cineast.core.data.StringDoublePair) Collectors(java.util.stream.Collectors) List(java.util.List) Logger(org.apache.logging.log4j.Logger) SegmentScoreElement(org.vitrivr.cineast.core.data.score.SegmentScoreElement) LogManager(org.apache.logging.log4j.LogManager) Config(org.vitrivr.cineast.standalone.config.Config) AbstractQueryTermContainer(org.vitrivr.cineast.core.data.query.containers.AbstractQueryTermContainer) ArrayList(java.util.ArrayList) QueryTerm(org.vitrivr.cineast.api.messages.query.QueryTerm) StringDoublePair(org.vitrivr.cineast.core.data.StringDoublePair) CompletableFuture(java.util.concurrent.CompletableFuture) QueryStage(org.vitrivr.cineast.api.messages.query.QueryStage) ArrayList(java.util.ArrayList) List(java.util.List) HashSet(java.util.HashSet) QueryConfig(org.vitrivr.cineast.core.config.QueryConfig) IntObjectHashMap(io.netty.util.collection.IntObjectHashMap) TemporalObject(org.vitrivr.cineast.core.data.TemporalObject) StagedSimilarityQuery(org.vitrivr.cineast.api.messages.query.StagedSimilarityQuery) SegmentScoreElement(org.vitrivr.cineast.core.data.score.SegmentScoreElement) MediaSegmentDescriptor(org.vitrivr.cineast.core.data.entities.MediaSegmentDescriptor) HashMap(java.util.HashMap) Map(java.util.Map) IntObjectHashMap(io.netty.util.collection.IntObjectHashMap)

Aggregations

ArrayList (java.util.ArrayList)9 StringDoublePair (org.vitrivr.cineast.core.data.StringDoublePair)9 HashMap (java.util.HashMap)6 List (java.util.List)6 MediaSegmentDescriptor (org.vitrivr.cineast.core.data.entities.MediaSegmentDescriptor)6 TemporalObject (org.vitrivr.cineast.core.data.TemporalObject)5 AbstractQueryTermContainer (org.vitrivr.cineast.core.data.query.containers.AbstractQueryTermContainer)3 TObjectDoubleHashMap (gnu.trove.map.hash.TObjectDoubleHashMap)2 HashSet (java.util.HashSet)2 Map (java.util.Map)2 Set (java.util.Set)2 CompletableFuture (java.util.concurrent.CompletableFuture)2 Collectors (java.util.stream.Collectors)2 StopWatch (org.apache.commons.lang3.time.StopWatch)2 LogManager (org.apache.logging.log4j.LogManager)2 Logger (org.apache.logging.log4j.Logger)2 QueryConfig (org.vitrivr.cineast.core.config.QueryConfig)2 ReadableQueryConfig (org.vitrivr.cineast.core.config.ReadableQueryConfig)2 SegmentScoreElement (org.vitrivr.cineast.core.data.score.SegmentScoreElement)2 Config (org.vitrivr.cineast.standalone.config.Config)2