Merge branch 'develop' into features/mchmmer

[jalview.git] / src / jalview / datamodel / Sequence.java
diff --git a/src/jalview/datamodel/Sequence.java b/src/jalview/datamodel/Sequence.java

index 9e81d81..7972626 100755 (executable)
--- a/src/jalview/datamodel/Sequence.java
+++ b/src/jalview/datamodel/Sequence.java
@@ -28,12 +28,14 @@ import jalview.util.Comparison;
  import jalview.util.DBRefUtils;
  import jalview.util.MapList;
  import jalview.util.StringUtils;
+import jalview.workers.InformationThread;
  
  import java.util.ArrayList;
  import java.util.Arrays;
  import java.util.BitSet;
  import java.util.Collections;
  import java.util.Enumeration;
+import java.util.Iterator;
  import java.util.List;
  import java.util.ListIterator;
  import java.util.Vector;
@@ -61,6 +63,10 @@ public class Sequence extends ASequence implements SequenceI
  
    int end;
  
+  HiddenMarkovModel hmm;
+
+  boolean isHMMConsensusSequence = false;
+
    Vector<PDBEntry> pdbIds;
  
    String vamsasId;
@@ -337,6 +343,11 @@ public class Sequence extends ASequence implements SequenceI
          this.addPDBId(new PDBEntry(pdb));
        }
      }
+    if (seq.getHMM() != null)
+    {
+      this.hmm = new HiddenMarkovModel(seq.getHMM(), this);
+    }
+
    }
  
    @Override
@@ -408,7 +419,7 @@ public class Sequence extends ASequence implements SequenceI
    {
      if (pdbIds == null)
      {
-      pdbIds = new Vector<PDBEntry>();
+      pdbIds = new Vector<>();
        pdbIds.add(entry);
        return true;
      }
@@ -444,7 +455,7 @@ public class Sequence extends ASequence implements SequenceI
    @Override
    public Vector<PDBEntry> getAllPDBEntries()
    {
-    return pdbIds == null ? new Vector<PDBEntry>() : pdbIds;
+    return pdbIds == null ? new Vector<>() : pdbIds;
    }
  
    /**
@@ -814,7 +825,7 @@ public class Sequence extends ASequence implements SequenceI
     * @param curs
     * @return
     */
-  protected int findIndex(int pos, SequenceCursor curs)
+  protected int findIndex(final int pos, SequenceCursor curs)
    {
      if (!isValidCursor(curs))
      {
@@ -840,8 +851,13 @@ public class Sequence extends ASequence implements SequenceI
      while (newPos != pos)
      {
        col += delta; // shift one column left or right
-      if (col < 0 || col == sequence.length)
+      if (col < 0)
+      {
+        break;
+      }
+      if (col == sequence.length)
        {
+        col--; // return last column if we failed to reach pos
          break;
        }
        if (!Comparison.isGap(sequence[col]))
@@ -851,7 +867,14 @@ public class Sequence extends ASequence implements SequenceI
      }
  
      col++; // convert back to base 1
-    updateCursor(pos, col, curs.firstColumnPosition);
+
+    /*
+     * only update cursor if we found the target position
+     */
+    if (newPos == pos)
+    {
+      updateCursor(pos, col, curs.firstColumnPosition);
+    }
  
      return col;
    }
@@ -1128,6 +1151,27 @@ public class Sequence extends ASequence implements SequenceI
      return map;
    }
  
+  /**
+   * Build a bitset corresponding to sequence gaps
+   * 
+   * @return a BitSet where set values correspond to gaps in the sequence
+   */
+  @Override
+  public BitSet gapBitset()
+  {
+    BitSet gaps = new BitSet(sequence.length);
+    int j = 0;
+    while (j < sequence.length)
+    {
+      if (jalview.util.Comparison.isGap(sequence[j]))
+      {
+        gaps.set(j);
+      }
+      j++;
+    }
+    return gaps;
+  }
+
    @Override
    public int[] findPositionMap()
    {
@@ -1151,7 +1195,7 @@ public class Sequence extends ASequence implements SequenceI
    @Override
    public List<int[]> getInsertions()
    {
-    ArrayList<int[]> map = new ArrayList<int[]>();
+    ArrayList<int[]> map = new ArrayList<>();
      int lastj = -1, j = 0;
      int pos = start;
      int seqlen = sequence.length;
@@ -1217,7 +1261,7 @@ public class Sequence extends ASequence implements SequenceI
    }
  
    @Override
-  public void deleteChars(int i, int j)
+  public void deleteChars(final int i, final int j)
    {
      int newstart = start, newend = end;
      if (i >= sequence.length || i < 0)
@@ -1229,62 +1273,75 @@ public class Sequence extends ASequence implements SequenceI
      boolean createNewDs = false;
      // TODO: take a (second look) at the dataset creation validation method for
      // the very large sequence case
-    int eindex = -1, sindex = -1;
-    boolean ecalc = false, scalc = false;
+    int startIndex = findIndex(start) - 1;
+    int endIndex = findIndex(end) - 1;
+    int startDeleteColumn = -1; // for dataset sequence deletions
+    int deleteCount = 0;
+
      for (int s = i; s < j; s++)
      {
-      if (jalview.schemes.ResidueProperties.aaIndex[sequence[s]] != 23)
+      if (Comparison.isGap(sequence[s]))
+      {
+        continue;
+      }
+      deleteCount++;
+      if (startDeleteColumn == -1)
+      {
+        startDeleteColumn = findPosition(s) - start;
+      }
+      if (createNewDs)
+      {
+        newend--;
+      }
+      else
        {
-        if (createNewDs)
+        if (startIndex == s)
          {
-          newend--;
+          /*
+           * deleting characters from start of sequence; new start is the
+           * sequence position of the next column (position to the right
+           * if the column position is gapped)
+           */
+          newstart = findPosition(j);
+          break;
          }
          else
          {
-          if (!scalc)
-          {
-            sindex = findIndex(start) - 1;
-            scalc = true;
-          }
-          if (sindex == s)
+          if (endIndex < j)
            {
-            // delete characters including start of sequence
-            newstart = findPosition(j);
-            break; // don't need to search for any more residue characters.
+            /*
+             * deleting characters at end of sequence; new end is the sequence
+             * position of the column before the deletion; subtract 1 if this is
+             * gapped since findPosition returns the next sequence position
+             */
+            newend = findPosition(i - 1);
+            if (Comparison.isGap(sequence[i - 1]))
+            {
+              newend--;
+            }
+            break;
            }
            else
            {
-            // delete characters after start.
-            if (!ecalc)
-            {
-              eindex = findIndex(end) - 1;
-              ecalc = true;
-            }
-            if (eindex < j)
-            {
-              // delete characters at end of sequence
-              newend = findPosition(i - 1);
-              break; // don't need to search for any more residue characters.
-            }
-            else
-            {
-              createNewDs = true;
-              newend--; // decrease end position by one for the deleted residue
-              // and search further
-            }
+            createNewDs = true;
+            newend--;
            }
          }
        }
      }
-    // deletion occured in the middle of the sequence
+
      if (createNewDs && this.datasetSequence != null)
      {
-      // construct a new sequence
+      /*
+       * if deletion occured in the middle of the sequence,
+       * construct a new dataset sequence and delete the residues
+       * that were deleted from the aligned sequence
+       */
        Sequence ds = new Sequence(datasetSequence);
+      ds.deleteChars(startDeleteColumn, startDeleteColumn + deleteCount);
+      datasetSequence = ds;
        // TODO: remove any non-inheritable properties ?
        // TODO: create a sequence mapping (since there is a relation here ?)
-      ds.deleteChars(i, j);
-      datasetSequence = ds;
      }
      start = newstart;
      end = newend;
@@ -1449,7 +1506,7 @@ public class Sequence extends ASequence implements SequenceI
    {
      if (this.annotation == null)
      {
-      this.annotation = new Vector<AlignmentAnnotation>();
+      this.annotation = new Vector<>();
      }
      if (!this.annotation.contains(annotation))
      {
@@ -1616,7 +1673,7 @@ public class Sequence extends ASequence implements SequenceI
        return null;
      }
  
-    Vector<AlignmentAnnotation> subset = new Vector<AlignmentAnnotation>();
+    Vector<AlignmentAnnotation> subset = new Vector<>();
      Enumeration<AlignmentAnnotation> e = annotation.elements();
      while (e.hasMoreElements())
      {
@@ -1750,12 +1807,13 @@ public class Sequence extends ASequence implements SequenceI
    public List<AlignmentAnnotation> getAlignmentAnnotations(String calcId,
            String label)
    {
-    List<AlignmentAnnotation> result = new ArrayList<AlignmentAnnotation>();
+    List<AlignmentAnnotation> result = new ArrayList<>();
      if (this.annotation != null)
      {
        for (AlignmentAnnotation ann : annotation)
        {
-        if (ann.calcId != null && ann.calcId.equals(calcId)
+        String id = ann.getCalcId();
+        if (id != null && id.equals(calcId)
                  && ann.label != null && ann.label.equals(label))
          {
            result.add(ann);
@@ -1806,7 +1864,7 @@ public class Sequence extends ASequence implements SequenceI
      }
      synchronized (dbrefs)
      {
-      List<DBRefEntry> primaries = new ArrayList<DBRefEntry>();
+      List<DBRefEntry> primaries = new ArrayList<>();
        DBRefEntry[] tmp = new DBRefEntry[1];
        for (DBRefEntry ref : dbrefs)
        {
@@ -1853,6 +1911,34 @@ public class Sequence extends ASequence implements SequenceI
      }
    }
  
+  @Override
+  public HiddenMarkovModel getHMM()
+  {
+    return hmm;
+  }
+
+  @Override
+  public void setHMM(HiddenMarkovModel hmm)
+  {
+    this.hmm = hmm;
+  }
+
+  @Override
+  public boolean hasHMMAnnotation()
+  {
+    if (this.annotation == null) {
+      return false;
+    }
+    for (AlignmentAnnotation ann : annotation)
+    {
+      if (InformationThread.HMM_CALC_ID.equals(ann.getCalcId()))
+      {
+        return true;
+      }
+    }
+    return false;
+  }
+
    /**
     * {@inheritDoc}
     */
@@ -1948,4 +2034,79 @@ public class Sequence extends ASequence implements SequenceI
  
      return count;
    }
+
+  @Override
+  public String getSequenceStringFromIterator(Iterator<int[]> it)
+  {
+    StringBuilder newSequence = new StringBuilder();
+    while (it.hasNext())
+    {
+      int[] block = it.next();
+      if (it.hasNext())
+      {
+        newSequence.append(getSequence(block[0], block[1] + 1));
+      }
+      else
+      {
+        newSequence.append(getSequence(block[0], block[1]));
+      }
+    }
+
+    return newSequence.toString();
+  }
+
+  @Override
+  public int firstResidueOutsideIterator(Iterator<int[]> regions)
+  {
+    int start = 0;
+
+    if (!regions.hasNext())
+    {
+      return findIndex(getStart()) - 1;
+    }
+
+    // Simply walk along the sequence whilst watching for region
+    // boundaries
+    int hideStart = getLength();
+    int hideEnd = -1;
+    boolean foundStart = false;
+
+    // step through the non-gapped positions of the sequence
+    for (int i = getStart(); i <= getEnd() && (!foundStart); i++)
+    {
+      // get alignment position of this residue in the sequence
+      int p = findIndex(i) - 1;
+
+      // update region start/end
+      while (hideEnd < p && regions.hasNext())
+      {
+        int[] region = regions.next();
+        hideStart = region[0];
+        hideEnd = region[1];
+      }
+      if (hideEnd < p)
+      {
+        hideStart = getLength();
+      }
+      // update boundary for sequence
+      if (p < hideStart)
+      {
+        start = p;
+        foundStart = true;
+      }
+    }
+
+    if (foundStart)
+    {
+      return start;
+    }
+    // otherwise, sequence was completely hidden
+    return 0;
+  }
+
+  @Override
+  public boolean hasHMMProfile()
+  {
+    return hmm != null;
+  }
  }